hippo-memory 1.56.0 → 1.58.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (129) hide show
  1. package/README.md +11 -0
  2. package/dist/agent-memories/claude-code.js +1 -1
  3. package/dist/agent-memories/gemini.js +1 -1
  4. package/dist/api-errors.d.ts +27 -0
  5. package/dist/api-errors.js +37 -0
  6. package/dist/api.d.ts +21 -14
  7. package/dist/api.js +97 -71
  8. package/dist/audit.d.ts +4 -0
  9. package/dist/audit.js +11 -0
  10. package/dist/autolearn.d.ts +1 -1
  11. package/dist/autolearn.js +7 -5
  12. package/dist/capture-contract.d.ts +47 -0
  13. package/dist/capture-contract.js +49 -0
  14. package/dist/capture-error.js +2 -1
  15. package/dist/capture.d.ts +0 -13
  16. package/dist/capture.js +5 -66
  17. package/dist/card-detail.d.ts +1 -1
  18. package/dist/card-detail.js +1 -1
  19. package/dist/cli/shared.d.ts +137 -0
  20. package/dist/cli/shared.js +834 -0
  21. package/dist/cli/sleep.d.ts +10 -0
  22. package/dist/cli/sleep.js +171 -0
  23. package/dist/cli.d.ts +0 -7
  24. package/dist/cli.js +322 -1827
  25. package/dist/client.js +9 -0
  26. package/dist/codex-patch.js +1 -1
  27. package/dist/compaction-record.d.ts +1 -1
  28. package/dist/compaction-record.js +3 -2
  29. package/dist/config.d.ts +5 -0
  30. package/dist/config.js +17 -0
  31. package/dist/connectors/github/dlq.js +5 -2
  32. package/dist/connectors/github/octokit-client.js +4 -2
  33. package/dist/connectors/github/webhook.d.ts +19 -0
  34. package/dist/connectors/github/webhook.js +313 -0
  35. package/dist/connectors/slack/dlq.js +6 -2
  36. package/dist/connectors/slack/web-client.js +7 -5
  37. package/dist/connectors/slack/webhook.d.ts +22 -0
  38. package/dist/connectors/slack/webhook.js +203 -0
  39. package/dist/consolidate.d.ts +10 -0
  40. package/dist/consolidate.js +38 -35
  41. package/dist/context-auto.d.ts +3 -0
  42. package/dist/context-auto.js +34 -0
  43. package/dist/customer-notes.js +16 -14
  44. package/dist/dag.js +3 -2
  45. package/dist/dashboard.js +3 -2
  46. package/dist/db.d.ts +12 -0
  47. package/dist/db.js +62 -1
  48. package/dist/decisions.js +11 -9
  49. package/dist/doctor.js +5 -0
  50. package/dist/embedding-provider.js +3 -3
  51. package/dist/embeddings.d.ts +4 -4
  52. package/dist/embeddings.js +72 -16
  53. package/dist/eval-stats.d.ts +58 -0
  54. package/dist/eval-stats.js +111 -0
  55. package/dist/extract.js +3 -2
  56. package/dist/goals.d.ts +49 -25
  57. package/dist/goals.js +39 -22
  58. package/dist/graph-extract.js +1 -1
  59. package/dist/graph-recall.d.ts +1 -1
  60. package/dist/graph-recall.js +1 -1
  61. package/dist/graph.js +1 -1
  62. package/dist/hooks.d.ts +1 -3
  63. package/dist/hooks.js +2 -4
  64. package/dist/http-retry.d.ts +21 -0
  65. package/dist/http-retry.js +50 -0
  66. package/dist/http-util.d.ts +39 -0
  67. package/dist/http-util.js +56 -0
  68. package/dist/importers.d.ts +2 -0
  69. package/dist/importers.js +16 -5
  70. package/dist/incidents.js +13 -11
  71. package/dist/index.d.ts +5 -2
  72. package/dist/index.js +5 -2
  73. package/dist/judgment.js +10 -17
  74. package/dist/log.d.ts +25 -0
  75. package/dist/log.js +48 -0
  76. package/dist/mcp/server.js +224 -308
  77. package/dist/mcp/tool-args.d.ts +21 -0
  78. package/dist/mcp/tool-args.js +80 -0
  79. package/dist/memory.d.ts +19 -0
  80. package/dist/memory.js +41 -2
  81. package/dist/overlap-index.d.ts +7 -0
  82. package/dist/overlap-index.js +38 -0
  83. package/dist/pilot-arm.d.ts +9 -0
  84. package/dist/pilot-arm.js +47 -0
  85. package/dist/policies.js +14 -12
  86. package/dist/predictions.js +11 -9
  87. package/dist/processes.js +16 -14
  88. package/dist/project-briefs.js +19 -16
  89. package/dist/project-identity.d.ts +1 -1
  90. package/dist/project-identity.js +25 -1
  91. package/dist/prompt-recall.js +1 -1
  92. package/dist/raw-archive.js +7 -6
  93. package/dist/recall-history.d.ts +5 -0
  94. package/dist/recall-history.js +9 -0
  95. package/dist/recall-pipeline.d.ts +101 -0
  96. package/dist/recall-pipeline.js +313 -0
  97. package/dist/recall-scope.d.ts +24 -1
  98. package/dist/recall-scope.js +29 -2
  99. package/dist/refine-llm.js +3 -2
  100. package/dist/reject-flow.js +6 -9
  101. package/dist/rejection.d.ts +2 -1
  102. package/dist/rejection.js +2 -1
  103. package/dist/search.d.ts +0 -20
  104. package/dist/search.js +16 -51
  105. package/dist/secret-detect.d.ts +13 -1
  106. package/dist/secret-detect.js +33 -1
  107. package/dist/server.d.ts +3 -1
  108. package/dist/server.js +1854 -2566
  109. package/dist/session-digest.js +2 -1
  110. package/dist/shared.js +7 -6
  111. package/dist/skills.js +17 -15
  112. package/dist/store-cards.d.ts +53 -0
  113. package/dist/store-cards.js +512 -0
  114. package/dist/store.d.ts +2 -89
  115. package/dist/store.js +10 -566
  116. package/dist/tenant.d.ts +22 -0
  117. package/dist/tenant.js +26 -0
  118. package/dist/token-ledger.d.ts +4 -2
  119. package/dist/token-ledger.js +2 -2
  120. package/dist/tokenize.d.ts +2 -0
  121. package/dist/tokenize.js +8 -0
  122. package/dist/version.d.ts +1 -1
  123. package/dist/version.js +1 -1
  124. package/extensions/openclaw-plugin/openclaw.plugin.json +1 -1
  125. package/extensions/openclaw-plugin/package.json +1 -1
  126. package/openclaw.plugin.json +1 -1
  127. package/package.json +1 -1
  128. package/dist/connectors/slack/ratelimit.d.ts +0 -9
  129. package/dist/connectors/slack/ratelimit.js +0 -18
@@ -10,24 +10,23 @@
10
10
  import * as fs from 'fs';
11
11
  import * as path from 'path';
12
12
  import { createMemory, Layer, calculateStrength, } from '../memory.js';
13
- import { hybridSearch, physicsSearch, estimateTokens } from '../search.js';
13
+ import { fitBudget, estimateTokens } from '../search.js';
14
14
  import { evalNow } from '../ablation.js';
15
- import { loadAllEntries, writeEntry, strengthenRetrieved, readEntry, loadFreshActiveTaskSnapshot, listMemoryConflicts, resolveConflict, countCreatedSinceLastSleep } from '../store.js';
15
+ import { loadAllEntries, writeEntry, readEntry, listMemoryConflicts, resolveConflict, countCreatedSinceLastSleep } from '../store.js';
16
16
  import { shareMemory, listPeers, getGlobalRoot, initGlobal } from '../shared.js';
17
17
  import { consolidate } from '../consolidate.js';
18
- import { execSync } from 'child_process';
19
18
  import { fetchGitLog, extractLessons, partitionLessons, isGitRepo } from '../autolearn.js';
20
19
  import { dropHeldCopies, duplicateKey, storedTextKeys } from '../same-text.js';
21
20
  import { loadConfig } from '../config.js';
22
21
  import { confidenceLabel } from '../memory.js';
23
22
  import { resolveTenantId } from '../tenant.js';
24
- import { recall as apiRecall, remember as apiRemember, outcome as apiOutcome, drillDown as apiDrillDown, assemble as apiAssemble, passesScopeFilterForRecall, buildSuppressionSummary, ambientSecretAdmit } from '../api.js';
25
- import { assertScopeRequestAllowed } from '../recall-scope.js';
26
- import { resolveProjectIdentity, classifyOriginProject, findHippoStoreDir } from '../project-identity.js';
23
+ import { retrieve as apiRetrieve, remember as apiRemember, outcome as apiOutcome, drillDown as apiDrillDown, assemble as apiAssemble, getContext as apiGetContext, buildSuppressionSummary } from '../api.js';
24
+ import { autoDetectContext } from '../context-auto.js';
25
+ import { resolveProjectIdentity, findHippoStoreDir } from '../project-identity.js';
27
26
  import { computePredictionBaserate } from '../predictions.js';
28
27
  import { appendAuditEvent, auditQueryFields } from '../audit.js';
29
28
  import { RejectedValueError } from '../rejection.js';
30
- import { detectAnchoring, hashQueryText, buildSessionKey, getOrCreateRing, appendRecall, snapshotRing, } from '../recall-history.js';
29
+ import { detectAnchoring, hashQueryText, biasHintEnabled, buildSessionKey, getOrCreateRing, appendRecall, snapshotRing, } from '../recall-history.js';
31
30
  import { detectAvailabilityBias } from '../availability.js';
32
31
  // v0.33 / J1 — Module-level per-(tenant, session) recall-history ring map
33
32
  // for the MCP pipeline. Separate from CLI/HTTP rings per plan v3
@@ -37,10 +36,10 @@ const sessionRecallHistoryMcp = new Map();
37
36
  export function __resetSessionRecallHistoryMcp() {
38
37
  sessionRecallHistoryMcp.clear();
39
38
  }
40
- import { applyGoalStackBoost } from '../goals.js';
41
39
  import { openHippoDb, closeHippoDb } from '../db.js';
42
40
  import { recordTokenUse } from '../token-ledger.js';
43
41
  import { PACKAGE_VERSION } from '../version.js';
42
+ import { validateToolArgs } from './tool-args.js';
44
43
  // ── Find hippo root ──
45
44
  /** Same bounded walk as the CLI (ends at home, so HIPPO_HOME wins over ~/.hippo); cwd/opts are the test seam. */
46
45
  export function findHippoRoot(cwd = process.cwd(), opts) {
@@ -78,6 +77,30 @@ function isJsonObjectRecord(v) {
78
77
  }
79
78
  import { formatHandoffEvidenceLine } from '../handoff.js';
80
79
  import { assembleCost, assembleText, drillCost, drillText, printedTokens } from '../context-render.js';
80
+ function handoffLines(h) {
81
+ const lines = [`- Summary: ${h.summary}`];
82
+ if (h.nextAction)
83
+ lines.push(`- Next action: ${h.nextAction}`);
84
+ if ((h.artifacts ?? []).length > 0)
85
+ lines.push(`- Artifacts: ${(h.artifacts ?? []).join(', ')}`);
86
+ if (h.outcome)
87
+ lines.push(`- Outcome: ${h.outcome}`);
88
+ if (h.targetRuntime)
89
+ lines.push(`- Target runtime: ${h.targetRuntime}`);
90
+ if (h.cardId)
91
+ lines.push(`- Card: ${h.cardId}`);
92
+ if ((h.constraints ?? []).length > 0)
93
+ lines.push(`- Constraints: ${(h.constraints ?? []).join(', ')}`);
94
+ if (h.evidence)
95
+ lines.push(`- Evidence: ${formatHandoffEvidenceLine(h.evidence)}`);
96
+ return lines;
97
+ }
98
+ function trailLines(events) {
99
+ return events.map((e) => {
100
+ const preview = e.content.length > 200 ? e.content.slice(0, 200) + '…' : e.content;
101
+ return `- [${e.event_type}] ${preview}`;
102
+ });
103
+ }
81
104
  function formatContinuityBlock(block) {
82
105
  const lines = ['## Continuity'];
83
106
  if (block.activeSnapshot) {
@@ -90,36 +113,12 @@ function formatContinuityBlock(block) {
90
113
  if (block.sessionHandoff) {
91
114
  lines.push('');
92
115
  lines.push('### Session Handoff');
93
- lines.push(`- Summary: ${block.sessionHandoff.summary}`);
94
- if (block.sessionHandoff.nextAction) {
95
- lines.push(`- Next action: ${block.sessionHandoff.nextAction}`);
96
- }
97
- if ((block.sessionHandoff.artifacts ?? []).length > 0) {
98
- lines.push(`- Artifacts: ${(block.sessionHandoff.artifacts ?? []).join(', ')}`);
99
- }
100
- if (block.sessionHandoff.outcome) {
101
- lines.push(`- Outcome: ${block.sessionHandoff.outcome}`);
102
- }
103
- if (block.sessionHandoff.targetRuntime) {
104
- lines.push(`- Target runtime: ${block.sessionHandoff.targetRuntime}`);
105
- }
106
- if (block.sessionHandoff.cardId) {
107
- lines.push(`- Card: ${block.sessionHandoff.cardId}`);
108
- }
109
- if ((block.sessionHandoff.constraints ?? []).length > 0) {
110
- lines.push(`- Constraints: ${(block.sessionHandoff.constraints ?? []).join(', ')}`);
111
- }
112
- if (block.sessionHandoff.evidence) {
113
- lines.push(`- Evidence: ${formatHandoffEvidenceLine(block.sessionHandoff.evidence)}`);
114
- }
116
+ lines.push(...handoffLines(block.sessionHandoff));
115
117
  }
116
118
  if (block.recentSessionEvents.length > 0) {
117
119
  lines.push('');
118
120
  lines.push('### Recent Session Trail');
119
- for (const e of block.recentSessionEvents) {
120
- const preview = e.content.length > 200 ? e.content.slice(0, 200) + '…' : e.content;
121
- lines.push(`- [${e.event_type}] ${preview}`);
122
- }
121
+ lines.push(...trailLines(block.recentSessionEvents));
123
122
  }
124
123
  if (lines.length === 1) {
125
124
  lines.push('');
@@ -147,6 +146,36 @@ const memoryCost = (r) => printedTokens(formatMemory(r));
147
146
  function memoriesReserve(budget) {
148
147
  return Math.max(printedTokens(memoriesHeading(budget)), estimateTokens(NO_MEMORIES));
149
148
  }
149
+ function snapshotPiece(s) {
150
+ return [
151
+ '## Active Task Snapshot',
152
+ `- Task: ${s.task}`,
153
+ `- Status: ${s.status}`,
154
+ `- Updated: ${s.updated_at}`,
155
+ '',
156
+ '### Summary',
157
+ s.summary,
158
+ '',
159
+ '### Next step',
160
+ s.next_step,
161
+ '',
162
+ '',
163
+ ].join('\n');
164
+ }
165
+ function handoffPiece(h) {
166
+ return ['## Session Handoff', ...handoffLines(h), '', ''].join('\n');
167
+ }
168
+ function trailPiece(events) {
169
+ return ['## Recent Session Trail', ...trailLines(events), '', ''].join('\n');
170
+ }
171
+ // Sections print ahead of the memories in hippo_context, so getContext pays for each as printed before any memory.
172
+ const contextCost = {
173
+ entry: memoryCost,
174
+ fixed: (budget) => memoriesReserve(budget),
175
+ snapshot: (s) => estimateTokens(snapshotPiece(s)),
176
+ handoff: (h) => estimateTokens(handoffPiece(h)),
177
+ trail: (events) => estimateTokens(trailPiece(events)),
178
+ };
150
179
  // Rows the ranked list already shows drop out of this section, so pricing every row bounds what it prints.
151
180
  function tailSection(rows) {
152
181
  if (rows.length === 0)
@@ -164,7 +193,7 @@ function tailSection(rows) {
164
193
  }
165
194
  return '\n' + lines.join('\n');
166
195
  }
167
- // J3.2: the hint depends on the query alone, so api.recall's copy is the one shown; JSON.stringify fences the phrase.
196
+ // J3.2: the hint depends on the query alone, so api.retrieve's copy is the one shown; JSON.stringify fences the phrase.
168
197
  function planningSection(r) {
169
198
  if (r.planningFallacyHint) {
170
199
  const h = r.planningFallacyHint;
@@ -177,6 +206,10 @@ function planningSection(r) {
177
206
  return '';
178
207
  }
179
208
  // ── Tool definitions ──
209
+ // HTTP sets no budget cap; 25x the 4000 recall default leaves room for large-context clients while bounding one call's work.
210
+ const MAX_BUDGET_TOKENS = 100_000;
211
+ // Same ceiling as the HTTP list routes' parseListLimit.
212
+ const MAX_LIST_LIMIT = 1000;
180
213
  const TOOLS = [
181
214
  {
182
215
  name: 'hippo_recall',
@@ -185,7 +218,12 @@ const TOOLS = [
185
218
  type: 'object',
186
219
  properties: {
187
220
  query: { type: 'string', description: 'What to search for in memory (natural language)' },
188
- budget: { type: 'number', description: 'Max tokens to return (default: config.defaultBudget, 4000)' },
221
+ budget: {
222
+ type: 'number',
223
+ minimum: 0,
224
+ maximum: MAX_BUDGET_TOKENS,
225
+ description: `Max tokens to return (default: config.defaultBudget, 4000; max ${MAX_BUDGET_TOKENS})`,
226
+ },
189
227
  include_continuity: {
190
228
  type: 'boolean',
191
229
  description: 'Append continuity context (active snapshot + handoff + last 5 session events) below the memory results. Useful at session boot.',
@@ -208,12 +246,12 @@ const TOOLS = [
208
246
  },
209
247
  scorer_window: {
210
248
  type: 'number',
211
- description: 'Candidate pool size that api.recall evaluates. Affects fresh-tail / summarize-overflow appendix paths and continuity hits. Note: the primary ranked block over MCP is driven by a separate physics/hybrid scorer over the full tenant store, so scorer_window does NOT narrow the main results — only the appendix. Default 200. Rejected as RecallContractError code=invalid_scorer_window if 0/negative/non-finite/non-numeric.',
249
+ description: 'How many of the top-ranked memories the fresh-tail and summarize-overflow appendix is worked out against. The main list ranks the whole tenant store, so scorer_window does not narrow it. Default 200. Rejected as RecallContractError code=invalid_scorer_window if 0/negative/non-finite/non-numeric.',
212
250
  },
213
251
  session_id: {
214
252
  type: 'string',
215
253
  maxLength: 256,
216
- description: 'Optional session id (v1.7.4). When set AND (tenant, session) has active goals, applies the dlPFC goal-stack boost to the primary physics/hybrid result band before formatting AND to api.recall\'s primary BM25 band (so the audit + appendix paths see the same session). Mirrors fresh_tail_session_id shape (256-char cap).',
254
+ description: 'Optional session id (v1.7.4). When set AND (tenant, session) has active goals, applies the dlPFC goal-stack boost to the ranked memories before formatting. Mirrors fresh_tail_session_id shape (256-char cap).',
217
255
  },
218
256
  },
219
257
  required: ['query'],
@@ -231,7 +269,9 @@ const TOOLS = [
231
269
  },
232
270
  budget: {
233
271
  type: 'number',
234
- description: 'Token budget for the assembled context (default 4000). Eviction kicks in over budget.',
272
+ minimum: 0,
273
+ maximum: MAX_BUDGET_TOKENS,
274
+ description: `Token budget for the assembled context (default 4000; max ${MAX_BUDGET_TOKENS}). Eviction kicks in over budget.`,
235
275
  },
236
276
  fresh_tail_count: {
237
277
  type: 'number',
@@ -261,11 +301,15 @@ const TOOLS = [
261
301
  },
262
302
  limit: {
263
303
  type: 'number',
264
- description: 'Max children to return (default 50).',
304
+ minimum: 0,
305
+ maximum: MAX_LIST_LIMIT,
306
+ description: `Max children to return (default 50; max ${MAX_LIST_LIMIT}).`,
265
307
  },
266
308
  budget: {
267
309
  type: 'number',
268
- description: 'Max total token cost (~ chars/4) of returned children. Truncates chronologically.',
310
+ minimum: 0,
311
+ maximum: MAX_BUDGET_TOKENS,
312
+ description: `Max total token cost (~ chars/4) of returned children (max ${MAX_BUDGET_TOKENS}). Truncates chronologically.`,
269
313
  },
270
314
  depth: {
271
315
  type: 'integer',
@@ -310,14 +354,19 @@ const TOOLS = [
310
354
  },
311
355
  {
312
356
  name: 'hippo_context',
313
- description: 'Smart context injection: auto-detects current task from git state and returns relevant memories plus the active task snapshot. Use at the start of any session. Memories and snapshot are scope-filtered: a no-scope caller does NOT see ANY <source>:private:* (slack, github, ...) or legacy-quarantine rows.',
357
+ description: 'Smart context injection: auto-detects current task from git state and returns relevant memories plus the active task snapshot, session handoff and recent session trail (the same bundle as GET /v1/context). Use at the start of any session. Memories and those sections are scope-filtered: a no-scope caller does NOT see ANY <source>:private:* (slack, github, ...) or legacy-quarantine rows.',
314
358
  inputSchema: {
315
359
  type: 'object',
316
360
  properties: {
317
- budget: { type: 'number', minimum: 0, description: 'Max tokens (default: config.defaultContextBudget, 3000)' },
361
+ budget: {
362
+ type: 'number',
363
+ minimum: 0,
364
+ maximum: MAX_BUDGET_TOKENS,
365
+ description: `Max tokens (default: config.defaultContextBudget, 3000; max ${MAX_BUDGET_TOKENS})`,
366
+ },
318
367
  scope: {
319
368
  type: 'string',
320
- description: 'Restrict memories and snapshot to this scope exactly. When omitted, default-deny applies to ANY <source>:private:* (slack, github, ...) and unknown-legacy rows.',
369
+ description: 'Restrict memories, snapshot, handoff and trail to this scope exactly. When omitted, default-deny applies to ANY <source>:private:* (slack, github, ...) and unknown-legacy rows.',
321
370
  },
322
371
  },
323
372
  },
@@ -398,6 +447,9 @@ const TOOLS = [
398
447
  },
399
448
  },
400
449
  ];
450
+ const TOOLS_BY_NAME = new Map(TOOLS.map((t) => [t.name, t]));
451
+ // api.retrieve rejects these itself, so MCP and HTTP callers get the same typed error code for the same bad value.
452
+ const ARGS_CHECKED_BY_API = new Map([['hippo_recall', new Set(['scorer_window'])]]);
401
453
  // ── Track last recalled IDs for outcome feedback ──
402
454
  //
403
455
  // Keyed per-client so two HTTP-MCP clients hitting the same tenant cannot
@@ -494,22 +546,11 @@ async function executeTool(name, args, ctx) {
494
546
  const summarizeOverflow = isJsonBoolean(args.summarize_overflow)
495
547
  ? args.summarize_overflow
496
548
  : undefined;
497
- // v1.7.2 T4 — scorer_window: Number-coerce so non-numeric input
498
- // (string 'abc', boolean, etc.) reaches api.recall() and produces
499
- // the same typed RecallContractError(code='invalid_scorer_window')
500
- // as HTTP. Codex CRITICAL[2]: do NOT use `typeof === 'number'` — that
501
- // would silently default-200 on string `"5"` while HTTP 400s on the
502
- // same value. Both transports must agree.
549
+ // Number-coerce, never typeof-check: "abc" must reach api.retrieve and fail as invalid_scorer_window, the same code HTTP returns.
503
550
  const scorerWindow = args.scorer_window === undefined
504
551
  ? undefined
505
552
  : Number(args.scorer_window);
506
- // v1.7.4 -- session_id for the dlPFC goal-stack boost. Mirrors
507
- // fresh_tail_session_id shape: trim, 256-char cap. When set and the
508
- // (tenant, session) has active goals, the boost is applied (a) inside
509
- // api.recall on its primary BM25 band (so the audit + fresh-tail /
510
- // summary appendix paths see consistent ranking), and (b) below on the
511
- // physics/hybrid result list before formatMemories (since MCP's
512
- // user-visible primary ordering does NOT come from api.recall).
553
+ // session_id drives the goal-stack boost inside api.retrieve; same trim and 256-char cap as fresh_tail_session_id.
513
554
  const sessionIdRaw = isJsonString(args.session_id) ? args.session_id.trim() : '';
514
555
  const sessionId = sessionIdRaw.length > 0 && sessionIdRaw.length <= 256
515
556
  ? sessionIdRaw
@@ -519,11 +560,6 @@ async function executeTool(name, args, ctx) {
519
560
  tenantId,
520
561
  actor: mcpActor(ctx),
521
562
  };
522
- // Route through api.recall for audit + (when requested) continuity block.
523
- // api.recall already applies the same default-deny / exact-match rules
524
- // we want here, so its continuity output is the source of truth.
525
- // RecallContractError throws propagate raw to the MCP caller (per the
526
- // v1.6.5 F5 contract documented in mcp-recall-fresh-tail-policy.test.ts).
527
563
  const recallExtra = {};
528
564
  if (freshTailCount !== undefined)
529
565
  recallExtra.freshTailCount = freshTailCount;
@@ -535,161 +571,101 @@ async function executeTool(name, args, ctx) {
535
571
  recallExtra.scorerWindow = scorerWindow;
536
572
  if (sessionId !== undefined)
537
573
  recallExtra.sessionId = sessionId;
538
- const apiResult = apiRecall(apiCtx, {
574
+ const anchorRing = biasHintEnabled('anchoring') && sessionId
575
+ ? getOrCreateRing(sessionRecallHistoryMcp, buildSessionKey(tenantId, sessionId))
576
+ : null;
577
+ const queryHash = hashQueryText(query);
578
+ const out = {};
579
+ // RecallContractError throws reach the MCP caller raw, as mcp-recall-fresh-tail-policy.test.ts pins.
580
+ await apiRetrieve(apiCtx, {
539
581
  query,
540
582
  limit: 50,
541
583
  scope: explicitScope,
542
584
  includeContinuity,
543
- // v1.13.x / J2 — MCP computes its OWN availability hint over the
544
- // physics/hybrid result set below; suppress api.recall's BM25-band copy
545
- // so one MCP recall does not emit recall_availability_detected twice.
585
+ mode: config.physics?.enabled !== false ? 'physics' : 'hybrid',
586
+ // The hint is computed below over the list MCP shows; the window band's copy would emit its audit row twice.
546
587
  suppressAvailabilityHint: true,
547
- // LC1 F2 fix — MCP's user-visible primary ordering comes from the
548
- // physics/hybrid scorer below, NOT this api.recall call's BM25 band
549
- // (see the comment above apiRecall). Tracing this call as pipeline
550
- // 'api' would mislabel training data with ids/ranks/scores the user
551
- // never actually saw. Real MCP tracing is the reserved 'mcp'
552
- // pipeline value (schema v40) — a follow-up, not v1 scope.
553
- suppressRecallTrace: true,
554
588
  keepHeldCopies: true,
555
589
  ...recallExtra,
590
+ showRanked: ({ ranked, pool, droppedByScope }, apiResult) => {
591
+ // Sections are paid in print order, ahead of the memories and after the heading; one that does not fit is dropped whole.
592
+ let left = budget - memoriesReserve(budget);
593
+ const pays = (piece) => {
594
+ const tokens = estimateTokens(piece);
595
+ if (tokens > left)
596
+ return false;
597
+ left -= tokens;
598
+ return true;
599
+ };
600
+ const planPiece = planningSection(apiResult);
601
+ const showPlan = planPiece !== '' && pays(planPiece);
602
+ const tailRows = apiResult.results.filter((r) => r.isFreshTail || r.isSummary);
603
+ const showTail = tailRows.length > 0 && pays(tailSection(tailRows));
604
+ const continuityPiece = includeContinuity && apiResult.continuity ? `\n\n${formatContinuityBlock(apiResult.continuity)}` : '';
605
+ const showContinuity = continuityPiece !== '' && pays(continuityPiece);
606
+ // J1, J2 and C5: the hints and Cutoff block describe the list MCP shows, not the window band in apiResult.
607
+ const render = (cut) => {
608
+ const list = dropHeldCopies(cut, (r) => r.entry); // after every cut, so a merged row cut here never hides its sources
609
+ const anchoring = anchorRing ? detectAnchoring(snapshotRing(anchorRing), queryHash, list[0]?.entry.id ?? null) : null;
610
+ const availability = biasHintEnabled('availability')
611
+ ? detectAvailabilityBias({
612
+ topK: list.map((r) => ({ id: r.entry.id, created: r.entry.created })),
613
+ pool: pool.map((e) => ({ id: e.id, created: e.created })),
614
+ })
615
+ : null;
616
+ const shownIds = new Set(list.map((r) => r.entry.id));
617
+ const shownKeys = storedTextKeys(list.map((r) => r.entry));
618
+ const tail = showTail
619
+ ? dropHeldCopies(tailRows.filter((r) => !shownIds.has(r.id) && !shownKeys.has(duplicateKey(r.content))), (r) => r)
620
+ : [];
621
+ const s = buildSuppressionSummary({
622
+ totalCandidates: pool.length + droppedByScope,
623
+ droppedPreRank: droppedByScope + cut.length - list.length, // the bucket CLI and API recall put hidden copies in
624
+ droppedByBudget: Math.max(0, pool.length - cut.length), // an upper bound: rows that never matched count too
625
+ summarySubstitutionsAdded: tail.filter((r) => r.isSummary).length,
626
+ freshTailAdded: tail.filter((r) => r.isFreshTail && !r.isSummary).length,
627
+ suppressedByInterference: anchoring?.reason === 'memory_dominance' ? 1 : 0,
628
+ });
629
+ // Anchoring is the stronger pull, so it prints first; the Cutoff block sits above the list, where the agent reads it.
630
+ let text = anchoring ? `## Anchoring hint\n${anchoring.summary}\n[anchored_on: ${anchoring.memoryId}]\n\n---\n\n` : '';
631
+ if (availability)
632
+ text += `## Availability bias\n${availability.summary}\n\n---\n\n`;
633
+ if (showPlan)
634
+ text += planPiece;
635
+ const cutoffClauses = [];
636
+ if (s.droppedByBudget > 0)
637
+ cutoffClauses.push(`${s.droppedByBudget} dropped to fit limit`);
638
+ if (s.droppedPreRank > 0)
639
+ cutoffClauses.push(`${s.droppedPreRank} filtered pre-rank`);
640
+ if (s.summarySubstitutionsAdded > 0)
641
+ cutoffClauses.push(`${s.summarySubstitutionsAdded} summary substitutions added`);
642
+ if (s.freshTailAdded > 0)
643
+ cutoffClauses.push(`${s.freshTailAdded} fresh-tail added`);
644
+ if (s.suppressedByInterference > 0)
645
+ cutoffClauses.push(`${s.suppressedByInterference} suppressed by interference`);
646
+ if (cutoffClauses.length > 0) {
647
+ text += `## Cutoff\nShowing ${list.length} of ${s.totalCandidates} candidates; ${cutoffClauses.join('; ')}.\n\n---\n\n`;
648
+ }
649
+ // The window band's fresh-tail and summary rows follow the ranked list, or the MCP fields go unanswered.
650
+ text += formatMemories(list) + tailSection(tail) + (showContinuity ? continuityPiece : '');
651
+ return { anchoring, availability, text, list };
652
+ };
653
+ let results = fitBudget(ranked, Math.max(0, left), 1, memoryCost);
654
+ let rendered = render(results);
655
+ // The hints, Cutoff block and heading vary with the list, so the lowest-ranked entry goes until the whole response fits.
656
+ while (results.length > 1 && estimateTokens(rendered.text) > budget) {
657
+ results = results.slice(0, -1);
658
+ rendered = render(results);
659
+ }
660
+ out.rendered = rendered;
661
+ return rendered.list.map((r) => r.entry.id);
662
+ },
556
663
  });
557
- // Existing physics/hybrid scorer continues to drive user-visible
558
- // ordering and the strength bump on retrieval. Apply the same scope
559
- // rule as api.recall: explicit scope = exact match; no scope =
560
- // default-deny on ANY `<source>:private:*` AND 'unknown:legacy'.
561
- // EI2: one shared predicate (passesScopeFilterForRecall) so MCP never admits what SQL hides.
562
- const allEntries = loadAllEntries(hippoRoot, tenantId);
563
- // v1.12.13 / C5 — WYSIATI counters for the MCP physics/hybrid pipeline.
564
- // Per the plan-eng-critic round 1 CRIT resolution: MCP's user-visible
565
- // memory list comes from THIS pipeline (loadAllEntries -> scope filter
566
- // -> physicsSearch/hybridSearch), NOT from apiResult. The MCP
567
- // suppressionSummary must describe what the user actually sees, so we
568
- // track filter activity here and replace apiResult.suppressionSummary
569
- // in the user-facing response.
570
- const totalCandidatesCountMcp = allEntries.length;
571
- const entries = explicitScope
572
- ? allEntries.filter((e) => e.scope === explicitScope)
573
- : allEntries.filter((e) => passesScopeFilterForRecall(e.scope ?? null, undefined));
574
- const droppedPreRankCountMcp = allEntries.length - entries.length;
575
- // Sections are paid in print order, ahead of the memories and after the heading; one that does not fit is dropped whole.
576
- let left = budget - memoriesReserve(budget);
577
- const pays = (piece) => {
578
- const tokens = estimateTokens(piece);
579
- if (tokens > left)
580
- return false;
581
- left -= tokens;
582
- return true;
583
- };
584
- const planPiece = planningSection(apiResult);
585
- const showPlan = planPiece !== '' && pays(planPiece);
586
- const tailRows = apiResult.results.filter((r) => r.isFreshTail || r.isSummary);
587
- const showTail = tailRows.length > 0 && pays(tailSection(tailRows));
588
- const continuityPiece = includeContinuity && apiResult.continuity ? `\n\n${formatContinuityBlock(apiResult.continuity)}` : '';
589
- const showContinuity = continuityPiece !== '' && pays(continuityPiece);
590
- const usePhysics = config.physics?.enabled !== false;
591
- const fit = { budget: Math.max(0, left), cost: memoryCost, hippoRoot };
592
- let results = usePhysics
593
- ? await physicsSearch(query, entries, { ...fit, physicsConfig: config.physics })
594
- : await hybridSearch(query, entries, fit);
595
- // v1.12.13 / C5 — droppedByBudget for MCP is an UPPER BOUND. The
596
- // difference (entries.length - results.length) lumps three things
597
- // together: rows hybridSearch/physicsSearch internally dropped because
598
- // they scored zero (didn't match the query at all), rows the search
599
- // engine filtered internally (e.g. superseded when --include-
600
- // superseded isn't set), and rows that genuinely didn't fit the
601
- // `budget` token cap. The honest fix needs hybridSearch/physicsSearch
602
- // to expose their pre-budget-cut scored-count. Until then this is an
603
- // upper bound that conflates "not relevant" with "no budget" on
604
- // no-match / sparse-match queries. Plan-eng-critic round 1 MED and
605
- // codex-review-critic P2 both flagged this; documented + tracked as
606
- // a v1.12.14 follow-up. Independent-review-critic and code-review-
607
- // critic both graded as non-blocking for v1.12.13 ship.
608
- // TODO(c5.1): expose scoredCount from hybridSearch/physicsSearch and
609
- // compute droppedByBudget = scoredCount - results.length, with the
610
- // remainder (entries.length - scoredCount) attributed to
611
- // droppedPreRank or a new "noQueryMatch" counter.
612
- const droppedByBudgetFor = (shown) => Math.max(0, entries.length - shown);
613
- // v1.7.4 -- dlPFC goal-stack boost on the MCP physics/hybrid result
614
- // list BEFORE formatMemories. MCP's user-visible primary ordering does
615
- // NOT come from api.recall (apiResult above), so the boost has to run
616
- // here too. Helper signature accepts any { entry, score } shape; the
617
- // physics/hybrid result rows are already in that shape.
618
- if (sessionId !== undefined) {
619
- const dbForBoost = openHippoDb(hippoRoot);
620
- try {
621
- results = applyGoalStackBoost(dbForBoost, results, {
622
- sessionId,
623
- tenantId,
624
- limit: results.length,
625
- });
626
- }
627
- finally {
628
- closeHippoDb(dbForBoost);
629
- }
630
- }
631
- // J1, J2 and C5: MCP ranks its own list (its top-1 can differ from api.recall's), so its hints and Cutoff block are its own.
632
- const anchorRing = process.env.HIPPO_ANCHORING !== 'off' && sessionId
633
- ? getOrCreateRing(sessionRecallHistoryMcp, buildSessionKey(tenantId, sessionId))
634
- : null;
635
- const queryHash = hashQueryText(query);
636
- const render = (cut) => {
637
- const list = dropHeldCopies(cut, (r) => r.entry); // after every cut, so a merged row cut here never hides its sources
638
- const anchoring = anchorRing ? detectAnchoring(snapshotRing(anchorRing), queryHash, list[0]?.entry.id ?? null) : null;
639
- const availability = process.env.HIPPO_AVAILABILITY !== 'off'
640
- ? detectAvailabilityBias({
641
- topK: list.map((r) => ({ id: r.entry.id, created: r.entry.created })),
642
- pool: entries.map((e) => ({ id: e.id, created: e.created })),
643
- })
644
- : null;
645
- const shownIds = new Set(list.map((r) => r.entry.id));
646
- const shownKeys = storedTextKeys(list.map((r) => r.entry));
647
- const tail = showTail
648
- ? dropHeldCopies(tailRows.filter((r) => !shownIds.has(r.id) && !shownKeys.has(duplicateKey(r.content))), (r) => r)
649
- : [];
650
- const s = buildSuppressionSummary({
651
- totalCandidates: totalCandidatesCountMcp,
652
- droppedPreRank: droppedPreRankCountMcp + cut.length - list.length, // the bucket CLI and API recall put hidden copies in
653
- droppedByBudget: droppedByBudgetFor(cut.length),
654
- summarySubstitutionsAdded: tail.filter((r) => r.isSummary).length,
655
- freshTailAdded: tail.filter((r) => r.isFreshTail && !r.isSummary).length,
656
- suppressedByInterference: anchoring?.reason === 'memory_dominance' ? 1 : 0,
657
- });
658
- // Anchoring is the stronger pull, so it prints first; the Cutoff block sits above the list, where the agent reads it.
659
- let text = anchoring ? `## Anchoring hint\n${anchoring.summary}\n[anchored_on: ${anchoring.memoryId}]\n\n---\n\n` : '';
660
- if (availability)
661
- text += `## Availability bias\n${availability.summary}\n\n---\n\n`;
662
- if (showPlan)
663
- text += planPiece;
664
- const cutoffClauses = [];
665
- if (s.droppedByBudget > 0)
666
- cutoffClauses.push(`${s.droppedByBudget} dropped to fit limit`);
667
- if (s.droppedPreRank > 0)
668
- cutoffClauses.push(`${s.droppedPreRank} filtered pre-rank`);
669
- if (s.summarySubstitutionsAdded > 0)
670
- cutoffClauses.push(`${s.summarySubstitutionsAdded} summary substitutions added`);
671
- if (s.freshTailAdded > 0)
672
- cutoffClauses.push(`${s.freshTailAdded} fresh-tail added`);
673
- if (s.suppressedByInterference > 0)
674
- cutoffClauses.push(`${s.suppressedByInterference} suppressed by interference`);
675
- if (cutoffClauses.length > 0) {
676
- text += `## Cutoff\nShowing ${list.length} of ${s.totalCandidates} candidates; ${cutoffClauses.join('; ')}.\n\n---\n\n`;
677
- }
678
- // v1.6.3: the fresh-tail and summary rows api.recall produced follow the ranked list, or the MCP fields go unanswered.
679
- text += formatMemories(list) + tailSection(tail) + (showContinuity ? continuityPiece : '');
680
- return { anchoring, availability, text, list };
681
- };
682
- let rendered = render(results);
683
- // The hints, Cutoff block and heading vary with the list, so the lowest-ranked entry goes until the whole response fits.
684
- while (results.length > 1 && estimateTokens(rendered.text) > budget) {
685
- results = results.slice(0, -1);
686
- rendered = render(results);
687
- }
688
- const { anchoring: mcpAnchoringHint, availability: mcpAvailabilityHint, list: shown } = rendered;
689
- const retrievedIds = shown.map((r) => r.entry.id);
690
- strengthenRetrieved(hippoRoot, retrievedIds);
691
- lastRecalledIds.set(resolveClientKey(ctx), retrievedIds);
692
- if (process.env.HIPPO_ANCHORING !== 'off') {
664
+ if (!out.rendered)
665
+ throw new Error('hippo_recall: api.retrieve returned without calling showRanked');
666
+ const { anchoring: mcpAnchoringHint, availability: mcpAvailabilityHint, list: shown, text: recallText } = out.rendered;
667
+ lastRecalledIds.set(resolveClientKey(ctx), shown.map((r) => r.entry.id));
668
+ if (biasHintEnabled('anchoring')) {
693
669
  if (anchorRing) {
694
670
  // Appended after the final detect: anchoredOn feeds the cooldown for the next recall on this session.
695
671
  appendRecall(anchorRing, queryHash, shown[0]?.entry.id ?? null, mcpAnchoringHint?.memoryId);
@@ -766,7 +742,7 @@ async function executeTool(name, args, ctx) {
766
742
  closeHippoDb(dbForAudit);
767
743
  }
768
744
  }
769
- return rendered.text;
745
+ return recallText;
770
746
  }
771
747
  case 'hippo_assemble': {
772
748
  const sessionId = String(args.session_id || '');
@@ -803,17 +779,8 @@ async function executeTool(name, args, ctx) {
803
779
  return 'No summary_id provided.';
804
780
  const limit = Number(args.limit);
805
781
  const budget = Number(args.budget);
806
- // v0.30 / E5: depth walks N levels (default 1, hard cap 10).
807
- // independent-review MED #5 fold: reject out-of-range explicitly
808
- // (no silent clamp) so MCP callers see the constraint at their layer.
809
- let depth;
810
- if (args.depth !== undefined) {
811
- const depthRaw = Number(args.depth);
812
- if (!Number.isInteger(depthRaw) || depthRaw < 1 || depthRaw > 10) {
813
- return `depth must be an integer between 1 and 10 (got ${args.depth})`;
814
- }
815
- depth = depthRaw;
816
- }
782
+ // The inputSchema rejects a depth outside 1..10 before this runs, so no silent clamp hides the cap.
783
+ const depth = args.depth === undefined ? undefined : Number(args.depth);
817
784
  const apiCtx = {
818
785
  hippoRoot,
819
786
  tenantId,
@@ -905,7 +872,8 @@ async function executeTool(name, args, ctx) {
905
872
  }
906
873
  const halfLife = entry?.half_life_days ?? config.defaultHalfLifeDays;
907
874
  const tagStr = entry?.tags.join(', ') || tags.join(', ') || 'none';
908
- return `Remembered [${result.id}] (half-life: ${halfLife}d, tags: ${tagStr})`;
875
+ const warnings = (result.warnings ?? []).map((w) => `\nWarning: ${w}`).join('');
876
+ return `Remembered [${result.id}] (half-life: ${halfLife}d, tags: ${tagStr})${warnings}`;
909
877
  }
910
878
  case 'hippo_outcome': {
911
879
  const good = Boolean(args.good);
@@ -933,94 +901,25 @@ async function executeTool(name, args, ctx) {
933
901
  return 'budget must be a non-negative number.';
934
902
  if (budget === 0)
935
903
  return '';
936
- const explicitScope = isJsonString(args.scope) && args.scope.length > 0
904
+ if (budget < memoriesReserve(budget))
905
+ return ''; // not even the heading fits, so nothing prints, as at budget 0
906
+ const exactScope = isJsonString(args.scope) && args.scope.length > 0
937
907
  ? args.scope
938
908
  : undefined;
939
- // Auto-detect query from git
940
- let query = '';
941
- try {
942
- const branch = execSync('git rev-parse --abbrev-ref HEAD 2>/dev/null', { encoding: 'utf-8', windowsHide: true }).trim();
943
- const diff = execSync('git diff --cached --stat 2>/dev/null', { encoding: 'utf-8', windowsHide: true }).trim();
944
- const log = execSync('git log -1 --pretty=format:"%s" 2>/dev/null', { encoding: 'utf-8', windowsHide: true }).trim();
945
- query = [branch, log, diff].filter(Boolean).join(' ');
946
- }
947
- catch { /* not a git repo */ }
948
- if (!query)
949
- query = 'project context general';
950
- // v1.2 codex audit: same scope filter as hippo_recall on BOTH the memory
951
- // results and the snapshot. Pre-v1.2 this surface returned all memories
952
- // and the snapshot unfiltered, which would have leaked private-channel
953
- // content to no-scope MCP callers once scope writers shipped.
954
- assertScopeRequestAllowed(mcpActor(ctx), explicitScope);
955
- const allEntries = loadAllEntries(hippoRoot, tenantId);
956
- // v39 memory scope isolation: this surface reads the LOCAL store only,
957
- // but synced-down or legacy rows can still carry another project's
958
- // origin, and secrets must never ambient-inject outside their owner.
959
- // Same policy as api.getContext; the scope filter above keeps this
960
- // surface's own explicit-scope exact-match semantics.
961
- //
962
- // Identity resolution handles both transports (codex rounds 4+5):
963
- // - The SERVED store is authoritative when it is a project store -
964
- // an HTTP /mcp daemon launched from anywhere still isolates the
965
- // project it serves.
966
- // - When the served store is the global root (stdio in a git repo
967
- // with no local .hippo falls back to it), dirname(store) is home
968
- // ('' would admit everything), so fall back to the launch cwd -
969
- // stdio servers launch in the project they serve.
970
- const mcpStoreIdentity = resolveProjectIdentity(path.dirname(path.resolve(hippoRoot)));
971
- const mcpProjectName = mcpStoreIdentity.name !== ''
972
- ? mcpStoreIdentity.name
973
- : resolveProjectIdentity(process.cwd()).name;
974
- const isolationOff = config.contextProjectIsolation === false;
975
- const entries = allEntries.filter((e) => {
976
- if (!passesScopeFilterForRecall(e.scope ?? null, explicitScope))
977
- return false;
978
- if (!ambientSecretAdmit(e, mcpProjectName))
979
- return false;
980
- if (isolationOff)
981
- return true;
982
- return classifyOriginProject(e.origin_project, mcpProjectName) !== 'cross-project';
909
+ // The served store names the project (an HTTP daemon runs from anywhere); the global root names none, so stdio falls back to its launch cwd.
910
+ const storeProject = resolveProjectIdentity(path.dirname(path.resolve(hippoRoot))).name;
911
+ const result = await apiGetContext({ hippoRoot, tenantId, actor: mcpActor(ctx) }, {
912
+ q: autoDetectContext(),
913
+ budget,
914
+ exactScope,
915
+ currentProject: storeProject !== '' ? storeProject : resolveProjectIdentity(process.cwd()).name,
916
+ cost: contextCost,
983
917
  });
984
- // DF1 (docs/plans/2026-08-23-df1-snapshot-lifecycle.md, T2): bounded
985
- // read, no session id available on this surface (freshness bound
986
- // only) — an orphaned snapshot must age out here too, not just on the
987
- // UserPromptSubmit path.
988
- const rawSnapshot = loadFreshActiveTaskSnapshot(hippoRoot, tenantId);
989
- const snapshot = rawSnapshot && passesScopeFilterForRecall(rawSnapshot.scope, explicitScope)
990
- ? rawSnapshot
991
- : null;
992
- const snapshotText = snapshot
993
- ? [
994
- '## Active Task Snapshot',
995
- `- Task: ${snapshot.task}`,
996
- `- Status: ${snapshot.status}`,
997
- `- Updated: ${snapshot.updated_at}`,
998
- '',
999
- '### Summary',
1000
- snapshot.summary,
1001
- '',
1002
- '### Next step',
1003
- snapshot.next_step,
1004
- '',
1005
- ].join('\n')
1006
- : '';
1007
- // The snapshot prints first, so it is paid first after the heading; context keeps no hit past the budget, even the top one.
1008
- let left = budget - memoriesReserve(budget);
1009
- if (left < 0)
1010
- return ''; // not even the heading fits, so nothing prints, as at budget 0
1011
- const snapshotPiece = snapshotText ? `${snapshotText}\n` : '';
1012
- const showSnapshot = snapshotPiece !== '' && estimateTokens(snapshotPiece) <= left;
1013
- if (showSnapshot)
1014
- left -= estimateTokens(snapshotPiece);
1015
- const usePhysicsCtx = config.physics?.enabled !== false;
1016
- const fit = { budget: left, minResults: 0, cost: memoryCost, hippoRoot };
1017
- const results = dropHeldCopies(usePhysicsCtx
1018
- ? await physicsSearch(query, entries, { ...fit, physicsConfig: config.physics })
1019
- : await hybridSearch(query, entries, fit), (r) => r.entry);
1020
- const retrievedIds = results.map((r) => r.entry.id);
1021
- strengthenRetrieved(hippoRoot, retrievedIds);
1022
- lastRecalledIds.set(resolveClientKey(ctx), retrievedIds);
1023
- return (showSnapshot ? snapshotPiece : '') + formatMemories(results);
918
+ lastRecalledIds.set(resolveClientKey(ctx), result.entries.map((r) => r.entry.id));
919
+ return (result.activeSnapshot ? snapshotPiece(result.activeSnapshot) : '')
920
+ + (result.sessionHandoff ? handoffPiece(result.sessionHandoff) : '')
921
+ + (result.recentEvents ? trailPiece(result.recentEvents) : '')
922
+ + formatMemories(result.entries);
1024
923
  }
1025
924
  case 'hippo_status': {
1026
925
  const entries = loadAllEntries(hippoRoot, tenantId);
@@ -1155,7 +1054,8 @@ async function executeTool(name, args, ctx) {
1155
1054
  return peers.map((p) => `${p.project}: ${p.count} memories (latest: ${p.latest.slice(0, 10)})`).join('\n');
1156
1055
  }
1157
1056
  default:
1158
- return `Unknown tool: ${name}`;
1057
+ // handleMcpRequest rejects names missing from TOOLS, so reaching here means TOOLS and this switch drifted apart.
1058
+ throw new Error(`hippo-mcp: tool ${name} is declared but has no handler`);
1159
1059
  }
1160
1060
  }
1161
1061
  // ── Request handling ──
@@ -1186,8 +1086,24 @@ export async function handleMcpRequest(req, ctx) {
1186
1086
  case 'tools/call': {
1187
1087
  const nameValue = params?.name;
1188
1088
  const toolName = isJsonString(nameValue) ? nameValue : '';
1089
+ const tool = TOOLS_BY_NAME.get(toolName);
1090
+ if (!tool) {
1091
+ return { jsonrpc: '2.0', id, error: { code: -32602, message: `Unknown tool: ${toolName.slice(0, 128)}` } };
1092
+ }
1189
1093
  const argumentsValue = params?.arguments;
1094
+ if (argumentsValue !== undefined && argumentsValue !== null && !isJsonObjectRecord(argumentsValue)) {
1095
+ return { jsonrpc: '2.0', id, error: { code: -32602, message: `${toolName}: arguments must be an object` } };
1096
+ }
1190
1097
  const toolArgs = isJsonObjectRecord(argumentsValue) ? argumentsValue : {};
1098
+ // The MCP spec reports input validation as a tool result with isError, so the model can read it and retry.
1099
+ const problems = validateToolArgs(tool.inputSchema, toolArgs, ARGS_CHECKED_BY_API.get(toolName));
1100
+ if (problems.length > 0) {
1101
+ return {
1102
+ jsonrpc: '2.0',
1103
+ id,
1104
+ result: { content: [{ type: 'text', text: `Invalid arguments for ${toolName}: ${problems.join('; ')}` }], isError: true },
1105
+ };
1106
+ }
1191
1107
  const output = await executeTool(toolName, toolArgs, ctx);
1192
1108
  recordMcpTokens(toolName, output, ctx);
1193
1109
  return {