@jasonwch/nodebb-plugin-meilisearch-r 1.1.2 → 1.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/client.js ADDED
@@ -0,0 +1,497 @@
1
+ 'use strict';
2
+
3
+ const winston = nodebb.require('winston');
4
+ const settings = nodebb.require('./src/meta/settings');
5
+ const Topics = nodebb.require('./src/topics');
6
+ const { drainPending } = require('./pending-queue');
7
+ const {
8
+ EMBEDDER_NAME, buildEmbedderConfig, validateEmbedderConfig, deepEqual,
9
+ getAppliedConfig, getAppliedTaskUid, setAppliedConfigs, clearAppliedConfig,
10
+ } = require('./embedder');
11
+ const {
12
+ EMBEDDER_TASK_TIMEOUT_MS, EMBEDDER_TASK_BG_TIMEOUT_MS, EMBEDDER_TASK_BG_INTERVAL_MS,
13
+ MEILI_HTTP_TIMEOUT_MS,
14
+ } = require('./constants');
15
+ const { redactSecrets } = require('./redact');
16
+
17
+ async function ensureIndex(plugin, uid, primaryKey) {
18
+ try {
19
+ await plugin.client.getIndex(uid);
20
+ } catch (e) {
21
+ await plugin.client.createIndex(uid, { primaryKey });
22
+ }
23
+ }
24
+
25
+ // Meilisearch settings/document mutations are enqueued tasks - the initial call resolving
26
+ // just means Meilisearch accepted the request, not that it succeeded. waitForTask() itself
27
+ // only ever throws on ITS OWN polling timeout (confirmed against the installed
28
+ // meilisearch-js's TaskClient#waitForTask, node_modules/meilisearch/dist/index.js): a task
29
+ // that fails on the server resolves normally with status "failed" and an `error` object, so
30
+ // callers that don't check `.status` (as every waitForTask call in this plugin used to)
31
+ // silently treat a failed settings/document change as if it had gone through.
32
+ async function waitForSucceededTask(plugin, taskUidOrTask, options) {
33
+ const task = await plugin.client.tasks.waitForTask(taskUidOrTask, options);
34
+ if (task.status !== 'succeeded') {
35
+ throw new Error(task.error?.message || `Meilisearch task ${task.status}`);
36
+ }
37
+ return task;
38
+ }
39
+
40
+ // Module-level tracker for in-flight embedder background pollers. Keyed by taskUid so the
41
+ // same task can't be polled twice (e.g. if updateEmbedders is somehow re-entered during
42
+ // the 60s sync wait). Cleared on terminal status (success or failure) by the poller itself.
43
+ const embedderPollers = new Set();
44
+
45
+ // Background-verify an embedder task that exceeded the sync wait window. Per the Meilisearch
46
+ // SDK (node_modules/meilisearch/dist/index.js:152), a poll timeout throws
47
+ // MeilisearchTaskTimeOutError — task is still running server-side. Re-embedding every
48
+ // document legitimately exceeds 60s on a forum of any size; the task may yet succeed.
49
+ // Strategy: poll at a longer interval with a longer timeout; alert only on delayed failure.
50
+ // On delayed success: update the persisted taskUid to null (sync-verified) so reattachPollers
51
+ // skips it on future restarts instead of hitting a 404 when Meilisearch purges task history.
52
+ // On delayed failure: revert the optimistic fingerprint via clearAppliedConfig so the next
53
+ // save pushes fresh config, alert admins, flip plugin.healthy=false to defer subsequent
54
+ // writes to the pending queue.
55
+ // On poll timeout (MeilisearchTaskTimeOutError): task is STILL RUNNING — do NOT clear the
56
+ // fingerprint (would re-open the double-paid-API-cost loop). Log and let the task continue
57
+ // server-side; if it ultimately fails, the next restart's reattachPollers will catch it.
58
+ // On network error: task status is genuinely unknown — log warning but do NOT clear, since
59
+ // clearing on a transient blip would cause an unnecessary re-push. reattachPollers catches
60
+ // actual failures on the next restart.
61
+ function startEmbedderPoller(plugin, plan, taskUid) {
62
+ if (embedderPollers.has(taskUid)) return;
63
+ embedderPollers.add(taskUid);
64
+ const startTime = Date.now();
65
+ const poll = async () => {
66
+ try {
67
+ const task = await plugin.client.tasks.waitForTask(taskUid, {
68
+ timeout: EMBEDDER_TASK_BG_TIMEOUT_MS,
69
+ interval: EMBEDDER_TASK_BG_INTERVAL_MS,
70
+ });
71
+ embedderPollers.delete(taskUid);
72
+ if (task.status !== 'succeeded') {
73
+ // C1c fix: redact any apiKey that Meilisearch may echo in task.error.message before
74
+ // sending to admin toasts. Issue 2 admin-room filter already limits toast recipients,
75
+ // but defense-in-depth handles the case where Meilisearch wraps the apiKey in the error.
76
+ // A3-style inner try/catch: if redaction itself fails, fall back to raw reason rather
77
+ // than crashing the poller's failure-handling path.
78
+ const rawReason = task.error?.message || `task ${task.status}`;
79
+ let safeReason;
80
+ try {
81
+ safeReason = await redactSecrets(rawReason, plugin);
82
+ } catch {
83
+ safeReason = rawReason;
84
+ }
85
+ await clearAppliedConfig(plugin, plan.indexName);
86
+ plugin.healthy = false;
87
+ plugin.notifyAdmins(`embedder:${plan.indexName}`, {
88
+ type: 'danger',
89
+ titleKey: '[[meilisearch:admin.semanticEmbedderDelayedFailed]]',
90
+ message: `${plan.indexName}: ${safeReason}`,
91
+ });
92
+ } else {
93
+ // Delayed success: mark taskUid as null (sync-verified) so reattachPollers
94
+ // skips this index on future restarts instead of hitting a 404 when Meilisearch
95
+ // purges task history (~24h). Fingerprint config is already correct — no change.
96
+ winston.info(`[plugin/meilisearch] Embedder task for "${plan.indexName}" succeeded after ${Date.now() - startTime}ms (background-poll)`);
97
+ await setAppliedConfigs(plugin, [{ indexName: plan.indexName, config: plan.config, taskUid: null }]);
98
+ }
99
+ } catch (err) {
100
+ embedderPollers.delete(taskUid);
101
+ if (err.name === 'MeilisearchTaskTimeOutError') {
102
+ // Poller's own 30-min timeout fired — task is STILL RUNNING server-side.
103
+ // Do NOT clear the fingerprint: that would make the next save re-push identical
104
+ // config, re-opening the double-paid-API-cost loop this whole mechanism prevents.
105
+ // The task will eventually succeed or fail; if NodeBB restarts before that,
106
+ // reattachPollers will pick it up via the persisted taskUid.
107
+ winston.warn(`[plugin/meilisearch] Embedder task ${taskUid} for "${plan.indexName}" still running after ${EMBEDDER_TASK_BG_TIMEOUT_MS}ms; giving up poll. Task continues server-side — reattachPollers will reconcile on next restart.`);
108
+ return;
109
+ }
110
+ // Network error or other non-timeout failure — task status is genuinely unknown.
111
+ // C1 fix: redact any apiKey echoed in err.message before logging.
112
+ // A3-style inner try/catch: fall back to raw err.message if redaction itself fails.
113
+ const rawPollerErr = err.message;
114
+ let safePollerErr;
115
+ try {
116
+ safePollerErr = await redactSecrets(rawPollerErr, plugin);
117
+ } catch {
118
+ safePollerErr = rawPollerErr;
119
+ }
120
+ winston.warn(`[plugin/meilisearch] Embedder poller for "${plan.indexName}" (task ${taskUid}) hit an error (non-fatal, fingerprint preserved): ${safePollerErr}`);
121
+ }
122
+ };
123
+ poll().catch(async err => {
124
+ // A3: inner try/catch makes the redaction layer airtight against future regressions
125
+ // where redactSecrets itself might throw (currently fully guarded internally, but
126
+ // defense-in-depth ensures the log path still runs).
127
+ let safeCrashErr;
128
+ try {
129
+ safeCrashErr = await redactSecrets(err.message, plugin);
130
+ } catch {
131
+ safeCrashErr = '<redaction failed>';
132
+ }
133
+ winston.error(`[plugin/meilisearch] embedder poller crashed: ${safeCrashErr}`);
134
+ });
135
+ }
136
+
137
+ // Recover embedder pollers after process restart. embedderPollers (line 42) is module-level
138
+ // in-memory; a NodeBB restart/crash/container-redeploy during the 30-min background-poller
139
+ // window (EMBEDDER_TASK_BG_TIMEOUT_MS) wipes it. This sweep reads the persisted taskUid
140
+ // (stored alongside each fingerprint in appliedEmbedders via setAppliedConfigs) and
141
+ // reconciles directly via plugin.client.tasks.getTask(uid):
142
+ // succeeded → no-op (fingerprint correct, taskUid was null from sync-success path)
143
+ // enqueued/processing → reattach startEmbedderPoller (task still running server-side)
144
+ // failed → clearAppliedConfig + alert (Scenario A: task failed during downtime)
145
+ // 404/network error → assume success, log warning (self-heals on next config change)
146
+ // For old-format fingerprints (no taskUid), falls through to query-based sweep as fallback.
147
+ async function reattachPollers(plugin) {
148
+ if (!plugin.client) return;
149
+ const ourIndexes = ['post', 'topic', 'chat_message'];
150
+ // Phase 1: Direct reconciliation via persisted taskUid (new-format fingerprints)
151
+ for (const indexName of ourIndexes) {
152
+ try {
153
+ const taskUid = await getAppliedTaskUid(plugin, indexName);
154
+ if (taskUid === undefined || taskUid === null) continue;
155
+ const task = await plugin.client.tasks.getTask(taskUid);
156
+ if (task.status === 'succeeded') continue;
157
+ if (task.status === 'enqueued' || task.status === 'processing') {
158
+ if (embedderPollers.has(taskUid)) continue;
159
+ const config = await buildEmbedderConfig(plugin, null, indexName);
160
+ winston.info(`[plugin/meilisearch] Reattaching embedder poller for "${indexName}" (task ${taskUid} still ${task.status})`);
161
+ startEmbedderPoller(plugin, { indexName, config }, taskUid);
162
+ continue;
163
+ }
164
+ // Task failed during downtime — clear stale fingerprint + alert
165
+ // C1b fix: redact any apiKey from task.error.message before logging + alerting.
166
+ // A3-style inner try/catch: fall back to raw reason if redaction itself fails.
167
+ const rawFailReason = task.error?.message || `task ${task.status}`;
168
+ let safeFailReason;
169
+ try {
170
+ safeFailReason = await redactSecrets(rawFailReason, plugin);
171
+ } catch {
172
+ safeFailReason = rawFailReason;
173
+ }
174
+ winston.warn(`[plugin/meilisearch] Embedder task ${taskUid} for "${indexName}" failed during downtime: ${safeFailReason}`);
175
+ await clearAppliedConfig(plugin, indexName);
176
+ plugin.healthy = false;
177
+ plugin.notifyAdmins(`embedder:${indexName}`, {
178
+ type: 'danger',
179
+ titleKey: '[[meilisearch:admin.semanticEmbedderDelayedFailed]]',
180
+ message: `${indexName}: ${safeFailReason}`,
181
+ });
182
+ } catch (err) {
183
+ // Non-fatal: taskUid may be 404 (purged from Meilisearch's ~24h history), or
184
+ // Meilisearch may be momentarily unreachable. Assume success — self-heals on next
185
+ // legitimate semantic-key change.
186
+ winston.warn(`[plugin/meilisearch] reattachPollers: could not reconcile taskUid for "${indexName}" (non-fatal): ${err.message}`);
187
+ }
188
+ }
189
+ // Phase 2: Query-based fallback for old-format fingerprints (no taskUid persisted).
190
+ // Catches in-flight embedder tasks that have a persisted fingerprint but no taskUid
191
+ // (pre-Option-1 format). Will be a no-op once all fingerprints are migrated to new format.
192
+ try {
193
+ const tasksResp = await plugin.client.tasks.getTasks({
194
+ statuses: ['enqueued', 'processing'],
195
+ types: ['settingsUpdate'],
196
+ limit: 50,
197
+ });
198
+ const candidates = [];
199
+ for (const t of (tasksResp.results || [])) {
200
+ if (!ourIndexes.includes(t.indexUid)) continue;
201
+ if (embedderPollers.has(t.uid)) continue;
202
+ if (!(t.details && t.details.embedders)) continue;
203
+ const tu = await getAppliedTaskUid(plugin, t.indexUid);
204
+ if (tu !== undefined) continue; // new-format fingerprint — already reconciled in Phase 1
205
+ candidates.push(t);
206
+ }
207
+ if (!candidates.length) return;
208
+ winston.info(`[plugin/meilisearch] Query fallback: reattaching ${candidates.length} embedder poller(s) for old-format fingerprints`);
209
+ for (const task of candidates) {
210
+ const applied = await getAppliedConfig(plugin, task.indexUid);
211
+ if (applied === undefined) continue;
212
+ const config = await buildEmbedderConfig(plugin, null, task.indexUid);
213
+ startEmbedderPoller(plugin, { indexName: task.indexUid, config }, task.uid);
214
+ }
215
+ } catch (err) {
216
+ winston.warn(`[plugin/meilisearch] reattachPollers query fallback failed (non-fatal): ${err.message}`);
217
+ }
218
+ }
219
+
220
+ // MeiliSearch connection lifecycle: connecting, health checks, and index/settings sync.
221
+ module.exports = function attachClient(plugin) {
222
+ plugin.patchTopicsSearch = function () {
223
+ if (Topics.__meiliOriginalSearch) { return; }
224
+ Topics.__meiliOriginalSearch = Topics.search;
225
+ Topics.search = async function (tid, term) {
226
+ if (!tid || !term || !String(term).trim()) { return []; }
227
+ return Topics.__meiliOriginalSearch.call(Topics, tid, term);
228
+ };
229
+ };
230
+
231
+ plugin.checkHealth = async function () {
232
+ try {
233
+ plugin.healthy = await plugin.client.isHealthy();
234
+ if (!plugin.healthy) {
235
+ winston.warn('[plugin/meilisearch] MeiliSearch host is unhealthy');
236
+ } else {
237
+ // B2b: SDK write/health error path — drainPending replay errors may contain
238
+ // document content snippets from already-indexed posts (rare; only if
239
+ // Meilisearch rejects a batch quoting the bad doc). Accepted risk for
240
+ // low-frequency replay failures; document content is already-in-DB data.
241
+ drainPending(plugin).catch(err => winston.error(`[plugin/meilisearch] drain failed: ${err.message}`));
242
+ }
243
+ return plugin.healthy;
244
+ } catch (err) {
245
+ plugin.healthy = false;
246
+ winston.warn(`[plugin/meilisearch] ${err.message}`);
247
+ return false;
248
+ }
249
+ };
250
+
251
+ // Ensure the chat_message index + filterable attributes exist, independent of reindex state.
252
+ // prepareSearch() returns early when `indexed=true`, so updateIndexSettings (which configures
253
+ // chat_message) is skipped on normal restarts — but indexMessage auto-creates the index via
254
+ // document add, leaving filterableAttributes empty. Without roomId/uid filterable, searchMessages
255
+ // would error on its filter clause. Idempotent + cheap (2 calls, settings deduped by Meilisearch).
256
+ plugin.ensureChatMessageIndex = async function () {
257
+ if (!plugin.client) return;
258
+ try {
259
+ await ensureIndex(plugin, 'chat_message', 'mid');
260
+ await plugin.client.index('chat_message').updateFilterableAttributes(['roomId', 'uid', 'timestamp']);
261
+ } catch (err) {
262
+ // B2b: SDK settings error path — updateFilterableAttributes errors don't typically
263
+ // echo credentials or document content; accepted risk for low-frequency settings failures.
264
+ winston.error(`[plugin/meilisearch] ensureChatMessageIndex failed: ${err.message}`);
265
+ plugin.healthy = false;
266
+ }
267
+ };
268
+
269
+ plugin.prepareSearch = async function (data, connectionChanged = false) {
270
+ winston.debug(`[plugin/meilisearch] Connecting to MeiliSearch host: ${data?.host || await settings.getOne(plugin.id, 'host')}`);
271
+ const { Meilisearch } = await import('meilisearch');
272
+ plugin.client = new Meilisearch({
273
+ host: data?.host || await settings.getOne(plugin.id, 'host'),
274
+ apiKey: data?.apiKey || await settings.getOne(plugin.id, 'apiKey') || undefined,
275
+ timeout: MEILI_HTTP_TIMEOUT_MS,
276
+ });
277
+ if (plugin.healthCheckTask) clearInterval(plugin.healthCheckTask);
278
+ const intervalRaw = parseInt(data?.healthCheckInterval || await settings.getOne(plugin.id, 'healthCheckInterval'), 10);
279
+ const intervalSeconds = Number.isFinite(intervalRaw) && intervalRaw > 0
280
+ ? Math.min(Math.max(intervalRaw, 10), 3600)
281
+ : 60;
282
+ plugin.healthCheckTask = setInterval(plugin.checkHealth, intervalSeconds * 1000);
283
+ // #15: Always sync plugin.healthy on (re)connect — otherwise it stays false from
284
+ // initialization until the first setInterval tick, causing stale "MeiliSearch
285
+ // Unreachable" warnings on save during the startup window (MS online, flag stale).
286
+ await plugin.checkHealth();
287
+ // G1: Recover in-flight embedder pollers after process restart (runs once on startup,
288
+ // not on every health-check tick — the sweep is unnecessary after the first run since
289
+ // embedderPollers Set is populated and deduplicates subsequent calls).
290
+ reattachPollers(plugin).catch(err => winston.warn(`[plugin/meilisearch] reattachPollers failed: ${err.message}`));
291
+ // #14: Don't auto-reindex after a failed reindex unless connection settings changed.
292
+ const indexed = await settings.getOne(plugin.id, 'indexed');
293
+ if (indexed) return;
294
+ if (!plugin.healthy || plugin.initializingOnAnotherInstance) return;
295
+ const lastResult = await settings.getOne(plugin.id, 'lastReindexResult') || {};
296
+ const allowAutoReindex = connectionChanged || lastResult.success !== false;
297
+ if (!allowAutoReindex) {
298
+ winston.warn('[plugin/meilisearch] Skipping auto-reindex: last reindex failed. Trigger manually from ACP.');
299
+ return;
300
+ }
301
+ await plugin.updateIndexSettings();
302
+ await plugin.updateEmbedders();
303
+ await plugin.reindex(false);
304
+ };
305
+
306
+ // Keyword-search index settings only (ranking rules, stop words, typo tolerance,
307
+ // synonyms, pagination). Deliberately does NOT touch embedders - see updateEmbedders
308
+ // below for why that has to stay a separate, independently-gated call.
309
+ plugin.updateIndexSettings = async (data) => {
310
+ await ensureIndex(plugin, 'post', 'pid');
311
+ await ensureIndex(plugin, 'topic', 'tid');
312
+ await ensureIndex(plugin, 'chat_message', 'mid');
313
+ data = {
314
+ maxDocuments: parseInt(data?.maxDocuments || await settings.getOne(plugin.id, 'maxDocuments') || 500, 10),
315
+ rankingRules: (data?.rankingRules || await settings.getOne(plugin.id, 'rankingRules'))?.map(value => value.rule),
316
+ stopWords: (data?.stopWords || await settings.getOne(plugin.id, 'stopWords'))?.map(value => value.word),
317
+ typoTolerance: ['on', true].includes(
318
+ data?.typoTolerance || await settings.getOne(plugin.id, 'typoTolerance') || undefined,
319
+ ),
320
+ typoToleranceMinWordSizeOneTypo: parseInt(
321
+ data?.typoToleranceMinWordSizeOneTypo ||
322
+ await settings.getOne(plugin.id, 'typoToleranceMinWordSizeOneTypo') || 5,
323
+ 10,
324
+ ),
325
+ typoToleranceMinWordSizeTwoTypos: parseInt(
326
+ data?.typoToleranceMinWordSizeTwoTypos ||
327
+ await settings.getOne(plugin.id, 'typoToleranceMinWordSizeTwoTypos') || 9,
328
+ 10,
329
+ ),
330
+ typoToleranceDisableOnWords:
331
+ (data?.typoToleranceDisableOnWords || await settings.getOne(plugin.id, 'typoToleranceDisableOnWords') ||
332
+ undefined)?.map(value => value.word),
333
+ synonyms: Object.fromEntries(
334
+ (data?.synonyms || await settings.getOne(plugin.id, 'synonyms') || [])?.map((
335
+ { word, synonyms },
336
+ ) => [word, synonyms?.split(',').map(synonym => synonym.trim())]),
337
+ ),
338
+ };
339
+ const postTask = await plugin.client.index('post').updateSettings({
340
+ filterableAttributes: ['tid', 'cid', 'uid', 'timestamp'],
341
+ sortableAttributes: ['timestamp', 'cid'],
342
+ searchableAttributes: ['content'],
343
+ pagination: {
344
+ maxTotalHits: data.maxDocuments,
345
+ },
346
+ rankingRules: data.rankingRules,
347
+ stopWords: data.stopWords,
348
+ typoTolerance: {
349
+ enabled: data.typoTolerance,
350
+ minWordSizeForTypos: {
351
+ oneTypo: data.typoToleranceMinWordSizeOneTypo,
352
+ twoTypos: data.typoToleranceMinWordSizeTwoTypos,
353
+ },
354
+ disableOnWords: data.typoToleranceDisableOnWords,
355
+ },
356
+ synonyms: data.synonyms,
357
+ });
358
+ const topicTask = await plugin.client.index('topic').updateSettings({
359
+ filterableAttributes: ['cid', 'uid', 'timestamp'],
360
+ sortableAttributes: ['cid', 'title', 'timestamp'],
361
+ searchableAttributes: ['title'],
362
+ pagination: {
363
+ maxTotalHits: data.maxDocuments,
364
+ },
365
+ rankingRules: data.rankingRules,
366
+ stopWords: data.stopWords,
367
+ typoTolerance: {
368
+ enabled: data.typoTolerance,
369
+ minWordSizeForTypos: {
370
+ oneTypo: data.typoToleranceMinWordSizeOneTypo,
371
+ twoTypos: data.typoToleranceMinWordSizeTwoTypos,
372
+ },
373
+ disableOnWords: data.typoToleranceDisableOnWords,
374
+ },
375
+ synonyms: data.synonyms,
376
+ });
377
+ const messageTask = await plugin.client.index('chat_message').updateSettings({
378
+ filterableAttributes: ['roomId', 'uid', 'timestamp'],
379
+ sortableAttributes: ['timestamp'],
380
+ searchableAttributes: ['content'],
381
+ pagination: {
382
+ maxTotalHits: data.maxDocuments,
383
+ },
384
+ rankingRules: data.rankingRules,
385
+ stopWords: data.stopWords,
386
+ typoTolerance: {
387
+ enabled: data.typoTolerance,
388
+ minWordSizeForTypos: {
389
+ oneTypo: data.typoToleranceMinWordSizeOneTypo,
390
+ twoTypos: data.typoToleranceMinWordSizeTwoTypos,
391
+ },
392
+ disableOnWords: data.typoToleranceDisableOnWords,
393
+ },
394
+ synonyms: data.synonyms,
395
+ });
396
+ return [postTask, topicTask, messageTask];
397
+ };
398
+
399
+ // Pushes (or removes) the "default" embedder on all three indexes based on the
400
+ // semantic search ACP settings. Meilisearch re-embeds every existing document
401
+ // through the (often paid) embedding API whenever an index's embedder config
402
+ // actually changes, so this is intentionally its own call rather than something
403
+ // updateIndexSettings does automatically - it must only run when semantic settings
404
+ // were deliberately changed (see lib/settings.js), or as part of an explicit
405
+ // reindex/connect flow, never as a side effect of unrelated settings (ranking rules,
406
+ // stop words, ...) being saved. Skips the network call entirely when the resolved
407
+ // config is identical to what was last pushed, as a second line of defense against
408
+ // needless re-embedding regardless of caller. Older/self-hosted Meilisearch instances
409
+ // without embedder support will reject the call; that's caught here so it never
410
+ // blocks the rest of the settings-save flow.
411
+ plugin.updateEmbedders = async (rawData) => {
412
+ // Pre-flight validation: fail fast with an admin-facing alert instead of letting
413
+ // Meilisearch reject the task server-side (which can take up to 60s on slow/unreachable
414
+ // embedding endpoints, leaving the ACP Save button stuck). Client-side validation in
415
+ // static/lib/admin.js is the primary UX path; this is defense-in-depth for reindex,
416
+ // prepareSearch-on-host-change, and any programmatic caller.
417
+ const validationError = await validateEmbedderConfig(plugin, rawData);
418
+ if (validationError) {
419
+ winston.warn(`[plugin/meilisearch] Embedder config validation failed: ${validationError}`);
420
+ plugin.notifyAdmins('embedder:validation', {
421
+ type: 'danger',
422
+ titleKey: '[[meilisearch:admin.semanticConfigInvalid]]',
423
+ message: validationError,
424
+ });
425
+ return; // no fingerprint persisted, no task enqueued, save proceeds immediately
426
+ }
427
+ const indexNames = ['post', 'topic', 'chat_message'];
428
+ // Resolve configs + diff-against-applied first (read-only), THEN push and persist -
429
+ // keeps the three indexes from racing on the same shared "applied" bookkeeping object
430
+ // (parallel read-modify-write on one settings key would let one index's write silently
431
+ // clobber another's).
432
+ const plans = await Promise.all(indexNames.map(async (indexName) => {
433
+ const config = await buildEmbedderConfig(plugin, rawData, indexName);
434
+ const applied = await getAppliedConfig(plugin, indexName);
435
+ return { indexName, config, needsPush: applied === undefined || !deepEqual(config, applied) };
436
+ }));
437
+ const results = await Promise.all(plans.map(async (plan) => {
438
+ if (!plan.needsPush) return null;
439
+ try {
440
+ // updateEmbedders() only enqueues a Meilisearch task - a bad URL/model/response
441
+ // shape for "rest"/"openAi"-with-custom-url configs only surfaces once Meilisearch
442
+ // actually runs the task (e.g. its probe call to auto-detect dimensions), so this
443
+ // must wait for and check the task's real outcome, not just the enqueue response.
444
+ const enqueued = await plugin.client.index(plan.indexName).updateEmbedders({ [EMBEDDER_NAME]: plan.config });
445
+ try {
446
+ await waitForSucceededTask(plugin, enqueued.taskUid, {
447
+ timeout: EMBEDDER_TASK_TIMEOUT_MS,
448
+ interval: 200,
449
+ });
450
+ } catch (err) {
451
+ // Distinguish "task is still running" (poll timeout) from "task definitively failed"
452
+ // (Meili rejected the call, 4xx, malformed config, network error). Meilisearch SDK
453
+ // throws MeilisearchTaskTimeOutError (name property, see node_modules/meilisearch/dist/index.js:152)
454
+ // on poll timeout; this means the task is still running server-side and may yet succeed.
455
+ if (err.name === 'MeilisearchTaskTimeOutError') {
456
+ // Re-embedding every document legitimately exceeds 60s on a forum of any size.
457
+ // Optimistically persist the fingerprint so the next ACP save doesn't re-push
458
+ // identical config (which would double paid-API cost — the exact thing we're
459
+ // preventing). startEmbedderPoller verifies the real outcome.
460
+ winston.warn(`[plugin/meilisearch] Embedder task for "${plan.indexName}" still running after ${EMBEDDER_TASK_TIMEOUT_MS}ms; background-polling.`);
461
+ startEmbedderPoller(plugin, plan, enqueued.taskUid);
462
+ return { ...plan, taskUid: enqueued.taskUid }; // optimistic — taskUid for later reconciliation
463
+ }
464
+ throw err; // real failure — outer catch handles alerting + return null
465
+ }
466
+ return { ...plan, taskUid: null }; // sync-verified success — no reconciliation needed
467
+ } catch (err) {
468
+ // C1a fix: redact any apiKey from err.message before logging + alerting.
469
+ // Meilisearch may echo the configured apiKey in 401/403 error responses.
470
+ // A3-style inner try/catch: fall back to raw err.message if redaction itself fails,
471
+ // preserving the catch block's responsibility (alert + return null).
472
+ const rawReason = err.message;
473
+ let reason;
474
+ try {
475
+ reason = await redactSecrets(rawReason, plugin);
476
+ } catch {
477
+ reason = rawReason;
478
+ }
479
+ winston.error(`[plugin/meilisearch] Failed to apply "${plan.indexName}" embedder (semantic search): ${reason}`);
480
+ plugin.notifyAdmins(`embedder:${plan.indexName}`, {
481
+ type: 'danger',
482
+ titleKey: '[[meilisearch:admin.semanticEmbedderFailed]]',
483
+ message: `${plan.indexName}: ${reason}`,
484
+ });
485
+ return null;
486
+ }
487
+ }));
488
+ const succeeded = results.filter(Boolean);
489
+ if (succeeded.length) {
490
+ await setAppliedConfigs(plugin, succeeded);
491
+ }
492
+ };
493
+ };
494
+
495
+ // Exposed so other call sites that enqueue Meilisearch tasks (lib/reindex.js) get the same
496
+ // "actually check the task succeeded" behavior instead of trusting waitForTask() to throw.
497
+ module.exports.waitForSucceededTask = waitForSucceededTask;
@@ -0,0 +1,42 @@
1
+ 'use strict';
2
+
3
+ module.exports = {
4
+ REINDEX_BATCH_SIZE: 500,
5
+ PENDING_KEY: 'plugin:meilisearch:pending',
6
+ PENDING_MAX: 100000,
7
+ DRAIN_MAX: 500,
8
+ REINDEX_LOCK_KEY: 'plugin:meilisearch:reindex:lock',
9
+ REINDEX_LOCK_TTL: 3600,
10
+ LOCK_REFRESH_INTERVAL: 5 * 60 * 1000,
11
+ REPLAY_OPS: [
12
+ 'indexPost', 'deindexPost', 'indexTopic', 'deindexTopic', 'deindexPostsPurge',
13
+ 'deindexTopicsPurge', 'reindexTopicPosts', 'deindexTopicPosts', 'restoreTopic',
14
+ 'changePostOwner', 'changeTopicOwner', 'onTopicMerge', 'onScheduledPublish',
15
+ 'indexMessage', 'deindexMessage',
16
+ ],
17
+ GLOBAL_CHAT_SEARCH_LIMIT: 200,
18
+ GLOBAL_CHAT_SEARCH_FETCH: 300,
19
+ MEMBERSHIP_CHUNK_SIZE: 100,
20
+ DECORATION_BATCH_SIZE: 10,
21
+ // Sync wait window for plugin.updateEmbedders' Meilisearch task before falling back to
22
+ // background poller. 60s keeps ACP save responsive; if the embedder task is still
23
+ // running server-side (re-embedding every doc legitimately exceeds 60s on a forum of
24
+ // any size), updateEmbedders persists the fingerprint optimistically and startEmbedderPoller
25
+ // verifies the eventual outcome.
26
+ EMBEDDER_TASK_TIMEOUT_MS: 60000,
27
+ // Background poller ceiling (60 min) and interval (5 s) for verifying embedder tasks
28
+ // that exceeded the sync wait window. 60 min covers re-embedding up to ~1M posts at
29
+ // OpenAI Tier 1 rate limits (1M TPM). For forums larger than that, the poller gives up
30
+ // and relies on reattachPollers to reconcile the task outcome on the next restart
31
+ // (within Meilisearch's ~24h task retention window). No double-paid-API-cost risk —
32
+ // fingerprint is preserved, not cleared, on poller timeout (see startEmbedderPoller
33
+ // MeilisearchTaskTimeOutError handling in lib/client.js).
34
+ EMBEDDER_TASK_BG_TIMEOUT_MS: 3600000,
35
+ EMBEDDER_TASK_BG_INTERVAL_MS: 5000,
36
+ // Meilisearch SDK client-side HTTP timeout. Caps every fetch() the SDK makes (including
37
+ // the initial enqueue POST for embedder tasks) at 30s — without this, the SDK uses no
38
+ // AbortSignal (node_modules/meilisearch/dist/index.js:340), so an unreachable Meilisearch
39
+ // host can leave plugin.updateEmbedders (and therefore the ACP Save button) stuck
40
+ // indefinitely.
41
+ MEILI_HTTP_TIMEOUT_MS: 30000,
42
+ };
@@ -0,0 +1,73 @@
1
+ 'use strict';
2
+
3
+ const defaults = {
4
+ host: 'http://localhost:7700',
5
+ apiKey: undefined,
6
+ maxDocuments: undefined,
7
+ indexed: false,
8
+ rankingRules: [
9
+ { rule: 'words' },
10
+ { rule: 'typo' },
11
+ { rule: 'proximity' },
12
+ { rule: 'attribute' },
13
+ { rule: 'sort' },
14
+ { rule: 'exactness' },
15
+ ],
16
+ stopWords: [],
17
+ typoTolerance: 'on',
18
+ typoToleranceMinWordSizeOneTypo: 5,
19
+ typoToleranceMinWordSizeTwoTypos: 9,
20
+ typoToleranceDisableOnWords: [],
21
+ synonyms: [],
22
+ healthCheckInterval: 60,
23
+ searchMinTermLength: 2,
24
+ globalChatSearchEnabled: 'on',
25
+ semanticSearchEnabled: 'off',
26
+ semanticSearchProvider: 'openAi',
27
+ semanticSearchApiKey: undefined,
28
+ semanticSearchModel: undefined,
29
+ semanticSearchUrl: undefined,
30
+ semanticSearchDimensions: undefined,
31
+ semanticSearchRatio: 0.5,
32
+ semanticSearchScoreThreshold: 0.2,
33
+ semanticSearchRestRequest: undefined,
34
+ semanticSearchRestResponse: undefined,
35
+ // Internal bookkeeping (not an ACP field): last embedder config actually pushed to
36
+ // Meilisearch per index, so unrelated settings saves can't re-trigger paid re-embedding.
37
+ appliedEmbedders: {},
38
+ lastReindexResult: {
39
+ success: false,
40
+ finishedAt: null,
41
+ topic_progress: { current: null, total: null },
42
+ post_progress: { current: null, total: null },
43
+ message_progress: { current: null, total: null },
44
+ skippedDeletedTopics: 0,
45
+ skippedDeletedPosts: 0,
46
+ skippedDeletedMessages: 0,
47
+ skippedSystemMessages: 0,
48
+ skippedOrphanMessages: 0,
49
+ error: null,
50
+ },
51
+ };
52
+
53
+ const breakingSettings = [
54
+ 'maxDocuments',
55
+ 'rankingRules',
56
+ 'stopWords',
57
+ 'typoTolerance',
58
+ 'typoToleranceMinWordSizeOneTypo',
59
+ 'typoToleranceMinWordSizeTwoTypos',
60
+ 'typoToleranceDisableOnWords',
61
+ 'typoToleranceDisableOnAttributes',
62
+ 'synonyms',
63
+ 'semanticSearchEnabled',
64
+ 'semanticSearchProvider',
65
+ 'semanticSearchApiKey',
66
+ 'semanticSearchModel',
67
+ 'semanticSearchUrl',
68
+ 'semanticSearchDimensions',
69
+ 'semanticSearchRestRequest',
70
+ 'semanticSearchRestResponse',
71
+ ];
72
+
73
+ module.exports = { defaults, breakingSettings };