@memberjunction/ai-vector-dupe 6.2.0-edge.0 → 6.2.0-edge.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -5
- package/dist/duplicateEntryCheckTypes.d.ts +80 -0
- package/dist/duplicateEntryCheckTypes.d.ts.map +1 -0
- package/dist/duplicateEntryCheckTypes.js +29 -0
- package/dist/duplicateEntryCheckTypes.js.map +1 -0
- package/dist/duplicateRecordDetector.d.ts +153 -13
- package/dist/duplicateRecordDetector.d.ts.map +1 -1
- package/dist/duplicateRecordDetector.js +436 -50
- package/dist/duplicateRecordDetector.js.map +1 -1
- package/dist/entryCheckDeadline.d.ts +73 -0
- package/dist/entryCheckDeadline.d.ts.map +1 -0
- package/dist/entryCheckDeadline.js +118 -0
- package/dist/entryCheckDeadline.js.map +1 -0
- package/dist/index.d.ts +6 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +4 -1
- package/dist/index.js.map +1 -1
- package/dist/reasoning/DecisionReasoningProvider.d.ts +177 -0
- package/dist/reasoning/DecisionReasoningProvider.d.ts.map +1 -0
- package/dist/reasoning/DecisionReasoningProvider.js +306 -0
- package/dist/reasoning/DecisionReasoningProvider.js.map +1 -0
- package/dist/reasoning/DecisionThenPromptReasoningProvider.d.ts +88 -0
- package/dist/reasoning/DecisionThenPromptReasoningProvider.d.ts.map +1 -0
- package/dist/reasoning/DecisionThenPromptReasoningProvider.js +159 -0
- package/dist/reasoning/DecisionThenPromptReasoningProvider.js.map +1 -0
- package/dist/reasoning/DuplicateReasoningProvider.d.ts +12 -6
- package/dist/reasoning/DuplicateReasoningProvider.d.ts.map +1 -1
- package/dist/reasoning/DuplicateReasoningProvider.js +12 -6
- package/dist/reasoning/DuplicateReasoningProvider.js.map +1 -1
- package/dist/reasoning/DuplicateReasoningTypes.d.ts +21 -2
- package/dist/reasoning/DuplicateReasoningTypes.d.ts.map +1 -1
- package/dist/reasoning/MatchedSetDeltaBuilder.d.ts +17 -3
- package/dist/reasoning/MatchedSetDeltaBuilder.d.ts.map +1 -1
- package/dist/reasoning/MatchedSetDeltaBuilder.js +49 -2
- package/dist/reasoning/MatchedSetDeltaBuilder.js.map +1 -1
- package/package.json +14 -14
|
@@ -14,17 +14,21 @@
|
|
|
14
14
|
*
|
|
15
15
|
* @module @memberjunction/ai-vector-dupe
|
|
16
16
|
*/
|
|
17
|
-
import {
|
|
17
|
+
import { GetAIAPIKey } from "@memberjunction/ai";
|
|
18
|
+
import { AIEmbeddingRunner } from "@memberjunction/ai-prompts";
|
|
18
19
|
import { PotentialDuplicateResponse, CompositeKey, PotentialDuplicateResult, LogError, LogStatus, RecordMergeRequest, PotentialDuplicate, } from "@memberjunction/core";
|
|
19
20
|
import { VectorDBBase } from "@memberjunction/ai-vectordb";
|
|
20
21
|
import { MJGlobal, UUIDsEqual, NormalizeUUID } from "@memberjunction/global";
|
|
21
22
|
import { KnowledgeHubMetadataEngine, } from "@memberjunction/core-entities";
|
|
22
23
|
import { VectorBase } from "@memberjunction/ai-vectors";
|
|
23
24
|
import { EntityVectorSyncer } from "@memberjunction/ai-vector-sync";
|
|
25
|
+
import { AIEngine } from "@memberjunction/aiengine";
|
|
24
26
|
import { EntityDocumentTemplateParser } from "@memberjunction/entity-documents";
|
|
25
27
|
import { TemplateEngineServer } from "@memberjunction/templates";
|
|
26
|
-
import { DuplicateReasoningProvider } from "./reasoning/DuplicateReasoningProvider.js";
|
|
28
|
+
import { DuplicateReasoningProvider, DECISION_REASONING_PROVIDER_KEY, DECISION_THEN_PROMPT_REASONING_PROVIDER_KEY, } from "./reasoning/DuplicateReasoningProvider.js";
|
|
27
29
|
import { MatchedSetDeltaBuilder } from "./reasoning/MatchedSetDeltaBuilder.js";
|
|
30
|
+
import { DUPLICATE_ENTRY_CHECK_MAX_DECISION_FIELDS, DUPLICATE_ENTRY_CHECK_MAX_FIELD_TEXT_LENGTH, DUPLICATE_ENTRY_CHECK_SERVER_BUDGET_MS, } from "./duplicateEntryCheckTypes.js";
|
|
31
|
+
import { EntryCheckDeadline, EntryCheckStoppedError } from "./entryCheckDeadline.js";
|
|
28
32
|
/** Default number of nearest neighbors to retrieve per record */
|
|
29
33
|
const DEFAULT_TOP_K = 5;
|
|
30
34
|
/** Default concurrency limit for parallel vector queries */
|
|
@@ -38,6 +42,14 @@ const DEFAULT_BATCH_SIZE = 500;
|
|
|
38
42
|
const VECTOR_QUERY_BATCH_SIZE = 100;
|
|
39
43
|
/** Default batch size for parallel database saves */
|
|
40
44
|
const SAVE_BATCH_SIZE = 20;
|
|
45
|
+
/**
|
|
46
|
+
* The entity-document ReasoningModes that turn on the entry-time check. Both carry a typed
|
|
47
|
+
* decision, which is fast enough to answer while a person types.
|
|
48
|
+
*/
|
|
49
|
+
const ENTRY_CHECK_REASONING_MODES = new Set([
|
|
50
|
+
DECISION_REASONING_PROVIDER_KEY,
|
|
51
|
+
DECISION_THEN_PROMPT_REASONING_PROVIDER_KEY,
|
|
52
|
+
]);
|
|
41
53
|
/**
|
|
42
54
|
* Modernized duplicate record detection engine.
|
|
43
55
|
*
|
|
@@ -45,6 +57,7 @@ const SAVE_BATCH_SIZE = 20;
|
|
|
45
57
|
* - List-based batch detection (getDuplicateRecords)
|
|
46
58
|
* - View/filter/full-entity batch detection (vector-first approach)
|
|
47
59
|
* - Single-record duplicate check (CheckSingleRecord)
|
|
60
|
+
* - Entry-time check of a new record's unsaved values, flag only (CheckRecordValues)
|
|
48
61
|
* - Hybrid search via RRF when vector DB supports it
|
|
49
62
|
* - Optional post-retrieval reranking via MJ's BaseReranker
|
|
50
63
|
* - Configurable topK, thresholds, and progress reporting
|
|
@@ -52,10 +65,11 @@ const SAVE_BATCH_SIZE = 20;
|
|
|
52
65
|
export class DuplicateRecordDetector extends VectorBase {
|
|
53
66
|
constructor() {
|
|
54
67
|
super(...arguments);
|
|
68
|
+
this._embeddingRunner = null;
|
|
69
|
+
this.embeddingModelID = null;
|
|
55
70
|
/**
|
|
56
71
|
* The embedding model's API identifier (e.g. 'Xenova/gte-small', 'text-embedding-3-small'),
|
|
57
|
-
* resolved from the entity document's AIModel.
|
|
58
|
-
* providers (which load the named ONNX pipeline) and honored by cloud providers.
|
|
72
|
+
* resolved from the entity document's AIModel.
|
|
59
73
|
*/
|
|
60
74
|
this.embeddingModelAPIName = null;
|
|
61
75
|
/**
|
|
@@ -64,6 +78,18 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
64
78
|
*/
|
|
65
79
|
this._seenPairs = new Set();
|
|
66
80
|
}
|
|
81
|
+
/**
|
|
82
|
+
* Protected getter/setter for the AIEmbeddingRunner instance used by duplicate detection.
|
|
83
|
+
*/
|
|
84
|
+
get EmbeddingRunner() {
|
|
85
|
+
if (!this._embeddingRunner) {
|
|
86
|
+
this._embeddingRunner = new AIEmbeddingRunner();
|
|
87
|
+
}
|
|
88
|
+
return this._embeddingRunner;
|
|
89
|
+
}
|
|
90
|
+
set EmbeddingRunner(value) {
|
|
91
|
+
this._embeddingRunner = value;
|
|
92
|
+
}
|
|
67
93
|
/**
|
|
68
94
|
* Run duplicate detection for records identified by ListID, ViewID, ExtraFilter,
|
|
69
95
|
* or all records in the entity (vector-first approach).
|
|
@@ -154,7 +180,7 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
154
180
|
return response;
|
|
155
181
|
}
|
|
156
182
|
const batchIDs = recordIDs.slice(offset, offset + batchSize);
|
|
157
|
-
const batchResults = await this.
|
|
183
|
+
const batchResults = await this.processBatch(batchIDs, entityInfo, entityDocument, templateParser, duplicateRun.ID, topK, concurrency, options, startTime, recordIDs.length, offset, totalMatchesFound, contextUser);
|
|
158
184
|
response.PotentialDuplicateResult.push(...batchResults.Results);
|
|
159
185
|
totalMatchesFound += batchResults.MatchesFound;
|
|
160
186
|
// Update cursor for resume support
|
|
@@ -202,25 +228,317 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
202
228
|
throw new Error(`Record not found: ${RecordID.ToString()}`);
|
|
203
229
|
}
|
|
204
230
|
const record = records.Results[0];
|
|
205
|
-
const
|
|
206
|
-
|
|
207
|
-
const embedResult = await this.embedding.EmbedTexts({ texts: templateTexts, model: this.embeddingModelAPIName });
|
|
208
|
-
const topK = options.TopK ?? DEFAULT_TOP_K;
|
|
209
|
-
const queryResults = await this.QueryDuplicatesForRecords([record], embedResult.vectors, templateTexts, entityDocument, topK, options, this.GetQueryConcurrency(entityDocument));
|
|
210
|
-
if (queryResults.length === 0) {
|
|
231
|
+
const single = await this.QueryCandidatesForRecord(record, entityDocument, options, ContextUser);
|
|
232
|
+
if (!single) {
|
|
211
233
|
return new PotentialDuplicateResult();
|
|
212
234
|
}
|
|
213
235
|
// Annotate the (non-persisted) single-record result with LLM reasoning when the entity
|
|
214
236
|
// has it enabled and the set clears the gate — same gate as the batch path. There are no
|
|
215
237
|
// match rows to stamp run ids onto here; the verdict rides on the returned result so a
|
|
216
238
|
// real-time "is this a duplicate?" caller gets recommendation + resolved field map too.
|
|
217
|
-
const single = queryResults[0];
|
|
218
239
|
const reasoning = await this.RunReasoningForSet(single, entityInfo, entityDocument, ContextUser);
|
|
219
240
|
if (reasoning) {
|
|
220
241
|
this.applyReasoningToResult(single.Duplicates, reasoning.Output, reasoning.FieldMap);
|
|
221
242
|
}
|
|
222
243
|
return single.Duplicates;
|
|
223
244
|
}
|
|
245
|
+
/**
|
|
246
|
+
* Check values a person is entering for a new record against the entity's existing records, and
|
|
247
|
+
* flag the plausible duplicates. It only flags: nothing is saved, merged or blocked.
|
|
248
|
+
*
|
|
249
|
+
* 1. **The switch.** The check runs only for an entity with an Active entity document that has
|
|
250
|
+
* `EnableLLMReasoning` on and whose `ReasoningMode` is `'Decision'` or `'DecisionThenPrompt'`
|
|
251
|
+
* (see {@link FindEntryCheckDocument}). Without one the result is `NotConfigured` and nothing
|
|
252
|
+
* else runs.
|
|
253
|
+
* 2. **Candidates.** The values, each text cut to {@link DUPLICATE_ENTRY_CHECK_MAX_FIELD_TEXT_LENGTH},
|
|
254
|
+
* become an unsaved entity object, rendered by the document's template, embedded, and queried
|
|
255
|
+
* with the same TopK and threshold as {@link CheckSingleRecord}.
|
|
256
|
+
* 3. **Permission.** The vector index does not know who is asking, so the candidates are narrowed
|
|
257
|
+
* to those the context user can read, with one RunView as that user.
|
|
258
|
+
* 4. **Decision.** One {@link DecisionReasoningProvider.DecideCandidates} call over the readable
|
|
259
|
+
* candidates, in either mode: the prompt half of `'DecisionThenPrompt'` is too slow for entry,
|
|
260
|
+
* and it can recommend a merge. Its state is bounded by
|
|
261
|
+
* {@link DUPLICATE_ENTRY_CHECK_MAX_DECISION_FIELDS} and the field-text limit. A candidate is
|
|
262
|
+
* flagged when the provider bands it `Uncertain`, by its own threshold, on the probability
|
|
263
|
+
* calibrated for the model that answered. A candidate the decision gave no answer for is
|
|
264
|
+
* flagged. A failed decision flags nothing, and neither does a decision from a model with no
|
|
265
|
+
* calibration: its probabilities can't be banded, and flagging every candidate would be the
|
|
266
|
+
* vector threshold alone. Both give a `Failed` result with the reason.
|
|
267
|
+
*
|
|
268
|
+
* **Budget.** The whole check is bounded by `options.TimeoutMS`
|
|
269
|
+
* ({@link DUPLICATE_ENTRY_CHECK_SERVER_BUDGET_MS} by default) and by `options.CancellationToken`.
|
|
270
|
+
* No step starts once either has fired, the check stops waiting for a running step, and the
|
|
271
|
+
* decision call receives both, so the model call is aborted. A stopped check is `Failed`.
|
|
272
|
+
*
|
|
273
|
+
* Never throws: any failure is a `Failed` result with the reason.
|
|
274
|
+
*
|
|
275
|
+
* @param entityName the entity the new record belongs to
|
|
276
|
+
* @param values the entered values by field name; fields the entity does not have are ignored
|
|
277
|
+
* @param contextUser the person entering the record. Required: the candidates are narrowed to the
|
|
278
|
+
* records this user can read, and without a user the check fails rather than skip that step.
|
|
279
|
+
* @param options overrides for TopK and the potential-match threshold, the budget and cancellation
|
|
280
|
+
* @returns the flagged candidates, most probable first, or why there are none
|
|
281
|
+
*/
|
|
282
|
+
async CheckRecordValues(entityName, values, contextUser, options) {
|
|
283
|
+
const startTime = Date.now();
|
|
284
|
+
if (!contextUser) {
|
|
285
|
+
return this.entryCheckResult('Failed', startTime, [], 'A context user is required: the check shows only the records that user can read');
|
|
286
|
+
}
|
|
287
|
+
this.CurrentUser = contextUser;
|
|
288
|
+
const deadline = new EntryCheckDeadline(options?.TimeoutMS ?? DUPLICATE_ENTRY_CHECK_SERVER_BUDGET_MS, options?.CancellationToken);
|
|
289
|
+
try {
|
|
290
|
+
return await this.runEntryCheck(entityName, values, options ?? {}, contextUser, deadline, startTime);
|
|
291
|
+
}
|
|
292
|
+
catch (e) {
|
|
293
|
+
if (e instanceof EntryCheckStoppedError) {
|
|
294
|
+
// Running out of the budget, or being cancelled, is how an abandoned check ends: not a fault.
|
|
295
|
+
return this.entryCheckResult('Failed', startTime, [], e.message);
|
|
296
|
+
}
|
|
297
|
+
const message = e instanceof Error ? e.message : String(e);
|
|
298
|
+
LogError(`Duplicate entry check for ${entityName} failed: ${message}`);
|
|
299
|
+
return this.entryCheckResult('Failed', startTime, [], message);
|
|
300
|
+
}
|
|
301
|
+
finally {
|
|
302
|
+
deadline.Dispose();
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
/**
|
|
306
|
+
* Render one record with the entity document's template, embed it, and query the vector index
|
|
307
|
+
* for its candidates. Shared by {@link CheckSingleRecord} (a saved record) and
|
|
308
|
+
* {@link CheckRecordValues} (an unsaved one).
|
|
309
|
+
*
|
|
310
|
+
* @param deadline bounds each step, for the entry-time check; without one the steps are unbounded
|
|
311
|
+
* @returns the record's query result, or null when the query produced none
|
|
312
|
+
*/
|
|
313
|
+
async QueryCandidatesForRecord(record, entityDocument, options, contextUser, deadline) {
|
|
314
|
+
const step = (name, work) => deadline ? deadline.Run(name, work) : work();
|
|
315
|
+
const templateParser = EntityDocumentTemplateParser.CreateInstance();
|
|
316
|
+
const templateTexts = await step('rendering the template', () => this.GenerateTemplateTexts(templateParser, entityDocument, [record], contextUser));
|
|
317
|
+
const embedResult = await step('embedding the record', () => this.EmbeddingRunner.RunEmbedding({
|
|
318
|
+
Texts: templateTexts,
|
|
319
|
+
ModelID: this.embeddingModelID ?? undefined,
|
|
320
|
+
ContextUser: contextUser,
|
|
321
|
+
Description: `Duplicate detection single record (${entityDocument.Name})`
|
|
322
|
+
}));
|
|
323
|
+
if (!embedResult.Success || !embedResult.Vectors || embedResult.Vectors.length === 0) {
|
|
324
|
+
throw new Error(`Embedding failed for duplicate detection: ${embedResult.ErrorMessage ?? 'Unknown error'}`);
|
|
325
|
+
}
|
|
326
|
+
const vectors = embedResult.Vectors;
|
|
327
|
+
const topK = options.TopK ?? DEFAULT_TOP_K;
|
|
328
|
+
const queryResults = await step('querying the vector index', () => this.QueryDuplicatesForRecords([record], vectors, templateTexts, entityDocument, topK, options, this.GetQueryConcurrency(entityDocument)));
|
|
329
|
+
return queryResults[0] ?? null;
|
|
330
|
+
}
|
|
331
|
+
// ─────────────────────────────────────────────
|
|
332
|
+
// Entry-Time Check
|
|
333
|
+
// ─────────────────────────────────────────────
|
|
334
|
+
/**
|
|
335
|
+
* The entity document that turns the entry-time check on for an entity: an Active document with
|
|
336
|
+
* `EnableLLMReasoning` on whose `ReasoningMode` is `'Decision'` or `'DecisionThenPrompt'`. The
|
|
337
|
+
* check makes a decision-model call on every pause in typing, so switching reasoning off on the
|
|
338
|
+
* document turns the check off too, as it turns off reasoning in batch runs
|
|
339
|
+
* ({@link IsReasoningGateOpen}). When several qualify, the oldest wins (earliest
|
|
340
|
+
* `__mj_CreatedAt`, then lowest ID), so adding a document never silently changes which one an
|
|
341
|
+
* entity uses.
|
|
342
|
+
*
|
|
343
|
+
* @returns the document, or null when the check is not configured for the entity
|
|
344
|
+
*/
|
|
345
|
+
async FindEntryCheckDocument(entityName) {
|
|
346
|
+
await KnowledgeHubMetadataEngine.Instance.Config(false, this.CurrentUser);
|
|
347
|
+
const documents = KnowledgeHubMetadataEngine.Instance.GetEntityDocumentsForEntity(entityName)
|
|
348
|
+
.filter(d => d.Status === 'Active' && d.EnableLLMReasoning && ENTRY_CHECK_REASONING_MODES.has(d.ReasoningMode));
|
|
349
|
+
return documents.sort(compareOldestFirst)[0] ?? null;
|
|
350
|
+
}
|
|
351
|
+
/**
|
|
352
|
+
* Build an unsaved entity object holding the entered values, so the entity document's template
|
|
353
|
+
* renders it as it renders a saved record. Values for fields the entity does not have are ignored,
|
|
354
|
+
* and text longer than {@link DUPLICATE_ENTRY_CHECK_MAX_FIELD_TEXT_LENGTH} is cut, so the
|
|
355
|
+
* template, the embedding and the decision state never see more of a field than that.
|
|
356
|
+
*/
|
|
357
|
+
async BuildUnsavedRecord(entityInfo, values, contextUser) {
|
|
358
|
+
const record = await this.Metadata.GetEntityObject(entityInfo.Name, contextUser);
|
|
359
|
+
record.NewRecord();
|
|
360
|
+
record.SetMany(boundEnteredValues(values), true);
|
|
361
|
+
return record;
|
|
362
|
+
}
|
|
363
|
+
/**
|
|
364
|
+
* The display names of the candidates the context user can read, from one RunView over the
|
|
365
|
+
* candidate keys run as that user, so entity permissions and row-level security apply. A
|
|
366
|
+
* candidate missing from the map must not be shown. When the RunView fails the map is empty:
|
|
367
|
+
* a candidate is shown only when the user is known to be able to read it.
|
|
368
|
+
*
|
|
369
|
+
* @param contextUser the person the candidates are narrowed for; the RunView runs as this user
|
|
370
|
+
* @returns display name by normalized compact record id ({@link NormalizeUUID})
|
|
371
|
+
*/
|
|
372
|
+
async LoadReadableCandidateNames(candidates, entityInfo, contextUser) {
|
|
373
|
+
const names = new Map();
|
|
374
|
+
if (candidates.length === 0) {
|
|
375
|
+
return names;
|
|
376
|
+
}
|
|
377
|
+
const nameFields = this.DisplayNameFields(entityInfo);
|
|
378
|
+
const result = await this.RunView.RunView({
|
|
379
|
+
EntityName: entityInfo.Name,
|
|
380
|
+
ExtraFilter: candidates.map(c => `(${c.ToWhereClause()})`).join(' OR '),
|
|
381
|
+
Fields: [...new Set([...entityInfo.PrimaryKeys.map(pk => pk.Name), ...nameFields.map(f => f.Name)])],
|
|
382
|
+
ResultType: 'simple',
|
|
383
|
+
MaxRows: candidates.length,
|
|
384
|
+
}, contextUser);
|
|
385
|
+
if (!result.Success) {
|
|
386
|
+
LogError(`Duplicate entry check: could not confirm which ${entityInfo.Name} candidates are readable: ${result.ErrorMessage}`);
|
|
387
|
+
return names;
|
|
388
|
+
}
|
|
389
|
+
for (const row of result.Results) {
|
|
390
|
+
const recordID = CompositeKey.FromEntityRecord(entityInfo, row).ToCompactURLSegment();
|
|
391
|
+
names.set(NormalizeUUID(recordID), this.rowDisplayName(row, nameFields) || recordID);
|
|
392
|
+
}
|
|
393
|
+
return names;
|
|
394
|
+
}
|
|
395
|
+
/**
|
|
396
|
+
* The decision provider the entry-time check asks: whatever the class factory holds for the
|
|
397
|
+
* `'Decision'` mode, which is {@link DecisionReasoningProvider} unless an app registers a
|
|
398
|
+
* subclass. Null when that registration is not a decision provider.
|
|
399
|
+
*/
|
|
400
|
+
ResolveEntryDecisionProvider() {
|
|
401
|
+
const provider = MJGlobal.Instance.ClassFactory.CreateInstance(DuplicateReasoningProvider, DECISION_REASONING_PROVIDER_KEY);
|
|
402
|
+
return isDecisionReasoningProvider(provider) ? provider : null;
|
|
403
|
+
}
|
|
404
|
+
/**
|
|
405
|
+
* The fields whose values make up a record's display name: every `IsNameField` field in
|
|
406
|
+
* Sequence order, or the entity's single `NameField` when none is flagged.
|
|
407
|
+
*/
|
|
408
|
+
DisplayNameFields(entityInfo) {
|
|
409
|
+
const nameFields = entityInfo.Fields
|
|
410
|
+
.filter(f => f.IsNameField)
|
|
411
|
+
.sort((a, b) => (a.Sequence ?? 9999) - (b.Sequence ?? 9999));
|
|
412
|
+
// Fall back to singular NameField if no IsNameField flags
|
|
413
|
+
if (nameFields.length === 0 && entityInfo.NameField) {
|
|
414
|
+
nameFields.push(entityInfo.NameField);
|
|
415
|
+
}
|
|
416
|
+
return nameFields;
|
|
417
|
+
}
|
|
418
|
+
/**
|
|
419
|
+
* The steps of {@link CheckRecordValues} after it has recorded the start time, each bounded by
|
|
420
|
+
* the deadline. May throw, {@link EntryCheckStoppedError} included.
|
|
421
|
+
*/
|
|
422
|
+
async runEntryCheck(entityName, values, options, contextUser, deadline, startTime) {
|
|
423
|
+
const entityDocument = await deadline.Run('finding the entity document', () => this.FindEntryCheckDocument(entityName));
|
|
424
|
+
if (!entityDocument) {
|
|
425
|
+
return this.entryCheckResult('NotConfigured', startTime);
|
|
426
|
+
}
|
|
427
|
+
const entityInfo = this.Metadata.EntityByID(entityDocument.EntityID);
|
|
428
|
+
if (!entityInfo) {
|
|
429
|
+
return this.entryCheckResult('Failed', startTime, [], `Entity not found for ID ${entityDocument.EntityID}`);
|
|
430
|
+
}
|
|
431
|
+
await deadline.Run('loading the embedding and vector providers', () => this.InitializeProviders(entityDocument));
|
|
432
|
+
const record = await deadline.Run('building the unsaved record', () => this.BuildUnsavedRecord(entityInfo, values, contextUser));
|
|
433
|
+
const query = await this.QueryCandidatesForRecord(record, entityDocument, options, contextUser, deadline);
|
|
434
|
+
if (!query) {
|
|
435
|
+
return this.entryCheckResult('Checked', startTime);
|
|
436
|
+
}
|
|
437
|
+
const displayNames = await this.keepReadableCandidates(query, entityInfo, contextUser, deadline);
|
|
438
|
+
if (query.Duplicates.Duplicates.length === 0) {
|
|
439
|
+
return this.entryCheckResult('Checked', startTime);
|
|
440
|
+
}
|
|
441
|
+
const decision = await this.decideEntryCandidates({
|
|
442
|
+
EntityInfo: entityInfo, EntityDocument: entityDocument, Record: record, Query: query,
|
|
443
|
+
DisplayNames: displayNames, ContextUser: contextUser, Deadline: deadline,
|
|
444
|
+
});
|
|
445
|
+
return 'Error' in decision
|
|
446
|
+
? this.entryCheckResult('Failed', startTime, [], decision.Error)
|
|
447
|
+
: this.entryCheckResult('Checked', startTime, decision.Candidates);
|
|
448
|
+
}
|
|
449
|
+
/** Narrow a query result's candidates, in place, to those the context user can read. */
|
|
450
|
+
async keepReadableCandidates(query, entityInfo, contextUser, deadline) {
|
|
451
|
+
const names = await deadline.Run('checking which candidates the user can read', () => this.LoadReadableCandidateNames(query.Duplicates.Duplicates, entityInfo, contextUser));
|
|
452
|
+
query.Duplicates.Duplicates = query.Duplicates.Duplicates.filter(d => names.has(NormalizeUUID(d.ToCompactURLSegment())));
|
|
453
|
+
return names;
|
|
454
|
+
}
|
|
455
|
+
/**
|
|
456
|
+
* Ask the decision about the readable candidates, with the unsaved record's own values as the
|
|
457
|
+
* source, and flag the ones the provider bands `Uncertain`. The decision call gets the deadline's
|
|
458
|
+
* signal and remaining time, so the model call is aborted when the check must stop.
|
|
459
|
+
*/
|
|
460
|
+
async decideEntryCandidates(state) {
|
|
461
|
+
const provider = this.ResolveEntryDecisionProvider();
|
|
462
|
+
if (!provider) {
|
|
463
|
+
return { Error: `No DecisionReasoningProvider is registered for the '${DECISION_REASONING_PROVIDER_KEY}' reasoning mode` };
|
|
464
|
+
}
|
|
465
|
+
const input = await state.Deadline.Run('loading the candidates\' field values', () => this.BuildReasoningInput(state.Query, state.EntityInfo, state.EntityDocument, state.ContextUser, state.Record));
|
|
466
|
+
const decision = await state.Deadline.Run('asking the decision model', () => provider.DecideCandidates(this.BoundEntryDecisionInput(input), {
|
|
467
|
+
Provider: this._provider,
|
|
468
|
+
ContextUser: state.ContextUser,
|
|
469
|
+
CancellationToken: state.Deadline.Signal,
|
|
470
|
+
TimeoutMS: state.Deadline.RemainingMS,
|
|
471
|
+
}));
|
|
472
|
+
if (!decision.Success) {
|
|
473
|
+
const message = `Decision failed: ${decision.ErrorMessage ?? 'unknown error'}`;
|
|
474
|
+
LogError(`Duplicate entry check for ${state.EntityInfo.Name}: ${message}. Nothing is flagged.`);
|
|
475
|
+
return { Error: message };
|
|
476
|
+
}
|
|
477
|
+
if (decision.UncalibratedModel) {
|
|
478
|
+
// Not logged here: the provider logs a missing calibration once per model, not per check.
|
|
479
|
+
return { Error: `The decision model "${decision.UncalibratedModel}" has no calibration, so nothing is flagged` };
|
|
480
|
+
}
|
|
481
|
+
return { Candidates: this.flagEntryCandidates(provider, decision.Candidates, state) };
|
|
482
|
+
}
|
|
483
|
+
/**
|
|
484
|
+
* The entry-time decision's input, bounded: at most {@link DUPLICATE_ENTRY_CHECK_MAX_DECISION_FIELDS}
|
|
485
|
+
* differing fields, the ones the person entered first, and every value and label cut to
|
|
486
|
+
* {@link DUPLICATE_ENTRY_CHECK_MAX_FIELD_TEXT_LENGTH}. A candidate's stored values are cut too:
|
|
487
|
+
* a long Notes field on an existing record would otherwise reach the model in full.
|
|
488
|
+
*/
|
|
489
|
+
BoundEntryDecisionInput(input) {
|
|
490
|
+
const sourceID = input.SourceRecord.RecordID;
|
|
491
|
+
const entered = (delta) => delta.Values.some(v => this.recordIdMatches(v.RecordID, sourceID) && v.Value != null && v.Value.trim().length > 0);
|
|
492
|
+
const fields = [...input.FieldDeltas.filter(entered), ...input.FieldDeltas.filter(d => !entered(d))]
|
|
493
|
+
.slice(0, DUPLICATE_ENTRY_CHECK_MAX_DECISION_FIELDS)
|
|
494
|
+
.map(delta => ({
|
|
495
|
+
FieldName: delta.FieldName,
|
|
496
|
+
Values: delta.Values.map(v => ({ RecordID: v.RecordID, Value: v.Value == null ? null : truncateText(v.Value) })),
|
|
497
|
+
}));
|
|
498
|
+
return {
|
|
499
|
+
...input,
|
|
500
|
+
SourceRecord: { ...input.SourceRecord, Label: truncateText(input.SourceRecord.Label) },
|
|
501
|
+
Candidates: input.Candidates.map(c => ({ ...c, Label: truncateText(c.Label) })),
|
|
502
|
+
FieldDeltas: fields,
|
|
503
|
+
};
|
|
504
|
+
}
|
|
505
|
+
/**
|
|
506
|
+
* The candidates the provider bands `Uncertain` (they cannot be ruled out), most probable first.
|
|
507
|
+
* The decision answers in input order, so each answer lines up with the query's candidate at the
|
|
508
|
+
* same index.
|
|
509
|
+
*/
|
|
510
|
+
flagEntryCandidates(provider, answers, state) {
|
|
511
|
+
const flagged = [];
|
|
512
|
+
answers.forEach((answer, index) => {
|
|
513
|
+
const duplicate = state.Query.Duplicates.Duplicates[index];
|
|
514
|
+
if (!duplicate || provider.BandCandidate(answer).Recommendation !== 'Uncertain') {
|
|
515
|
+
return;
|
|
516
|
+
}
|
|
517
|
+
const recordID = duplicate.ToCompactURLSegment();
|
|
518
|
+
flagged.push({
|
|
519
|
+
RecordID: recordID,
|
|
520
|
+
DisplayName: state.DisplayNames.get(NormalizeUUID(recordID)) ?? recordID,
|
|
521
|
+
VectorScore: duplicate.ProbabilityScore,
|
|
522
|
+
Probability: answer.Probability,
|
|
523
|
+
});
|
|
524
|
+
});
|
|
525
|
+
return flagged.sort(compareMostProbableFirst);
|
|
526
|
+
}
|
|
527
|
+
/** A row's name-field values joined with spaces; empty when it has none the user may read. */
|
|
528
|
+
rowDisplayName(row, nameFields) {
|
|
529
|
+
return nameFields
|
|
530
|
+
.map(f => row[f.Name])
|
|
531
|
+
.filter(v => v != null && String(v).trim() !== '')
|
|
532
|
+
.map(v => String(v))
|
|
533
|
+
.join(' ');
|
|
534
|
+
}
|
|
535
|
+
entryCheckResult(status, startTime, candidates = [], errorMessage) {
|
|
536
|
+
const result = { Status: status, Candidates: candidates, ElapsedMs: Date.now() - startTime };
|
|
537
|
+
if (errorMessage) {
|
|
538
|
+
result.ErrorMessage = errorMessage;
|
|
539
|
+
}
|
|
540
|
+
return result;
|
|
541
|
+
}
|
|
224
542
|
// ─────────────────────────────────────────────
|
|
225
543
|
// Batch Processing
|
|
226
544
|
// ─────────────────────────────────────────────
|
|
@@ -250,7 +568,7 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
250
568
|
* @param contextUser - The user context for entity operations
|
|
251
569
|
* @returns Combined results and match count for this batch
|
|
252
570
|
*/
|
|
253
|
-
async
|
|
571
|
+
async processBatch(batchIDs, entityInfo, entityDocument, templateParser, duplicateRunID, topK, concurrency, options, startTime, totalRecords, processedSoFar, matchesSoFar, contextUser) {
|
|
254
572
|
// 6a: Load full record data for this batch (needed for template rendering). Each id is a
|
|
255
573
|
// compact key segment (see LoadRecordIDsToCheck), so this rebuilds the entity's real
|
|
256
574
|
// key — a single column of any name, or a composite key — rather than assuming one column.
|
|
@@ -272,10 +590,18 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
272
590
|
// Embed this sub-batch
|
|
273
591
|
this.reportProgress(options, 'Embedding', totalRecords, processedSoFar, matchesSoFar, startTime);
|
|
274
592
|
const subTemplateTexts = await this.GenerateTemplateTexts(templateParser, entityDocument, subRecords, contextUser);
|
|
275
|
-
const subEmbedResult = await this.
|
|
593
|
+
const subEmbedResult = await this.EmbeddingRunner.RunEmbedding({
|
|
594
|
+
Texts: subTemplateTexts,
|
|
595
|
+
ModelID: this.embeddingModelID ?? undefined,
|
|
596
|
+
ContextUser: contextUser,
|
|
597
|
+
Description: `Duplicate detection batch (${entityDocument.Name})`
|
|
598
|
+
});
|
|
599
|
+
if (!subEmbedResult.Success || !subEmbedResult.Vectors || subEmbedResult.Vectors.length === 0) {
|
|
600
|
+
throw new Error(`Embedding failed for duplicate detection batch: ${subEmbedResult.ErrorMessage ?? 'Unknown error'}`);
|
|
601
|
+
}
|
|
276
602
|
// Query vector DB for each record in the sub-batch with concurrency control
|
|
277
603
|
this.reportProgress(options, 'Querying', totalRecords, processedSoFar, matchesSoFar, startTime);
|
|
278
|
-
const subQueryResults = await this.QueryDuplicatesForRecords(subRecords, subEmbedResult.
|
|
604
|
+
const subQueryResults = await this.QueryDuplicatesForRecords(subRecords, subEmbedResult.Vectors, subTemplateTexts, entityDocument, topK, options, concurrency);
|
|
279
605
|
allQueryResults.push(...subQueryResults);
|
|
280
606
|
}
|
|
281
607
|
// 6d.5: Drop matches that point to records which no longer exist in the source
|
|
@@ -387,7 +713,7 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
387
713
|
*/
|
|
388
714
|
async InitializeProviders(entityDocument) {
|
|
389
715
|
// Skip re-initialization if providers are already set for this entity document
|
|
390
|
-
if (this.
|
|
716
|
+
if (this._embeddingRunner && this.vectorDB && this.indexName) {
|
|
391
717
|
return;
|
|
392
718
|
}
|
|
393
719
|
if (!entityDocument.AIModelID) {
|
|
@@ -397,27 +723,21 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
397
723
|
throw new Error(`Entity Document '${entityDocument.Name}' (${entityDocument.ID}) cannot be used for duplicate detection: VectorDatabaseID is required but is null.`);
|
|
398
724
|
}
|
|
399
725
|
const aiModel = this.GetAIModel(entityDocument.AIModelID);
|
|
400
|
-
|
|
401
|
-
// to load the right ONNX pipeline; cloud providers fall back to their own default if absent.
|
|
726
|
+
this.embeddingModelID = entityDocument.AIModelID;
|
|
402
727
|
this.embeddingModelAPIName = aiModel.APIName;
|
|
403
728
|
// Captured for the vector query id when the provider keys by EntityDocumentID (SVS).
|
|
404
729
|
this.entityDocumentID = entityDocument.ID;
|
|
405
730
|
const vectorDB = this.GetVectorDatabase(entityDocument.VectorDatabaseID);
|
|
406
|
-
//
|
|
407
|
-
//
|
|
408
|
-
//
|
|
409
|
-
// DB genuinely needs one is decided AFTER instantiation
|
|
410
|
-
//
|
|
411
|
-
// with a more actionable provider-level error. Mirrors EntityVectorSyncer.
|
|
412
|
-
const embeddingAPIKey = GetAIAPIKey(aiModel.DriverClass) || '';
|
|
731
|
+
// Only the vector DB key is resolved here. AIEmbeddingRunner resolves embedding
|
|
732
|
+
// credentials on each call and runs a keyless driver (LocalEmbedding) without one.
|
|
733
|
+
// An empty vector-DB key is legitimate for in-process and colocated providers, so we
|
|
734
|
+
// don't pre-throw; whether the DB genuinely needs one is decided AFTER instantiation
|
|
735
|
+
// via VectorDBBase.RequiresAPIKey. Mirrors EntityVectorSyncer.
|
|
413
736
|
const vectorDBAPIKey = GetAIAPIKey(vectorDB.ClassKey) || '';
|
|
414
|
-
this.
|
|
737
|
+
this._embeddingRunner = new AIEmbeddingRunner();
|
|
415
738
|
// Sentinel when keyless so the base ctor's non-empty requirement is satisfied for
|
|
416
739
|
// local providers (which authenticate via the host process, not a key).
|
|
417
740
|
this.vectorDB = MJGlobal.Instance.ClassFactory.CreateInstance(VectorDBBase, vectorDB.ClassKey, vectorDBAPIKey || 'colocated');
|
|
418
|
-
if (!this.embedding) {
|
|
419
|
-
throw new Error(`Failed to create Embeddings instance for ${aiModel.DriverClass}`);
|
|
420
|
-
}
|
|
421
741
|
if (!this.vectorDB) {
|
|
422
742
|
throw new Error(`Failed to create VectorDB instance for ${vectorDB.ClassKey}`);
|
|
423
743
|
}
|
|
@@ -425,11 +745,11 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
425
745
|
throw new Error(`No API Key found for Vector Database ${vectorDB.ClassKey}`);
|
|
426
746
|
}
|
|
427
747
|
// Resolve the vector index name from the entity document's VectorIndexID
|
|
428
|
-
// Uses
|
|
748
|
+
// Uses the AIEngine vector index cache instead of a RunView query
|
|
429
749
|
if (entityDocument.VectorIndexID) {
|
|
430
|
-
const vectorIndex =
|
|
750
|
+
const vectorIndex = AIEngine.Instance.GetVectorIndexByID(entityDocument.VectorIndexID);
|
|
431
751
|
if (vectorIndex) {
|
|
432
|
-
this.indexName = vectorIndex
|
|
752
|
+
this.indexName = AIEngine.Instance.GetProviderIndexName(vectorIndex);
|
|
433
753
|
}
|
|
434
754
|
}
|
|
435
755
|
if (!this.indexName) {
|
|
@@ -787,13 +1107,7 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
787
1107
|
buildSourceMetadataMap(records, entityInfo) {
|
|
788
1108
|
const metadataMap = new Map();
|
|
789
1109
|
// Combine all IsNameField fields in Sequence order for the display name
|
|
790
|
-
const nameFields = entityInfo
|
|
791
|
-
.filter(f => f.IsNameField)
|
|
792
|
-
.sort((a, b) => (a.Sequence ?? 9999) - (b.Sequence ?? 9999));
|
|
793
|
-
// Fall back to singular NameField if no IsNameField flags
|
|
794
|
-
if (nameFields.length === 0 && entityInfo.NameField) {
|
|
795
|
-
nameFields.push(entityInfo.NameField);
|
|
796
|
-
}
|
|
1110
|
+
const nameFields = this.DisplayNameFields(entityInfo);
|
|
797
1111
|
// Use DefaultInView fields for display, plus IsNameField fields
|
|
798
1112
|
const internalNames = new Set(['ID', '__mj_CreatedAt', '__mj_UpdatedAt']);
|
|
799
1113
|
const displayFields = entityInfo.Fields
|
|
@@ -1065,12 +1379,13 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
1065
1379
|
* independently), so a false-positive candidate reads NotDuplicate even when another candidate
|
|
1066
1380
|
* in the same set is a confident Merge. Falls back to the set-level summary only when the
|
|
1067
1381
|
* reasoner returned no per-candidate verdict for this record. The proposed survivor + field
|
|
1068
|
-
* map stay set-level (they describe the merge of the set's true duplicates).
|
|
1382
|
+
* map stay set-level (they describe the merge of the set's true duplicates). The run id is
|
|
1383
|
+
* the set's, unless this candidate's verdict came from a run of its own.
|
|
1069
1384
|
*
|
|
1070
1385
|
* @param candidateRecordID this candidate's record id (matches the input candidate RecordID)
|
|
1071
1386
|
*/
|
|
1072
1387
|
applyReasoningToMatch(match, reasoning, candidateRecordID) {
|
|
1073
|
-
const verdict =
|
|
1388
|
+
const verdict = this.findCandidateVerdict(reasoning, candidateRecordID);
|
|
1074
1389
|
match.LLMRecommendation = verdict ? verdict.Recommendation : reasoning.Recommendation;
|
|
1075
1390
|
match.LLMConfidence = verdict ? verdict.Confidence : reasoning.Confidence;
|
|
1076
1391
|
match.LLMReasoning = (verdict?.Reasoning || reasoning.Reasoning) || null;
|
|
@@ -1078,9 +1393,25 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
1078
1393
|
match.LLMProposedFieldMap = reasoning.FieldChoices.length > 0
|
|
1079
1394
|
? JSON.stringify(reasoning.FieldChoices)
|
|
1080
1395
|
: null;
|
|
1396
|
+
this.applyRunIDsToMatch(match, reasoning, verdict);
|
|
1397
|
+
}
|
|
1398
|
+
/**
|
|
1399
|
+
* Point the match row at the run that produced its verdict: the verdict's own prompt run when
|
|
1400
|
+
* it carries one (a candidate `DecisionThenPrompt`'s decision dropped), else the set's run.
|
|
1401
|
+
*/
|
|
1402
|
+
applyRunIDsToMatch(match, reasoning, verdict) {
|
|
1403
|
+
if (verdict?.AIPromptRunID) {
|
|
1404
|
+
match.AIPromptRunID = verdict.AIPromptRunID;
|
|
1405
|
+
match.AIAgentRunID = null;
|
|
1406
|
+
return;
|
|
1407
|
+
}
|
|
1081
1408
|
match.AIPromptRunID = reasoning.AIPromptRunID ?? null;
|
|
1082
1409
|
match.AIAgentRunID = reasoning.AIAgentRunID ?? null;
|
|
1083
1410
|
}
|
|
1411
|
+
/** This candidate's own verdict, matched by record id, or undefined when the reasoner returned none for it. */
|
|
1412
|
+
findCandidateVerdict(reasoning, candidateRecordID) {
|
|
1413
|
+
return reasoning.CandidateVerdicts.find(v => this.recordIdMatches(v.RecordID, candidateRecordID));
|
|
1414
|
+
}
|
|
1084
1415
|
// ─────────────────────────────────────────────
|
|
1085
1416
|
// LLM Reasoning (gated, per matched set)
|
|
1086
1417
|
// ─────────────────────────────────────────────
|
|
@@ -1198,14 +1529,17 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
1198
1529
|
/**
|
|
1199
1530
|
* Assemble the reasoning input for a matched set: source description, candidate
|
|
1200
1531
|
* descriptions, and the differing-field deltas loaded for the whole set.
|
|
1532
|
+
*
|
|
1533
|
+
* @param unsavedSource the source record when it is not saved yet (the entry-time check): its
|
|
1534
|
+
* in-memory values stand in for the database row the delta load cannot find
|
|
1201
1535
|
*/
|
|
1202
|
-
async BuildReasoningInput(qr, entityInfo, entityDocument, contextUser) {
|
|
1536
|
+
async BuildReasoningInput(qr, entityInfo, entityDocument, contextUser, unsavedSource) {
|
|
1203
1537
|
// Load the field deltas first — the same pass yields a record-id → name label map, which
|
|
1204
1538
|
// gives BOTH the source and the candidates real names (loaded from the records) instead of
|
|
1205
1539
|
// raw GUIDs. The vector-metadata name is a weaker fallback; the GUID is the last resort.
|
|
1206
1540
|
const deltaBuilder = new MatchedSetDeltaBuilder(this.RunView);
|
|
1207
1541
|
const allKeys = [qr.SourceKey, ...qr.Duplicates.Duplicates];
|
|
1208
|
-
const { FieldDeltas: fieldDeltas, Labels: labels } = await deltaBuilder.Build(entityInfo, allKeys, contextUser);
|
|
1542
|
+
const { FieldDeltas: fieldDeltas, Labels: labels } = await deltaBuilder.Build(entityInfo, allKeys, contextUser, unsavedSource);
|
|
1209
1543
|
const candidates = qr.Duplicates.Duplicates.map(d => ({
|
|
1210
1544
|
RecordID: d.Values(),
|
|
1211
1545
|
Label: this.resolveRecordLabel(labels, d.Values(), this.reasoningLabel(d.VectorMetadata)),
|
|
@@ -1252,11 +1586,12 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
1252
1586
|
return null;
|
|
1253
1587
|
}
|
|
1254
1588
|
/**
|
|
1255
|
-
* Carry the set-level verdict + resolved survivor field map onto the result
|
|
1256
|
-
*
|
|
1257
|
-
*
|
|
1258
|
-
*
|
|
1259
|
-
*
|
|
1589
|
+
* Carry the set-level verdict + resolved survivor field map onto the result, and each
|
|
1590
|
+
* candidate's own verdict onto its {@link PotentialDuplicate}, so the auto-merge step
|
|
1591
|
+
* (AutoMergeAboveAbsolute) can consult both recommendations and apply the literal
|
|
1592
|
+
* {FieldName, Value} overrides via {@link RecordMergeRequest.FieldMap}. The UI still reads
|
|
1593
|
+
* the persisted per-row {@link MJDuplicateRunDetailMatchEntity.LLMProposedFieldMap} (the raw
|
|
1594
|
+
* choices) and lets the reviewer override before a manual merge.
|
|
1260
1595
|
*/
|
|
1261
1596
|
applyReasoningToResult(result, output, fieldMap) {
|
|
1262
1597
|
if (!output.Success) {
|
|
@@ -1265,6 +1600,9 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
1265
1600
|
result.ReasoningRecommendation = output.Recommendation;
|
|
1266
1601
|
result.ReasoningFieldMap = fieldMap.length > 0 ? fieldMap : undefined;
|
|
1267
1602
|
result.ReasoningText = output.Reasoning?.trim() ? output.Reasoning : undefined;
|
|
1603
|
+
for (const dupe of result.Duplicates) {
|
|
1604
|
+
dupe.ReasoningRecommendation = this.findCandidateVerdict(output, dupe.Values())?.Recommendation;
|
|
1605
|
+
}
|
|
1268
1606
|
}
|
|
1269
1607
|
// ─────────────────────────────────────────────
|
|
1270
1608
|
// Auto-Merge
|
|
@@ -1301,7 +1639,9 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
1301
1639
|
* Reasoning path (EnableLLMReasoning = true): `AutomationLevel` governs.
|
|
1302
1640
|
* - ReviewAll / LLMGated → never auto-merge (everything goes to human review).
|
|
1303
1641
|
* - AutoMergeAboveAbsolute → at/above the absolute threshold AND the set's LLM
|
|
1304
|
-
* recommendation is 'Merge'.
|
|
1642
|
+
* recommendation is 'Merge' AND this candidate's own verdict is 'Merge'. A set-level
|
|
1643
|
+
* 'Merge' only says that SOME candidate is a duplicate, so it never merges a candidate
|
|
1644
|
+
* the reasoner judged otherwise, or returned no verdict for.
|
|
1305
1645
|
*/
|
|
1306
1646
|
IsAutoMergeEligible(dupe, dupeResult, entityDocument, absoluteThreshold) {
|
|
1307
1647
|
const aboveAbsolute = dupe.ProbabilityScore >= absoluteThreshold;
|
|
@@ -1311,7 +1651,9 @@ export class DuplicateRecordDetector extends VectorBase {
|
|
|
1311
1651
|
if (entityDocument.AutomationLevel !== 'AutoMergeAboveAbsolute') {
|
|
1312
1652
|
return false;
|
|
1313
1653
|
}
|
|
1314
|
-
return aboveAbsolute
|
|
1654
|
+
return aboveAbsolute
|
|
1655
|
+
&& dupeResult.ReasoningRecommendation === 'Merge'
|
|
1656
|
+
&& dupe.ReasoningRecommendation === 'Merge';
|
|
1315
1657
|
}
|
|
1316
1658
|
/**
|
|
1317
1659
|
* Execute one guarded auto-merge for an eligible candidate, applying the LLM's resolved
|
|
@@ -1387,6 +1729,50 @@ function chunkArray(array, chunkSize) {
|
|
|
1387
1729
|
}
|
|
1388
1730
|
return chunks;
|
|
1389
1731
|
}
|
|
1732
|
+
/**
|
|
1733
|
+
* Whether a reasoning provider is a decision provider. Checked by shape, not `instanceof`, so this
|
|
1734
|
+
* module need not load the decision provider and the AI engine behind it; the package index
|
|
1735
|
+
* registers the provider, as it does every other reasoning mode's.
|
|
1736
|
+
*/
|
|
1737
|
+
function isDecisionReasoningProvider(provider) {
|
|
1738
|
+
return provider != null
|
|
1739
|
+
&& 'DecideCandidates' in provider && typeof provider.DecideCandidates === 'function'
|
|
1740
|
+
&& 'BandCandidate' in provider && typeof provider.BandCandidate === 'function';
|
|
1741
|
+
}
|
|
1742
|
+
/**
|
|
1743
|
+
* The entered values with every string cut to {@link DUPLICATE_ENTRY_CHECK_MAX_FIELD_TEXT_LENGTH}.
|
|
1744
|
+
* Other values pass through as they are.
|
|
1745
|
+
*/
|
|
1746
|
+
function boundEnteredValues(values) {
|
|
1747
|
+
const bounded = {};
|
|
1748
|
+
for (const [name, value] of Object.entries(values)) {
|
|
1749
|
+
bounded[name] = typeof value === 'string' ? truncateText(value) : value;
|
|
1750
|
+
}
|
|
1751
|
+
return bounded;
|
|
1752
|
+
}
|
|
1753
|
+
/**
|
|
1754
|
+
* Text cut to at most `max` characters; a cut text ends with an ellipsis, so a reader (the decision
|
|
1755
|
+
* model included) can tell it was cut.
|
|
1756
|
+
*/
|
|
1757
|
+
function truncateText(text, max = DUPLICATE_ENTRY_CHECK_MAX_FIELD_TEXT_LENGTH) {
|
|
1758
|
+
return text.length <= max ? text : `${text.slice(0, max - 1)}\u2026`;
|
|
1759
|
+
}
|
|
1760
|
+
/**
|
|
1761
|
+
* Orders entity documents oldest first: by `__mj_CreatedAt`, then by ID, so the order is total
|
|
1762
|
+
* and stable across calls.
|
|
1763
|
+
*/
|
|
1764
|
+
function compareOldestFirst(a, b) {
|
|
1765
|
+
const byAge = (a.__mj_CreatedAt?.getTime() ?? 0) - (b.__mj_CreatedAt?.getTime() ?? 0);
|
|
1766
|
+
return byAge !== 0 ? byAge : NormalizeUUID(a.ID).localeCompare(NormalizeUUID(b.ID));
|
|
1767
|
+
}
|
|
1768
|
+
/**
|
|
1769
|
+
* Orders flagged candidates most probable first. A candidate the decision gave no answer for sorts
|
|
1770
|
+
* after every answered one; ties fall back to the vector score.
|
|
1771
|
+
*/
|
|
1772
|
+
function compareMostProbableFirst(a, b) {
|
|
1773
|
+
const byProbability = (b.Probability ?? -1) - (a.Probability ?? -1);
|
|
1774
|
+
return byProbability !== 0 ? byProbability : b.VectorScore - a.VectorScore;
|
|
1775
|
+
}
|
|
1390
1776
|
/**
|
|
1391
1777
|
* Run async tasks with a concurrency limit.
|
|
1392
1778
|
* Executes up to `limit` tasks in parallel, queuing the rest.
|