@mailwoman/match 10.0.0 → 10.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/lib/blocking.ts +41 -36
- package/lib/clustering.ts +43 -26
- package/lib/comparators.ts +11 -34
- package/lib/distance.ts +22 -48
- package/lib/em.ts +14 -6
- package/lib/fellegi-sunter.ts +39 -22
- package/lib/gbt.ts +8 -32
- package/lib/tf.ts +26 -14
- package/out/blocking.d.ts +40 -35
- package/out/blocking.d.ts.map +1 -1
- package/out/blocking.js +34 -25
- package/out/blocking.js.map +1 -1
- package/out/clustering.d.ts +30 -17
- package/out/clustering.d.ts.map +1 -1
- package/out/clustering.js +31 -19
- package/out/clustering.js.map +1 -1
- package/out/comparators.d.ts +11 -33
- package/out/comparators.d.ts.map +1 -1
- package/out/comparators.js +11 -34
- package/out/comparators.js.map +1 -1
- package/out/distance.d.ts +18 -43
- package/out/distance.d.ts.map +1 -1
- package/out/distance.js +16 -41
- package/out/distance.js.map +1 -1
- package/out/em.d.ts +14 -6
- package/out/em.d.ts.map +1 -1
- package/out/em.js +5 -3
- package/out/em.js.map +1 -1
- package/out/fellegi-sunter.d.ts +39 -22
- package/out/fellegi-sunter.d.ts.map +1 -1
- package/out/fellegi-sunter.js +17 -12
- package/out/fellegi-sunter.js.map +1 -1
- package/out/gbt.d.ts +7 -17
- package/out/gbt.d.ts.map +1 -1
- package/out/gbt.js +7 -31
- package/out/gbt.js.map +1 -1
- package/out/tf.d.ts +24 -13
- package/out/tf.d.ts.map +1 -1
- package/out/tf.js +17 -11
- package/out/tf.js.map +1 -1
- package/package.json +13 -88
package/lib/fellegi-sunter.ts
CHANGED
|
@@ -5,9 +5,9 @@
|
|
|
5
5
|
*
|
|
6
6
|
* The Fellegi-Sunter scorer — the matcher's decision layer.
|
|
7
7
|
*
|
|
8
|
-
* Each field comparison
|
|
9
|
-
*
|
|
10
|
-
*
|
|
8
|
+
* Each field comparison assigns a record pair to an _agreement level_: exact, high, low, different, or missing.
|
|
9
|
+
* Each level has two probabilities. `m` is P(this level | the pair really matches).
|
|
10
|
+
* `u` is P(this level | the pair does not match). Their ratio is a Bayes factor. Its
|
|
11
11
|
* log is the level's contribution to the total match weight in bits:
|
|
12
12
|
*
|
|
13
13
|
* ```
|
|
@@ -16,13 +16,12 @@
|
|
|
16
16
|
*
|
|
17
17
|
* — a prior (how likely any two random records match) plus an additive, per-field-attributable
|
|
18
18
|
* stack of evidence. Convert `M` to a probability and threshold it: above the upper bound is a
|
|
19
|
-
* link
|
|
20
|
-
* calibrated abstain zone the whole design leans on.
|
|
19
|
+
* link. Below the lower bound, it is a non-link. The band between them is _clerical review_, the calibrated abstain zone.
|
|
21
20
|
*
|
|
22
|
-
* The `m`/`u` numbers here are
|
|
21
|
+
* The `m`/`u` numbers here are not universal constants. They are estimated from the data — by EM,
|
|
23
22
|
* unsupervised (the next increment) — and the term-frequency adjustment that makes a rare-name
|
|
24
23
|
* agreement count more than a common one layers on top. This module is the deterministic core
|
|
25
|
-
* those build on
|
|
24
|
+
* those build on. Given the levels, it produces the weights and probability. It also produces the decision.
|
|
26
25
|
*/
|
|
27
26
|
|
|
28
27
|
import { nameSimilarity } from "#comparators"
|
|
@@ -40,7 +39,7 @@ export interface ComparisonLevel {
|
|
|
40
39
|
*/
|
|
41
40
|
m: number
|
|
42
41
|
/**
|
|
43
|
-
* P(a pair lands in this level | it is
|
|
42
|
+
* P(a pair lands in this level | it is not a match). A measure of coincidence / cardinality.
|
|
44
43
|
*/
|
|
45
44
|
u: number
|
|
46
45
|
/**
|
|
@@ -70,22 +69,30 @@ export interface Comparison<R> {
|
|
|
70
69
|
*/
|
|
71
70
|
assess(a: R, b: R): number
|
|
72
71
|
/**
|
|
73
|
-
* Optional term-frequency adjustment: on the levels it names, replace the level's
|
|
74
|
-
* value's actual frequency, so agreement on a rare
|
|
72
|
+
* Optional term-frequency adjustment: on the levels it names, replace the level's
|
|
73
|
+
* average `u` with the agreeing value's actual frequency, so agreement on a rare
|
|
74
|
+
* value (`Vijayan`) outweighs agreement on a common one (`Smith`).
|
|
75
|
+
*
|
|
75
76
|
* See `withTermFrequency`.
|
|
76
77
|
*/
|
|
77
78
|
termFrequency?: TermFrequencyAdjustment<R>
|
|
78
79
|
}
|
|
79
80
|
|
|
80
81
|
/**
|
|
81
|
-
* Per-value term-frequency adjustment for a comparison (the Splink/Winkler mechanism).
|
|
82
|
-
*
|
|
83
|
-
*
|
|
84
|
-
*
|
|
82
|
+
* Per-value term-frequency adjustment for a comparison (the Splink/Winkler mechanism).
|
|
83
|
+
*
|
|
84
|
+
* `m` is unchanged.
|
|
85
|
+
* On an agreement level the effective `u` becomes the value's own frequency, adding
|
|
86
|
+
* `log2(u_level / frequency)` to the weight — large and positive for rare values, negative for common ones.
|
|
87
|
+
*
|
|
88
|
+
* Floored at {@link TermFrequencyAdjustment.minimumFrequency} so an ultra-rare
|
|
89
|
+
* value can't produce an unbounded boost.
|
|
85
90
|
*/
|
|
86
91
|
export interface TermFrequencyAdjustment<R> {
|
|
87
92
|
/**
|
|
88
|
-
* Relative frequency of a value in the data, in (0, 1].
|
|
93
|
+
* Relative frequency of a value in the data, in (0, 1].
|
|
94
|
+
*
|
|
95
|
+
* Typically computed on-the-fly.
|
|
89
96
|
*/
|
|
90
97
|
frequency(value: string): number
|
|
91
98
|
/**
|
|
@@ -97,11 +104,15 @@ export interface TermFrequencyAdjustment<R> {
|
|
|
97
104
|
*/
|
|
98
105
|
value(a: R, b: R): string | null | undefined
|
|
99
106
|
/**
|
|
100
|
-
* Scale the adjustment in [0, 1].
|
|
107
|
+
* Scale the adjustment in [0, 1].
|
|
108
|
+
*
|
|
109
|
+
* Default 1.
|
|
101
110
|
*/
|
|
102
111
|
weight?: number
|
|
103
112
|
/**
|
|
104
|
-
* Floor for the looked-up frequency, bounding the boost on ultra-rare values.
|
|
113
|
+
* Floor for the looked-up frequency, bounding the boost on ultra-rare values.
|
|
114
|
+
*
|
|
115
|
+
* Default 1e-4.
|
|
105
116
|
*/
|
|
106
117
|
minimumFrequency?: number
|
|
107
118
|
}
|
|
@@ -168,8 +179,11 @@ export function probabilityFromWeight(weight: number): number {
|
|
|
168
179
|
}
|
|
169
180
|
|
|
170
181
|
/**
|
|
171
|
-
* A comparison driven by a similarity function and a tier of `minSimilarity`
|
|
172
|
-
*
|
|
182
|
+
* A comparison driven by a similarity function and a tier of `minSimilarity`
|
|
183
|
+
* thresholds (the StatCan/Splink recipe).
|
|
184
|
+
*
|
|
185
|
+
* Levels must be ordered highest → lowest similarity, the last acting as the
|
|
186
|
+
* `different` catch-all (`minSimilarity` 0).
|
|
173
187
|
* A missing value on either side yields no evidence.
|
|
174
188
|
*/
|
|
175
189
|
export function similarityComparison<R>(config: {
|
|
@@ -204,7 +218,8 @@ export function similarityComparison<R>(config: {
|
|
|
204
218
|
}
|
|
205
219
|
|
|
206
220
|
/**
|
|
207
|
-
*
|
|
221
|
+
* Scores a record pair and returns its total match weight and probability.
|
|
222
|
+
* The result also includes per-field contributions.
|
|
208
223
|
*/
|
|
209
224
|
export function scorePair<R>(model: FellegiSunterModel<R>, a: R, b: R): PairScore {
|
|
210
225
|
let weight = priorWeight(model.lambda)
|
|
@@ -245,8 +260,10 @@ export function scorePair<R>(model: FellegiSunterModel<R>, a: R, b: R): PairScor
|
|
|
245
260
|
}
|
|
246
261
|
|
|
247
262
|
/**
|
|
248
|
-
* Classify a score against upper / lower match-weight thresholds (in bits): at or above `upper` is a link
|
|
249
|
-
*
|
|
263
|
+
* Classify a score against upper / lower match-weight thresholds (in bits): at or above `upper` is a link.
|
|
264
|
+
*
|
|
265
|
+
* A score at or below `lower` is a non-link.
|
|
266
|
+
* The band between them means clerical review (abstain).
|
|
250
267
|
*/
|
|
251
268
|
export function decide(score: PairScore, thresholds: { upper: number; lower: number }): MatchDecision {
|
|
252
269
|
if (score.weight >= thresholds.upper) return "match"
|
package/lib/gbt.ts
CHANGED
|
@@ -2,30 +2,10 @@
|
|
|
2
2
|
* @copyright Sister Software
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
|
-
*
|
|
6
|
-
* Gradient-boosted shallow regression trees (logistic loss), pure-Node — the learned scorer #603
|
|
7
|
-
* names: an offline-trained model (this trainer, or XGBoost/LightGBM exported to the same
|
|
8
|
-
* {@link GBT} shape) plus a trivial evaluator, no new runtime dependency. It sits behind the
|
|
9
|
-
* matcher's `scorer` hook to replace the Fellegi-Sunter link weight where labels (or a held-out
|
|
10
|
-
* truth like an NPI) let a tree learn the over-merge signature the hand-weights miss.
|
|
11
|
-
*
|
|
12
|
-
* This module is feature-agnostic: feature vectors are caller-defined `number[]` (the record
|
|
13
|
-
* matcher builds them in `@mailwoman/registry`'s learned-scorer module — one-hot agreement
|
|
14
|
-
* levels
|
|
15
|
-
*
|
|
16
|
-
* - Interaction terms + corpus statistics). It only fits ({@link trainGBT}) and scores
|
|
17
|
-
* ({@link gbtScore}). The trained {@link GBT} is plain JSON (`{trees, lr, base}`), so a model
|
|
18
|
-
* trains offline once and ships as a data file.
|
|
19
5
|
*/
|
|
20
6
|
|
|
21
|
-
/**
|
|
22
|
-
* Distinct values at or below which every split point is tried exactly rather than by quantile.
|
|
23
|
-
*/
|
|
24
7
|
const MAX_EXACT_SPLIT_VALUES = 5
|
|
25
8
|
|
|
26
|
-
/**
|
|
27
|
-
* Quantile split points evaluated for a continuous feature.
|
|
28
|
-
*/
|
|
29
9
|
const QUANTILE_SPLIT_COUNT = 6
|
|
30
10
|
|
|
31
11
|
/**
|
|
@@ -33,11 +13,12 @@ const QUANTILE_SPLIT_COUNT = 6
|
|
|
33
13
|
*/
|
|
34
14
|
export type TreeNode = { leaf: number } | { f: number; thr: number; lo: TreeNode; hi: TreeNode }
|
|
35
15
|
|
|
36
|
-
// Local by design: `@mailwoman/registry`'s tools/shared.ts exports this sigmoid; registry depends on match, not the reverse.
|
|
37
16
|
const sigmoid = (z: number): number => 1 / (1 + Math.exp(-Math.max(-30, Math.min(30, z))))
|
|
38
17
|
|
|
39
18
|
/**
|
|
40
|
-
*
|
|
19
|
+
* Returns each feature's candidate split thresholds: midpoints between values
|
|
20
|
+
* for features with at most five distinct values.
|
|
21
|
+
* It returns six quantiles otherwise.
|
|
41
22
|
*/
|
|
42
23
|
export function buildThresholds(X: number[][]): number[][] {
|
|
43
24
|
const dim = X[0]?.length ?? 0
|
|
@@ -72,9 +53,6 @@ export function buildThresholds(X: number[][]): number[][] {
|
|
|
72
53
|
return out
|
|
73
54
|
}
|
|
74
55
|
|
|
75
|
-
/**
|
|
76
|
-
* Weighted SSE of target `g` over `rows` around their weighted mean.
|
|
77
|
-
*/
|
|
78
56
|
function nodeSSE(rows: number[], g: number[], w: number[]): number {
|
|
79
57
|
let wsum = 0
|
|
80
58
|
let wg = 0
|
|
@@ -95,9 +73,6 @@ function nodeSSE(rows: number[], g: number[], w: number[]): number {
|
|
|
95
73
|
return sse
|
|
96
74
|
}
|
|
97
75
|
|
|
98
|
-
/**
|
|
99
|
-
* Greedy depth-limited weighted regression tree on target `g` (the boosting residual).
|
|
100
|
-
*/
|
|
101
76
|
function fitRegTree(
|
|
102
77
|
rows: number[],
|
|
103
78
|
X: number[][],
|
|
@@ -168,7 +143,7 @@ function predictTree(t: TreeNode, x: number[]): number {
|
|
|
168
143
|
}
|
|
169
144
|
|
|
170
145
|
/**
|
|
171
|
-
* A trained gradient-boosted
|
|
146
|
+
* A trained gradient-boosted tree ensemble whose trees add to a base log-odds, stored as plain JSON.
|
|
172
147
|
*/
|
|
173
148
|
export interface GBT {
|
|
174
149
|
trees: TreeNode[]
|
|
@@ -204,7 +179,7 @@ export function trainGBT(X: number[][], y: number[], w: number[], opts: GBTOpts)
|
|
|
204
179
|
}
|
|
205
180
|
}
|
|
206
181
|
|
|
207
|
-
const base = Math.log((wpos + 1) / (wtot - wpos + 1))
|
|
182
|
+
const base = Math.log((wpos + 1) / (wtot - wpos + 1))
|
|
208
183
|
const F = new Array<number>(N).fill(base)
|
|
209
184
|
const trees: TreeNode[] = []
|
|
210
185
|
|
|
@@ -215,7 +190,6 @@ export function trainGBT(X: number[][], y: number[], w: number[], opts: GBTOpts)
|
|
|
215
190
|
g[i] = y[i]! - sigmoid(F[i]!)
|
|
216
191
|
}
|
|
217
192
|
|
|
218
|
-
// negative gradient of logistic loss
|
|
219
193
|
const tree = fitRegTree(rowsAll, X, g, w, thresholds, opts.depth, opts.minLeaf)
|
|
220
194
|
|
|
221
195
|
for (let i = 0; i < N; i++) {
|
|
@@ -229,7 +203,9 @@ export function trainGBT(X: number[][], y: number[], w: number[], opts: GBTOpts)
|
|
|
229
203
|
}
|
|
230
204
|
|
|
231
205
|
/**
|
|
232
|
-
*
|
|
206
|
+
* Returns the model's logit for one feature vector.
|
|
207
|
+
*
|
|
208
|
+
* Compare it against a threshold like a Fellegi-Sunter match weight.
|
|
233
209
|
*/
|
|
234
210
|
export function gbtScore(m: GBT, x: number[]): number {
|
|
235
211
|
let f = m.base
|
package/lib/tf.ts
CHANGED
|
@@ -5,15 +5,15 @@
|
|
|
5
5
|
*
|
|
6
6
|
* Term-frequency adjustment — making a rare-value agreement count more than a common one.
|
|
7
7
|
*
|
|
8
|
-
* Two people
|
|
8
|
+
* Two people who share the name "Vijayan" are far stronger evidence of a match than two who share the name "Smith",
|
|
9
9
|
* because "Smith" agreements happen by chance all the time and "Vijayan" agreements don't. The
|
|
10
|
-
* Fellegi-Sunter `m` (how often a true match agrees) is roughly the same either way
|
|
10
|
+
* Fellegi-Sunter `m` (how often a true match agrees) is roughly the same either way. what differs
|
|
11
11
|
* is `u` — the chance a _non_-match agrees — which for an exact agreement on value `v` is just
|
|
12
12
|
* how common `v` is. So we leave `m`, and replace the level's average `u` with `frequency(v)`,
|
|
13
13
|
* adding `log2(u_level / frequency(v))` to the weight: a big positive bump for rare values, a
|
|
14
14
|
* penalty for common ones.
|
|
15
15
|
*
|
|
16
|
-
* Crucially for a label-free matcher: the frequencies are computed
|
|
16
|
+
* Crucially for a label-free matcher: the frequencies are computed on-the-FLY from the input column
|
|
17
17
|
* (the Splink approach) — no external Census table required. Build a {@link TermFrequencyTable}
|
|
18
18
|
* from the values you're matching, then attach it to a comparison with {@link withTermFrequency}.
|
|
19
19
|
*/
|
|
@@ -38,13 +38,15 @@ export interface TermFrequencyTable {
|
|
|
38
38
|
readonly distinct: number
|
|
39
39
|
}
|
|
40
40
|
|
|
41
|
-
// Local by design: `@mailwoman/record`'s per-field normalizers are the shared home
|
|
41
|
+
// Local by design: `@mailwoman/record`'s per-field normalizers are the shared home.
|
|
42
|
+
// Match takes no record dependency.
|
|
42
43
|
const defaultNormalize = (value: string): string => value.trim().toLowerCase().replaceAll(/\s+/g, " ")
|
|
43
44
|
|
|
44
45
|
/**
|
|
45
|
-
* Build a {@link TermFrequencyTable} from an iterable of values (e.g.
|
|
46
|
-
*
|
|
47
|
-
*
|
|
46
|
+
* Build a {@link TermFrequencyTable} from an iterable of values (e.g. Every `given` name in the dataset).
|
|
47
|
+
*
|
|
48
|
+
* Values are normalized (default: trim + lowercase + collapse whitespace) before counting.
|
|
49
|
+
* `frequency()` normalizes its argument the same way, so callers pass raw field values.
|
|
48
50
|
*/
|
|
49
51
|
export function buildTermFrequencyTable(
|
|
50
52
|
values: Iterable<string | null | undefined>,
|
|
@@ -76,10 +78,14 @@ export function buildTermFrequencyTable(
|
|
|
76
78
|
}
|
|
77
79
|
|
|
78
80
|
/**
|
|
79
|
-
* Attach a term-frequency adjustment to a comparison.
|
|
80
|
-
*
|
|
81
|
-
*
|
|
82
|
-
*
|
|
81
|
+
* Attach a term-frequency adjustment to a comparison.
|
|
82
|
+
*
|
|
83
|
+
* By default it applies to the exact level (index 0) and looks up the value via
|
|
84
|
+
* `value(a, b)` — usually the agreeing field extracted from one side.
|
|
85
|
+
* Returns a new comparison.
|
|
86
|
+
*
|
|
87
|
+
* The underlying `assess` and levels are untouched, so this composes with EM
|
|
88
|
+
* (which re-estimates the base `m`/`u` the adjustment sits on top of).
|
|
83
89
|
*/
|
|
84
90
|
export function withTermFrequency<R>(
|
|
85
91
|
comparison: Comparison<R>,
|
|
@@ -87,15 +93,21 @@ export function withTermFrequency<R>(
|
|
|
87
93
|
table: TermFrequencyTable
|
|
88
94
|
value: (a: R, b: R) => string | null | undefined
|
|
89
95
|
/**
|
|
90
|
-
* Level indices to adjust.
|
|
96
|
+
* Level indices to adjust.
|
|
97
|
+
*
|
|
98
|
+
* Default `[0]` (the exact level).
|
|
91
99
|
*/
|
|
92
100
|
levels?: Iterable<number>
|
|
93
101
|
/**
|
|
94
|
-
* Scale in [0, 1].
|
|
102
|
+
* Scale in [0, 1].
|
|
103
|
+
*
|
|
104
|
+
* Default 1.
|
|
95
105
|
*/
|
|
96
106
|
weight?: number
|
|
97
107
|
/**
|
|
98
|
-
* Frequency floor bounding the boost on ultra-rare values.
|
|
108
|
+
* Frequency floor bounding the boost on ultra-rare values.
|
|
109
|
+
*
|
|
110
|
+
* Default 1e-4.
|
|
99
111
|
*/
|
|
100
112
|
minimumFrequency?: number
|
|
101
113
|
}
|
package/out/blocking.d.ts
CHANGED
|
@@ -3,58 +3,62 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
* comparisons
|
|
6
|
+
* The blocking stage generates candidate pairs. An all-pairs comparison is O(n²): a million records require a trillion
|
|
7
|
+
* comparisons. This stage scores pairs that share a cheap key. This is where the geocode-first
|
|
8
8
|
* bet pays off: two records resolving to the same place land in the same spatial cell regardless
|
|
9
9
|
* of how their address strings are spelled, so geography is the primary block.
|
|
10
10
|
*
|
|
11
|
-
* A {@link BlockingKey} maps a record to zero or more string keys
|
|
11
|
+
* A {@link BlockingKey} maps a record to zero or more string keys. records sharing any key become
|
|
12
12
|
* candidates. Keys compose as a _union_ (the standard multi-pass approach — high recall from
|
|
13
|
-
* cheap rules): block on the spatial cell
|
|
14
|
-
* any rule
|
|
15
|
-
*
|
|
13
|
+
* cheap rules): block on the spatial cell, canonical key, or postcode. A pair caught by
|
|
14
|
+
* any rule is scored. {@link conjunction} builds the and-style key Geo-ER uses
|
|
15
|
+
* (`name-cell and geo-cell`) when a single rule is too loose.
|
|
16
16
|
*
|
|
17
|
-
* Recall is the priority
|
|
18
|
-
* silent failure in record linkage.
|
|
19
|
-
* default
|
|
17
|
+
* Recall is the priority. A pair the blocker never proposes can never match, the most dangerous
|
|
18
|
+
* silent failure in record linkage. The spatial grid is generous and neighbor-expanded by
|
|
19
|
+
* default. The code reports any block too large to scan instead of dropping it silently.
|
|
20
20
|
*/
|
|
21
|
+
import type { GeoCoordinate } from "@mailwoman/spatial";
|
|
21
22
|
/**
|
|
22
|
-
* Maps a record to zero or more block keys.
|
|
23
|
+
* Maps a record to zero or more block keys.
|
|
24
|
+
*
|
|
25
|
+
* Two records sharing any key become a candidate pair.
|
|
23
26
|
*/
|
|
24
27
|
export type BlockingKey<R> = (record: R) => string[];
|
|
25
28
|
/**
|
|
26
|
-
* A
|
|
27
|
-
*/
|
|
28
|
-
export interface LatLon {
|
|
29
|
-
latitude: number;
|
|
30
|
-
longitude: number;
|
|
31
|
-
}
|
|
32
|
-
/**
|
|
33
|
-
* A spatial-cell block key: a configurable lat/lon grid. `precisionDegrees` sets the cell size (default 0.05° ≈ 5.5 km
|
|
34
|
-
* of latitude — deliberately generous, per the literature, so same-place records reliably co-block). With `neighbors`
|
|
35
|
-
* (default `true`) a record also keys its 8 adjacent cells, so a pair straddling a cell boundary still meets.
|
|
29
|
+
* A spatial-cell block key: a configurable lat/lon grid.
|
|
36
30
|
*
|
|
37
|
-
*
|
|
38
|
-
*
|
|
39
|
-
*
|
|
31
|
+
* `precisionDegrees` sets the cell size (default 0.05° ≈ 5.5 km of latitude —
|
|
32
|
+
* deliberately generous, per the literature, so same-place records reliably co-block).
|
|
33
|
+
* With `neighbors` (default `true`) a record also keys its 8 adjacent cells,
|
|
34
|
+
* so a pair straddling a cell boundary still meets.
|
|
35
|
+
*
|
|
36
|
+
* Note: an equal-_degree_ grid (longitude cells shrink toward the poles)
|
|
37
|
+
* and neighbor expansion inflates block sizes ~9×; an equal-area H3/geohash index
|
|
38
|
+
* with a single-cell + neighbor-query is the refinement.
|
|
39
|
+
* Behavior — proximity co-blocking — is the same.
|
|
40
40
|
*/
|
|
41
|
-
export declare function geoCellKey<R>(extract: (record: R) =>
|
|
41
|
+
export declare function geoCellKey<R>(extract: (record: R) => GeoCoordinate | null | undefined, opts?: {
|
|
42
42
|
precisionDegrees?: number;
|
|
43
43
|
neighbors?: boolean;
|
|
44
44
|
}): BlockingKey<R>;
|
|
45
45
|
/**
|
|
46
|
-
* An exact-value block key (the canonical address key, a postcode, an email domain…), normalized
|
|
47
|
-
* truncated to a leading `prefix` of characters (a cheaper, higher-recall rule).
|
|
48
|
-
*
|
|
46
|
+
* An exact-value block key (the canonical address key, a postcode, an email domain…), normalized
|
|
47
|
+
* and optionally truncated to a leading `prefix` of characters (a cheaper, higher-recall rule).
|
|
48
|
+
*
|
|
49
|
+
* A missing or empty value produces no key.
|
|
49
50
|
*/
|
|
50
51
|
export declare function exactKey<R>(extract: (record: R) => string | null | undefined, opts?: {
|
|
51
52
|
prefix?: number;
|
|
52
53
|
normalize?: (value: string) => string;
|
|
53
54
|
}): BlockingKey<R>;
|
|
54
55
|
/**
|
|
55
|
-
* A conjunctive block key — the cross-product of its sub-keys, joined (Geo-ER's "name
|
|
56
|
-
*
|
|
57
|
-
*
|
|
56
|
+
* A conjunctive block key — the cross-product of its sub-keys, joined (Geo-ER's "name and distance").
|
|
57
|
+
*
|
|
58
|
+
* A record is keyed by every combination of one sub-key from each input,
|
|
59
|
+
* so two records co-block only when they agree on _all_ inputs.
|
|
60
|
+
* Tighter blocks, lower recall.
|
|
61
|
+
* Use when a single rule is too loose.
|
|
58
62
|
*/
|
|
59
63
|
export declare function conjunction<R>(...keys: BlockingKey<R>[]): BlockingKey<R>;
|
|
60
64
|
/**
|
|
@@ -62,7 +66,7 @@ export declare function conjunction<R>(...keys: BlockingKey<R>[]): BlockingKey<R
|
|
|
62
66
|
*/
|
|
63
67
|
export interface BlockResult<R> {
|
|
64
68
|
/**
|
|
65
|
-
* Deduplicated candidate pairs (no self-pairs
|
|
69
|
+
* Deduplicated candidate pairs (no self-pairs. A pair caught by multiple keys appears once).
|
|
66
70
|
*/
|
|
67
71
|
pairs: Array<[R, R]>;
|
|
68
72
|
/**
|
|
@@ -74,10 +78,11 @@ export interface BlockResult<R> {
|
|
|
74
78
|
}>;
|
|
75
79
|
}
|
|
76
80
|
/**
|
|
77
|
-
* Generate candidate pairs from `records` via one or more blocking keys (their union).
|
|
78
|
-
*
|
|
79
|
-
*
|
|
80
|
-
*
|
|
81
|
+
* Generate candidate pairs from `records` via one or more blocking keys (their union).
|
|
82
|
+
*
|
|
83
|
+
* Builds an inverted index (key → records) and emits the unique within-block pairs.
|
|
84
|
+
* A block larger than `maxBlockSize` is skipped and reported in `droppedBlocks` rather than blowing
|
|
85
|
+
* up into a quadratic scan — an explicit, visible coverage limit rather than a silent drop.
|
|
81
86
|
*/
|
|
82
87
|
export declare function block<R>(records: readonly R[], blockingKeys: BlockingKey<R> | BlockingKey<R>[], opts?: {
|
|
83
88
|
maxBlockSize?: number;
|
package/out/blocking.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"blocking.d.ts","sourceRoot":"","sources":["../lib/blocking.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AAEH
|
|
1
|
+
{"version":3,"file":"blocking.d.ts","sourceRoot":"","sources":["../lib/blocking.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AAEH,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,oBAAoB,CAAA;AAEvD;;;;GAIG;AACH,MAAM,MAAM,WAAW,CAAC,CAAC,IAAI,CAAC,MAAM,EAAE,CAAC,KAAK,MAAM,EAAE,CAAA;AAEpD;;;;;;;;;;;;GAYG;AACH,wBAAgB,UAAU,CAAC,CAAC,EAC3B,OAAO,EAAE,CAAC,MAAM,EAAE,CAAC,KAAK,aAAa,GAAG,IAAI,GAAG,SAAS,EACxD,IAAI,GAAE;IAAE,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAAC,SAAS,CAAC,EAAE,OAAO,CAAA;CAAO,GAC3D,WAAW,CAAC,CAAC,CAAC,CAwBhB;AAED;;;;;GAKG;AACH,wBAAgB,QAAQ,CAAC,CAAC,EACzB,OAAO,EAAE,CAAC,MAAM,EAAE,CAAC,KAAK,MAAM,GAAG,IAAI,GAAG,SAAS,EACjD,IAAI,GAAE;IAAE,MAAM,CAAC,EAAE,MAAM,CAAC;IAAC,SAAS,CAAC,EAAE,CAAC,KAAK,EAAE,MAAM,KAAK,MAAM,CAAA;CAAO,GACnE,WAAW,CAAC,CAAC,CAAC,CAahB;AAED;;;;;;;GAOG;AACH,wBAAgB,WAAW,CAAC,CAAC,EAAE,GAAG,IAAI,EAAE,WAAW,CAAC,CAAC,CAAC,EAAE,GAAG,WAAW,CAAC,CAAC,CAAC,CAaxE;AAED;;GAEG;AACH,MAAM,WAAW,WAAW,CAAC,CAAC;IAC7B;;OAEG;IACH,KAAK,EAAE,KAAK,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,CAAA;IACpB;;OAEG;IACH,aAAa,EAAE,KAAK,CAAC;QAAE,GAAG,EAAE,MAAM,CAAC;QAAC,IAAI,EAAE,MAAM,CAAA;KAAE,CAAC,CAAA;CACnD;AAED;;;;;;GAMG;AACH,wBAAgB,KAAK,CAAC,CAAC,EACtB,OAAO,EAAE,SAAS,CAAC,EAAE,EACrB,YAAY,EAAE,WAAW,CAAC,CAAC,CAAC,GAAG,WAAW,CAAC,CAAC,CAAC,EAAE,EAC/C,IAAI,GAAE;IAAE,YAAY,CAAC,EAAE,MAAM,CAAA;CAAO,GAClC,WAAW,CAAC,CAAC,CAAC,CAmDhB"}
|
package/out/blocking.js
CHANGED
|
@@ -3,29 +3,33 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
* comparisons
|
|
6
|
+
* The blocking stage generates candidate pairs. An all-pairs comparison is O(n²): a million records require a trillion
|
|
7
|
+
* comparisons. This stage scores pairs that share a cheap key. This is where the geocode-first
|
|
8
8
|
* bet pays off: two records resolving to the same place land in the same spatial cell regardless
|
|
9
9
|
* of how their address strings are spelled, so geography is the primary block.
|
|
10
10
|
*
|
|
11
|
-
* A {@link BlockingKey} maps a record to zero or more string keys
|
|
11
|
+
* A {@link BlockingKey} maps a record to zero or more string keys. records sharing any key become
|
|
12
12
|
* candidates. Keys compose as a _union_ (the standard multi-pass approach — high recall from
|
|
13
|
-
* cheap rules): block on the spatial cell
|
|
14
|
-
* any rule
|
|
15
|
-
*
|
|
13
|
+
* cheap rules): block on the spatial cell, canonical key, or postcode. A pair caught by
|
|
14
|
+
* any rule is scored. {@link conjunction} builds the and-style key Geo-ER uses
|
|
15
|
+
* (`name-cell and geo-cell`) when a single rule is too loose.
|
|
16
16
|
*
|
|
17
|
-
* Recall is the priority
|
|
18
|
-
* silent failure in record linkage.
|
|
19
|
-
* default
|
|
17
|
+
* Recall is the priority. A pair the blocker never proposes can never match, the most dangerous
|
|
18
|
+
* silent failure in record linkage. The spatial grid is generous and neighbor-expanded by
|
|
19
|
+
* default. The code reports any block too large to scan instead of dropping it silently.
|
|
20
20
|
*/
|
|
21
21
|
/**
|
|
22
|
-
* A spatial-cell block key: a configurable lat/lon grid.
|
|
23
|
-
* of latitude — deliberately generous, per the literature, so same-place records reliably co-block). With `neighbors`
|
|
24
|
-
* (default `true`) a record also keys its 8 adjacent cells, so a pair straddling a cell boundary still meets.
|
|
22
|
+
* A spatial-cell block key: a configurable lat/lon grid.
|
|
25
23
|
*
|
|
26
|
-
*
|
|
27
|
-
*
|
|
28
|
-
*
|
|
24
|
+
* `precisionDegrees` sets the cell size (default 0.05° ≈ 5.5 km of latitude —
|
|
25
|
+
* deliberately generous, per the literature, so same-place records reliably co-block).
|
|
26
|
+
* With `neighbors` (default `true`) a record also keys its 8 adjacent cells,
|
|
27
|
+
* so a pair straddling a cell boundary still meets.
|
|
28
|
+
*
|
|
29
|
+
* Note: an equal-_degree_ grid (longitude cells shrink toward the poles)
|
|
30
|
+
* and neighbor expansion inflates block sizes ~9×; an equal-area H3/geohash index
|
|
31
|
+
* with a single-cell + neighbor-query is the refinement.
|
|
32
|
+
* Behavior — proximity co-blocking — is the same.
|
|
29
33
|
*/
|
|
30
34
|
export function geoCellKey(extract, opts = {}) {
|
|
31
35
|
const step = opts.precisionDegrees ?? 0.05;
|
|
@@ -48,9 +52,10 @@ export function geoCellKey(extract, opts = {}) {
|
|
|
48
52
|
};
|
|
49
53
|
}
|
|
50
54
|
/**
|
|
51
|
-
* An exact-value block key (the canonical address key, a postcode, an email domain…), normalized
|
|
52
|
-
* truncated to a leading `prefix` of characters (a cheaper, higher-recall rule).
|
|
53
|
-
*
|
|
55
|
+
* An exact-value block key (the canonical address key, a postcode, an email domain…), normalized
|
|
56
|
+
* and optionally truncated to a leading `prefix` of characters (a cheaper, higher-recall rule).
|
|
57
|
+
*
|
|
58
|
+
* A missing or empty value produces no key.
|
|
54
59
|
*/
|
|
55
60
|
export function exactKey(extract, opts = {}) {
|
|
56
61
|
const normalize = opts.normalize ?? ((v) => v.trim().toLowerCase().replaceAll(/\s+/g, " "));
|
|
@@ -65,9 +70,12 @@ export function exactKey(extract, opts = {}) {
|
|
|
65
70
|
};
|
|
66
71
|
}
|
|
67
72
|
/**
|
|
68
|
-
* A conjunctive block key — the cross-product of its sub-keys, joined (Geo-ER's "name
|
|
69
|
-
*
|
|
70
|
-
*
|
|
73
|
+
* A conjunctive block key — the cross-product of its sub-keys, joined (Geo-ER's "name and distance").
|
|
74
|
+
*
|
|
75
|
+
* A record is keyed by every combination of one sub-key from each input,
|
|
76
|
+
* so two records co-block only when they agree on _all_ inputs.
|
|
77
|
+
* Tighter blocks, lower recall.
|
|
78
|
+
* Use when a single rule is too loose.
|
|
71
79
|
*/
|
|
72
80
|
export function conjunction(...keys) {
|
|
73
81
|
return (record) => {
|
|
@@ -82,10 +90,11 @@ export function conjunction(...keys) {
|
|
|
82
90
|
};
|
|
83
91
|
}
|
|
84
92
|
/**
|
|
85
|
-
* Generate candidate pairs from `records` via one or more blocking keys (their union).
|
|
86
|
-
*
|
|
87
|
-
*
|
|
88
|
-
*
|
|
93
|
+
* Generate candidate pairs from `records` via one or more blocking keys (their union).
|
|
94
|
+
*
|
|
95
|
+
* Builds an inverted index (key → records) and emits the unique within-block pairs.
|
|
96
|
+
* A block larger than `maxBlockSize` is skipped and reported in `droppedBlocks` rather than blowing
|
|
97
|
+
* up into a quadratic scan — an explicit, visible coverage limit rather than a silent drop.
|
|
89
98
|
*/
|
|
90
99
|
export function block(records, blockingKeys, opts = {}) {
|
|
91
100
|
const keys = Array.isArray(blockingKeys) ? blockingKeys : [blockingKeys];
|
package/out/blocking.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"blocking.js","sourceRoot":"","sources":["../lib/blocking.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;
|
|
1
|
+
{"version":3,"file":"blocking.js","sourceRoot":"","sources":["../lib/blocking.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AAWH;;;;;;;;;;;;GAYG;AACH,MAAM,UAAU,UAAU,CACzB,OAAwD,EACxD,IAAI,GAAuD,EAAE;IAE7D,MAAM,IAAI,GAAG,IAAI,CAAC,gBAAgB,IAAI,IAAI,CAAA;IAC1C,MAAM,MAAM,GAAG,IAAI,CAAC,SAAS,IAAI,IAAI,CAAA;IAErC,OAAO,CAAC,MAAM,EAAE,EAAE;QACjB,MAAM,UAAU,GAAG,OAAO,CAAC,MAAM,CAAC,CAAA;QAElC,IAAI,CAAC,UAAU,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,UAAU,CAAC,QAAQ,CAAC,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,UAAU,CAAC,SAAS,CAAC;YAAE,OAAO,EAAE,CAAA;QAE7G,MAAM,OAAO,GAAG,IAAI,CAAC,KAAK,CAAC,UAAU,CAAC,QAAQ,GAAG,IAAI,CAAC,CAAA;QACtD,MAAM,OAAO,GAAG,IAAI,CAAC,KAAK,CAAC,UAAU,CAAC,SAAS,GAAG,IAAI,CAAC,CAAA;QAEvD,IAAI,CAAC,MAAM;YAAE,OAAO,CAAC,GAAG,OAAO,IAAI,OAAO,EAAE,CAAC,CAAA;QAE7C,MAAM,IAAI,GAAa,EAAE,CAAA;QAEzB,KAAK,IAAI,IAAI,GAAG,CAAC,CAAC,EAAE,IAAI,IAAI,CAAC,EAAE,IAAI,EAAE,EAAE,CAAC;YACvC,KAAK,IAAI,IAAI,GAAG,CAAC,CAAC,EAAE,IAAI,IAAI,CAAC,EAAE,IAAI,EAAE,EAAE,CAAC;gBACvC,IAAI,CAAC,IAAI,CAAC,GAAG,OAAO,GAAG,IAAI,IAAI,OAAO,GAAG,IAAI,EAAE,CAAC,CAAA;YACjD,CAAC;QACF,CAAC;QAED,OAAO,IAAI,CAAA;IACZ,CAAC,CAAA;AACF,CAAC;AAED;;;;;GAKG;AACH,MAAM,UAAU,QAAQ,CACvB,OAAiD,EACjD,IAAI,GAA+D,EAAE;IAErE,MAAM,SAAS,GAAG,IAAI,CAAC,SAAS,IAAI,CAAC,CAAC,CAAS,EAAE,EAAE,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC,WAAW,EAAE,CAAC,UAAU,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC,CAAA;IAEnG,OAAO,CAAC,MAAM,EAAE,EAAE;QACjB,MAAM,KAAK,GAAG,OAAO,CAAC,MAAM,CAAC,CAAA;QAE7B,IAAI,CAAC,KAAK;YAAE,OAAO,EAAE,CAAA;QACrB,MAAM,UAAU,GAAG,SAAS,CAAC,KAAK,CAAC,CAAA;QAEnC,IAAI,CAAC,UAAU;YAAE,OAAO,EAAE,CAAA;QAE1B,OAAO,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC,CAAC,UAAU,CAAC,KAAK,CAAC,CAAC,EAAE,IAAI,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,UAAU,CAAC,CAAA;IACrE,CAAC,CAAA;AACF,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,WAAW,CAAI,GAAG,IAAsB;IACvD,OAAO,CAAC,MAAM,EAAE,EAAE;QACjB,IAAI,MAAM,GAAG,CAAC,EAAE,CAAC,CAAA;QAEjB,KAAK,MAAM,GAAG,IAAI,IAAI,EAAE,CAAC;YACxB,MAAM,KAAK,GAAG,GAAG,CAAC,MAAM,CAAC,CAAA;YAEzB,IAAI,CAAC,KAAK,CAAC,MAAM;gBAAE,OAAO,EAAE,CAAA;YAC5B,MAAM,GAAG,MAAM,CAAC,OAAO,CAAC,CAAC,MAAM,EAAE,EAAE,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,GAAG,MAAM,IAAI,IAAI,EAAE,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,CAAA;QAChG,CAAC;QAED,OAAO,MAAM,CAAA;IACd,CAAC,CAAA;AACF,CAAC;AAgBD;;;;;;GAMG;AACH,MAAM,UAAU,KAAK,CACpB,OAAqB,EACrB,YAA+C,EAC/C,IAAI,GAA8B,EAAE;IAEpC,MAAM,IAAI,GAAG,KAAK,CAAC,OAAO,CAAC,YAAY,CAAC,CAAC,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,CAAC,YAAY,CAAC,CAAA;IACxE,MAAM,YAAY,GAAG,IAAI,CAAC,YAAY,IAAI,QAAQ,CAAA;IAClD,MAAM,KAAK,GAAG,IAAI,GAAG,EAAoB,CAAA;IAEzC,OAAO,CAAC,OAAO,CAAC,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE;QAC7B,MAAM,IAAI,GAAG,IAAI,GAAG,EAAU,CAAA;QAE9B,KAAK,MAAM,KAAK,IAAI,IAAI,EAAE,CAAC;YAC1B,KAAK,MAAM,GAAG,IAAI,KAAK,CAAC,MAAM,CAAC,EAAE,CAAC;gBACjC,IAAI,CAAC,GAAG,IAAI,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC;oBAAE,SAAQ;gBACnC,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,CAAA;gBACb,MAAM,MAAM,GAAG,KAAK,CAAC,GAAG,CAAC,GAAG,CAAC,CAAA;gBAE7B,IAAI,MAAM,EAAE,CAAC;oBACZ,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,CAAA;gBACf,CAAC;qBAAM,CAAC;oBACP,KAAK,CAAC,GAAG,CAAC,GAAG,EAAE,CAAC,CAAC,CAAC,CAAC,CAAA;gBACpB,CAAC;YACF,CAAC;QACF,CAAC;IACF,CAAC,CAAC,CAAA;IAEF,MAAM,CAAC,GAAG,OAAO,CAAC,MAAM,CAAA;IACxB,MAAM,OAAO,GAAG,IAAI,GAAG,EAAU,CAAA;IACjC,MAAM,KAAK,GAAkB,EAAE,CAAA;IAC/B,MAAM,aAAa,GAAoC,EAAE,CAAA;IAEzD,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,KAAK,EAAE,CAAC;QACnC,IAAI,MAAM,CAAC,MAAM,GAAG,CAAC;YAAE,SAAQ;QAE/B,IAAI,MAAM,CAAC,MAAM,GAAG,YAAY,EAAE,CAAC;YAClC,aAAa,CAAC,IAAI,CAAC,EAAE,GAAG,EAAE,IAAI,EAAE,MAAM,CAAC,MAAM,EAAE,CAAC,CAAA;YAEhD,SAAQ;QACT,CAAC;QAED,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,MAAM,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;YACxC,KAAK,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,MAAM,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;gBAC5C,MAAM,EAAE,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,CAAC,CAAE,EAAE,MAAM,CAAC,CAAC,CAAE,CAAC,CAAA;gBAC3C,MAAM,EAAE,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,CAAC,CAAE,EAAE,MAAM,CAAC,CAAC,CAAE,CAAC,CAAA;gBAC3C,MAAM,EAAE,GAAG,EAAE,GAAG,CAAC,GAAG,EAAE,CAAA;gBAEtB,IAAI,OAAO,CAAC,GAAG,CAAC,EAAE,CAAC;oBAAE,SAAQ;gBAC7B,OAAO,CAAC,GAAG,CAAC,EAAE,CAAC,CAAA;gBACf,KAAK,CAAC,IAAI,CAAC,CAAC,OAAO,CAAC,EAAE,CAAE,EAAE,OAAO,CAAC,EAAE,CAAE,CAAC,CAAC,CAAA;YACzC,CAAC;QACF,CAAC;IACF,CAAC;IAED,OAAO,EAAE,KAAK,EAAE,aAAa,EAAE,CAAA;AAChC,CAAC"}
|