@mailwoman/match 9.4.0 → 10.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/lib/blocking.ts +41 -36
- package/lib/clustering.ts +43 -26
- package/lib/comparators.ts +11 -34
- package/lib/distance.ts +22 -48
- package/lib/em.ts +14 -6
- package/lib/fellegi-sunter.ts +39 -22
- package/lib/gbt.ts +8 -32
- package/lib/tf.ts +26 -14
- package/out/blocking.d.ts +40 -35
- package/out/blocking.d.ts.map +1 -1
- package/out/blocking.js +34 -25
- package/out/blocking.js.map +1 -1
- package/out/clustering.d.ts +30 -17
- package/out/clustering.d.ts.map +1 -1
- package/out/clustering.js +31 -19
- package/out/clustering.js.map +1 -1
- package/out/comparators.d.ts +11 -33
- package/out/comparators.d.ts.map +1 -1
- package/out/comparators.js +11 -34
- package/out/comparators.js.map +1 -1
- package/out/distance.d.ts +18 -43
- package/out/distance.d.ts.map +1 -1
- package/out/distance.js +16 -41
- package/out/distance.js.map +1 -1
- package/out/em.d.ts +14 -6
- package/out/em.d.ts.map +1 -1
- package/out/em.js +5 -3
- package/out/em.js.map +1 -1
- package/out/fellegi-sunter.d.ts +39 -22
- package/out/fellegi-sunter.d.ts.map +1 -1
- package/out/fellegi-sunter.js +17 -12
- package/out/fellegi-sunter.js.map +1 -1
- package/out/gbt.d.ts +7 -17
- package/out/gbt.d.ts.map +1 -1
- package/out/gbt.js +7 -31
- package/out/gbt.js.map +1 -1
- package/out/tf.d.ts +24 -13
- package/out/tf.d.ts.map +1 -1
- package/out/tf.js +17 -11
- package/out/tf.js.map +1 -1
- package/package.json +13 -88
package/README.md
CHANGED
|
@@ -36,7 +36,7 @@ candidate pairs via cheap, high-recall keys:
|
|
|
36
36
|
|
|
37
37
|
The `scorePair` function computes a match probability using:
|
|
38
38
|
|
|
39
|
-
- **
|
|
39
|
+
- **Text comparators** — Jaro-Winkler similarity over names and addresses
|
|
40
40
|
- **Distance comparison** — great-circle distance bucketed into same-building /
|
|
41
41
|
same-block / same-area / far
|
|
42
42
|
- **Fellegi-Sunter weight model** — agreement-level log-likelihood ratios
|
|
@@ -92,4 +92,4 @@ withTermFrequency(model: FSModel, records: SourceRecord[]): FSModel
|
|
|
92
92
|
|
|
93
93
|
## License
|
|
94
94
|
|
|
95
|
-
[AGPL-3.0-only](https://www.gnu.org/licenses/
|
|
95
|
+
[AGPL-3.0-only](https://www.gnu.org/licenses/AGPL-3.0.html)
|
package/lib/blocking.ts
CHANGED
|
@@ -3,46 +3,46 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
* comparisons
|
|
6
|
+
* The blocking stage generates candidate pairs. An all-pairs comparison is O(n²): a million records require a trillion
|
|
7
|
+
* comparisons. This stage scores pairs that share a cheap key. This is where the geocode-first
|
|
8
8
|
* bet pays off: two records resolving to the same place land in the same spatial cell regardless
|
|
9
9
|
* of how their address strings are spelled, so geography is the primary block.
|
|
10
10
|
*
|
|
11
|
-
* A {@link BlockingKey} maps a record to zero or more string keys
|
|
11
|
+
* A {@link BlockingKey} maps a record to zero or more string keys. records sharing any key become
|
|
12
12
|
* candidates. Keys compose as a _union_ (the standard multi-pass approach — high recall from
|
|
13
|
-
* cheap rules): block on the spatial cell
|
|
14
|
-
* any rule
|
|
15
|
-
*
|
|
13
|
+
* cheap rules): block on the spatial cell, canonical key, or postcode. A pair caught by
|
|
14
|
+
* any rule is scored. {@link conjunction} builds the and-style key Geo-ER uses
|
|
15
|
+
* (`name-cell and geo-cell`) when a single rule is too loose.
|
|
16
16
|
*
|
|
17
|
-
* Recall is the priority
|
|
18
|
-
* silent failure in record linkage.
|
|
19
|
-
* default
|
|
17
|
+
* Recall is the priority. A pair the blocker never proposes can never match, the most dangerous
|
|
18
|
+
* silent failure in record linkage. The spatial grid is generous and neighbor-expanded by
|
|
19
|
+
* default. The code reports any block too large to scan instead of dropping it silently.
|
|
20
20
|
*/
|
|
21
21
|
|
|
22
|
-
|
|
23
|
-
* Maps a record to zero or more block keys. Two records sharing any key become a candidate pair.
|
|
24
|
-
*/
|
|
25
|
-
export type BlockingKey<R> = (record: R) => string[]
|
|
22
|
+
import type { GeoCoordinate } from "@mailwoman/spatial"
|
|
26
23
|
|
|
27
24
|
/**
|
|
28
|
-
*
|
|
25
|
+
* Maps a record to zero or more block keys.
|
|
26
|
+
*
|
|
27
|
+
* Two records sharing any key become a candidate pair.
|
|
29
28
|
*/
|
|
30
|
-
export
|
|
31
|
-
latitude: number
|
|
32
|
-
longitude: number
|
|
33
|
-
}
|
|
29
|
+
export type BlockingKey<R> = (record: R) => string[]
|
|
34
30
|
|
|
35
31
|
/**
|
|
36
|
-
* A spatial-cell block key: a configurable lat/lon grid.
|
|
37
|
-
*
|
|
38
|
-
*
|
|
32
|
+
* A spatial-cell block key: a configurable lat/lon grid.
|
|
33
|
+
*
|
|
34
|
+
* `precisionDegrees` sets the cell size (default 0.05° ≈ 5.5 km of latitude —
|
|
35
|
+
* deliberately generous, per the literature, so same-place records reliably co-block).
|
|
36
|
+
* With `neighbors` (default `true`) a record also keys its 8 adjacent cells,
|
|
37
|
+
* so a pair straddling a cell boundary still meets.
|
|
39
38
|
*
|
|
40
|
-
* Note: an equal-_degree_ grid (longitude cells shrink toward the poles)
|
|
41
|
-
* ~9×; an equal-area H3/geohash index
|
|
42
|
-
*
|
|
39
|
+
* Note: an equal-_degree_ grid (longitude cells shrink toward the poles)
|
|
40
|
+
* and neighbor expansion inflates block sizes ~9×; an equal-area H3/geohash index
|
|
41
|
+
* with a single-cell + neighbor-query is the refinement.
|
|
42
|
+
* Behavior — proximity co-blocking — is the same.
|
|
43
43
|
*/
|
|
44
44
|
export function geoCellKey<R>(
|
|
45
|
-
extract: (record: R) =>
|
|
45
|
+
extract: (record: R) => GeoCoordinate | null | undefined,
|
|
46
46
|
opts: { precisionDegrees?: number; neighbors?: boolean } = {}
|
|
47
47
|
): BlockingKey<R> {
|
|
48
48
|
const step = opts.precisionDegrees ?? 0.05
|
|
@@ -71,9 +71,10 @@ export function geoCellKey<R>(
|
|
|
71
71
|
}
|
|
72
72
|
|
|
73
73
|
/**
|
|
74
|
-
* An exact-value block key (the canonical address key, a postcode, an email domain…), normalized
|
|
75
|
-
* truncated to a leading `prefix` of characters (a cheaper, higher-recall rule).
|
|
76
|
-
*
|
|
74
|
+
* An exact-value block key (the canonical address key, a postcode, an email domain…), normalized
|
|
75
|
+
* and optionally truncated to a leading `prefix` of characters (a cheaper, higher-recall rule).
|
|
76
|
+
*
|
|
77
|
+
* A missing or empty value produces no key.
|
|
77
78
|
*/
|
|
78
79
|
export function exactKey<R>(
|
|
79
80
|
extract: (record: R) => string | null | undefined,
|
|
@@ -94,9 +95,12 @@ export function exactKey<R>(
|
|
|
94
95
|
}
|
|
95
96
|
|
|
96
97
|
/**
|
|
97
|
-
* A conjunctive block key — the cross-product of its sub-keys, joined (Geo-ER's "name
|
|
98
|
-
*
|
|
99
|
-
*
|
|
98
|
+
* A conjunctive block key — the cross-product of its sub-keys, joined (Geo-ER's "name and distance").
|
|
99
|
+
*
|
|
100
|
+
* A record is keyed by every combination of one sub-key from each input,
|
|
101
|
+
* so two records co-block only when they agree on _all_ inputs.
|
|
102
|
+
* Tighter blocks, lower recall.
|
|
103
|
+
* Use when a single rule is too loose.
|
|
100
104
|
*/
|
|
101
105
|
export function conjunction<R>(...keys: BlockingKey<R>[]): BlockingKey<R> {
|
|
102
106
|
return (record) => {
|
|
@@ -118,7 +122,7 @@ export function conjunction<R>(...keys: BlockingKey<R>[]): BlockingKey<R> {
|
|
|
118
122
|
*/
|
|
119
123
|
export interface BlockResult<R> {
|
|
120
124
|
/**
|
|
121
|
-
* Deduplicated candidate pairs (no self-pairs
|
|
125
|
+
* Deduplicated candidate pairs (no self-pairs. A pair caught by multiple keys appears once).
|
|
122
126
|
*/
|
|
123
127
|
pairs: Array<[R, R]>
|
|
124
128
|
/**
|
|
@@ -128,10 +132,11 @@ export interface BlockResult<R> {
|
|
|
128
132
|
}
|
|
129
133
|
|
|
130
134
|
/**
|
|
131
|
-
* Generate candidate pairs from `records` via one or more blocking keys (their union).
|
|
132
|
-
*
|
|
133
|
-
*
|
|
134
|
-
*
|
|
135
|
+
* Generate candidate pairs from `records` via one or more blocking keys (their union).
|
|
136
|
+
*
|
|
137
|
+
* Builds an inverted index (key → records) and emits the unique within-block pairs.
|
|
138
|
+
* A block larger than `maxBlockSize` is skipped and reported in `droppedBlocks` rather than blowing
|
|
139
|
+
* up into a quadratic scan — an explicit, visible coverage limit rather than a silent drop.
|
|
135
140
|
*/
|
|
136
141
|
export function block<R>(
|
|
137
142
|
records: readonly R[],
|
package/lib/clustering.ts
CHANGED
|
@@ -3,10 +3,10 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
6
|
+
* The clustering stage resolves scored pairs into canonical entities.
|
|
7
7
|
*
|
|
8
|
-
* The pairwise scorer treats each pair independently
|
|
9
|
-
* high weight and B~C at a high weight
|
|
8
|
+
* The pairwise scorer treats each pair independently. Its scores are not transitive: A~B at a
|
|
9
|
+
* high weight and B~C at a high weight do not guarantee that A~C is a match. A separate stage is
|
|
10
10
|
* required to turn the graph of above-threshold links into coherent groups — skip it and your
|
|
11
11
|
* "entities" silently fracture or fuse.
|
|
12
12
|
*
|
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
* for more recall. Its known weakness is over-merging via transitive chains (a string of weak
|
|
16
16
|
* links can pull unrelated records into one component); the principled fix is
|
|
17
17
|
* centroid-/average-linkage hierarchical clustering (Dedupe), which uses the full within-cluster
|
|
18
|
-
* score matrix — a documented refinement
|
|
18
|
+
* score matrix — a documented refinement rather than this first version. For a geocode-first matcher the
|
|
19
19
|
* over-merge risk is already damped: blocking keeps candidate sets local, so chains can't run
|
|
20
20
|
* across the whole dataset.
|
|
21
21
|
*/
|
|
@@ -40,25 +40,32 @@ export interface ClusterOptions {
|
|
|
40
40
|
/**
|
|
41
41
|
* How the above-threshold link graph resolves into clusters:
|
|
42
42
|
*
|
|
43
|
-
* - `"single"` (default) — connected components (union-find).
|
|
44
|
-
*
|
|
45
|
-
* -
|
|
46
|
-
*
|
|
47
|
-
*
|
|
48
|
-
*
|
|
43
|
+
* - `"single"` (default) — connected components (union-find).
|
|
44
|
+
* Fast.
|
|
45
|
+
* Any above-threshold link fuses two groups, so a single weak link can over-merge
|
|
46
|
+
* unrelated records through a transitive chain.
|
|
47
|
+
* - `"average"` — agglomerative average-linkage refinement within each connected component:
|
|
48
|
+
* two sub-clusters merge only when the average weight of the links between them clears
|
|
49
|
+
* the threshold, so a lone weak bridge no longer fuses two otherwise-dense groups.
|
|
50
|
+
* The documented over-merge fix (Dedupe).
|
|
51
|
+
* Falls back to single-linkage for any component larger than {@link maxAverageLinkageComponent}.
|
|
49
52
|
*/
|
|
50
53
|
linkage?: "single" | "average"
|
|
51
54
|
/**
|
|
52
|
-
* Components larger than this skip the O(k³) average-linkage refine and keep single-linkage.
|
|
55
|
+
* Components larger than this skip the O(k³) average-linkage refine and keep single-linkage.
|
|
56
|
+
*
|
|
57
|
+
* Default 64.
|
|
53
58
|
*/
|
|
54
59
|
maxAverageLinkageComponent?: number
|
|
55
60
|
}
|
|
56
61
|
|
|
57
62
|
/**
|
|
58
|
-
* Refine one connected component by agglomerative average-linkage.
|
|
59
|
-
*
|
|
60
|
-
*
|
|
61
|
-
*
|
|
63
|
+
* Refine one connected component by agglomerative average-linkage.
|
|
64
|
+
*
|
|
65
|
+
* Starts with every member a singleton and repeatedly merges the cluster pair
|
|
66
|
+
* with the highest _average_ inter-cluster link weight while that average is at
|
|
67
|
+
* or above `threshold`; clusters with no link between them never merge.
|
|
68
|
+
* O(k³) in the component size, so callers boundate it on a size cap.
|
|
62
69
|
*/
|
|
63
70
|
function averageLinkageRefine<R>(members: R[], edges: Array<[number, number, number]>, threshold: number): R[][] {
|
|
64
71
|
const clusters = members.map((_, i) => [i])
|
|
@@ -105,15 +112,20 @@ function averageLinkageRefine<R>(members: R[], edges: Array<[number, number, num
|
|
|
105
112
|
}
|
|
106
113
|
|
|
107
114
|
/**
|
|
108
|
-
* Cluster records into canonical entities by connected components of the above-threshold link graph.
|
|
109
|
-
*
|
|
110
|
-
*
|
|
115
|
+
* Cluster records into canonical entities by connected components of the above-threshold link graph.
|
|
116
|
+
*
|
|
117
|
+
* Every input record lands in exactly one cluster.
|
|
118
|
+
* A record with no qualifying link is a singleton.
|
|
119
|
+
* Links referencing a record not in `records` are ignored.
|
|
120
|
+
*
|
|
121
|
+
* Reference identity is used, so pass the same record objects to both arguments.
|
|
111
122
|
*/
|
|
112
123
|
export function cluster<R>(records: readonly R[], links: Iterable<ScoredLink<R>>, opts: ClusterOptions): R[][] {
|
|
113
124
|
const index = new Map<R, number>()
|
|
114
125
|
records.forEach((record, i) => index.set(record, i))
|
|
115
126
|
|
|
116
|
-
// Local by design: the shared union-find is `@mailwoman/core/utils/union-find`;
|
|
127
|
+
// Local by design: the shared union-find is `@mailwoman/core/utils/union-find`;
|
|
128
|
+
// match takes no core dependency (~11 MB).
|
|
117
129
|
const parent = records.map((_, i) => i)
|
|
118
130
|
const rank = new Array<number>(records.length).fill(0)
|
|
119
131
|
|
|
@@ -151,9 +163,10 @@ export function cluster<R>(records: readonly R[], links: Iterable<ScoredLink<R>>
|
|
|
151
163
|
}
|
|
152
164
|
}
|
|
153
165
|
|
|
154
|
-
// Collect
|
|
155
|
-
// above-threshold ones, but the average-linkage refinement needs the full sub-graph
|
|
156
|
-
// disagreeing below-threshold edge between two sub-clusters is
|
|
166
|
+
// Collect all valid links (not just above-threshold): connected components form from the
|
|
167
|
+
// above-threshold ones, but the average-linkage refinement needs the full sub-graph.
|
|
168
|
+
// A weak or disagreeing below-threshold edge between two sub-clusters is
|
|
169
|
+
// exactly what should pull them apart.
|
|
157
170
|
const allLinks: ScoredLink<R>[] = []
|
|
158
171
|
|
|
159
172
|
for (const link of links) {
|
|
@@ -184,7 +197,7 @@ export function cluster<R>(records: readonly R[], links: Iterable<ScoredLink<R>>
|
|
|
184
197
|
if (opts.linkage !== "average") return [...groups.values()]
|
|
185
198
|
|
|
186
199
|
// Average-linkage refinement: split each component where its sub-clusters are joined only by a weak
|
|
187
|
-
// bridge (the average inter-cluster link weight, over
|
|
200
|
+
// bridge (the average inter-cluster link weight, over all edges between them, falls below the threshold).
|
|
188
201
|
const maxComponent = opts.maxAverageLinkageComponent ?? 64
|
|
189
202
|
const localOf = new Map<R, number>()
|
|
190
203
|
|
|
@@ -222,9 +235,13 @@ export function cluster<R>(records: readonly R[], links: Iterable<ScoredLink<R>>
|
|
|
222
235
|
}
|
|
223
236
|
|
|
224
237
|
/**
|
|
225
|
-
* Pick a cluster's most complete record as its canonical representative —
|
|
226
|
-
* / `undefined` / `""`).
|
|
227
|
-
*
|
|
238
|
+
* Pick a cluster's most complete record as its canonical representative —
|
|
239
|
+
* the one with the fewest empty fields (`null` / `undefined` / `""`).
|
|
240
|
+
*
|
|
241
|
+
* Ties keep the earliest.
|
|
242
|
+
* A basic, generic canonicalizer.
|
|
243
|
+
*
|
|
244
|
+
* Field-level merging across the cluster is the application's job (it knows which source to trust).
|
|
228
245
|
*/
|
|
229
246
|
export function representative<R extends object>(group: readonly R[]): R | undefined {
|
|
230
247
|
let best: R | undefined
|
package/lib/comparators.ts
CHANGED
|
@@ -3,25 +3,14 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
* The record-linkage literature (Winkler/Census; Belin 1993) settles on the prefix-weighted Jaro
|
|
9
|
-
* comparator (Jaro-Winkler) as the default for names: it tolerates the typographical error real
|
|
10
|
-
* data is full of better than raw character-edit distance. But J-W has a documented blind spot on
|
|
11
|
-
* compound / double surnames (e.g. Hispanic `Garcia Lopez`): the second half of the compound
|
|
12
|
-
* falls outside J-W's match window, so `Lopez` vs `Garcia Lopez` scores ~0. The fix the
|
|
13
|
-
* literature prescribes is an edit-distance / token fallback for single-vs-compound pairs —
|
|
14
|
-
* implemented in {@link nameSimilarity}.
|
|
15
|
-
*
|
|
16
|
-
* These are pure similarity primitives in [0, 1]. The mapping of a similarity onto discrete
|
|
17
|
-
* Fellegi-Sunter agreement levels (and the m/u weights) is the scorer's job, not theirs.
|
|
6
|
+
* Jaro-Winkler is the default record-linkage comparator for names. {@link nameSimilarity} adds
|
|
7
|
+
* the literature's token/edit fallback because J-W scores a compound surname's second half near zero.
|
|
18
8
|
*/
|
|
19
9
|
|
|
20
10
|
import { distance as levenshteinDistance } from "fastest-levenshtein"
|
|
21
11
|
|
|
22
12
|
/**
|
|
23
|
-
* Jaro similarity in [0, 1]
|
|
24
|
-
* sliding window of `floor(max(len)/2) - 1`, discounting half-transpositions.
|
|
13
|
+
* Jaro similarity in [0, 1], where two empty strings are identical (1) and one empty is 0.
|
|
25
14
|
*/
|
|
26
15
|
export function jaro(a: string, b: string): number {
|
|
27
16
|
if (a === b) return 1
|
|
@@ -53,7 +42,6 @@ export function jaro(a: string, b: string): number {
|
|
|
53
42
|
|
|
54
43
|
if (matches === 0) return 0
|
|
55
44
|
|
|
56
|
-
// Count transpositions: matched chars of `a` and `b`, in order, that disagree (halved).
|
|
57
45
|
let transpositions = 0
|
|
58
46
|
let k = 0
|
|
59
47
|
|
|
@@ -77,9 +65,8 @@ export function jaro(a: string, b: string): number {
|
|
|
77
65
|
}
|
|
78
66
|
|
|
79
67
|
/**
|
|
80
|
-
* Jaro-Winkler similarity in [0, 1]: Jaro
|
|
81
|
-
*
|
|
82
|
-
* when `jaro` already clears `boostThreshold` (0.7), per Winkler.
|
|
68
|
+
* Jaro-Winkler similarity in [0, 1]: Jaro plus a shared-prefix bonus, boosting only when Jaro
|
|
69
|
+
* already clears the 0.7 threshold, with Winkler's standard prefix cap 4 and weight 0.1.
|
|
83
70
|
*/
|
|
84
71
|
export function jaroWinkler(
|
|
85
72
|
a: string,
|
|
@@ -105,14 +92,8 @@ export function jaroWinkler(
|
|
|
105
92
|
}
|
|
106
93
|
|
|
107
94
|
/**
|
|
108
|
-
* Jaccard similarity
|
|
109
|
-
*
|
|
110
|
-
* The set-of-tokens complement to the string comparators above. Where {@link nameSimilarity} asks how close two names
|
|
111
|
-
* LOOK, this asks how much two token bags OVERLAP — the right question for organization names and address bags, where
|
|
112
|
-
* word order carries no information and a shared rare token is worth more than character-level proximity.
|
|
113
|
-
*
|
|
114
|
-
* Either side empty scores 0 rather than 1: an empty bag agrees with nothing, and treating "no evidence" as "perfect
|
|
115
|
-
* agreement" is how a blocking pass floods with false pairs.
|
|
95
|
+
* Jaccard similarity `|a ∩ b| / |a ∪ b|` over two token sets, where an empty side scores 0 rather than 1
|
|
96
|
+
* because treating no evidence as perfect agreement floods a blocking pass with false pairs.
|
|
116
97
|
*/
|
|
117
98
|
export function jaccard(a: ReadonlySet<string>, b: ReadonlySet<string>): number {
|
|
118
99
|
if (!a.size || !b.size) return 0
|
|
@@ -140,15 +121,11 @@ export function levenshteinSimilarity(a: string, b: string): number {
|
|
|
140
121
|
}
|
|
141
122
|
|
|
142
123
|
/**
|
|
143
|
-
* Name-aware similarity in [0, 1]
|
|
144
|
-
*
|
|
145
|
-
*
|
|
146
|
-
* - If one name's tokens are a strict subset of the other's (`Lopez` ⊂ `Garcia Lopez`), that is strong partial agreement
|
|
147
|
-
* J-W misses — floor the score at 0.9.
|
|
148
|
-
* - Otherwise return the better of Jaro-Winkler and normalized edit similarity, so a single token that is a substring of
|
|
149
|
-
* a longer compound (`Garcia` vs `Garcialopez`) still scores sensibly.
|
|
124
|
+
* Name-aware similarity in [0, 1] that floors the score at 0.9 when one name's
|
|
125
|
+
* tokens are a strict subset of the other's.
|
|
150
126
|
*
|
|
151
|
-
*
|
|
127
|
+
* In other cases, it returns the better of Jaro-Winkler and normalized edit
|
|
128
|
+
* similarity without case or whitespace sensitivity.
|
|
152
129
|
*/
|
|
153
130
|
export function nameSimilarity(a: string, b: string): number {
|
|
154
131
|
const x = a.trim().toLowerCase().replaceAll(/\s+/g, " ")
|
package/lib/distance.ts
CHANGED
|
@@ -2,46 +2,31 @@
|
|
|
2
2
|
* @copyright Sister Software
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
|
-
*
|
|
6
|
-
* Geographic distance as a scoring feature — the other half of geocode-first matching.
|
|
7
|
-
*
|
|
8
|
-
* Blocking uses geography to _propose_ candidates; this scores them on it. The research is explicit
|
|
9
|
-
* that an address must be matched as a SPATIAL attribute, not by string similarity (a
|
|
10
|
-
* one-character edit can be 650 m apart), and that distance measurably helps as a comparison
|
|
11
|
-
* feature. So we bucket the great-circle distance between two records' coordinates into ordered
|
|
12
|
-
* Fellegi-Sunter agreement levels (Splink's `DistanceInKMAtThresholds`): "same building" / "same
|
|
13
|
-
* block" / "same area" / far, each with its own m/u and weight.
|
|
14
|
-
*
|
|
15
|
-
* Calibrate the bucket boundaries to the geocoder's OWN error, which is heavy-tailed and density-
|
|
16
|
-
* dependent (≈38 m urban, ≈200 m rural). A weakening of this evidence by geocode quality (a
|
|
17
|
-
* shared interpolated centroid is softer than a shared rooftop point) is the documented
|
|
18
|
-
* refinement.
|
|
19
5
|
*/
|
|
20
6
|
|
|
21
|
-
import { haversineKm as greatCircleKm } from "@mailwoman/spatial"
|
|
7
|
+
import { haversineKm as greatCircleKm, type GeoCoordinate } from "@mailwoman/spatial"
|
|
22
8
|
|
|
23
|
-
import type { LatLon } from "#blocking"
|
|
24
9
|
import type { Comparison, ComparisonLevel } from "#fellegi-sunter"
|
|
25
10
|
|
|
26
11
|
/**
|
|
27
|
-
*
|
|
28
|
-
*
|
|
29
|
-
* helper — not a second implementation.
|
|
12
|
+
* Computes the great-circle distance in kilometers between two `GeoCoordinate` records.
|
|
13
|
+
* It delegates to the scalar helper in `@mailwoman/spatial`.
|
|
30
14
|
*/
|
|
31
|
-
export const haversineKm = (a:
|
|
15
|
+
export const haversineKm = (a: GeoCoordinate, b: GeoCoordinate): number =>
|
|
32
16
|
greatCircleKm(a.latitude, a.longitude, b.latitude, b.longitude)
|
|
33
17
|
|
|
34
18
|
/**
|
|
35
|
-
*
|
|
36
|
-
* levels
|
|
37
|
-
*
|
|
19
|
+
* Creates a comparison that buckets the great-circle distance between two records into
|
|
20
|
+
* levels ordered nearest first by `maxKm`, with the last level catching everything farther.
|
|
21
|
+
*
|
|
22
|
+
* A missing or non-finite coordinate on either side yields no evidence.
|
|
38
23
|
*/
|
|
39
24
|
export function distanceComparison<R>(config: {
|
|
40
25
|
name: string
|
|
41
|
-
extract: (record: R) =>
|
|
26
|
+
extract: (record: R) => GeoCoordinate | null | undefined
|
|
42
27
|
levels: ComparisonLevel[]
|
|
43
28
|
}): Comparison<R> {
|
|
44
|
-
const valid = (c:
|
|
29
|
+
const valid = (c: GeoCoordinate | null | undefined): c is GeoCoordinate =>
|
|
45
30
|
!!c && Number.isFinite(c.latitude) && Number.isFinite(c.longitude)
|
|
46
31
|
|
|
47
32
|
return {
|
|
@@ -65,8 +50,8 @@ export function distanceComparison<R>(config: {
|
|
|
65
50
|
}
|
|
66
51
|
|
|
67
52
|
/**
|
|
68
|
-
*
|
|
69
|
-
*
|
|
53
|
+
* Provides default distance levels at building, block and area scales.
|
|
54
|
+
* Their `m` and `u` values seed EM re-estimation.
|
|
70
55
|
*/
|
|
71
56
|
export const DEFAULT_DISTANCE_LEVELS: ComparisonLevel[] = [
|
|
72
57
|
{ label: "same-building", maxKm: 0.05, m: 0.7, u: 0.001 },
|
|
@@ -76,30 +61,19 @@ export const DEFAULT_DISTANCE_LEVELS: ComparisonLevel[] = [
|
|
|
76
61
|
]
|
|
77
62
|
|
|
78
63
|
/**
|
|
79
|
-
*
|
|
64
|
+
* Creates a single spatial comparison whose level 0 is an exact canonical-key match
|
|
65
|
+
* and whose remaining levels bucket great-circle distance for pairs with different keys.
|
|
80
66
|
*
|
|
81
|
-
*
|
|
82
|
-
*
|
|
83
|
-
* redundant vote is exactly what over-merges distinct providers at a shared clinic address. This folds them into one
|
|
84
|
-
* comparison:
|
|
85
|
-
*
|
|
86
|
-
* - **level 0 `same-key`** — an EXACT canonical-key match: the strongest tier, and the one the inverse-address-frequency
|
|
87
|
-
* adjustment rides ({@link withTermFrequency} on level 0), so agreement on a crowded shared key is down-weighted
|
|
88
|
-
* toward worthless while a rare one keeps full weight.
|
|
89
|
-
* - **levels 1…n** — great-circle distance buckets for pairs whose keys DIFFER, so "123 Main St" vs "123 Main Street Apt
|
|
90
|
-
* 2" that geocode to the same rooftop still earns near-agreement (the geo-first point of the whole design).
|
|
91
|
-
* - Keys differ and no usable coordinate → no evidence.
|
|
92
|
-
*
|
|
93
|
-
* Exactly one spatial vote, no redundancy. Pass {@link DEFAULT_SPATIAL_LEVELS} or your own; index 0 must be the
|
|
94
|
-
* exact-key tier, indices 1…n the distance buckets nearest → far by `maxKm` (last = `far`).
|
|
67
|
+
* Separate key and distance comparisons count a co-located pair's evidence twice.
|
|
68
|
+
* They also over-merge distinct entities at a shared address.
|
|
95
69
|
*/
|
|
96
70
|
export function spatialComparison<R>(config: {
|
|
97
71
|
name: string
|
|
98
72
|
key: (record: R) => string | null | undefined
|
|
99
|
-
coordinate: (record: R) =>
|
|
73
|
+
coordinate: (record: R) => GeoCoordinate | null | undefined
|
|
100
74
|
levels: ComparisonLevel[]
|
|
101
75
|
}): Comparison<R> {
|
|
102
|
-
const valid = (c:
|
|
76
|
+
const valid = (c: GeoCoordinate | null | undefined): c is GeoCoordinate =>
|
|
103
77
|
!!c && Number.isFinite(c.latitude) && Number.isFinite(c.longitude)
|
|
104
78
|
|
|
105
79
|
return {
|
|
@@ -109,12 +83,12 @@ export function spatialComparison<R>(config: {
|
|
|
109
83
|
const ka = config.key(a)
|
|
110
84
|
const kb = config.key(b)
|
|
111
85
|
|
|
112
|
-
if (ka && kb && ka.trim() && ka === kb) return 0
|
|
86
|
+
if (ka && kb && ka.trim() && ka === kb) return 0
|
|
113
87
|
|
|
114
88
|
const ca = config.coordinate(a)
|
|
115
89
|
const cb = config.coordinate(b)
|
|
116
90
|
|
|
117
|
-
if (!valid(ca) || !valid(cb)) return -1
|
|
91
|
+
if (!valid(ca) || !valid(cb)) return -1
|
|
118
92
|
|
|
119
93
|
const km = haversineKm(ca, cb)
|
|
120
94
|
|
|
@@ -128,8 +102,8 @@ export function spatialComparison<R>(config: {
|
|
|
128
102
|
}
|
|
129
103
|
|
|
130
104
|
/**
|
|
131
|
-
*
|
|
132
|
-
*
|
|
105
|
+
* Provides default levels for {@link spatialComparison}: an exact same-key tier followed
|
|
106
|
+
* by the building, block, area and far distance buckets, with seed `m` and `u` values.
|
|
133
107
|
*/
|
|
134
108
|
export const DEFAULT_SPATIAL_LEVELS: ComparisonLevel[] = [
|
|
135
109
|
{ label: "same-key", m: 0.85, u: 0.01 },
|
package/lib/em.ts
CHANGED
|
@@ -41,15 +41,21 @@ export function agreementPattern<R>(comparisons: Comparison<R>[], a: R, b: R): n
|
|
|
41
41
|
*/
|
|
42
42
|
export interface EmOptions {
|
|
43
43
|
/**
|
|
44
|
-
* Hard iteration cap.
|
|
44
|
+
* Hard iteration cap.
|
|
45
|
+
*
|
|
46
|
+
* Default 100.
|
|
45
47
|
*/
|
|
46
48
|
maxIterations?: number
|
|
47
49
|
/**
|
|
48
|
-
* Convergence tolerance on the largest parameter change between iterations.
|
|
50
|
+
* Convergence tolerance on the largest parameter change between iterations.
|
|
51
|
+
*
|
|
52
|
+
* Default 1e-6.
|
|
49
53
|
*/
|
|
50
54
|
tolerance?: number
|
|
51
55
|
/**
|
|
52
|
-
*
|
|
56
|
+
* Prior match rate used to initialize the model.
|
|
57
|
+
*
|
|
58
|
+
* Defaults to the model's `lambda`.
|
|
53
59
|
*/
|
|
54
60
|
initialLambda?: number
|
|
55
61
|
}
|
|
@@ -71,9 +77,11 @@ export interface EmResult<R> {
|
|
|
71
77
|
}
|
|
72
78
|
|
|
73
79
|
/**
|
|
74
|
-
* Estimate `m`/`u` and the prior `λ` from unlabeled agreement patterns via EM.
|
|
75
|
-
*
|
|
76
|
-
*
|
|
80
|
+
* Estimate `m`/`u` and the prior `λ` from unlabeled agreement patterns via EM.
|
|
81
|
+
*
|
|
82
|
+
* The patterns are per-comparison level indices (as produced by {@link agreementPattern});
|
|
83
|
+
* a `-1` (missing) field contributes no evidence to either class.
|
|
84
|
+
* The model's existing level `m`/`u` seed the iteration.
|
|
77
85
|
*/
|
|
78
86
|
export function estimateParameters<R>(
|
|
79
87
|
model: FellegiSunterModel<R>,
|