@mailwoman/match 9.4.0 → 10.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -36,7 +36,7 @@ candidate pairs via cheap, high-recall keys:
36
36
 
37
37
  The `scorePair` function computes a match probability using:
38
38
 
39
- - **String comparators** — Jaro-Winkler similarity over names and addresses
39
+ - **Text comparators** — Jaro-Winkler similarity over names and addresses
40
40
  - **Distance comparison** — great-circle distance bucketed into same-building /
41
41
  same-block / same-area / far
42
42
  - **Fellegi-Sunter weight model** — agreement-level log-likelihood ratios
@@ -92,4 +92,4 @@ withTermFrequency(model: FSModel, records: SourceRecord[]): FSModel
92
92
 
93
93
  ## License
94
94
 
95
- [AGPL-3.0-only](https://www.gnu.org/licenses/agpl-3.0.html)
95
+ [AGPL-3.0-only](https://www.gnu.org/licenses/AGPL-3.0.html)
package/lib/blocking.ts CHANGED
@@ -3,46 +3,46 @@
3
3
  * @license AGPL-3.0
4
4
  * @author Teffen Ellis, et al.
5
5
  *
6
- * Blocking — candidate generation. Comparing every pair is O(n²) (a million records is a trillion
7
- * comparisons), so we only score pairs that share a cheap key. This is where the geocode-first
6
+ * The blocking stage generates candidate pairs. An all-pairs comparison is O(n²): a million records require a trillion
7
+ * comparisons. This stage scores pairs that share a cheap key. This is where the geocode-first
8
8
  * bet pays off: two records resolving to the same place land in the same spatial cell regardless
9
9
  * of how their address strings are spelled, so geography is the primary block.
10
10
  *
11
- * A {@link BlockingKey} maps a record to zero or more string keys; records sharing any key become
11
+ * A {@link BlockingKey} maps a record to zero or more string keys. records sharing any key become
12
12
  * candidates. Keys compose as a _union_ (the standard multi-pass approach — high recall from
13
- * cheap rules): block on the spatial cell OR the canonical key OR the postcode, and a pair that
14
- * any rule catches is scored. {@link conjunction} builds the AND-style key Geo-ER uses (`name-cell
15
- * AND geo-cell`) when a single rule is too loose.
13
+ * cheap rules): block on the spatial cell, canonical key, or postcode. A pair caught by
14
+ * any rule is scored. {@link conjunction} builds the and-style key Geo-ER uses
15
+ * (`name-cell and geo-cell`) when a single rule is too loose.
16
16
  *
17
- * Recall is the priority — a pair the blocker never proposes can never match, the most dangerous
18
- * silent failure in record linkage. So the spatial grid is generous and neighbour-expanded by
19
- * default, and any block too large to scan is _reported_, never silently dropped.
17
+ * Recall is the priority. A pair the blocker never proposes can never match, the most dangerous
18
+ * silent failure in record linkage. The spatial grid is generous and neighbor-expanded by
19
+ * default. The code reports any block too large to scan instead of dropping it silently.
20
20
  */
21
21
 
22
- /**
23
- * Maps a record to zero or more block keys. Two records sharing any key become a candidate pair.
24
- */
25
- export type BlockingKey<R> = (record: R) => string[]
22
+ import type { GeoCoordinate } from "@mailwoman/spatial"
26
23
 
27
24
  /**
28
- * A geographic coordinate (WGS84 decimal degrees).
25
+ * Maps a record to zero or more block keys.
26
+ *
27
+ * Two records sharing any key become a candidate pair.
29
28
  */
30
- export interface LatLon {
31
- latitude: number
32
- longitude: number
33
- }
29
+ export type BlockingKey<R> = (record: R) => string[]
34
30
 
35
31
  /**
36
- * A spatial-cell block key: a configurable lat/lon grid. `precisionDegrees` sets the cell size (default 0.05° ≈ 5.5 km
37
- * of latitude — deliberately generous, per the literature, so same-place records reliably co-block). With `neighbors`
38
- * (default `true`) a record also keys its 8 adjacent cells, so a pair straddling a cell boundary still meets.
32
+ * A spatial-cell block key: a configurable lat/lon grid.
33
+ *
34
+ * `precisionDegrees` sets the cell size (default 0.05° ≈ 5.5 km of latitude —
35
+ * deliberately generous, per the literature, so same-place records reliably co-block).
36
+ * With `neighbors` (default `true`) a record also keys its 8 adjacent cells,
37
+ * so a pair straddling a cell boundary still meets.
39
38
  *
40
- * Note: an equal-_degree_ grid (longitude cells shrink toward the poles) and neighbour expansion inflates block sizes
41
- * ~9×; an equal-area H3/geohash index with a single-cell + neighbour-query is the refinement. Behaviour — proximity
42
- * co-blocking — is the same.
39
+ * Note: an equal-_degree_ grid (longitude cells shrink toward the poles)
40
+ * and neighbor expansion inflates block sizes ~9×; an equal-area H3/geohash index
41
+ * with a single-cell + neighbor-query is the refinement.
42
+ * Behavior — proximity co-blocking — is the same.
43
43
  */
44
44
  export function geoCellKey<R>(
45
- extract: (record: R) => LatLon | null | undefined,
45
+ extract: (record: R) => GeoCoordinate | null | undefined,
46
46
  opts: { precisionDegrees?: number; neighbors?: boolean } = {}
47
47
  ): BlockingKey<R> {
48
48
  const step = opts.precisionDegrees ?? 0.05
@@ -71,9 +71,10 @@ export function geoCellKey<R>(
71
71
  }
72
72
 
73
73
  /**
74
- * An exact-value block key (the canonical address key, a postcode, an email domain…), normalized and optionally
75
- * truncated to a leading `prefix` of characters (a cheaper, higher-recall rule). A missing or empty value produces no
76
- * key.
74
+ * An exact-value block key (the canonical address key, a postcode, an email domain…), normalized
75
+ * and optionally truncated to a leading `prefix` of characters (a cheaper, higher-recall rule).
76
+ *
77
+ * A missing or empty value produces no key.
77
78
  */
78
79
  export function exactKey<R>(
79
80
  extract: (record: R) => string | null | undefined,
@@ -94,9 +95,12 @@ export function exactKey<R>(
94
95
  }
95
96
 
96
97
  /**
97
- * A conjunctive block key — the cross-product of its sub-keys, joined (Geo-ER's "name AND distance"). A record is keyed
98
- * by every combination of one sub-key from each input, so two records co-block only when they agree on _all_ inputs.
99
- * Tighter blocks, lower recall — use when a single rule is too loose.
98
+ * A conjunctive block key — the cross-product of its sub-keys, joined (Geo-ER's "name and distance").
99
+ *
100
+ * A record is keyed by every combination of one sub-key from each input,
101
+ * so two records co-block only when they agree on _all_ inputs.
102
+ * Tighter blocks, lower recall.
103
+ * Use when a single rule is too loose.
100
104
  */
101
105
  export function conjunction<R>(...keys: BlockingKey<R>[]): BlockingKey<R> {
102
106
  return (record) => {
@@ -118,7 +122,7 @@ export function conjunction<R>(...keys: BlockingKey<R>[]): BlockingKey<R> {
118
122
  */
119
123
  export interface BlockResult<R> {
120
124
  /**
121
- * Deduplicated candidate pairs (no self-pairs; a pair caught by multiple keys appears once).
125
+ * Deduplicated candidate pairs (no self-pairs. A pair caught by multiple keys appears once).
122
126
  */
123
127
  pairs: Array<[R, R]>
124
128
  /**
@@ -128,10 +132,11 @@ export interface BlockResult<R> {
128
132
  }
129
133
 
130
134
  /**
131
- * Generate candidate pairs from `records` via one or more blocking keys (their union). Builds an inverted index (key →
132
- * records) and emits the unique within-block pairs. A block larger than `maxBlockSize` is skipped and reported in
133
- * `droppedBlocks` rather than blowing up into a quadratic scan — an explicit, visible coverage limit, not a silent
134
- * drop.
135
+ * Generate candidate pairs from `records` via one or more blocking keys (their union).
136
+ *
137
+ * Builds an inverted index (key → records) and emits the unique within-block pairs.
138
+ * A block larger than `maxBlockSize` is skipped and reported in `droppedBlocks` rather than blowing
139
+ * up into a quadratic scan — an explicit, visible coverage limit rather than a silent drop.
135
140
  */
136
141
  export function block<R>(
137
142
  records: readonly R[],
package/lib/clustering.ts CHANGED
@@ -3,10 +3,10 @@
3
3
  * @license AGPL-3.0
4
4
  * @author Teffen Ellis, et al.
5
5
  *
6
- * Clustering — the third and final matcher stage: resolve scored pairs into canonical entities.
6
+ * The clustering stage resolves scored pairs into canonical entities.
7
7
  *
8
- * The pairwise scorer treats each pair independently, and its scores are NOT transitive: A~B at a
9
- * high weight and B~C at a high weight does not guarantee A~C is a match. So a distinct stage is
8
+ * The pairwise scorer treats each pair independently. Its scores are not transitive: A~B at a
9
+ * high weight and B~C at a high weight do not guarantee that A~C is a match. A separate stage is
10
10
  * required to turn the graph of above-threshold links into coherent groups — skip it and your
11
11
  * "entities" silently fracture or fuse.
12
12
  *
@@ -15,7 +15,7 @@
15
15
  * for more recall. Its known weakness is over-merging via transitive chains (a string of weak
16
16
  * links can pull unrelated records into one component); the principled fix is
17
17
  * centroid-/average-linkage hierarchical clustering (Dedupe), which uses the full within-cluster
18
- * score matrix — a documented refinement, not this first version. For a geocode-first matcher the
18
+ * score matrix — a documented refinement rather than this first version. For a geocode-first matcher the
19
19
  * over-merge risk is already damped: blocking keeps candidate sets local, so chains can't run
20
20
  * across the whole dataset.
21
21
  */
@@ -40,25 +40,32 @@ export interface ClusterOptions {
40
40
  /**
41
41
  * How the above-threshold link graph resolves into clusters:
42
42
  *
43
- * - `"single"` (default) — connected components (union-find). Fast; ANY above-threshold link fuses two groups, so a
44
- * single weak link can over-merge unrelated records through a transitive chain.
45
- * - `"average"` — agglomerative average-linkage refinement WITHIN each connected component: two sub-clusters merge only
46
- * when the AVERAGE weight of the links between them clears the threshold, so a lone weak bridge no longer fuses two
47
- * otherwise-dense groups. The documented over-merge fix (Dedupe). Falls back to single-linkage for any component
48
- * larger than {@link maxAverageLinkageComponent}.
43
+ * - `"single"` (default) — connected components (union-find).
44
+ * Fast.
45
+ * Any above-threshold link fuses two groups, so a single weak link can over-merge
46
+ * unrelated records through a transitive chain.
47
+ * - `"average"` — agglomerative average-linkage refinement within each connected component:
48
+ * two sub-clusters merge only when the average weight of the links between them clears
49
+ * the threshold, so a lone weak bridge no longer fuses two otherwise-dense groups.
50
+ * The documented over-merge fix (Dedupe).
51
+ * Falls back to single-linkage for any component larger than {@link maxAverageLinkageComponent}.
49
52
  */
50
53
  linkage?: "single" | "average"
51
54
  /**
52
- * Components larger than this skip the O(k³) average-linkage refine and keep single-linkage. Default 64.
55
+ * Components larger than this skip the O(k³) average-linkage refine and keep single-linkage.
56
+ *
57
+ * Default 64.
53
58
  */
54
59
  maxAverageLinkageComponent?: number
55
60
  }
56
61
 
57
62
  /**
58
- * Refine one connected component by agglomerative average-linkage. Starts with every member a singleton and repeatedly
59
- * merges the cluster pair with the highest _average_ inter-cluster link weight while that average is at or above
60
- * `threshold`; clusters with no link between them never merge. O(k³) in the component size, so callers boundate it on a
61
- * size cap.
63
+ * Refine one connected component by agglomerative average-linkage.
64
+ *
65
+ * Starts with every member a singleton and repeatedly merges the cluster pair
66
+ * with the highest _average_ inter-cluster link weight while that average is at
67
+ * or above `threshold`; clusters with no link between them never merge.
68
+ * O(k³) in the component size, so callers boundate it on a size cap.
62
69
  */
63
70
  function averageLinkageRefine<R>(members: R[], edges: Array<[number, number, number]>, threshold: number): R[][] {
64
71
  const clusters = members.map((_, i) => [i])
@@ -105,15 +112,20 @@ function averageLinkageRefine<R>(members: R[], edges: Array<[number, number, num
105
112
  }
106
113
 
107
114
  /**
108
- * Cluster records into canonical entities by connected components of the above-threshold link graph. Every input record
109
- * lands in exactly one cluster — a record with no qualifying link is a singleton. Links referencing a record not in
110
- * `records` are ignored. Reference identity is used, so pass the same record objects to both arguments.
115
+ * Cluster records into canonical entities by connected components of the above-threshold link graph.
116
+ *
117
+ * Every input record lands in exactly one cluster.
118
+ * A record with no qualifying link is a singleton.
119
+ * Links referencing a record not in `records` are ignored.
120
+ *
121
+ * Reference identity is used, so pass the same record objects to both arguments.
111
122
  */
112
123
  export function cluster<R>(records: readonly R[], links: Iterable<ScoredLink<R>>, opts: ClusterOptions): R[][] {
113
124
  const index = new Map<R, number>()
114
125
  records.forEach((record, i) => index.set(record, i))
115
126
 
116
- // Local by design: the shared union-find is `@mailwoman/core/utils/union-find`; match takes no core dependency (~11 MB).
127
+ // Local by design: the shared union-find is `@mailwoman/core/utils/union-find`;
128
+ // match takes no core dependency (~11 MB).
117
129
  const parent = records.map((_, i) => i)
118
130
  const rank = new Array<number>(records.length).fill(0)
119
131
 
@@ -151,9 +163,10 @@ export function cluster<R>(records: readonly R[], links: Iterable<ScoredLink<R>>
151
163
  }
152
164
  }
153
165
 
154
- // Collect ALL valid links (not just above-threshold): connected components form from the
155
- // above-threshold ones, but the average-linkage refinement needs the full sub-graph — a weak or
156
- // disagreeing below-threshold edge between two sub-clusters is exactly what should pull them apart.
166
+ // Collect all valid links (not just above-threshold): connected components form from the
167
+ // above-threshold ones, but the average-linkage refinement needs the full sub-graph.
168
+ // A weak or disagreeing below-threshold edge between two sub-clusters is
169
+ // exactly what should pull them apart.
157
170
  const allLinks: ScoredLink<R>[] = []
158
171
 
159
172
  for (const link of links) {
@@ -184,7 +197,7 @@ export function cluster<R>(records: readonly R[], links: Iterable<ScoredLink<R>>
184
197
  if (opts.linkage !== "average") return [...groups.values()]
185
198
 
186
199
  // Average-linkage refinement: split each component where its sub-clusters are joined only by a weak
187
- // bridge (the average inter-cluster link weight, over ALL edges between them, falls below the threshold).
200
+ // bridge (the average inter-cluster link weight, over all edges between them, falls below the threshold).
188
201
  const maxComponent = opts.maxAverageLinkageComponent ?? 64
189
202
  const localOf = new Map<R, number>()
190
203
 
@@ -222,9 +235,13 @@ export function cluster<R>(records: readonly R[], links: Iterable<ScoredLink<R>>
222
235
  }
223
236
 
224
237
  /**
225
- * Pick a cluster's most complete record as its canonical representative — the one with the fewest empty fields (`null`
226
- * / `undefined` / `""`). Ties keep the earliest. A basic, generic canonicalizer; field-level merging across the cluster
227
- * is the application's job (it knows which source to trust).
238
+ * Pick a cluster's most complete record as its canonical representative —
239
+ * the one with the fewest empty fields (`null` / `undefined` / `""`).
240
+ *
241
+ * Ties keep the earliest.
242
+ * A basic, generic canonicalizer.
243
+ *
244
+ * Field-level merging across the cluster is the application's job (it knows which source to trust).
228
245
  */
229
246
  export function representative<R extends object>(group: readonly R[]): R | undefined {
230
247
  let best: R | undefined
@@ -3,25 +3,14 @@
3
3
  * @license AGPL-3.0
4
4
  * @author Teffen Ellis, et al.
5
5
  *
6
- * String comparators for the matcher's scoring stage.
7
- *
8
- * The record-linkage literature (Winkler/Census; Belin 1993) settles on the prefix-weighted Jaro
9
- * comparator (Jaro-Winkler) as the default for names: it tolerates the typographical error real
10
- * data is full of better than raw character-edit distance. But J-W has a documented blind spot on
11
- * compound / double surnames (e.g. Hispanic `Garcia Lopez`): the second half of the compound
12
- * falls outside J-W's match window, so `Lopez` vs `Garcia Lopez` scores ~0. The fix the
13
- * literature prescribes is an edit-distance / token fallback for single-vs-compound pairs —
14
- * implemented in {@link nameSimilarity}.
15
- *
16
- * These are pure similarity primitives in [0, 1]. The mapping of a similarity onto discrete
17
- * Fellegi-Sunter agreement levels (and the m/u weights) is the scorer's job, not theirs.
6
+ * Jaro-Winkler is the default record-linkage comparator for names. {@link nameSimilarity} adds
7
+ * the literature's token/edit fallback because J-W scores a compound surname's second half near zero.
18
8
  */
19
9
 
20
10
  import { distance as levenshteinDistance } from "fastest-levenshtein"
21
11
 
22
12
  /**
23
- * Jaro similarity in [0, 1]. Two empty strings are identical (1); one empty is 0. Counts matching characters within a
24
- * sliding window of `floor(max(len)/2) - 1`, discounting half-transpositions.
13
+ * Jaro similarity in [0, 1], where two empty strings are identical (1) and one empty is 0.
25
14
  */
26
15
  export function jaro(a: string, b: string): number {
27
16
  if (a === b) return 1
@@ -53,7 +42,6 @@ export function jaro(a: string, b: string): number {
53
42
 
54
43
  if (matches === 0) return 0
55
44
 
56
- // Count transpositions: matched chars of `a` and `b`, in order, that disagree (halved).
57
45
  let transpositions = 0
58
46
  let k = 0
59
47
 
@@ -77,9 +65,8 @@ export function jaro(a: string, b: string): number {
77
65
  }
78
66
 
79
67
  /**
80
- * Jaro-Winkler similarity in [0, 1]: Jaro with a bonus for a shared prefix — `jw = jaro + prefix * weight * (1 -
81
- * jaro)`, prefix capped at `maxPrefix` (Winkler's standard 4), `weight` the scaling factor (standard 0.1). Only boosts
82
- * when `jaro` already clears `boostThreshold` (0.7), per Winkler.
68
+ * Jaro-Winkler similarity in [0, 1]: Jaro plus a shared-prefix bonus, boosting only when Jaro
69
+ * already clears the 0.7 threshold, with Winkler's standard prefix cap 4 and weight 0.1.
83
70
  */
84
71
  export function jaroWinkler(
85
72
  a: string,
@@ -105,14 +92,8 @@ export function jaroWinkler(
105
92
  }
106
93
 
107
94
  /**
108
- * Jaccard similarity between two token sets in [0, 1]: `|a ∩ b| / |a ∪ b|`.
109
- *
110
- * The set-of-tokens complement to the string comparators above. Where {@link nameSimilarity} asks how close two names
111
- * LOOK, this asks how much two token bags OVERLAP — the right question for organization names and address bags, where
112
- * word order carries no information and a shared rare token is worth more than character-level proximity.
113
- *
114
- * Either side empty scores 0 rather than 1: an empty bag agrees with nothing, and treating "no evidence" as "perfect
115
- * agreement" is how a blocking pass floods with false pairs.
95
+ * Jaccard similarity `|a ∩ b| / |a ∪ b|` over two token sets, where an empty side scores 0 rather than 1
96
+ * because treating no evidence as perfect agreement floods a blocking pass with false pairs.
116
97
  */
117
98
  export function jaccard(a: ReadonlySet<string>, b: ReadonlySet<string>): number {
118
99
  if (!a.size || !b.size) return 0
@@ -140,15 +121,11 @@ export function levenshteinSimilarity(a: string, b: string): number {
140
121
  }
141
122
 
142
123
  /**
143
- * Name-aware similarity in [0, 1]. Jaro-Winkler by default, with the compound-surname fallback the literature
144
- * prescribes:
145
- *
146
- * - If one name's tokens are a strict subset of the other's (`Lopez` ⊂ `Garcia Lopez`), that is strong partial agreement
147
- * J-W misses — floor the score at 0.9.
148
- * - Otherwise return the better of Jaro-Winkler and normalized edit similarity, so a single token that is a substring of
149
- * a longer compound (`Garcia` vs `Garcialopez`) still scores sensibly.
124
+ * Name-aware similarity in [0, 1] that floors the score at 0.9 when one name's
125
+ * tokens are a strict subset of the other's.
150
126
  *
151
- * Case- and whitespace-insensitive. Empty input scores 0.
127
+ * In other cases, it returns the better of Jaro-Winkler and normalized edit
128
+ * similarity without case or whitespace sensitivity.
152
129
  */
153
130
  export function nameSimilarity(a: string, b: string): number {
154
131
  const x = a.trim().toLowerCase().replaceAll(/\s+/g, " ")
package/lib/distance.ts CHANGED
@@ -2,46 +2,31 @@
2
2
  * @copyright Sister Software
3
3
  * @license AGPL-3.0
4
4
  * @author Teffen Ellis, et al.
5
- *
6
- * Geographic distance as a scoring feature — the other half of geocode-first matching.
7
- *
8
- * Blocking uses geography to _propose_ candidates; this scores them on it. The research is explicit
9
- * that an address must be matched as a SPATIAL attribute, not by string similarity (a
10
- * one-character edit can be 650 m apart), and that distance measurably helps as a comparison
11
- * feature. So we bucket the great-circle distance between two records' coordinates into ordered
12
- * Fellegi-Sunter agreement levels (Splink's `DistanceInKMAtThresholds`): "same building" / "same
13
- * block" / "same area" / far, each with its own m/u and weight.
14
- *
15
- * Calibrate the bucket boundaries to the geocoder's OWN error, which is heavy-tailed and density-
16
- * dependent (≈38 m urban, ≈200 m rural). A weakening of this evidence by geocode quality (a
17
- * shared interpolated centroid is softer than a shared rooftop point) is the documented
18
- * refinement.
19
5
  */
20
6
 
21
- import { haversineKm as greatCircleKm } from "@mailwoman/spatial"
7
+ import { haversineKm as greatCircleKm, type GeoCoordinate } from "@mailwoman/spatial"
22
8
 
23
- import type { LatLon } from "#blocking"
24
9
  import type { Comparison, ComparisonLevel } from "#fellegi-sunter"
25
10
 
26
11
  /**
27
- * Great-circle (haversine) distance in km between two coordinates. The formula's one true home is `@mailwoman/spatial`;
28
- * this is a thin domain-typed adapter from `match`'s `LatLon` ({ latitude, longitude }) onto the canonical scalar
29
- * helper — not a second implementation.
12
+ * Computes the great-circle distance in kilometers between two `GeoCoordinate` records.
13
+ * It delegates to the scalar helper in `@mailwoman/spatial`.
30
14
  */
31
- export const haversineKm = (a: LatLon, b: LatLon): number =>
15
+ export const haversineKm = (a: GeoCoordinate, b: GeoCoordinate): number =>
32
16
  greatCircleKm(a.latitude, a.longitude, b.latitude, b.longitude)
33
17
 
34
18
  /**
35
- * A geo-distance comparison: bucket the great-circle distance between two records' coordinates into ordered agreement
36
- * levels. Levels must be ordered NEAREST first by `maxKm`, the last acting as the `far` catch-all (`maxKm` omitted →
37
- * unbounded). A missing/invalid coordinate on either side yields no evidence.
19
+ * Creates a comparison that buckets the great-circle distance between two records into
20
+ * levels ordered nearest first by `maxKm`, with the last level catching everything farther.
21
+ *
22
+ * A missing or non-finite coordinate on either side yields no evidence.
38
23
  */
39
24
  export function distanceComparison<R>(config: {
40
25
  name: string
41
- extract: (record: R) => LatLon | null | undefined
26
+ extract: (record: R) => GeoCoordinate | null | undefined
42
27
  levels: ComparisonLevel[]
43
28
  }): Comparison<R> {
44
- const valid = (c: LatLon | null | undefined): c is LatLon =>
29
+ const valid = (c: GeoCoordinate | null | undefined): c is GeoCoordinate =>
45
30
  !!c && Number.isFinite(c.latitude) && Number.isFinite(c.longitude)
46
31
 
47
32
  return {
@@ -65,8 +50,8 @@ export function distanceComparison<R>(config: {
65
50
  }
66
51
 
67
52
  /**
68
- * Default distance levels, nearest → far, with boundaries at rooftop / block / locality scale. The m/u are illustrative
69
- * seeds (EM re-estimates them); the boundaries reflect typical geocoder error.
53
+ * Provides default distance levels at building, block and area scales.
54
+ * Their `m` and `u` values seed EM re-estimation.
70
55
  */
71
56
  export const DEFAULT_DISTANCE_LEVELS: ComparisonLevel[] = [
72
57
  { label: "same-building", maxKm: 0.05, m: 0.7, u: 0.001 },
@@ -76,30 +61,19 @@ export const DEFAULT_DISTANCE_LEVELS: ComparisonLevel[] = [
76
61
  ]
77
62
 
78
63
  /**
79
- * The collapsed spatial-agreement comparison — ONE non-redundant geographic signal.
64
+ * Creates a single spatial comparison whose level 0 is an exact canonical-key match
65
+ * and whose remaining levels bucket great-circle distance for pairs with different keys.
80
66
  *
81
- * The first matcher carried TWO spatial comparisons: canonical-address-key similarity AND great-circle distance. They
82
- * double-count — an exact key match implies distance ≈ 0, so a co-located pair banked the same evidence twice, and the
83
- * redundant vote is exactly what over-merges distinct providers at a shared clinic address. This folds them into one
84
- * comparison:
85
- *
86
- * - **level 0 `same-key`** — an EXACT canonical-key match: the strongest tier, and the one the inverse-address-frequency
87
- * adjustment rides ({@link withTermFrequency} on level 0), so agreement on a crowded shared key is down-weighted
88
- * toward worthless while a rare one keeps full weight.
89
- * - **levels 1…n** — great-circle distance buckets for pairs whose keys DIFFER, so "123 Main St" vs "123 Main Street Apt
90
- * 2" that geocode to the same rooftop still earns near-agreement (the geo-first point of the whole design).
91
- * - Keys differ and no usable coordinate → no evidence.
92
- *
93
- * Exactly one spatial vote, no redundancy. Pass {@link DEFAULT_SPATIAL_LEVELS} or your own; index 0 must be the
94
- * exact-key tier, indices 1…n the distance buckets nearest → far by `maxKm` (last = `far`).
67
+ * Separate key and distance comparisons count a co-located pair's evidence twice.
68
+ * They also over-merge distinct entities at a shared address.
95
69
  */
96
70
  export function spatialComparison<R>(config: {
97
71
  name: string
98
72
  key: (record: R) => string | null | undefined
99
- coordinate: (record: R) => LatLon | null | undefined
73
+ coordinate: (record: R) => GeoCoordinate | null | undefined
100
74
  levels: ComparisonLevel[]
101
75
  }): Comparison<R> {
102
- const valid = (c: LatLon | null | undefined): c is LatLon =>
76
+ const valid = (c: GeoCoordinate | null | undefined): c is GeoCoordinate =>
103
77
  !!c && Number.isFinite(c.latitude) && Number.isFinite(c.longitude)
104
78
 
105
79
  return {
@@ -109,12 +83,12 @@ export function spatialComparison<R>(config: {
109
83
  const ka = config.key(a)
110
84
  const kb = config.key(b)
111
85
 
112
- if (ka && kb && ka.trim() && ka === kb) return 0 // exact canonical-key match — one strong vote
86
+ if (ka && kb && ka.trim() && ka === kb) return 0
113
87
 
114
88
  const ca = config.coordinate(a)
115
89
  const cb = config.coordinate(b)
116
90
 
117
- if (!valid(ca) || !valid(cb)) return -1 // keys differ and no coordinate → no spatial evidence
91
+ if (!valid(ca) || !valid(cb)) return -1
118
92
 
119
93
  const km = haversineKm(ca, cb)
120
94
 
@@ -128,8 +102,8 @@ export function spatialComparison<R>(config: {
128
102
  }
129
103
 
130
104
  /**
131
- * Default levels for {@link spatialComparison}: an exact same-key tier on top of the distance buckets. `m`/`u` are
132
- * EM-estimable seeds (m decreasing, u increasing down the tiers; each column ≈ sums to 1).
105
+ * Provides default levels for {@link spatialComparison}: an exact same-key tier followed
106
+ * by the building, block, area and far distance buckets, with seed `m` and `u` values.
133
107
  */
134
108
  export const DEFAULT_SPATIAL_LEVELS: ComparisonLevel[] = [
135
109
  { label: "same-key", m: 0.85, u: 0.01 },
package/lib/em.ts CHANGED
@@ -41,15 +41,21 @@ export function agreementPattern<R>(comparisons: Comparison<R>[], a: R, b: R): n
41
41
  */
42
42
  export interface EmOptions {
43
43
  /**
44
- * Hard iteration cap. Default 100.
44
+ * Hard iteration cap.
45
+ *
46
+ * Default 100.
45
47
  */
46
48
  maxIterations?: number
47
49
  /**
48
- * Convergence tolerance on the largest parameter change between iterations. Default 1e-6.
50
+ * Convergence tolerance on the largest parameter change between iterations.
51
+ *
52
+ * Default 1e-6.
49
53
  */
50
54
  tolerance?: number
51
55
  /**
52
- * Starting prior match rate. Defaults to the model's `lambda`.
56
+ * Prior match rate used to initialize the model.
57
+ *
58
+ * Defaults to the model's `lambda`.
53
59
  */
54
60
  initialLambda?: number
55
61
  }
@@ -71,9 +77,11 @@ export interface EmResult<R> {
71
77
  }
72
78
 
73
79
  /**
74
- * Estimate `m`/`u` and the prior `λ` from unlabeled agreement patterns via EM. The patterns are per-comparison level
75
- * indices (as produced by {@link agreementPattern}); a `-1` (missing) field contributes no evidence to either class. The
76
- * model's existing level `m`/`u` seed the iteration.
80
+ * Estimate `m`/`u` and the prior `λ` from unlabeled agreement patterns via EM.
81
+ *
82
+ * The patterns are per-comparison level indices (as produced by {@link agreementPattern});
83
+ * a `-1` (missing) field contributes no evidence to either class.
84
+ * The model's existing level `m`/`u` seed the iteration.
77
85
  */
78
86
  export function estimateParameters<R>(
79
87
  model: FellegiSunterModel<R>,