@mailwoman/match 9.4.0 → 10.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/lib/blocking.ts +41 -36
- package/lib/clustering.ts +43 -26
- package/lib/comparators.ts +11 -34
- package/lib/distance.ts +22 -48
- package/lib/em.ts +14 -6
- package/lib/fellegi-sunter.ts +39 -22
- package/lib/gbt.ts +8 -32
- package/lib/tf.ts +26 -14
- package/out/blocking.d.ts +40 -35
- package/out/blocking.d.ts.map +1 -1
- package/out/blocking.js +34 -25
- package/out/blocking.js.map +1 -1
- package/out/clustering.d.ts +30 -17
- package/out/clustering.d.ts.map +1 -1
- package/out/clustering.js +31 -19
- package/out/clustering.js.map +1 -1
- package/out/comparators.d.ts +11 -33
- package/out/comparators.d.ts.map +1 -1
- package/out/comparators.js +11 -34
- package/out/comparators.js.map +1 -1
- package/out/distance.d.ts +18 -43
- package/out/distance.d.ts.map +1 -1
- package/out/distance.js +16 -41
- package/out/distance.js.map +1 -1
- package/out/em.d.ts +14 -6
- package/out/em.d.ts.map +1 -1
- package/out/em.js +5 -3
- package/out/em.js.map +1 -1
- package/out/fellegi-sunter.d.ts +39 -22
- package/out/fellegi-sunter.d.ts.map +1 -1
- package/out/fellegi-sunter.js +17 -12
- package/out/fellegi-sunter.js.map +1 -1
- package/out/gbt.d.ts +7 -17
- package/out/gbt.d.ts.map +1 -1
- package/out/gbt.js +7 -31
- package/out/gbt.js.map +1 -1
- package/out/tf.d.ts +24 -13
- package/out/tf.d.ts.map +1 -1
- package/out/tf.js +17 -11
- package/out/tf.js.map +1 -1
- package/package.json +13 -88
package/out/clustering.d.ts
CHANGED
|
@@ -3,10 +3,10 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
6
|
+
* The clustering stage resolves scored pairs into canonical entities.
|
|
7
7
|
*
|
|
8
|
-
* The pairwise scorer treats each pair independently
|
|
9
|
-
* high weight and B~C at a high weight
|
|
8
|
+
* The pairwise scorer treats each pair independently. Its scores are not transitive: A~B at a
|
|
9
|
+
* high weight and B~C at a high weight do not guarantee that A~C is a match. A separate stage is
|
|
10
10
|
* required to turn the graph of above-threshold links into coherent groups — skip it and your
|
|
11
11
|
* "entities" silently fracture or fuse.
|
|
12
12
|
*
|
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
* for more recall. Its known weakness is over-merging via transitive chains (a string of weak
|
|
16
16
|
* links can pull unrelated records into one component); the principled fix is
|
|
17
17
|
* centroid-/average-linkage hierarchical clustering (Dedupe), which uses the full within-cluster
|
|
18
|
-
* score matrix — a documented refinement
|
|
18
|
+
* score matrix — a documented refinement rather than this first version. For a geocode-first matcher the
|
|
19
19
|
* over-merge risk is already damped: blocking keeps candidate sets local, so chains can't run
|
|
20
20
|
* across the whole dataset.
|
|
21
21
|
*/
|
|
@@ -38,29 +38,42 @@ export interface ClusterOptions {
|
|
|
38
38
|
/**
|
|
39
39
|
* How the above-threshold link graph resolves into clusters:
|
|
40
40
|
*
|
|
41
|
-
* - `"single"` (default) — connected components (union-find).
|
|
42
|
-
*
|
|
43
|
-
* -
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
*
|
|
41
|
+
* - `"single"` (default) — connected components (union-find).
|
|
42
|
+
* Fast.
|
|
43
|
+
* Any above-threshold link fuses two groups, so a single weak link can over-merge
|
|
44
|
+
* unrelated records through a transitive chain.
|
|
45
|
+
* - `"average"` — agglomerative average-linkage refinement within each connected component:
|
|
46
|
+
* two sub-clusters merge only when the average weight of the links between them clears
|
|
47
|
+
* the threshold, so a lone weak bridge no longer fuses two otherwise-dense groups.
|
|
48
|
+
* The documented over-merge fix (Dedupe).
|
|
49
|
+
* Falls back to single-linkage for any component larger than {@link maxAverageLinkageComponent}.
|
|
47
50
|
*/
|
|
48
51
|
linkage?: "single" | "average";
|
|
49
52
|
/**
|
|
50
|
-
* Components larger than this skip the O(k³) average-linkage refine and keep single-linkage.
|
|
53
|
+
* Components larger than this skip the O(k³) average-linkage refine and keep single-linkage.
|
|
54
|
+
*
|
|
55
|
+
* Default 64.
|
|
51
56
|
*/
|
|
52
57
|
maxAverageLinkageComponent?: number;
|
|
53
58
|
}
|
|
54
59
|
/**
|
|
55
|
-
* Cluster records into canonical entities by connected components of the above-threshold link graph.
|
|
56
|
-
*
|
|
57
|
-
*
|
|
60
|
+
* Cluster records into canonical entities by connected components of the above-threshold link graph.
|
|
61
|
+
*
|
|
62
|
+
* Every input record lands in exactly one cluster.
|
|
63
|
+
* A record with no qualifying link is a singleton.
|
|
64
|
+
* Links referencing a record not in `records` are ignored.
|
|
65
|
+
*
|
|
66
|
+
* Reference identity is used, so pass the same record objects to both arguments.
|
|
58
67
|
*/
|
|
59
68
|
export declare function cluster<R>(records: readonly R[], links: Iterable<ScoredLink<R>>, opts: ClusterOptions): R[][];
|
|
60
69
|
/**
|
|
61
|
-
* Pick a cluster's most complete record as its canonical representative —
|
|
62
|
-
* / `undefined` / `""`).
|
|
63
|
-
*
|
|
70
|
+
* Pick a cluster's most complete record as its canonical representative —
|
|
71
|
+
* the one with the fewest empty fields (`null` / `undefined` / `""`).
|
|
72
|
+
*
|
|
73
|
+
* Ties keep the earliest.
|
|
74
|
+
* A basic, generic canonicalizer.
|
|
75
|
+
*
|
|
76
|
+
* Field-level merging across the cluster is the application's job (it knows which source to trust).
|
|
64
77
|
*/
|
|
65
78
|
export declare function representative<R extends object>(group: readonly R[]): R | undefined;
|
|
66
79
|
//# sourceMappingURL=clustering.d.ts.map
|
package/out/clustering.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"clustering.d.ts","sourceRoot":"","sources":["../lib/clustering.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAEH;;GAEG;AACH,MAAM,WAAW,UAAU,CAAC,CAAC;IAC5B,CAAC,EAAE,CAAC,CAAA;IACJ,CAAC,EAAE,CAAC,CAAA;IACJ,MAAM,EAAE,MAAM,CAAA;CACd;AAED;;GAEG;AACH,MAAM,WAAW,cAAc;IAC9B;;OAEG;IACH,SAAS,EAAE,MAAM,CAAA;IACjB
|
|
1
|
+
{"version":3,"file":"clustering.d.ts","sourceRoot":"","sources":["../lib/clustering.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAEH;;GAEG;AACH,MAAM,WAAW,UAAU,CAAC,CAAC;IAC5B,CAAC,EAAE,CAAC,CAAA;IACJ,CAAC,EAAE,CAAC,CAAA;IACJ,MAAM,EAAE,MAAM,CAAA;CACd;AAED;;GAEG;AACH,MAAM,WAAW,cAAc;IAC9B;;OAEG;IACH,SAAS,EAAE,MAAM,CAAA;IACjB;;;;;;;;;;;;OAYG;IACH,OAAO,CAAC,EAAE,QAAQ,GAAG,SAAS,CAAA;IAC9B;;;;OAIG;IACH,0BAA0B,CAAC,EAAE,MAAM,CAAA;CACnC;AAsDD;;;;;;;;GAQG;AACH,wBAAgB,OAAO,CAAC,CAAC,EAAE,OAAO,EAAE,SAAS,CAAC,EAAE,EAAE,KAAK,EAAE,QAAQ,CAAC,UAAU,CAAC,CAAC,CAAC,CAAC,EAAE,IAAI,EAAE,cAAc,GAAG,CAAC,EAAE,EAAE,CAgH7G;AAED;;;;;;;;GAQG;AACH,wBAAgB,cAAc,CAAC,CAAC,SAAS,MAAM,EAAE,KAAK,EAAE,SAAS,CAAC,EAAE,GAAG,CAAC,GAAG,SAAS,CAoBnF"}
|
package/out/clustering.js
CHANGED
|
@@ -3,10 +3,10 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
6
|
+
* The clustering stage resolves scored pairs into canonical entities.
|
|
7
7
|
*
|
|
8
|
-
* The pairwise scorer treats each pair independently
|
|
9
|
-
* high weight and B~C at a high weight
|
|
8
|
+
* The pairwise scorer treats each pair independently. Its scores are not transitive: A~B at a
|
|
9
|
+
* high weight and B~C at a high weight do not guarantee that A~C is a match. A separate stage is
|
|
10
10
|
* required to turn the graph of above-threshold links into coherent groups — skip it and your
|
|
11
11
|
* "entities" silently fracture or fuse.
|
|
12
12
|
*
|
|
@@ -15,15 +15,17 @@
|
|
|
15
15
|
* for more recall. Its known weakness is over-merging via transitive chains (a string of weak
|
|
16
16
|
* links can pull unrelated records into one component); the principled fix is
|
|
17
17
|
* centroid-/average-linkage hierarchical clustering (Dedupe), which uses the full within-cluster
|
|
18
|
-
* score matrix — a documented refinement
|
|
18
|
+
* score matrix — a documented refinement rather than this first version. For a geocode-first matcher the
|
|
19
19
|
* over-merge risk is already damped: blocking keeps candidate sets local, so chains can't run
|
|
20
20
|
* across the whole dataset.
|
|
21
21
|
*/
|
|
22
22
|
/**
|
|
23
|
-
* Refine one connected component by agglomerative average-linkage.
|
|
24
|
-
*
|
|
25
|
-
*
|
|
26
|
-
*
|
|
23
|
+
* Refine one connected component by agglomerative average-linkage.
|
|
24
|
+
*
|
|
25
|
+
* Starts with every member a singleton and repeatedly merges the cluster pair
|
|
26
|
+
* with the highest _average_ inter-cluster link weight while that average is at
|
|
27
|
+
* or above `threshold`; clusters with no link between them never merge.
|
|
28
|
+
* O(k³) in the component size, so callers boundate it on a size cap.
|
|
27
29
|
*/
|
|
28
30
|
function averageLinkageRefine(members, edges, threshold) {
|
|
29
31
|
const clusters = members.map((_, i) => [i]);
|
|
@@ -61,14 +63,19 @@ function averageLinkageRefine(members, edges, threshold) {
|
|
|
61
63
|
return clusters.map((local) => local.map((i) => members[i]));
|
|
62
64
|
}
|
|
63
65
|
/**
|
|
64
|
-
* Cluster records into canonical entities by connected components of the above-threshold link graph.
|
|
65
|
-
*
|
|
66
|
-
*
|
|
66
|
+
* Cluster records into canonical entities by connected components of the above-threshold link graph.
|
|
67
|
+
*
|
|
68
|
+
* Every input record lands in exactly one cluster.
|
|
69
|
+
* A record with no qualifying link is a singleton.
|
|
70
|
+
* Links referencing a record not in `records` are ignored.
|
|
71
|
+
*
|
|
72
|
+
* Reference identity is used, so pass the same record objects to both arguments.
|
|
67
73
|
*/
|
|
68
74
|
export function cluster(records, links, opts) {
|
|
69
75
|
const index = new Map();
|
|
70
76
|
records.forEach((record, i) => index.set(record, i));
|
|
71
|
-
// Local by design: the shared union-find is `@mailwoman/core/utils/union-find`;
|
|
77
|
+
// Local by design: the shared union-find is `@mailwoman/core/utils/union-find`;
|
|
78
|
+
// match takes no core dependency (~11 MB).
|
|
72
79
|
const parent = records.map((_, i) => i);
|
|
73
80
|
const rank = new Array(records.length).fill(0);
|
|
74
81
|
const find = (x) => {
|
|
@@ -100,9 +107,10 @@ export function cluster(records, links, opts) {
|
|
|
100
107
|
rank[rx]++;
|
|
101
108
|
}
|
|
102
109
|
};
|
|
103
|
-
// Collect
|
|
104
|
-
// above-threshold ones, but the average-linkage refinement needs the full sub-graph
|
|
105
|
-
// disagreeing below-threshold edge between two sub-clusters is
|
|
110
|
+
// Collect all valid links (not just above-threshold): connected components form from the
|
|
111
|
+
// above-threshold ones, but the average-linkage refinement needs the full sub-graph.
|
|
112
|
+
// A weak or disagreeing below-threshold edge between two sub-clusters is
|
|
113
|
+
// exactly what should pull them apart.
|
|
106
114
|
const allLinks = [];
|
|
107
115
|
for (const link of links) {
|
|
108
116
|
const ia = index.get(link.a);
|
|
@@ -128,7 +136,7 @@ export function cluster(records, links, opts) {
|
|
|
128
136
|
if (opts.linkage !== "average")
|
|
129
137
|
return [...groups.values()];
|
|
130
138
|
// Average-linkage refinement: split each component where its sub-clusters are joined only by a weak
|
|
131
|
-
// bridge (the average inter-cluster link weight, over
|
|
139
|
+
// bridge (the average inter-cluster link weight, over all edges between them, falls below the threshold).
|
|
132
140
|
const maxComponent = opts.maxAverageLinkageComponent ?? 64;
|
|
133
141
|
const localOf = new Map();
|
|
134
142
|
// member → its index within its own group
|
|
@@ -157,9 +165,13 @@ export function cluster(records, links, opts) {
|
|
|
157
165
|
return result;
|
|
158
166
|
}
|
|
159
167
|
/**
|
|
160
|
-
* Pick a cluster's most complete record as its canonical representative —
|
|
161
|
-
* / `undefined` / `""`).
|
|
162
|
-
*
|
|
168
|
+
* Pick a cluster's most complete record as its canonical representative —
|
|
169
|
+
* the one with the fewest empty fields (`null` / `undefined` / `""`).
|
|
170
|
+
*
|
|
171
|
+
* Ties keep the earliest.
|
|
172
|
+
* A basic, generic canonicalizer.
|
|
173
|
+
*
|
|
174
|
+
* Field-level merging across the cluster is the application's job (it knows which source to trust).
|
|
163
175
|
*/
|
|
164
176
|
export function representative(group) {
|
|
165
177
|
let best;
|
package/out/clustering.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"clustering.js","sourceRoot":"","sources":["../lib/clustering.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;
|
|
1
|
+
{"version":3,"file":"clustering.js","sourceRoot":"","sources":["../lib/clustering.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAyCH;;;;;;;GAOG;AACH,SAAS,oBAAoB,CAAI,OAAY,EAAE,KAAsC,EAAE,SAAiB;IACvG,MAAM,QAAQ,GAAG,OAAO,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC,CAAA;IAE3C,MAAM,YAAY,GAAG,CAAC,CAAW,EAAE,CAAW,EAAiB,EAAE;QAChE,MAAM,GAAG,GAAG,IAAI,GAAG,CAAC,CAAC,CAAC,CAAA;QACtB,MAAM,GAAG,GAAG,IAAI,GAAG,CAAC,CAAC,CAAC,CAAA;QACtB,IAAI,GAAG,GAAG,CAAC,CAAA;QACX,IAAI,KAAK,GAAG,CAAC,CAAA;QAEb,KAAK,MAAM,CAAC,CAAC,EAAE,CAAC,EAAE,CAAC,CAAC,IAAI,KAAK,EAAE,CAAC;YAC/B,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,CAAC,IAAI,GAAG,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,CAAC,IAAI,GAAG,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC;gBAC9D,GAAG,IAAI,CAAC,CAAA;gBAER,KAAK,EAAE,CAAA;YACR,CAAC;QACF,CAAC;QAED,OAAO,KAAK,GAAG,CAAC,CAAC,CAAC,CAAC,GAAG,GAAG,KAAK,CAAC,CAAC,CAAC,IAAI,CAAA;IACtC,CAAC,CAAA;IAED,SAAS,CAAC;QACT,IAAI,OAAO,GAAG,CAAC,QAAQ,CAAA;QACvB,IAAI,QAAQ,GAA4B,IAAI,CAAA;QAE5C,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,QAAQ,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;YAC1C,KAAK,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,QAAQ,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;gBAC9C,MAAM,GAAG,GAAG,YAAY,CAAC,QAAQ,CAAC,CAAC,CAAE,EAAE,QAAQ,CAAC,CAAC,CAAE,CAAC,CAAA;gBAEpD,IAAI,GAAG,KAAK,IAAI,IAAI,GAAG,GAAG,OAAO,EAAE,CAAC;oBACnC,OAAO,GAAG,GAAG,CAAA;oBACb,QAAQ,GAAG,CAAC,CAAC,EAAE,CAAC,CAAC,CAAA;gBAClB,CAAC;YACF,CAAC;QACF,CAAC;QAED,IAAI,CAAC,QAAQ,IAAI,OAAO,GAAG,SAAS;YAAE,MAAK;QAC3C,MAAM,CAAC,CAAC,EAAE,CAAC,CAAC,GAAG,QAAQ,CAAA;QACvB,QAAQ,CAAC,CAAC,CAAC,GAAG,QAAQ,CAAC,CAAC,CAAE,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAE,CAAC,CAAA;QAC/C,QAAQ,CAAC,MAAM,CAAC,CAAC,EAAE,CAAC,CAAC,CAAA;IACtB,CAAC;IAED,OAAO,QAAQ,CAAC,GAAG,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,OAAO,CAAC,CAAC,CAAE,CAAC,CAAC,CAAA;AAC9D,CAAC;AAED;;;;;;;;GAQG;AACH,MAAM,UAAU,OAAO,CAAI,OAAqB,EAAE,KAA8B,EAAE,IAAoB;IACrG,MAAM,KAAK,GAAG,IAAI,GAAG,EAAa,CAAA;IAClC,OAAO,CAAC,OAAO,CAAC,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC,KAAK,CAAC,GAAG,CAAC,MAAM,EAAE,CAAC,CAAC,CAAC,CAAA;IAEpD,gFAAgF;IAChF,2CAA2C;IAC3C,MAAM,MAAM,GAAG,OAAO,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAA;IACvC,MAAM,IAAI,GAAG,IAAI,KAAK,CAAS,OAAO,CAAC,MAAM,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,CAAA;IAEtD,MAAM,IAAI,GAAG,CAAC,CAAS,EAAU,EAAE;QAClC,IAAI,IAAI,GAAG,CAAC,CAAA;QAEZ,OAAO,MAAM,CAAC,IAAI,CAAC,KAAK,IAAI,EAAE,CAAC;YAC9B,IAAI,GAAG,MAAM,CAAC,IAAI,CAAE,CAAA;QACrB,CAAC;QAED,oBAAoB;QACpB,OAAO,MAAM,CAAC,CAAC,CAAC,KAAK,IAAI,EAAE,CAAC;YAC3B,MAAM,IAAI,GAAG,MAAM,CAAC,CAAC,CAAE,CAAA;YACvB,MAAM,CAAC,CAAC,CAAC,GAAG,IAAI,CAAA;YAChB,CAAC,GAAG,IAAI,CAAA;QACT,CAAC;QAED,OAAO,IAAI,CAAA;IACZ,CAAC,CAAA;IAED,MAAM,KAAK,GAAG,CAAC,CAAS,EAAE,CAAS,EAAQ,EAAE;QAC5C,MAAM,EAAE,GAAG,IAAI,CAAC,CAAC,CAAC,CAAA;QAClB,MAAM,EAAE,GAAG,IAAI,CAAC,CAAC,CAAC,CAAA;QAElB,IAAI,EAAE,KAAK,EAAE;YAAE,OAAM;QAErB,IAAI,IAAI,CAAC,EAAE,CAAE,GAAG,IAAI,CAAC,EAAE,CAAE,EAAE,CAAC;YAC3B,MAAM,CAAC,EAAE,CAAC,GAAG,EAAE,CAAA;QAChB,CAAC;aAAM,IAAI,IAAI,CAAC,EAAE,CAAE,GAAG,IAAI,CAAC,EAAE,CAAE,EAAE,CAAC;YAClC,MAAM,CAAC,EAAE,CAAC,GAAG,EAAE,CAAA;QAChB,CAAC;aAAM,CAAC;YACP,MAAM,CAAC,EAAE,CAAC,GAAG,EAAE,CAAA;YAEf,IAAI,CAAC,EAAE,CAAE,EAAE,CAAA;QACZ,CAAC;IACF,CAAC,CAAA;IAED,yFAAyF;IACzF,qFAAqF;IACrF,yEAAyE;IACzE,uCAAuC;IACvC,MAAM,QAAQ,GAAoB,EAAE,CAAA;IAEpC,KAAK,MAAM,IAAI,IAAI,KAAK,EAAE,CAAC;QAC1B,MAAM,EAAE,GAAG,KAAK,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC,CAAC,CAAA;QAC5B,MAAM,EAAE,GAAG,KAAK,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC,CAAC,CAAA;QAE5B,IAAI,EAAE,KAAK,SAAS,IAAI,EAAE,KAAK,SAAS;YAAE,SAAQ;QAClD,QAAQ,CAAC,IAAI,CAAC,IAAI,CAAC,CAAA;QAEnB,IAAI,IAAI,CAAC,MAAM,IAAI,IAAI,CAAC,SAAS,EAAE,CAAC;YACnC,KAAK,CAAC,EAAE,EAAE,EAAE,CAAC,CAAA;QACd,CAAC;IACF,CAAC;IAED,MAAM,MAAM,GAAG,IAAI,GAAG,EAAe,CAAA;IAErC,OAAO,CAAC,OAAO,CAAC,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE;QAC7B,MAAM,IAAI,GAAG,IAAI,CAAC,CAAC,CAAC,CAAA;QACpB,MAAM,KAAK,GAAG,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,CAAA;QAE9B,IAAI,KAAK,EAAE,CAAC;YACX,KAAK,CAAC,IAAI,CAAC,MAAM,CAAC,CAAA;QACnB,CAAC;aAAM,CAAC;YACP,MAAM,CAAC,GAAG,CAAC,IAAI,EAAE,CAAC,MAAM,CAAC,CAAC,CAAA;QAC3B,CAAC;IACF,CAAC,CAAC,CAAA;IAEF,IAAI,IAAI,CAAC,OAAO,KAAK,SAAS;QAAE,OAAO,CAAC,GAAG,MAAM,CAAC,MAAM,EAAE,CAAC,CAAA;IAE3D,oGAAoG;IACpG,0GAA0G;IAC1G,MAAM,YAAY,GAAG,IAAI,CAAC,0BAA0B,IAAI,EAAE,CAAA;IAC1D,MAAM,OAAO,GAAG,IAAI,GAAG,EAAa,CAAA;IAEpC,0CAA0C;IAC1C,KAAK,MAAM,OAAO,IAAI,MAAM,CAAC,MAAM,EAAE,EAAE,CAAC;QACvC,OAAO,CAAC,OAAO,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,CAAA;IAC7C,CAAC;IAED,MAAM,UAAU,GAAG,IAAI,GAAG,EAA2C,CAAA;IAErE,KAAK,MAAM,IAAI,IAAI,QAAQ,EAAE,CAAC;QAC7B,MAAM,IAAI,GAAG,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC,CAAE,CAAC,CAAA;QAErC,IAAI,IAAI,KAAK,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC,CAAE,CAAC;YAAE,SAAQ,CAAC,oDAAoD;QACpG,MAAM,IAAI,GAAG,UAAU,CAAC,GAAG,CAAC,IAAI,CAAC,IAAI,EAAE,CAAA;QACvC,IAAI,CAAC,IAAI,CAAC,CAAC,OAAO,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC,CAAE,EAAE,OAAO,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC,CAAE,EAAE,IAAI,CAAC,MAAM,CAAC,CAAC,CAAA;QACpE,UAAU,CAAC,GAAG,CAAC,IAAI,EAAE,IAAI,CAAC,CAAA;IAC3B,CAAC;IAED,MAAM,MAAM,GAAU,EAAE,CAAA;IAExB,KAAK,MAAM,CAAC,IAAI,EAAE,OAAO,CAAC,IAAI,MAAM,EAAE,CAAC;QACtC,IAAI,OAAO,CAAC,MAAM,IAAI,CAAC,IAAI,OAAO,CAAC,MAAM,GAAG,YAAY,EAAE,CAAC;YAC1D,MAAM,CAAC,IAAI,CAAC,OAAO,CAAC,CAAA;YAEpB,SAAQ;QACT,CAAC;QAED,KAAK,MAAM,GAAG,IAAI,oBAAoB,CAAC,OAAO,EAAE,UAAU,CAAC,GAAG,CAAC,IAAI,CAAC,IAAI,EAAE,EAAE,IAAI,CAAC,SAAS,CAAC,EAAE,CAAC;YAC7F,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,CAAA;QACjB,CAAC;IACF,CAAC;IAED,OAAO,MAAM,CAAA;AACd,CAAC;AAED;;;;;;;;GAQG;AACH,MAAM,UAAU,cAAc,CAAmB,KAAmB;IACnE,IAAI,IAAmB,CAAA;IACvB,IAAI,UAAU,GAAG,CAAC,CAAC,CAAA;IAEnB,KAAK,MAAM,MAAM,IAAI,KAAK,EAAE,CAAC;QAC5B,IAAI,MAAM,GAAG,CAAC,CAAA;QAEd,KAAK,MAAM,KAAK,IAAI,MAAM,CAAC,MAAM,CAAC,MAAM,CAAC,EAAE,CAAC;YAC3C,IAAI,KAAK,KAAK,IAAI,IAAI,KAAK,KAAK,SAAS,IAAI,KAAK,KAAK,EAAE,EAAE,CAAC;gBAC3D,MAAM,EAAE,CAAA;YACT,CAAC;QACF,CAAC;QAED,IAAI,MAAM,GAAG,UAAU,EAAE,CAAC;YACzB,UAAU,GAAG,MAAM,CAAA;YACnB,IAAI,GAAG,MAAM,CAAA;QACd,CAAC;IACF,CAAC;IAED,OAAO,IAAI,CAAA;AACZ,CAAC"}
|
package/out/comparators.d.ts
CHANGED
|
@@ -3,28 +3,16 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
* The record-linkage literature (Winkler/Census; Belin 1993) settles on the prefix-weighted Jaro
|
|
9
|
-
* comparator (Jaro-Winkler) as the default for names: it tolerates the typographical error real
|
|
10
|
-
* data is full of better than raw character-edit distance. But J-W has a documented blind spot on
|
|
11
|
-
* compound / double surnames (e.g. Hispanic `Garcia Lopez`): the second half of the compound
|
|
12
|
-
* falls outside J-W's match window, so `Lopez` vs `Garcia Lopez` scores ~0. The fix the
|
|
13
|
-
* literature prescribes is an edit-distance / token fallback for single-vs-compound pairs —
|
|
14
|
-
* implemented in {@link nameSimilarity}.
|
|
15
|
-
*
|
|
16
|
-
* These are pure similarity primitives in [0, 1]. The mapping of a similarity onto discrete
|
|
17
|
-
* Fellegi-Sunter agreement levels (and the m/u weights) is the scorer's job, not theirs.
|
|
6
|
+
* Jaro-Winkler is the default record-linkage comparator for names. {@link nameSimilarity} adds
|
|
7
|
+
* the literature's token/edit fallback because J-W scores a compound surname's second half near zero.
|
|
18
8
|
*/
|
|
19
9
|
/**
|
|
20
|
-
* Jaro similarity in [0, 1]
|
|
21
|
-
* sliding window of `floor(max(len)/2) - 1`, discounting half-transpositions.
|
|
10
|
+
* Jaro similarity in [0, 1], where two empty strings are identical (1) and one empty is 0.
|
|
22
11
|
*/
|
|
23
12
|
export declare function jaro(a: string, b: string): number;
|
|
24
13
|
/**
|
|
25
|
-
* Jaro-Winkler similarity in [0, 1]: Jaro
|
|
26
|
-
*
|
|
27
|
-
* when `jaro` already clears `boostThreshold` (0.7), per Winkler.
|
|
14
|
+
* Jaro-Winkler similarity in [0, 1]: Jaro plus a shared-prefix bonus, boosting only when Jaro
|
|
15
|
+
* already clears the 0.7 threshold, with Winkler's standard prefix cap 4 and weight 0.1.
|
|
28
16
|
*/
|
|
29
17
|
export declare function jaroWinkler(a: string, b: string, opts?: {
|
|
30
18
|
weight?: number;
|
|
@@ -32,14 +20,8 @@ export declare function jaroWinkler(a: string, b: string, opts?: {
|
|
|
32
20
|
boostThreshold?: number;
|
|
33
21
|
}): number;
|
|
34
22
|
/**
|
|
35
|
-
* Jaccard similarity
|
|
36
|
-
*
|
|
37
|
-
* The set-of-tokens complement to the string comparators above. Where {@link nameSimilarity} asks how close two names
|
|
38
|
-
* LOOK, this asks how much two token bags OVERLAP — the right question for organization names and address bags, where
|
|
39
|
-
* word order carries no information and a shared rare token is worth more than character-level proximity.
|
|
40
|
-
*
|
|
41
|
-
* Either side empty scores 0 rather than 1: an empty bag agrees with nothing, and treating "no evidence" as "perfect
|
|
42
|
-
* agreement" is how a blocking pass floods with false pairs.
|
|
23
|
+
* Jaccard similarity `|a ∩ b| / |a ∪ b|` over two token sets, where an empty side scores 0 rather than 1
|
|
24
|
+
* because treating no evidence as perfect agreement floods a blocking pass with false pairs.
|
|
43
25
|
*/
|
|
44
26
|
export declare function jaccard(a: ReadonlySet<string>, b: ReadonlySet<string>): number;
|
|
45
27
|
/**
|
|
@@ -47,15 +29,11 @@ export declare function jaccard(a: ReadonlySet<string>, b: ReadonlySet<string>):
|
|
|
47
29
|
*/
|
|
48
30
|
export declare function levenshteinSimilarity(a: string, b: string): number;
|
|
49
31
|
/**
|
|
50
|
-
* Name-aware similarity in [0, 1]
|
|
51
|
-
*
|
|
52
|
-
*
|
|
53
|
-
* - If one name's tokens are a strict subset of the other's (`Lopez` ⊂ `Garcia Lopez`), that is strong partial agreement
|
|
54
|
-
* J-W misses — floor the score at 0.9.
|
|
55
|
-
* - Otherwise return the better of Jaro-Winkler and normalized edit similarity, so a single token that is a substring of
|
|
56
|
-
* a longer compound (`Garcia` vs `Garcialopez`) still scores sensibly.
|
|
32
|
+
* Name-aware similarity in [0, 1] that floors the score at 0.9 when one name's
|
|
33
|
+
* tokens are a strict subset of the other's.
|
|
57
34
|
*
|
|
58
|
-
*
|
|
35
|
+
* In other cases, it returns the better of Jaro-Winkler and normalized edit
|
|
36
|
+
* similarity without case or whitespace sensitivity.
|
|
59
37
|
*/
|
|
60
38
|
export declare function nameSimilarity(a: string, b: string): number;
|
|
61
39
|
//# sourceMappingURL=comparators.d.ts.map
|
package/out/comparators.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"comparators.d.ts","sourceRoot":"","sources":["../lib/comparators.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"comparators.d.ts","sourceRoot":"","sources":["../lib/comparators.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAIH;;GAEG;AACH,wBAAgB,IAAI,CAAC,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,MAAM,GAAG,MAAM,CAkDjD;AAED;;;GAGG;AACH,wBAAgB,WAAW,CAC1B,CAAC,EAAE,MAAM,EACT,CAAC,EAAE,MAAM,EACT,IAAI,GAAE;IAAE,MAAM,CAAC,EAAE,MAAM,CAAC;IAAC,SAAS,CAAC,EAAE,MAAM,CAAC;IAAC,cAAc,CAAC,EAAE,MAAM,CAAA;CAAO,GACzE,MAAM,CAiBR;AAED;;;GAGG;AACH,wBAAgB,OAAO,CAAC,CAAC,EAAE,WAAW,CAAC,MAAM,CAAC,EAAE,CAAC,EAAE,WAAW,CAAC,MAAM,CAAC,GAAG,MAAM,CAW9E;AAED;;GAEG;AACH,wBAAgB,qBAAqB,CAAC,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,MAAM,GAAG,MAAM,CAOlE;AAED;;;;;;GAMG;AACH,wBAAgB,cAAc,CAAC,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,MAAM,GAAG,MAAM,CAkB3D"}
|
package/out/comparators.js
CHANGED
|
@@ -3,23 +3,12 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
* The record-linkage literature (Winkler/Census; Belin 1993) settles on the prefix-weighted Jaro
|
|
9
|
-
* comparator (Jaro-Winkler) as the default for names: it tolerates the typographical error real
|
|
10
|
-
* data is full of better than raw character-edit distance. But J-W has a documented blind spot on
|
|
11
|
-
* compound / double surnames (e.g. Hispanic `Garcia Lopez`): the second half of the compound
|
|
12
|
-
* falls outside J-W's match window, so `Lopez` vs `Garcia Lopez` scores ~0. The fix the
|
|
13
|
-
* literature prescribes is an edit-distance / token fallback for single-vs-compound pairs —
|
|
14
|
-
* implemented in {@link nameSimilarity}.
|
|
15
|
-
*
|
|
16
|
-
* These are pure similarity primitives in [0, 1]. The mapping of a similarity onto discrete
|
|
17
|
-
* Fellegi-Sunter agreement levels (and the m/u weights) is the scorer's job, not theirs.
|
|
6
|
+
* Jaro-Winkler is the default record-linkage comparator for names. {@link nameSimilarity} adds
|
|
7
|
+
* the literature's token/edit fallback because J-W scores a compound surname's second half near zero.
|
|
18
8
|
*/
|
|
19
9
|
import { distance as levenshteinDistance } from "fastest-levenshtein";
|
|
20
10
|
/**
|
|
21
|
-
* Jaro similarity in [0, 1]
|
|
22
|
-
* sliding window of `floor(max(len)/2) - 1`, discounting half-transpositions.
|
|
11
|
+
* Jaro similarity in [0, 1], where two empty strings are identical (1) and one empty is 0.
|
|
23
12
|
*/
|
|
24
13
|
export function jaro(a, b) {
|
|
25
14
|
if (a === b)
|
|
@@ -46,7 +35,6 @@ export function jaro(a, b) {
|
|
|
46
35
|
}
|
|
47
36
|
if (matches === 0)
|
|
48
37
|
return 0;
|
|
49
|
-
// Count transpositions: matched chars of `a` and `b`, in order, that disagree (halved).
|
|
50
38
|
let transpositions = 0;
|
|
51
39
|
let k = 0;
|
|
52
40
|
for (let i = 0; i < la; i++) {
|
|
@@ -64,9 +52,8 @@ export function jaro(a, b) {
|
|
|
64
52
|
return (matches / la + matches / lb + (matches - transpositions) / matches) / 3;
|
|
65
53
|
}
|
|
66
54
|
/**
|
|
67
|
-
* Jaro-Winkler similarity in [0, 1]: Jaro
|
|
68
|
-
*
|
|
69
|
-
* when `jaro` already clears `boostThreshold` (0.7), per Winkler.
|
|
55
|
+
* Jaro-Winkler similarity in [0, 1]: Jaro plus a shared-prefix bonus, boosting only when Jaro
|
|
56
|
+
* already clears the 0.7 threshold, with Winkler's standard prefix cap 4 and weight 0.1.
|
|
70
57
|
*/
|
|
71
58
|
export function jaroWinkler(a, b, opts = {}) {
|
|
72
59
|
const weight = opts.weight ?? 0.1;
|
|
@@ -83,14 +70,8 @@ export function jaroWinkler(a, b, opts = {}) {
|
|
|
83
70
|
return base + prefix * weight * (1 - base);
|
|
84
71
|
}
|
|
85
72
|
/**
|
|
86
|
-
* Jaccard similarity
|
|
87
|
-
*
|
|
88
|
-
* The set-of-tokens complement to the string comparators above. Where {@link nameSimilarity} asks how close two names
|
|
89
|
-
* LOOK, this asks how much two token bags OVERLAP — the right question for organization names and address bags, where
|
|
90
|
-
* word order carries no information and a shared rare token is worth more than character-level proximity.
|
|
91
|
-
*
|
|
92
|
-
* Either side empty scores 0 rather than 1: an empty bag agrees with nothing, and treating "no evidence" as "perfect
|
|
93
|
-
* agreement" is how a blocking pass floods with false pairs.
|
|
73
|
+
* Jaccard similarity `|a ∩ b| / |a ∪ b|` over two token sets, where an empty side scores 0 rather than 1
|
|
74
|
+
* because treating no evidence as perfect agreement floods a blocking pass with false pairs.
|
|
94
75
|
*/
|
|
95
76
|
export function jaccard(a, b) {
|
|
96
77
|
if (!a.size || !b.size)
|
|
@@ -115,15 +96,11 @@ export function levenshteinSimilarity(a, b) {
|
|
|
115
96
|
return 1 - levenshteinDistance(a, b) / longest;
|
|
116
97
|
}
|
|
117
98
|
/**
|
|
118
|
-
* Name-aware similarity in [0, 1]
|
|
119
|
-
*
|
|
120
|
-
*
|
|
121
|
-
* - If one name's tokens are a strict subset of the other's (`Lopez` ⊂ `Garcia Lopez`), that is strong partial agreement
|
|
122
|
-
* J-W misses — floor the score at 0.9.
|
|
123
|
-
* - Otherwise return the better of Jaro-Winkler and normalized edit similarity, so a single token that is a substring of
|
|
124
|
-
* a longer compound (`Garcia` vs `Garcialopez`) still scores sensibly.
|
|
99
|
+
* Name-aware similarity in [0, 1] that floors the score at 0.9 when one name's
|
|
100
|
+
* tokens are a strict subset of the other's.
|
|
125
101
|
*
|
|
126
|
-
*
|
|
102
|
+
* In other cases, it returns the better of Jaro-Winkler and normalized edit
|
|
103
|
+
* similarity without case or whitespace sensitivity.
|
|
127
104
|
*/
|
|
128
105
|
export function nameSimilarity(a, b) {
|
|
129
106
|
const x = a.trim().toLowerCase().replaceAll(/\s+/g, " ");
|
package/out/comparators.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"comparators.js","sourceRoot":"","sources":["../lib/comparators.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"comparators.js","sourceRoot":"","sources":["../lib/comparators.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAEH,OAAO,EAAE,QAAQ,IAAI,mBAAmB,EAAE,MAAM,qBAAqB,CAAA;AAErE;;GAEG;AACH,MAAM,UAAU,IAAI,CAAC,CAAS,EAAE,CAAS;IACxC,IAAI,CAAC,KAAK,CAAC;QAAE,OAAO,CAAC,CAAA;IACrB,MAAM,EAAE,GAAG,CAAC,CAAC,MAAM,CAAA;IACnB,MAAM,EAAE,GAAG,CAAC,CAAC,MAAM,CAAA;IAEnB,IAAI,EAAE,KAAK,CAAC,IAAI,EAAE,KAAK,CAAC;QAAE,OAAO,CAAC,CAAA;IAElC,MAAM,MAAM,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,EAAE,EAAE,EAAE,CAAC,GAAG,CAAC,CAAC,GAAG,CAAC,CAAC,CAAA;IAChE,MAAM,QAAQ,GAAG,IAAI,KAAK,CAAU,EAAE,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,CAAA;IACnD,MAAM,QAAQ,GAAG,IAAI,KAAK,CAAU,EAAE,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,CAAA;IAEnD,IAAI,OAAO,GAAG,CAAC,CAAA;IAEf,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,EAAE,EAAE,CAAC,EAAE,EAAE,CAAC;QAC7B,MAAM,KAAK,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,CAAC,GAAG,MAAM,CAAC,CAAA;QACrC,MAAM,GAAG,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,GAAG,MAAM,GAAG,CAAC,EAAE,EAAE,CAAC,CAAA;QAExC,KAAK,IAAI,CAAC,GAAG,KAAK,EAAE,CAAC,GAAG,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC;YAClC,IAAI,QAAQ,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC;gBAAE,SAAQ;YAC1C,QAAQ,CAAC,CAAC,CAAC,GAAG,IAAI,CAAA;YAClB,QAAQ,CAAC,CAAC,CAAC,GAAG,IAAI,CAAA;YAElB,OAAO,EAAE,CAAA;YAET,MAAK;QACN,CAAC;IACF,CAAC;IAED,IAAI,OAAO,KAAK,CAAC;QAAE,OAAO,CAAC,CAAA;IAE3B,IAAI,cAAc,GAAG,CAAC,CAAA;IACtB,IAAI,CAAC,GAAG,CAAC,CAAA;IAET,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,EAAE,EAAE,CAAC,EAAE,EAAE,CAAC;QAC7B,IAAI,CAAC,QAAQ,CAAC,CAAC,CAAC;YAAE,SAAQ;QAE1B,OAAO,CAAC,QAAQ,CAAC,CAAC,CAAC,EAAE,CAAC;YACrB,CAAC,EAAE,CAAA;QACJ,CAAC;QAED,IAAI,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC;YACnB,cAAc,EAAE,CAAA;QACjB,CAAC;QAED,CAAC,EAAE,CAAA;IACJ,CAAC;IAED,cAAc,IAAI,CAAC,CAAA;IAEnB,OAAO,CAAC,OAAO,GAAG,EAAE,GAAG,OAAO,GAAG,EAAE,GAAG,CAAC,OAAO,GAAG,cAAc,CAAC,GAAG,OAAO,CAAC,GAAG,CAAC,CAAA;AAChF,CAAC;AAED;;;GAGG;AACH,MAAM,UAAU,WAAW,CAC1B,CAAS,EACT,CAAS,EACT,IAAI,GAAqE,EAAE;IAE3E,MAAM,MAAM,GAAG,IAAI,CAAC,MAAM,IAAI,GAAG,CAAA;IACjC,MAAM,SAAS,GAAG,IAAI,CAAC,SAAS,IAAI,CAAC,CAAA;IACrC,MAAM,cAAc,GAAG,IAAI,CAAC,cAAc,IAAI,GAAG,CAAA;IAEjD,MAAM,IAAI,GAAG,IAAI,CAAC,CAAC,EAAE,CAAC,CAAC,CAAA;IAEvB,IAAI,IAAI,GAAG,cAAc;QAAE,OAAO,IAAI,CAAA;IAEtC,IAAI,MAAM,GAAG,CAAC,CAAA;IACd,MAAM,KAAK,GAAG,IAAI,CAAC,GAAG,CAAC,SAAS,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,CAAC,MAAM,CAAC,CAAA;IAErD,OAAO,MAAM,GAAG,KAAK,IAAI,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC,CAAC,MAAM,CAAC,EAAE,CAAC;QAClD,MAAM,EAAE,CAAA;IACT,CAAC;IAED,OAAO,IAAI,GAAG,MAAM,GAAG,MAAM,GAAG,CAAC,CAAC,GAAG,IAAI,CAAC,CAAA;AAC3C,CAAC;AAED;;;GAGG;AACH,MAAM,UAAU,OAAO,CAAC,CAAsB,EAAE,CAAsB;IACrE,IAAI,CAAC,CAAC,CAAC,IAAI,IAAI,CAAC,CAAC,CAAC,IAAI;QAAE,OAAO,CAAC,CAAA;IAChC,IAAI,YAAY,GAAG,CAAC,CAAA;IAEpB,KAAK,MAAM,KAAK,IAAI,CAAC,EAAE,CAAC;QACvB,IAAI,CAAC,CAAC,GAAG,CAAC,KAAK,CAAC,EAAE,CAAC;YAClB,YAAY,EAAE,CAAA;QACf,CAAC;IACF,CAAC;IAED,OAAO,YAAY,GAAG,CAAC,CAAC,CAAC,IAAI,GAAG,CAAC,CAAC,IAAI,GAAG,YAAY,CAAC,CAAA;AACvD,CAAC;AAED;;GAEG;AACH,MAAM,UAAU,qBAAqB,CAAC,CAAS,EAAE,CAAS;IACzD,IAAI,CAAC,KAAK,CAAC;QAAE,OAAO,CAAC,CAAA;IACrB,MAAM,OAAO,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC,MAAM,EAAE,CAAC,CAAC,MAAM,CAAC,CAAA;IAE5C,IAAI,OAAO,KAAK,CAAC;QAAE,OAAO,CAAC,CAAA;IAE3B,OAAO,CAAC,GAAG,mBAAmB,CAAC,CAAC,EAAE,CAAC,CAAC,GAAG,OAAO,CAAA;AAC/C,CAAC;AAED;;;;;;GAMG;AACH,MAAM,UAAU,cAAc,CAAC,CAAS,EAAE,CAAS;IAClD,MAAM,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,CAAC,WAAW,EAAE,CAAC,UAAU,CAAC,MAAM,EAAE,GAAG,CAAC,CAAA;IACxD,MAAM,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,CAAC,WAAW,EAAE,CAAC,UAAU,CAAC,MAAM,EAAE,GAAG,CAAC,CAAA;IAExD,IAAI,CAAC,CAAC,IAAI,CAAC,CAAC;QAAE,OAAO,CAAC,CAAA;IAEtB,IAAI,CAAC,KAAK,CAAC;QAAE,OAAO,CAAC,CAAA;IAErB,MAAM,EAAE,GAAG,WAAW,CAAC,CAAC,EAAE,CAAC,CAAC,CAAA;IAE5B,MAAM,OAAO,GAAG,IAAI,GAAG,CAAC,CAAC,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,CAAA;IACrC,MAAM,OAAO,GAAG,IAAI,GAAG,CAAC,CAAC,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,CAAA;IACrC,MAAM,CAAC,KAAK,EAAE,GAAG,CAAC,GAAG,OAAO,CAAC,IAAI,IAAI,OAAO,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,OAAO,EAAE,OAAO,CAAC,CAAC,CAAC,CAAC,CAAC,OAAO,EAAE,OAAO,CAAC,CAAA;IAC3F,MAAM,MAAM,GAAG,KAAK,CAAC,IAAI,GAAG,GAAG,CAAC,IAAI,IAAI,CAAC,GAAG,KAAK,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAA;IAE3E,IAAI,MAAM;QAAE,OAAO,IAAI,CAAC,GAAG,CAAC,EAAE,EAAE,GAAG,CAAC,CAAA;IAEpC,OAAO,IAAI,CAAC,GAAG,CAAC,EAAE,EAAE,qBAAqB,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,CAAA;AACjD,CAAC"}
|
package/out/distance.d.ts
CHANGED
|
@@ -2,71 +2,46 @@
|
|
|
2
2
|
* @copyright Sister Software
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
|
-
*
|
|
6
|
-
* Geographic distance as a scoring feature — the other half of geocode-first matching.
|
|
7
|
-
*
|
|
8
|
-
* Blocking uses geography to _propose_ candidates; this scores them on it. The research is explicit
|
|
9
|
-
* that an address must be matched as a SPATIAL attribute, not by string similarity (a
|
|
10
|
-
* one-character edit can be 650 m apart), and that distance measurably helps as a comparison
|
|
11
|
-
* feature. So we bucket the great-circle distance between two records' coordinates into ordered
|
|
12
|
-
* Fellegi-Sunter agreement levels (Splink's `DistanceInKMAtThresholds`): "same building" / "same
|
|
13
|
-
* block" / "same area" / far, each with its own m/u and weight.
|
|
14
|
-
*
|
|
15
|
-
* Calibrate the bucket boundaries to the geocoder's OWN error, which is heavy-tailed and density-
|
|
16
|
-
* dependent (≈38 m urban, ≈200 m rural). A weakening of this evidence by geocode quality (a
|
|
17
|
-
* shared interpolated centroid is softer than a shared rooftop point) is the documented
|
|
18
|
-
* refinement.
|
|
19
5
|
*/
|
|
20
|
-
import type
|
|
6
|
+
import { type GeoCoordinate } from "@mailwoman/spatial";
|
|
21
7
|
import type { Comparison, ComparisonLevel } from "#fellegi-sunter";
|
|
22
8
|
/**
|
|
23
|
-
*
|
|
24
|
-
*
|
|
25
|
-
* helper — not a second implementation.
|
|
9
|
+
* Computes the great-circle distance in kilometers between two `GeoCoordinate` records.
|
|
10
|
+
* It delegates to the scalar helper in `@mailwoman/spatial`.
|
|
26
11
|
*/
|
|
27
|
-
export declare const haversineKm: (a:
|
|
12
|
+
export declare const haversineKm: (a: GeoCoordinate, b: GeoCoordinate) => number;
|
|
28
13
|
/**
|
|
29
|
-
*
|
|
30
|
-
* levels
|
|
31
|
-
*
|
|
14
|
+
* Creates a comparison that buckets the great-circle distance between two records into
|
|
15
|
+
* levels ordered nearest first by `maxKm`, with the last level catching everything farther.
|
|
16
|
+
*
|
|
17
|
+
* A missing or non-finite coordinate on either side yields no evidence.
|
|
32
18
|
*/
|
|
33
19
|
export declare function distanceComparison<R>(config: {
|
|
34
20
|
name: string;
|
|
35
|
-
extract: (record: R) =>
|
|
21
|
+
extract: (record: R) => GeoCoordinate | null | undefined;
|
|
36
22
|
levels: ComparisonLevel[];
|
|
37
23
|
}): Comparison<R>;
|
|
38
24
|
/**
|
|
39
|
-
*
|
|
40
|
-
*
|
|
25
|
+
* Provides default distance levels at building, block and area scales.
|
|
26
|
+
* Their `m` and `u` values seed EM re-estimation.
|
|
41
27
|
*/
|
|
42
28
|
export declare const DEFAULT_DISTANCE_LEVELS: ComparisonLevel[];
|
|
43
29
|
/**
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
* The first matcher carried TWO spatial comparisons: canonical-address-key similarity AND great-circle distance. They
|
|
47
|
-
* double-count — an exact key match implies distance ≈ 0, so a co-located pair banked the same evidence twice, and the
|
|
48
|
-
* redundant vote is exactly what over-merges distinct providers at a shared clinic address. This folds them into one
|
|
49
|
-
* comparison:
|
|
50
|
-
*
|
|
51
|
-
* - **level 0 `same-key`** — an EXACT canonical-key match: the strongest tier, and the one the inverse-address-frequency
|
|
52
|
-
* adjustment rides ({@link withTermFrequency} on level 0), so agreement on a crowded shared key is down-weighted
|
|
53
|
-
* toward worthless while a rare one keeps full weight.
|
|
54
|
-
* - **levels 1…n** — great-circle distance buckets for pairs whose keys DIFFER, so "123 Main St" vs "123 Main Street Apt
|
|
55
|
-
* 2" that geocode to the same rooftop still earns near-agreement (the geo-first point of the whole design).
|
|
56
|
-
* - Keys differ and no usable coordinate → no evidence.
|
|
30
|
+
* Creates a single spatial comparison whose level 0 is an exact canonical-key match
|
|
31
|
+
* and whose remaining levels bucket great-circle distance for pairs with different keys.
|
|
57
32
|
*
|
|
58
|
-
*
|
|
59
|
-
*
|
|
33
|
+
* Separate key and distance comparisons count a co-located pair's evidence twice.
|
|
34
|
+
* They also over-merge distinct entities at a shared address.
|
|
60
35
|
*/
|
|
61
36
|
export declare function spatialComparison<R>(config: {
|
|
62
37
|
name: string;
|
|
63
38
|
key: (record: R) => string | null | undefined;
|
|
64
|
-
coordinate: (record: R) =>
|
|
39
|
+
coordinate: (record: R) => GeoCoordinate | null | undefined;
|
|
65
40
|
levels: ComparisonLevel[];
|
|
66
41
|
}): Comparison<R>;
|
|
67
42
|
/**
|
|
68
|
-
*
|
|
69
|
-
*
|
|
43
|
+
* Provides default levels for {@link spatialComparison}: an exact same-key tier followed
|
|
44
|
+
* by the building, block, area and far distance buckets, with seed `m` and `u` values.
|
|
70
45
|
*/
|
|
71
46
|
export declare const DEFAULT_SPATIAL_LEVELS: ComparisonLevel[];
|
|
72
47
|
//# sourceMappingURL=distance.d.ts.map
|
package/out/distance.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"distance.d.ts","sourceRoot":"","sources":["../lib/distance.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"distance.d.ts","sourceRoot":"","sources":["../lib/distance.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAEH,OAAO,EAAgC,KAAK,aAAa,EAAE,MAAM,oBAAoB,CAAA;AAErF,OAAO,KAAK,EAAE,UAAU,EAAE,eAAe,EAAE,MAAM,iBAAiB,CAAA;AAElE;;;GAGG;AACH,eAAO,MAAM,WAAW,MAAO,aAAa,KAAK,aAAa,KAAG,MACD,CAAA;AAEhE;;;;;GAKG;AACH,wBAAgB,kBAAkB,CAAC,CAAC,EAAE,MAAM,EAAE;IAC7C,IAAI,EAAE,MAAM,CAAA;IACZ,OAAO,EAAE,CAAC,MAAM,EAAE,CAAC,KAAK,aAAa,GAAG,IAAI,GAAG,SAAS,CAAA;IACxD,MAAM,EAAE,eAAe,EAAE,CAAA;CACzB,GAAG,UAAU,CAAC,CAAC,CAAC,CAsBhB;AAED;;;GAGG;AACH,eAAO,MAAM,uBAAuB,EAAE,eAAe,EAKpD,CAAA;AAED;;;;;;GAMG;AACH,wBAAgB,iBAAiB,CAAC,CAAC,EAAE,MAAM,EAAE;IAC5C,IAAI,EAAE,MAAM,CAAA;IACZ,GAAG,EAAE,CAAC,MAAM,EAAE,CAAC,KAAK,MAAM,GAAG,IAAI,GAAG,SAAS,CAAA;IAC7C,UAAU,EAAE,CAAC,MAAM,EAAE,CAAC,KAAK,aAAa,GAAG,IAAI,GAAG,SAAS,CAAA;IAC3D,MAAM,EAAE,eAAe,EAAE,CAAA;CACzB,GAAG,UAAU,CAAC,CAAC,CAAC,CA2BhB;AAED;;;GAGG;AACH,eAAO,MAAM,sBAAsB,EAAE,eAAe,EAMnD,CAAA"}
|
package/out/distance.js
CHANGED
|
@@ -2,32 +2,18 @@
|
|
|
2
2
|
* @copyright Sister Software
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
|
-
*
|
|
6
|
-
* Geographic distance as a scoring feature — the other half of geocode-first matching.
|
|
7
|
-
*
|
|
8
|
-
* Blocking uses geography to _propose_ candidates; this scores them on it. The research is explicit
|
|
9
|
-
* that an address must be matched as a SPATIAL attribute, not by string similarity (a
|
|
10
|
-
* one-character edit can be 650 m apart), and that distance measurably helps as a comparison
|
|
11
|
-
* feature. So we bucket the great-circle distance between two records' coordinates into ordered
|
|
12
|
-
* Fellegi-Sunter agreement levels (Splink's `DistanceInKMAtThresholds`): "same building" / "same
|
|
13
|
-
* block" / "same area" / far, each with its own m/u and weight.
|
|
14
|
-
*
|
|
15
|
-
* Calibrate the bucket boundaries to the geocoder's OWN error, which is heavy-tailed and density-
|
|
16
|
-
* dependent (≈38 m urban, ≈200 m rural). A weakening of this evidence by geocode quality (a
|
|
17
|
-
* shared interpolated centroid is softer than a shared rooftop point) is the documented
|
|
18
|
-
* refinement.
|
|
19
5
|
*/
|
|
20
6
|
import { haversineKm as greatCircleKm } from "@mailwoman/spatial";
|
|
21
7
|
/**
|
|
22
|
-
*
|
|
23
|
-
*
|
|
24
|
-
* helper — not a second implementation.
|
|
8
|
+
* Computes the great-circle distance in kilometers between two `GeoCoordinate` records.
|
|
9
|
+
* It delegates to the scalar helper in `@mailwoman/spatial`.
|
|
25
10
|
*/
|
|
26
11
|
export const haversineKm = (a, b) => greatCircleKm(a.latitude, a.longitude, b.latitude, b.longitude);
|
|
27
12
|
/**
|
|
28
|
-
*
|
|
29
|
-
* levels
|
|
30
|
-
*
|
|
13
|
+
* Creates a comparison that buckets the great-circle distance between two records into
|
|
14
|
+
* levels ordered nearest first by `maxKm`, with the last level catching everything farther.
|
|
15
|
+
*
|
|
16
|
+
* A missing or non-finite coordinate on either side yields no evidence.
|
|
31
17
|
*/
|
|
32
18
|
export function distanceComparison(config) {
|
|
33
19
|
const valid = (c) => !!c && Number.isFinite(c.latitude) && Number.isFinite(c.longitude);
|
|
@@ -49,8 +35,8 @@ export function distanceComparison(config) {
|
|
|
49
35
|
};
|
|
50
36
|
}
|
|
51
37
|
/**
|
|
52
|
-
*
|
|
53
|
-
*
|
|
38
|
+
* Provides default distance levels at building, block and area scales.
|
|
39
|
+
* Their `m` and `u` values seed EM re-estimation.
|
|
54
40
|
*/
|
|
55
41
|
export const DEFAULT_DISTANCE_LEVELS = [
|
|
56
42
|
{ label: "same-building", maxKm: 0.05, m: 0.7, u: 0.001 },
|
|
@@ -59,22 +45,11 @@ export const DEFAULT_DISTANCE_LEVELS = [
|
|
|
59
45
|
{ label: "far", m: 0.02, u: 0.779 },
|
|
60
46
|
];
|
|
61
47
|
/**
|
|
62
|
-
*
|
|
48
|
+
* Creates a single spatial comparison whose level 0 is an exact canonical-key match
|
|
49
|
+
* and whose remaining levels bucket great-circle distance for pairs with different keys.
|
|
63
50
|
*
|
|
64
|
-
*
|
|
65
|
-
*
|
|
66
|
-
* redundant vote is exactly what over-merges distinct providers at a shared clinic address. This folds them into one
|
|
67
|
-
* comparison:
|
|
68
|
-
*
|
|
69
|
-
* - **level 0 `same-key`** — an EXACT canonical-key match: the strongest tier, and the one the inverse-address-frequency
|
|
70
|
-
* adjustment rides ({@link withTermFrequency} on level 0), so agreement on a crowded shared key is down-weighted
|
|
71
|
-
* toward worthless while a rare one keeps full weight.
|
|
72
|
-
* - **levels 1…n** — great-circle distance buckets for pairs whose keys DIFFER, so "123 Main St" vs "123 Main Street Apt
|
|
73
|
-
* 2" that geocode to the same rooftop still earns near-agreement (the geo-first point of the whole design).
|
|
74
|
-
* - Keys differ and no usable coordinate → no evidence.
|
|
75
|
-
*
|
|
76
|
-
* Exactly one spatial vote, no redundancy. Pass {@link DEFAULT_SPATIAL_LEVELS} or your own; index 0 must be the
|
|
77
|
-
* exact-key tier, indices 1…n the distance buckets nearest → far by `maxKm` (last = `far`).
|
|
51
|
+
* Separate key and distance comparisons count a co-located pair's evidence twice.
|
|
52
|
+
* They also over-merge distinct entities at a shared address.
|
|
78
53
|
*/
|
|
79
54
|
export function spatialComparison(config) {
|
|
80
55
|
const valid = (c) => !!c && Number.isFinite(c.latitude) && Number.isFinite(c.longitude);
|
|
@@ -85,11 +60,11 @@ export function spatialComparison(config) {
|
|
|
85
60
|
const ka = config.key(a);
|
|
86
61
|
const kb = config.key(b);
|
|
87
62
|
if (ka && kb && ka.trim() && ka === kb)
|
|
88
|
-
return 0;
|
|
63
|
+
return 0;
|
|
89
64
|
const ca = config.coordinate(a);
|
|
90
65
|
const cb = config.coordinate(b);
|
|
91
66
|
if (!valid(ca) || !valid(cb))
|
|
92
|
-
return -1;
|
|
67
|
+
return -1;
|
|
93
68
|
const km = haversineKm(ca, cb);
|
|
94
69
|
for (let i = 1; i < config.levels.length; i++) {
|
|
95
70
|
if (km <= (config.levels[i].maxKm ?? Infinity))
|
|
@@ -100,8 +75,8 @@ export function spatialComparison(config) {
|
|
|
100
75
|
};
|
|
101
76
|
}
|
|
102
77
|
/**
|
|
103
|
-
*
|
|
104
|
-
*
|
|
78
|
+
* Provides default levels for {@link spatialComparison}: an exact same-key tier followed
|
|
79
|
+
* by the building, block, area and far distance buckets, with seed `m` and `u` values.
|
|
105
80
|
*/
|
|
106
81
|
export const DEFAULT_SPATIAL_LEVELS = [
|
|
107
82
|
{ label: "same-key", m: 0.85, u: 0.01 },
|