custmatch 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- custmatch/__init__.py +1 -0
- custmatch/__main__.py +5 -0
- custmatch/blocking.py +480 -0
- custmatch/cli.py +124 -0
- custmatch/core.py +652 -0
- custmatch/countries.py +82 -0
- custmatch/country.py +393 -0
- custmatch/data/nicknames_au_uk.csv +91 -0
- custmatch/explain.py +214 -0
- custmatch/golden.py +236 -0
- custmatch/households.py +236 -0
- custmatch/identity.py +187 -0
- custmatch/incremental.py +305 -0
- custmatch/io.py +282 -0
- custmatch/labels.py +132 -0
- custmatch/matchweight.py +217 -0
- custmatch/models.py +130 -0
- custmatch/monitor.py +101 -0
- custmatch/pipeline.py +527 -0
- custmatch/placeholders.py +112 -0
- custmatch/schema.py +58 -0
- custmatch/stewardship.py +306 -0
- custmatch/suspect.py +111 -0
- custmatch-0.1.0.dist-info/METADATA +277 -0
- custmatch-0.1.0.dist-info/RECORD +30 -0
- custmatch-0.1.0.dist-info/WHEEL +5 -0
- custmatch-0.1.0.dist-info/entry_points.txt +2 -0
- custmatch-0.1.0.dist-info/licenses/LICENSE +202 -0
- custmatch-0.1.0.dist-info/licenses/NOTICE +29 -0
- custmatch-0.1.0.dist-info/top_level.txt +1 -0
custmatch/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Customer matching: the pipeline measured in docs/report/, as a tool. See docs/guide.md."""
|
custmatch/__main__.py
ADDED
custmatch/blocking.py
ADDED
|
@@ -0,0 +1,480 @@
|
|
|
1
|
+
"""Blocking for the tool: rules, learned rule sets, split oversized blocks, SQL self-join, diagnostics.
|
|
2
|
+
|
|
3
|
+
What the three leading projects do and custmatch's first blocker did not (docs/guide.md):
|
|
4
|
+
|
|
5
|
+
learned rules dedupe: pick the cheapest set of blocking rules that still covers a target
|
|
6
|
+
share (default 95%) of the labelled matches - a greedy weighted set cover
|
|
7
|
+
over a library of rules - instead of fixed hand-picked keys
|
|
8
|
+
split, not drop Zingg: a block larger than max_block is refined with the next predicate
|
|
9
|
+
(first letters of the first name, birth year, street, ...) until it fits,
|
|
10
|
+
and dropped only if it still cannot be split - custmatch dropped it outright,
|
|
11
|
+
which lost the 173-record London block in Splink's fake_1000
|
|
12
|
+
SQL self-join dedupe / Splink: pairs come from a DuckDB self-join on (rule, key) with
|
|
13
|
+
DISTINCT, not Python loops over every block
|
|
14
|
+
diagnostics Splink: pairs contributed by each rule, its largest blocks, and what was split
|
|
15
|
+
or dropped - blocking_report.json in the run's output
|
|
16
|
+
|
|
17
|
+
core.block (the fixed keys the report measured) is unchanged; this module is used by the tool.
|
|
18
|
+
A rule is a tuple of predicate names; records agree on a rule when every predicate gives the
|
|
19
|
+
same non-empty key.
|
|
20
|
+
"""
|
|
21
|
+
import numpy as np
|
|
22
|
+
import pandas as pd
|
|
23
|
+
|
|
24
|
+
# predicate name -> (column, transform). Transforms map a cleaned string column to keys.
|
|
25
|
+
_TRANSFORMS = {
|
|
26
|
+
"exact": lambda s: s,
|
|
27
|
+
"first3": lambda s: s.str[:3].where(s.str.len() >= 3, ""),
|
|
28
|
+
"first1": lambda s: s.str[:1],
|
|
29
|
+
"year": lambda s: s.str.extract(r"(\d{4})", expand=False).fillna(""),
|
|
30
|
+
"yearmonth": None, # set below: needs date parsing
|
|
31
|
+
"soundex": None,
|
|
32
|
+
}
|
|
33
|
+
NAME_FIELDS = ("first_name", "last_name")
|
|
34
|
+
FIXED_RULES = [("last_name:exact", "first_name:first3"), ("last_name:exact", "addr_street:exact"),
|
|
35
|
+
("addr_number:exact", "addr_street:exact"), ("first_name:exact", "addr_street:exact"),
|
|
36
|
+
("email:exact",), ("phone:exact",)]
|
|
37
|
+
# refinements tried, in order, to split an oversized block
|
|
38
|
+
REFINEMENTS = ["first_name:first3", "date_of_birth:year", "last_name:first3", "addr_street:exact",
|
|
39
|
+
"first_name:exact", "date_of_birth:exact", "addr_number:exact"]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
_SOUNDEX = {**dict.fromkeys("bfpv", "1"), **dict.fromkeys("cgjkqsxz", "2"), **dict.fromkeys("dt", "3"),
|
|
43
|
+
"l": "4", **dict.fromkeys("mn", "5"), "r": "6"}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def soundex(word):
|
|
47
|
+
"""American Soundex: the first letter, then up to three digits for the consonant sounds that
|
|
48
|
+
follow (Robert, Rupert -> R163). Letters with one code next to each other, or separated
|
|
49
|
+
only by h or w, count once; a vowel, space or hyphen between them counts them twice - as
|
|
50
|
+
jellyfish.soundex, which this replaces, does."""
|
|
51
|
+
w = word.lower()
|
|
52
|
+
first = next((n for n, c in enumerate(w) if "a" <= c <= "z"), None)
|
|
53
|
+
if first is None:
|
|
54
|
+
return ""
|
|
55
|
+
out, last = [], _SOUNDEX.get(w[first], "")
|
|
56
|
+
for c in w[first + 1:]:
|
|
57
|
+
code = _SOUNDEX.get(c, "") # vowels, spaces and hyphens separate: code ""
|
|
58
|
+
if code and code != last:
|
|
59
|
+
out.append(code)
|
|
60
|
+
if c not in "hw": # h and w do not separate two equal codes
|
|
61
|
+
last = code
|
|
62
|
+
return (w[first].upper() + "".join(out) + "000")[:4]
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _soundex(s):
|
|
66
|
+
return s.map(lambda v: soundex(v) if v and v[0].isalpha() else "")
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _yearmonth(s):
|
|
70
|
+
from .core import parse_dates
|
|
71
|
+
d = parse_dates(s.to_numpy())
|
|
72
|
+
return pd.Series([f"{y:04d}{m:02d}" if y >= 0 else "" for y, m, _ in d], index=s.index)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class Keys:
|
|
76
|
+
"""Lazily computed predicate keys per record (cleaned strings, "" = no key)."""
|
|
77
|
+
|
|
78
|
+
def __init__(self, df, profiles=None):
|
|
79
|
+
self.df, self.cache = df, {}
|
|
80
|
+
self.junk = {}
|
|
81
|
+
for f in ("email", "phone"):
|
|
82
|
+
if profiles and f in profiles:
|
|
83
|
+
self.junk[f] = df[f].map(lambda v: profiles[f].get(v, ("personal",))[0] == "junk").to_numpy()
|
|
84
|
+
|
|
85
|
+
def get(self, pred):
|
|
86
|
+
if pred not in self.cache:
|
|
87
|
+
col, how = pred.split(":")
|
|
88
|
+
if col not in self.df.columns:
|
|
89
|
+
self.cache[pred] = np.full(len(self.df), "", dtype=object)
|
|
90
|
+
else:
|
|
91
|
+
s = self.df[col].astype(str)
|
|
92
|
+
if how == "soundex":
|
|
93
|
+
k = _soundex(s)
|
|
94
|
+
elif how == "yearmonth":
|
|
95
|
+
k = _yearmonth(s)
|
|
96
|
+
else:
|
|
97
|
+
k = _TRANSFORMS[how](s)
|
|
98
|
+
k = k.to_numpy(dtype=object)
|
|
99
|
+
if col in self.junk:
|
|
100
|
+
k = np.where(self.junk[col], "", k)
|
|
101
|
+
self.cache[pred] = k
|
|
102
|
+
return self.cache[pred]
|
|
103
|
+
|
|
104
|
+
def codes(self, pred):
|
|
105
|
+
"""Integer code per record for a predicate; -1 where it has no key."""
|
|
106
|
+
ck = ("codes", pred)
|
|
107
|
+
if ck not in self.cache:
|
|
108
|
+
k = self.get(pred)
|
|
109
|
+
c, _ = pd.factorize(k)
|
|
110
|
+
self.cache[ck] = np.where(k == "", -1, c).astype(np.int64)
|
|
111
|
+
return self.cache[ck]
|
|
112
|
+
|
|
113
|
+
def rule(self, rule):
|
|
114
|
+
"""Integer key per record for a rule (a conjunction of predicates); -1 where any
|
|
115
|
+
predicate has no key. Integer codes, not joined strings: at 1M records joining strings
|
|
116
|
+
made blocking the slowest stage."""
|
|
117
|
+
parts = [self.codes(p) for p in rule]
|
|
118
|
+
empty = np.zeros(len(self.df), bool)
|
|
119
|
+
for c in parts:
|
|
120
|
+
empty |= c < 0
|
|
121
|
+
if len(parts) == 1:
|
|
122
|
+
key = parts[0]
|
|
123
|
+
else:
|
|
124
|
+
key, _ = pd.MultiIndex.from_arrays(parts).factorize()
|
|
125
|
+
key = key.astype(np.int64)
|
|
126
|
+
return np.where(empty, -1, key)
|
|
127
|
+
|
|
128
|
+
def label(self, rule, i):
|
|
129
|
+
"""Readable key of record i under a rule, for diagnostics."""
|
|
130
|
+
return " | ".join(str(self.get(p)[i]) for p in rule)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def rule_library(df, cfg=None):
|
|
134
|
+
"""Candidate rules over the fields this data actually has."""
|
|
135
|
+
from .schema import extra_fields
|
|
136
|
+
# country.py's derived name columns are blocked on through country.blocking_keys, not as rules
|
|
137
|
+
derived = {"name_script", "first_roots", "name_skeleton", "country", "updated_at"}
|
|
138
|
+
have = [c for c in df.columns if c not in ("source", "record_id") and c not in derived
|
|
139
|
+
and (df[c] != "").sum() > 1]
|
|
140
|
+
single = []
|
|
141
|
+
for c in have:
|
|
142
|
+
single.append(f"{c}:exact")
|
|
143
|
+
if c in NAME_FIELDS or c in ("addr_street",) or c in {n for n, _ in extra_fields(cfg or {})}:
|
|
144
|
+
single.append(f"{c}:first3")
|
|
145
|
+
if c in NAME_FIELDS:
|
|
146
|
+
single.append(f"{c}:soundex")
|
|
147
|
+
if c == "date_of_birth":
|
|
148
|
+
single += ["date_of_birth:year", "date_of_birth:yearmonth"]
|
|
149
|
+
rules = [(p,) for p in single]
|
|
150
|
+
strong = [p for p in single if p.endswith(":exact") or p.endswith(":soundex")]
|
|
151
|
+
for a in range(len(single)):
|
|
152
|
+
for b in range(a + 1, len(single)):
|
|
153
|
+
pa, pb = single[a], single[b]
|
|
154
|
+
if pa.split(":")[0] == pb.split(":")[0]:
|
|
155
|
+
continue
|
|
156
|
+
if pa in strong or pb in strong:
|
|
157
|
+
rules.append((pa, pb))
|
|
158
|
+
rules += fixed_rules(cfg, df)
|
|
159
|
+
return list(dict.fromkeys(rules))
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def fixed_rules(cfg=None, df=None):
|
|
163
|
+
"""The fixed keys as rules. A predicate on a field the data never fills is dropped from its
|
|
164
|
+
rule, so (last_name, addr_street) becomes (last_name) where there is no address - what the
|
|
165
|
+
report's core.block did implicitly (an empty street was a shared key), and the recall it gave
|
|
166
|
+
on historical_50k (87.0 vs 84.0 without) depends on it."""
|
|
167
|
+
from .schema import extra_keys
|
|
168
|
+
rules = [tuple(r) for r in FIXED_RULES] + [tuple(c if ":" in c else f"{c}:exact" for c in k) for k in extra_keys(cfg or {})]
|
|
169
|
+
if df is None:
|
|
170
|
+
return rules
|
|
171
|
+
filled = {c for c in df.columns if (df[c] != "").any()}
|
|
172
|
+
out = []
|
|
173
|
+
for r in rules:
|
|
174
|
+
kept = tuple(p for p in r if p.split(":")[0] in filled)
|
|
175
|
+
if kept:
|
|
176
|
+
out.append(kept)
|
|
177
|
+
return list(dict.fromkeys(out))
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _n2(sizes):
|
|
181
|
+
sizes = np.asarray(sizes, dtype=np.float64)
|
|
182
|
+
return float(np.sum(sizes * (sizes - 1) / 2))
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def learn_rules(df, keys, pos_pairs, cfg=None, target=0.95, max_block=60, max_rules=8,
|
|
186
|
+
sample=50000, seed=0):
|
|
187
|
+
"""Greedy weighted set cover: repeatedly add the rule covering the most still-uncovered
|
|
188
|
+
labelled matches per candidate pair it would generate, until `target` of the labelled
|
|
189
|
+
matches that any rule can cover are covered. Returns (rules, report).
|
|
190
|
+
|
|
191
|
+
Like dedupe, it learns on a sample: the records in labelled pairs plus up to `sample`
|
|
192
|
+
random others, with block sizes scaled up to the full file for the cost - evaluating the
|
|
193
|
+
whole rule library on a million records took 143 s."""
|
|
194
|
+
lib = rule_library(df, cfg)
|
|
195
|
+
n = len(df)
|
|
196
|
+
rng = np.random.default_rng(seed)
|
|
197
|
+
in_labels = np.unique(pos_pairs.ravel())
|
|
198
|
+
others = np.setdiff1d(np.arange(n), in_labels)
|
|
199
|
+
extra = rng.choice(others, min(sample, len(others)), replace=False) if len(others) else others
|
|
200
|
+
idx = np.sort(np.concatenate([in_labels, extra]))
|
|
201
|
+
sub = Keys(df.iloc[idx].reset_index(drop=True), None)
|
|
202
|
+
sub.junk = {f: v[idx] for f, v in keys.junk.items()}
|
|
203
|
+
pos_local = np.searchsorted(idx, pos_pairs)
|
|
204
|
+
i, j = pos_local[:, 0], pos_local[:, 1]
|
|
205
|
+
scale = n / len(idx)
|
|
206
|
+
covers, cost = {}, {}
|
|
207
|
+
for r in lib:
|
|
208
|
+
k = sub.rule(r)
|
|
209
|
+
covers[r] = (k[i] == k[j]) & (k[i] >= 0)
|
|
210
|
+
sizes = np.bincount(k[k >= 0]) * scale if (k >= 0).any() else np.zeros(0)
|
|
211
|
+
sizes = sizes[sizes > 0]
|
|
212
|
+
# oversized blocks will be split; charge them roughly as pieces of max_block
|
|
213
|
+
small, big = sizes[sizes <= max_block], sizes[sizes > max_block]
|
|
214
|
+
cost[r] = _n2(small) + float(np.sum(np.ceil(big / max_block))) * _n2([max_block])
|
|
215
|
+
coverable = np.any([covers[r] for r in lib], axis=0) if lib else np.zeros(len(i), bool)
|
|
216
|
+
need = target * coverable.sum()
|
|
217
|
+
chosen, covered = [], np.zeros(len(i), bool)
|
|
218
|
+
while covered.sum() < need and len(chosen) < max_rules:
|
|
219
|
+
best = max(lib, key=lambda r: ((covers[r] & ~covered).sum() / (cost[r] + 1.0), -cost[r]))
|
|
220
|
+
if not (covers[best] & ~covered).any():
|
|
221
|
+
break
|
|
222
|
+
chosen.append(best)
|
|
223
|
+
covered |= covers[best]
|
|
224
|
+
report = {"labelled_matches": int(len(i)), "coverable_by_library": int(coverable.sum()),
|
|
225
|
+
"covered": int(covered.sum()), "target": target, "library_size": len(lib),
|
|
226
|
+
"rules": [{"rule": list(r), "covers": int(covers[r].sum()), "est_pairs": int(cost[r])}
|
|
227
|
+
for r in chosen]}
|
|
228
|
+
return chosen, report
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _split(keys, idx_block, rule, max_block):
|
|
232
|
+
"""Refine an oversized block with the next predicates; returns (list of record-index arrays,
|
|
233
|
+
number of pieces dropped because nothing was left to split them on)."""
|
|
234
|
+
if len(idx_block) <= max_block:
|
|
235
|
+
return [idx_block], 0
|
|
236
|
+
for p in REFINEMENTS:
|
|
237
|
+
if p in rule:
|
|
238
|
+
continue
|
|
239
|
+
sub = keys.codes(p)[idx_block]
|
|
240
|
+
if (sub < 0).all() or len(np.unique(sub)) == 1:
|
|
241
|
+
continue
|
|
242
|
+
out, dropped = [], 0
|
|
243
|
+
# records with no value for this refinement (a first name shorter than three letters,
|
|
244
|
+
# no birth year) stay together as their own piece rather than being lost
|
|
245
|
+
for v in np.unique(sub):
|
|
246
|
+
part = idx_block[sub == v]
|
|
247
|
+
if len(part) < 2:
|
|
248
|
+
continue
|
|
249
|
+
o, d = _split(keys, part, rule + (p,), max_block)
|
|
250
|
+
out += o
|
|
251
|
+
dropped += d
|
|
252
|
+
return out, dropped
|
|
253
|
+
return [], 1 # nothing left to split on: drop it as a hub
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def generate(df, keys, rules, max_block=60):
|
|
257
|
+
"""Candidate pairs for the rules (splitting oversized blocks), plus per-rule diagnostics.
|
|
258
|
+
Returns (pairs int32 array sorted, report)."""
|
|
259
|
+
rows, report = [], []
|
|
260
|
+
for n, r in enumerate(rules):
|
|
261
|
+
k = keys.rule(r)
|
|
262
|
+
has = np.flatnonzero(k >= 0)
|
|
263
|
+
kk = k[has]
|
|
264
|
+
sizes = np.bincount(kk) if len(kk) else np.zeros(0, int)
|
|
265
|
+
big_codes = np.flatnonzero(sizes > max_block)
|
|
266
|
+
small = sizes[kk] <= max_block
|
|
267
|
+
rk, ri = kk[small], has[small]
|
|
268
|
+
next_key = int(sizes.size)
|
|
269
|
+
split_blocks, dropped, extra_k, extra_i = 0, 0, [], []
|
|
270
|
+
for bc in big_codes:
|
|
271
|
+
parts, d = _split(keys, has[kk == bc], tuple(r), max_block)
|
|
272
|
+
split_blocks += 1
|
|
273
|
+
dropped += d
|
|
274
|
+
for pidx in parts:
|
|
275
|
+
extra_k.append(np.full(len(pidx), next_key, np.int64))
|
|
276
|
+
extra_i.append(pidx)
|
|
277
|
+
next_key += 1
|
|
278
|
+
rows.append(pd.DataFrame({"rule": n, "key": np.concatenate([rk] + extra_k) if extra_k else rk,
|
|
279
|
+
"rid": np.concatenate([ri] + extra_i) if extra_i else ri}))
|
|
280
|
+
order = np.argsort(-sizes)[:5]
|
|
281
|
+
report.append({"rule": list(r), "records_with_key": int(len(has)), "blocks": int((sizes > 0).sum()),
|
|
282
|
+
"largest_blocks": [{"key": keys.label(r, int(has[kk == c][0])), "size": int(sizes[c])}
|
|
283
|
+
for c in order if sizes[c] > 0],
|
|
284
|
+
"oversized_blocks_split": split_blocks, "blocks_dropped_as_hubs": dropped})
|
|
285
|
+
table = pd.concat(rows, ignore_index=True) if rows else pd.DataFrame({"rule": [], "key": [], "rid": []})
|
|
286
|
+
per_rule = _block_pairs(table.rule.to_numpy(np.int64), table.key.to_numpy(np.int64), table.rid.to_numpy(np.int64))
|
|
287
|
+
first = per_rule.groupby(["i", "j"]).rule.min().reset_index()
|
|
288
|
+
counts = per_rule.rule.value_counts()
|
|
289
|
+
new = first.rule.value_counts()
|
|
290
|
+
for n, entry in enumerate(report):
|
|
291
|
+
entry["pairs"] = int(counts.get(n, 0))
|
|
292
|
+
entry["new_pairs"] = int(new.get(n, 0)) # not already produced by an earlier rule
|
|
293
|
+
pairs = first.sort_values(["i", "j"])[["i", "j"]].to_numpy(dtype=np.int32) if len(first) \
|
|
294
|
+
else np.zeros((0, 2), np.int32)
|
|
295
|
+
return pairs, {"rules": report, "candidate_pairs": int(len(pairs)), "max_block": max_block}
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def _block_pairs(rule, key, rid):
|
|
299
|
+
"""Every pair of records sharing a (rule, key) block: a DataFrame of rule, i < j. Sorted by
|
|
300
|
+
block, a record pairs with the k-th record after it while that one is still in its block;
|
|
301
|
+
blocks are capped at max_block, so k stays small and each step is one vectorised pass."""
|
|
302
|
+
if not len(rid):
|
|
303
|
+
return pd.DataFrame({"rule": np.zeros(0, np.int64), "i": np.zeros(0, np.int64), "j": np.zeros(0, np.int64)})
|
|
304
|
+
order = np.lexsort((rid, key, rule))
|
|
305
|
+
rule, key, rid = rule[order], key[order], rid[order]
|
|
306
|
+
new = np.r_[True, (rule[1:] != rule[:-1]) | (key[1:] != key[:-1])]
|
|
307
|
+
starts = np.flatnonzero(new)
|
|
308
|
+
end = np.repeat(np.r_[starts[1:], len(rid)], np.diff(np.r_[starts, len(rid)])) # block end per row
|
|
309
|
+
pos = np.arange(len(rid))
|
|
310
|
+
out_r, out_i, out_j = [], [], []
|
|
311
|
+
for k in range(1, int((end - pos).max())):
|
|
312
|
+
a = pos[pos + k < end]
|
|
313
|
+
b = a + k
|
|
314
|
+
out_r.append(rule[a]); out_i.append(np.minimum(rid[a], rid[b])); out_j.append(np.maximum(rid[a], rid[b]))
|
|
315
|
+
if not out_r:
|
|
316
|
+
return pd.DataFrame({"rule": np.zeros(0, np.int64), "i": np.zeros(0, np.int64), "j": np.zeros(0, np.int64)})
|
|
317
|
+
return pd.DataFrame({"rule": np.concatenate(out_r), "i": np.concatenate(out_i), "j": np.concatenate(out_j)})
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
def country_pairs(df, profiles, max_block):
|
|
321
|
+
"""Pairs from country.blocking_keys (first-name root + DOB or surname, email user name, name
|
|
322
|
+
skeleton), blocks larger than max_block skipped - the same keys core.block uses."""
|
|
323
|
+
from collections import defaultdict
|
|
324
|
+
from . import country
|
|
325
|
+
groups = defaultdict(list)
|
|
326
|
+
for k, i in country.blocking_keys(df, profiles):
|
|
327
|
+
groups[k].append(i)
|
|
328
|
+
out = set()
|
|
329
|
+
for idx in groups.values():
|
|
330
|
+
if 2 <= len(idx) <= max_block:
|
|
331
|
+
idx = sorted(set(idx))
|
|
332
|
+
out.update((a, b) for x, a in enumerate(idx) for b in idx[x + 1:])
|
|
333
|
+
return np.array(sorted(out), dtype=np.int32).reshape(-1, 2)
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def candidates_for(df, profiles, cfg, labelled_pos=None, rules=None):
|
|
337
|
+
"""The tool's candidate pairs. Returns (pairs, hubs_dropped, report, rules_used).
|
|
338
|
+
|
|
339
|
+
cfg "blocking": absent -> the fixed keys the report measured (core.block, blocks over
|
|
340
|
+
max_block dropped); "fixed" -> the same kind of keys as rules, oversized blocks split;
|
|
341
|
+
"learned" -> rules learned from labelled_pos (falls back to fixed without labels).
|
|
342
|
+
rules: reuse a saved rule set (incremental matching)."""
|
|
343
|
+
from .core import block
|
|
344
|
+
from .schema import extra_keys, max_block as cfg_max_block
|
|
345
|
+
mode = cfg.get("blocking")
|
|
346
|
+
mb = cfg_max_block(cfg) or 60
|
|
347
|
+
if rules is None and not mode:
|
|
348
|
+
skipped, by_key = [], []
|
|
349
|
+
pairs, hubs = block(df, profiles, extra_keys=extra_keys(cfg), max_block=cfg_max_block(cfg),
|
|
350
|
+
lsh=cfg.get("lsh_blocking"), skipped=skipped, by_key=by_key)
|
|
351
|
+
# records never compared with anyone because their only blocks were skipped - a record
|
|
352
|
+
# whose broad block (postcode alone) was skipped but whose narrow one was used is fine
|
|
353
|
+
compared = np.zeros(len(df), bool)
|
|
354
|
+
compared[pairs.ravel()] = True
|
|
355
|
+
records = len({i for _, idx in skipped for i in idx if not compared[i]})
|
|
356
|
+
top = sorted(skipped, key=lambda s: -len(s[1]))[:5]
|
|
357
|
+
return pairs, hubs, {"mode": "core", "candidate_pairs": int(len(pairs)), "hubs_dropped": int(hubs),
|
|
358
|
+
"skipped_blocks_with_values": len(skipped), "records_only_in_skipped_blocks": records,
|
|
359
|
+
"largest_skipped_blocks": [{"key": " / ".join(str(v) for v in k), "records": len(idx)}
|
|
360
|
+
for k, idx in top],
|
|
361
|
+
"pairs_by_key": by_key, "_skipped": skipped}, None
|
|
362
|
+
if cfg.get("lsh_blocking"):
|
|
363
|
+
from .io import ConfigError
|
|
364
|
+
raise ConfigError('"lsh_blocking" works with the default blocking only; remove "blocking": '
|
|
365
|
+
f'"{mode}" or "lsh_blocking"')
|
|
366
|
+
keys = Keys(df, profiles)
|
|
367
|
+
learn = None
|
|
368
|
+
if rules is None:
|
|
369
|
+
if mode == "learned" and labelled_pos is not None and len(labelled_pos):
|
|
370
|
+
rules, learn = learn_rules(df, keys, labelled_pos, cfg, target=cfg.get("blocking_recall", 0.95),
|
|
371
|
+
max_block=mb)
|
|
372
|
+
else:
|
|
373
|
+
rules = fixed_rules(cfg, df)
|
|
374
|
+
pairs, report = generate(df, keys, rules, max_block=mb)
|
|
375
|
+
extra = country_pairs(df, profiles, mb) # country.py keys: nickname + DOB, email user...
|
|
376
|
+
if len(extra):
|
|
377
|
+
allp = np.vstack([pairs, extra]) if len(pairs) else extra
|
|
378
|
+
pairs = np.unique(allp, axis=0)
|
|
379
|
+
report.update(mode=mode or "saved", learned=learn, country_key_pairs=int(len(extra)))
|
|
380
|
+
hubs = sum(r["blocks_dropped_as_hubs"] for r in report["rules"])
|
|
381
|
+
return pairs, hubs, report, [list(r) for r in rules]
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
# ------------------------------------------------------------------ similarity (LSH) blocking
|
|
385
|
+
def lsh_keys(df, spec, seed=0):
|
|
386
|
+
"""Blocking keys from MinHash locality-sensitive hashing of letter pairs - the idea behind
|
|
387
|
+
anonlink's blocklib, on plain text. Records whose `fields` (joined) share most of their
|
|
388
|
+
letter pairs get the same signature in at least one band with high probability, so names
|
|
389
|
+
with typos in both still meet; exact keys miss them. `with` adds a field to every key, to
|
|
390
|
+
keep the blocks small (a band of common names state-wide is a hub otherwise).
|
|
391
|
+
|
|
392
|
+
"lsh_blocking": {"fields": ["first_name", "last_name"], "with": "town",
|
|
393
|
+
"bands": 16, "rows": 3}
|
|
394
|
+
|
|
395
|
+
With b bands of r rows, two records whose letter-pair sets have Jaccard similarity s share
|
|
396
|
+
a band with probability 1 - (1 - s^r)^b: for 16 x 3, 0.97 at s = 0.6 and 0.32 at s = 0.2.
|
|
397
|
+
Returns a list of (record index, key) pairs."""
|
|
398
|
+
import zlib
|
|
399
|
+
fields, bands, rows = spec["fields"], int(spec.get("bands", 16)), int(spec.get("rows", 3))
|
|
400
|
+
text = df[fields[0]].astype(str)
|
|
401
|
+
for f in fields[1:]:
|
|
402
|
+
text = text + " " + df[f].astype(str)
|
|
403
|
+
text = text.str.strip()
|
|
404
|
+
rng = np.random.default_rng(seed)
|
|
405
|
+
k = bands * rows
|
|
406
|
+
p = np.uint64((1 << 61) - 1)
|
|
407
|
+
a = rng.integers(1, int(p), k, dtype=np.uint64) # k hash functions (a*x + b) mod p
|
|
408
|
+
b = rng.integers(0, int(p), k, dtype=np.uint64)
|
|
409
|
+
other = df[spec["with"]].astype(str).to_numpy() if spec.get("with") else None
|
|
410
|
+
out = []
|
|
411
|
+
cache = {}
|
|
412
|
+
for i, t in enumerate(text):
|
|
413
|
+
if len(t) < 3 or (other is not None and not other[i]):
|
|
414
|
+
continue
|
|
415
|
+
if t not in cache:
|
|
416
|
+
g = {f"{t[x:x + 2]}" for x in range(len(t) - 1)}
|
|
417
|
+
ids = np.array([zlib.crc32(s.encode()) for s in g], dtype=np.uint64)
|
|
418
|
+
# min over the record's letter pairs of k random hash functions
|
|
419
|
+
sig = ((ids[:, None] * a[None, :] + b[None, :]) % p).min(axis=0)
|
|
420
|
+
cache[t] = [hash(tuple(sig[x * rows:(x + 1) * rows].tolist())) for x in range(bands)]
|
|
421
|
+
extra = other[i] if other is not None else ""
|
|
422
|
+
out += [(i, ("lsh", band, extra, h)) for band, h in enumerate(cache[t])]
|
|
423
|
+
return out
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
LARGE_BLOCK_SAMPLE = 50 # pairs scored per oversized block
|
|
427
|
+
LARGE_BLOCK_MATCH_SHARE = 0.5 # kept when at least this share of them are matches
|
|
428
|
+
LARGE_BLOCK_MAX = 5000 # records: larger blocks are not checked (12.5 million pairs each)
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def large_block_check_on(cfg):
|
|
432
|
+
"""On unless "check_large_blocks": false - it found no large entity, and changed nothing, on
|
|
433
|
+
six person benchmarks, and lifted dedupe's patent-applicant file from 6.93 to 75.88."""
|
|
434
|
+
return cfg.get("check_large_blocks", True) and not cfg.get("blocking")
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
def saved_run_large_blocks(cfg, report, predict, th, pairs):
|
|
438
|
+
"""For `add` and `explain`: the same check over a saved run's model, so a large entity kept by
|
|
439
|
+
the run is kept again. Returns `pairs` with the kept blocks' pairs added (sorted, unique)."""
|
|
440
|
+
skipped = report.pop("_skipped", [])
|
|
441
|
+
if not (large_block_check_on(cfg) and skipped):
|
|
442
|
+
return pairs
|
|
443
|
+
extra, _ = check_large_blocks(skipped, predict, th, budget=max(1_000_000, len(pairs)))
|
|
444
|
+
return np.unique(np.vstack([pairs, extra]), axis=0).astype(np.int32) if len(extra) else pairs
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def check_large_blocks(skipped, predict, th, budget, seed=0, share=LARGE_BLOCK_MATCH_SHARE):
|
|
448
|
+
"""Blocks skipped as hubs that are one real entity - an organisation with hundreds of records -
|
|
449
|
+
rather than junk (a@a.com: unrelated people agreeing on nothing else) or many people (an
|
|
450
|
+
apartment building). The trained model scores a sample of each block's pairs; a block whose
|
|
451
|
+
sampled pairs are mostly matches is kept, and all its pairs become candidates, largest-first
|
|
452
|
+
until `budget` pairs. Returns (new pairs int32 (m, 2), i < j; report rows)."""
|
|
453
|
+
rng = np.random.default_rng(seed)
|
|
454
|
+
blocks = [(key, np.unique(np.asarray(idx))) for key, idx in skipped if len(idx) <= LARGE_BLOCK_MAX]
|
|
455
|
+
if not blocks:
|
|
456
|
+
return np.zeros((0, 2), np.int32), []
|
|
457
|
+
samples, owner = [], []
|
|
458
|
+
for b, (_, idx) in enumerate(blocks):
|
|
459
|
+
a, c = rng.choice(idx, LARGE_BLOCK_SAMPLE), rng.choice(idx, LARGE_BLOCK_SAMPLE)
|
|
460
|
+
ok = a != c
|
|
461
|
+
samples.append(np.c_[np.minimum(a[ok], c[ok]), np.maximum(a[ok], c[ok])])
|
|
462
|
+
owner.append(np.full(int(ok.sum()), b))
|
|
463
|
+
p = predict(np.vstack(samples).astype(np.int32)) # one scoring pass for every block
|
|
464
|
+
owner = np.concatenate(owner)
|
|
465
|
+
match_share = np.bincount(owner, weights=p > th, minlength=len(blocks)) / np.maximum(
|
|
466
|
+
np.bincount(owner, minlength=len(blocks)), 1)
|
|
467
|
+
rows, new, used = [], [], 0
|
|
468
|
+
for b in np.argsort(-np.array([len(idx) for _, idx in blocks]), kind="stable"):
|
|
469
|
+
key, idx = blocks[b]
|
|
470
|
+
n_pairs = len(idx) * (len(idx) - 1) // 2
|
|
471
|
+
keep = match_share[b] >= share and used + n_pairs <= budget
|
|
472
|
+
rows.append({"key": " / ".join(str(v) for v in key), "records": int(len(idx)),
|
|
473
|
+
"sampled_match_share": round(float(match_share[b]), 3), "kept": bool(keep)})
|
|
474
|
+
if keep:
|
|
475
|
+
i, j = np.triu_indices(len(idx), 1)
|
|
476
|
+
new.append(np.c_[idx[i], idx[j]])
|
|
477
|
+
used += n_pairs
|
|
478
|
+
if not new:
|
|
479
|
+
return np.zeros((0, 2), np.int32), rows
|
|
480
|
+
return np.unique(np.vstack(new), axis=0).astype(np.int32), rows
|
custmatch/cli.py
ADDED
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""custmatch - match customer records across sources.
|
|
2
|
+
|
|
3
|
+
custmatch sample CONFIG --n 500 --out to_label.csv [--active labels_so_far.csv]
|
|
4
|
+
random candidate pairs, side by side, for you to label (fill `label` with 1 or 0);
|
|
5
|
+
--active picks the pairs the forest is least sure of instead
|
|
6
|
+
|
|
7
|
+
custmatch run CONFIG --labels labels.csv --out results/ [--previous DIR] [--model DIR]
|
|
8
|
+
clusters.csv, golden_records.csv, review.csv, history.csv, summary.json;
|
|
9
|
+
--previous keeps that run's customer IDs; --model reuses its trained model (no labels)
|
|
10
|
+
|
|
11
|
+
custmatch blocking CONFIG
|
|
12
|
+
blocking diagnostics before you label: pairs per rule, largest blocks, splits and drops
|
|
13
|
+
|
|
14
|
+
custmatch add NEW_CONFIG --state results/ --out results2/ [--delete deleted.csv]
|
|
15
|
+
match new or changed records against a previous run's customers without rerunning it,
|
|
16
|
+
and remove deleted ones (--delete: a CSV of source, record_id); writes changes.csv,
|
|
17
|
+
merges.csv and history.csv
|
|
18
|
+
|
|
19
|
+
custmatch forget RECORDS.csv --state results/ --out results2/
|
|
20
|
+
an erasure request: remove these records (source, record_id) and rebuild the affected
|
|
21
|
+
customers without them; delete the old output folders yourself
|
|
22
|
+
|
|
23
|
+
custmatch explain --state results/ RECORD_A RECORD_B
|
|
24
|
+
why two records were or were not matched: blocking, evidence, score, overrides
|
|
25
|
+
|
|
26
|
+
custmatch monitor results/2026-09 results/2026-10 ... [--out trend.csv] [--fail-on-drift]
|
|
27
|
+
match rates, field fill and score drift across outputs, oldest first; exits 3 on drift
|
|
28
|
+
with --fail-on-drift
|
|
29
|
+
|
|
30
|
+
CONFIG maps your files and columns onto the pipeline's (see custmatch/io.py and docs/guide.md).
|
|
31
|
+
"""
|
|
32
|
+
import argparse
|
|
33
|
+
import json
|
|
34
|
+
import sys
|
|
35
|
+
|
|
36
|
+
from . import io, labels, pipeline
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def main(argv=None):
|
|
40
|
+
ap = argparse.ArgumentParser(prog="custmatch", description=__doc__,
|
|
41
|
+
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
42
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
43
|
+
s = sub.add_parser("sample", help="draw candidate pairs for hand-labelling")
|
|
44
|
+
s.add_argument("config")
|
|
45
|
+
s.add_argument("--n", type=int, default=500)
|
|
46
|
+
s.add_argument("--out", required=True)
|
|
47
|
+
s.add_argument("--seed", type=int, default=0)
|
|
48
|
+
s.add_argument("--active", metavar="LABELS",
|
|
49
|
+
help="instead of random pairs, the ones the forest trained on LABELS is least sure of")
|
|
50
|
+
s.add_argument("--strategy", choices=["uncertainty", "disagreement"], default="uncertainty",
|
|
51
|
+
help="with --active: disagreement also offers pairs blocking missed that the forest calls matches")
|
|
52
|
+
s.add_argument("--mix", type=float, default=0.0,
|
|
53
|
+
help="with --active: share of each batch drawn at random instead (e.g. 0.5)")
|
|
54
|
+
r = sub.add_parser("run", help="match, cluster, households, golden records")
|
|
55
|
+
r.add_argument("config")
|
|
56
|
+
r.add_argument("--labels", help="labelled pairs; optional with --model")
|
|
57
|
+
r.add_argument("--previous", metavar="DIR", help="a previous run's output: keep its customer IDs")
|
|
58
|
+
r.add_argument("--model", metavar="DIR", help="reuse that run's trained model and threshold")
|
|
59
|
+
r.add_argument("--out", required=True)
|
|
60
|
+
bl = sub.add_parser("blocking", help="blocking diagnostics: pairs per rule, largest blocks, splits")
|
|
61
|
+
bl.add_argument("config")
|
|
62
|
+
ad = sub.add_parser("add", help="match new records against a previous run's customers")
|
|
63
|
+
ad.add_argument("config", help="config for the NEW records only")
|
|
64
|
+
ad.add_argument("--state", required=True, help="output folder of a previous run or add")
|
|
65
|
+
ad.add_argument("--out", required=True)
|
|
66
|
+
ad.add_argument("--delete", metavar="CSV", help="records to remove: columns source, record_id")
|
|
67
|
+
fg = sub.add_parser("forget", help="erase records: remove them and rebuild their customers")
|
|
68
|
+
fg.add_argument("records", help="CSV of source, record_id")
|
|
69
|
+
fg.add_argument("--state", required=True)
|
|
70
|
+
fg.add_argument("--out", required=True)
|
|
71
|
+
ex = sub.add_parser("explain", help="why two records were or were not matched")
|
|
72
|
+
ex.add_argument("a", help="first record, source:record_id")
|
|
73
|
+
ex.add_argument("b", help="second record, source:record_id")
|
|
74
|
+
ex.add_argument("--state", required=True, help="output folder of a run or add")
|
|
75
|
+
mo = sub.add_parser("monitor", help="match rates and drift across outputs, oldest first")
|
|
76
|
+
mo.add_argument("outputs", nargs="+", help="output folders of runs or adds, in time order")
|
|
77
|
+
mo.add_argument("--out", help="also write the table to this CSV")
|
|
78
|
+
mo.add_argument("--fail-on-drift", action="store_true", help="exit 3 when any drift alert fires")
|
|
79
|
+
a = ap.parse_args(argv)
|
|
80
|
+
try:
|
|
81
|
+
if a.cmd == "sample":
|
|
82
|
+
k = (labels.sample_active(a.config, a.active, a.n, a.out, a.seed, a.mix, a.strategy) if a.active
|
|
83
|
+
else labels.sample(a.config, a.n, a.out, a.seed))
|
|
84
|
+
print(f"wrote {k} pairs to {a.out}: fill the label column with 1 (same person) or 0")
|
|
85
|
+
elif a.cmd == "blocking":
|
|
86
|
+
from .blocking import candidates_for
|
|
87
|
+
from .core import profile_contacts
|
|
88
|
+
df, cfg = io.load(a.config)
|
|
89
|
+
# the blocker `run` will use: default, "fixed" or "learned" (which, before labels
|
|
90
|
+
# exist, falls back to fixed rules)
|
|
91
|
+
_, _, rep, _ = candidates_for(df, {f: profile_contacts(df, f) for f in ("email", "phone")}, cfg)
|
|
92
|
+
rep.pop("_skipped", None)
|
|
93
|
+
print(json.dumps(rep, indent=1, default=str))
|
|
94
|
+
elif a.cmd == "add":
|
|
95
|
+
from . import incremental
|
|
96
|
+
print(json.dumps(incremental.add(a.state, a.config, a.out, delete=a.delete), indent=1))
|
|
97
|
+
elif a.cmd == "forget":
|
|
98
|
+
from . import incremental
|
|
99
|
+
print(json.dumps(incremental.add(a.state, None, a.out, delete=a.records), indent=1))
|
|
100
|
+
elif a.cmd == "monitor":
|
|
101
|
+
from . import monitor
|
|
102
|
+
table, alerts = monitor.compare(a.outputs)
|
|
103
|
+
print(table.to_string(index=False))
|
|
104
|
+
print("\n".join(["", "drift alerts:"] + alerts) if alerts else "\nno drift alerts")
|
|
105
|
+
if a.out:
|
|
106
|
+
table.to_csv(a.out, index=False)
|
|
107
|
+
if alerts and a.fail_on_drift:
|
|
108
|
+
return 3
|
|
109
|
+
elif a.cmd == "explain":
|
|
110
|
+
from . import explain
|
|
111
|
+
print(json.dumps(explain.why(a.state, a.a, a.b), indent=1))
|
|
112
|
+
else:
|
|
113
|
+
if a.labels is None and a.model is None:
|
|
114
|
+
print("custmatch: run needs --labels (or --model to reuse a trained model)", file=sys.stderr)
|
|
115
|
+
return 2
|
|
116
|
+
print(json.dumps(pipeline.run(a.config, a.labels, a.out, previous_dir=a.previous, model_dir=a.model), indent=1))
|
|
117
|
+
except (io.ConfigError, FileNotFoundError) as e:
|
|
118
|
+
print(f"custmatch: {e}", file=sys.stderr)
|
|
119
|
+
return 2
|
|
120
|
+
return 0
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
if __name__ == "__main__":
|
|
124
|
+
sys.exit(main())
|