custmatch 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
custmatch/__init__.py ADDED
@@ -0,0 +1 @@
1
+ """Customer matching: the pipeline measured in docs/report/, as a tool. See docs/guide.md."""
custmatch/__main__.py ADDED
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from .cli import main
4
+
5
+ sys.exit(main())
custmatch/blocking.py ADDED
@@ -0,0 +1,480 @@
1
+ """Blocking for the tool: rules, learned rule sets, split oversized blocks, SQL self-join, diagnostics.
2
+
3
+ What the three leading projects do and custmatch's first blocker did not (docs/guide.md):
4
+
5
+ learned rules dedupe: pick the cheapest set of blocking rules that still covers a target
6
+ share (default 95%) of the labelled matches - a greedy weighted set cover
7
+ over a library of rules - instead of fixed hand-picked keys
8
+ split, not drop Zingg: a block larger than max_block is refined with the next predicate
9
+ (first letters of the first name, birth year, street, ...) until it fits,
10
+ and dropped only if it still cannot be split - custmatch dropped it outright,
11
+ which lost the 173-record London block in Splink's fake_1000
12
+ SQL self-join dedupe / Splink: pairs come from a DuckDB self-join on (rule, key) with
13
+ DISTINCT, not Python loops over every block
14
+ diagnostics Splink: pairs contributed by each rule, its largest blocks, and what was split
15
+ or dropped - blocking_report.json in the run's output
16
+
17
+ core.block (the fixed keys the report measured) is unchanged; this module is used by the tool.
18
+ A rule is a tuple of predicate names; records agree on a rule when every predicate gives the
19
+ same non-empty key.
20
+ """
21
+ import numpy as np
22
+ import pandas as pd
23
+
24
+ # predicate name -> (column, transform). Transforms map a cleaned string column to keys.
25
+ _TRANSFORMS = {
26
+ "exact": lambda s: s,
27
+ "first3": lambda s: s.str[:3].where(s.str.len() >= 3, ""),
28
+ "first1": lambda s: s.str[:1],
29
+ "year": lambda s: s.str.extract(r"(\d{4})", expand=False).fillna(""),
30
+ "yearmonth": None, # set below: needs date parsing
31
+ "soundex": None,
32
+ }
33
+ NAME_FIELDS = ("first_name", "last_name")
34
+ FIXED_RULES = [("last_name:exact", "first_name:first3"), ("last_name:exact", "addr_street:exact"),
35
+ ("addr_number:exact", "addr_street:exact"), ("first_name:exact", "addr_street:exact"),
36
+ ("email:exact",), ("phone:exact",)]
37
+ # refinements tried, in order, to split an oversized block
38
+ REFINEMENTS = ["first_name:first3", "date_of_birth:year", "last_name:first3", "addr_street:exact",
39
+ "first_name:exact", "date_of_birth:exact", "addr_number:exact"]
40
+
41
+
42
+ _SOUNDEX = {**dict.fromkeys("bfpv", "1"), **dict.fromkeys("cgjkqsxz", "2"), **dict.fromkeys("dt", "3"),
43
+ "l": "4", **dict.fromkeys("mn", "5"), "r": "6"}
44
+
45
+
46
+ def soundex(word):
47
+ """American Soundex: the first letter, then up to three digits for the consonant sounds that
48
+ follow (Robert, Rupert -> R163). Letters with one code next to each other, or separated
49
+ only by h or w, count once; a vowel, space or hyphen between them counts them twice - as
50
+ jellyfish.soundex, which this replaces, does."""
51
+ w = word.lower()
52
+ first = next((n for n, c in enumerate(w) if "a" <= c <= "z"), None)
53
+ if first is None:
54
+ return ""
55
+ out, last = [], _SOUNDEX.get(w[first], "")
56
+ for c in w[first + 1:]:
57
+ code = _SOUNDEX.get(c, "") # vowels, spaces and hyphens separate: code ""
58
+ if code and code != last:
59
+ out.append(code)
60
+ if c not in "hw": # h and w do not separate two equal codes
61
+ last = code
62
+ return (w[first].upper() + "".join(out) + "000")[:4]
63
+
64
+
65
+ def _soundex(s):
66
+ return s.map(lambda v: soundex(v) if v and v[0].isalpha() else "")
67
+
68
+
69
+ def _yearmonth(s):
70
+ from .core import parse_dates
71
+ d = parse_dates(s.to_numpy())
72
+ return pd.Series([f"{y:04d}{m:02d}" if y >= 0 else "" for y, m, _ in d], index=s.index)
73
+
74
+
75
+ class Keys:
76
+ """Lazily computed predicate keys per record (cleaned strings, "" = no key)."""
77
+
78
+ def __init__(self, df, profiles=None):
79
+ self.df, self.cache = df, {}
80
+ self.junk = {}
81
+ for f in ("email", "phone"):
82
+ if profiles and f in profiles:
83
+ self.junk[f] = df[f].map(lambda v: profiles[f].get(v, ("personal",))[0] == "junk").to_numpy()
84
+
85
+ def get(self, pred):
86
+ if pred not in self.cache:
87
+ col, how = pred.split(":")
88
+ if col not in self.df.columns:
89
+ self.cache[pred] = np.full(len(self.df), "", dtype=object)
90
+ else:
91
+ s = self.df[col].astype(str)
92
+ if how == "soundex":
93
+ k = _soundex(s)
94
+ elif how == "yearmonth":
95
+ k = _yearmonth(s)
96
+ else:
97
+ k = _TRANSFORMS[how](s)
98
+ k = k.to_numpy(dtype=object)
99
+ if col in self.junk:
100
+ k = np.where(self.junk[col], "", k)
101
+ self.cache[pred] = k
102
+ return self.cache[pred]
103
+
104
+ def codes(self, pred):
105
+ """Integer code per record for a predicate; -1 where it has no key."""
106
+ ck = ("codes", pred)
107
+ if ck not in self.cache:
108
+ k = self.get(pred)
109
+ c, _ = pd.factorize(k)
110
+ self.cache[ck] = np.where(k == "", -1, c).astype(np.int64)
111
+ return self.cache[ck]
112
+
113
+ def rule(self, rule):
114
+ """Integer key per record for a rule (a conjunction of predicates); -1 where any
115
+ predicate has no key. Integer codes, not joined strings: at 1M records joining strings
116
+ made blocking the slowest stage."""
117
+ parts = [self.codes(p) for p in rule]
118
+ empty = np.zeros(len(self.df), bool)
119
+ for c in parts:
120
+ empty |= c < 0
121
+ if len(parts) == 1:
122
+ key = parts[0]
123
+ else:
124
+ key, _ = pd.MultiIndex.from_arrays(parts).factorize()
125
+ key = key.astype(np.int64)
126
+ return np.where(empty, -1, key)
127
+
128
+ def label(self, rule, i):
129
+ """Readable key of record i under a rule, for diagnostics."""
130
+ return " | ".join(str(self.get(p)[i]) for p in rule)
131
+
132
+
133
+ def rule_library(df, cfg=None):
134
+ """Candidate rules over the fields this data actually has."""
135
+ from .schema import extra_fields
136
+ # country.py's derived name columns are blocked on through country.blocking_keys, not as rules
137
+ derived = {"name_script", "first_roots", "name_skeleton", "country", "updated_at"}
138
+ have = [c for c in df.columns if c not in ("source", "record_id") and c not in derived
139
+ and (df[c] != "").sum() > 1]
140
+ single = []
141
+ for c in have:
142
+ single.append(f"{c}:exact")
143
+ if c in NAME_FIELDS or c in ("addr_street",) or c in {n for n, _ in extra_fields(cfg or {})}:
144
+ single.append(f"{c}:first3")
145
+ if c in NAME_FIELDS:
146
+ single.append(f"{c}:soundex")
147
+ if c == "date_of_birth":
148
+ single += ["date_of_birth:year", "date_of_birth:yearmonth"]
149
+ rules = [(p,) for p in single]
150
+ strong = [p for p in single if p.endswith(":exact") or p.endswith(":soundex")]
151
+ for a in range(len(single)):
152
+ for b in range(a + 1, len(single)):
153
+ pa, pb = single[a], single[b]
154
+ if pa.split(":")[0] == pb.split(":")[0]:
155
+ continue
156
+ if pa in strong or pb in strong:
157
+ rules.append((pa, pb))
158
+ rules += fixed_rules(cfg, df)
159
+ return list(dict.fromkeys(rules))
160
+
161
+
162
+ def fixed_rules(cfg=None, df=None):
163
+ """The fixed keys as rules. A predicate on a field the data never fills is dropped from its
164
+ rule, so (last_name, addr_street) becomes (last_name) where there is no address - what the
165
+ report's core.block did implicitly (an empty street was a shared key), and the recall it gave
166
+ on historical_50k (87.0 vs 84.0 without) depends on it."""
167
+ from .schema import extra_keys
168
+ rules = [tuple(r) for r in FIXED_RULES] + [tuple(c if ":" in c else f"{c}:exact" for c in k) for k in extra_keys(cfg or {})]
169
+ if df is None:
170
+ return rules
171
+ filled = {c for c in df.columns if (df[c] != "").any()}
172
+ out = []
173
+ for r in rules:
174
+ kept = tuple(p for p in r if p.split(":")[0] in filled)
175
+ if kept:
176
+ out.append(kept)
177
+ return list(dict.fromkeys(out))
178
+
179
+
180
+ def _n2(sizes):
181
+ sizes = np.asarray(sizes, dtype=np.float64)
182
+ return float(np.sum(sizes * (sizes - 1) / 2))
183
+
184
+
185
+ def learn_rules(df, keys, pos_pairs, cfg=None, target=0.95, max_block=60, max_rules=8,
186
+ sample=50000, seed=0):
187
+ """Greedy weighted set cover: repeatedly add the rule covering the most still-uncovered
188
+ labelled matches per candidate pair it would generate, until `target` of the labelled
189
+ matches that any rule can cover are covered. Returns (rules, report).
190
+
191
+ Like dedupe, it learns on a sample: the records in labelled pairs plus up to `sample`
192
+ random others, with block sizes scaled up to the full file for the cost - evaluating the
193
+ whole rule library on a million records took 143 s."""
194
+ lib = rule_library(df, cfg)
195
+ n = len(df)
196
+ rng = np.random.default_rng(seed)
197
+ in_labels = np.unique(pos_pairs.ravel())
198
+ others = np.setdiff1d(np.arange(n), in_labels)
199
+ extra = rng.choice(others, min(sample, len(others)), replace=False) if len(others) else others
200
+ idx = np.sort(np.concatenate([in_labels, extra]))
201
+ sub = Keys(df.iloc[idx].reset_index(drop=True), None)
202
+ sub.junk = {f: v[idx] for f, v in keys.junk.items()}
203
+ pos_local = np.searchsorted(idx, pos_pairs)
204
+ i, j = pos_local[:, 0], pos_local[:, 1]
205
+ scale = n / len(idx)
206
+ covers, cost = {}, {}
207
+ for r in lib:
208
+ k = sub.rule(r)
209
+ covers[r] = (k[i] == k[j]) & (k[i] >= 0)
210
+ sizes = np.bincount(k[k >= 0]) * scale if (k >= 0).any() else np.zeros(0)
211
+ sizes = sizes[sizes > 0]
212
+ # oversized blocks will be split; charge them roughly as pieces of max_block
213
+ small, big = sizes[sizes <= max_block], sizes[sizes > max_block]
214
+ cost[r] = _n2(small) + float(np.sum(np.ceil(big / max_block))) * _n2([max_block])
215
+ coverable = np.any([covers[r] for r in lib], axis=0) if lib else np.zeros(len(i), bool)
216
+ need = target * coverable.sum()
217
+ chosen, covered = [], np.zeros(len(i), bool)
218
+ while covered.sum() < need and len(chosen) < max_rules:
219
+ best = max(lib, key=lambda r: ((covers[r] & ~covered).sum() / (cost[r] + 1.0), -cost[r]))
220
+ if not (covers[best] & ~covered).any():
221
+ break
222
+ chosen.append(best)
223
+ covered |= covers[best]
224
+ report = {"labelled_matches": int(len(i)), "coverable_by_library": int(coverable.sum()),
225
+ "covered": int(covered.sum()), "target": target, "library_size": len(lib),
226
+ "rules": [{"rule": list(r), "covers": int(covers[r].sum()), "est_pairs": int(cost[r])}
227
+ for r in chosen]}
228
+ return chosen, report
229
+
230
+
231
+ def _split(keys, idx_block, rule, max_block):
232
+ """Refine an oversized block with the next predicates; returns (list of record-index arrays,
233
+ number of pieces dropped because nothing was left to split them on)."""
234
+ if len(idx_block) <= max_block:
235
+ return [idx_block], 0
236
+ for p in REFINEMENTS:
237
+ if p in rule:
238
+ continue
239
+ sub = keys.codes(p)[idx_block]
240
+ if (sub < 0).all() or len(np.unique(sub)) == 1:
241
+ continue
242
+ out, dropped = [], 0
243
+ # records with no value for this refinement (a first name shorter than three letters,
244
+ # no birth year) stay together as their own piece rather than being lost
245
+ for v in np.unique(sub):
246
+ part = idx_block[sub == v]
247
+ if len(part) < 2:
248
+ continue
249
+ o, d = _split(keys, part, rule + (p,), max_block)
250
+ out += o
251
+ dropped += d
252
+ return out, dropped
253
+ return [], 1 # nothing left to split on: drop it as a hub
254
+
255
+
256
+ def generate(df, keys, rules, max_block=60):
257
+ """Candidate pairs for the rules (splitting oversized blocks), plus per-rule diagnostics.
258
+ Returns (pairs int32 array sorted, report)."""
259
+ rows, report = [], []
260
+ for n, r in enumerate(rules):
261
+ k = keys.rule(r)
262
+ has = np.flatnonzero(k >= 0)
263
+ kk = k[has]
264
+ sizes = np.bincount(kk) if len(kk) else np.zeros(0, int)
265
+ big_codes = np.flatnonzero(sizes > max_block)
266
+ small = sizes[kk] <= max_block
267
+ rk, ri = kk[small], has[small]
268
+ next_key = int(sizes.size)
269
+ split_blocks, dropped, extra_k, extra_i = 0, 0, [], []
270
+ for bc in big_codes:
271
+ parts, d = _split(keys, has[kk == bc], tuple(r), max_block)
272
+ split_blocks += 1
273
+ dropped += d
274
+ for pidx in parts:
275
+ extra_k.append(np.full(len(pidx), next_key, np.int64))
276
+ extra_i.append(pidx)
277
+ next_key += 1
278
+ rows.append(pd.DataFrame({"rule": n, "key": np.concatenate([rk] + extra_k) if extra_k else rk,
279
+ "rid": np.concatenate([ri] + extra_i) if extra_i else ri}))
280
+ order = np.argsort(-sizes)[:5]
281
+ report.append({"rule": list(r), "records_with_key": int(len(has)), "blocks": int((sizes > 0).sum()),
282
+ "largest_blocks": [{"key": keys.label(r, int(has[kk == c][0])), "size": int(sizes[c])}
283
+ for c in order if sizes[c] > 0],
284
+ "oversized_blocks_split": split_blocks, "blocks_dropped_as_hubs": dropped})
285
+ table = pd.concat(rows, ignore_index=True) if rows else pd.DataFrame({"rule": [], "key": [], "rid": []})
286
+ per_rule = _block_pairs(table.rule.to_numpy(np.int64), table.key.to_numpy(np.int64), table.rid.to_numpy(np.int64))
287
+ first = per_rule.groupby(["i", "j"]).rule.min().reset_index()
288
+ counts = per_rule.rule.value_counts()
289
+ new = first.rule.value_counts()
290
+ for n, entry in enumerate(report):
291
+ entry["pairs"] = int(counts.get(n, 0))
292
+ entry["new_pairs"] = int(new.get(n, 0)) # not already produced by an earlier rule
293
+ pairs = first.sort_values(["i", "j"])[["i", "j"]].to_numpy(dtype=np.int32) if len(first) \
294
+ else np.zeros((0, 2), np.int32)
295
+ return pairs, {"rules": report, "candidate_pairs": int(len(pairs)), "max_block": max_block}
296
+
297
+
298
+ def _block_pairs(rule, key, rid):
299
+ """Every pair of records sharing a (rule, key) block: a DataFrame of rule, i < j. Sorted by
300
+ block, a record pairs with the k-th record after it while that one is still in its block;
301
+ blocks are capped at max_block, so k stays small and each step is one vectorised pass."""
302
+ if not len(rid):
303
+ return pd.DataFrame({"rule": np.zeros(0, np.int64), "i": np.zeros(0, np.int64), "j": np.zeros(0, np.int64)})
304
+ order = np.lexsort((rid, key, rule))
305
+ rule, key, rid = rule[order], key[order], rid[order]
306
+ new = np.r_[True, (rule[1:] != rule[:-1]) | (key[1:] != key[:-1])]
307
+ starts = np.flatnonzero(new)
308
+ end = np.repeat(np.r_[starts[1:], len(rid)], np.diff(np.r_[starts, len(rid)])) # block end per row
309
+ pos = np.arange(len(rid))
310
+ out_r, out_i, out_j = [], [], []
311
+ for k in range(1, int((end - pos).max())):
312
+ a = pos[pos + k < end]
313
+ b = a + k
314
+ out_r.append(rule[a]); out_i.append(np.minimum(rid[a], rid[b])); out_j.append(np.maximum(rid[a], rid[b]))
315
+ if not out_r:
316
+ return pd.DataFrame({"rule": np.zeros(0, np.int64), "i": np.zeros(0, np.int64), "j": np.zeros(0, np.int64)})
317
+ return pd.DataFrame({"rule": np.concatenate(out_r), "i": np.concatenate(out_i), "j": np.concatenate(out_j)})
318
+
319
+
320
+ def country_pairs(df, profiles, max_block):
321
+ """Pairs from country.blocking_keys (first-name root + DOB or surname, email user name, name
322
+ skeleton), blocks larger than max_block skipped - the same keys core.block uses."""
323
+ from collections import defaultdict
324
+ from . import country
325
+ groups = defaultdict(list)
326
+ for k, i in country.blocking_keys(df, profiles):
327
+ groups[k].append(i)
328
+ out = set()
329
+ for idx in groups.values():
330
+ if 2 <= len(idx) <= max_block:
331
+ idx = sorted(set(idx))
332
+ out.update((a, b) for x, a in enumerate(idx) for b in idx[x + 1:])
333
+ return np.array(sorted(out), dtype=np.int32).reshape(-1, 2)
334
+
335
+
336
+ def candidates_for(df, profiles, cfg, labelled_pos=None, rules=None):
337
+ """The tool's candidate pairs. Returns (pairs, hubs_dropped, report, rules_used).
338
+
339
+ cfg "blocking": absent -> the fixed keys the report measured (core.block, blocks over
340
+ max_block dropped); "fixed" -> the same kind of keys as rules, oversized blocks split;
341
+ "learned" -> rules learned from labelled_pos (falls back to fixed without labels).
342
+ rules: reuse a saved rule set (incremental matching)."""
343
+ from .core import block
344
+ from .schema import extra_keys, max_block as cfg_max_block
345
+ mode = cfg.get("blocking")
346
+ mb = cfg_max_block(cfg) or 60
347
+ if rules is None and not mode:
348
+ skipped, by_key = [], []
349
+ pairs, hubs = block(df, profiles, extra_keys=extra_keys(cfg), max_block=cfg_max_block(cfg),
350
+ lsh=cfg.get("lsh_blocking"), skipped=skipped, by_key=by_key)
351
+ # records never compared with anyone because their only blocks were skipped - a record
352
+ # whose broad block (postcode alone) was skipped but whose narrow one was used is fine
353
+ compared = np.zeros(len(df), bool)
354
+ compared[pairs.ravel()] = True
355
+ records = len({i for _, idx in skipped for i in idx if not compared[i]})
356
+ top = sorted(skipped, key=lambda s: -len(s[1]))[:5]
357
+ return pairs, hubs, {"mode": "core", "candidate_pairs": int(len(pairs)), "hubs_dropped": int(hubs),
358
+ "skipped_blocks_with_values": len(skipped), "records_only_in_skipped_blocks": records,
359
+ "largest_skipped_blocks": [{"key": " / ".join(str(v) for v in k), "records": len(idx)}
360
+ for k, idx in top],
361
+ "pairs_by_key": by_key, "_skipped": skipped}, None
362
+ if cfg.get("lsh_blocking"):
363
+ from .io import ConfigError
364
+ raise ConfigError('"lsh_blocking" works with the default blocking only; remove "blocking": '
365
+ f'"{mode}" or "lsh_blocking"')
366
+ keys = Keys(df, profiles)
367
+ learn = None
368
+ if rules is None:
369
+ if mode == "learned" and labelled_pos is not None and len(labelled_pos):
370
+ rules, learn = learn_rules(df, keys, labelled_pos, cfg, target=cfg.get("blocking_recall", 0.95),
371
+ max_block=mb)
372
+ else:
373
+ rules = fixed_rules(cfg, df)
374
+ pairs, report = generate(df, keys, rules, max_block=mb)
375
+ extra = country_pairs(df, profiles, mb) # country.py keys: nickname + DOB, email user...
376
+ if len(extra):
377
+ allp = np.vstack([pairs, extra]) if len(pairs) else extra
378
+ pairs = np.unique(allp, axis=0)
379
+ report.update(mode=mode or "saved", learned=learn, country_key_pairs=int(len(extra)))
380
+ hubs = sum(r["blocks_dropped_as_hubs"] for r in report["rules"])
381
+ return pairs, hubs, report, [list(r) for r in rules]
382
+
383
+
384
+ # ------------------------------------------------------------------ similarity (LSH) blocking
385
+ def lsh_keys(df, spec, seed=0):
386
+ """Blocking keys from MinHash locality-sensitive hashing of letter pairs - the idea behind
387
+ anonlink's blocklib, on plain text. Records whose `fields` (joined) share most of their
388
+ letter pairs get the same signature in at least one band with high probability, so names
389
+ with typos in both still meet; exact keys miss them. `with` adds a field to every key, to
390
+ keep the blocks small (a band of common names state-wide is a hub otherwise).
391
+
392
+ "lsh_blocking": {"fields": ["first_name", "last_name"], "with": "town",
393
+ "bands": 16, "rows": 3}
394
+
395
+ With b bands of r rows, two records whose letter-pair sets have Jaccard similarity s share
396
+ a band with probability 1 - (1 - s^r)^b: for 16 x 3, 0.97 at s = 0.6 and 0.32 at s = 0.2.
397
+ Returns a list of (record index, key) pairs."""
398
+ import zlib
399
+ fields, bands, rows = spec["fields"], int(spec.get("bands", 16)), int(spec.get("rows", 3))
400
+ text = df[fields[0]].astype(str)
401
+ for f in fields[1:]:
402
+ text = text + " " + df[f].astype(str)
403
+ text = text.str.strip()
404
+ rng = np.random.default_rng(seed)
405
+ k = bands * rows
406
+ p = np.uint64((1 << 61) - 1)
407
+ a = rng.integers(1, int(p), k, dtype=np.uint64) # k hash functions (a*x + b) mod p
408
+ b = rng.integers(0, int(p), k, dtype=np.uint64)
409
+ other = df[spec["with"]].astype(str).to_numpy() if spec.get("with") else None
410
+ out = []
411
+ cache = {}
412
+ for i, t in enumerate(text):
413
+ if len(t) < 3 or (other is not None and not other[i]):
414
+ continue
415
+ if t not in cache:
416
+ g = {f"{t[x:x + 2]}" for x in range(len(t) - 1)}
417
+ ids = np.array([zlib.crc32(s.encode()) for s in g], dtype=np.uint64)
418
+ # min over the record's letter pairs of k random hash functions
419
+ sig = ((ids[:, None] * a[None, :] + b[None, :]) % p).min(axis=0)
420
+ cache[t] = [hash(tuple(sig[x * rows:(x + 1) * rows].tolist())) for x in range(bands)]
421
+ extra = other[i] if other is not None else ""
422
+ out += [(i, ("lsh", band, extra, h)) for band, h in enumerate(cache[t])]
423
+ return out
424
+
425
+
426
+ LARGE_BLOCK_SAMPLE = 50 # pairs scored per oversized block
427
+ LARGE_BLOCK_MATCH_SHARE = 0.5 # kept when at least this share of them are matches
428
+ LARGE_BLOCK_MAX = 5000 # records: larger blocks are not checked (12.5 million pairs each)
429
+
430
+
431
+ def large_block_check_on(cfg):
432
+ """On unless "check_large_blocks": false - it found no large entity, and changed nothing, on
433
+ six person benchmarks, and lifted dedupe's patent-applicant file from 6.93 to 75.88."""
434
+ return cfg.get("check_large_blocks", True) and not cfg.get("blocking")
435
+
436
+
437
+ def saved_run_large_blocks(cfg, report, predict, th, pairs):
438
+ """For `add` and `explain`: the same check over a saved run's model, so a large entity kept by
439
+ the run is kept again. Returns `pairs` with the kept blocks' pairs added (sorted, unique)."""
440
+ skipped = report.pop("_skipped", [])
441
+ if not (large_block_check_on(cfg) and skipped):
442
+ return pairs
443
+ extra, _ = check_large_blocks(skipped, predict, th, budget=max(1_000_000, len(pairs)))
444
+ return np.unique(np.vstack([pairs, extra]), axis=0).astype(np.int32) if len(extra) else pairs
445
+
446
+
447
+ def check_large_blocks(skipped, predict, th, budget, seed=0, share=LARGE_BLOCK_MATCH_SHARE):
448
+ """Blocks skipped as hubs that are one real entity - an organisation with hundreds of records -
449
+ rather than junk (a@a.com: unrelated people agreeing on nothing else) or many people (an
450
+ apartment building). The trained model scores a sample of each block's pairs; a block whose
451
+ sampled pairs are mostly matches is kept, and all its pairs become candidates, largest-first
452
+ until `budget` pairs. Returns (new pairs int32 (m, 2), i < j; report rows)."""
453
+ rng = np.random.default_rng(seed)
454
+ blocks = [(key, np.unique(np.asarray(idx))) for key, idx in skipped if len(idx) <= LARGE_BLOCK_MAX]
455
+ if not blocks:
456
+ return np.zeros((0, 2), np.int32), []
457
+ samples, owner = [], []
458
+ for b, (_, idx) in enumerate(blocks):
459
+ a, c = rng.choice(idx, LARGE_BLOCK_SAMPLE), rng.choice(idx, LARGE_BLOCK_SAMPLE)
460
+ ok = a != c
461
+ samples.append(np.c_[np.minimum(a[ok], c[ok]), np.maximum(a[ok], c[ok])])
462
+ owner.append(np.full(int(ok.sum()), b))
463
+ p = predict(np.vstack(samples).astype(np.int32)) # one scoring pass for every block
464
+ owner = np.concatenate(owner)
465
+ match_share = np.bincount(owner, weights=p > th, minlength=len(blocks)) / np.maximum(
466
+ np.bincount(owner, minlength=len(blocks)), 1)
467
+ rows, new, used = [], [], 0
468
+ for b in np.argsort(-np.array([len(idx) for _, idx in blocks]), kind="stable"):
469
+ key, idx = blocks[b]
470
+ n_pairs = len(idx) * (len(idx) - 1) // 2
471
+ keep = match_share[b] >= share and used + n_pairs <= budget
472
+ rows.append({"key": " / ".join(str(v) for v in key), "records": int(len(idx)),
473
+ "sampled_match_share": round(float(match_share[b]), 3), "kept": bool(keep)})
474
+ if keep:
475
+ i, j = np.triu_indices(len(idx), 1)
476
+ new.append(np.c_[idx[i], idx[j]])
477
+ used += n_pairs
478
+ if not new:
479
+ return np.zeros((0, 2), np.int32), rows
480
+ return np.unique(np.vstack(new), axis=0).astype(np.int32), rows
custmatch/cli.py ADDED
@@ -0,0 +1,124 @@
1
+ """custmatch - match customer records across sources.
2
+
3
+ custmatch sample CONFIG --n 500 --out to_label.csv [--active labels_so_far.csv]
4
+ random candidate pairs, side by side, for you to label (fill `label` with 1 or 0);
5
+ --active picks the pairs the forest is least sure of instead
6
+
7
+ custmatch run CONFIG --labels labels.csv --out results/ [--previous DIR] [--model DIR]
8
+ clusters.csv, golden_records.csv, review.csv, history.csv, summary.json;
9
+ --previous keeps that run's customer IDs; --model reuses its trained model (no labels)
10
+
11
+ custmatch blocking CONFIG
12
+ blocking diagnostics before you label: pairs per rule, largest blocks, splits and drops
13
+
14
+ custmatch add NEW_CONFIG --state results/ --out results2/ [--delete deleted.csv]
15
+ match new or changed records against a previous run's customers without rerunning it,
16
+ and remove deleted ones (--delete: a CSV of source, record_id); writes changes.csv,
17
+ merges.csv and history.csv
18
+
19
+ custmatch forget RECORDS.csv --state results/ --out results2/
20
+ an erasure request: remove these records (source, record_id) and rebuild the affected
21
+ customers without them; delete the old output folders yourself
22
+
23
+ custmatch explain --state results/ RECORD_A RECORD_B
24
+ why two records were or were not matched: blocking, evidence, score, overrides
25
+
26
+ custmatch monitor results/2026-09 results/2026-10 ... [--out trend.csv] [--fail-on-drift]
27
+ match rates, field fill and score drift across outputs, oldest first; exits 3 on drift
28
+ with --fail-on-drift
29
+
30
+ CONFIG maps your files and columns onto the pipeline's (see custmatch/io.py and docs/guide.md).
31
+ """
32
+ import argparse
33
+ import json
34
+ import sys
35
+
36
+ from . import io, labels, pipeline
37
+
38
+
39
+ def main(argv=None):
40
+ ap = argparse.ArgumentParser(prog="custmatch", description=__doc__,
41
+ formatter_class=argparse.RawDescriptionHelpFormatter)
42
+ sub = ap.add_subparsers(dest="cmd", required=True)
43
+ s = sub.add_parser("sample", help="draw candidate pairs for hand-labelling")
44
+ s.add_argument("config")
45
+ s.add_argument("--n", type=int, default=500)
46
+ s.add_argument("--out", required=True)
47
+ s.add_argument("--seed", type=int, default=0)
48
+ s.add_argument("--active", metavar="LABELS",
49
+ help="instead of random pairs, the ones the forest trained on LABELS is least sure of")
50
+ s.add_argument("--strategy", choices=["uncertainty", "disagreement"], default="uncertainty",
51
+ help="with --active: disagreement also offers pairs blocking missed that the forest calls matches")
52
+ s.add_argument("--mix", type=float, default=0.0,
53
+ help="with --active: share of each batch drawn at random instead (e.g. 0.5)")
54
+ r = sub.add_parser("run", help="match, cluster, households, golden records")
55
+ r.add_argument("config")
56
+ r.add_argument("--labels", help="labelled pairs; optional with --model")
57
+ r.add_argument("--previous", metavar="DIR", help="a previous run's output: keep its customer IDs")
58
+ r.add_argument("--model", metavar="DIR", help="reuse that run's trained model and threshold")
59
+ r.add_argument("--out", required=True)
60
+ bl = sub.add_parser("blocking", help="blocking diagnostics: pairs per rule, largest blocks, splits")
61
+ bl.add_argument("config")
62
+ ad = sub.add_parser("add", help="match new records against a previous run's customers")
63
+ ad.add_argument("config", help="config for the NEW records only")
64
+ ad.add_argument("--state", required=True, help="output folder of a previous run or add")
65
+ ad.add_argument("--out", required=True)
66
+ ad.add_argument("--delete", metavar="CSV", help="records to remove: columns source, record_id")
67
+ fg = sub.add_parser("forget", help="erase records: remove them and rebuild their customers")
68
+ fg.add_argument("records", help="CSV of source, record_id")
69
+ fg.add_argument("--state", required=True)
70
+ fg.add_argument("--out", required=True)
71
+ ex = sub.add_parser("explain", help="why two records were or were not matched")
72
+ ex.add_argument("a", help="first record, source:record_id")
73
+ ex.add_argument("b", help="second record, source:record_id")
74
+ ex.add_argument("--state", required=True, help="output folder of a run or add")
75
+ mo = sub.add_parser("monitor", help="match rates and drift across outputs, oldest first")
76
+ mo.add_argument("outputs", nargs="+", help="output folders of runs or adds, in time order")
77
+ mo.add_argument("--out", help="also write the table to this CSV")
78
+ mo.add_argument("--fail-on-drift", action="store_true", help="exit 3 when any drift alert fires")
79
+ a = ap.parse_args(argv)
80
+ try:
81
+ if a.cmd == "sample":
82
+ k = (labels.sample_active(a.config, a.active, a.n, a.out, a.seed, a.mix, a.strategy) if a.active
83
+ else labels.sample(a.config, a.n, a.out, a.seed))
84
+ print(f"wrote {k} pairs to {a.out}: fill the label column with 1 (same person) or 0")
85
+ elif a.cmd == "blocking":
86
+ from .blocking import candidates_for
87
+ from .core import profile_contacts
88
+ df, cfg = io.load(a.config)
89
+ # the blocker `run` will use: default, "fixed" or "learned" (which, before labels
90
+ # exist, falls back to fixed rules)
91
+ _, _, rep, _ = candidates_for(df, {f: profile_contacts(df, f) for f in ("email", "phone")}, cfg)
92
+ rep.pop("_skipped", None)
93
+ print(json.dumps(rep, indent=1, default=str))
94
+ elif a.cmd == "add":
95
+ from . import incremental
96
+ print(json.dumps(incremental.add(a.state, a.config, a.out, delete=a.delete), indent=1))
97
+ elif a.cmd == "forget":
98
+ from . import incremental
99
+ print(json.dumps(incremental.add(a.state, None, a.out, delete=a.records), indent=1))
100
+ elif a.cmd == "monitor":
101
+ from . import monitor
102
+ table, alerts = monitor.compare(a.outputs)
103
+ print(table.to_string(index=False))
104
+ print("\n".join(["", "drift alerts:"] + alerts) if alerts else "\nno drift alerts")
105
+ if a.out:
106
+ table.to_csv(a.out, index=False)
107
+ if alerts and a.fail_on_drift:
108
+ return 3
109
+ elif a.cmd == "explain":
110
+ from . import explain
111
+ print(json.dumps(explain.why(a.state, a.a, a.b), indent=1))
112
+ else:
113
+ if a.labels is None and a.model is None:
114
+ print("custmatch: run needs --labels (or --model to reuse a trained model)", file=sys.stderr)
115
+ return 2
116
+ print(json.dumps(pipeline.run(a.config, a.labels, a.out, previous_dir=a.previous, model_dir=a.model), indent=1))
117
+ except (io.ConfigError, FileNotFoundError) as e:
118
+ print(f"custmatch: {e}", file=sys.stderr)
119
+ return 2
120
+ return 0
121
+
122
+
123
+ if __name__ == "__main__":
124
+ sys.exit(main())