custmatch 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. custmatch-0.1.0/LICENSE +202 -0
  2. custmatch-0.1.0/NOTICE +29 -0
  3. custmatch-0.1.0/PKG-INFO +277 -0
  4. custmatch-0.1.0/README.md +235 -0
  5. custmatch-0.1.0/custmatch/__init__.py +1 -0
  6. custmatch-0.1.0/custmatch/__main__.py +5 -0
  7. custmatch-0.1.0/custmatch/blocking.py +480 -0
  8. custmatch-0.1.0/custmatch/cli.py +124 -0
  9. custmatch-0.1.0/custmatch/core.py +652 -0
  10. custmatch-0.1.0/custmatch/countries.py +82 -0
  11. custmatch-0.1.0/custmatch/country.py +393 -0
  12. custmatch-0.1.0/custmatch/data/nicknames_au_uk.csv +91 -0
  13. custmatch-0.1.0/custmatch/explain.py +214 -0
  14. custmatch-0.1.0/custmatch/golden.py +236 -0
  15. custmatch-0.1.0/custmatch/households.py +236 -0
  16. custmatch-0.1.0/custmatch/identity.py +187 -0
  17. custmatch-0.1.0/custmatch/incremental.py +305 -0
  18. custmatch-0.1.0/custmatch/io.py +282 -0
  19. custmatch-0.1.0/custmatch/labels.py +132 -0
  20. custmatch-0.1.0/custmatch/matchweight.py +217 -0
  21. custmatch-0.1.0/custmatch/models.py +130 -0
  22. custmatch-0.1.0/custmatch/monitor.py +101 -0
  23. custmatch-0.1.0/custmatch/pipeline.py +527 -0
  24. custmatch-0.1.0/custmatch/placeholders.py +112 -0
  25. custmatch-0.1.0/custmatch/schema.py +58 -0
  26. custmatch-0.1.0/custmatch/stewardship.py +306 -0
  27. custmatch-0.1.0/custmatch/suspect.py +111 -0
  28. custmatch-0.1.0/custmatch.egg-info/PKG-INFO +277 -0
  29. custmatch-0.1.0/custmatch.egg-info/SOURCES.txt +46 -0
  30. custmatch-0.1.0/custmatch.egg-info/dependency_links.txt +1 -0
  31. custmatch-0.1.0/custmatch.egg-info/entry_points.txt +2 -0
  32. custmatch-0.1.0/custmatch.egg-info/requires.txt +19 -0
  33. custmatch-0.1.0/custmatch.egg-info/top_level.txt +1 -0
  34. custmatch-0.1.0/pyproject.toml +43 -0
  35. custmatch-0.1.0/setup.cfg +4 -0
  36. custmatch-0.1.0/tests/test_clustering.py +114 -0
  37. custmatch-0.1.0/tests/test_contacts.py +83 -0
  38. custmatch-0.1.0/tests/test_correctness.py +241 -0
  39. custmatch-0.1.0/tests/test_coverage.py +337 -0
  40. custmatch-0.1.0/tests/test_custmatch.py +1770 -0
  41. custmatch-0.1.0/tests/test_data_integrity.py +86 -0
  42. custmatch-0.1.0/tests/test_metrics.py +63 -0
  43. custmatch-0.1.0/tests/test_packaging.py +55 -0
  44. custmatch-0.1.0/tests/test_reported_results.py +1107 -0
  45. custmatch-0.1.0/tests/test_server_ops.py +143 -0
  46. custmatch-0.1.0/tests/test_server_rules.py +177 -0
  47. custmatch-0.1.0/tests/test_server_smoke.py +40 -0
  48. custmatch-0.1.0/tests/test_server_steward.py +284 -0
@@ -0,0 +1,202 @@
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright [yyyy] [name of copyright owner]
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.
custmatch-0.1.0/NOTICE ADDED
@@ -0,0 +1,29 @@
1
+ custmatch
2
+ Copyright 2026 Andrew Goodchild
3
+
4
+ Licensed under the Apache License, Version 2.0 (see LICENSE).
5
+
6
+ patches/ditto-mps-and-fixes.patch modifies Ditto (https://github.com/megagonlabs/ditto),
7
+ Copyright Megagon Labs, licensed under the Apache License, Version 2.0. The patch contains
8
+ lines of Ditto's source as diff context.
9
+
10
+ No third-party datasets are included in the current tree. (Earlier commits in the history
11
+ contained the nickname list below and a Wikidata name list, CC0, since removed.) custmatch
12
+ downloads it into ~/.cache/custmatch (or $CUSTMATCH_DATA) when first needed:
13
+
14
+ - nicknames.csv: the carltonnorthern/nicknames list (https://github.com/carltonnorthern/nicknames,
15
+ commit ed160b3), Copyright its contributors, licensed under the Apache License, Version 2.0.
16
+ Downloaded unmodified on first use and checked against its SHA-256.
17
+ - profanity_en.txt: the English list of LDNOOBW, the List of Dirty, Naughty, Obscene, and
18
+ Otherwise Bad Words (https://github.com/LDNOOBW/List-of-Dirty-Naughty-Obscene-and-Otherwise-Bad-Words,
19
+ commit 5faf2ba), Copyright its contributors, licensed under CC BY 4.0. Downloaded unmodified
20
+ on first use and checked against its SHA-256; custmatch uses it without the words that are
21
+ real names (custmatch/suspect.py).
22
+
23
+
24
+ Other datasets used by the research scripts are not included either. Scripts download or generate them (FEBRL and
25
+ historical_50k via recordlinkage and splink, the Magellan / DeepMatcher and MatchGPT
26
+ benchmarks from their own repositories, synthetic people via pseudopeople, name frequencies
27
+ from the US Census and Social Security via the FiveThirtyEight data repository (CC BY 4.0)
28
+ for the Australian test file), each under its own licence; see docs/reproducing.md. Files in results/ are metrics and logs computed from
29
+ them.
@@ -0,0 +1,277 @@
1
+ Metadata-Version: 2.4
2
+ Name: custmatch
3
+ Version: 0.1.0
4
+ Summary: Customer matching: contact profiling, blocking, a forest matcher tuned on clusters, households and golden records
5
+ Author: Andrew Goodchild
6
+ License-Expression: Apache-2.0
7
+ Project-URL: Repository, https://github.com/andrewgoodchild/custmatch
8
+ Project-URL: Documentation, https://github.com/andrewgoodchild/custmatch/blob/main/docs/guide.md
9
+ Project-URL: Issues, https://github.com/andrewgoodchild/custmatch/issues
10
+ Keywords: entity resolution,record linkage,deduplication,customer matching,householding
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Programming Language :: Python :: 3.10
14
+ Classifier: Programming Language :: Python :: 3.11
15
+ Classifier: Programming Language :: Python :: 3.12
16
+ Classifier: Programming Language :: Python :: 3.13
17
+ Classifier: Intended Audience :: Developers
18
+ Classifier: Intended Audience :: Science/Research
19
+ Classifier: Operating System :: OS Independent
20
+ Classifier: Topic :: Database
21
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
22
+ Requires-Python: >=3.10
23
+ Description-Content-Type: text/markdown
24
+ License-File: LICENSE
25
+ License-File: NOTICE
26
+ Requires-Dist: numpy
27
+ Requires-Dist: pandas
28
+ Requires-Dist: scipy
29
+ Requires-Dist: scikit-learn
30
+ Requires-Dist: rapidfuzz
31
+ Requires-Dist: certifi
32
+ Requires-Dist: pyarrow
33
+ Requires-Dist: phonenumbers
34
+ Requires-Dist: anyascii
35
+ Requires-Dist: networkx
36
+ Provides-Extra: test
37
+ Requires-Dist: pytest; extra == "test"
38
+ Requires-Dist: tomli; python_version < "3.11" and extra == "test"
39
+ Provides-Extra: catboost
40
+ Requires-Dist: catboost>=1.2; extra == "catboost"
41
+ Dynamic: license-file
42
+
43
+ # custmatch
44
+
45
+ [![tests](https://github.com/andrewgoodchild/custmatch/actions/workflows/tests.yml/badge.svg)](https://github.com/andrewgoodchild/custmatch/actions/workflows/tests.yml)
46
+ [![licence: Apache 2.0](https://img.shields.io/badge/licence-Apache%202.0-blue.svg)](https://github.com/andrewgoodchild/custmatch/blob/main/LICENSE)
47
+ ![python 3.10+](https://img.shields.io/badge/python-3.10%2B-blue.svg)
48
+
49
+ Find the records that are the same customer - across your CRM, billing and web sign-ups -
50
+ group them into households, and write one golden record per customer.
51
+
52
+ ![One customer in the steward dashboard: the golden record, and the links between four records from three sources](https://raw.githubusercontent.com/andrewgoodchild/custmatch/main/docs/images/dashboard-customer.png)
53
+
54
+ *One customer, John Citizen, found in four records from a web sign-up, the app and a store.
55
+ The golden record at the top takes the best value of each field; "2 value(s)" marks where the
56
+ records disagree, and a steward can pin the right one. Below it, each line is a link the model
57
+ made between two records, green for a score of 0.9 or more, with the evidence in the table:
58
+ "jack" in the app joined through the same email and phone. The dashed line is a* bridge *- the
59
+ only link holding app:a003772 to the rest - so it is the one to check if the customer ever
60
+ looks wrong.*
61
+
62
+ ## Why
63
+
64
+ Customer-matching projects stumble on the same problems. Records arrive in different formats,
65
+ with typos, nicknames and swapped names, and web sign-ups carry little more than a first name
66
+ and an email. Contact details lie: "noemail@noemail.com" is on a hundred records, a family shares
67
+ one email, a call centre typed in its own phone number. Spouses, parents and children, twins
68
+ and two John Smiths look like one person; one weak link chains strangers into a giant customer;
69
+ nobody can say why two records were merged; a steward's correction is lost on the next run;
70
+ customer IDs churn downstream; and the data is personal. custmatch is built around those
71
+ problems: junk and shared contacts are recognised rather than trusted, households are kept
72
+ apart from people, the match threshold is judged on whole customers so it cannot chain, every
73
+ link is explained, decisions and IDs survive reruns, and contacts can be matched hashed.
74
+
75
+ ## How it works
76
+
77
+ ```
78
+ records ─► clean & normalise ─► profile contacts ─► blocking ─► compare pairs ─► random forest
79
+ │
80
+ golden records ◄─ households ◄─ customers ◄─ threshold chosen on whole customers ◄────┘
81
+ ```
82
+
83
+ - **Clean and normalise** names, phones, dates, emails and addresses by country (US, AU, NZ, GB,
84
+ CA): nicknames, transliteration, Gmail dots, phone formats.
85
+ - **Profile contacts**: each email and phone is personal, a household's, or junk - judged by who
86
+ shares it, so junk and shared values stop counting as proof.
87
+ - **Blocking** compares only records that share a key (a surname and street, an email...), so a
88
+ million records make millions of pairs, not half a trillion.
89
+ - **Compare pairs** with name similarity, rarity and graded dates; a **random forest** learns
90
+ from a few hundred labelled pairs (active learning picks the useful ones).
91
+ - **Threshold on whole customers**: links are joined into customers, and the threshold is the
92
+ one whose *customers*, not pairs, are most accurate - the project's central finding.
93
+ - **Households** group customers by address and shared contacts; a **golden record** takes the
94
+ best value of each field.
95
+
96
+ [docs/guide.md](https://github.com/andrewgoodchild/custmatch/blob/main/docs/guide.md) explains each step.
97
+
98
+ ## Results
99
+
100
+ | 10 million records | 1 million records | Messy customer data | Ensemble option |
101
+ | :---: | :---: | :---: | :---: |
102
+ | **31 minutes** on one 16-core machine, **96.89** clustered F1 | **71 seconds** on a laptop, **97.99** clustered F1 | **97.3** clustered F1, against [Splink](https://github.com/moj-analytical-services/splink)'s 84.4 | **+0.41** F1 on average, **+1.8** on sparse records |
103
+
104
+ <sub>The two sizes ran on different machines: per record the laptop was about 2 to 2.6 times faster than
105
+ the cloud machine. On one machine, time per million records rose by about a quarter from 2 to
106
+ 10 million, most of it in the household and golden-record stage.</sub>
107
+
108
+ **Where it is stronger**
109
+
110
+ - **Messy customer data.** Placeholder, shared and household contacts are recognised, not
111
+ trusted, so they stop chaining strangers together: at a million records custmatch scores
112
+ 97.99 clustered F1 where Splink scores 78.88, and against a CDP's exact-identifier matching it
113
+ finds 22 to 26 more points of the true matches.
114
+ - **It keeps going.** Accuracy holds as the file grows - 97.99 at a million records, 96.89 at
115
+ ten million, on one machine. [dedupe](https://github.com/dedupeio/dedupe), about as accurate on 20,134 records (97.75), took 499
116
+ seconds and 8.5 GB there against custmatch's 5.4 seconds and 0.35 GB, so a million records was
117
+ out of reach on the same 16 GB laptop.
118
+ - **Modest memory.** A million records peak at 3.64 GB on a laptop (Splink: 5.96 GB); ten million
119
+ at 34.0 GB, about 3.4 GB per million, so a 16 GB laptop holds about four million. dedupe
120
+ needed 8.5 GB for 20,134 records.
121
+ - **An ensemble for hard files.** `"matcher_model": "forest_catboost"` averages the random forest
122
+ with CatBoost: +0.41 clustered F1 on average over six datasets, +1.3 on dense genealogy records
123
+ and +1.8 on sparse sign-ups, never more than 0.03 behind on any dataset, for runs about 30%
124
+ longer.
125
+
126
+ **Where others win.** Splink needs no labels and is faster once its large blocks are skipped
127
+ (21.6 seconds at a million records), and it can run on Spark across a cluster: its developers
128
+ matched a billion records that way. custmatch runs on one machine - about 3.4 GB per million
129
+ records, so tens of millions, not hundreds. On clean records with names and places only, the tools
130
+ are close.
131
+
132
+ The [report](https://github.com/andrewgoodchild/custmatch/tree/main/docs/report) has the full
133
+ comparison - against Splink, dedupe and a CDP's exact matching, on these files and on each tool's
134
+ own datasets - with every caveat: the synthetic and Australian files were generated for this
135
+ project, so their errors are ones the tool was designed around, and it needs good labels. The
136
+ test suite checks every number against `results/*.json`.
137
+
138
+ ## Quick start
139
+
140
+ ```bash
141
+ pip install custmatch # Python 3.10+ (or from a clone: pip install -e .)
142
+ # try it on the bundled demo (1,766 synthetic records, 600 pairs already labelled):
143
+ custmatch run examples/demo/config.json --labels examples/demo/labels.csv --out demo_out/
144
+
145
+ # on your own files: map the columns in a config, label pairs 1 or 0, run
146
+ custmatch sample config.json --n 300 --out to_label.csv
147
+ custmatch sample config.json --active to_label.csv --n 100 --out more.csv # the most useful pairs next
148
+ custmatch run config.json --labels labels.csv --out results/
149
+ ```
150
+
151
+ `config.json` maps your files' columns onto custmatch's fields; `examples/demo/config.json` is a
152
+ small working example, and all example data is synthetic (`examples/README.md`). For sparse
153
+ records - only a first name plus an email or mobile - add `"country": "AU"` (or your country's
154
+ code) and label with `sample --active`: on a sparse Australian test file that took clustered F1
155
+ from 54.9 to 81.7 (`examples/au/config.json`). The run writes `clusters.csv` (customer and
156
+ household per record), `golden_records.csv`, `links.csv` (every link with its evidence, weakest
157
+ first), `suspect_records.csv` and a `summary.json` with warnings.
158
+
159
+ ## The service: custmatch-server
160
+
161
+ The library runs a batch over files. **custmatch-server** runs it as a self-hosted service next to
162
+ your data - one Docker image, one database (Postgres, or SQLite for a trial):
163
+
164
+ - **Source records and their loads** kept in the database, every version of every record; files
165
+ dropped in an inbox loaded and the batch run on a schedule, with health checks and alerts.
166
+ - **Real-time matching and lookups** over HTTP (`POST /match`, `GET /records/...`), consistent
167
+ with the batch run, which reconciles every real-time decision.
168
+ - **A steward dashboard**: the review queue with a breakdown of every score, customers as a
169
+ network of links, merge and split with previews, pinned golden-record values, decisions that
170
+ train the next run, approval by a second steward, the model's accuracy and the data's quality
171
+ - every chart with an explanation of how it is calculated and what good looks like.
172
+ - API keys with roles, an audit log of every steward action.
173
+
174
+ ```bash
175
+ docker compose -f server/docker-compose.yml up -d --build db api # then see server/README.md
176
+ ```
177
+
178
+ ![The review queue: two records side by side, and why the model scored them 53%](https://raw.githubusercontent.com/andrewgoodchild/custmatch/main/docs/images/dashboard-review.png)
179
+
180
+ *The review queue shows the pairs the model was least sure of, closest calls first. Both records
181
+ here say Jane, but the web sign-up has only an email and a postcode, the app record a
182
+ surname, a birth date and a mobile, and they share nothing else. The chart explains the 53%:
183
+ from the base rate, the agreeing first name pushes the score up, and the missing surname and
184
+ unshared contacts pull it back below the threshold (the yellow line), where the model kept them
185
+ apart. A steward answers* Same person *or* Different
186
+ people*; the answer applies at once and teaches the next run.*
187
+
188
+ ![The model's accuracy against the threshold, and how its scores are spread](https://raw.githubusercontent.com/andrewgoodchild/custmatch/main/docs/images/dashboard-model.png)
189
+
190
+ *How well the model is doing, and where its line is drawn. Each pair of records gets a score from
191
+ 0 to 1, and the* threshold *is the score above which two records count as one person. The left
192
+ chart moves that threshold from 0 to 1: raise it and* precision *(blue - of the records joined,
193
+ the share that truly are one person) rises while* recall *(green - of the true matches, the share
194
+ found) falls. The dashed line is where custmatch set it, the point where the two balance best
195
+ (F1, white), judged on whole customers: here an estimated 91.65% precision and 93.51% recall.
196
+ The right chart counts the pairs at each score. A confident model puts most pairs near 0
197
+ (clearly different) or near 1 (clearly the same); the few in the middle are the ones it sends
198
+ to the review queue.*
199
+
200
+ [server/README.md](https://github.com/andrewgoodchild/custmatch/blob/main/server/README.md) is the operator guide;
201
+ [the report's chapter on the service](https://github.com/andrewgoodchild/custmatch/blob/main/docs/report/12-the-service.md) what it was measured at.
202
+
203
+ ## Features
204
+
205
+ **Key:** ✅ done and measured · ◐ opt-in · ❌ not built yet.
206
+
207
+ ### Matching
208
+
209
+ | Feature | Status | Notes |
210
+ | --- | --- | --- |
211
+ | Learns from your labelled pairs; the threshold is chosen on whole customers, which stops chaining | ✅ | The project's central finding |
212
+ | Active learning: it picks the pairs worth labelling | ✅ | Essential on sparse records |
213
+ | Country profiles: US, AU, NZ, GB, CA, several per file | ✅ | Phones, dates, emails, postcodes, SSN checks, nicknames |
214
+ | Name variation: nicknames, typos, swapped names, accents, Chinese and Arabic script | ✅ | |
215
+ | Junk and placeholder contacts ignored (test@test.com, 0400 000 000) | ✅ | Bought "gold" numbers kept; placeholder names and addresses opt-in |
216
+ | Sparse records: a first name and an email or mobile | ✅ | |
217
+ | Households: couples, families, shared phones and emails | ✅ | |
218
+ | Large entities that share every key (an organisation with hundreds of records) | ✅ | Checked by the model, not skipped as junk |
219
+ | Suspect records flagged for review: profanity, disposable or random-looking emails | ✅ | `suspect_records.csv`; never changes a match |
220
+ | Twins kept apart | ◐ | `"twin_guard"` |
221
+ | Safe contact details: a precision target, and contacts taken only through certain links | ◐ | `"min_precision"`, `"contact_confidence"`, `contacts.csv` |
222
+ | One record per person in a duplicate-free source (a voter roll, a loyalty scheme) | ◐ | `"unique_sources"` |
223
+ | Trusted IDs: same loyalty or customer number, same person | ✅ | `"trusted_ids"` |
224
+ | Record dates: moves, name changes, reissued mobiles | ◐ | `"record_dates"` |
225
+ | Cookie and device-ID stitching, with shared-device and call-centre guards | ✅ | `"identity_events"` |
226
+ | A forest and CatBoost averaged, for hard and sparse files | ◐ | `"matcher_model": "forest_catboost"` |
227
+ | Address verification against official files (G-NAF, USPS) | ❌ | Needs licensed reference data |
228
+ | Business (B2B) matching: companies and their contacts | ❌ | People and households |
229
+
230
+ ### Golden records
231
+
232
+ | Feature | Status | Notes |
233
+ | --- | --- | --- |
234
+ | A golden record per customer, with survivorship rules per field | ✅ | Most recent, most common, preferred sources with a fallback |
235
+
236
+ ### Operating over time
237
+
238
+ | Feature | Status | Notes |
239
+ | --- | --- | --- |
240
+ | New, updated and deleted records without a full rerun | ✅ | `custmatch add`, `--delete` |
241
+ | Stable customer IDs with a history of merges and splits | ✅ | Records stop swapping between people as data arrives |
242
+ | Review queue for uncertain pairs | ✅ | `review.csv` |
243
+ | Steward overrides and unmerge, kept across runs | ✅ | `"overrides"` |
244
+ | Monitoring: match rates and drift across runs | ✅ | `custmatch monitor` |
245
+ | Real-time matching and lookup | ✅ | custmatch-server: `POST /match`, `GET /records/...` |
246
+ | Steward dashboard, scheduled runs, health checks and alerts | ✅ | custmatch-server |
247
+
248
+ ### Trust and privacy
249
+
250
+ | Feature | Status | Notes |
251
+ | --- | --- | --- |
252
+ | Why two records were, or were not, matched | ✅ | `links.csv`, `custmatch explain` |
253
+ | Matching on hashed emails, phones and IDs | ✅ | `"hash_fields"` |
254
+ | Result files without names or contacts: record keys and evidence only | ◐ | `"minimal_outputs"`; the saved state that `custmatch add` needs still holds the records |
255
+ | Erasure requests | ✅ | `custmatch forget` |
256
+
257
+ ## Documentation
258
+
259
+ | | |
260
+ | --- | --- |
261
+ | [docs/guide.md](https://github.com/andrewgoodchild/custmatch/blob/main/docs/guide.md) | Using custmatch: how it works, step by step; your config, labels and a run; running it over time; every option; a glossary |
262
+ | [docs/report/](https://github.com/andrewgoodchild/custmatch/tree/main/docs/report) | The research: what was found, then one chapter per question - whole customers not pairs, which matcher, labels, fake contacts and households, blocking and scale, golden records, CDPs, other tools - and reproductions of published work |
263
+ | [docs/reproducing.md](https://github.com/andrewgoodchild/custmatch/blob/main/docs/reproducing.md) | Rebuilding the datasets, rerunning every experiment, the test suite, and reproduction traps |
264
+ | [server/README.md](https://github.com/andrewgoodchild/custmatch/blob/main/server/README.md) | Running custmatch-server: start it, load data, keys, schedules, the dashboard, monitoring, backup, upgrades |
265
+
266
+ ## Tests
267
+
268
+ ```bash
269
+ pip install -e '.[test]' && pytest tests/
270
+ pip install -e ./server httpx # the service's tests too
271
+ ```
272
+
273
+ The suite includes a check that every number quoted in the docs still matches `results/*.json`.
274
+
275
+ ## License
276
+
277
+ [Apache 2.0](https://github.com/andrewgoodchild/custmatch/blob/main/LICENSE).