iporigin 1.0.0__tar.gz → 1.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. iporigin-1.1.1/CHANGELOG.md +32 -0
  2. iporigin-1.1.1/MANIFEST.in +7 -0
  3. iporigin-1.1.1/NOTICE +78 -0
  4. {iporigin-1.0.0/src/iporigin.egg-info → iporigin-1.1.1}/PKG-INFO +41 -14
  5. {iporigin-1.0.0 → iporigin-1.1.1}/README.md +39 -13
  6. {iporigin-1.0.0 → iporigin-1.1.1}/pyproject.toml +4 -1
  7. {iporigin-1.0.0 → iporigin-1.1.1}/src/iporigin/__init__.py +1 -1
  8. iporigin-1.1.1/src/iporigin/data/ranges.bin +0 -0
  9. {iporigin-1.0.0 → iporigin-1.1.1/src/iporigin.egg-info}/PKG-INFO +41 -14
  10. {iporigin-1.0.0 → iporigin-1.1.1}/src/iporigin.egg-info/SOURCES.txt +7 -1
  11. {iporigin-1.0.0 → iporigin-1.1.1}/tests/test_iporigin.py +145 -0
  12. iporigin-1.1.1/tools/build_dataset.py +231 -0
  13. iporigin-1.1.1/tools/bump_version.py +84 -0
  14. iporigin-1.1.1/tools/sources.py +321 -0
  15. iporigin-1.0.0/src/iporigin/data/ranges.bin +0 -0
  16. {iporigin-1.0.0 → iporigin-1.1.1}/LICENSE +0 -0
  17. {iporigin-1.0.0 → iporigin-1.1.1}/setup.cfg +0 -0
  18. {iporigin-1.0.0 → iporigin-1.1.1}/src/iporigin/_data.py +0 -0
  19. {iporigin-1.0.0 → iporigin-1.1.1}/src/iporigin/cli.py +0 -0
  20. {iporigin-1.0.0 → iporigin-1.1.1}/src/iporigin/core.py +0 -0
  21. {iporigin-1.0.0 → iporigin-1.1.1}/src/iporigin/online.py +0 -0
  22. {iporigin-1.0.0 → iporigin-1.1.1}/src/iporigin/py.typed +0 -0
  23. {iporigin-1.0.0 → iporigin-1.1.1}/src/iporigin.egg-info/dependency_links.txt +0 -0
  24. {iporigin-1.0.0 → iporigin-1.1.1}/src/iporigin.egg-info/entry_points.txt +0 -0
  25. {iporigin-1.0.0 → iporigin-1.1.1}/src/iporigin.egg-info/requires.txt +0 -0
  26. {iporigin-1.0.0 → iporigin-1.1.1}/src/iporigin.egg-info/top_level.txt +0 -0
@@ -0,0 +1,32 @@
1
+ # Changelog
2
+
3
+ ## 1.1.0
4
+
5
+ - 11 more providers, from two repositories used with their maintainers'
6
+ permission: Cogent, DataCamp, Contabo, Vercel, CDN77, GleSYS, Scalaxy,
7
+ GTHost, Melbicom, BuyVM, BunnyCDN, plus extra Akamai ranges.
8
+ 37 providers to 48; 33,647 ranges to 38,379; IPv4 coverage 248M to 287M
9
+ addresses.
10
+ - NOTICE records every data source, its licence, and — where a repository
11
+ has none — the permission it is used under.
12
+ - `jhassine/server-ip-addresses` was measured and deliberately left out:
13
+ every one of 211,616 sampled addresses was already covered, so it adds
14
+ nothing but another endpoint that can break the weekly rebuild.
15
+
16
+ ## 1.0.0
17
+
18
+ First release.
19
+
20
+ - Offline classification of IPv4 and IPv6 addresses into `hosting`, `cdn`,
21
+ `vpn`, `tor`, `bot`, `reserved` and `unknown`, from a table of 33,647
22
+ disjoint ranges covering 37 providers.
23
+ - Ten providers come from their own published feeds; the rest — Hetzner,
24
+ OVHcloud, Akamai, Mullvad, ProtonVPN, Tor, Googlebot and others — from two
25
+ CC0 community lists, because those providers publish nothing
26
+ machine-readable.
27
+ - `classify`, `is_datacenter`, `classify_many`, `dataset_info`, plus the
28
+ `DATACENTER_KINDS` and `ANONYMIZER_KINDS` sets.
29
+ - `iporigin` command line tool, reads addresses from arguments or stdin.
30
+ - Optional live lookup in `iporigin.online` for VPN providers the offline
31
+ table does not reach.
32
+ - No runtime dependencies.
@@ -0,0 +1,7 @@
1
+ include LICENSE
2
+ include NOTICE
3
+ include CHANGELOG.md
4
+ include README.md
5
+ include src/iporigin/data/ranges.bin
6
+ recursive-include tools *.py
7
+ recursive-include tests *.py
iporigin-1.1.1/NOTICE ADDED
@@ -0,0 +1,78 @@
1
+ Third-party data sources
2
+ ========================
3
+
4
+ iporigin's bundled dataset (src/iporigin/data/ranges.bin) is compiled from
5
+ the sources listed in tools/sources.py. iporigin's own code is MIT; the data
6
+ comes from elsewhere and is recorded here so that anyone redistributing this
7
+ package knows exactly what they are redistributing.
8
+
9
+
10
+ 1. Published by the provider
11
+ ----------------------------
12
+
13
+ Amazon AWS, Google, Google Cloud, Microsoft Azure, DigitalOcean, Linode,
14
+ Vultr, Oracle Cloud, GitHub, Cloudflare, Fastly.
15
+
16
+ Each is the operator's own published feed, issued for the purpose of being
17
+ consumed programmatically. No separate permission is needed or claimed.
18
+
19
+
20
+ 2. CC0-1.0 community aggregations
21
+ ---------------------------------
22
+
23
+ rezmoss/cloud-provider-ip-addresses CC0-1.0
24
+ https://github.com/rezmoss/cloud-provider-ip-addresses
25
+
26
+ lord-alfred/ipranges CC0-1.0
27
+ https://github.com/lord-alfred/ipranges
28
+
29
+ CC0 is a public-domain dedication, so these carry no conditions.
30
+
31
+
32
+ 3. Used with the maintainer's permission
33
+ ----------------------------------------
34
+
35
+ These repositories publish no licence file. Under default copyright that
36
+ means all rights reserved, and they would otherwise be unusable here.
37
+ They are included because permission was obtained from each maintainer
38
+ directly, by Serdar Akarca of Yuix Networks, prior to the 1.1.0 release
39
+ (2026-09-13). The maintainers' position as relayed was that the lists were
40
+ published for general use.
41
+
42
+ 123jjck/cdn-ip-ranges
43
+ https://github.com/123jjck/cdn-ip-ranges
44
+ Used for: Cogent, DataCamp, Contabo, Vercel, CDN77, GleSYS, Scalaxy,
45
+ GTHost, Melbicom, BuyVM, BunnyCDN
46
+
47
+ SecOps-Institute/Akamai-ASN-and-IPs-List
48
+ https://github.com/SecOps-Institute/Akamai-ASN-and-IPs-List
49
+ Used for: additional Akamai ranges
50
+
51
+ TODO for the maintainer: attach the written record of each permission below
52
+ (issue link, PR link or email date). Until that is here, the grant rests on
53
+ a verbal account, which is thinner than a licence and worth replacing. The
54
+ cleanest fix is a PR to each repository adding a licence file — that removes
55
+ the question for everyone downstream, not just for us.
56
+
57
+ 123jjck/cdn-ip-ranges evidence: <to be added>
58
+ SecOps-Institute/Akamai-ASN-and-IPs-List evidence: <to be added>
59
+
60
+
61
+ Sources deliberately excluded
62
+ -----------------------------
63
+
64
+ Measured and left out because they add nothing, not for licensing reasons:
65
+
66
+ jhassine/server-ip-addresses
67
+ 52,772 prefixes covering 227M addresses. Every one of 211,616 sampled
68
+ addresses was already covered by the tiers above — it is a subset of
69
+ the provider feeds it was itself built from.
70
+
71
+ Pymmdrza/Datacenter_List_DataBase_IP
72
+ 1,280 new addresses beyond Hetzner's other sources.
73
+
74
+ SM443/IP-Prefix-List
75
+ No parseable CIDR content at the referenced paths.
76
+
77
+ A source that adds nothing still costs something: another endpoint that can
78
+ break the weekly rebuild.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: iporigin
3
- Version: 1.0.0
3
+ Version: 1.1.1
4
4
  Summary: Is this IP a datacenter, VPN, Tor exit or a home connection? Offline lookup, no API key.
5
5
  Author-email: Yuix Networks <info@yuix.org>
6
6
  License: MIT
@@ -28,6 +28,7 @@ Classifier: Typing :: Typed
28
28
  Requires-Python: >=3.8
29
29
  Description-Content-Type: text/markdown
30
30
  License-File: LICENSE
31
+ License-File: NOTICE
31
32
  Provides-Extra: dev
32
33
  Requires-Dist: pytest>=7; extra == "dev"
33
34
  Dynamic: license-file
@@ -47,7 +48,7 @@ False
47
48
  ```
48
49
 
49
50
  No network calls. No signup. No runtime dependencies. The answer comes from
50
- a bundled table of 33,000 ranges covering 37 hosting providers, CDNs,
51
+ a bundled table of 38,000 ranges covering 48 hosting providers, CDNs,
51
52
  consumer VPNs, Tor and declared crawlers.
52
53
 
53
54
  ## Install
@@ -154,7 +155,7 @@ import this module on purpose. No key required.
154
155
 
155
156
  ## What is in the dataset
156
157
 
157
- 37 providers, in two tiers.
158
+ 48 providers, in three tiers.
158
159
 
159
160
  **Published by the provider.** The authoritative tier — each of these is the
160
161
  company's own feed, fetched at build time:
@@ -186,19 +187,44 @@ Covering Hetzner, OVHcloud, Scaleway, Alibaba Cloud, Leaseweb, UpCloud, IBM
186
187
  Cloud, Huawei Cloud, Tencent Cloud, Rackspace, Akamai, Gcore, Mullvad,
187
188
  ProtonVPN, Apple Private Relay, Tor, and ten declared crawlers.
188
189
 
189
- Both are CC0, which is why these two and not the half-dozen other repos
190
- covering the same ground. Redistributing an unlicensed list inside an MIT
191
- package is not something a dependency should ask of the people who install
192
- it.
190
+ Both are CC0, a public-domain dedication, so they carry no conditions.
193
191
 
194
- About 449,000 published prefixes collapse into 33,647 disjoint ranges
195
- (17,169 IPv4, 16,478 IPv6). Rebuild it yourself at any time:
192
+ **Used with the maintainer's permission.** Two further repositories publish
193
+ no licence file, which normally rules them out. They are included because
194
+ permission was obtained from each maintainer directly — see
195
+ [NOTICE](NOTICE), which records what was granted and by whom:
196
+
197
+ - [`123jjck/cdn-ip-ranges`](https://github.com/123jjck/cdn-ip-ranges) —
198
+ Cogent, DataCamp, Contabo, Vercel, CDN77, GleSYS, Scalaxy, GTHost,
199
+ Melbicom, BuyVM, BunnyCDN
200
+ - [`SecOps-Institute/Akamai-ASN-and-IPs-List`](https://github.com/SecOps-Institute/Akamai-ASN-and-IPs-List) —
201
+ additional Akamai ranges
202
+
203
+ Only what is additive is taken. `jhassine/server-ip-addresses` is the
204
+ best-known list of this kind and is **not** included: 52,772 prefixes,
205
+ 227M addresses, and every one of 211,616 sampled addresses was already
206
+ covered. It is a subset of the provider feeds it was itself built from, and
207
+ a source that adds nothing is still another endpoint that can break the
208
+ weekly rebuild.
209
+
210
+ About 454,000 published prefixes collapse into roughly 38,400 disjoint
211
+ ranges (about 21,900 IPv4 and 16,500 IPv6) covering 287 million IPv4
212
+ addresses. The exact figures move every week with the feeds; the build
213
+ prints them. Rebuild it yourself at any time:
196
214
 
197
215
  ```
198
216
  python tools/build_dataset.py
199
217
  ```
200
218
 
201
- A GitHub Action re-runs that weekly and opens a PR when the ranges move.
219
+ A GitHub Action re-runs that weekly and commits the result when the ranges
220
+ move, and a second one cuts a patch release on the 6th of each month if the
221
+ dataset changed since the last release — so `pip install --upgrade iporigin`
222
+ is never more than about a month behind the feeds, and a quiet month
223
+ produces no release rather than a version whose only content is a new
224
+ number. The build refuses to replace the committed dataset if it shrinks by
225
+ more than 20% — a feed that starts answering with an empty body looks
226
+ exactly like a provider giving up its address space, and nothing else
227
+ would catch it. Pass `--allow-shrink` when the drop is genuine.
202
228
 
203
229
  ### Known gaps
204
230
 
@@ -209,8 +235,8 @@ Being explicit about these is more useful than pretending they are not there:
209
235
  - **Some provider-owned addresses** sit outside the ranges the provider
210
236
  publishes. `1.1.1.1` is Cloudflare's resolver but is not in Cloudflare's
211
237
  published edge list, so it comes back `unknown`.
212
- - **Second-hand data is second-hand.** The community tier is as good as
213
- those repos are, and they are not the provider speaking.
238
+ - **Second-hand data is second-hand.** Tiers 2 and 3 are as good as those
239
+ repos are, and they are not the provider speaking.
214
240
  - The data is **as accurate as the feeds**. A range reassigned yesterday is
215
241
  wrong until the next rebuild.
216
242
 
@@ -237,8 +263,9 @@ Python 3.8+. No dependencies.
237
263
 
238
264
  ## License
239
265
 
240
- MIT. The compiled dataset is derived from the providers' own public feeds,
241
- each published for exactly this purpose.
266
+ The code is MIT. The bundled dataset comes from third parties; every source,
267
+ its licence, and — where there is none — the permission it is used under are
268
+ recorded in [NOTICE](NOTICE).
242
269
 
243
270
  ---
244
271
 
@@ -13,7 +13,7 @@ False
13
13
  ```
14
14
 
15
15
  No network calls. No signup. No runtime dependencies. The answer comes from
16
- a bundled table of 33,000 ranges covering 37 hosting providers, CDNs,
16
+ a bundled table of 38,000 ranges covering 48 hosting providers, CDNs,
17
17
  consumer VPNs, Tor and declared crawlers.
18
18
 
19
19
  ## Install
@@ -120,7 +120,7 @@ import this module on purpose. No key required.
120
120
 
121
121
  ## What is in the dataset
122
122
 
123
- 37 providers, in two tiers.
123
+ 48 providers, in three tiers.
124
124
 
125
125
  **Published by the provider.** The authoritative tier — each of these is the
126
126
  company's own feed, fetched at build time:
@@ -152,19 +152,44 @@ Covering Hetzner, OVHcloud, Scaleway, Alibaba Cloud, Leaseweb, UpCloud, IBM
152
152
  Cloud, Huawei Cloud, Tencent Cloud, Rackspace, Akamai, Gcore, Mullvad,
153
153
  ProtonVPN, Apple Private Relay, Tor, and ten declared crawlers.
154
154
 
155
- Both are CC0, which is why these two and not the half-dozen other repos
156
- covering the same ground. Redistributing an unlicensed list inside an MIT
157
- package is not something a dependency should ask of the people who install
158
- it.
155
+ Both are CC0, a public-domain dedication, so they carry no conditions.
159
156
 
160
- About 449,000 published prefixes collapse into 33,647 disjoint ranges
161
- (17,169 IPv4, 16,478 IPv6). Rebuild it yourself at any time:
157
+ **Used with the maintainer's permission.** Two further repositories publish
158
+ no licence file, which normally rules them out. They are included because
159
+ permission was obtained from each maintainer directly — see
160
+ [NOTICE](NOTICE), which records what was granted and by whom:
161
+
162
+ - [`123jjck/cdn-ip-ranges`](https://github.com/123jjck/cdn-ip-ranges) —
163
+ Cogent, DataCamp, Contabo, Vercel, CDN77, GleSYS, Scalaxy, GTHost,
164
+ Melbicom, BuyVM, BunnyCDN
165
+ - [`SecOps-Institute/Akamai-ASN-and-IPs-List`](https://github.com/SecOps-Institute/Akamai-ASN-and-IPs-List) —
166
+ additional Akamai ranges
167
+
168
+ Only what is additive is taken. `jhassine/server-ip-addresses` is the
169
+ best-known list of this kind and is **not** included: 52,772 prefixes,
170
+ 227M addresses, and every one of 211,616 sampled addresses was already
171
+ covered. It is a subset of the provider feeds it was itself built from, and
172
+ a source that adds nothing is still another endpoint that can break the
173
+ weekly rebuild.
174
+
175
+ About 454,000 published prefixes collapse into roughly 38,400 disjoint
176
+ ranges (about 21,900 IPv4 and 16,500 IPv6) covering 287 million IPv4
177
+ addresses. The exact figures move every week with the feeds; the build
178
+ prints them. Rebuild it yourself at any time:
162
179
 
163
180
  ```
164
181
  python tools/build_dataset.py
165
182
  ```
166
183
 
167
- A GitHub Action re-runs that weekly and opens a PR when the ranges move.
184
+ A GitHub Action re-runs that weekly and commits the result when the ranges
185
+ move, and a second one cuts a patch release on the 6th of each month if the
186
+ dataset changed since the last release — so `pip install --upgrade iporigin`
187
+ is never more than about a month behind the feeds, and a quiet month
188
+ produces no release rather than a version whose only content is a new
189
+ number. The build refuses to replace the committed dataset if it shrinks by
190
+ more than 20% — a feed that starts answering with an empty body looks
191
+ exactly like a provider giving up its address space, and nothing else
192
+ would catch it. Pass `--allow-shrink` when the drop is genuine.
168
193
 
169
194
  ### Known gaps
170
195
 
@@ -175,8 +200,8 @@ Being explicit about these is more useful than pretending they are not there:
175
200
  - **Some provider-owned addresses** sit outside the ranges the provider
176
201
  publishes. `1.1.1.1` is Cloudflare's resolver but is not in Cloudflare's
177
202
  published edge list, so it comes back `unknown`.
178
- - **Second-hand data is second-hand.** The community tier is as good as
179
- those repos are, and they are not the provider speaking.
203
+ - **Second-hand data is second-hand.** Tiers 2 and 3 are as good as those
204
+ repos are, and they are not the provider speaking.
180
205
  - The data is **as accurate as the feeds**. A range reassigned yesterday is
181
206
  wrong until the next rebuild.
182
207
 
@@ -203,8 +228,9 @@ Python 3.8+. No dependencies.
203
228
 
204
229
  ## License
205
230
 
206
- MIT. The compiled dataset is derived from the providers' own public feeds,
207
- each published for exactly this purpose.
231
+ The code is MIT. The bundled dataset comes from third parties; every source,
232
+ its licence, and — where there is none — the permission it is used under are
233
+ recorded in [NOTICE](NOTICE).
208
234
 
209
235
  ---
210
236
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "iporigin"
7
- version = "1.0.0"
7
+ version = "1.1.1"
8
8
  description = "Is this IP a datacenter, VPN, Tor exit or a home connection? Offline lookup, no API key."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.8"
@@ -49,6 +49,9 @@ iporigin = "iporigin.cli:main"
49
49
  [project.optional-dependencies]
50
50
  dev = ["pytest>=7"]
51
51
 
52
+ [tool.setuptools]
53
+ license-files = ["LICENSE", "NOTICE"]
54
+
52
55
  [tool.setuptools.packages.find]
53
56
  where = ["src"]
54
57
 
@@ -29,7 +29,7 @@ from .core import (
29
29
  is_datacenter,
30
30
  )
31
31
 
32
- __version__ = "1.0.0"
32
+ __version__ = "1.1.1"
33
33
 
34
34
  __all__ = [
35
35
  "ANONYMIZER_KINDS",
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: iporigin
3
- Version: 1.0.0
3
+ Version: 1.1.1
4
4
  Summary: Is this IP a datacenter, VPN, Tor exit or a home connection? Offline lookup, no API key.
5
5
  Author-email: Yuix Networks <info@yuix.org>
6
6
  License: MIT
@@ -28,6 +28,7 @@ Classifier: Typing :: Typed
28
28
  Requires-Python: >=3.8
29
29
  Description-Content-Type: text/markdown
30
30
  License-File: LICENSE
31
+ License-File: NOTICE
31
32
  Provides-Extra: dev
32
33
  Requires-Dist: pytest>=7; extra == "dev"
33
34
  Dynamic: license-file
@@ -47,7 +48,7 @@ False
47
48
  ```
48
49
 
49
50
  No network calls. No signup. No runtime dependencies. The answer comes from
50
- a bundled table of 33,000 ranges covering 37 hosting providers, CDNs,
51
+ a bundled table of 38,000 ranges covering 48 hosting providers, CDNs,
51
52
  consumer VPNs, Tor and declared crawlers.
52
53
 
53
54
  ## Install
@@ -154,7 +155,7 @@ import this module on purpose. No key required.
154
155
 
155
156
  ## What is in the dataset
156
157
 
157
- 37 providers, in two tiers.
158
+ 48 providers, in three tiers.
158
159
 
159
160
  **Published by the provider.** The authoritative tier — each of these is the
160
161
  company's own feed, fetched at build time:
@@ -186,19 +187,44 @@ Covering Hetzner, OVHcloud, Scaleway, Alibaba Cloud, Leaseweb, UpCloud, IBM
186
187
  Cloud, Huawei Cloud, Tencent Cloud, Rackspace, Akamai, Gcore, Mullvad,
187
188
  ProtonVPN, Apple Private Relay, Tor, and ten declared crawlers.
188
189
 
189
- Both are CC0, which is why these two and not the half-dozen other repos
190
- covering the same ground. Redistributing an unlicensed list inside an MIT
191
- package is not something a dependency should ask of the people who install
192
- it.
190
+ Both are CC0, a public-domain dedication, so they carry no conditions.
193
191
 
194
- About 449,000 published prefixes collapse into 33,647 disjoint ranges
195
- (17,169 IPv4, 16,478 IPv6). Rebuild it yourself at any time:
192
+ **Used with the maintainer's permission.** Two further repositories publish
193
+ no licence file, which normally rules them out. They are included because
194
+ permission was obtained from each maintainer directly — see
195
+ [NOTICE](NOTICE), which records what was granted and by whom:
196
+
197
+ - [`123jjck/cdn-ip-ranges`](https://github.com/123jjck/cdn-ip-ranges) —
198
+ Cogent, DataCamp, Contabo, Vercel, CDN77, GleSYS, Scalaxy, GTHost,
199
+ Melbicom, BuyVM, BunnyCDN
200
+ - [`SecOps-Institute/Akamai-ASN-and-IPs-List`](https://github.com/SecOps-Institute/Akamai-ASN-and-IPs-List) —
201
+ additional Akamai ranges
202
+
203
+ Only what is additive is taken. `jhassine/server-ip-addresses` is the
204
+ best-known list of this kind and is **not** included: 52,772 prefixes,
205
+ 227M addresses, and every one of 211,616 sampled addresses was already
206
+ covered. It is a subset of the provider feeds it was itself built from, and
207
+ a source that adds nothing is still another endpoint that can break the
208
+ weekly rebuild.
209
+
210
+ About 454,000 published prefixes collapse into roughly 38,400 disjoint
211
+ ranges (about 21,900 IPv4 and 16,500 IPv6) covering 287 million IPv4
212
+ addresses. The exact figures move every week with the feeds; the build
213
+ prints them. Rebuild it yourself at any time:
196
214
 
197
215
  ```
198
216
  python tools/build_dataset.py
199
217
  ```
200
218
 
201
- A GitHub Action re-runs that weekly and opens a PR when the ranges move.
219
+ A GitHub Action re-runs that weekly and commits the result when the ranges
220
+ move, and a second one cuts a patch release on the 6th of each month if the
221
+ dataset changed since the last release — so `pip install --upgrade iporigin`
222
+ is never more than about a month behind the feeds, and a quiet month
223
+ produces no release rather than a version whose only content is a new
224
+ number. The build refuses to replace the committed dataset if it shrinks by
225
+ more than 20% — a feed that starts answering with an empty body looks
226
+ exactly like a provider giving up its address space, and nothing else
227
+ would catch it. Pass `--allow-shrink` when the drop is genuine.
202
228
 
203
229
  ### Known gaps
204
230
 
@@ -209,8 +235,8 @@ Being explicit about these is more useful than pretending they are not there:
209
235
  - **Some provider-owned addresses** sit outside the ranges the provider
210
236
  publishes. `1.1.1.1` is Cloudflare's resolver but is not in Cloudflare's
211
237
  published edge list, so it comes back `unknown`.
212
- - **Second-hand data is second-hand.** The community tier is as good as
213
- those repos are, and they are not the provider speaking.
238
+ - **Second-hand data is second-hand.** Tiers 2 and 3 are as good as those
239
+ repos are, and they are not the provider speaking.
214
240
  - The data is **as accurate as the feeds**. A range reassigned yesterday is
215
241
  wrong until the next rebuild.
216
242
 
@@ -237,8 +263,9 @@ Python 3.8+. No dependencies.
237
263
 
238
264
  ## License
239
265
 
240
- MIT. The compiled dataset is derived from the providers' own public feeds,
241
- each published for exactly this purpose.
266
+ The code is MIT. The bundled dataset comes from third parties; every source,
267
+ its licence, and — where there is none — the permission it is used under are
268
+ recorded in [NOTICE](NOTICE).
242
269
 
243
270
  ---
244
271
 
@@ -1,4 +1,7 @@
1
+ CHANGELOG.md
1
2
  LICENSE
3
+ MANIFEST.in
4
+ NOTICE
2
5
  README.md
3
6
  pyproject.toml
4
7
  src/iporigin/__init__.py
@@ -14,4 +17,7 @@ src/iporigin.egg-info/entry_points.txt
14
17
  src/iporigin.egg-info/requires.txt
15
18
  src/iporigin.egg-info/top_level.txt
16
19
  src/iporigin/data/ranges.bin
17
- tests/test_iporigin.py
20
+ tests/test_iporigin.py
21
+ tools/build_dataset.py
22
+ tools/bump_version.py
23
+ tools/sources.py
@@ -363,3 +363,148 @@ def _stub_api(monkeypatch, payload):
363
363
  yield Response()
364
364
 
365
365
  monkeypatch.setattr(online.urllib.request, "urlopen", fake_urlopen)
366
+
367
+
368
+ # --- the build guard -------------------------------------------------------
369
+ #
370
+ # The weekly refresh commits straight to main, so this check is the only
371
+ # thing standing between a broken upstream feed and a shipped dataset.
372
+
373
+
374
+ def _build_module():
375
+ import importlib.util
376
+ import pathlib
377
+
378
+ path = pathlib.Path(__file__).parent.parent / "tools" / "build_dataset.py"
379
+ spec = importlib.util.spec_from_file_location("build_dataset", path)
380
+ module = importlib.util.module_from_spec(spec)
381
+ spec.loader.exec_module(module)
382
+ return module
383
+
384
+
385
+ def test_a_large_shrink_is_refused():
386
+ build = _build_module()
387
+ assert build.shrink_complaint(30000, 5000) is not None
388
+
389
+
390
+ def test_a_normal_week_is_allowed():
391
+ """Feeds move by a few percent all the time."""
392
+ build = _build_module()
393
+ assert build.shrink_complaint(30000, 29000) is None
394
+ assert build.shrink_complaint(30000, 31500) is None
395
+
396
+
397
+ def test_growth_is_never_refused():
398
+ build = _build_module()
399
+ assert build.shrink_complaint(1000, 100000) is None
400
+
401
+
402
+ def test_the_first_ever_build_is_allowed():
403
+ """No committed dataset to compare against."""
404
+ build = _build_module()
405
+ assert build.shrink_complaint(None, 10) is None
406
+ assert build.shrink_complaint(0, 10) is None
407
+
408
+
409
+ def test_the_threshold_is_the_documented_one():
410
+ build = _build_module()
411
+ assert build.MAX_SHRINK == 0.20
412
+ # Exactly at the limit passes; a hair below does not.
413
+ assert build.shrink_complaint(1000, 800) is None
414
+ assert build.shrink_complaint(1000, 799) is not None
415
+
416
+
417
+ def test_the_complaint_says_what_to_do():
418
+ build = _build_module()
419
+ message = build.shrink_complaint(30000, 5000)
420
+ assert "--allow-shrink" in message
421
+ assert "5000" in message and "30000" in message
422
+
423
+
424
+ def test_existing_range_count_reads_the_committed_dataset():
425
+ build = _build_module()
426
+ count = build.existing_range_count()
427
+ assert count and count > 1000
428
+
429
+
430
+ def test_existing_range_count_survives_a_missing_or_corrupt_file(tmp_path, monkeypatch):
431
+ build = _build_module()
432
+ monkeypatch.setattr(build, "OUT", tmp_path / "absent.bin")
433
+ assert build.existing_range_count() is None
434
+
435
+ corrupt = tmp_path / "ranges.bin"
436
+ corrupt.write_bytes(b"nope")
437
+ monkeypatch.setattr(build, "OUT", corrupt)
438
+ assert build.existing_range_count() is None
439
+
440
+
441
+ # --- the release bump ------------------------------------------------------
442
+ #
443
+ # Releases are cut by a scheduled job, so a bump that goes half-way lands on
444
+ # PyPI before anyone looks at it.
445
+
446
+
447
+ def _bump_module():
448
+ import importlib.util
449
+ import pathlib
450
+
451
+ path = pathlib.Path(__file__).parent.parent / "tools" / "bump_version.py"
452
+ spec = importlib.util.spec_from_file_location("bump_version", path)
453
+ module = importlib.util.module_from_spec(spec)
454
+ spec.loader.exec_module(module)
455
+ return module
456
+
457
+
458
+ def test_the_two_declared_versions_agree():
459
+ """pyproject.toml and __init__.py each hold the version separately."""
460
+ bump = _bump_module()
461
+ declared, exported = bump.read_versions(
462
+ bump.PYPROJECT.read_text(), bump.INIT.read_text()
463
+ )
464
+ assert declared == exported == iporigin.__version__
465
+
466
+
467
+ def test_patch_bump():
468
+ bump = _bump_module()
469
+ assert bump.next_patch("1.1.0") == "1.1.1"
470
+ assert bump.next_patch("1.1.9") == "1.1.10"
471
+ assert bump.next_patch("0.0.0") == "0.0.1"
472
+
473
+
474
+ def test_a_non_numeric_version_is_left_alone():
475
+ bump = _bump_module()
476
+ for bad in ("2.0.0rc1", "1.1", "1.1.0.post1", "v1.1.0"):
477
+ with pytest.raises(ValueError):
478
+ bump.next_patch(bad)
479
+
480
+
481
+ def test_bump_rewrites_both_files():
482
+ bump = _bump_module()
483
+ pyproject, init, new = bump.bump(
484
+ 'name = "iporigin"\nversion = "1.2.3"\n', '__version__ = "1.2.3"\n'
485
+ )
486
+ assert new == "1.2.4"
487
+ assert 'version = "1.2.4"' in pyproject
488
+ assert '__version__ = "1.2.4"' in init
489
+
490
+
491
+ def test_bump_refuses_when_the_files_disagree():
492
+ bump = _bump_module()
493
+ with pytest.raises(ValueError, match="fix that before releasing"):
494
+ bump.bump('version = "1.2.3"\n', '__version__ = "1.0.0"\n')
495
+
496
+
497
+ def test_bump_refuses_when_a_declaration_is_missing():
498
+ bump = _bump_module()
499
+ with pytest.raises(ValueError, match="pyproject"):
500
+ bump.read_versions("name = \"iporigin\"\n", '__version__ = "1.0.0"\n')
501
+ with pytest.raises(ValueError, match="__init__"):
502
+ bump.read_versions('version = "1.0.0"\n', "x = 1\n")
503
+
504
+
505
+ def test_bump_only_touches_the_version_line():
506
+ """pyproject holds other quoted values; a greedy substitution eats them."""
507
+ bump = _bump_module()
508
+ source = 'version = "1.2.3"\nrequires-python = ">=3.8"\n'
509
+ pyproject, _init, _new = bump.bump(source, '__version__ = "1.2.3"\n')
510
+ assert 'requires-python = ">=3.8"' in pyproject
@@ -0,0 +1,231 @@
1
+ #!/usr/bin/env python3
2
+ """Compile the published provider feeds into src/iporigin/data/ranges.bin.
3
+
4
+ Run it by hand or let .github/workflows/update-data.yml do it weekly:
5
+
6
+ python tools/build_dataset.py
7
+
8
+ Format (all integers little-endian):
9
+
10
+ magic 8s b"IPORIGIN"
11
+ version H format version, currently 1
12
+ built_at Q unix timestamp of the build
13
+ n_labels H
14
+ n_v4 I
15
+ n_v6 I
16
+ labels n_labels x (H length + utf-8 "provider\\tkind")
17
+ v4 table n_v4 x (I start, I end, H label)
18
+ v6 table n_v6 x (16s start, 16s end, H label)
19
+
20
+ Ranges are stored as inclusive start/end integers rather than CIDR prefixes
21
+ because a lookup is then one bisect over a sorted column, with no prefix
22
+ arithmetic at query time. They are sorted by start and merged where a
23
+ provider's own feed overlaps itself, which the cloud feeds do constantly.
24
+ """
25
+
26
+ import ipaddress
27
+ import struct
28
+ import sys
29
+ import time
30
+ from pathlib import Path
31
+
32
+ sys.path.insert(0, str(Path(__file__).parent))
33
+
34
+ from sources import SOURCES # noqa: E402
35
+
36
+ MAGIC = b"IPORIGIN"
37
+ VERSION = 1
38
+ OUT = Path(__file__).parent.parent / "src" / "iporigin" / "data" / "ranges.bin"
39
+
40
+
41
+ def collect():
42
+ """Fetch every source. Returns (v4, v6) lists of (start, end, label)."""
43
+ v4, v6 = [], []
44
+ for name, (kind, fetch) in SOURCES.items():
45
+ label = "%s\t%s" % (name, kind)
46
+ count = 0
47
+ try:
48
+ for prefix in fetch():
49
+ try:
50
+ net = ipaddress.ip_network(prefix.strip(), strict=False)
51
+ except ValueError:
52
+ continue
53
+ record = (
54
+ int(net.network_address),
55
+ int(net.broadcast_address),
56
+ label,
57
+ )
58
+ (v4 if net.version == 4 else v6).append(record)
59
+ count += 1
60
+ except Exception as exc: # noqa: BLE001 — see module docstring
61
+ print(" !! %-18s skipped: %s" % (name, exc), file=sys.stderr)
62
+ continue
63
+ print(" %-18s %6d prefixes" % (name, count))
64
+ return v4, v6
65
+
66
+
67
+ def flatten(records):
68
+ """Turn overlapping ranges into a disjoint, sorted table.
69
+
70
+ Two things make this necessary rather than a nicety:
71
+
72
+ * Feeds overlap each other. GitHub runs on Azure and AWS, so its
73
+ prefixes sit inside theirs; Google publishes goog.json and cloud.json
74
+ which share space.
75
+ * The lookup is a bisect that inspects exactly one candidate — the last
76
+ range whose start is <= the address. With overlapping ranges that
77
+ candidate can be a narrow range that ends before the address while a
78
+ wider range still contains it, and the lookup returns "unknown" for an
79
+ address that is plainly in the table.
80
+
81
+ Where ranges overlap, the narrowest one wins: GitHub inside Azure should
82
+ answer GitHub, which is the more specific truth.
83
+ """
84
+ import heapq
85
+
86
+ if not records:
87
+ return []
88
+
89
+ events = []
90
+ for index, (start, end, label) in enumerate(records):
91
+ events.append((start, 0, index)) # range opens
92
+ events.append((end + 1, 1, index)) # range closes
93
+ events.sort()
94
+
95
+ active = [] # heap of (width, index)
96
+ ends = {}
97
+ for index, (start, end, _) in enumerate(records):
98
+ ends[index] = end
99
+
100
+ segments = []
101
+ position = events[0][0]
102
+ event_index = 0
103
+ total = len(events)
104
+
105
+ while event_index < total:
106
+ point = events[event_index][0]
107
+
108
+ # Emit the segment that ends where this event begins.
109
+ if point > position:
110
+ while active and ends[active[0][1]] < position:
111
+ heapq.heappop(active)
112
+ if active:
113
+ _, winner = active[0]
114
+ segments.append((position, point - 1, records[winner][2]))
115
+ position = point
116
+
117
+ while event_index < total and events[event_index][0] == point:
118
+ _, kind, index = events[event_index]
119
+ if kind == 0:
120
+ start, end, _ = records[index]
121
+ heapq.heappush(active, (end - start, index))
122
+ event_index += 1
123
+
124
+ # Closing events are handled lazily by the pop above; pushing the
125
+ # close event only serves to create a segment boundary here.
126
+
127
+ # Coalesce neighbours that ended up with the same label.
128
+ merged = []
129
+ for start, end, label in segments:
130
+ if merged and merged[-1][2] == label and start <= merged[-1][1] + 1:
131
+ merged[-1] = (merged[-1][0], max(merged[-1][1], end), label)
132
+ else:
133
+ merged.append((start, end, label))
134
+ return merged
135
+
136
+
137
+ def pack(v4, v6):
138
+ labels = []
139
+ index = {}
140
+ for _, _, label in v4 + v6:
141
+ if label not in index:
142
+ index[label] = len(labels)
143
+ labels.append(label)
144
+
145
+ out = bytearray()
146
+ out += struct.pack("<8sHQHII", MAGIC, VERSION, int(time.time()), len(labels), len(v4), len(v6))
147
+ for label in labels:
148
+ raw = label.encode("utf-8")
149
+ out += struct.pack("<H", len(raw)) + raw
150
+ for start, end, label in v4:
151
+ out += struct.pack("<IIH", start, end, index[label])
152
+ for start, end, label in v6:
153
+ out += struct.pack("<16s16sH",
154
+ start.to_bytes(16, "big"), end.to_bytes(16, "big"), index[label])
155
+ return bytes(out)
156
+
157
+
158
+ #: Refuse to replace the dataset if it shrinks by more than this. A source
159
+ #: that starts answering 200 with an empty body, or quietly changes its JSON
160
+ #: shape, looks exactly like a provider giving up its address space — and
161
+ #: nothing else in the pipeline would notice.
162
+ MAX_SHRINK = 0.20
163
+
164
+
165
+ def existing_range_count():
166
+ """How many ranges the committed dataset holds, or None if there is none."""
167
+ if not OUT.exists():
168
+ return None
169
+ try:
170
+ blob = OUT.read_bytes()
171
+ magic, _version, _built, _labels, n_v4, n_v6 = struct.unpack_from("<8sHQHII", blob, 0)
172
+ except Exception:
173
+ return None
174
+ if magic != MAGIC:
175
+ return None
176
+ return n_v4 + n_v6
177
+
178
+
179
+ def shrink_complaint(previous, current, max_shrink=None):
180
+ """Why this build must not replace the committed one, or None.
181
+
182
+ A source that starts answering 200 with an empty body, or quietly
183
+ changes its JSON shape, looks exactly like a provider giving up its
184
+ address space. Nothing else in the pipeline notices, and since the
185
+ refresh now commits straight to main there is no review that would.
186
+ """
187
+ limit = MAX_SHRINK if max_shrink is None else max_shrink
188
+ if not previous or current >= previous * (1 - limit):
189
+ return None
190
+ return (
191
+ "Refusing to write: %d ranges is %.0f%% below the %d already "
192
+ "committed. Either several sources failed at once, or one changed "
193
+ "shape and is now parsing to nothing. Check the per-source counts "
194
+ "above, then override with --allow-shrink if it is genuine."
195
+ % (current, 100 * (1 - current / previous), previous)
196
+ )
197
+
198
+
199
+ def main(allow_shrink=False):
200
+ print("Fetching provider feeds...")
201
+ v4, v6 = collect()
202
+ if not v4:
203
+ # Every source failing at once means something is wrong with the
204
+ # runner, not with ten providers. Refuse to ship an empty dataset
205
+ # over a good one.
206
+ print("No IPv4 ranges collected; refusing to write an empty dataset.", file=sys.stderr)
207
+ return 1
208
+
209
+ before = len(v4) + len(v6)
210
+ v4, v6 = flatten(v4), flatten(v6)
211
+ blob = pack(v4, v6)
212
+
213
+ # The last line of defence now that nothing downstream reviews this.
214
+ complaint = None if allow_shrink else shrink_complaint(
215
+ existing_range_count(), len(v4) + len(v6)
216
+ )
217
+ if complaint:
218
+ print(complaint, file=sys.stderr)
219
+ return 1
220
+
221
+ OUT.parent.mkdir(parents=True, exist_ok=True)
222
+ OUT.write_bytes(blob)
223
+
224
+ print("\n%d prefixes -> %d ranges (%d v4, %d v6), %.1f KB"
225
+ % (before, len(v4) + len(v6), len(v4), len(v6), len(blob) / 1024))
226
+ print("written to %s" % OUT)
227
+ return 0
228
+
229
+
230
+ if __name__ == "__main__":
231
+ raise SystemExit(main(allow_shrink="--allow-shrink" in sys.argv))
@@ -0,0 +1,84 @@
1
+ """Bump the patch version, in both of the places that hold it.
2
+
3
+ The version lives in pyproject.toml (what PyPI sees) and in
4
+ src/iporigin/__init__.py (what `iporigin --version` and any caller sees).
5
+ Nothing enforces that they agree, and a release that bumps one of them
6
+ ships a package that lies about itself — so this refuses to touch either
7
+ unless they start out identical.
8
+ """
9
+ import pathlib
10
+ import re
11
+ import sys
12
+
13
+ ROOT = pathlib.Path(__file__).resolve().parent.parent
14
+ PYPROJECT = ROOT / "pyproject.toml"
15
+ INIT = ROOT / "src" / "iporigin" / "__init__.py"
16
+
17
+ PYPROJECT_RE = re.compile(r'^version = "([^"]+)"', re.M)
18
+ INIT_RE = re.compile(r'^__version__ = "([^"]+)"', re.M)
19
+
20
+
21
+ def read_versions(pyproject_text, init_text):
22
+ """The declared version in each file, as a pair."""
23
+ found = []
24
+ for pattern, text, where in (
25
+ (PYPROJECT_RE, pyproject_text, "pyproject.toml"),
26
+ (INIT_RE, init_text, "__init__.py"),
27
+ ):
28
+ match = pattern.search(text)
29
+ if not match:
30
+ raise ValueError("no version declaration found in %s" % where)
31
+ found.append(match.group(1))
32
+ return tuple(found)
33
+
34
+
35
+ def next_patch(version):
36
+ """1.1.0 -> 1.1.1. Refuses anything that is not three plain numbers,
37
+ because guessing at what comes after 2.0.0rc1 is not this script's job."""
38
+ parts = version.split(".")
39
+ if len(parts) != 3 or not all(p.isdigit() for p in parts):
40
+ raise ValueError(
41
+ "%r is not a plain major.minor.patch version; bump it by hand" % version
42
+ )
43
+ major, minor, patch = parts
44
+ return "%s.%s.%d" % (major, minor, int(patch) + 1)
45
+
46
+
47
+ def bump(pyproject_text, init_text):
48
+ """Both files' new contents, plus the version they now declare."""
49
+ current_pyproject, current_init = read_versions(pyproject_text, init_text)
50
+ if current_pyproject != current_init:
51
+ raise ValueError(
52
+ "pyproject.toml says %s but __init__.py says %s; fix that before "
53
+ "releasing" % (current_pyproject, current_init)
54
+ )
55
+ new = next_patch(current_pyproject)
56
+ return (
57
+ PYPROJECT_RE.sub('version = "%s"' % new, pyproject_text, count=1),
58
+ INIT_RE.sub('__version__ = "%s"' % new, init_text, count=1),
59
+ new,
60
+ )
61
+
62
+
63
+ def main(argv):
64
+ pyproject_text = PYPROJECT.read_text()
65
+ init_text = INIT.read_text()
66
+
67
+ if "--current" in argv:
68
+ print(read_versions(pyproject_text, init_text)[0])
69
+ return 0
70
+
71
+ new_pyproject, new_init, new = bump(pyproject_text, init_text)
72
+ if "--dry-run" not in argv:
73
+ PYPROJECT.write_text(new_pyproject)
74
+ INIT.write_text(new_init)
75
+ print(new)
76
+ return 0
77
+
78
+
79
+ if __name__ == "__main__":
80
+ try:
81
+ raise SystemExit(main(sys.argv[1:]))
82
+ except ValueError as exc:
83
+ print("error: %s" % exc, file=sys.stderr)
84
+ raise SystemExit(1)
@@ -0,0 +1,321 @@
1
+ """Where the address ranges come from.
2
+
3
+ Every source here is the provider's own published feed, so the dataset is
4
+ reproducible: anyone can re-run tools/build_dataset.py and get the same
5
+ thing. That matters more than coverage — a bundled blob nobody can verify
6
+ is a blob nobody should trust.
7
+
8
+ A source that fails is skipped with a warning rather than failing the
9
+ build. These are ten third-party endpoints; on any given day one of them
10
+ is having a bad time, and that must not block a release.
11
+ """
12
+
13
+ import csv
14
+ import io
15
+ import json
16
+ import re
17
+ import urllib.request
18
+
19
+ USER_AGENT = "iporigin-dataset-builder/1.0 (+https://github.com/Yuix-Networks/iporigin)"
20
+ TIMEOUT = 60
21
+
22
+
23
+ def _get(url):
24
+ request = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})
25
+ with urllib.request.urlopen(request, timeout=TIMEOUT) as response:
26
+ return response.read()
27
+
28
+
29
+ def _get_json(url):
30
+ return json.loads(_get(url))
31
+
32
+
33
+ def _get_text(url):
34
+ return _get(url).decode("utf-8", "replace")
35
+
36
+
37
+ def aws():
38
+ data = _get_json("https://ip-ranges.amazonaws.com/ip-ranges.json")
39
+ for row in data.get("prefixes", []):
40
+ yield row["ip_prefix"]
41
+ for row in data.get("ipv6_prefixes", []):
42
+ yield row["ipv6_prefix"]
43
+
44
+
45
+ def gcp():
46
+ data = _get_json("https://www.gstatic.com/ipranges/cloud.json")
47
+ for row in data.get("prefixes", []):
48
+ prefix = row.get("ipv4Prefix") or row.get("ipv6Prefix")
49
+ if prefix:
50
+ yield prefix
51
+
52
+
53
+ def google():
54
+ # cloud.json is only the ranges Google Cloud hands to customers.
55
+ # goog.json is every range Google announces, which is what covers
56
+ # things like 8.8.8.8 — so both are needed, and they overlap.
57
+ data = _get_json("https://www.gstatic.com/ipranges/goog.json")
58
+ for row in data.get("prefixes", []):
59
+ prefix = row.get("ipv4Prefix") or row.get("ipv6Prefix")
60
+ if prefix:
61
+ yield prefix
62
+
63
+
64
+ def azure():
65
+ # Microsoft publishes a dated JSON behind a download page and changes the
66
+ # filename every week, so the URL has to be discovered rather than pinned.
67
+ page = _get_text("https://www.microsoft.com/en-us/download/details.aspx?id=56519")
68
+ match = re.search(
69
+ r"https://download\.microsoft\.com/download/[^\"']*?ServiceTags_Public_\d+\.json",
70
+ page,
71
+ )
72
+ if not match:
73
+ raise RuntimeError("could not find the ServiceTags JSON link on the page")
74
+ data = _get_json(match.group(0))
75
+ for value in data.get("values", []):
76
+ for prefix in value.get("properties", {}).get("addressPrefixes", []):
77
+ yield prefix
78
+
79
+
80
+ def digitalocean():
81
+ text = _get_text("https://www.digitalocean.com/geo/google.csv")
82
+ for row in csv.reader(io.StringIO(text)):
83
+ if row and "/" in row[0]:
84
+ yield row[0]
85
+
86
+
87
+ def linode():
88
+ # RFC 8805 geofeed: "prefix,country,region,city,postcode", # for comments.
89
+ for line in _get_text("https://geoip.linode.com/").splitlines():
90
+ line = line.strip()
91
+ if not line or line.startswith("#"):
92
+ continue
93
+ prefix = line.split(",")[0].strip()
94
+ if "/" in prefix:
95
+ yield prefix
96
+
97
+
98
+ def vultr():
99
+ data = _get_json("https://geofeed.constant.com/?json")
100
+ for row in data.get("subnets", []):
101
+ prefix = row.get("ip_prefix")
102
+ if prefix:
103
+ yield prefix
104
+
105
+
106
+ def oracle():
107
+ data = _get_json("https://docs.oracle.com/en-us/iaas/tools/public_ip_ranges.json")
108
+ for region in data.get("regions", []):
109
+ for cidr in region.get("cidrs", []):
110
+ prefix = cidr.get("cidr")
111
+ if prefix:
112
+ yield prefix
113
+
114
+
115
+ def cloudflare():
116
+ for url in ("https://www.cloudflare.com/ips-v4", "https://www.cloudflare.com/ips-v6"):
117
+ for line in _get_text(url).splitlines():
118
+ line = line.strip()
119
+ if "/" in line:
120
+ yield line
121
+
122
+
123
+ def fastly():
124
+ data = _get_json("https://api.fastly.com/public-ip-list")
125
+ for key in ("addresses", "ipv6_addresses"):
126
+ for prefix in data.get(key, []):
127
+ yield prefix
128
+
129
+
130
+ def github():
131
+ data = _get_json("https://api.github.com/meta")
132
+ seen = set()
133
+ for key, values in data.items():
134
+ if key in ("verifiable_password_authentication", "ssh_key_fingerprints", "domains"):
135
+ continue
136
+ if not isinstance(values, list):
137
+ continue
138
+ for prefix in values:
139
+ if isinstance(prefix, str) and "/" in prefix and prefix not in seen:
140
+ seen.add(prefix)
141
+ yield prefix
142
+
143
+
144
+ # name -> (kind, fetcher). "kind" is what the range means to a caller:
145
+ # hosting is a machine in a datacenter, cdn is edge infrastructure that
146
+ # fronts other people's sites and says nothing about who the visitor is.
147
+ SOURCES = {
148
+ "Amazon AWS": ("hosting", aws),
149
+ "Google Cloud": ("hosting", gcp),
150
+ "Google": ("hosting", google),
151
+ "Microsoft Azure": ("hosting", azure),
152
+ "DigitalOcean": ("hosting", digitalocean),
153
+ "Linode": ("hosting", linode),
154
+ "Vultr": ("hosting", vultr),
155
+ "Oracle Cloud": ("hosting", oracle),
156
+ "GitHub": ("hosting", github),
157
+ "Cloudflare": ("cdn", cloudflare),
158
+ "Fastly": ("cdn", fastly),
159
+ }
160
+
161
+
162
+ # ---------------------------------------------------------------------------
163
+ # Community aggregations
164
+ #
165
+ # Some providers publish no machine-readable range feed at all — Hetzner and
166
+ # OVH are the two that matter most, since a large share of abusive traffic
167
+ # comes from them. Consumer VPN exits and crawler ranges have the same
168
+ # problem for a different reason: nobody has an interest in publishing them.
169
+ #
170
+ # These come from two community repos instead. Both are CC0, which is why
171
+ # these two and not the half-dozen others that cover the same ground:
172
+ # redistributing an unlicensed list inside an MIT package is not something a
173
+ # dependency should ask of the people who install it.
174
+ #
175
+ # rezmoss/cloud-provider-ip-addresses CC0-1.0
176
+ # lord-alfred/ipranges CC0-1.0
177
+ #
178
+ # They are second-hand by definition, so they are labelled as such in the
179
+ # README rather than presented as the provider's own word.
180
+ # ---------------------------------------------------------------------------
181
+
182
+ REZMOSS = "https://raw.githubusercontent.com/rezmoss/cloud-provider-ip-addresses/main/%s/%s_ips_v%d.txt"
183
+ LORD_ALFRED = "https://raw.githubusercontent.com/lord-alfred/ipranges/main/%s/ipv%d_merged.txt"
184
+
185
+
186
+ def _plain_cidr_lines(url, required=True):
187
+ try:
188
+ text = _get_text(url)
189
+ except Exception:
190
+ if required:
191
+ raise
192
+ return
193
+ for line in text.splitlines():
194
+ line = line.strip()
195
+ if line and not line.startswith("#") and "/" in line:
196
+ yield line
197
+
198
+
199
+ def _rezmoss(slug):
200
+ def fetch():
201
+ yield from _plain_cidr_lines(REZMOSS % (slug, slug, 4))
202
+ # Not every provider has a v6 file with content; a missing one must
203
+ # not lose the v4 ranges we already collected.
204
+ yield from _plain_cidr_lines(REZMOSS % (slug, slug, 6), required=False)
205
+
206
+ return fetch
207
+
208
+
209
+ def _lord_alfred(slug):
210
+ def fetch():
211
+ yield from _plain_cidr_lines(LORD_ALFRED % (slug, 4))
212
+ yield from _plain_cidr_lines(LORD_ALFRED % (slug, 6), required=False)
213
+
214
+ return fetch
215
+
216
+
217
+ COMMUNITY = {
218
+ # Hosting with no official feed.
219
+ "Hetzner": ("hosting", _rezmoss("hetzner")),
220
+ "OVHcloud": ("hosting", _rezmoss("ovhcloud")),
221
+ "Scaleway": ("hosting", _rezmoss("scaleway")),
222
+ "Alibaba Cloud": ("hosting", _rezmoss("alibaba")),
223
+ "Leaseweb": ("hosting", _rezmoss("leaseweb")),
224
+ "UpCloud": ("hosting", _rezmoss("upcloud")),
225
+ "IBM Cloud": ("hosting", _rezmoss("ibmcloud")),
226
+ "Huawei Cloud": ("hosting", _rezmoss("huawei")),
227
+ "Tencent Cloud": ("hosting", _rezmoss("tencent")),
228
+ "Rackspace": ("hosting", _rezmoss("rackspace")),
229
+ # CDN.
230
+ "Akamai": ("cdn", _rezmoss("akamai")),
231
+ "Gcore": ("cdn", _rezmoss("gcore")),
232
+ # Consumer VPN exits — the thing no provider feed can give us.
233
+ "Mullvad": ("vpn", _rezmoss("mullvad")),
234
+ "ProtonVPN": ("vpn", _lord_alfred("protonvpn")),
235
+ "Apple Private Relay": ("vpn", _rezmoss("apple_private_relay")),
236
+ # Tor is its own thing: not a VPN, not a datacenter, and a caller
237
+ # usually wants to treat it differently from both.
238
+ "Tor": ("tor", _rezmoss("tor")),
239
+ # Declared crawlers. Separate from hosting because "a bot" and "someone
240
+ # on a server" call for different handling — you rate-limit one and
241
+ # block the other.
242
+ "Googlebot": ("bot", _rezmoss("googlebot")),
243
+ "Bingbot": ("bot", _rezmoss("bingbot")),
244
+ "GPTBot": ("bot", _rezmoss("gptbot")),
245
+ "ClaudeBot": ("bot", _rezmoss("claudebot")),
246
+ "PerplexityBot": ("bot", _rezmoss("perplexitybot")),
247
+ "DuckDuckBot": ("bot", _rezmoss("duckduckbot")),
248
+ "Amazonbot": ("bot", _rezmoss("amazonbot")),
249
+ "Applebot": ("bot", _rezmoss("applebot")),
250
+ "Common Crawl": ("bot", _rezmoss("commoncrawl")),
251
+ "Internet Archive": ("bot", _rezmoss("internetarchive")),
252
+ }
253
+
254
+ SOURCES.update(COMMUNITY)
255
+
256
+
257
+ # ---------------------------------------------------------------------------
258
+ # Sources used with the upstream maintainers' permission
259
+ #
260
+ # These repositories carry no licence file, which normally means "all rights
261
+ # reserved" and rules them out of an MIT package. They are included because
262
+ # permission was obtained from each maintainer directly — see NOTICE.
263
+ #
264
+ # Only what is actually additive is taken. Two well-known lists were measured
265
+ # and left out rather than included for the sake of it:
266
+ #
267
+ # jhassine/server-ip-addresses 52,772 prefixes, 227M addresses, and every
268
+ # one of 211,616 sampled addresses already
269
+ # covered — it is a subset of the provider
270
+ # feeds it was itself built from.
271
+ # Pymmdrza/Datacenter_List... 1,280 new addresses over what Hetzner's
272
+ # other sources already give.
273
+ #
274
+ # A source that adds nothing is not free: it is another endpoint that can
275
+ # break the weekly rebuild.
276
+ # ---------------------------------------------------------------------------
277
+
278
+ JJCK = "https://raw.githubusercontent.com/123jjck/cdn-ip-ranges/main/%s/%s_plain_ipv4.txt"
279
+
280
+
281
+ def _jjck(slug):
282
+ # IPv4 only; this repo publishes no v6 files.
283
+ return lambda: _plain_cidr_lines(JJCK % (slug, slug))
284
+
285
+
286
+ def akamai_secops():
287
+ yield from _plain_cidr_lines(
288
+ "https://raw.githubusercontent.com/SecOps-Institute/"
289
+ "Akamai-ASN-and-IPs-List/master/akamai_ip_cidr_blocks.lst"
290
+ )
291
+
292
+
293
+ BY_PERMISSION = {
294
+ # Networks with no feed anywhere else. Measured additions over the
295
+ # CC0 tier, largest first.
296
+ "Cogent": ("hosting", _jjck("cogent")), # +36.2M addresses
297
+ "DataCamp": ("cdn", _jjck("datacamp")), # +1.07M
298
+ "Contabo": ("hosting", _jjck("contabo")), # +603k
299
+ "Vercel": ("hosting", _jjck("vercel")), # +134k
300
+ "CDN77": ("cdn", _jjck("cdn77")), # +128k
301
+ "GleSYS": ("hosting", _jjck("glesys")), # +158k
302
+ "Scalaxy": ("hosting", _jjck("scalaxy")), # +105k
303
+ "GTHost": ("hosting", _jjck("gthost")), # +89k
304
+ "Melbicom": ("hosting", _jjck("melbicom")), # +63k
305
+ "BuyVM": ("hosting", _jjck("buyvm")), # +32k
306
+ "BunnyCDN": ("cdn", _jjck("bunny")), # +4k
307
+ }
308
+
309
+ # Akamai already has a CC0 source; this one adds ~309k addresses on top, so
310
+ # it feeds the same label rather than creating a second Akamai entry.
311
+ _akamai_cc0 = COMMUNITY["Akamai"][1]
312
+
313
+
314
+ def akamai_combined():
315
+ yield from _akamai_cc0()
316
+ yield from akamai_secops()
317
+
318
+
319
+ COMMUNITY["Akamai"] = ("cdn", akamai_combined)
320
+
321
+ SOURCES.update(BY_PERMISSION)
Binary file
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes