crawlcheck 1.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crawlcheck-1.0.1/PKG-INFO +36 -0
- crawlcheck-1.0.1/README.md +25 -0
- crawlcheck-1.0.1/crawlcheck/__init__.py +6 -0
- crawlcheck-1.0.1/crawlcheck/__main__.py +31 -0
- crawlcheck-1.0.1/crawlcheck/_ed25519.py +84 -0
- crawlcheck-1.0.1/crawlcheck/client.py +115 -0
- crawlcheck-1.0.1/crawlcheck/verify.py +100 -0
- crawlcheck-1.0.1/crawlcheck.egg-info/PKG-INFO +36 -0
- crawlcheck-1.0.1/crawlcheck.egg-info/SOURCES.txt +12 -0
- crawlcheck-1.0.1/crawlcheck.egg-info/dependency_links.txt +1 -0
- crawlcheck-1.0.1/crawlcheck.egg-info/top_level.txt +1 -0
- crawlcheck-1.0.1/pyproject.toml +16 -0
- crawlcheck-1.0.1/setup.cfg +4 -0
- crawlcheck-1.0.1/tests/test_live.py +65 -0
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: crawlcheck
|
|
3
|
+
Version: 1.0.1
|
|
4
|
+
Summary: CrawlCheck client: resolve a domain, get a signed decision for an agent action, verify CrawlCheck evidence offline. No dependencies.
|
|
5
|
+
Author: VSNARY
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://crawlcheck.io/sdk
|
|
8
|
+
Project-URL: Documentation, https://crawlcheck.io/docs/api
|
|
9
|
+
Requires-Python: >=3.8
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
|
|
12
|
+
# crawlcheck (Python)
|
|
13
|
+
|
|
14
|
+
Before your agent reads, cites, connects to or transacts with a domain, ask CrawlCheck. No dependencies; Python 3.8+.
|
|
15
|
+
|
|
16
|
+
```
|
|
17
|
+
pip install https://crawlcheck.io/sdk/crawlcheck-1.0.1-py3-none-any.whl
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
```python
|
|
21
|
+
from crawlcheck import CrawlCheck
|
|
22
|
+
|
|
23
|
+
cc = CrawlCheck() # CrawlCheck(key="cc_...") for licensed detail
|
|
24
|
+
answer = cc.resolve("example.com") # signed ResolveV1: policy, delivery, files, capabilities, entity, freshness, actions
|
|
25
|
+
print(answer["actions"]["read"]["decision"])
|
|
26
|
+
|
|
27
|
+
receipt = cc.preflight("example.com", action="cite", agent="GPTBot", template="safe_citation")
|
|
28
|
+
print(receipt["decision"], [s["why"] for s in receipt["steps"]])
|
|
29
|
+
assert cc.verify(receipt)["verified"] # Ed25519, checked locally; the key must be in the published directory
|
|
30
|
+
|
|
31
|
+
ok, receipt = cc.guard("shop.example", action="transact", confirm=lambda r: input(r["meaning"] + " y/n? ") == "y")
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Command line: `python -m crawlcheck preflight example.com cite GPTBot`, `python -m crawlcheck verify receipt.json`.
|
|
35
|
+
|
|
36
|
+
Decisions: `allow`, `warn`, `require_confirmation`, `block`, `unsupported`. A policy (`max_age_hours`, `on_warn`, `require_entity`, `human_approval`, `allow`, `deny`) can only make a decision stricter.
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
# crawlcheck (Python)
|
|
2
|
+
|
|
3
|
+
Before your agent reads, cites, connects to or transacts with a domain, ask CrawlCheck. No dependencies; Python 3.8+.
|
|
4
|
+
|
|
5
|
+
```
|
|
6
|
+
pip install https://crawlcheck.io/sdk/crawlcheck-1.0.1-py3-none-any.whl
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
```python
|
|
10
|
+
from crawlcheck import CrawlCheck
|
|
11
|
+
|
|
12
|
+
cc = CrawlCheck() # CrawlCheck(key="cc_...") for licensed detail
|
|
13
|
+
answer = cc.resolve("example.com") # signed ResolveV1: policy, delivery, files, capabilities, entity, freshness, actions
|
|
14
|
+
print(answer["actions"]["read"]["decision"])
|
|
15
|
+
|
|
16
|
+
receipt = cc.preflight("example.com", action="cite", agent="GPTBot", template="safe_citation")
|
|
17
|
+
print(receipt["decision"], [s["why"] for s in receipt["steps"]])
|
|
18
|
+
assert cc.verify(receipt)["verified"] # Ed25519, checked locally; the key must be in the published directory
|
|
19
|
+
|
|
20
|
+
ok, receipt = cc.guard("shop.example", action="transact", confirm=lambda r: input(r["meaning"] + " y/n? ") == "y")
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
Command line: `python -m crawlcheck preflight example.com cite GPTBot`, `python -m crawlcheck verify receipt.json`.
|
|
24
|
+
|
|
25
|
+
Decisions: `allow`, `warn`, `require_confirmation`, `block`, `unsupported`. A policy (`max_age_hours`, `on_warn`, `require_entity`, `human_approval`, `allow`, `deny`) can only make a decision stricter.
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""CrawlCheck: resolve a domain, decide an agent action, verify the evidence. https://crawlcheck.io/docs/api"""
|
|
2
|
+
from .client import CrawlCheck, CrawlCheckError, VERSION, ACTIONS
|
|
3
|
+
from .verify import verify_document, canon, key_id
|
|
4
|
+
|
|
5
|
+
__version__ = VERSION
|
|
6
|
+
__all__ = ["CrawlCheck", "CrawlCheckError", "verify_document", "canon", "key_id", "ACTIONS", "__version__"]
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""python -m crawlcheck resolve <domain> | preflight <domain> <action> [agent] | verify <file.json>"""
|
|
2
|
+
import json
|
|
3
|
+
import sys
|
|
4
|
+
|
|
5
|
+
from . import CrawlCheck, verify_document
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def main(argv=None):
|
|
9
|
+
a = list(sys.argv[1:] if argv is None else argv)
|
|
10
|
+
if not a or a[0] not in ("resolve", "preflight", "verify"):
|
|
11
|
+
print(__doc__)
|
|
12
|
+
return 2
|
|
13
|
+
cc = CrawlCheck()
|
|
14
|
+
if a[0] == "verify":
|
|
15
|
+
doc = json.load(open(a[1], encoding="utf-8"))
|
|
16
|
+
r = cc.verify(doc)
|
|
17
|
+
for c in r["checks"]:
|
|
18
|
+
print({True: "PASS", False: "FAIL", None: " -- "}[c["ok"]], c["id"].ljust(16), c["why"])
|
|
19
|
+
return 0 if r["verified"] else 1
|
|
20
|
+
if a[0] == "resolve":
|
|
21
|
+
print(json.dumps(cc.resolve(a[1]), indent=1))
|
|
22
|
+
return 0
|
|
23
|
+
rc = cc.preflight(a[1], a[2] if len(a) > 2 else "read", a[3] if len(a) > 3 else None)
|
|
24
|
+
print(rc["decision"], "-", rc.get("meaning"))
|
|
25
|
+
for s in rc.get("steps", []):
|
|
26
|
+
print(" ", s["rule"].ljust(26), s["effect"].ljust(20), s["why"])
|
|
27
|
+
return 0
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
if __name__ == "__main__":
|
|
31
|
+
sys.exit(main())
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""Pure-Python Ed25519 signature verification (RFC 8032, section 5.1.7). Verify only; no secret keys are handled."""
|
|
2
|
+
import hashlib
|
|
3
|
+
|
|
4
|
+
_p = 2 ** 255 - 19
|
|
5
|
+
_q = 2 ** 252 + 27742317777372353535851937790883648493
|
|
6
|
+
_d = -121665 * pow(121666, _p - 2, _p) % _p
|
|
7
|
+
_I = pow(2, (_p - 1) // 4, _p)
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _inv(x):
|
|
11
|
+
return pow(x, _p - 2, _p)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _add(P, Q):
|
|
15
|
+
A = (P[1] - P[0]) * (Q[1] - Q[0]) % _p
|
|
16
|
+
B = (P[1] + P[0]) * (Q[1] + Q[0]) % _p
|
|
17
|
+
C = 2 * P[3] * Q[3] * _d % _p
|
|
18
|
+
D = 2 * P[2] * Q[2] % _p
|
|
19
|
+
E, F, G, H = B - A, D - C, D + C, B + A
|
|
20
|
+
return (E * F % _p, G * H % _p, F * G % _p, E * H % _p)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _mul(s, P):
|
|
24
|
+
Q = (0, 1, 1, 0)
|
|
25
|
+
while s > 0:
|
|
26
|
+
if s & 1:
|
|
27
|
+
Q = _add(Q, P)
|
|
28
|
+
P = _add(P, P)
|
|
29
|
+
s >>= 1
|
|
30
|
+
return Q
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _equal(P, Q):
|
|
34
|
+
return (P[0] * Q[2] - Q[0] * P[2]) % _p == 0 and (P[1] * Q[2] - Q[1] * P[2]) % _p == 0
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _recover_x(y, sign):
|
|
38
|
+
if y >= _p:
|
|
39
|
+
return None
|
|
40
|
+
x2 = (y * y - 1) * _inv(_d * y * y + 1)
|
|
41
|
+
if x2 == 0:
|
|
42
|
+
return None if sign else 0
|
|
43
|
+
x = pow(x2, (_p + 3) // 8, _p)
|
|
44
|
+
if (x * x - x2) % _p != 0:
|
|
45
|
+
x = x * _I % _p
|
|
46
|
+
if (x * x - x2) % _p != 0:
|
|
47
|
+
return None
|
|
48
|
+
if (x & 1) != sign:
|
|
49
|
+
x = _p - x
|
|
50
|
+
return x
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
_gy = 4 * _inv(5) % _p
|
|
54
|
+
_gx = _recover_x(_gy, 0)
|
|
55
|
+
_G = (_gx, _gy, 1, _gx * _gy % _p)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _decompress(s):
|
|
59
|
+
if len(s) != 32:
|
|
60
|
+
return None
|
|
61
|
+
y = int.from_bytes(s, "little")
|
|
62
|
+
sign = y >> 255
|
|
63
|
+
y &= (1 << 255) - 1
|
|
64
|
+
x = _recover_x(y, sign)
|
|
65
|
+
if x is None:
|
|
66
|
+
return None
|
|
67
|
+
return (x, y, 1, x * y % _p)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def verify(public: bytes, msg: bytes, sig: bytes) -> bool:
|
|
71
|
+
if len(public) != 32 or len(sig) != 64:
|
|
72
|
+
return False
|
|
73
|
+
A = _decompress(public)
|
|
74
|
+
if not A:
|
|
75
|
+
return False
|
|
76
|
+
Rs = sig[:32]
|
|
77
|
+
R = _decompress(Rs)
|
|
78
|
+
if not R:
|
|
79
|
+
return False
|
|
80
|
+
s = int.from_bytes(sig[32:], "little")
|
|
81
|
+
if s >= _q:
|
|
82
|
+
return False
|
|
83
|
+
h = int.from_bytes(hashlib.sha512(Rs + public + msg).digest(), "little") % _q
|
|
84
|
+
return _equal(_mul(s, _G), _add(R, _mul(h, A)))
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
"""HTTP client for the CrawlCheck API. Standard library only."""
|
|
2
|
+
import json
|
|
3
|
+
import urllib.error
|
|
4
|
+
import urllib.parse
|
|
5
|
+
import urllib.request
|
|
6
|
+
|
|
7
|
+
from .verify import verify_document, key_id
|
|
8
|
+
|
|
9
|
+
VERSION = "1.0.1"
|
|
10
|
+
ACTIONS = ("read", "cite", "connect", "transact", "administer")
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class CrawlCheckError(Exception):
|
|
14
|
+
def __init__(self, status, body):
|
|
15
|
+
self.status, self.body = status, body
|
|
16
|
+
msg = body.get("error") if isinstance(body, dict) else None
|
|
17
|
+
if not msg and isinstance(body, dict) and body.get("errors"):
|
|
18
|
+
msg = "; ".join(body["errors"])
|
|
19
|
+
super().__init__("HTTP %s: %s" % (status, msg or str(body)[:200]))
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class CrawlCheck:
|
|
23
|
+
"""client = CrawlCheck(key=None) # a licence key is optional; it unlocks full findings for licensed domains"""
|
|
24
|
+
|
|
25
|
+
def __init__(self, key=None, base="https://crawlcheck.io", timeout=30):
|
|
26
|
+
self.key, self.base, self.timeout = key, base.rstrip("/"), timeout
|
|
27
|
+
self._kids = None
|
|
28
|
+
|
|
29
|
+
def _req(self, method, path, query=None, body=None):
|
|
30
|
+
url = self.base + path + ("?" + urllib.parse.urlencode({k: v for k, v in (query or {}).items() if v is not None}) if query else "")
|
|
31
|
+
h = {"user-agent": "crawlcheck-python/" + VERSION, "accept": "application/json"}
|
|
32
|
+
if self.key:
|
|
33
|
+
h["x-crawlcheck-key"] = self.key
|
|
34
|
+
data = None
|
|
35
|
+
if body is not None:
|
|
36
|
+
data = json.dumps(body).encode()
|
|
37
|
+
h["content-type"] = "application/json"
|
|
38
|
+
req = urllib.request.Request(url, data=data, headers=h, method=method)
|
|
39
|
+
try:
|
|
40
|
+
with urllib.request.urlopen(req, timeout=self.timeout) as r:
|
|
41
|
+
return json.loads(r.read().decode("utf-8"))
|
|
42
|
+
except urllib.error.HTTPError as e:
|
|
43
|
+
raw = e.read().decode("utf-8", "replace")
|
|
44
|
+
try:
|
|
45
|
+
b = json.loads(raw)
|
|
46
|
+
except ValueError:
|
|
47
|
+
b = {"error": raw[:300]}
|
|
48
|
+
raise CrawlCheckError(e.code, b) from None
|
|
49
|
+
|
|
50
|
+
# --- read before acting ---
|
|
51
|
+
def resolve(self, domain):
|
|
52
|
+
"""The signed ResolveV1 answer: crawl policy, delivery, machine files, capabilities, entity, findings, freshness and a decision per action."""
|
|
53
|
+
return self._req("GET", "/api/v1/resolve", {"domain": domain})
|
|
54
|
+
|
|
55
|
+
def resolve_batch(self, domains):
|
|
56
|
+
"""Up to 100 domains in one call; each answer is a full signed ResolveV1 document."""
|
|
57
|
+
return self._req("POST", "/api/v1/resolve", body={"domains": list(domains)})
|
|
58
|
+
|
|
59
|
+
def preflight(self, domain, action="read", agent=None, template=None, policy=None):
|
|
60
|
+
"""The policy engine: a signed decision receipt (allow | warn | require_confirmation | block | unsupported).
|
|
61
|
+
policy keys: max_age_hours, on_warn (warn|require_confirmation|block), require_entity, human_approval[], allow[], deny[].
|
|
62
|
+
A policy can only make the decision stricter."""
|
|
63
|
+
b = {"domain": domain, "action": action}
|
|
64
|
+
if agent:
|
|
65
|
+
b["agent"] = agent
|
|
66
|
+
if template:
|
|
67
|
+
b["template"] = template
|
|
68
|
+
if policy:
|
|
69
|
+
b["policy"] = policy
|
|
70
|
+
return self._req("POST", "/api/v1/preflight", body=b)
|
|
71
|
+
|
|
72
|
+
def changes(self, since=None, domain=None, limit=100):
|
|
73
|
+
return self._req("GET", "/api/v1/changes", {"since": since, "domain": domain, "limit": limit})
|
|
74
|
+
|
|
75
|
+
def outcome(self, domain, action, status=None, blocked=None, challenged=None, agent=None, resolve_sha256=None, **extra):
|
|
76
|
+
"""Report what happened after acting. Outcomes are kept apart from the signed record and never rewrite it."""
|
|
77
|
+
b = dict(extra, domain=domain, action=action)
|
|
78
|
+
for k, v in (("status", status), ("blocked", blocked), ("challenged", challenged), ("agent", agent), ("resolve_sha256", resolve_sha256)):
|
|
79
|
+
if v is not None:
|
|
80
|
+
b[k] = v
|
|
81
|
+
return self._req("POST", "/api/v1/outcome", body=b)
|
|
82
|
+
|
|
83
|
+
def domain(self, domain):
|
|
84
|
+
"""The registry record (RDAP) and lifecycle of a domain."""
|
|
85
|
+
return self._req("GET", "/api/v1/domain", {"domain": domain})
|
|
86
|
+
|
|
87
|
+
def scan(self, domain):
|
|
88
|
+
return self._req("POST", "/api/scan", body={"domain": domain})
|
|
89
|
+
|
|
90
|
+
def machine_record(self, report_id, fmt=None):
|
|
91
|
+
return self._req("GET", "/api/v1/machine-record", {"id": report_id, "format": fmt})
|
|
92
|
+
|
|
93
|
+
# --- verification ---
|
|
94
|
+
def published_key_ids(self, refresh=False):
|
|
95
|
+
if self._kids is None or refresh:
|
|
96
|
+
d = self._req("GET", "/.well-known/http-message-signatures-directory")
|
|
97
|
+
self._kids = [k.get("kid") or key_id(k) for k in d.get("keys", [])]
|
|
98
|
+
return self._kids
|
|
99
|
+
|
|
100
|
+
def verify(self, doc, online_keys=True):
|
|
101
|
+
"""Verify a signed document. With online_keys, the signing key must also be in the published directory."""
|
|
102
|
+
return verify_document(doc, self.published_key_ids() if online_keys else None)
|
|
103
|
+
|
|
104
|
+
def guard(self, domain, action="read", agent=None, template=None, policy=None, confirm=None, allow_warn=False):
|
|
105
|
+
"""Preflight, verify the receipt, and return (proceed: bool, receipt). Proceeds on allow. warn proceeds only with
|
|
106
|
+
allow_warn=True. require_confirmation proceeds only if confirm(receipt) returns True. block and unsupported never."""
|
|
107
|
+
rc = self.preflight(domain, action, agent, template, policy)
|
|
108
|
+
if not self.verify(rc)["verified"]:
|
|
109
|
+
return False, rc
|
|
110
|
+
d = rc.get("decision")
|
|
111
|
+
if d == "allow" or (d == "warn" and allow_warn):
|
|
112
|
+
return True, rc
|
|
113
|
+
if d == "require_confirmation" and confirm is not None:
|
|
114
|
+
return bool(confirm(rc)), rc
|
|
115
|
+
return False, rc
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""Offline verification of CrawlCheck signed documents: resolve answers, decision receipts, data inventories,
|
|
2
|
+
export parts and deletion records. Mirrors verifyDataDoc in crawlcheck-verify.mjs, check for check."""
|
|
3
|
+
import base64
|
|
4
|
+
import hashlib
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
from decimal import Decimal
|
|
8
|
+
|
|
9
|
+
from . import _ed25519
|
|
10
|
+
|
|
11
|
+
DATA_SIG_PREFIX = "crawlcheck-data-v1\n"
|
|
12
|
+
DATA_KINDS = ("crawlcheck-data-inventory", "crawlcheck-data-export-part", "crawlcheck-deletion-record", "crawlcheck-resolve", "crawlcheck-decision")
|
|
13
|
+
DECISIONS = ("allow", "warn", "require_confirmation", "block", "unsupported")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _js_number(v):
|
|
17
|
+
if isinstance(v, bool):
|
|
18
|
+
return "true" if v else "false"
|
|
19
|
+
if isinstance(v, int):
|
|
20
|
+
return str(v)
|
|
21
|
+
if v != v or v in (float("inf"), float("-inf")):
|
|
22
|
+
return "null"
|
|
23
|
+
if v == int(v) and abs(v) < 1e21:
|
|
24
|
+
return str(int(v))
|
|
25
|
+
r = repr(v)
|
|
26
|
+
if "e" in r or "E" in r:
|
|
27
|
+
m, e = r.lower().split("e")
|
|
28
|
+
e = int(e)
|
|
29
|
+
if -7 < e < 21:
|
|
30
|
+
s = format(Decimal(r), "f")
|
|
31
|
+
return s.rstrip("0").rstrip(".") if "." in s else s
|
|
32
|
+
return m + "e" + ("+" if e > 0 else "-") + str(abs(e))
|
|
33
|
+
return r
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def canon(v):
|
|
37
|
+
"""Canonical JSON exactly as the server computes it: keys sorted, no whitespace, JSON.stringify for scalars."""
|
|
38
|
+
if isinstance(v, dict):
|
|
39
|
+
return "{" + ",".join(json.dumps(k, ensure_ascii=False) + ":" + canon(v[k]) for k in sorted(v)) + "}"
|
|
40
|
+
if isinstance(v, (list, tuple)):
|
|
41
|
+
return "[" + ",".join(canon(x) for x in v) + "]"
|
|
42
|
+
if v is None:
|
|
43
|
+
return "null"
|
|
44
|
+
if isinstance(v, (bool, int, float)):
|
|
45
|
+
return _js_number(v)
|
|
46
|
+
return json.dumps(v, ensure_ascii=False)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _b64u_decode(s):
|
|
50
|
+
s = str(s)
|
|
51
|
+
return base64.urlsafe_b64decode(s + "=" * (-len(s) % 4))
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _b64u(b):
|
|
55
|
+
return base64.urlsafe_b64encode(b).decode().rstrip("=")
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def key_id(jwk):
|
|
59
|
+
"""RFC 7638 thumbprint of an Ed25519 JWK - the kid CrawlCheck publishes."""
|
|
60
|
+
return _b64u(hashlib.sha256(json.dumps({"crv": jwk["crv"], "kty": jwk["kty"], "x": jwk["x"]}, separators=(",", ":")).encode()).digest())
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def verify_document(doc, published_kids=None):
|
|
64
|
+
"""Verify a signed CrawlCheck document offline. Returns {"verified": bool, "checks": [...]}.
|
|
65
|
+
published_kids: optional list of key ids from https://crawlcheck.io/.well-known/http-message-signatures-directory;
|
|
66
|
+
without it key_published stays None (undecided), as in the JavaScript verifier."""
|
|
67
|
+
checks = []
|
|
68
|
+
|
|
69
|
+
def C(i, ok, why):
|
|
70
|
+
checks.append({"id": i, "ok": ok, "why": why})
|
|
71
|
+
|
|
72
|
+
if not isinstance(doc, dict) or doc.get("kind") not in DATA_KINDS or not doc.get("signature"):
|
|
73
|
+
raise ValueError("not a CrawlCheck data inventory, export part, deletion record, resolve answer or decision receipt")
|
|
74
|
+
body = {k: v for k, v in doc.items() if k not in ("sha256", "signature")}
|
|
75
|
+
sha = hashlib.sha256(canon(body).encode("utf-8")).hexdigest()
|
|
76
|
+
C("document_hash", sha == doc.get("sha256"), "the document hashes to its stated digest" if sha == doc.get("sha256") else "the document hashes to %s, it states %s - a field was changed" % (sha, doc.get("sha256")))
|
|
77
|
+
sig = doc["signature"]
|
|
78
|
+
jwk = (sig.get("key") or {}).get("jwk") or {}
|
|
79
|
+
if not jwk.get("x"):
|
|
80
|
+
C("key_id", None, "no key in the document")
|
|
81
|
+
C("signature", None, "no key to check against")
|
|
82
|
+
else:
|
|
83
|
+
kid = key_id(jwk)
|
|
84
|
+
C("key_id", kid == sig.get("kid"), "public key thumbprint (RFC 7638) is the key id" if kid == sig.get("kid") else "thumbprint %s does not match key id %s" % (kid, sig.get("kid")))
|
|
85
|
+
msg = DATA_SIG_PREFIX + sha
|
|
86
|
+
ok = sig.get("message") == msg and _ed25519.verify(_b64u_decode(jwk["x"]), msg.encode(), _b64u_decode(sig.get("sig", "")))
|
|
87
|
+
C("signature", ok, "Ed25519 signature over the document digest verifies" if ok else "signature does NOT verify over this content")
|
|
88
|
+
if published_kids is None:
|
|
89
|
+
C("key_published", None, "offline this cannot be decided: compare key id %s with the published key directory" % sig.get("kid"))
|
|
90
|
+
else:
|
|
91
|
+
C("key_published", sig.get("kid") in published_kids, "the signing key is published" if sig.get("kid") in published_kids else "the signing key is NOT in the published directory - reject this document")
|
|
92
|
+
if doc.get("kind") == "crawlcheck-decision":
|
|
93
|
+
r = doc.get("resolve") or {}
|
|
94
|
+
bound = bool(re.fullmatch(r"[0-9a-f]{64}", str(r.get("sha256") or ""))) and r.get("kid") == sig.get("kid")
|
|
95
|
+
C("resolve_bound", bound, "the decision names the resolve answer it was made from, signed by the same key" if bound else "the decision does not name a resolve answer digest")
|
|
96
|
+
C("decision_known", doc.get("decision") in DECISIONS, "decision is %s" % doc.get("decision"))
|
|
97
|
+
if doc.get("kind") == "crawlcheck-deletion-record":
|
|
98
|
+
C("counts", sum((f.get("keys") or 0) for f in doc.get("deleted") or []) == doc.get("deleted_total"), "the per-family counts add up to deleted_total")
|
|
99
|
+
ran = [c for c in checks if c["ok"] is not None]
|
|
100
|
+
return {"verified": bool(ran) and all(c["ok"] for c in ran), "checks": checks}
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: crawlcheck
|
|
3
|
+
Version: 1.0.1
|
|
4
|
+
Summary: CrawlCheck client: resolve a domain, get a signed decision for an agent action, verify CrawlCheck evidence offline. No dependencies.
|
|
5
|
+
Author: VSNARY
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://crawlcheck.io/sdk
|
|
8
|
+
Project-URL: Documentation, https://crawlcheck.io/docs/api
|
|
9
|
+
Requires-Python: >=3.8
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
|
|
12
|
+
# crawlcheck (Python)
|
|
13
|
+
|
|
14
|
+
Before your agent reads, cites, connects to or transacts with a domain, ask CrawlCheck. No dependencies; Python 3.8+.
|
|
15
|
+
|
|
16
|
+
```
|
|
17
|
+
pip install https://crawlcheck.io/sdk/crawlcheck-1.0.1-py3-none-any.whl
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
```python
|
|
21
|
+
from crawlcheck import CrawlCheck
|
|
22
|
+
|
|
23
|
+
cc = CrawlCheck() # CrawlCheck(key="cc_...") for licensed detail
|
|
24
|
+
answer = cc.resolve("example.com") # signed ResolveV1: policy, delivery, files, capabilities, entity, freshness, actions
|
|
25
|
+
print(answer["actions"]["read"]["decision"])
|
|
26
|
+
|
|
27
|
+
receipt = cc.preflight("example.com", action="cite", agent="GPTBot", template="safe_citation")
|
|
28
|
+
print(receipt["decision"], [s["why"] for s in receipt["steps"]])
|
|
29
|
+
assert cc.verify(receipt)["verified"] # Ed25519, checked locally; the key must be in the published directory
|
|
30
|
+
|
|
31
|
+
ok, receipt = cc.guard("shop.example", action="transact", confirm=lambda r: input(r["meaning"] + " y/n? ") == "y")
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Command line: `python -m crawlcheck preflight example.com cite GPTBot`, `python -m crawlcheck verify receipt.json`.
|
|
35
|
+
|
|
36
|
+
Decisions: `allow`, `warn`, `require_confirmation`, `block`, `unsupported`. A policy (`max_age_hours`, `on_warn`, `require_entity`, `human_approval`, `allow`, `deny`) can only make a decision stricter.
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
README.md
|
|
2
|
+
pyproject.toml
|
|
3
|
+
crawlcheck/__init__.py
|
|
4
|
+
crawlcheck/__main__.py
|
|
5
|
+
crawlcheck/_ed25519.py
|
|
6
|
+
crawlcheck/client.py
|
|
7
|
+
crawlcheck/verify.py
|
|
8
|
+
crawlcheck.egg-info/PKG-INFO
|
|
9
|
+
crawlcheck.egg-info/SOURCES.txt
|
|
10
|
+
crawlcheck.egg-info/dependency_links.txt
|
|
11
|
+
crawlcheck.egg-info/top_level.txt
|
|
12
|
+
tests/test_live.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
crawlcheck
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "crawlcheck"
|
|
7
|
+
version = "1.0.1"
|
|
8
|
+
description = "CrawlCheck client: resolve a domain, get a signed decision for an agent action, verify CrawlCheck evidence offline. No dependencies."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.8"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "VSNARY" }]
|
|
13
|
+
urls = { Homepage = "https://crawlcheck.io/sdk", Documentation = "https://crawlcheck.io/docs/api" }
|
|
14
|
+
|
|
15
|
+
[tool.setuptools]
|
|
16
|
+
packages = ["crawlcheck"]
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""Live tests against production plus offline tamper tests. python -m unittest discover -s tests"""
|
|
2
|
+
import copy
|
|
3
|
+
import unittest
|
|
4
|
+
|
|
5
|
+
from crawlcheck import CrawlCheck, CrawlCheckError, canon, verify_document
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class T(unittest.TestCase):
|
|
9
|
+
cc = CrawlCheck()
|
|
10
|
+
|
|
11
|
+
def test_canon_matches_js(self):
|
|
12
|
+
self.assertEqual(canon({"b": 1, "a": [True, None, 1.5, 1.0, "é\n"], "c": {"z": 1e-7, "y": 1e21, "x": 1e16}}),
|
|
13
|
+
'{"a":[true,null,1.5,1,"é\\n"],"b":1,"c":{"x":10000000000000000,"y":1e+21,"z":1e-7}}')
|
|
14
|
+
|
|
15
|
+
def test_resolve_verifies(self):
|
|
16
|
+
a = self.cc.resolve("supremefencingdenver.com")
|
|
17
|
+
self.assertEqual(a["kind"], "crawlcheck-resolve")
|
|
18
|
+
r = self.cc.verify(a)
|
|
19
|
+
self.assertTrue(r["verified"], r)
|
|
20
|
+
self.assertIn("read", a["actions"])
|
|
21
|
+
self.assertIn("fresh_until", a["freshness"])
|
|
22
|
+
|
|
23
|
+
def test_tampered_resolve_fails(self):
|
|
24
|
+
a = copy.deepcopy(self.cc.resolve("github.com"))
|
|
25
|
+
a["actions"]["read"]["decision"] = "allow" if a["actions"]["read"]["decision"] != "allow" else "block"
|
|
26
|
+
self.assertFalse(verify_document(a)["verified"])
|
|
27
|
+
|
|
28
|
+
def test_preflight_receipt(self):
|
|
29
|
+
rc = self.cc.preflight("amazon.com", "read", agent="GPTBot")
|
|
30
|
+
self.assertEqual(rc["decision"], "block")
|
|
31
|
+
self.assertTrue(self.cc.verify(rc)["verified"])
|
|
32
|
+
ids = {c["id"]: c["ok"] for c in self.cc.verify(rc)["checks"]}
|
|
33
|
+
self.assertTrue(ids["resolve_bound"] and ids["key_published"])
|
|
34
|
+
|
|
35
|
+
def test_policy_cannot_loosen(self):
|
|
36
|
+
with self.assertRaises(CrawlCheckError) as e:
|
|
37
|
+
self.cc.preflight("example.com", "read", policy={"on_warn": "allow"})
|
|
38
|
+
self.assertEqual(e.exception.status, 422)
|
|
39
|
+
|
|
40
|
+
def test_guard_requires_confirmation(self):
|
|
41
|
+
ok, rc = self.cc.guard("supremefencingdenver.com", "transact")
|
|
42
|
+
self.assertFalse(ok)
|
|
43
|
+
ok, rc = self.cc.guard("supremefencingdenver.com", "transact", confirm=lambda r: True)
|
|
44
|
+
self.assertTrue(ok)
|
|
45
|
+
|
|
46
|
+
def test_foreign_key_rejected_online(self):
|
|
47
|
+
a = self.cc.resolve("supremefencingdenver.com")
|
|
48
|
+
r = verify_document(a, published_kids=["someone-else"])
|
|
49
|
+
self.assertFalse(r["verified"])
|
|
50
|
+
|
|
51
|
+
def test_guard_warn_needs_opt_in(self):
|
|
52
|
+
rc = self.cc.preflight("github.com", "read")
|
|
53
|
+
if rc["decision"] == "warn":
|
|
54
|
+
self.assertFalse(self.cc.guard("github.com", "read")[0])
|
|
55
|
+
self.assertTrue(self.cc.guard("github.com", "read", allow_warn=True)[0])
|
|
56
|
+
|
|
57
|
+
def test_batch(self):
|
|
58
|
+
b = self.cc.resolve_batch(["github.com", "supremefencingdenver.com"])
|
|
59
|
+
self.assertEqual(b["count"], 2)
|
|
60
|
+
for x in b["answers"]:
|
|
61
|
+
self.assertTrue(self.cc.verify(x["answer"])["verified"])
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
if __name__ == "__main__":
|
|
65
|
+
unittest.main()
|