parserail 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parserail/__init__.py +663 -0
- parserail-0.6.0.dist-info/METADATA +107 -0
- parserail-0.6.0.dist-info/RECORD +4 -0
- parserail-0.6.0.dist-info/WHEEL +4 -0
parserail/__init__.py
ADDED
|
@@ -0,0 +1,663 @@
|
|
|
1
|
+
"""Official Python SDK for ParseRail — the AI back-end for your product.
|
|
2
|
+
|
|
3
|
+
from parserail import ParseRail, ParseRailError
|
|
4
|
+
|
|
5
|
+
client = ParseRail(api_key="ksk_live_...")
|
|
6
|
+
doc = client.parse(file_url="https://.../invoice.pdf")
|
|
7
|
+
print(doc["totalAmount"], doc["usage"]["balanceRemaining"])
|
|
8
|
+
|
|
9
|
+
Zero dependencies (stdlib only). Every method returns the endpoint result as a
|
|
10
|
+
dict including a ``usage`` envelope; failures raise a typed ``ParseRailError`` and
|
|
11
|
+
never burn credits. Get a key + 500 free credits every month at
|
|
12
|
+
https://api.thecompound.tech.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
# Generated from the ParseRail endpoint catalog.
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import json
|
|
20
|
+
import urllib.error
|
|
21
|
+
import urllib.request
|
|
22
|
+
from time import monotonic, sleep
|
|
23
|
+
from typing import Any, Dict, List, Optional
|
|
24
|
+
from urllib.parse import quote
|
|
25
|
+
|
|
26
|
+
__all__ = ["ParseRail", "ParseRailError", "__version__"]
|
|
27
|
+
__version__ = "0.6.0"
|
|
28
|
+
|
|
29
|
+
_DEFAULT_BASE_URL = "https://api.thecompound.tech"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class ParseRailError(Exception):
|
|
33
|
+
"""Raised for any non-2xx API response. Mirrors ``{"error": {"code", "message"}}``."""
|
|
34
|
+
|
|
35
|
+
def __init__(self, code: str, message: str, status: int) -> None:
|
|
36
|
+
super().__init__(message)
|
|
37
|
+
self.code = code
|
|
38
|
+
self.message = message
|
|
39
|
+
self.status = status
|
|
40
|
+
|
|
41
|
+
def __repr__(self) -> str: # pragma: no cover - trivial
|
|
42
|
+
return f"ParseRailError(code={self.code!r}, status={self.status}, message={self.message!r})"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _document(file_url: Optional[str], text: Optional[str], file: Optional[Dict[str, str]]) -> Dict[str, Any]:
|
|
46
|
+
if file_url:
|
|
47
|
+
return {"fileUrl": file_url}
|
|
48
|
+
if file:
|
|
49
|
+
return {"file": file}
|
|
50
|
+
if text:
|
|
51
|
+
return {"text": text}
|
|
52
|
+
raise ValueError("Provide one of: file_url, text, or file={'data','mimeType'}.")
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class ParseRail:
|
|
56
|
+
"""Synchronous client for the ParseRail API."""
|
|
57
|
+
|
|
58
|
+
def __init__(
|
|
59
|
+
self,
|
|
60
|
+
api_key: str,
|
|
61
|
+
base_url: str = _DEFAULT_BASE_URL,
|
|
62
|
+
timeout: float = 60.0,
|
|
63
|
+
) -> None:
|
|
64
|
+
if not api_key:
|
|
65
|
+
raise ValueError(
|
|
66
|
+
"ParseRail: `api_key` is required. Mint one at https://parserail.thecompound.tech/dashboard."
|
|
67
|
+
)
|
|
68
|
+
self.api_key = api_key
|
|
69
|
+
self.base_url = base_url.rstrip("/")
|
|
70
|
+
self.timeout = timeout
|
|
71
|
+
|
|
72
|
+
# -- transport ---------------------------------------------------------
|
|
73
|
+
def _request(self, method: str, path: str, body: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
|
|
74
|
+
url = f"{self.base_url}{path}"
|
|
75
|
+
data = json.dumps(body).encode("utf-8") if body is not None else None
|
|
76
|
+
headers = {"Authorization": f"Bearer {self.api_key}", "User-Agent": f"parserail-python/{__version__}"}
|
|
77
|
+
if data is not None:
|
|
78
|
+
headers["Content-Type"] = "application/json"
|
|
79
|
+
req = urllib.request.Request(url, data=data, headers=headers, method=method)
|
|
80
|
+
try:
|
|
81
|
+
with urllib.request.urlopen(req, timeout=self.timeout) as resp:
|
|
82
|
+
payload = resp.read().decode("utf-8")
|
|
83
|
+
return json.loads(payload) if payload else {}
|
|
84
|
+
except urllib.error.HTTPError as exc:
|
|
85
|
+
raw = exc.read().decode("utf-8", errors="replace")
|
|
86
|
+
code, message = "unknown", f"Request failed with status {exc.code}."
|
|
87
|
+
try:
|
|
88
|
+
err = json.loads(raw).get("error") or {}
|
|
89
|
+
code = err.get("code", code)
|
|
90
|
+
message = err.get("message", message)
|
|
91
|
+
except (ValueError, AttributeError):
|
|
92
|
+
pass
|
|
93
|
+
raise ParseRailError(code, message, exc.code) from None
|
|
94
|
+
except urllib.error.URLError as exc:
|
|
95
|
+
raise ParseRailError("unknown", str(exc.reason), 0) from None
|
|
96
|
+
|
|
97
|
+
def _post(self, path: str, body: Dict[str, Any]) -> Dict[str, Any]:
|
|
98
|
+
# Drop None values so optional params are omitted.
|
|
99
|
+
clean = {k: v for k, v in body.items() if v is not None}
|
|
100
|
+
return self._request("POST", path, clean)
|
|
101
|
+
|
|
102
|
+
# -- capabilities ------------------------------------------------------
|
|
103
|
+
def parse(
|
|
104
|
+
self,
|
|
105
|
+
*,
|
|
106
|
+
file_url: Optional[str] = None,
|
|
107
|
+
text: Optional[str] = None,
|
|
108
|
+
file: Optional[Dict[str, str]] = None,
|
|
109
|
+
doc_type: Optional[str] = None,
|
|
110
|
+
async_: bool = False,
|
|
111
|
+
callback_url: Optional[str] = None,
|
|
112
|
+
) -> Dict[str, Any]:
|
|
113
|
+
"""Any invoice, receipt, EOB, ERA, or COI — PDF or image — into structured, validated JSON.
|
|
114
|
+
|
|
115
|
+
Pass ``async_=True`` to queue the work and get ``{"jobId", "status"}``
|
|
116
|
+
back instead of the result; poll it with ``get_job`` / ``wait_for_job``."""
|
|
117
|
+
body: Dict[str, Any] = _document(file_url, text, file)
|
|
118
|
+
if doc_type is not None:
|
|
119
|
+
body["docType"] = doc_type
|
|
120
|
+
if async_:
|
|
121
|
+
body["async"] = True
|
|
122
|
+
if callback_url is not None:
|
|
123
|
+
body["callbackUrl"] = callback_url
|
|
124
|
+
return self._post("/v1/parse", body)
|
|
125
|
+
|
|
126
|
+
def extract(
|
|
127
|
+
self,
|
|
128
|
+
*,
|
|
129
|
+
text: str,
|
|
130
|
+
fields: List[str],
|
|
131
|
+
instructions: Optional[str] = None,
|
|
132
|
+
) -> Dict[str, Any]:
|
|
133
|
+
"""Pull a field set you define out of any block of text. You name the fields; you get typed values with confidence."""
|
|
134
|
+
body: Dict[str, Any] = {"text": text, "fields": fields, "instructions": instructions}
|
|
135
|
+
return self._post("/v1/extract", body)
|
|
136
|
+
|
|
137
|
+
def classify(
|
|
138
|
+
self,
|
|
139
|
+
*,
|
|
140
|
+
text: str,
|
|
141
|
+
labels: List[str],
|
|
142
|
+
multi: Optional[bool] = None,
|
|
143
|
+
instructions: Optional[str] = None,
|
|
144
|
+
) -> Dict[str, Any]:
|
|
145
|
+
"""Route or tag text against your own taxonomy — a label, a confidence, and a one-line rationale."""
|
|
146
|
+
body: Dict[str, Any] = {"text": text, "labels": labels, "multi": multi, "instructions": instructions}
|
|
147
|
+
return self._post("/v1/classify", body)
|
|
148
|
+
|
|
149
|
+
def summarize(
|
|
150
|
+
self,
|
|
151
|
+
*,
|
|
152
|
+
text: str,
|
|
153
|
+
length: Optional[str] = None,
|
|
154
|
+
action_items: Optional[bool] = None,
|
|
155
|
+
) -> Dict[str, Any]:
|
|
156
|
+
"""A meeting transcript, thread, or report → a tight summary, key points, and extracted action items."""
|
|
157
|
+
body: Dict[str, Any] = {"text": text, "length": length, "actionItems": action_items}
|
|
158
|
+
return self._post("/v1/summarize", body)
|
|
159
|
+
|
|
160
|
+
def redact(
|
|
161
|
+
self,
|
|
162
|
+
*,
|
|
163
|
+
text: str,
|
|
164
|
+
types: Optional[List[str]] = None,
|
|
165
|
+
placeholder: Optional[str] = None,
|
|
166
|
+
) -> Dict[str, Any]:
|
|
167
|
+
"""Detect and strip names, emails, phones, SSNs, cards, and PHI from text before you store or log it."""
|
|
168
|
+
body: Dict[str, Any] = {"text": text, "types": types, "placeholder": placeholder}
|
|
169
|
+
return self._post("/v1/redact", body)
|
|
170
|
+
|
|
171
|
+
def sentiment(
|
|
172
|
+
self,
|
|
173
|
+
*,
|
|
174
|
+
text: str,
|
|
175
|
+
aspects: Optional[List[str]] = None,
|
|
176
|
+
) -> Dict[str, Any]:
|
|
177
|
+
"""Turn a review or support message into a sentiment score, per-aspect breakdown, and the themes driving it."""
|
|
178
|
+
body: Dict[str, Any] = {"text": text, "aspects": aspects}
|
|
179
|
+
return self._post("/v1/sentiment", body)
|
|
180
|
+
|
|
181
|
+
def contract(
|
|
182
|
+
self,
|
|
183
|
+
*,
|
|
184
|
+
file_url: Optional[str] = None,
|
|
185
|
+
text: Optional[str] = None,
|
|
186
|
+
file: Optional[Dict[str, str]] = None,
|
|
187
|
+
async_: bool = False,
|
|
188
|
+
callback_url: Optional[str] = None,
|
|
189
|
+
) -> Dict[str, Any]:
|
|
190
|
+
"""A contract → parties, effective date, term, renewal, governing law, obligations, and flagged risk clauses.
|
|
191
|
+
|
|
192
|
+
Pass ``async_=True`` to queue the work and get ``{"jobId", "status"}``
|
|
193
|
+
back instead of the result; poll it with ``get_job`` / ``wait_for_job``."""
|
|
194
|
+
body: Dict[str, Any] = _document(file_url, text, file)
|
|
195
|
+
if async_:
|
|
196
|
+
body["async"] = True
|
|
197
|
+
if callback_url is not None:
|
|
198
|
+
body["callbackUrl"] = callback_url
|
|
199
|
+
return self._post("/v1/contract", body)
|
|
200
|
+
|
|
201
|
+
def chargeback(
|
|
202
|
+
self,
|
|
203
|
+
*,
|
|
204
|
+
reason: str,
|
|
205
|
+
transaction: Dict[str, Any],
|
|
206
|
+
evidence: Optional[List[str]] = None,
|
|
207
|
+
context: Optional[str] = None,
|
|
208
|
+
network: Optional[str] = None,
|
|
209
|
+
) -> Dict[str, Any]:
|
|
210
|
+
"""Dispute details → a representment narrative, an evidence checklist, the right reason code, and a win-likelihood."""
|
|
211
|
+
body: Dict[str, Any] = {"reason": reason, "transaction": transaction, "evidence": evidence, "context": context, "network": network}
|
|
212
|
+
return self._post("/v1/chargeback", body)
|
|
213
|
+
|
|
214
|
+
def enrich(
|
|
215
|
+
self,
|
|
216
|
+
*,
|
|
217
|
+
domain: Optional[str] = None,
|
|
218
|
+
email: Optional[str] = None,
|
|
219
|
+
) -> Dict[str, Any]:
|
|
220
|
+
"""A domain or work email → a structured company profile: name, description, industry, HQ, size, and links."""
|
|
221
|
+
body: Dict[str, Any] = {"domain": domain, "email": email}
|
|
222
|
+
return self._post("/v1/enrich", body)
|
|
223
|
+
|
|
224
|
+
def invoice(
|
|
225
|
+
self,
|
|
226
|
+
*,
|
|
227
|
+
file_url: Optional[str] = None,
|
|
228
|
+
text: Optional[str] = None,
|
|
229
|
+
file: Optional[Dict[str, str]] = None,
|
|
230
|
+
async_: bool = False,
|
|
231
|
+
callback_url: Optional[str] = None,
|
|
232
|
+
) -> Dict[str, Any]:
|
|
233
|
+
"""An invoice — PDF, photo, or text — into vendor, dates, PO refs, tax, totals, and clean line items.
|
|
234
|
+
|
|
235
|
+
Pass ``async_=True`` to queue the work and get ``{"jobId", "status"}``
|
|
236
|
+
back instead of the result; poll it with ``get_job`` / ``wait_for_job``."""
|
|
237
|
+
body: Dict[str, Any] = _document(file_url, text, file)
|
|
238
|
+
if async_:
|
|
239
|
+
body["async"] = True
|
|
240
|
+
if callback_url is not None:
|
|
241
|
+
body["callbackUrl"] = callback_url
|
|
242
|
+
return self._post("/v1/invoice", body)
|
|
243
|
+
|
|
244
|
+
def receipt(
|
|
245
|
+
self,
|
|
246
|
+
*,
|
|
247
|
+
file_url: Optional[str] = None,
|
|
248
|
+
text: Optional[str] = None,
|
|
249
|
+
file: Optional[Dict[str, str]] = None,
|
|
250
|
+
async_: bool = False,
|
|
251
|
+
callback_url: Optional[str] = None,
|
|
252
|
+
) -> Dict[str, Any]:
|
|
253
|
+
"""A receipt into merchant, items, totals, payment method, and an expense category — built for expense flows.
|
|
254
|
+
|
|
255
|
+
Pass ``async_=True`` to queue the work and get ``{"jobId", "status"}``
|
|
256
|
+
back instead of the result; poll it with ``get_job`` / ``wait_for_job``."""
|
|
257
|
+
body: Dict[str, Any] = _document(file_url, text, file)
|
|
258
|
+
if async_:
|
|
259
|
+
body["async"] = True
|
|
260
|
+
if callback_url is not None:
|
|
261
|
+
body["callbackUrl"] = callback_url
|
|
262
|
+
return self._post("/v1/receipt", body)
|
|
263
|
+
|
|
264
|
+
def statement(
|
|
265
|
+
self,
|
|
266
|
+
*,
|
|
267
|
+
file_url: Optional[str] = None,
|
|
268
|
+
text: Optional[str] = None,
|
|
269
|
+
file: Optional[Dict[str, str]] = None,
|
|
270
|
+
async_: bool = False,
|
|
271
|
+
callback_url: Optional[str] = None,
|
|
272
|
+
) -> Dict[str, Any]:
|
|
273
|
+
"""A bank or card statement into the account, the period, balances, and every transaction as a normalized row.
|
|
274
|
+
|
|
275
|
+
Pass ``async_=True`` to queue the work and get ``{"jobId", "status"}``
|
|
276
|
+
back instead of the result; poll it with ``get_job`` / ``wait_for_job``."""
|
|
277
|
+
body: Dict[str, Any] = _document(file_url, text, file)
|
|
278
|
+
if async_:
|
|
279
|
+
body["async"] = True
|
|
280
|
+
if callback_url is not None:
|
|
281
|
+
body["callbackUrl"] = callback_url
|
|
282
|
+
return self._post("/v1/statement", body)
|
|
283
|
+
|
|
284
|
+
def resume(
|
|
285
|
+
self,
|
|
286
|
+
*,
|
|
287
|
+
file_url: Optional[str] = None,
|
|
288
|
+
text: Optional[str] = None,
|
|
289
|
+
file: Optional[Dict[str, str]] = None,
|
|
290
|
+
async_: bool = False,
|
|
291
|
+
callback_url: Optional[str] = None,
|
|
292
|
+
) -> Dict[str, Any]:
|
|
293
|
+
"""A resume or CV into a structured candidate profile: contact, skills, experience, education, and links.
|
|
294
|
+
|
|
295
|
+
Pass ``async_=True`` to queue the work and get ``{"jobId", "status"}``
|
|
296
|
+
back instead of the result; poll it with ``get_job`` / ``wait_for_job``."""
|
|
297
|
+
body: Dict[str, Any] = _document(file_url, text, file)
|
|
298
|
+
if async_:
|
|
299
|
+
body["async"] = True
|
|
300
|
+
if callback_url is not None:
|
|
301
|
+
body["callbackUrl"] = callback_url
|
|
302
|
+
return self._post("/v1/resume", body)
|
|
303
|
+
|
|
304
|
+
def tables(
|
|
305
|
+
self,
|
|
306
|
+
*,
|
|
307
|
+
file_url: Optional[str] = None,
|
|
308
|
+
text: Optional[str] = None,
|
|
309
|
+
file: Optional[Dict[str, str]] = None,
|
|
310
|
+
async_: bool = False,
|
|
311
|
+
callback_url: Optional[str] = None,
|
|
312
|
+
) -> Dict[str, Any]:
|
|
313
|
+
"""Every table in a document — even scanned — as clean headers and rows, ready for your spreadsheet or DB.
|
|
314
|
+
|
|
315
|
+
Pass ``async_=True`` to queue the work and get ``{"jobId", "status"}``
|
|
316
|
+
back instead of the result; poll it with ``get_job`` / ``wait_for_job``."""
|
|
317
|
+
body: Dict[str, Any] = _document(file_url, text, file)
|
|
318
|
+
if async_:
|
|
319
|
+
body["async"] = True
|
|
320
|
+
if callback_url is not None:
|
|
321
|
+
body["callbackUrl"] = callback_url
|
|
322
|
+
return self._post("/v1/tables", body)
|
|
323
|
+
|
|
324
|
+
def split(
|
|
325
|
+
self,
|
|
326
|
+
*,
|
|
327
|
+
file_url: Optional[str] = None,
|
|
328
|
+
text: Optional[str] = None,
|
|
329
|
+
file: Optional[Dict[str, str]] = None,
|
|
330
|
+
async_: bool = False,
|
|
331
|
+
callback_url: Optional[str] = None,
|
|
332
|
+
) -> Dict[str, Any]:
|
|
333
|
+
"""A multi-document scan bundle classified and split: what each document is, where it starts and ends, and a summary.
|
|
334
|
+
|
|
335
|
+
Pass ``async_=True`` to queue the work and get ``{"jobId", "status"}``
|
|
336
|
+
back instead of the result; poll it with ``get_job`` / ``wait_for_job``."""
|
|
337
|
+
body: Dict[str, Any] = _document(file_url, text, file)
|
|
338
|
+
if async_:
|
|
339
|
+
body["async"] = True
|
|
340
|
+
if callback_url is not None:
|
|
341
|
+
body["callbackUrl"] = callback_url
|
|
342
|
+
return self._post("/v1/split", body)
|
|
343
|
+
|
|
344
|
+
def compare(
|
|
345
|
+
self,
|
|
346
|
+
*,
|
|
347
|
+
a: Dict[str, Any],
|
|
348
|
+
b: Dict[str, Any],
|
|
349
|
+
async_: bool = False,
|
|
350
|
+
callback_url: Optional[str] = None,
|
|
351
|
+
) -> Dict[str, Any]:
|
|
352
|
+
"""Two versions of a contract or document → every material change, what it means, and the risk it carries.
|
|
353
|
+
|
|
354
|
+
Pass ``async_=True`` to queue the work and get ``{"jobId", "status"}``
|
|
355
|
+
back instead of the result; poll it with ``get_job`` / ``wait_for_job``."""
|
|
356
|
+
body: Dict[str, Any] = {"a": a, "b": b}
|
|
357
|
+
if async_:
|
|
358
|
+
body["async"] = True
|
|
359
|
+
if callback_url is not None:
|
|
360
|
+
body["callbackUrl"] = callback_url
|
|
361
|
+
return self._post("/v1/compare", body)
|
|
362
|
+
|
|
363
|
+
def structure(
|
|
364
|
+
self,
|
|
365
|
+
*,
|
|
366
|
+
input: str,
|
|
367
|
+
schema: Dict[str, Any],
|
|
368
|
+
instructions: Optional[str] = None,
|
|
369
|
+
) -> Dict[str, Any]:
|
|
370
|
+
"""Any messy input — text, HTML, an email — plus YOUR JSON schema → output shaped to it, validated against your required fields and property types, with an automatic corrective retry and a `valid` flag."""
|
|
371
|
+
body: Dict[str, Any] = {"input": input, "schema": schema, "instructions": instructions}
|
|
372
|
+
return self._post("/v1/structure", body)
|
|
373
|
+
|
|
374
|
+
def normalize(
|
|
375
|
+
self,
|
|
376
|
+
*,
|
|
377
|
+
records: List[Dict[str, Any]],
|
|
378
|
+
fields: Optional[List[str]] = None,
|
|
379
|
+
instructions: Optional[str] = None,
|
|
380
|
+
) -> Dict[str, Any]:
|
|
381
|
+
"""A batch of messy records → clean canonical rows, with a change log of every fix (casing, formats, dedup-ready values)."""
|
|
382
|
+
body: Dict[str, Any] = {"records": records, "fields": fields, "instructions": instructions}
|
|
383
|
+
return self._post("/v1/normalize", body)
|
|
384
|
+
|
|
385
|
+
def match(
|
|
386
|
+
self,
|
|
387
|
+
*,
|
|
388
|
+
a: List[Dict[str, Any]],
|
|
389
|
+
b: List[Dict[str, Any]],
|
|
390
|
+
keys: Optional[List[str]] = None,
|
|
391
|
+
) -> Dict[str, Any]:
|
|
392
|
+
"""Two record sets → which rows are the same real-world thing, with confidence and reasoning. Fuzzy names, typos, aliases handled."""
|
|
393
|
+
body: Dict[str, Any] = {"a": a, "b": b, "keys": keys}
|
|
394
|
+
return self._post("/v1/match", body)
|
|
395
|
+
|
|
396
|
+
def categorize(
|
|
397
|
+
self,
|
|
398
|
+
*,
|
|
399
|
+
items: List[str],
|
|
400
|
+
taxonomy: List[str],
|
|
401
|
+
multi: Optional[bool] = None,
|
|
402
|
+
instructions: Optional[str] = None,
|
|
403
|
+
) -> Dict[str, Any]:
|
|
404
|
+
"""Up to a hundred items against your taxonomy in one call — products, transactions, tickets — each with a confidence."""
|
|
405
|
+
body: Dict[str, Any] = {"items": items, "taxonomy": taxonomy, "multi": multi, "instructions": instructions}
|
|
406
|
+
return self._post("/v1/categorize", body)
|
|
407
|
+
|
|
408
|
+
def dunning(
|
|
409
|
+
self,
|
|
410
|
+
*,
|
|
411
|
+
invoice: Dict[str, Any],
|
|
412
|
+
customer: Dict[str, Any],
|
|
413
|
+
tone: Optional[str] = None,
|
|
414
|
+
channel: Optional[str] = None,
|
|
415
|
+
steps: Optional[int] = None,
|
|
416
|
+
) -> Dict[str, Any]:
|
|
417
|
+
"""An overdue invoice → a ready-to-send collection sequence, escalating at the right pace and tone for how late it is."""
|
|
418
|
+
body: Dict[str, Any] = {"invoice": invoice, "customer": customer, "tone": tone, "channel": channel, "steps": steps}
|
|
419
|
+
return self._post("/v1/dunning", body)
|
|
420
|
+
|
|
421
|
+
def po_match(
|
|
422
|
+
self,
|
|
423
|
+
*,
|
|
424
|
+
invoice: str,
|
|
425
|
+
purchase_order: str,
|
|
426
|
+
receipt: Optional[str] = None,
|
|
427
|
+
tolerance_pct: Optional[float] = None,
|
|
428
|
+
) -> Dict[str, Any]:
|
|
429
|
+
"""Invoice vs purchase order vs receipt → matched, partial, or mismatched, with every discrepancy flagged and sized."""
|
|
430
|
+
body: Dict[str, Any] = {"invoice": invoice, "purchaseOrder": purchase_order, "receipt": receipt, "tolerancePct": tolerance_pct}
|
|
431
|
+
return self._post("/v1/po-match", body)
|
|
432
|
+
|
|
433
|
+
def quote(
|
|
434
|
+
self,
|
|
435
|
+
*,
|
|
436
|
+
job: str,
|
|
437
|
+
rates: Optional[str] = None,
|
|
438
|
+
past_quotes: Optional[str] = None,
|
|
439
|
+
currency: Optional[str] = None,
|
|
440
|
+
) -> Dict[str, Any]:
|
|
441
|
+
"""A job description plus your rates → an itemized, professional quote with assumptions and scope notes spelled out."""
|
|
442
|
+
body: Dict[str, Any] = {"job": job, "rates": rates, "pastQuotes": past_quotes, "currency": currency}
|
|
443
|
+
return self._post("/v1/quote", body)
|
|
444
|
+
|
|
445
|
+
def fraud_flag(
|
|
446
|
+
self,
|
|
447
|
+
*,
|
|
448
|
+
order: Dict[str, Any],
|
|
449
|
+
context: Optional[str] = None,
|
|
450
|
+
) -> Dict[str, Any]:
|
|
451
|
+
"""An order or transaction in context → a risk score, the signals driving it, and the checks worth running before you ship."""
|
|
452
|
+
body: Dict[str, Any] = {"order": order, "context": context}
|
|
453
|
+
return self._post("/v1/fraud-flag", body)
|
|
454
|
+
|
|
455
|
+
def reply(
|
|
456
|
+
self,
|
|
457
|
+
*,
|
|
458
|
+
thread: str,
|
|
459
|
+
context: Optional[str] = None,
|
|
460
|
+
tone: Optional[str] = None,
|
|
461
|
+
goal: Optional[str] = None,
|
|
462
|
+
sender_name: Optional[str] = None,
|
|
463
|
+
) -> Dict[str, Any]:
|
|
464
|
+
"""A customer thread plus your context → a ready-to-send reply in the right tone, with an internal note for the agent."""
|
|
465
|
+
body: Dict[str, Any] = {"thread": thread, "context": context, "tone": tone, "goal": goal, "senderName": sender_name}
|
|
466
|
+
return self._post("/v1/reply", body)
|
|
467
|
+
|
|
468
|
+
def triage(
|
|
469
|
+
self,
|
|
470
|
+
*,
|
|
471
|
+
ticket: str,
|
|
472
|
+
categories: Optional[List[str]] = None,
|
|
473
|
+
teams: Optional[List[str]] = None,
|
|
474
|
+
) -> Dict[str, Any]:
|
|
475
|
+
"""A support ticket → priority, category, the team it belongs to, sentiment, SLA risk, and a suggested first response."""
|
|
476
|
+
body: Dict[str, Any] = {"ticket": ticket, "categories": categories, "teams": teams}
|
|
477
|
+
return self._post("/v1/triage", body)
|
|
478
|
+
|
|
479
|
+
def minutes(
|
|
480
|
+
self,
|
|
481
|
+
*,
|
|
482
|
+
transcript: str,
|
|
483
|
+
attendees: Optional[List[str]] = None,
|
|
484
|
+
) -> Dict[str, Any]:
|
|
485
|
+
"""A meeting transcript → clean minutes: summary, decisions made, action items with owners, and open questions."""
|
|
486
|
+
body: Dict[str, Any] = {"transcript": transcript, "attendees": attendees}
|
|
487
|
+
return self._post("/v1/minutes", body)
|
|
488
|
+
|
|
489
|
+
def outreach(
|
|
490
|
+
self,
|
|
491
|
+
*,
|
|
492
|
+
lead: str,
|
|
493
|
+
product: str,
|
|
494
|
+
sender: Optional[Dict[str, Any]] = None,
|
|
495
|
+
channel: Optional[str] = None,
|
|
496
|
+
steps: Optional[int] = None,
|
|
497
|
+
) -> Dict[str, Any]:
|
|
498
|
+
"""An enriched lead plus what you sell → a personalized outreach sequence that references what actually makes them a fit."""
|
|
499
|
+
body: Dict[str, Any] = {"lead": lead, "product": product, "sender": sender, "channel": channel, "steps": steps}
|
|
500
|
+
return self._post("/v1/outreach", body)
|
|
501
|
+
|
|
502
|
+
def review_reply(
|
|
503
|
+
self,
|
|
504
|
+
*,
|
|
505
|
+
review: str,
|
|
506
|
+
rating: Optional[float] = None,
|
|
507
|
+
business: Optional[Dict[str, Any]] = None,
|
|
508
|
+
resolution: Optional[str] = None,
|
|
509
|
+
) -> Dict[str, Any]:
|
|
510
|
+
"""A customer review → a brand-safe public response, the issues to log, and whether it needs human escalation."""
|
|
511
|
+
body: Dict[str, Any] = {"review": review, "rating": rating, "business": business, "resolution": resolution}
|
|
512
|
+
return self._post("/v1/review-reply", body)
|
|
513
|
+
|
|
514
|
+
def moderate(
|
|
515
|
+
self,
|
|
516
|
+
*,
|
|
517
|
+
text: Optional[str] = None,
|
|
518
|
+
image: Optional[Dict[str, Any]] = None,
|
|
519
|
+
image_url: Optional[str] = None,
|
|
520
|
+
policy: Optional[str] = None,
|
|
521
|
+
) -> Dict[str, Any]:
|
|
522
|
+
"""Text or an image against YOUR policy → allow, review, or block, with the categories and excerpts that drove the call."""
|
|
523
|
+
body: Dict[str, Any] = {"text": text, "image": image, "imageUrl": image_url, "policy": policy}
|
|
524
|
+
return self._post("/v1/moderate", body)
|
|
525
|
+
|
|
526
|
+
def rewrite(
|
|
527
|
+
self,
|
|
528
|
+
*,
|
|
529
|
+
text: str,
|
|
530
|
+
voice: Optional[str] = None,
|
|
531
|
+
goal: Optional[str] = None,
|
|
532
|
+
audience: Optional[str] = None,
|
|
533
|
+
length: Optional[str] = None,
|
|
534
|
+
) -> Dict[str, Any]:
|
|
535
|
+
"""Any copy → your brand voice. Describe the voice or paste a sample; get the rewrite and what changed."""
|
|
536
|
+
body: Dict[str, Any] = {"text": text, "voice": voice, "goal": goal, "audience": audience, "length": length}
|
|
537
|
+
return self._post("/v1/rewrite", body)
|
|
538
|
+
|
|
539
|
+
def product_copy(
|
|
540
|
+
self,
|
|
541
|
+
*,
|
|
542
|
+
product: Dict[str, Any],
|
|
543
|
+
channel: Optional[str] = None,
|
|
544
|
+
) -> Dict[str, Any]:
|
|
545
|
+
"""Specs and an audience → listing-ready titles, bullets, a description, and SEO keywords — per channel."""
|
|
546
|
+
body: Dict[str, Any] = {"product": product, "channel": channel}
|
|
547
|
+
return self._post("/v1/product-copy", body)
|
|
548
|
+
|
|
549
|
+
def describe(
|
|
550
|
+
self,
|
|
551
|
+
*,
|
|
552
|
+
image: Optional[Dict[str, Any]] = None,
|
|
553
|
+
image_url: Optional[str] = None,
|
|
554
|
+
purpose: Optional[str] = None,
|
|
555
|
+
) -> Dict[str, Any]:
|
|
556
|
+
"""An image → accessible alt text, a caption, tags, and any text found inside it. Accessibility and catalogs in one call."""
|
|
557
|
+
body: Dict[str, Any] = {"image": image, "imageUrl": image_url, "purpose": purpose}
|
|
558
|
+
return self._post("/v1/describe", body)
|
|
559
|
+
|
|
560
|
+
def research(
|
|
561
|
+
self,
|
|
562
|
+
*,
|
|
563
|
+
query: str,
|
|
564
|
+
focus: Optional[str] = None,
|
|
565
|
+
depth: Optional[str] = None,
|
|
566
|
+
) -> Dict[str, Any]:
|
|
567
|
+
"""A company or topic → a multi-source, citation-backed brief: what it is, what changed lately, and what matters. Real web work."""
|
|
568
|
+
body: Dict[str, Any] = {"query": query, "focus": focus, "depth": depth}
|
|
569
|
+
return self._post("/v1/research", body)
|
|
570
|
+
|
|
571
|
+
def screen(
|
|
572
|
+
self,
|
|
573
|
+
*,
|
|
574
|
+
lead: str,
|
|
575
|
+
criteria: str,
|
|
576
|
+
) -> Dict[str, Any]:
|
|
577
|
+
"""A lead plus your ICP criteria → a web-grounded qualification verdict with the evidence for and against."""
|
|
578
|
+
body: Dict[str, Any] = {"lead": lead, "criteria": criteria}
|
|
579
|
+
return self._post("/v1/screen", body)
|
|
580
|
+
|
|
581
|
+
def transcribe(
|
|
582
|
+
self,
|
|
583
|
+
*,
|
|
584
|
+
audio: Optional[Dict[str, Any]] = None,
|
|
585
|
+
audio_url: Optional[str] = None,
|
|
586
|
+
diarize: Optional[bool] = None,
|
|
587
|
+
language: Optional[str] = None,
|
|
588
|
+
) -> Dict[str, Any]:
|
|
589
|
+
"""An audio recording → accurate text with speakers and paragraph timestamps. Meetings, calls, voice notes."""
|
|
590
|
+
body: Dict[str, Any] = {"audio": audio, "audioUrl": audio_url, "diarize": diarize, "language": language}
|
|
591
|
+
return self._post("/v1/transcribe", body)
|
|
592
|
+
|
|
593
|
+
def memory(
|
|
594
|
+
self,
|
|
595
|
+
*,
|
|
596
|
+
op: str,
|
|
597
|
+
namespace: Optional[str] = None,
|
|
598
|
+
content: Optional[str] = None,
|
|
599
|
+
metadata: Optional[Dict[str, Any]] = None,
|
|
600
|
+
query: Optional[str] = None,
|
|
601
|
+
limit: Optional[int] = None,
|
|
602
|
+
id: Optional[str] = None,
|
|
603
|
+
) -> Dict[str, Any]:
|
|
604
|
+
"""Store, search, and forget memories for your agents — semantic recall on your own namespace, no vector DB to run."""
|
|
605
|
+
body: Dict[str, Any] = {"op": op, "namespace": namespace, "content": content, "metadata": metadata, "query": query, "limit": limit, "id": id}
|
|
606
|
+
return self._post("/v1/memory", body)
|
|
607
|
+
|
|
608
|
+
def image(
|
|
609
|
+
self,
|
|
610
|
+
*,
|
|
611
|
+
prompt: str,
|
|
612
|
+
style: Optional[str] = None,
|
|
613
|
+
aspect_ratio: Optional[str] = None,
|
|
614
|
+
) -> Dict[str, Any]:
|
|
615
|
+
"""A prompt → a production-ready image. Marketing visuals, product scenes, and consistent brand imagery."""
|
|
616
|
+
body: Dict[str, Any] = {"prompt": prompt, "style": style, "aspectRatio": aspect_ratio}
|
|
617
|
+
return self._post("/v1/image", body)
|
|
618
|
+
|
|
619
|
+
def speak(
|
|
620
|
+
self,
|
|
621
|
+
*,
|
|
622
|
+
text: str,
|
|
623
|
+
voice: Optional[str] = None,
|
|
624
|
+
pace: Optional[str] = None,
|
|
625
|
+
) -> Dict[str, Any]:
|
|
626
|
+
"""Text → natural speech audio, ready to embed. Voice notes, IVR lines, narration."""
|
|
627
|
+
body: Dict[str, Any] = {"text": text, "voice": voice, "pace": pace}
|
|
628
|
+
return self._post("/v1/speak", body)
|
|
629
|
+
|
|
630
|
+
def account(self) -> Dict[str, Any]:
|
|
631
|
+
"""Fetch the authenticated account's credit balance."""
|
|
632
|
+
return self._request("GET", "/v1/account")
|
|
633
|
+
|
|
634
|
+
# -- async jobs --------------------------------------------------------
|
|
635
|
+
def get_job(self, job_id: str) -> Dict[str, Any]:
|
|
636
|
+
"""Read a job created with ``async_=True``. Never charges credits."""
|
|
637
|
+
return self._request("GET", f"/v1/jobs/{quote(job_id, safe='')}")
|
|
638
|
+
|
|
639
|
+
def wait_for_job(
|
|
640
|
+
self,
|
|
641
|
+
job_id: str,
|
|
642
|
+
*,
|
|
643
|
+
interval: float = 2.0,
|
|
644
|
+
timeout: float = 300.0,
|
|
645
|
+
) -> Dict[str, Any]:
|
|
646
|
+
"""Poll a job until it finishes, and return it.
|
|
647
|
+
|
|
648
|
+
Returns on ``"failed"`` as well as ``"succeeded"`` — check
|
|
649
|
+
``job["status"]``. A failed job was never charged. Raises
|
|
650
|
+
``TimeoutError`` only if the wait elapses; the job keeps running
|
|
651
|
+
server-side and can be polled again.
|
|
652
|
+
"""
|
|
653
|
+
deadline = monotonic() + timeout
|
|
654
|
+
while True:
|
|
655
|
+
job = self.get_job(job_id)
|
|
656
|
+
if job.get("status") in ("succeeded", "failed"):
|
|
657
|
+
return job
|
|
658
|
+
if monotonic() + interval >= deadline:
|
|
659
|
+
raise TimeoutError(
|
|
660
|
+
f"Job {job_id} did not finish within {timeout}s. "
|
|
661
|
+
"It is still running — poll get_job() to collect it."
|
|
662
|
+
)
|
|
663
|
+
sleep(interval)
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: parserail
|
|
3
|
+
Version: 0.6.0
|
|
4
|
+
Summary: Official Python SDK for ParseRail: parse documents, extract fields, redact PII, analyze contracts.
|
|
5
|
+
Project-URL: Homepage, https://parserail.thecompound.tech
|
|
6
|
+
Project-URL: Documentation, https://parserail.thecompound.tech/docs/sdk-python
|
|
7
|
+
Author: Compound Labs
|
|
8
|
+
License: MIT
|
|
9
|
+
Keywords: ai,api,chargeback,contract,document-parsing,llm,ocr,parserail,pii
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
15
|
+
Requires-Python: >=3.8
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
|
|
18
|
+
# parserail
|
|
19
|
+
|
|
20
|
+
Official Python SDK for **[ParseRail](https://parserail.thecompound.tech)** — the AI back-end for your product. Parse documents, extract fields, redact PII, analyze contracts, fight chargebacks, and enrich companies.
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
pip install parserail
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
## Quickstart
|
|
27
|
+
|
|
28
|
+
```python
|
|
29
|
+
from parserail import ParseRail
|
|
30
|
+
|
|
31
|
+
client = ParseRail(api_key="ksk_live_...")
|
|
32
|
+
|
|
33
|
+
doc = client.parse(file_url="https://.../invoice.pdf")
|
|
34
|
+
print(doc["totalAmount"]) # 4820.5
|
|
35
|
+
print(doc["usage"]["balanceRemaining"]) # 490
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
Get a key (and 500 free credits) at **[parserail.thecompound.tech](https://parserail.thecompound.tech)**. Zero dependencies — stdlib only, Python 3.8+.
|
|
39
|
+
|
|
40
|
+
## Methods
|
|
41
|
+
|
|
42
|
+
Every method returns the endpoint result as a dict, including a `usage` envelope. A non-2xx response raises a typed `ParseRailError` (and never burns credits).
|
|
43
|
+
|
|
44
|
+
```python
|
|
45
|
+
client.parse(file_url=...) # documents → JSON
|
|
46
|
+
client.extract(text=..., fields=["order", "total"]) # pull named fields
|
|
47
|
+
client.classify(text=..., labels=["billing", "tech"]) # label text
|
|
48
|
+
client.summarize(text=..., length="standard") # summary + actions
|
|
49
|
+
client.redact(text=...) # strip PII/PHI
|
|
50
|
+
client.sentiment(text=..., aspects=["product"]) # sentiment + aspects
|
|
51
|
+
client.contract(file_url=...) # contract → terms + risks
|
|
52
|
+
client.chargeback(reason=..., transaction={"amount": 129}) # representment packet
|
|
53
|
+
client.enrich(email="sam@stripe.com") # company profile
|
|
54
|
+
client.account() # balance
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## Async & webhooks
|
|
58
|
+
|
|
59
|
+
A hundred-page contract doesn't fit in a request/response cycle. The document endpoints — `parse`, `invoice`, `receipt`, `statement`, `resume`, `tables`, `split`, `compare`, `contract` — take `async_=True` and hand you a job instead of a result. (`async` is a Python keyword, hence the trailing underscore; the wire field is `async`.)
|
|
60
|
+
|
|
61
|
+
```python
|
|
62
|
+
job = client.parse(file_url="https://.../contract.pdf", async_=True)
|
|
63
|
+
done = client.wait_for_job(job["jobId"])
|
|
64
|
+
|
|
65
|
+
if done["status"] == "succeeded":
|
|
66
|
+
print(done["result"]["totalAmount"])
|
|
67
|
+
else:
|
|
68
|
+
print(done["error"]) # failed jobs are never charged
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Use `client.get_job(job_id)` for a single poll if you'd rather drive the loop yourself.
|
|
72
|
+
|
|
73
|
+
Every job reaches a terminal state. If the instance running yours dies mid-flight, it is marked `failed` with an explanation rather than left `running` forever — and you aren't billed for it. Nothing is silently retried; resubmit and you stay in control of the spend.
|
|
74
|
+
|
|
75
|
+
### Webhooks
|
|
76
|
+
|
|
77
|
+
Pass a `callback_url` (public https) and the finished job is POSTed to it, signed with your account's webhook secret from the [API keys page](https://parserail.thecompound.tech/dashboard/keys):
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
client.parse(file_url=..., async_=True, callback_url="https://you.example/hooks/parserail")
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
```python
|
|
84
|
+
import hmac, hashlib
|
|
85
|
+
|
|
86
|
+
# X-ParseRail-Signature: sha256=<hex HMAC-SHA256 of the RAW body>
|
|
87
|
+
def verify(raw_body: bytes, header: str, secret: str) -> bool:
|
|
88
|
+
expected = hmac.new(secret.encode(), raw_body, hashlib.sha256).hexdigest()
|
|
89
|
+
return hmac.compare_digest(header.removeprefix("sha256="), expected)
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
Delivery is best-effort and never retried — **polling is the source of truth**.
|
|
93
|
+
|
|
94
|
+
## Error handling
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
from parserail import ParseRail, ParseRailError
|
|
98
|
+
|
|
99
|
+
client = ParseRail(api_key="ksk_live_...")
|
|
100
|
+
try:
|
|
101
|
+
client.parse(file_url="https://.../invoice.pdf")
|
|
102
|
+
except ParseRailError as err:
|
|
103
|
+
# err.code: "insufficient_credits" | "rate_limited" | "unauthorized" | ...
|
|
104
|
+
print(err.code, err.status, err.message)
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
MIT © ParseRail Studios
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
parserail/__init__.py,sha256=R4LDq0JQdHINyQiBLSOzKgtsHS3BCtJTr6Q4g6rg7TU,26573
|
|
2
|
+
parserail-0.6.0.dist-info/METADATA,sha256=kirtqjXgOQLd-iOaLcJdo7DaQh4bhOjQbc4tDD9XmC8,4467
|
|
3
|
+
parserail-0.6.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
4
|
+
parserail-0.6.0.dist-info/RECORD,,
|