resumereaderapi 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- resumereaderapi/__init__.py +4 -0
- resumereaderapi/client.py +108 -0
- resumereaderapi-1.0.0.dist-info/LICENSE +21 -0
- resumereaderapi-1.0.0.dist-info/METADATA +161 -0
- resumereaderapi-1.0.0.dist-info/RECORD +7 -0
- resumereaderapi-1.0.0.dist-info/WHEEL +5 -0
- resumereaderapi-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""Client for the Resume Reader API (https://www.resumereaderapi.com)."""
|
|
2
|
+
import base64
|
|
3
|
+
import os
|
|
4
|
+
from typing import Any, Dict, Iterable, List, Optional
|
|
5
|
+
|
|
6
|
+
import requests
|
|
7
|
+
|
|
8
|
+
__version__ = "1.0.0"
|
|
9
|
+
BASE = "https://www.resumereaderapi.com/api/"
|
|
10
|
+
|
|
11
|
+
STATUS_TEXT = {
|
|
12
|
+
400: "Missing or invalid input",
|
|
13
|
+
401: "Invalid API key",
|
|
14
|
+
402: "Insufficient credits, nothing was billed",
|
|
15
|
+
413: "Document beyond 20 pages, about 30,000 tokens or 10 MB, nothing was billed",
|
|
16
|
+
422: "File could not be read as a resume",
|
|
17
|
+
429: "Rate limit exceeded, 30 requests per 60 seconds per IP",
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class ResumeReaderError(Exception):
|
|
22
|
+
"""The HTTP status is always 200, so errors come from the status field of the JSON body."""
|
|
23
|
+
|
|
24
|
+
def __init__(self, status: int, message: str, body: Optional[dict] = None):
|
|
25
|
+
super().__init__("[%s] %s" % (status, message))
|
|
26
|
+
self.status = status
|
|
27
|
+
self.body = body or {}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class ResumeReader:
|
|
31
|
+
def __init__(self, api_key: str, schema_version: int = 2, timeout: float = 120.0,
|
|
32
|
+
session: Optional[requests.Session] = None):
|
|
33
|
+
if not api_key:
|
|
34
|
+
raise ValueError("An API key is required")
|
|
35
|
+
self.api_key = api_key
|
|
36
|
+
self.schema_version = schema_version
|
|
37
|
+
self.timeout = timeout
|
|
38
|
+
self.session = session or requests.Session()
|
|
39
|
+
self.session.headers["User-Agent"] = "resumereaderapi-python/%s (+https://www.resumereaderapi.com)" % __version__
|
|
40
|
+
|
|
41
|
+
def _post(self, endpoint: str, payload: Dict[str, Any]) -> Dict[str, Any]:
|
|
42
|
+
resp = self.session.post(BASE + endpoint, json=payload, timeout=self.timeout)
|
|
43
|
+
try:
|
|
44
|
+
body = resp.json()
|
|
45
|
+
except ValueError:
|
|
46
|
+
raise ResumeReaderError(resp.status_code, "Response was not JSON")
|
|
47
|
+
status = body.get("status", 200) if isinstance(body, dict) else 200
|
|
48
|
+
if status != 200:
|
|
49
|
+
raise ResumeReaderError(status, STATUS_TEXT.get(status) or body.get("message", "API error"), body)
|
|
50
|
+
return body
|
|
51
|
+
|
|
52
|
+
def _parse(self, source: Dict[str, Any], field_names: Optional[str], exclude_sensitive: bool,
|
|
53
|
+
anonymize: bool, max_pages: Optional[int], sections: Optional[List[str]],
|
|
54
|
+
language: Optional[str]) -> Dict[str, Any]:
|
|
55
|
+
payload = {"api_key": self.api_key, "schema_version": self.schema_version}
|
|
56
|
+
payload.update(source)
|
|
57
|
+
if field_names:
|
|
58
|
+
payload["field_names"] = field_names
|
|
59
|
+
if exclude_sensitive:
|
|
60
|
+
payload["exclude_sensitive"] = True
|
|
61
|
+
if anonymize:
|
|
62
|
+
payload["anonymize"] = True
|
|
63
|
+
if max_pages:
|
|
64
|
+
payload["max_pages"] = max_pages
|
|
65
|
+
if sections:
|
|
66
|
+
payload["sections"] = sections
|
|
67
|
+
if language:
|
|
68
|
+
payload["language"] = language
|
|
69
|
+
return self._post("parse.php", payload)
|
|
70
|
+
|
|
71
|
+
def parse_text(self, text: str, field_names: Optional[str] = None, exclude_sensitive: bool = False,
|
|
72
|
+
anonymize: bool = False, max_pages: Optional[int] = None,
|
|
73
|
+
sections: Optional[List[str]] = None, language: Optional[str] = None) -> Dict[str, Any]:
|
|
74
|
+
return self._parse({"text": text}, field_names, exclude_sensitive, anonymize, max_pages, sections, language)
|
|
75
|
+
|
|
76
|
+
def parse_file(self, path: str, field_names: Optional[str] = None, exclude_sensitive: bool = False,
|
|
77
|
+
anonymize: bool = False, max_pages: Optional[int] = None,
|
|
78
|
+
sections: Optional[List[str]] = None, language: Optional[str] = None) -> Dict[str, Any]:
|
|
79
|
+
with open(path, "rb") as fh:
|
|
80
|
+
encoded = base64.b64encode(fh.read()).decode("ascii")
|
|
81
|
+
source = {"file_base64": encoded, "filename": os.path.basename(path)}
|
|
82
|
+
return self._parse(source, field_names, exclude_sensitive, anonymize, max_pages, sections, language)
|
|
83
|
+
|
|
84
|
+
def parse_url(self, file_url: str, field_names: Optional[str] = None, exclude_sensitive: bool = False,
|
|
85
|
+
anonymize: bool = False, max_pages: Optional[int] = None,
|
|
86
|
+
sections: Optional[List[str]] = None, language: Optional[str] = None) -> Dict[str, Any]:
|
|
87
|
+
return self._parse({"file_url": file_url}, field_names, exclude_sensitive, anonymize, max_pages, sections, language)
|
|
88
|
+
|
|
89
|
+
def _normalize(self, endpoint: str, key: str, items: Iterable[str]) -> Dict[str, Any]:
|
|
90
|
+
values = [items] if isinstance(items, str) else list(items)
|
|
91
|
+
results: List[Dict[str, Any]] = []
|
|
92
|
+
used = 0.0
|
|
93
|
+
remaining = None
|
|
94
|
+
for i in range(0, len(values), 100):
|
|
95
|
+
body = self._post(endpoint, {"api_key": self.api_key, key: values[i:i + 100]})
|
|
96
|
+
results.extend(body.get("results", []))
|
|
97
|
+
used += body.get("credits_used", 0)
|
|
98
|
+
remaining = body.get("remaining_credits", remaining)
|
|
99
|
+
return {"results": results, "credits_used": round(used, 4), "remaining_credits": remaining}
|
|
100
|
+
|
|
101
|
+
def normalize_titles(self, titles: Iterable[str]) -> Dict[str, Any]:
|
|
102
|
+
return self._normalize("normalize_title.php", "titles", titles)
|
|
103
|
+
|
|
104
|
+
def normalize_skills(self, skills: Iterable[str]) -> Dict[str, Any]:
|
|
105
|
+
return self._normalize("normalize_skills.php", "skills", skills)
|
|
106
|
+
|
|
107
|
+
def normalize_locations(self, locations: Iterable[str]) -> Dict[str, Any]:
|
|
108
|
+
return self._normalize("normalize_locations.php", "locations", locations)
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Alpha Quantum
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: resumereaderapi
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Python client for the Resume Reader API: parse PDF, DOCX and scanned resumes into 114 structured JSON fields, and normalize job titles, skills and locations.
|
|
5
|
+
Home-page: https://www.resumereaderapi.com
|
|
6
|
+
Author: Alpha Quantum
|
|
7
|
+
Author-email: info@alpha-quantum.com
|
|
8
|
+
License: MIT
|
|
9
|
+
Project-URL: Homepage, https://www.resumereaderapi.com
|
|
10
|
+
Project-URL: Documentation, https://www.resumereaderapi.com/api-v2.php
|
|
11
|
+
Project-URL: Source, https://github.com/explainableaixai/resumereaderapi
|
|
12
|
+
Project-URL: Mirror, https://gitlab.com/url-classifications/resumereaderapi
|
|
13
|
+
Project-URL: Pricing, https://www.resumereaderapi.com/pricing.php
|
|
14
|
+
Keywords: resume parser,cv parser,resume parsing api,resume to json,ats,recruiting,hr tech,job title normalization,skill normalization,location normalization,ocr,cv anonymization
|
|
15
|
+
Platform: UNKNOWN
|
|
16
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
17
|
+
Classifier: Intended Audience :: Developers
|
|
18
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
19
|
+
Classifier: Programming Language :: Python :: 3
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.7
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Topic :: Office/Business
|
|
23
|
+
Classifier: Topic :: Text Processing
|
|
24
|
+
Requires-Python: >=3.7
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
Requires-Dist: requests>=2.20.0
|
|
27
|
+
|
|
28
|
+
# resumereaderapi (Python)
|
|
29
|
+
|
|
30
|
+
Resume parser client for Python. Hand it a PDF, a DOCX, a scan or raw text and get structured candidate data back as a dictionary.
|
|
31
|
+
|
|
32
|
+
Needs `requests`. Python 3.7 and later.
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
pip install resumereaderapi
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
## Why use a hosted resume parser
|
|
39
|
+
|
|
40
|
+
Writing regular expressions for CVs is a trap. Layouts vary, languages vary, and half the documents are scans. A hosted resume parser gives you one stable shape no matter what the candidate uploaded.
|
|
41
|
+
|
|
42
|
+
The service behind this package is the [resume parser](https://www.resumereaderapi.com/) from Alpha Quantum. It returns 114 fields and marks missing facts as `None` instead of guessing.
|
|
43
|
+
|
|
44
|
+
## Parse in three lines
|
|
45
|
+
|
|
46
|
+
```python
|
|
47
|
+
from resumereaderapi import ResumeReader
|
|
48
|
+
|
|
49
|
+
rr = ResumeReader("YOUR_API_KEY")
|
|
50
|
+
data = rr.parse_file("cv.pdf")
|
|
51
|
+
|
|
52
|
+
print(data["resume"]["contact"]["emails"])
|
|
53
|
+
print(data["remaining_credits"])
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
## Calls
|
|
57
|
+
|
|
58
|
+
```python
|
|
59
|
+
rr.parse_text(text, field_names=None, exclude_sensitive=False, anonymize=False,
|
|
60
|
+
max_pages=None, sections=None, language=None)
|
|
61
|
+
rr.parse_file(path, ...same options...)
|
|
62
|
+
rr.parse_url(https_url, ...same options...)
|
|
63
|
+
|
|
64
|
+
rr.normalize_titles(["Sr. SWE II"])
|
|
65
|
+
rr.normalize_skills(["K8s", "JS"])
|
|
66
|
+
rr.normalize_locations(["NYC"])
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Normalizer methods batch in groups of 100 and return `{"results": [...], "credits_used": float, "remaining_credits": float}`.
|
|
70
|
+
|
|
71
|
+
## Skill normalization in practice
|
|
72
|
+
|
|
73
|
+
Candidates write the same skill ten ways. `JS`, `Javascript` and `java script` should be one filter value.
|
|
74
|
+
|
|
75
|
+
```python
|
|
76
|
+
raw = ["JS", "Javascript", "Python 3.11", "py", "MS Excel", "n/a"]
|
|
77
|
+
out = rr.normalize_skills(raw)
|
|
78
|
+
for item in out["results"]:
|
|
79
|
+
print(item["input"], "->", item["normalized_skill"], item["skill_type"])
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
Junk entries such as "n/a" come back with `normalized_skill` set to `None`. See the [skill normalization](https://www.resumereaderapi.com/normalization/skills.php) page for fifty real examples.
|
|
83
|
+
|
|
84
|
+
## Anonymize before a human sees the CV
|
|
85
|
+
|
|
86
|
+
```python
|
|
87
|
+
data = rr.parse_file(
|
|
88
|
+
"cv.docx",
|
|
89
|
+
anonymize=True,
|
|
90
|
+
exclude_sensitive=True,
|
|
91
|
+
)
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
Names, contact values and sensitive fields become placeholders such as `[NAME]` and `[EMAIL]`. This is the base of a blind review process, covered under [CV anonymization](https://www.resumereaderapi.com/use-cases/cv-anonymization.php).
|
|
95
|
+
|
|
96
|
+
## Handle errors
|
|
97
|
+
|
|
98
|
+
```python
|
|
99
|
+
from resumereaderapi import ResumeReaderError
|
|
100
|
+
|
|
101
|
+
try:
|
|
102
|
+
rr.parse_file(path)
|
|
103
|
+
except ResumeReaderError as exc:
|
|
104
|
+
if exc.status == 402:
|
|
105
|
+
print("buy more credits")
|
|
106
|
+
elif exc.status in (413, 422):
|
|
107
|
+
print("bad document:", exc)
|
|
108
|
+
else:
|
|
109
|
+
raise
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
## Example: into a DataFrame
|
|
113
|
+
|
|
114
|
+
```python
|
|
115
|
+
import pandas as pd
|
|
116
|
+
from pathlib import Path
|
|
117
|
+
|
|
118
|
+
rows = []
|
|
119
|
+
for p in Path("cvs").glob("*.pdf"):
|
|
120
|
+
r = rr.parse_file(str(p))["resume"]
|
|
121
|
+
rows.append({
|
|
122
|
+
"name": r["contact"]["full_name"],
|
|
123
|
+
"city": r["contact"]["address"]["city"],
|
|
124
|
+
"title": (r["career"]["current_position"] or {}).get("title"),
|
|
125
|
+
"years": r["career"]["total_experience_years"],
|
|
126
|
+
"skills": ", ".join(r["skills"]["technical"]),
|
|
127
|
+
})
|
|
128
|
+
|
|
129
|
+
pd.DataFrame(rows).to_csv("candidates.csv", index=False)
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
## Example: Flask upload endpoint
|
|
133
|
+
|
|
134
|
+
```python
|
|
135
|
+
from flask import Flask, request, jsonify
|
|
136
|
+
import tempfile, os
|
|
137
|
+
|
|
138
|
+
app = Flask(__name__)
|
|
139
|
+
|
|
140
|
+
@app.post("/parse")
|
|
141
|
+
def parse():
|
|
142
|
+
f = request.files["cv"]
|
|
143
|
+
with tempfile.NamedTemporaryFile(delete=False, suffix=os.path.splitext(f.filename)[1]) as tmp:
|
|
144
|
+
f.save(tmp.name)
|
|
145
|
+
try:
|
|
146
|
+
return jsonify(rr.parse_file(tmp.name)["resume"])
|
|
147
|
+
finally:
|
|
148
|
+
os.unlink(tmp.name)
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
## Questions
|
|
152
|
+
|
|
153
|
+
**What file types?** PDF, DOCX, TXT, RTF, HTML, ODT, spreadsheets and images including scanned PDFs.
|
|
154
|
+
|
|
155
|
+
**Is there a limit per call?** 10 MB, 20 pages and about 30,000 tokens.
|
|
156
|
+
|
|
157
|
+
**Does it work in French?** Pass `field_names="fr"` for French keys. Values keep the language of the resume.
|
|
158
|
+
|
|
159
|
+
MIT license. Contact: info@alpha-quantum.com
|
|
160
|
+
|
|
161
|
+
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
resumereaderapi/__init__.py,sha256=nMU49uyWRJdT1aHcV38rOs_6pVkS8TmXsAKOyvz7KI4,123
|
|
2
|
+
resumereaderapi/client.py,sha256=2kUXRmBH3eDn0DWv6207fvaYOpZcfXOwJSwAIXU1POM,5311
|
|
3
|
+
resumereaderapi-1.0.0.dist-info/LICENSE,sha256=-kjrwollysEMZkPQkJIq5zyefl9XyPw5egz8knSXiB4,1070
|
|
4
|
+
resumereaderapi-1.0.0.dist-info/METADATA,sha256=QEJMAA_AQ_itf1mWw-tlcFY8cS6LmUYbIYVGW0FQq_U,5262
|
|
5
|
+
resumereaderapi-1.0.0.dist-info/WHEEL,sha256=tZoeGjtWxWRfdplE7E3d45VPlLNQnvbKiYnx7gwAy8A,92
|
|
6
|
+
resumereaderapi-1.0.0.dist-info/top_level.txt,sha256=LVu9C0Qs8FspbvMkktipSuIdrzLDnDjblGrJ42v5xMY,16
|
|
7
|
+
resumereaderapi-1.0.0.dist-info/RECORD,,
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
resumereaderapi
|