linkfetch 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- linkfetch/__init__.py +3 -0
- linkfetch/capture/__init__.py +1 -0
- linkfetch/capture/browser.py +112 -0
- linkfetch/capture/expand.py +61 -0
- linkfetch/cli.py +274 -0
- linkfetch/config.py +105 -0
- linkfetch/emit.py +57 -0
- linkfetch/models.py +196 -0
- linkfetch/parse/__init__.py +43 -0
- linkfetch/parse/awards.py +57 -0
- linkfetch/parse/basics.py +89 -0
- linkfetch/parse/certifications.py +70 -0
- linkfetch/parse/common.py +251 -0
- linkfetch/parse/courses.py +38 -0
- linkfetch/parse/education.py +51 -0
- linkfetch/parse/experience.py +129 -0
- linkfetch/parse/honors.py +32 -0
- linkfetch/parse/languages.py +23 -0
- linkfetch/parse/patents.py +60 -0
- linkfetch/parse/projects.py +67 -0
- linkfetch/parse/recommendations.py +64 -0
- linkfetch/parse/skills.py +30 -0
- linkfetch/parse/volunteering.py +47 -0
- linkfetch/sections.py +84 -0
- linkfetch/text.py +78 -0
- linkfetch-0.1.0.dist-info/METADATA +174 -0
- linkfetch-0.1.0.dist-info/RECORD +30 -0
- linkfetch-0.1.0.dist-info/WHEEL +4 -0
- linkfetch-0.1.0.dist-info/entry_points.txt +2 -0
- linkfetch-0.1.0.dist-info/licenses/LICENSE +21 -0
linkfetch/models.py
ADDED
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
"""Models for scraped LinkedIn data.
|
|
2
|
+
|
|
3
|
+
Two families:
|
|
4
|
+
|
|
5
|
+
1. **tailor-mapped** models — a *subset* of tailor's profile models using the
|
|
6
|
+
**same field names** (``id``, ``summary``, ``bullets``, ``tags``, ``org``,
|
|
7
|
+
``positions``, ``timeline``, …). YAML emitted from these drops straight into
|
|
8
|
+
tailor's profile store. We intentionally omit fields tailor fills in by hand
|
|
9
|
+
(e.g. multiple position framings, emphasis) — the scrape is deterministic and
|
|
10
|
+
leaves enrichment to the user.
|
|
11
|
+
|
|
12
|
+
2. **extra** models — small purpose-built models for LinkedIn sections tailor
|
|
13
|
+
has no home for (education, volunteering, recommendations, …).
|
|
14
|
+
|
|
15
|
+
All ``id`` values are stable slugs derived from the item's identity so re-runs
|
|
16
|
+
produce diffable YAML.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from pydantic import BaseModel, Field
|
|
22
|
+
|
|
23
|
+
# ---------------------------------------------------------------------------
|
|
24
|
+
# shared
|
|
25
|
+
# ---------------------------------------------------------------------------
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class Timeline(BaseModel):
|
|
29
|
+
"""A start/end span. ``end`` omitted or 'present' means ongoing.
|
|
30
|
+
|
|
31
|
+
Mirrors tailor.profile.models.Timeline (field-compatible).
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
start: str | None = None
|
|
35
|
+
end: str | None = None
|
|
36
|
+
note: str | None = None
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class Position(BaseModel):
|
|
40
|
+
"""One title framing for a role. Mirrors tailor's Position.
|
|
41
|
+
|
|
42
|
+
The scrape produces exactly one framing (the real LinkedIn title); the user
|
|
43
|
+
adds alternative framings later in tailor.
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
title: str
|
|
47
|
+
seniority: str | None = None
|
|
48
|
+
emphasis: str | None = None
|
|
49
|
+
timeline: Timeline | None = None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class Link(BaseModel):
|
|
53
|
+
label: str
|
|
54
|
+
url: str
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
# ---------------------------------------------------------------------------
|
|
58
|
+
# tailor-mapped models (field names match tailor.profile.models)
|
|
59
|
+
# ---------------------------------------------------------------------------
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class Basics(BaseModel):
|
|
63
|
+
name: str
|
|
64
|
+
label: str | None = None # headline
|
|
65
|
+
email: str | None = None
|
|
66
|
+
phone: str | None = None
|
|
67
|
+
location: str | None = None
|
|
68
|
+
summary: str | None = None # the "About" section
|
|
69
|
+
links: list[Link] = Field(default_factory=list)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
class Experience(BaseModel):
|
|
73
|
+
id: str
|
|
74
|
+
org: str
|
|
75
|
+
location: str | None = None
|
|
76
|
+
positions: list[Position] = Field(default_factory=list)
|
|
77
|
+
timeline: Timeline | None = None
|
|
78
|
+
summary: str | None = None
|
|
79
|
+
bullets: list[str] = Field(default_factory=list)
|
|
80
|
+
tags: list[str] = Field(default_factory=list)
|
|
81
|
+
tech: list[str] = Field(default_factory=list)
|
|
82
|
+
links: list[Link] = Field(default_factory=list)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class Skill(BaseModel):
|
|
86
|
+
id: str
|
|
87
|
+
name: str
|
|
88
|
+
category: str | None = None
|
|
89
|
+
level: str | None = None
|
|
90
|
+
years: float | None = None
|
|
91
|
+
summary: str | None = None
|
|
92
|
+
bullets: list[str] = Field(default_factory=list)
|
|
93
|
+
tags: list[str] = Field(default_factory=list)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class Project(BaseModel):
|
|
97
|
+
id: str
|
|
98
|
+
name: str
|
|
99
|
+
role: str | None = None
|
|
100
|
+
timeline: Timeline | None = None
|
|
101
|
+
summary: str | None = None
|
|
102
|
+
bullets: list[str] = Field(default_factory=list)
|
|
103
|
+
tags: list[str] = Field(default_factory=list)
|
|
104
|
+
tech: list[str] = Field(default_factory=list)
|
|
105
|
+
links: list[Link] = Field(default_factory=list)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class Certification(BaseModel):
|
|
109
|
+
id: str
|
|
110
|
+
name: str
|
|
111
|
+
issuer: str | None = None
|
|
112
|
+
issued: str | None = None
|
|
113
|
+
expires: str | None = None
|
|
114
|
+
credential_id: str | None = None
|
|
115
|
+
summary: str | None = None
|
|
116
|
+
bullets: list[str] = Field(default_factory=list)
|
|
117
|
+
tags: list[str] = Field(default_factory=list)
|
|
118
|
+
links: list[Link] = Field(default_factory=list)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
class Award(BaseModel):
|
|
122
|
+
id: str
|
|
123
|
+
name: str
|
|
124
|
+
issuer: str | None = None
|
|
125
|
+
date: str | None = None
|
|
126
|
+
summary: str | None = None
|
|
127
|
+
bullets: list[str] = Field(default_factory=list)
|
|
128
|
+
tags: list[str] = Field(default_factory=list)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
# ---------------------------------------------------------------------------
|
|
132
|
+
# extra models (linkfetch-local schema; not in tailor)
|
|
133
|
+
# ---------------------------------------------------------------------------
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
class Education(BaseModel):
|
|
137
|
+
id: str
|
|
138
|
+
school: str
|
|
139
|
+
degree: str | None = None
|
|
140
|
+
field: str | None = None
|
|
141
|
+
timeline: Timeline | None = None
|
|
142
|
+
summary: str | None = None
|
|
143
|
+
bullets: list[str] = Field(default_factory=list)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
class Volunteering(BaseModel):
|
|
147
|
+
id: str
|
|
148
|
+
org: str
|
|
149
|
+
role: str | None = None
|
|
150
|
+
cause: str | None = None
|
|
151
|
+
timeline: Timeline | None = None
|
|
152
|
+
summary: str | None = None
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
class Recommendation(BaseModel):
|
|
156
|
+
id: str
|
|
157
|
+
author: str
|
|
158
|
+
author_title: str | None = None
|
|
159
|
+
relationship: str | None = None
|
|
160
|
+
text: str | None = None
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
class Patent(BaseModel):
|
|
164
|
+
id: str
|
|
165
|
+
title: str
|
|
166
|
+
number: str | None = None
|
|
167
|
+
status: str | None = None
|
|
168
|
+
date: str | None = None
|
|
169
|
+
summary: str | None = None
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
class Course(BaseModel):
|
|
173
|
+
id: str
|
|
174
|
+
name: str
|
|
175
|
+
number: str | None = None
|
|
176
|
+
associated_with: str | None = None
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
class Language(BaseModel):
|
|
180
|
+
id: str
|
|
181
|
+
name: str
|
|
182
|
+
proficiency: str | None = None
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
class Honor(BaseModel):
|
|
186
|
+
"""Honors & awards as a standalone extra section.
|
|
187
|
+
|
|
188
|
+
Note: tailor's Award is also populated from this section; this richer extra
|
|
189
|
+
file preserves everything (issuer, date, description) for the user.
|
|
190
|
+
"""
|
|
191
|
+
|
|
192
|
+
id: str
|
|
193
|
+
title: str
|
|
194
|
+
issuer: str | None = None
|
|
195
|
+
date: str | None = None
|
|
196
|
+
summary: str | None = None
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""Per-section HTML parsers.
|
|
2
|
+
|
|
3
|
+
Each section module exposes ``parse(html: str) -> list[Model]`` (or a single
|
|
4
|
+
model for basics). Parsers are **pure**: lxml in, pydantic out, no browser, no
|
|
5
|
+
network. They anchor on LinkedIn's *stable* structure — the ``<main>`` content
|
|
6
|
+
``<section>``, its repeated entry blocks, and the ordered ``<p>``/``<span>`` text
|
|
7
|
+
fragments inside each (see :mod:`linkfetch.parse.common`) — rather than the
|
|
8
|
+
randomized hashed utility class names, so they survive cosmetic markup churn.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from linkfetch.parse import (
|
|
14
|
+
awards,
|
|
15
|
+
basics,
|
|
16
|
+
certifications,
|
|
17
|
+
courses,
|
|
18
|
+
education,
|
|
19
|
+
experience,
|
|
20
|
+
honors,
|
|
21
|
+
languages,
|
|
22
|
+
patents,
|
|
23
|
+
projects,
|
|
24
|
+
recommendations,
|
|
25
|
+
skills,
|
|
26
|
+
volunteering,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
__all__ = [
|
|
30
|
+
"awards",
|
|
31
|
+
"basics",
|
|
32
|
+
"certifications",
|
|
33
|
+
"courses",
|
|
34
|
+
"education",
|
|
35
|
+
"experience",
|
|
36
|
+
"honors",
|
|
37
|
+
"languages",
|
|
38
|
+
"patents",
|
|
39
|
+
"projects",
|
|
40
|
+
"recommendations",
|
|
41
|
+
"skills",
|
|
42
|
+
"volunteering",
|
|
43
|
+
]
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""Parse Honors & awards into tailor's Award model.
|
|
2
|
+
|
|
3
|
+
Ordered fragments per entry::
|
|
4
|
+
|
|
5
|
+
[title, "Issued by <issuer> · <date>", "Associated with <entity>", <description>]
|
|
6
|
+
|
|
7
|
+
The issuer and date are pulled from the "Issued by … · …" fragment; the
|
|
8
|
+
expandable description (when present) becomes the bullets. See honors.py for the
|
|
9
|
+
richer extra-file view of the same section.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import re
|
|
15
|
+
|
|
16
|
+
from linkfetch.models import Award
|
|
17
|
+
from linkfetch.parse.common import description_text, entry_items, ordered_texts
|
|
18
|
+
from linkfetch.text import clean, slugify, split_bullets
|
|
19
|
+
|
|
20
|
+
_ISSUED_BY = re.compile(r"^issued by\s+", re.I)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def split_issuer_date(frag: str) -> tuple[str | None, str | None]:
|
|
24
|
+
"""``"Issued by Roche · Apr 2021"`` -> ``("Roche", "Apr 2021")``."""
|
|
25
|
+
body = _ISSUED_BY.sub("", frag)
|
|
26
|
+
issuer, sep, date = body.partition("·")
|
|
27
|
+
return clean(issuer), (clean(date) if sep else None)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def extract(texts: list[str]) -> tuple[str, str | None, str | None]:
|
|
31
|
+
"""Return ``(title, issuer, date)`` from an entry's fragments."""
|
|
32
|
+
title = texts[0]
|
|
33
|
+
issuer = date = None
|
|
34
|
+
for frag in texts[1:]:
|
|
35
|
+
if frag.lower().startswith("issued by"):
|
|
36
|
+
issuer, date = split_issuer_date(frag)
|
|
37
|
+
break
|
|
38
|
+
return title, issuer, date
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def parse(html: str) -> list[Award]:
|
|
42
|
+
out: list[Award] = []
|
|
43
|
+
for el in entry_items(html):
|
|
44
|
+
texts = ordered_texts(el)
|
|
45
|
+
if not texts:
|
|
46
|
+
continue
|
|
47
|
+
title, issuer, date = extract(texts)
|
|
48
|
+
out.append(
|
|
49
|
+
Award(
|
|
50
|
+
id=slugify(title, issuer),
|
|
51
|
+
name=title,
|
|
52
|
+
issuer=issuer,
|
|
53
|
+
date=date,
|
|
54
|
+
bullets=split_bullets(description_text(el)),
|
|
55
|
+
)
|
|
56
|
+
)
|
|
57
|
+
return out
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""Parse the top profile card for basics (name, headline, location, about).
|
|
2
|
+
|
|
3
|
+
Unlike the section details pages, basics come from the main profile page. The
|
|
4
|
+
captured DOM has no ``<h1>`` or ``#about`` anchor, so we anchor on stable signals
|
|
5
|
+
instead:
|
|
6
|
+
|
|
7
|
+
* **name** — the page ``<title>`` ("Jane Doe | LinkedIn"), with the first
|
|
8
|
+
matching visible fragment as a fallback;
|
|
9
|
+
* **headline** — the visible fragment immediately following the name;
|
|
10
|
+
* **location** — the first short "City, Region, Country" fragment;
|
|
11
|
+
* **about** — the single expandable text box on the page (the About summary).
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import re
|
|
17
|
+
|
|
18
|
+
from lxml import html as lhtml
|
|
19
|
+
|
|
20
|
+
from linkfetch.models import Basics
|
|
21
|
+
from linkfetch.parse.common import description_text, ordered_texts
|
|
22
|
+
from linkfetch.text import clean
|
|
23
|
+
|
|
24
|
+
# A "City, Region, Country" style location: 2-3 comma-separated short parts.
|
|
25
|
+
_LOCATION = re.compile(r"^[^,\n]{2,40}(?:,[^,\n]{2,40}){1,2}$")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def parse(html: str) -> Basics:
|
|
29
|
+
if not html or not html.strip():
|
|
30
|
+
return Basics(name="")
|
|
31
|
+
root = lhtml.fromstring(html)
|
|
32
|
+
|
|
33
|
+
name = _name(root)
|
|
34
|
+
fragments = ordered_texts(root)
|
|
35
|
+
|
|
36
|
+
headline = None
|
|
37
|
+
if name:
|
|
38
|
+
for i, frag in enumerate(fragments):
|
|
39
|
+
if frag == name and i + 1 < len(fragments):
|
|
40
|
+
headline = fragments[i + 1]
|
|
41
|
+
break
|
|
42
|
+
|
|
43
|
+
# Location: a "City, Region, Country" fragment in the top card, anchored by
|
|
44
|
+
# the "Contact info"/"connections" line that LinkedIn renders right after it.
|
|
45
|
+
location = None
|
|
46
|
+
for i, frag in enumerate(fragments):
|
|
47
|
+
if frag in (name, headline) or not _LOCATION.match(frag):
|
|
48
|
+
continue
|
|
49
|
+
if _looks_like_company(frag) or _is_location_junk(frag):
|
|
50
|
+
continue
|
|
51
|
+
nearby = " ".join(fragments[i + 1 : i + 4]).lower()
|
|
52
|
+
if "contact info" in nearby or "connection" in nearby:
|
|
53
|
+
location = frag
|
|
54
|
+
break
|
|
55
|
+
|
|
56
|
+
return Basics(
|
|
57
|
+
name=name or "",
|
|
58
|
+
label=headline,
|
|
59
|
+
location=location,
|
|
60
|
+
summary=description_text(root, exclude_linked=True),
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _name(root: lhtml.HtmlElement) -> str | None:
|
|
65
|
+
titles = root.xpath("//title/text()")
|
|
66
|
+
if titles:
|
|
67
|
+
title = clean(titles[0]) or ""
|
|
68
|
+
# "Jane Doe | LinkedIn" / "… | (5) LinkedIn"
|
|
69
|
+
name = re.split(r"\s*[|–—]\s*", title)[0]
|
|
70
|
+
name = clean(name)
|
|
71
|
+
if name and name.lower() != "linkedin":
|
|
72
|
+
return name
|
|
73
|
+
h1 = root.xpath("//h1")
|
|
74
|
+
for h in h1:
|
|
75
|
+
txt = clean(h.text_content())
|
|
76
|
+
if txt:
|
|
77
|
+
return txt
|
|
78
|
+
return None
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _looks_like_company(frag: str) -> bool:
|
|
82
|
+
low = frag.lower()
|
|
83
|
+
return "·" in frag or "|" in frag or "linkedin" in low
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _is_location_junk(frag: str) -> bool:
|
|
87
|
+
"""Reject media-player / dialog strings that happen to contain commas."""
|
|
88
|
+
low = frag.lower()
|
|
89
|
+
return any(w in low for w in ("live", "seek", "modal", "dialog", "opacity"))
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""Parse the Licenses & certifications details page.
|
|
2
|
+
|
|
3
|
+
Ordered fragments per entry::
|
|
4
|
+
|
|
5
|
+
[name, issuer, "Issued <date>" (· "Expires <date>"), "Show credential",
|
|
6
|
+
"Credential ID …", "Skills: …"]
|
|
7
|
+
|
|
8
|
+
The credential URL is the anchor whose visible text is "Show credential".
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
|
|
15
|
+
from linkfetch.models import Certification, Link
|
|
16
|
+
from linkfetch.parse.common import entry_items, ordered_texts
|
|
17
|
+
from linkfetch.text import clean, slugify
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def parse(html: str) -> list[Certification]:
|
|
21
|
+
out: list[Certification] = []
|
|
22
|
+
for el in entry_items(html):
|
|
23
|
+
texts = ordered_texts(el)
|
|
24
|
+
if not texts:
|
|
25
|
+
continue
|
|
26
|
+
name = texts[0]
|
|
27
|
+
issuer = texts[1] if len(texts) > 1 else None
|
|
28
|
+
|
|
29
|
+
issued = expires = credential_id = None
|
|
30
|
+
for frag in texts[2:]:
|
|
31
|
+
low = frag.lower()
|
|
32
|
+
if "issued" in low or "expires" in low or "expir" in low:
|
|
33
|
+
issued, expires = _split_dates(frag)
|
|
34
|
+
elif "credential id" in low:
|
|
35
|
+
credential_id = clean(frag.split(":", 1)[-1])
|
|
36
|
+
|
|
37
|
+
href = _credential_link(el)
|
|
38
|
+
links = [Link(label="Credential", url=href)] if href else []
|
|
39
|
+
out.append(
|
|
40
|
+
Certification(
|
|
41
|
+
id=slugify(name, issuer),
|
|
42
|
+
name=name,
|
|
43
|
+
issuer=issuer,
|
|
44
|
+
issued=issued,
|
|
45
|
+
expires=expires,
|
|
46
|
+
credential_id=credential_id,
|
|
47
|
+
links=links,
|
|
48
|
+
)
|
|
49
|
+
)
|
|
50
|
+
return out
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _credential_link(el) -> str | None:
|
|
54
|
+
"""Anchor whose text reads 'Show credential' (the external credential URL)."""
|
|
55
|
+
for a in el.xpath(".//a[@href]"):
|
|
56
|
+
text = clean(a.text_content()) or ""
|
|
57
|
+
if "credential" in text.lower():
|
|
58
|
+
return a.get("href").split("?")[0]
|
|
59
|
+
return None
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _split_dates(text: str) -> tuple[str | None, str | None]:
|
|
63
|
+
issued = expires = None
|
|
64
|
+
m = re.search(r"issued\s+(.+?)(?:·|$)", text, re.I)
|
|
65
|
+
if m:
|
|
66
|
+
issued = clean(m.group(1))
|
|
67
|
+
m = re.search(r"expir\w*\s+(.+?)(?:·|$)", text, re.I)
|
|
68
|
+
if m:
|
|
69
|
+
expires = clean(m.group(1))
|
|
70
|
+
return issued, expires
|