linkfetch 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- linkfetch/__init__.py +3 -0
- linkfetch/capture/__init__.py +1 -0
- linkfetch/capture/browser.py +112 -0
- linkfetch/capture/expand.py +61 -0
- linkfetch/cli.py +274 -0
- linkfetch/config.py +105 -0
- linkfetch/emit.py +57 -0
- linkfetch/models.py +196 -0
- linkfetch/parse/__init__.py +43 -0
- linkfetch/parse/awards.py +57 -0
- linkfetch/parse/basics.py +89 -0
- linkfetch/parse/certifications.py +70 -0
- linkfetch/parse/common.py +251 -0
- linkfetch/parse/courses.py +38 -0
- linkfetch/parse/education.py +51 -0
- linkfetch/parse/experience.py +129 -0
- linkfetch/parse/honors.py +32 -0
- linkfetch/parse/languages.py +23 -0
- linkfetch/parse/patents.py +60 -0
- linkfetch/parse/projects.py +67 -0
- linkfetch/parse/recommendations.py +64 -0
- linkfetch/parse/skills.py +30 -0
- linkfetch/parse/volunteering.py +47 -0
- linkfetch/sections.py +84 -0
- linkfetch/text.py +78 -0
- linkfetch-0.1.0.dist-info/METADATA +174 -0
- linkfetch-0.1.0.dist-info/RECORD +30 -0
- linkfetch-0.1.0.dist-info/WHEEL +4 -0
- linkfetch-0.1.0.dist-info/entry_points.txt +2 -0
- linkfetch-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,251 @@
|
|
|
1
|
+
"""Shared helpers for section parsers.
|
|
2
|
+
|
|
3
|
+
These target LinkedIn's **current** rendered DOM (captured via Playwright), which:
|
|
4
|
+
|
|
5
|
+
* uses **obfuscated, per-build hashed class names** (``_759d3eea`` …) — so we never
|
|
6
|
+
key off classes;
|
|
7
|
+
* puts visible text in ``<p>`` and ``<span>`` elements, frequently emitting the
|
|
8
|
+
**same string twice in a row** (a visible copy + a screen-reader copy);
|
|
9
|
+
* lays out each section under ``<main>`` inside a single content ``<section>``
|
|
10
|
+
(the section title is its first visible fragment), with the footer, ad widgets
|
|
11
|
+
and global nav living *outside* that section;
|
|
12
|
+
* renders each entry as a repeated sibling block (a ``<div>`` or ``<li>``) inside
|
|
13
|
+
that section — **not** the old single ``<ul>`` of ``<li>``s;
|
|
14
|
+
* nests multiple roles at one company in an inner ``<ul>`` of role ``<li>``s
|
|
15
|
+
inside the company's block;
|
|
16
|
+
* wraps long descriptions in a ``<span data-testid="expandable-text-box">`` whose
|
|
17
|
+
line breaks are real ``<br/>`` tags (which ``text_content()`` would otherwise
|
|
18
|
+
flatten).
|
|
19
|
+
|
|
20
|
+
We anchor on this **structure** (``main`` → content ``section`` → repeating entry
|
|
21
|
+
blocks → ordered text fragments), which is far more stable than the hashed
|
|
22
|
+
classes.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import re
|
|
28
|
+
|
|
29
|
+
from lxml import html as lhtml
|
|
30
|
+
|
|
31
|
+
from linkfetch.text import clean
|
|
32
|
+
|
|
33
|
+
# Chrome / nav / footer / ad boilerplate that leaks into the captured page.
|
|
34
|
+
# Matched as a lowercase *substring* against an entry's head text so an entry
|
|
35
|
+
# that is purely chrome is dropped.
|
|
36
|
+
_CHROME = (
|
|
37
|
+
"home", "my network", "jobs", "messaging", "notifications",
|
|
38
|
+
"skip to search", "skip to main content", "skip to active conversation",
|
|
39
|
+
"back to search results", "why am i seeing this ad", "manage your ad",
|
|
40
|
+
"hide or report this ad", "report this ad", "ad options",
|
|
41
|
+
"linkedin corporation", "questions?", "visit our help center",
|
|
42
|
+
"manage your account and privacy", "go to your settings",
|
|
43
|
+
"recommendation transparency", "learn more about recommended content",
|
|
44
|
+
"select language", "profile language", "accessibility", "talent solutions",
|
|
45
|
+
"community guidelines", "marketing solutions", "privacy & terms",
|
|
46
|
+
"ad choices", "sales solutions", "small business", "safety center",
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
# Section-title fragments that appear as the first child block of the content
|
|
50
|
+
# section and must not be treated as an entry.
|
|
51
|
+
_SECTION_TITLES = (
|
|
52
|
+
"experience", "education", "skills", "projects",
|
|
53
|
+
"licenses & certifications", "licenses and certifications", "certifications",
|
|
54
|
+
"honors & awards", "honors and awards", "volunteering",
|
|
55
|
+
"recommendations", "patents", "courses", "languages",
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _root(page_html: str) -> lhtml.HtmlElement | None:
|
|
60
|
+
if not page_html or not page_html.strip():
|
|
61
|
+
return None
|
|
62
|
+
return lhtml.fromstring(page_html)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _content_section(page_html: str) -> lhtml.HtmlElement | None:
|
|
66
|
+
"""Return the ``<main>``'s primary content ``<section>`` (the entry scope).
|
|
67
|
+
|
|
68
|
+
The captured page always carries one ``<section aria-label="Primary
|
|
69
|
+
content">`` — the real entries — alongside a fixed-size "People also
|
|
70
|
+
viewed" sidebar section that renders identically on every details page.
|
|
71
|
+
We take the labelled section outright when it is there. Falling back to
|
|
72
|
+
"most visible text fragments" picks the sidebar instead whenever a
|
|
73
|
+
person's real section (projects, certifications, honors, a short
|
|
74
|
+
education list, …) happens to be smaller than that sidebar's fixed
|
|
75
|
+
fragment count — not a rare case, since plenty of profiles are sparse in
|
|
76
|
+
exactly those sections. The fragment-count heuristic remains only as a
|
|
77
|
+
fallback for a page that, for whatever reason, has no labelled section.
|
|
78
|
+
"""
|
|
79
|
+
root = _root(page_html)
|
|
80
|
+
if root is None:
|
|
81
|
+
return None
|
|
82
|
+
mains = root.xpath("//main")
|
|
83
|
+
scope = mains[0] if mains else root
|
|
84
|
+
sections = scope.xpath(".//section")
|
|
85
|
+
if not sections:
|
|
86
|
+
return scope
|
|
87
|
+
labelled = [s for s in sections if s.get("aria-label") == "Primary content"]
|
|
88
|
+
if labelled:
|
|
89
|
+
return labelled[0]
|
|
90
|
+
return max(sections, key=lambda s: len(ordered_texts(s)))
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _is_chrome(text: str | None) -> bool:
|
|
94
|
+
if not text:
|
|
95
|
+
return True
|
|
96
|
+
low = text.lower()
|
|
97
|
+
return any(hint in low for hint in _CHROME)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def entry_items(page_html: str) -> list[lhtml.HtmlElement]:
|
|
101
|
+
"""Return the top-level entry blocks of a details page.
|
|
102
|
+
|
|
103
|
+
Finds the content ``<section>`` under ``<main>``, then the element whose
|
|
104
|
+
direct ``<div>``/``<li>``/``<section>`` children are the repeated entry
|
|
105
|
+
blocks. An entry child is one carrying at least two distinct visible text
|
|
106
|
+
fragments; we pick the container with the **most** such children (the
|
|
107
|
+
deepest one on ties — so we land on the real entry list rather than a
|
|
108
|
+
wrapper). Section-title and chrome blocks are then filtered out.
|
|
109
|
+
"""
|
|
110
|
+
scope = _content_section(page_html)
|
|
111
|
+
if scope is None:
|
|
112
|
+
return []
|
|
113
|
+
|
|
114
|
+
best: lhtml.HtmlElement | None = None
|
|
115
|
+
best_n = 0
|
|
116
|
+
best_depth = -1
|
|
117
|
+
for el in scope.iter():
|
|
118
|
+
entries = [
|
|
119
|
+
c
|
|
120
|
+
for c in el
|
|
121
|
+
if c.tag in ("div", "li", "section") and len(ordered_texts(c)) >= 2
|
|
122
|
+
]
|
|
123
|
+
n = len(entries)
|
|
124
|
+
if n == 0 or n < best_n:
|
|
125
|
+
continue
|
|
126
|
+
depth = sum(1 for _ in el.iterancestors())
|
|
127
|
+
if n > best_n or (n == best_n and depth > best_depth):
|
|
128
|
+
best_n = n
|
|
129
|
+
best = el
|
|
130
|
+
best_depth = depth
|
|
131
|
+
|
|
132
|
+
if best is None:
|
|
133
|
+
return []
|
|
134
|
+
|
|
135
|
+
items: list[lhtml.HtmlElement] = []
|
|
136
|
+
for kid in best:
|
|
137
|
+
if kid.tag not in ("div", "li", "section"):
|
|
138
|
+
continue
|
|
139
|
+
texts = ordered_texts(kid)
|
|
140
|
+
head = texts[0] if texts else None
|
|
141
|
+
if head is None or _is_chrome(head):
|
|
142
|
+
continue
|
|
143
|
+
if head.lower().strip() in _SECTION_TITLES:
|
|
144
|
+
continue
|
|
145
|
+
items.append(kid)
|
|
146
|
+
return items
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _head_text(el: lhtml.HtmlElement) -> str | None:
|
|
150
|
+
for frag in ordered_texts(el):
|
|
151
|
+
return frag
|
|
152
|
+
return None
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def ordered_texts(el: lhtml.HtmlElement) -> list[str]:
|
|
156
|
+
"""Ordered, de-duplicated visible text fragments inside ``el``.
|
|
157
|
+
|
|
158
|
+
Reads ``<p>`` and ``<span>`` text in document order, dropping consecutive
|
|
159
|
+
duplicates (LinkedIn's visible + screen-reader copies) and empties.
|
|
160
|
+
"""
|
|
161
|
+
out: list[str] = []
|
|
162
|
+
for node in el.iter():
|
|
163
|
+
if node.tag not in ("p", "span"):
|
|
164
|
+
continue
|
|
165
|
+
txt = clean(node.text_content())
|
|
166
|
+
if not txt:
|
|
167
|
+
continue
|
|
168
|
+
if out and out[-1] == txt:
|
|
169
|
+
continue
|
|
170
|
+
out.append(txt)
|
|
171
|
+
return out
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def sub_entries(li: lhtml.HtmlElement) -> list[lhtml.HtmlElement]:
|
|
175
|
+
"""Return nested sub-entry ``<li>``s (e.g. multiple roles at one company)."""
|
|
176
|
+
nested = li.xpath(".//ul/li")
|
|
177
|
+
return [n for n in nested if _head_text(n)]
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def first_link(el: lhtml.HtmlElement, *, contains: str | None = None) -> str | None:
|
|
181
|
+
"""First anchor ``href`` inside ``el``; optionally requiring a substring."""
|
|
182
|
+
for a in el.xpath(".//a[@href]"):
|
|
183
|
+
href = a.get("href")
|
|
184
|
+
if href and (contains is None or contains in href):
|
|
185
|
+
return href.split("?")[0]
|
|
186
|
+
return None
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def description_text(el: lhtml.HtmlElement, *, exclude_linked: bool = False) -> str | None:
|
|
190
|
+
"""Long-form description inside ``el``, preserving ``<br/>`` as newlines.
|
|
191
|
+
|
|
192
|
+
LinkedIn renders the description in a ``<span data-testid="expandable-text-box">``
|
|
193
|
+
using real ``<br/>`` tags for line breaks (which ``text_content()`` flattens).
|
|
194
|
+
We pick the longest such box and convert ``<br/>`` to newlines so the
|
|
195
|
+
description splits cleanly into bullets. Returns ``None`` when the entry has
|
|
196
|
+
no such box (i.e. genuinely has no description).
|
|
197
|
+
|
|
198
|
+
``exclude_linked`` drops any box that sits inside an ``<a>``. On a details
|
|
199
|
+
page every box in scope is the one genuine description, so this is off by
|
|
200
|
+
default; the main profile page also renders every recent Activity post's
|
|
201
|
+
preview through this same component, each wrapped in a link to the post,
|
|
202
|
+
which "pick the longest box" would otherwise treat as a candidate for the
|
|
203
|
+
person's About text.
|
|
204
|
+
"""
|
|
205
|
+
boxes = el.xpath('.//*[@data-testid="expandable-text-box"]')
|
|
206
|
+
if exclude_linked:
|
|
207
|
+
boxes = [b for b in boxes if not b.xpath("ancestor::a")]
|
|
208
|
+
best = ""
|
|
209
|
+
for box in boxes:
|
|
210
|
+
text = _text_with_breaks(box)
|
|
211
|
+
# Drop the trailing "… more" expand affordance LinkedIn appends.
|
|
212
|
+
text = re.sub(r"\s*…\s*more\s*$", "", text)
|
|
213
|
+
if len(text) > len(best):
|
|
214
|
+
best = text
|
|
215
|
+
return clean(best) or None
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _text_with_breaks(el: lhtml.HtmlElement) -> str:
|
|
219
|
+
"""``text_content()`` but with ``<br/>`` rendered as a newline."""
|
|
220
|
+
parts: list[str] = []
|
|
221
|
+
if el.text:
|
|
222
|
+
parts.append(el.text)
|
|
223
|
+
for child in el:
|
|
224
|
+
if child.tag == "br":
|
|
225
|
+
parts.append("\n")
|
|
226
|
+
else:
|
|
227
|
+
parts.append(_text_with_breaks(child))
|
|
228
|
+
if child.tail:
|
|
229
|
+
parts.append(child.tail)
|
|
230
|
+
return "".join(parts)
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def split_timeline(date_range: str | None) -> tuple[str | None, str | None]:
|
|
234
|
+
"""Split ``"Mar 2021 - Present · 3 yrs"`` into ``(start, end)``.
|
|
235
|
+
|
|
236
|
+
The duration suffix after ``·`` is dropped. A single date with no separator
|
|
237
|
+
is treated as the start.
|
|
238
|
+
"""
|
|
239
|
+
if not date_range:
|
|
240
|
+
return None, None
|
|
241
|
+
text = date_range.split("·")[0].strip()
|
|
242
|
+
for sep in (" – ", " - ", " — ", "–", "—", " to "):
|
|
243
|
+
if sep in text:
|
|
244
|
+
start, _, end = text.partition(sep)
|
|
245
|
+
return clean(start), clean(end)
|
|
246
|
+
return clean(text), None
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def looks_like_dates(text: str) -> bool:
|
|
250
|
+
"""True if the fragment carries a 4-digit year or 'Present'."""
|
|
251
|
+
return bool(re.search(r"\b(19|20)\d{2}\b", text)) or "present" in text.lower()
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Parse the Courses details page.
|
|
2
|
+
|
|
3
|
+
Ordered fragments per entry::
|
|
4
|
+
|
|
5
|
+
[name, "<course number>", "Associated with <school>"]
|
|
6
|
+
|
|
7
|
+
e.g. ``["Algorithm Design and Analysis", "CSE 100", "Associated with …"]``.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from linkfetch.models import Course
|
|
13
|
+
from linkfetch.parse.common import entry_items, ordered_texts
|
|
14
|
+
from linkfetch.text import clean, slugify
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def parse(html: str) -> list[Course]:
|
|
18
|
+
out: list[Course] = []
|
|
19
|
+
for el in entry_items(html):
|
|
20
|
+
texts = ordered_texts(el)
|
|
21
|
+
if not texts:
|
|
22
|
+
continue
|
|
23
|
+
name = texts[0]
|
|
24
|
+
number = associated = None
|
|
25
|
+
for frag in texts[1:]:
|
|
26
|
+
if frag.lower().startswith("associated with"):
|
|
27
|
+
associated = clean(frag.split("with", 1)[-1])
|
|
28
|
+
elif number is None:
|
|
29
|
+
number = frag
|
|
30
|
+
out.append(
|
|
31
|
+
Course(
|
|
32
|
+
id=slugify(name, number),
|
|
33
|
+
name=name,
|
|
34
|
+
number=number,
|
|
35
|
+
associated_with=associated,
|
|
36
|
+
)
|
|
37
|
+
)
|
|
38
|
+
return out
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""Parse the Education details page.
|
|
2
|
+
|
|
3
|
+
Ordered fragments per entry::
|
|
4
|
+
|
|
5
|
+
[school, "Degree, Field of study", "2016 – 2021", <activities/description>]
|
|
6
|
+
|
|
7
|
+
The degree/field fragment is split on the first comma; the timeline fragment is
|
|
8
|
+
a bare year range. The long-form "Activities and societies: …" / description
|
|
9
|
+
text (an expandable box) becomes the bullets.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from linkfetch.models import Education, Timeline
|
|
15
|
+
from linkfetch.parse.common import (
|
|
16
|
+
description_text,
|
|
17
|
+
entry_items,
|
|
18
|
+
looks_like_dates,
|
|
19
|
+
ordered_texts,
|
|
20
|
+
split_timeline,
|
|
21
|
+
)
|
|
22
|
+
from linkfetch.text import clean, slugify, split_bullets
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def parse(html: str) -> list[Education]:
|
|
26
|
+
out: list[Education] = []
|
|
27
|
+
for el in entry_items(html):
|
|
28
|
+
texts = ordered_texts(el)
|
|
29
|
+
if not texts:
|
|
30
|
+
continue
|
|
31
|
+
school = texts[0]
|
|
32
|
+
|
|
33
|
+
degree = field = None
|
|
34
|
+
start = end = None
|
|
35
|
+
for frag in texts[1:]:
|
|
36
|
+
if start is None and looks_like_dates(frag):
|
|
37
|
+
start, end = split_timeline(frag)
|
|
38
|
+
elif degree is None and not frag.lower().startswith("activities"):
|
|
39
|
+
degree, _, field = (clean(p) for p in frag.partition(","))
|
|
40
|
+
|
|
41
|
+
out.append(
|
|
42
|
+
Education(
|
|
43
|
+
id=slugify(school, degree),
|
|
44
|
+
school=school,
|
|
45
|
+
degree=degree,
|
|
46
|
+
field=field,
|
|
47
|
+
timeline=Timeline(start=start, end=end) if start else None,
|
|
48
|
+
bullets=split_bullets(description_text(el)),
|
|
49
|
+
)
|
|
50
|
+
)
|
|
51
|
+
return out
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
"""Parse the Experience details page.
|
|
2
|
+
|
|
3
|
+
Each entry block is a *company*. There are two shapes:
|
|
4
|
+
|
|
5
|
+
* **Single role** — the block's ordered fragments are::
|
|
6
|
+
|
|
7
|
+
[title, "Org · Employment-type", "date range · duration", location,
|
|
8
|
+
<description>, "Skills: …"]
|
|
9
|
+
|
|
10
|
+
* **Multiple roles** — the block heads with the company, then nests one role per
|
|
11
|
+
inner ``<li>``::
|
|
12
|
+
|
|
13
|
+
company head: ["Roche", "Full-time · 5 yrs 4 mos", "Hybrid", …]
|
|
14
|
+
each role li: [title, "date range · duration", location, <description>,
|
|
15
|
+
"Skills: …"]
|
|
16
|
+
|
|
17
|
+
We emit **one** :class:`Experience` per role; for multi-role companies the
|
|
18
|
+
``org`` is the company head. The ``· duration`` / employment-type suffixes are
|
|
19
|
+
stripped from ``org``; the description becomes ``bullets``; the trailing
|
|
20
|
+
``Skills: …`` fragment is parsed into ``tech``.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
from lxml import html as lhtml
|
|
26
|
+
|
|
27
|
+
from linkfetch.models import Experience, Link, Position, Timeline
|
|
28
|
+
from linkfetch.parse.common import (
|
|
29
|
+
description_text,
|
|
30
|
+
entry_items,
|
|
31
|
+
first_link,
|
|
32
|
+
looks_like_dates,
|
|
33
|
+
ordered_texts,
|
|
34
|
+
split_timeline,
|
|
35
|
+
sub_entries,
|
|
36
|
+
)
|
|
37
|
+
from linkfetch.text import clean, slugify, split_bullets
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _strip_suffix(text: str | None) -> str | None:
|
|
41
|
+
"""Drop the ``· …`` suffix LinkedIn appends (employment type, duration)."""
|
|
42
|
+
if not text:
|
|
43
|
+
return None
|
|
44
|
+
return clean(text.split("·")[0])
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _is_skills(frag: str) -> bool:
|
|
48
|
+
return frag.lower().startswith("skills:")
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _tech_from_skills(frag: str) -> list[str]:
|
|
52
|
+
"""``"Skills: Angular, Python, +21 skills"`` -> ``["Angular", "Python"]``."""
|
|
53
|
+
body = frag.split(":", 1)[-1]
|
|
54
|
+
out: list[str] = []
|
|
55
|
+
for part in body.split(","):
|
|
56
|
+
item = clean(part)
|
|
57
|
+
if not item or item.lower().endswith("skills") or item.startswith("+"):
|
|
58
|
+
continue
|
|
59
|
+
out.append(item)
|
|
60
|
+
return out
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def parse(html: str) -> list[Experience]:
|
|
64
|
+
out: list[Experience] = []
|
|
65
|
+
for block in entry_items(html):
|
|
66
|
+
roles = sub_entries(block)
|
|
67
|
+
if roles:
|
|
68
|
+
org = _strip_suffix(ordered_texts(block)[0]) if ordered_texts(block) else None
|
|
69
|
+
href = first_link(block, contains="/company/")
|
|
70
|
+
for role in roles:
|
|
71
|
+
exp = _build(role, org=org, company_href=href)
|
|
72
|
+
if exp:
|
|
73
|
+
out.append(exp)
|
|
74
|
+
else:
|
|
75
|
+
exp = _build(block, org=None, company_href=None)
|
|
76
|
+
if exp:
|
|
77
|
+
out.append(exp)
|
|
78
|
+
return out
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _build(
|
|
82
|
+
el: lhtml.HtmlElement,
|
|
83
|
+
*,
|
|
84
|
+
org: str | None,
|
|
85
|
+
company_href: str | None,
|
|
86
|
+
) -> Experience | None:
|
|
87
|
+
texts = ordered_texts(el)
|
|
88
|
+
if not texts:
|
|
89
|
+
return None
|
|
90
|
+
|
|
91
|
+
title = texts[0]
|
|
92
|
+
rest = texts[1:]
|
|
93
|
+
|
|
94
|
+
# For a single-role block the org is the 2nd fragment ("Org · Full-time").
|
|
95
|
+
if org is None and rest:
|
|
96
|
+
org = _strip_suffix(rest[0])
|
|
97
|
+
rest = rest[1:]
|
|
98
|
+
if not org:
|
|
99
|
+
return None
|
|
100
|
+
|
|
101
|
+
start = end = location = None
|
|
102
|
+
for frag in rest:
|
|
103
|
+
if _is_skills(frag):
|
|
104
|
+
continue
|
|
105
|
+
if start is None and looks_like_dates(frag):
|
|
106
|
+
start, end = split_timeline(frag)
|
|
107
|
+
elif location is None and not looks_like_dates(frag) and "\n" not in frag:
|
|
108
|
+
location = frag
|
|
109
|
+
|
|
110
|
+
bullets = split_bullets(description_text(el))
|
|
111
|
+
tech: list[str] = []
|
|
112
|
+
for frag in texts:
|
|
113
|
+
if _is_skills(frag):
|
|
114
|
+
tech = _tech_from_skills(frag)
|
|
115
|
+
break
|
|
116
|
+
|
|
117
|
+
href = company_href or first_link(el, contains="/company/")
|
|
118
|
+
links = [Link(label=org, url=href)] if href else []
|
|
119
|
+
|
|
120
|
+
return Experience(
|
|
121
|
+
id=slugify(org, title),
|
|
122
|
+
org=org,
|
|
123
|
+
location=location,
|
|
124
|
+
positions=[Position(title=title)],
|
|
125
|
+
timeline=Timeline(start=start, end=end) if start else None,
|
|
126
|
+
bullets=bullets,
|
|
127
|
+
tech=tech,
|
|
128
|
+
links=links,
|
|
129
|
+
)
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""Parse Honors & awards into the richer extra Honor model.
|
|
2
|
+
|
|
3
|
+
Same source section as awards.py (same fragment order), but kept as a standalone
|
|
4
|
+
extra file so the full description survives in ``summary``. tailor's Award
|
|
5
|
+
(awards.py) is the trimmed, mappable view.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from linkfetch.models import Honor
|
|
11
|
+
from linkfetch.parse.awards import extract
|
|
12
|
+
from linkfetch.parse.common import description_text, entry_items, ordered_texts
|
|
13
|
+
from linkfetch.text import slugify
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def parse(html: str) -> list[Honor]:
|
|
17
|
+
out: list[Honor] = []
|
|
18
|
+
for el in entry_items(html):
|
|
19
|
+
texts = ordered_texts(el)
|
|
20
|
+
if not texts:
|
|
21
|
+
continue
|
|
22
|
+
title, issuer, date = extract(texts)
|
|
23
|
+
out.append(
|
|
24
|
+
Honor(
|
|
25
|
+
id=slugify(title, issuer),
|
|
26
|
+
title=title,
|
|
27
|
+
issuer=issuer,
|
|
28
|
+
date=date,
|
|
29
|
+
summary=description_text(el),
|
|
30
|
+
)
|
|
31
|
+
)
|
|
32
|
+
return out
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""Parse the Languages details page.
|
|
2
|
+
|
|
3
|
+
Ordered fragments per entry: ``[language name, proficiency]`` — e.g.
|
|
4
|
+
``["English", "Full professional proficiency"]``.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from linkfetch.models import Language
|
|
10
|
+
from linkfetch.parse.common import entry_items, ordered_texts
|
|
11
|
+
from linkfetch.text import slugify
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def parse(html: str) -> list[Language]:
|
|
15
|
+
out: list[Language] = []
|
|
16
|
+
for el in entry_items(html):
|
|
17
|
+
texts = ordered_texts(el)
|
|
18
|
+
if not texts:
|
|
19
|
+
continue
|
|
20
|
+
name = texts[0]
|
|
21
|
+
proficiency = texts[1] if len(texts) > 1 else None
|
|
22
|
+
out.append(Language(id=slugify(name), name=name, proficiency=proficiency))
|
|
23
|
+
return out
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""Parse the Patents details page.
|
|
2
|
+
|
|
3
|
+
Ordered fragments per entry::
|
|
4
|
+
|
|
5
|
+
[title, "APPLICATION Filed · Filed Mar 26, 2025", <description>, …]
|
|
6
|
+
|
|
7
|
+
The status ("Application Filed" / "Granted") and date are read from the second
|
|
8
|
+
fragment; an explicit patent number (e.g. "US 1,234,567") is captured when
|
|
9
|
+
present. The description is the expandable text box.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import re
|
|
15
|
+
|
|
16
|
+
from linkfetch.models import Patent
|
|
17
|
+
from linkfetch.parse.common import description_text, entry_items, ordered_texts
|
|
18
|
+
from linkfetch.text import clean, slugify
|
|
19
|
+
|
|
20
|
+
_DATE = re.compile(r"\b([A-Z][a-z]{2,8}\.?\s+\d{1,2},?\s+(?:19|20)\d{2}|(?:19|20)\d{2})\b")
|
|
21
|
+
_NUMBER = re.compile(r"\b([A-Z]{2}\s?[\d,]{5,}|\d{1,3}[,/]\d{3}[,/\d]+)\b")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def parse(html: str) -> list[Patent]:
|
|
25
|
+
out: list[Patent] = []
|
|
26
|
+
for el in entry_items(html):
|
|
27
|
+
texts = ordered_texts(el)
|
|
28
|
+
if not texts:
|
|
29
|
+
continue
|
|
30
|
+
title = texts[0]
|
|
31
|
+
|
|
32
|
+
number = status = date = None
|
|
33
|
+
meta = texts[1] if len(texts) > 1 else ""
|
|
34
|
+
if meta:
|
|
35
|
+
head = meta.split("·")[0]
|
|
36
|
+
if head and not _DATE.fullmatch(head.strip()):
|
|
37
|
+
cleaned = clean(head)
|
|
38
|
+
# "APPLICATION Filed" -> "Application Filed" (preserve real casing
|
|
39
|
+
# of mixed-case tokens; only fix all-caps words).
|
|
40
|
+
status = " ".join(
|
|
41
|
+
w.capitalize() if w.isupper() else w for w in cleaned.split()
|
|
42
|
+
) if cleaned else None
|
|
43
|
+
m = _DATE.search(meta)
|
|
44
|
+
if m:
|
|
45
|
+
date = clean(m.group(1).lstrip("Filed ").strip())
|
|
46
|
+
m = _NUMBER.search(meta)
|
|
47
|
+
if m:
|
|
48
|
+
number = clean(m.group(1))
|
|
49
|
+
|
|
50
|
+
out.append(
|
|
51
|
+
Patent(
|
|
52
|
+
id=slugify(title),
|
|
53
|
+
title=title,
|
|
54
|
+
number=number,
|
|
55
|
+
status=status,
|
|
56
|
+
date=date,
|
|
57
|
+
summary=description_text(el),
|
|
58
|
+
)
|
|
59
|
+
)
|
|
60
|
+
return out
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Parse the Projects details page.
|
|
2
|
+
|
|
3
|
+
Ordered fragments per entry::
|
|
4
|
+
|
|
5
|
+
[name, "date range", <description>, <external link label>]
|
|
6
|
+
|
|
7
|
+
The description is an expandable box whose first line is usually ``Role: …`` —
|
|
8
|
+
we lift that into ``role`` and split the remainder into bullets. An associated
|
|
9
|
+
external URL (e.g. a GitHub repo) is captured as a link.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from linkfetch.models import Link, Project, Timeline
|
|
15
|
+
from linkfetch.parse.common import (
|
|
16
|
+
description_text,
|
|
17
|
+
entry_items,
|
|
18
|
+
looks_like_dates,
|
|
19
|
+
ordered_texts,
|
|
20
|
+
split_timeline,
|
|
21
|
+
)
|
|
22
|
+
from linkfetch.text import clean, slugify, split_bullets
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def parse(html: str) -> list[Project]:
|
|
26
|
+
out: list[Project] = []
|
|
27
|
+
for el in entry_items(html):
|
|
28
|
+
texts = ordered_texts(el)
|
|
29
|
+
if not texts:
|
|
30
|
+
continue
|
|
31
|
+
name = texts[0]
|
|
32
|
+
|
|
33
|
+
start = end = None
|
|
34
|
+
for frag in texts[1:]:
|
|
35
|
+
if looks_like_dates(frag):
|
|
36
|
+
start, end = split_timeline(frag)
|
|
37
|
+
break
|
|
38
|
+
|
|
39
|
+
role = None
|
|
40
|
+
bullets = split_bullets(description_text(el))
|
|
41
|
+
if bullets and bullets[0].lower().startswith("role:"):
|
|
42
|
+
role = clean(bullets[0].split(":", 1)[-1])
|
|
43
|
+
bullets = bullets[1:]
|
|
44
|
+
|
|
45
|
+
href = _external_link(el)
|
|
46
|
+
links = [Link(label=name, url=href)] if href else []
|
|
47
|
+
|
|
48
|
+
out.append(
|
|
49
|
+
Project(
|
|
50
|
+
id=slugify(name),
|
|
51
|
+
name=name,
|
|
52
|
+
role=role,
|
|
53
|
+
timeline=Timeline(start=start, end=end) if start else None,
|
|
54
|
+
bullets=bullets,
|
|
55
|
+
links=links,
|
|
56
|
+
)
|
|
57
|
+
)
|
|
58
|
+
return out
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _external_link(el) -> str | None:
|
|
62
|
+
"""First non-LinkedIn anchor (a real project/repo URL), if any."""
|
|
63
|
+
for a in el.xpath(".//a[@href]"):
|
|
64
|
+
href = a.get("href")
|
|
65
|
+
if href and href.startswith("http") and "linkedin.com" not in href:
|
|
66
|
+
return href.split("?")[0]
|
|
67
|
+
return None
|