linkfetch 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,251 @@
1
+ """Shared helpers for section parsers.
2
+
3
+ These target LinkedIn's **current** rendered DOM (captured via Playwright), which:
4
+
5
+ * uses **obfuscated, per-build hashed class names** (``_759d3eea`` …) — so we never
6
+ key off classes;
7
+ * puts visible text in ``<p>`` and ``<span>`` elements, frequently emitting the
8
+ **same string twice in a row** (a visible copy + a screen-reader copy);
9
+ * lays out each section under ``<main>`` inside a single content ``<section>``
10
+ (the section title is its first visible fragment), with the footer, ad widgets
11
+ and global nav living *outside* that section;
12
+ * renders each entry as a repeated sibling block (a ``<div>`` or ``<li>``) inside
13
+ that section — **not** the old single ``<ul>`` of ``<li>``s;
14
+ * nests multiple roles at one company in an inner ``<ul>`` of role ``<li>``s
15
+ inside the company's block;
16
+ * wraps long descriptions in a ``<span data-testid="expandable-text-box">`` whose
17
+ line breaks are real ``<br/>`` tags (which ``text_content()`` would otherwise
18
+ flatten).
19
+
20
+ We anchor on this **structure** (``main`` → content ``section`` → repeating entry
21
+ blocks → ordered text fragments), which is far more stable than the hashed
22
+ classes.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import re
28
+
29
+ from lxml import html as lhtml
30
+
31
+ from linkfetch.text import clean
32
+
33
+ # Chrome / nav / footer / ad boilerplate that leaks into the captured page.
34
+ # Matched as a lowercase *substring* against an entry's head text so an entry
35
+ # that is purely chrome is dropped.
36
+ _CHROME = (
37
+ "home", "my network", "jobs", "messaging", "notifications",
38
+ "skip to search", "skip to main content", "skip to active conversation",
39
+ "back to search results", "why am i seeing this ad", "manage your ad",
40
+ "hide or report this ad", "report this ad", "ad options",
41
+ "linkedin corporation", "questions?", "visit our help center",
42
+ "manage your account and privacy", "go to your settings",
43
+ "recommendation transparency", "learn more about recommended content",
44
+ "select language", "profile language", "accessibility", "talent solutions",
45
+ "community guidelines", "marketing solutions", "privacy & terms",
46
+ "ad choices", "sales solutions", "small business", "safety center",
47
+ )
48
+
49
+ # Section-title fragments that appear as the first child block of the content
50
+ # section and must not be treated as an entry.
51
+ _SECTION_TITLES = (
52
+ "experience", "education", "skills", "projects",
53
+ "licenses & certifications", "licenses and certifications", "certifications",
54
+ "honors & awards", "honors and awards", "volunteering",
55
+ "recommendations", "patents", "courses", "languages",
56
+ )
57
+
58
+
59
+ def _root(page_html: str) -> lhtml.HtmlElement | None:
60
+ if not page_html or not page_html.strip():
61
+ return None
62
+ return lhtml.fromstring(page_html)
63
+
64
+
65
+ def _content_section(page_html: str) -> lhtml.HtmlElement | None:
66
+ """Return the ``<main>``'s primary content ``<section>`` (the entry scope).
67
+
68
+ The captured page always carries one ``<section aria-label="Primary
69
+ content">`` — the real entries — alongside a fixed-size "People also
70
+ viewed" sidebar section that renders identically on every details page.
71
+ We take the labelled section outright when it is there. Falling back to
72
+ "most visible text fragments" picks the sidebar instead whenever a
73
+ person's real section (projects, certifications, honors, a short
74
+ education list, …) happens to be smaller than that sidebar's fixed
75
+ fragment count — not a rare case, since plenty of profiles are sparse in
76
+ exactly those sections. The fragment-count heuristic remains only as a
77
+ fallback for a page that, for whatever reason, has no labelled section.
78
+ """
79
+ root = _root(page_html)
80
+ if root is None:
81
+ return None
82
+ mains = root.xpath("//main")
83
+ scope = mains[0] if mains else root
84
+ sections = scope.xpath(".//section")
85
+ if not sections:
86
+ return scope
87
+ labelled = [s for s in sections if s.get("aria-label") == "Primary content"]
88
+ if labelled:
89
+ return labelled[0]
90
+ return max(sections, key=lambda s: len(ordered_texts(s)))
91
+
92
+
93
+ def _is_chrome(text: str | None) -> bool:
94
+ if not text:
95
+ return True
96
+ low = text.lower()
97
+ return any(hint in low for hint in _CHROME)
98
+
99
+
100
+ def entry_items(page_html: str) -> list[lhtml.HtmlElement]:
101
+ """Return the top-level entry blocks of a details page.
102
+
103
+ Finds the content ``<section>`` under ``<main>``, then the element whose
104
+ direct ``<div>``/``<li>``/``<section>`` children are the repeated entry
105
+ blocks. An entry child is one carrying at least two distinct visible text
106
+ fragments; we pick the container with the **most** such children (the
107
+ deepest one on ties — so we land on the real entry list rather than a
108
+ wrapper). Section-title and chrome blocks are then filtered out.
109
+ """
110
+ scope = _content_section(page_html)
111
+ if scope is None:
112
+ return []
113
+
114
+ best: lhtml.HtmlElement | None = None
115
+ best_n = 0
116
+ best_depth = -1
117
+ for el in scope.iter():
118
+ entries = [
119
+ c
120
+ for c in el
121
+ if c.tag in ("div", "li", "section") and len(ordered_texts(c)) >= 2
122
+ ]
123
+ n = len(entries)
124
+ if n == 0 or n < best_n:
125
+ continue
126
+ depth = sum(1 for _ in el.iterancestors())
127
+ if n > best_n or (n == best_n and depth > best_depth):
128
+ best_n = n
129
+ best = el
130
+ best_depth = depth
131
+
132
+ if best is None:
133
+ return []
134
+
135
+ items: list[lhtml.HtmlElement] = []
136
+ for kid in best:
137
+ if kid.tag not in ("div", "li", "section"):
138
+ continue
139
+ texts = ordered_texts(kid)
140
+ head = texts[0] if texts else None
141
+ if head is None or _is_chrome(head):
142
+ continue
143
+ if head.lower().strip() in _SECTION_TITLES:
144
+ continue
145
+ items.append(kid)
146
+ return items
147
+
148
+
149
+ def _head_text(el: lhtml.HtmlElement) -> str | None:
150
+ for frag in ordered_texts(el):
151
+ return frag
152
+ return None
153
+
154
+
155
+ def ordered_texts(el: lhtml.HtmlElement) -> list[str]:
156
+ """Ordered, de-duplicated visible text fragments inside ``el``.
157
+
158
+ Reads ``<p>`` and ``<span>`` text in document order, dropping consecutive
159
+ duplicates (LinkedIn's visible + screen-reader copies) and empties.
160
+ """
161
+ out: list[str] = []
162
+ for node in el.iter():
163
+ if node.tag not in ("p", "span"):
164
+ continue
165
+ txt = clean(node.text_content())
166
+ if not txt:
167
+ continue
168
+ if out and out[-1] == txt:
169
+ continue
170
+ out.append(txt)
171
+ return out
172
+
173
+
174
+ def sub_entries(li: lhtml.HtmlElement) -> list[lhtml.HtmlElement]:
175
+ """Return nested sub-entry ``<li>``s (e.g. multiple roles at one company)."""
176
+ nested = li.xpath(".//ul/li")
177
+ return [n for n in nested if _head_text(n)]
178
+
179
+
180
+ def first_link(el: lhtml.HtmlElement, *, contains: str | None = None) -> str | None:
181
+ """First anchor ``href`` inside ``el``; optionally requiring a substring."""
182
+ for a in el.xpath(".//a[@href]"):
183
+ href = a.get("href")
184
+ if href and (contains is None or contains in href):
185
+ return href.split("?")[0]
186
+ return None
187
+
188
+
189
+ def description_text(el: lhtml.HtmlElement, *, exclude_linked: bool = False) -> str | None:
190
+ """Long-form description inside ``el``, preserving ``<br/>`` as newlines.
191
+
192
+ LinkedIn renders the description in a ``<span data-testid="expandable-text-box">``
193
+ using real ``<br/>`` tags for line breaks (which ``text_content()`` flattens).
194
+ We pick the longest such box and convert ``<br/>`` to newlines so the
195
+ description splits cleanly into bullets. Returns ``None`` when the entry has
196
+ no such box (i.e. genuinely has no description).
197
+
198
+ ``exclude_linked`` drops any box that sits inside an ``<a>``. On a details
199
+ page every box in scope is the one genuine description, so this is off by
200
+ default; the main profile page also renders every recent Activity post's
201
+ preview through this same component, each wrapped in a link to the post,
202
+ which "pick the longest box" would otherwise treat as a candidate for the
203
+ person's About text.
204
+ """
205
+ boxes = el.xpath('.//*[@data-testid="expandable-text-box"]')
206
+ if exclude_linked:
207
+ boxes = [b for b in boxes if not b.xpath("ancestor::a")]
208
+ best = ""
209
+ for box in boxes:
210
+ text = _text_with_breaks(box)
211
+ # Drop the trailing "… more" expand affordance LinkedIn appends.
212
+ text = re.sub(r"\s*…\s*more\s*$", "", text)
213
+ if len(text) > len(best):
214
+ best = text
215
+ return clean(best) or None
216
+
217
+
218
+ def _text_with_breaks(el: lhtml.HtmlElement) -> str:
219
+ """``text_content()`` but with ``<br/>`` rendered as a newline."""
220
+ parts: list[str] = []
221
+ if el.text:
222
+ parts.append(el.text)
223
+ for child in el:
224
+ if child.tag == "br":
225
+ parts.append("\n")
226
+ else:
227
+ parts.append(_text_with_breaks(child))
228
+ if child.tail:
229
+ parts.append(child.tail)
230
+ return "".join(parts)
231
+
232
+
233
+ def split_timeline(date_range: str | None) -> tuple[str | None, str | None]:
234
+ """Split ``"Mar 2021 - Present · 3 yrs"`` into ``(start, end)``.
235
+
236
+ The duration suffix after ``·`` is dropped. A single date with no separator
237
+ is treated as the start.
238
+ """
239
+ if not date_range:
240
+ return None, None
241
+ text = date_range.split("·")[0].strip()
242
+ for sep in (" – ", " - ", " — ", "–", "—", " to "):
243
+ if sep in text:
244
+ start, _, end = text.partition(sep)
245
+ return clean(start), clean(end)
246
+ return clean(text), None
247
+
248
+
249
+ def looks_like_dates(text: str) -> bool:
250
+ """True if the fragment carries a 4-digit year or 'Present'."""
251
+ return bool(re.search(r"\b(19|20)\d{2}\b", text)) or "present" in text.lower()
@@ -0,0 +1,38 @@
1
+ """Parse the Courses details page.
2
+
3
+ Ordered fragments per entry::
4
+
5
+ [name, "<course number>", "Associated with <school>"]
6
+
7
+ e.g. ``["Algorithm Design and Analysis", "CSE 100", "Associated with …"]``.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from linkfetch.models import Course
13
+ from linkfetch.parse.common import entry_items, ordered_texts
14
+ from linkfetch.text import clean, slugify
15
+
16
+
17
+ def parse(html: str) -> list[Course]:
18
+ out: list[Course] = []
19
+ for el in entry_items(html):
20
+ texts = ordered_texts(el)
21
+ if not texts:
22
+ continue
23
+ name = texts[0]
24
+ number = associated = None
25
+ for frag in texts[1:]:
26
+ if frag.lower().startswith("associated with"):
27
+ associated = clean(frag.split("with", 1)[-1])
28
+ elif number is None:
29
+ number = frag
30
+ out.append(
31
+ Course(
32
+ id=slugify(name, number),
33
+ name=name,
34
+ number=number,
35
+ associated_with=associated,
36
+ )
37
+ )
38
+ return out
@@ -0,0 +1,51 @@
1
+ """Parse the Education details page.
2
+
3
+ Ordered fragments per entry::
4
+
5
+ [school, "Degree, Field of study", "2016 – 2021", <activities/description>]
6
+
7
+ The degree/field fragment is split on the first comma; the timeline fragment is
8
+ a bare year range. The long-form "Activities and societies: …" / description
9
+ text (an expandable box) becomes the bullets.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from linkfetch.models import Education, Timeline
15
+ from linkfetch.parse.common import (
16
+ description_text,
17
+ entry_items,
18
+ looks_like_dates,
19
+ ordered_texts,
20
+ split_timeline,
21
+ )
22
+ from linkfetch.text import clean, slugify, split_bullets
23
+
24
+
25
+ def parse(html: str) -> list[Education]:
26
+ out: list[Education] = []
27
+ for el in entry_items(html):
28
+ texts = ordered_texts(el)
29
+ if not texts:
30
+ continue
31
+ school = texts[0]
32
+
33
+ degree = field = None
34
+ start = end = None
35
+ for frag in texts[1:]:
36
+ if start is None and looks_like_dates(frag):
37
+ start, end = split_timeline(frag)
38
+ elif degree is None and not frag.lower().startswith("activities"):
39
+ degree, _, field = (clean(p) for p in frag.partition(","))
40
+
41
+ out.append(
42
+ Education(
43
+ id=slugify(school, degree),
44
+ school=school,
45
+ degree=degree,
46
+ field=field,
47
+ timeline=Timeline(start=start, end=end) if start else None,
48
+ bullets=split_bullets(description_text(el)),
49
+ )
50
+ )
51
+ return out
@@ -0,0 +1,129 @@
1
+ """Parse the Experience details page.
2
+
3
+ Each entry block is a *company*. There are two shapes:
4
+
5
+ * **Single role** — the block's ordered fragments are::
6
+
7
+ [title, "Org · Employment-type", "date range · duration", location,
8
+ <description>, "Skills: …"]
9
+
10
+ * **Multiple roles** — the block heads with the company, then nests one role per
11
+ inner ``<li>``::
12
+
13
+ company head: ["Roche", "Full-time · 5 yrs 4 mos", "Hybrid", …]
14
+ each role li: [title, "date range · duration", location, <description>,
15
+ "Skills: …"]
16
+
17
+ We emit **one** :class:`Experience` per role; for multi-role companies the
18
+ ``org`` is the company head. The ``· duration`` / employment-type suffixes are
19
+ stripped from ``org``; the description becomes ``bullets``; the trailing
20
+ ``Skills: …`` fragment is parsed into ``tech``.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ from lxml import html as lhtml
26
+
27
+ from linkfetch.models import Experience, Link, Position, Timeline
28
+ from linkfetch.parse.common import (
29
+ description_text,
30
+ entry_items,
31
+ first_link,
32
+ looks_like_dates,
33
+ ordered_texts,
34
+ split_timeline,
35
+ sub_entries,
36
+ )
37
+ from linkfetch.text import clean, slugify, split_bullets
38
+
39
+
40
+ def _strip_suffix(text: str | None) -> str | None:
41
+ """Drop the ``· …`` suffix LinkedIn appends (employment type, duration)."""
42
+ if not text:
43
+ return None
44
+ return clean(text.split("·")[0])
45
+
46
+
47
+ def _is_skills(frag: str) -> bool:
48
+ return frag.lower().startswith("skills:")
49
+
50
+
51
+ def _tech_from_skills(frag: str) -> list[str]:
52
+ """``"Skills: Angular, Python, +21 skills"`` -> ``["Angular", "Python"]``."""
53
+ body = frag.split(":", 1)[-1]
54
+ out: list[str] = []
55
+ for part in body.split(","):
56
+ item = clean(part)
57
+ if not item or item.lower().endswith("skills") or item.startswith("+"):
58
+ continue
59
+ out.append(item)
60
+ return out
61
+
62
+
63
+ def parse(html: str) -> list[Experience]:
64
+ out: list[Experience] = []
65
+ for block in entry_items(html):
66
+ roles = sub_entries(block)
67
+ if roles:
68
+ org = _strip_suffix(ordered_texts(block)[0]) if ordered_texts(block) else None
69
+ href = first_link(block, contains="/company/")
70
+ for role in roles:
71
+ exp = _build(role, org=org, company_href=href)
72
+ if exp:
73
+ out.append(exp)
74
+ else:
75
+ exp = _build(block, org=None, company_href=None)
76
+ if exp:
77
+ out.append(exp)
78
+ return out
79
+
80
+
81
+ def _build(
82
+ el: lhtml.HtmlElement,
83
+ *,
84
+ org: str | None,
85
+ company_href: str | None,
86
+ ) -> Experience | None:
87
+ texts = ordered_texts(el)
88
+ if not texts:
89
+ return None
90
+
91
+ title = texts[0]
92
+ rest = texts[1:]
93
+
94
+ # For a single-role block the org is the 2nd fragment ("Org · Full-time").
95
+ if org is None and rest:
96
+ org = _strip_suffix(rest[0])
97
+ rest = rest[1:]
98
+ if not org:
99
+ return None
100
+
101
+ start = end = location = None
102
+ for frag in rest:
103
+ if _is_skills(frag):
104
+ continue
105
+ if start is None and looks_like_dates(frag):
106
+ start, end = split_timeline(frag)
107
+ elif location is None and not looks_like_dates(frag) and "\n" not in frag:
108
+ location = frag
109
+
110
+ bullets = split_bullets(description_text(el))
111
+ tech: list[str] = []
112
+ for frag in texts:
113
+ if _is_skills(frag):
114
+ tech = _tech_from_skills(frag)
115
+ break
116
+
117
+ href = company_href or first_link(el, contains="/company/")
118
+ links = [Link(label=org, url=href)] if href else []
119
+
120
+ return Experience(
121
+ id=slugify(org, title),
122
+ org=org,
123
+ location=location,
124
+ positions=[Position(title=title)],
125
+ timeline=Timeline(start=start, end=end) if start else None,
126
+ bullets=bullets,
127
+ tech=tech,
128
+ links=links,
129
+ )
@@ -0,0 +1,32 @@
1
+ """Parse Honors & awards into the richer extra Honor model.
2
+
3
+ Same source section as awards.py (same fragment order), but kept as a standalone
4
+ extra file so the full description survives in ``summary``. tailor's Award
5
+ (awards.py) is the trimmed, mappable view.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from linkfetch.models import Honor
11
+ from linkfetch.parse.awards import extract
12
+ from linkfetch.parse.common import description_text, entry_items, ordered_texts
13
+ from linkfetch.text import slugify
14
+
15
+
16
+ def parse(html: str) -> list[Honor]:
17
+ out: list[Honor] = []
18
+ for el in entry_items(html):
19
+ texts = ordered_texts(el)
20
+ if not texts:
21
+ continue
22
+ title, issuer, date = extract(texts)
23
+ out.append(
24
+ Honor(
25
+ id=slugify(title, issuer),
26
+ title=title,
27
+ issuer=issuer,
28
+ date=date,
29
+ summary=description_text(el),
30
+ )
31
+ )
32
+ return out
@@ -0,0 +1,23 @@
1
+ """Parse the Languages details page.
2
+
3
+ Ordered fragments per entry: ``[language name, proficiency]`` — e.g.
4
+ ``["English", "Full professional proficiency"]``.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from linkfetch.models import Language
10
+ from linkfetch.parse.common import entry_items, ordered_texts
11
+ from linkfetch.text import slugify
12
+
13
+
14
+ def parse(html: str) -> list[Language]:
15
+ out: list[Language] = []
16
+ for el in entry_items(html):
17
+ texts = ordered_texts(el)
18
+ if not texts:
19
+ continue
20
+ name = texts[0]
21
+ proficiency = texts[1] if len(texts) > 1 else None
22
+ out.append(Language(id=slugify(name), name=name, proficiency=proficiency))
23
+ return out
@@ -0,0 +1,60 @@
1
+ """Parse the Patents details page.
2
+
3
+ Ordered fragments per entry::
4
+
5
+ [title, "APPLICATION Filed · Filed Mar 26, 2025", <description>, …]
6
+
7
+ The status ("Application Filed" / "Granted") and date are read from the second
8
+ fragment; an explicit patent number (e.g. "US 1,234,567") is captured when
9
+ present. The description is the expandable text box.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import re
15
+
16
+ from linkfetch.models import Patent
17
+ from linkfetch.parse.common import description_text, entry_items, ordered_texts
18
+ from linkfetch.text import clean, slugify
19
+
20
+ _DATE = re.compile(r"\b([A-Z][a-z]{2,8}\.?\s+\d{1,2},?\s+(?:19|20)\d{2}|(?:19|20)\d{2})\b")
21
+ _NUMBER = re.compile(r"\b([A-Z]{2}\s?[\d,]{5,}|\d{1,3}[,/]\d{3}[,/\d]+)\b")
22
+
23
+
24
+ def parse(html: str) -> list[Patent]:
25
+ out: list[Patent] = []
26
+ for el in entry_items(html):
27
+ texts = ordered_texts(el)
28
+ if not texts:
29
+ continue
30
+ title = texts[0]
31
+
32
+ number = status = date = None
33
+ meta = texts[1] if len(texts) > 1 else ""
34
+ if meta:
35
+ head = meta.split("·")[0]
36
+ if head and not _DATE.fullmatch(head.strip()):
37
+ cleaned = clean(head)
38
+ # "APPLICATION Filed" -> "Application Filed" (preserve real casing
39
+ # of mixed-case tokens; only fix all-caps words).
40
+ status = " ".join(
41
+ w.capitalize() if w.isupper() else w for w in cleaned.split()
42
+ ) if cleaned else None
43
+ m = _DATE.search(meta)
44
+ if m:
45
+ date = clean(m.group(1).lstrip("Filed ").strip())
46
+ m = _NUMBER.search(meta)
47
+ if m:
48
+ number = clean(m.group(1))
49
+
50
+ out.append(
51
+ Patent(
52
+ id=slugify(title),
53
+ title=title,
54
+ number=number,
55
+ status=status,
56
+ date=date,
57
+ summary=description_text(el),
58
+ )
59
+ )
60
+ return out
@@ -0,0 +1,67 @@
1
+ """Parse the Projects details page.
2
+
3
+ Ordered fragments per entry::
4
+
5
+ [name, "date range", <description>, <external link label>]
6
+
7
+ The description is an expandable box whose first line is usually ``Role: …`` —
8
+ we lift that into ``role`` and split the remainder into bullets. An associated
9
+ external URL (e.g. a GitHub repo) is captured as a link.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from linkfetch.models import Link, Project, Timeline
15
+ from linkfetch.parse.common import (
16
+ description_text,
17
+ entry_items,
18
+ looks_like_dates,
19
+ ordered_texts,
20
+ split_timeline,
21
+ )
22
+ from linkfetch.text import clean, slugify, split_bullets
23
+
24
+
25
+ def parse(html: str) -> list[Project]:
26
+ out: list[Project] = []
27
+ for el in entry_items(html):
28
+ texts = ordered_texts(el)
29
+ if not texts:
30
+ continue
31
+ name = texts[0]
32
+
33
+ start = end = None
34
+ for frag in texts[1:]:
35
+ if looks_like_dates(frag):
36
+ start, end = split_timeline(frag)
37
+ break
38
+
39
+ role = None
40
+ bullets = split_bullets(description_text(el))
41
+ if bullets and bullets[0].lower().startswith("role:"):
42
+ role = clean(bullets[0].split(":", 1)[-1])
43
+ bullets = bullets[1:]
44
+
45
+ href = _external_link(el)
46
+ links = [Link(label=name, url=href)] if href else []
47
+
48
+ out.append(
49
+ Project(
50
+ id=slugify(name),
51
+ name=name,
52
+ role=role,
53
+ timeline=Timeline(start=start, end=end) if start else None,
54
+ bullets=bullets,
55
+ links=links,
56
+ )
57
+ )
58
+ return out
59
+
60
+
61
+ def _external_link(el) -> str | None:
62
+ """First non-LinkedIn anchor (a real project/repo URL), if any."""
63
+ for a in el.xpath(".//a[@href]"):
64
+ href = a.get("href")
65
+ if href and href.startswith("http") and "linkedin.com" not in href:
66
+ return href.split("?")[0]
67
+ return None