linkfetch 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
linkfetch/models.py ADDED
@@ -0,0 +1,196 @@
1
+ """Models for scraped LinkedIn data.
2
+
3
+ Two families:
4
+
5
+ 1. **tailor-mapped** models — a *subset* of tailor's profile models using the
6
+ **same field names** (``id``, ``summary``, ``bullets``, ``tags``, ``org``,
7
+ ``positions``, ``timeline``, …). YAML emitted from these drops straight into
8
+ tailor's profile store. We intentionally omit fields tailor fills in by hand
9
+ (e.g. multiple position framings, emphasis) — the scrape is deterministic and
10
+ leaves enrichment to the user.
11
+
12
+ 2. **extra** models — small purpose-built models for LinkedIn sections tailor
13
+ has no home for (education, volunteering, recommendations, …).
14
+
15
+ All ``id`` values are stable slugs derived from the item's identity so re-runs
16
+ produce diffable YAML.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ from pydantic import BaseModel, Field
22
+
23
+ # ---------------------------------------------------------------------------
24
+ # shared
25
+ # ---------------------------------------------------------------------------
26
+
27
+
28
+ class Timeline(BaseModel):
29
+ """A start/end span. ``end`` omitted or 'present' means ongoing.
30
+
31
+ Mirrors tailor.profile.models.Timeline (field-compatible).
32
+ """
33
+
34
+ start: str | None = None
35
+ end: str | None = None
36
+ note: str | None = None
37
+
38
+
39
+ class Position(BaseModel):
40
+ """One title framing for a role. Mirrors tailor's Position.
41
+
42
+ The scrape produces exactly one framing (the real LinkedIn title); the user
43
+ adds alternative framings later in tailor.
44
+ """
45
+
46
+ title: str
47
+ seniority: str | None = None
48
+ emphasis: str | None = None
49
+ timeline: Timeline | None = None
50
+
51
+
52
+ class Link(BaseModel):
53
+ label: str
54
+ url: str
55
+
56
+
57
+ # ---------------------------------------------------------------------------
58
+ # tailor-mapped models (field names match tailor.profile.models)
59
+ # ---------------------------------------------------------------------------
60
+
61
+
62
+ class Basics(BaseModel):
63
+ name: str
64
+ label: str | None = None # headline
65
+ email: str | None = None
66
+ phone: str | None = None
67
+ location: str | None = None
68
+ summary: str | None = None # the "About" section
69
+ links: list[Link] = Field(default_factory=list)
70
+
71
+
72
+ class Experience(BaseModel):
73
+ id: str
74
+ org: str
75
+ location: str | None = None
76
+ positions: list[Position] = Field(default_factory=list)
77
+ timeline: Timeline | None = None
78
+ summary: str | None = None
79
+ bullets: list[str] = Field(default_factory=list)
80
+ tags: list[str] = Field(default_factory=list)
81
+ tech: list[str] = Field(default_factory=list)
82
+ links: list[Link] = Field(default_factory=list)
83
+
84
+
85
+ class Skill(BaseModel):
86
+ id: str
87
+ name: str
88
+ category: str | None = None
89
+ level: str | None = None
90
+ years: float | None = None
91
+ summary: str | None = None
92
+ bullets: list[str] = Field(default_factory=list)
93
+ tags: list[str] = Field(default_factory=list)
94
+
95
+
96
+ class Project(BaseModel):
97
+ id: str
98
+ name: str
99
+ role: str | None = None
100
+ timeline: Timeline | None = None
101
+ summary: str | None = None
102
+ bullets: list[str] = Field(default_factory=list)
103
+ tags: list[str] = Field(default_factory=list)
104
+ tech: list[str] = Field(default_factory=list)
105
+ links: list[Link] = Field(default_factory=list)
106
+
107
+
108
+ class Certification(BaseModel):
109
+ id: str
110
+ name: str
111
+ issuer: str | None = None
112
+ issued: str | None = None
113
+ expires: str | None = None
114
+ credential_id: str | None = None
115
+ summary: str | None = None
116
+ bullets: list[str] = Field(default_factory=list)
117
+ tags: list[str] = Field(default_factory=list)
118
+ links: list[Link] = Field(default_factory=list)
119
+
120
+
121
+ class Award(BaseModel):
122
+ id: str
123
+ name: str
124
+ issuer: str | None = None
125
+ date: str | None = None
126
+ summary: str | None = None
127
+ bullets: list[str] = Field(default_factory=list)
128
+ tags: list[str] = Field(default_factory=list)
129
+
130
+
131
+ # ---------------------------------------------------------------------------
132
+ # extra models (linkfetch-local schema; not in tailor)
133
+ # ---------------------------------------------------------------------------
134
+
135
+
136
+ class Education(BaseModel):
137
+ id: str
138
+ school: str
139
+ degree: str | None = None
140
+ field: str | None = None
141
+ timeline: Timeline | None = None
142
+ summary: str | None = None
143
+ bullets: list[str] = Field(default_factory=list)
144
+
145
+
146
+ class Volunteering(BaseModel):
147
+ id: str
148
+ org: str
149
+ role: str | None = None
150
+ cause: str | None = None
151
+ timeline: Timeline | None = None
152
+ summary: str | None = None
153
+
154
+
155
+ class Recommendation(BaseModel):
156
+ id: str
157
+ author: str
158
+ author_title: str | None = None
159
+ relationship: str | None = None
160
+ text: str | None = None
161
+
162
+
163
+ class Patent(BaseModel):
164
+ id: str
165
+ title: str
166
+ number: str | None = None
167
+ status: str | None = None
168
+ date: str | None = None
169
+ summary: str | None = None
170
+
171
+
172
+ class Course(BaseModel):
173
+ id: str
174
+ name: str
175
+ number: str | None = None
176
+ associated_with: str | None = None
177
+
178
+
179
+ class Language(BaseModel):
180
+ id: str
181
+ name: str
182
+ proficiency: str | None = None
183
+
184
+
185
+ class Honor(BaseModel):
186
+ """Honors & awards as a standalone extra section.
187
+
188
+ Note: tailor's Award is also populated from this section; this richer extra
189
+ file preserves everything (issuer, date, description) for the user.
190
+ """
191
+
192
+ id: str
193
+ title: str
194
+ issuer: str | None = None
195
+ date: str | None = None
196
+ summary: str | None = None
@@ -0,0 +1,43 @@
1
+ """Per-section HTML parsers.
2
+
3
+ Each section module exposes ``parse(html: str) -> list[Model]`` (or a single
4
+ model for basics). Parsers are **pure**: lxml in, pydantic out, no browser, no
5
+ network. They anchor on LinkedIn's *stable* structure — the ``<main>`` content
6
+ ``<section>``, its repeated entry blocks, and the ordered ``<p>``/``<span>`` text
7
+ fragments inside each (see :mod:`linkfetch.parse.common`) — rather than the
8
+ randomized hashed utility class names, so they survive cosmetic markup churn.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from linkfetch.parse import (
14
+ awards,
15
+ basics,
16
+ certifications,
17
+ courses,
18
+ education,
19
+ experience,
20
+ honors,
21
+ languages,
22
+ patents,
23
+ projects,
24
+ recommendations,
25
+ skills,
26
+ volunteering,
27
+ )
28
+
29
+ __all__ = [
30
+ "awards",
31
+ "basics",
32
+ "certifications",
33
+ "courses",
34
+ "education",
35
+ "experience",
36
+ "honors",
37
+ "languages",
38
+ "patents",
39
+ "projects",
40
+ "recommendations",
41
+ "skills",
42
+ "volunteering",
43
+ ]
@@ -0,0 +1,57 @@
1
+ """Parse Honors & awards into tailor's Award model.
2
+
3
+ Ordered fragments per entry::
4
+
5
+ [title, "Issued by <issuer> · <date>", "Associated with <entity>", <description>]
6
+
7
+ The issuer and date are pulled from the "Issued by … · …" fragment; the
8
+ expandable description (when present) becomes the bullets. See honors.py for the
9
+ richer extra-file view of the same section.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import re
15
+
16
+ from linkfetch.models import Award
17
+ from linkfetch.parse.common import description_text, entry_items, ordered_texts
18
+ from linkfetch.text import clean, slugify, split_bullets
19
+
20
+ _ISSUED_BY = re.compile(r"^issued by\s+", re.I)
21
+
22
+
23
+ def split_issuer_date(frag: str) -> tuple[str | None, str | None]:
24
+ """``"Issued by Roche · Apr 2021"`` -> ``("Roche", "Apr 2021")``."""
25
+ body = _ISSUED_BY.sub("", frag)
26
+ issuer, sep, date = body.partition("·")
27
+ return clean(issuer), (clean(date) if sep else None)
28
+
29
+
30
+ def extract(texts: list[str]) -> tuple[str, str | None, str | None]:
31
+ """Return ``(title, issuer, date)`` from an entry's fragments."""
32
+ title = texts[0]
33
+ issuer = date = None
34
+ for frag in texts[1:]:
35
+ if frag.lower().startswith("issued by"):
36
+ issuer, date = split_issuer_date(frag)
37
+ break
38
+ return title, issuer, date
39
+
40
+
41
+ def parse(html: str) -> list[Award]:
42
+ out: list[Award] = []
43
+ for el in entry_items(html):
44
+ texts = ordered_texts(el)
45
+ if not texts:
46
+ continue
47
+ title, issuer, date = extract(texts)
48
+ out.append(
49
+ Award(
50
+ id=slugify(title, issuer),
51
+ name=title,
52
+ issuer=issuer,
53
+ date=date,
54
+ bullets=split_bullets(description_text(el)),
55
+ )
56
+ )
57
+ return out
@@ -0,0 +1,89 @@
1
+ """Parse the top profile card for basics (name, headline, location, about).
2
+
3
+ Unlike the section details pages, basics come from the main profile page. The
4
+ captured DOM has no ``<h1>`` or ``#about`` anchor, so we anchor on stable signals
5
+ instead:
6
+
7
+ * **name** — the page ``<title>`` ("Jane Doe | LinkedIn"), with the first
8
+ matching visible fragment as a fallback;
9
+ * **headline** — the visible fragment immediately following the name;
10
+ * **location** — the first short "City, Region, Country" fragment;
11
+ * **about** — the single expandable text box on the page (the About summary).
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import re
17
+
18
+ from lxml import html as lhtml
19
+
20
+ from linkfetch.models import Basics
21
+ from linkfetch.parse.common import description_text, ordered_texts
22
+ from linkfetch.text import clean
23
+
24
+ # A "City, Region, Country" style location: 2-3 comma-separated short parts.
25
+ _LOCATION = re.compile(r"^[^,\n]{2,40}(?:,[^,\n]{2,40}){1,2}$")
26
+
27
+
28
+ def parse(html: str) -> Basics:
29
+ if not html or not html.strip():
30
+ return Basics(name="")
31
+ root = lhtml.fromstring(html)
32
+
33
+ name = _name(root)
34
+ fragments = ordered_texts(root)
35
+
36
+ headline = None
37
+ if name:
38
+ for i, frag in enumerate(fragments):
39
+ if frag == name and i + 1 < len(fragments):
40
+ headline = fragments[i + 1]
41
+ break
42
+
43
+ # Location: a "City, Region, Country" fragment in the top card, anchored by
44
+ # the "Contact info"/"connections" line that LinkedIn renders right after it.
45
+ location = None
46
+ for i, frag in enumerate(fragments):
47
+ if frag in (name, headline) or not _LOCATION.match(frag):
48
+ continue
49
+ if _looks_like_company(frag) or _is_location_junk(frag):
50
+ continue
51
+ nearby = " ".join(fragments[i + 1 : i + 4]).lower()
52
+ if "contact info" in nearby or "connection" in nearby:
53
+ location = frag
54
+ break
55
+
56
+ return Basics(
57
+ name=name or "",
58
+ label=headline,
59
+ location=location,
60
+ summary=description_text(root, exclude_linked=True),
61
+ )
62
+
63
+
64
+ def _name(root: lhtml.HtmlElement) -> str | None:
65
+ titles = root.xpath("//title/text()")
66
+ if titles:
67
+ title = clean(titles[0]) or ""
68
+ # "Jane Doe | LinkedIn" / "… | (5) LinkedIn"
69
+ name = re.split(r"\s*[|–—]\s*", title)[0]
70
+ name = clean(name)
71
+ if name and name.lower() != "linkedin":
72
+ return name
73
+ h1 = root.xpath("//h1")
74
+ for h in h1:
75
+ txt = clean(h.text_content())
76
+ if txt:
77
+ return txt
78
+ return None
79
+
80
+
81
+ def _looks_like_company(frag: str) -> bool:
82
+ low = frag.lower()
83
+ return "·" in frag or "|" in frag or "linkedin" in low
84
+
85
+
86
+ def _is_location_junk(frag: str) -> bool:
87
+ """Reject media-player / dialog strings that happen to contain commas."""
88
+ low = frag.lower()
89
+ return any(w in low for w in ("live", "seek", "modal", "dialog", "opacity"))
@@ -0,0 +1,70 @@
1
+ """Parse the Licenses & certifications details page.
2
+
3
+ Ordered fragments per entry::
4
+
5
+ [name, issuer, "Issued <date>" (· "Expires <date>"), "Show credential",
6
+ "Credential ID …", "Skills: …"]
7
+
8
+ The credential URL is the anchor whose visible text is "Show credential".
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import re
14
+
15
+ from linkfetch.models import Certification, Link
16
+ from linkfetch.parse.common import entry_items, ordered_texts
17
+ from linkfetch.text import clean, slugify
18
+
19
+
20
+ def parse(html: str) -> list[Certification]:
21
+ out: list[Certification] = []
22
+ for el in entry_items(html):
23
+ texts = ordered_texts(el)
24
+ if not texts:
25
+ continue
26
+ name = texts[0]
27
+ issuer = texts[1] if len(texts) > 1 else None
28
+
29
+ issued = expires = credential_id = None
30
+ for frag in texts[2:]:
31
+ low = frag.lower()
32
+ if "issued" in low or "expires" in low or "expir" in low:
33
+ issued, expires = _split_dates(frag)
34
+ elif "credential id" in low:
35
+ credential_id = clean(frag.split(":", 1)[-1])
36
+
37
+ href = _credential_link(el)
38
+ links = [Link(label="Credential", url=href)] if href else []
39
+ out.append(
40
+ Certification(
41
+ id=slugify(name, issuer),
42
+ name=name,
43
+ issuer=issuer,
44
+ issued=issued,
45
+ expires=expires,
46
+ credential_id=credential_id,
47
+ links=links,
48
+ )
49
+ )
50
+ return out
51
+
52
+
53
+ def _credential_link(el) -> str | None:
54
+ """Anchor whose text reads 'Show credential' (the external credential URL)."""
55
+ for a in el.xpath(".//a[@href]"):
56
+ text = clean(a.text_content()) or ""
57
+ if "credential" in text.lower():
58
+ return a.get("href").split("?")[0]
59
+ return None
60
+
61
+
62
+ def _split_dates(text: str) -> tuple[str | None, str | None]:
63
+ issued = expires = None
64
+ m = re.search(r"issued\s+(.+?)(?:·|$)", text, re.I)
65
+ if m:
66
+ issued = clean(m.group(1))
67
+ m = re.search(r"expir\w*\s+(.+?)(?:·|$)", text, re.I)
68
+ if m:
69
+ expires = clean(m.group(1))
70
+ return issued, expires