linkfetch 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- linkfetch/__init__.py +3 -0
- linkfetch/capture/__init__.py +1 -0
- linkfetch/capture/browser.py +112 -0
- linkfetch/capture/expand.py +61 -0
- linkfetch/cli.py +274 -0
- linkfetch/config.py +105 -0
- linkfetch/emit.py +57 -0
- linkfetch/models.py +196 -0
- linkfetch/parse/__init__.py +43 -0
- linkfetch/parse/awards.py +57 -0
- linkfetch/parse/basics.py +89 -0
- linkfetch/parse/certifications.py +70 -0
- linkfetch/parse/common.py +251 -0
- linkfetch/parse/courses.py +38 -0
- linkfetch/parse/education.py +51 -0
- linkfetch/parse/experience.py +129 -0
- linkfetch/parse/honors.py +32 -0
- linkfetch/parse/languages.py +23 -0
- linkfetch/parse/patents.py +60 -0
- linkfetch/parse/projects.py +67 -0
- linkfetch/parse/recommendations.py +64 -0
- linkfetch/parse/skills.py +30 -0
- linkfetch/parse/volunteering.py +47 -0
- linkfetch/sections.py +84 -0
- linkfetch/text.py +78 -0
- linkfetch-0.1.0.dist-info/METADATA +174 -0
- linkfetch-0.1.0.dist-info/RECORD +30 -0
- linkfetch-0.1.0.dist-info/WHEEL +4 -0
- linkfetch-0.1.0.dist-info/entry_points.txt +2 -0
- linkfetch-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""Parse the Recommendations (received) details page.
|
|
2
|
+
|
|
3
|
+
Ordered fragments per entry::
|
|
4
|
+
|
|
5
|
+
[author, "· 1st", author_title, "<date>, <relationship>",
|
|
6
|
+
"All LinkedIn members", "On", <body>]
|
|
7
|
+
|
|
8
|
+
The connection-degree ("· 1st"), visibility ("All LinkedIn members", "On") and
|
|
9
|
+
expand ("… more") fragments are skipped. The relationship is the clause after
|
|
10
|
+
the date; the body is the expandable text box.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from linkfetch.models import Recommendation
|
|
16
|
+
from linkfetch.parse.common import description_text, entry_items, ordered_texts
|
|
17
|
+
from linkfetch.text import clean, slugify
|
|
18
|
+
|
|
19
|
+
_SKIP = {"all linkedin members", "on", "off", "… more", "…", "more"}
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def parse(html: str) -> list[Recommendation]:
|
|
23
|
+
out: list[Recommendation] = []
|
|
24
|
+
for el in entry_items(html):
|
|
25
|
+
texts = ordered_texts(el)
|
|
26
|
+
if not texts:
|
|
27
|
+
continue
|
|
28
|
+
author = texts[0]
|
|
29
|
+
|
|
30
|
+
author_title = relationship = None
|
|
31
|
+
for frag in texts[1:]:
|
|
32
|
+
low = frag.lower()
|
|
33
|
+
if frag.startswith("·") or low in _SKIP:
|
|
34
|
+
continue
|
|
35
|
+
if relationship is None and _is_relationship(frag):
|
|
36
|
+
relationship = _relationship_clause(frag)
|
|
37
|
+
elif author_title is None:
|
|
38
|
+
author_title = frag
|
|
39
|
+
|
|
40
|
+
out.append(
|
|
41
|
+
Recommendation(
|
|
42
|
+
id=slugify(author),
|
|
43
|
+
author=author,
|
|
44
|
+
author_title=author_title,
|
|
45
|
+
relationship=relationship,
|
|
46
|
+
text=description_text(el),
|
|
47
|
+
)
|
|
48
|
+
)
|
|
49
|
+
return out
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _is_relationship(frag: str) -> bool:
|
|
53
|
+
low = frag.lower()
|
|
54
|
+
return ("managed" in low or "worked with" in low or "was senior" in low
|
|
55
|
+
or "studied" in low or "taught" in low or "reported" in low)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _relationship_clause(frag: str) -> str | None:
|
|
59
|
+
"""``"April 4, 2020, John was senior to Jane"`` -> the clause after the date."""
|
|
60
|
+
# Date prefix ends at the second comma ("Month D, YYYY,"); keep the rest.
|
|
61
|
+
parts = frag.split(",")
|
|
62
|
+
if len(parts) >= 3:
|
|
63
|
+
return clean(",".join(parts[2:]))
|
|
64
|
+
return clean(frag)
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""Parse the Skills details page.
|
|
2
|
+
|
|
3
|
+
Each entry's first visible fragment is the skill name; a second fragment gives
|
|
4
|
+
the context where it was used ("Senior Software Development Engineer at Roche"),
|
|
5
|
+
which we keep as the skill ``summary``. Duplicates are dropped; categorization is
|
|
6
|
+
left to the user in tailor.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from linkfetch.models import Skill
|
|
12
|
+
from linkfetch.parse.common import entry_items, ordered_texts
|
|
13
|
+
from linkfetch.text import slugify
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def parse(html: str) -> list[Skill]:
|
|
17
|
+
out: list[Skill] = []
|
|
18
|
+
seen: set[str] = set()
|
|
19
|
+
for el in entry_items(html):
|
|
20
|
+
texts = ordered_texts(el)
|
|
21
|
+
if not texts:
|
|
22
|
+
continue
|
|
23
|
+
name = texts[0]
|
|
24
|
+
key = name.lower()
|
|
25
|
+
if key in seen:
|
|
26
|
+
continue
|
|
27
|
+
seen.add(key)
|
|
28
|
+
summary = texts[1] if len(texts) > 1 else None
|
|
29
|
+
out.append(Skill(id=slugify(name), name=name, summary=summary))
|
|
30
|
+
return out
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""Parse the Volunteering details page.
|
|
2
|
+
|
|
3
|
+
Ordered fragments per entry::
|
|
4
|
+
|
|
5
|
+
[role, org, "date range · duration", cause, <description>]
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from linkfetch.models import Timeline, Volunteering
|
|
11
|
+
from linkfetch.parse.common import (
|
|
12
|
+
description_text,
|
|
13
|
+
entry_items,
|
|
14
|
+
looks_like_dates,
|
|
15
|
+
ordered_texts,
|
|
16
|
+
split_timeline,
|
|
17
|
+
)
|
|
18
|
+
from linkfetch.text import slugify
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def parse(html: str) -> list[Volunteering]:
|
|
22
|
+
out: list[Volunteering] = []
|
|
23
|
+
for el in entry_items(html):
|
|
24
|
+
texts = ordered_texts(el)
|
|
25
|
+
if not texts:
|
|
26
|
+
continue
|
|
27
|
+
role = texts[0]
|
|
28
|
+
org = texts[1] if len(texts) > 1 else None
|
|
29
|
+
|
|
30
|
+
start = end = cause = None
|
|
31
|
+
for frag in texts[2:]:
|
|
32
|
+
if start is None and looks_like_dates(frag):
|
|
33
|
+
start, end = split_timeline(frag)
|
|
34
|
+
elif cause is None and not looks_like_dates(frag):
|
|
35
|
+
cause = frag
|
|
36
|
+
|
|
37
|
+
out.append(
|
|
38
|
+
Volunteering(
|
|
39
|
+
id=slugify(org, role),
|
|
40
|
+
org=org or role,
|
|
41
|
+
role=role,
|
|
42
|
+
cause=cause,
|
|
43
|
+
timeline=Timeline(start=start, end=end) if start else None,
|
|
44
|
+
summary=description_text(el),
|
|
45
|
+
)
|
|
46
|
+
)
|
|
47
|
+
return out
|
linkfetch/sections.py
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""Section registry — the single source of truth for what linkfetch handles.
|
|
2
|
+
|
|
3
|
+
Each Section ties together:
|
|
4
|
+
- ``slug``: the linkfetch identifier (used on the CLI and as the capture filename)
|
|
5
|
+
- ``detail_path``: LinkedIn details-page path template ('{vanity}' filled in);
|
|
6
|
+
None for ``basics`` which comes from the main profile page
|
|
7
|
+
- ``parser``: the pure ``parse(html)`` callable
|
|
8
|
+
- ``out_file`` / ``top_key``: where parsed YAML is written and under what key
|
|
9
|
+
- ``mapped``: True if the output file is one of the profile files THRIVE and
|
|
10
|
+
tailor read (``data/profile/<name>.yaml``); ``honors`` is an extra,
|
|
11
|
+
richer copy of awards and is not
|
|
12
|
+
|
|
13
|
+
Note ``honors`` and ``awards`` share the same source page ('honors') but produce
|
|
14
|
+
two different output files (the trimmed tailor Award + the richer extra Honor).
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from dataclasses import dataclass
|
|
20
|
+
from typing import Callable
|
|
21
|
+
|
|
22
|
+
from linkfetch.parse import (
|
|
23
|
+
awards,
|
|
24
|
+
basics,
|
|
25
|
+
certifications,
|
|
26
|
+
courses,
|
|
27
|
+
education,
|
|
28
|
+
experience,
|
|
29
|
+
honors,
|
|
30
|
+
languages,
|
|
31
|
+
patents,
|
|
32
|
+
projects,
|
|
33
|
+
recommendations,
|
|
34
|
+
skills,
|
|
35
|
+
volunteering,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass(frozen=True)
|
|
40
|
+
class Section:
|
|
41
|
+
slug: str
|
|
42
|
+
detail_path: str | None # '{vanity}' placeholder; None => main profile page
|
|
43
|
+
parser: Callable[[str], object]
|
|
44
|
+
out_file: str
|
|
45
|
+
top_key: str
|
|
46
|
+
mapped: bool
|
|
47
|
+
source: str | None = None # capture filename to reuse; defaults to slug
|
|
48
|
+
|
|
49
|
+
@property
|
|
50
|
+
def capture_name(self) -> str:
|
|
51
|
+
return self.source or self.slug
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
# Ordered for a sensible run sequence (identity first).
|
|
55
|
+
SECTIONS: list[Section] = [
|
|
56
|
+
Section("basics", None, basics.parse, "basics.yaml", "basics", True),
|
|
57
|
+
Section("experience", "/in/{vanity}/details/experience/", experience.parse, "experiences.yaml", "experiences", True),
|
|
58
|
+
Section("education", "/in/{vanity}/details/education/", education.parse, "education.yaml", "education", True),
|
|
59
|
+
Section("skills", "/in/{vanity}/details/skills/", skills.parse, "skills.yaml", "skills", True),
|
|
60
|
+
Section("projects", "/in/{vanity}/details/projects/", projects.parse, "projects.yaml", "projects", True),
|
|
61
|
+
Section("certifications", "/in/{vanity}/details/certifications/", certifications.parse, "certifications.yaml", "certifications", True),
|
|
62
|
+
Section("awards", "/in/{vanity}/details/honors/", awards.parse, "awards.yaml", "awards", True, source="honors"),
|
|
63
|
+
Section("honors", "/in/{vanity}/details/honors/", honors.parse, "honors.yaml", "honors", False),
|
|
64
|
+
Section("volunteering", "/in/{vanity}/details/volunteering-experiences/", volunteering.parse, "volunteering.yaml", "volunteering", True),
|
|
65
|
+
Section("recommendations", "/in/{vanity}/details/recommendations/", recommendations.parse, "recommendations.yaml", "recommendations", True),
|
|
66
|
+
Section("patents", "/in/{vanity}/details/patents/", patents.parse, "patents.yaml", "patents", True),
|
|
67
|
+
Section("courses", "/in/{vanity}/details/courses/", courses.parse, "courses.yaml", "courses", True),
|
|
68
|
+
Section("languages", "/in/{vanity}/details/languages/", languages.parse, "languages.yaml", "languages", True),
|
|
69
|
+
]
|
|
70
|
+
|
|
71
|
+
BY_SLUG: dict[str, Section] = {s.slug: s for s in SECTIONS}
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def resolve(slugs: list[str] | None) -> list[Section]:
|
|
75
|
+
"""Return the requested sections (all if ``slugs`` is None/empty)."""
|
|
76
|
+
if not slugs:
|
|
77
|
+
return list(SECTIONS)
|
|
78
|
+
unknown = [s for s in slugs if s not in BY_SLUG]
|
|
79
|
+
if unknown:
|
|
80
|
+
raise ValueError(
|
|
81
|
+
f"Unknown section(s): {', '.join(unknown)}. "
|
|
82
|
+
f"Valid: {', '.join(BY_SLUG)}"
|
|
83
|
+
)
|
|
84
|
+
return [BY_SLUG[s] for s in slugs]
|
linkfetch/text.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Small text helpers shared across parsers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
import unicodedata
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def slugify(*parts: str | None) -> str:
|
|
10
|
+
"""Build a stable, filesystem/YAML-friendly slug from one or more parts.
|
|
11
|
+
|
|
12
|
+
Used for model ``id`` fields so re-running the scrape yields diffable YAML.
|
|
13
|
+
Empty/None parts are dropped. Falls back to ``item`` if nothing usable.
|
|
14
|
+
"""
|
|
15
|
+
joined = " ".join(p for p in parts if p)
|
|
16
|
+
norm = unicodedata.normalize("NFKD", joined)
|
|
17
|
+
norm = norm.encode("ascii", "ignore").decode("ascii")
|
|
18
|
+
norm = norm.lower()
|
|
19
|
+
norm = re.sub(r"[^a-z0-9]+", "-", norm).strip("-")
|
|
20
|
+
return norm or "item"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def clean(text: str | None) -> str | None:
|
|
24
|
+
"""Collapse whitespace; return None for empty results."""
|
|
25
|
+
if text is None:
|
|
26
|
+
return None
|
|
27
|
+
collapsed = re.sub(r"[ \t ]+", " ", text)
|
|
28
|
+
lines = [ln.strip() for ln in collapsed.splitlines()]
|
|
29
|
+
result = "\n".join(ln for ln in lines if ln).strip()
|
|
30
|
+
return result or None
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
_BULLET_GLYPH = re.compile(r"[•‣◦]")
|
|
34
|
+
_LEADING_PUNCT = re.compile(r"^[\-\*·]+\s*")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def split_bullets(text: str | None) -> list[str]:
|
|
38
|
+
"""Split a description blob into bullets.
|
|
39
|
+
|
|
40
|
+
A LinkedIn description mixes two things, sometimes in the same blob:
|
|
41
|
+
|
|
42
|
+
* Plain ``<br/>``-separated lines with no bullet glyph — often short
|
|
43
|
+
headers an author writes before their real bullet list starts (``Role:
|
|
44
|
+
…``, ``What it is:``, ``Highlights:``), or the entire description when
|
|
45
|
+
it never uses glyphs at all. Each line here is its own bullet.
|
|
46
|
+
* ``•``/``‣``/``◦``-prefixed bullets, which are **not** reliably
|
|
47
|
+
one-per-line: a long bullet is often wrapped across two or more
|
|
48
|
+
``<br/>``-separated lines by the author (the wrap is not a second
|
|
49
|
+
bullet), and two short bullets can sit on the very same line with no
|
|
50
|
+
break between them at all (the glyph, not the line break, is the
|
|
51
|
+
boundary).
|
|
52
|
+
|
|
53
|
+
So: text up to the first glyph is split one bullet per line, same as a
|
|
54
|
+
glyph-free description; everything from the first glyph onward is split
|
|
55
|
+
on the glyph instead, with any line break inside one glyph bullet
|
|
56
|
+
collapsed back into a space rather than read as a second bullet.
|
|
57
|
+
|
|
58
|
+
Deterministic: no AI. Returns [] for empty input.
|
|
59
|
+
"""
|
|
60
|
+
if not text:
|
|
61
|
+
return []
|
|
62
|
+
|
|
63
|
+
head, *glyph_bullets = _BULLET_GLYPH.split(text)
|
|
64
|
+
out: list[str] = _plain_lines(head)
|
|
65
|
+
for part in glyph_bullets:
|
|
66
|
+
line = " ".join(part.split())
|
|
67
|
+
if line:
|
|
68
|
+
out.append(line)
|
|
69
|
+
return out
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _plain_lines(text: str) -> list[str]:
|
|
73
|
+
out: list[str] = []
|
|
74
|
+
for raw in text.splitlines():
|
|
75
|
+
line = _LEADING_PUNCT.sub("", raw.strip()).strip()
|
|
76
|
+
if line:
|
|
77
|
+
out.append(line)
|
|
78
|
+
return out
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: linkfetch
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Capture your own LinkedIn profile on your own computer and turn it into structured profile files
|
|
5
|
+
Project-URL: Homepage, https://github.com/Prosperis/linkfetch
|
|
6
|
+
Project-URL: Issues, https://github.com/Prosperis/linkfetch/issues
|
|
7
|
+
Author: Prosperis
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: cv,export,linkedin,playwright,profile,resume
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: End Users/Desktop
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Topic :: Office/Business
|
|
21
|
+
Requires-Python: >=3.10
|
|
22
|
+
Requires-Dist: lxml>=5.0
|
|
23
|
+
Requires-Dist: playwright>=1.44
|
|
24
|
+
Requires-Dist: pydantic>=2.6
|
|
25
|
+
Requires-Dist: pyyaml>=6.0
|
|
26
|
+
Requires-Dist: rich>=13.7
|
|
27
|
+
Requires-Dist: tomli>=2.0; python_version < '3.11'
|
|
28
|
+
Requires-Dist: typer>=0.12
|
|
29
|
+
Provides-Extra: browser
|
|
30
|
+
Provides-Extra: dev
|
|
31
|
+
Requires-Dist: pytest-mock>=3.12; extra == 'dev'
|
|
32
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
33
|
+
Description-Content-Type: text/markdown
|
|
34
|
+
|
|
35
|
+
# linkfetch
|
|
36
|
+
|
|
37
|
+
Capture **your own** LinkedIn profile on **your own computer** and turn it into
|
|
38
|
+
structured profile files — every role, description, skill, project,
|
|
39
|
+
certification and more — ready to import into
|
|
40
|
+
[THRIVE](https://prosperis-thrive.vercel.app) or any tool that reads YAML.
|
|
41
|
+
|
|
42
|
+
LinkedIn's official data export leaves out details the profile page shows.
|
|
43
|
+
linkfetch reads the rendered pages instead, in a real browser window that you
|
|
44
|
+
log into yourself.
|
|
45
|
+
|
|
46
|
+
## What it does and does not do
|
|
47
|
+
|
|
48
|
+
- **Runs only on your machine.** linkfetch is not a website or a service. It
|
|
49
|
+
sends your data nowhere; the files it writes stay in a folder on your
|
|
50
|
+
computer until you choose to upload them.
|
|
51
|
+
- **Reads only your own profile**, in a browser you log into. It never sees
|
|
52
|
+
your password: you type it into LinkedIn's own login page.
|
|
53
|
+
- **Deterministic, no AI.** Every field comes straight from the page.
|
|
54
|
+
- **Open source** (MIT), so you can read exactly what it does.
|
|
55
|
+
|
|
56
|
+
## Install
|
|
57
|
+
|
|
58
|
+
You need Python 3.10 or newer and [uv](https://docs.astral.sh/uv/) (or
|
|
59
|
+
[pipx](https://pipx.pypa.io/)).
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
uv tool install linkfetch # or: pipx install linkfetch
|
|
63
|
+
linkfetch doctor # shows where data is kept and whether Playwright and a login exist
|
|
64
|
+
playwright install chromium # one-time download of the browser linkfetch drives
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
## Use
|
|
68
|
+
|
|
69
|
+
```bash
|
|
70
|
+
linkfetch login # a browser window opens: log in to LinkedIn, then close it
|
|
71
|
+
linkfetch run --vanity your-name # capture your profile and build the files
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
`your-name` is the part after `/in/` in your profile address:
|
|
75
|
+
`linkedin.com/in/your-name`.
|
|
76
|
+
|
|
77
|
+
`run` visits each section of your profile (experience, education, skills, …),
|
|
78
|
+
expands every "see more", saves the pages, and turns them into one YAML file
|
|
79
|
+
per section plus **`linkfetch-profile.zip`**. `linkfetch` prints where they
|
|
80
|
+
are.
|
|
81
|
+
|
|
82
|
+
### Import into THRIVE
|
|
83
|
+
|
|
84
|
+
In THRIVE, open **Tailor → Profile → Import** and choose
|
|
85
|
+
`linkfetch-profile.zip`. Review each section before saving.
|
|
86
|
+
|
|
87
|
+
## Where your data is kept
|
|
88
|
+
|
|
89
|
+
Installed with `uv tool` or `pipx`, linkfetch keeps everything in your user
|
|
90
|
+
data folder:
|
|
91
|
+
|
|
92
|
+
| OS | Folder |
|
|
93
|
+
|---|---|
|
|
94
|
+
| Windows | `%LOCALAPPDATA%\linkfetch` |
|
|
95
|
+
| macOS | `~/Library/Application Support/linkfetch` |
|
|
96
|
+
| Linux | `$XDG_DATA_HOME/linkfetch` (usually `~/.local/share/linkfetch`) |
|
|
97
|
+
|
|
98
|
+
Set `LINKFETCH_HOME` to use a different folder. Inside it:
|
|
99
|
+
|
|
100
|
+
- `data/browser/` — the browser profile that keeps you logged in to LinkedIn.
|
|
101
|
+
Treat it like a password; delete it to log out.
|
|
102
|
+
- `data/captures/` — the saved profile pages.
|
|
103
|
+
- `data/output/` — the YAML files and `linkfetch-profile.zip`.
|
|
104
|
+
|
|
105
|
+
Delete the folder to remove everything linkfetch stored.
|
|
106
|
+
|
|
107
|
+
## Commands
|
|
108
|
+
|
|
109
|
+
| Command | What it does |
|
|
110
|
+
|---|---|
|
|
111
|
+
| `linkfetch doctor` | Show the data folders and check Playwright and your login |
|
|
112
|
+
| `linkfetch login` | Open a browser window to log in to LinkedIn once |
|
|
113
|
+
| `linkfetch capture --vanity <you>` | Save your profile pages (needs the browser) |
|
|
114
|
+
| `linkfetch parse` | Turn saved pages into YAML and the ZIP (no browser; re-runnable) |
|
|
115
|
+
| `linkfetch run --vanity <you>` | `capture` then `parse` |
|
|
116
|
+
|
|
117
|
+
Use `--section experience --section skills` to limit a run to some sections,
|
|
118
|
+
and `parse --no-zip` to skip the ZIP.
|
|
119
|
+
|
|
120
|
+
## Output
|
|
121
|
+
|
|
122
|
+
One file per section, each a list under a top-level key:
|
|
123
|
+
`basics.yaml`, `experiences.yaml`, `education.yaml`, `skills.yaml`,
|
|
124
|
+
`projects.yaml`, `certifications.yaml`, `awards.yaml`, `languages.yaml`,
|
|
125
|
+
`courses.yaml`, `patents.yaml`, `recommendations.yaml`, `volunteering.yaml`,
|
|
126
|
+
plus `honors.yaml` (a richer copy of awards, not included in the ZIP).
|
|
127
|
+
Experiences keep their full description as a summary plus bullets, and roles
|
|
128
|
+
at the same company stay grouped.
|
|
129
|
+
|
|
130
|
+
## How it works
|
|
131
|
+
|
|
132
|
+
1. **Capture** (needs the browser, runs rarely): a persistent Chromium profile
|
|
133
|
+
you logged into visits each section's details page
|
|
134
|
+
(`/in/<you>/details/experience/`, …), expands collapsed content, and saves
|
|
135
|
+
the rendered HTML.
|
|
136
|
+
2. **Parse** (pure, offline): reads the saved HTML with `lxml` and writes YAML.
|
|
137
|
+
If LinkedIn changes its pages, a parser fix plus `linkfetch parse` is enough
|
|
138
|
+
— no need to capture again.
|
|
139
|
+
|
|
140
|
+
## Development
|
|
141
|
+
|
|
142
|
+
```bash
|
|
143
|
+
git clone https://github.com/Prosperis/linkfetch
|
|
144
|
+
cd linkfetch
|
|
145
|
+
uv venv && uv pip install -e ".[dev]"
|
|
146
|
+
playwright install chromium
|
|
147
|
+
uv run pytest
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
A checkout keeps its data in `data/` next to `config.toml`, which also holds
|
|
151
|
+
browser settings (headless mode, timeouts, scroll pacing). Test fixtures are
|
|
152
|
+
anonymized excerpts of LinkedIn's page structure.
|
|
153
|
+
|
|
154
|
+
## Releasing
|
|
155
|
+
|
|
156
|
+
Bump `version` in `pyproject.toml`, commit, then push a matching tag:
|
|
157
|
+
|
|
158
|
+
```bash
|
|
159
|
+
git tag v0.1.1 && git push origin v0.1.1
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
The Release workflow tests, builds and publishes to PyPI through trusted
|
|
163
|
+
publishing (no stored token), then creates the GitHub release.
|
|
164
|
+
|
|
165
|
+
## Legal note
|
|
166
|
+
|
|
167
|
+
LinkedIn's User Agreement restricts automated access. linkfetch is meant for
|
|
168
|
+
**personal use on your own profile**: it uses a browser you logged into, moves
|
|
169
|
+
at a human pace, reads only your own data and stores nothing remotely. Use it
|
|
170
|
+
on your own account and at your own discretion.
|
|
171
|
+
|
|
172
|
+
## License
|
|
173
|
+
|
|
174
|
+
[MIT](LICENSE) © Prosperis
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
linkfetch/__init__.py,sha256=9gf-qPA4WRfBjcbdAvhTSEmhTASb8VYYM3KwwNI6YTE,107
|
|
2
|
+
linkfetch/cli.py,sha256=TUJ7EIC83Jm_bkW1aZmFKExPzo-qsO-6qI7IYX1Lkjk,9797
|
|
3
|
+
linkfetch/config.py,sha256=lrpZmwYFcpLirc4qcAjkw6Q-Q4YBudZjScQ4VRsyA5Y,3436
|
|
4
|
+
linkfetch/emit.py,sha256=PoKfOHgeqM-wDC0q2JrUOnew1dNtlkYD9HWceCnSNFk,1856
|
|
5
|
+
linkfetch/models.py,sha256=QWOEFik9HP0NARwo26lTYCS4cPEsY4-1fqdwR-H0Qg4,5355
|
|
6
|
+
linkfetch/sections.py,sha256=DdL9wwB7CM3weftkgco5UFCJdawy6J6Rd7d1S67pjjM,3601
|
|
7
|
+
linkfetch/text.py,sha256=dyzenT4CqDLbY9KBECMzV-0NrtkO00OyIa-ZW-c5j38,2770
|
|
8
|
+
linkfetch/capture/__init__.py,sha256=rZMoqGlSznk4wbaOPn7iYgIz8q5IXaoUVa5bWDWLpRk,79
|
|
9
|
+
linkfetch/capture/browser.py,sha256=NW_-XWgzRnRvtG8iyam5bSuRmc3yV9QC4dQMeMaUwMQ,3865
|
|
10
|
+
linkfetch/capture/expand.py,sha256=38fsTiKztiaNxRWfafhHHGRVVu7r_Pqik9jzw7Rt4bQ,2571
|
|
11
|
+
linkfetch/parse/__init__.py,sha256=UNL-bykW3Lb2Gq25zgOZX64Fw7IsJlR4X-zewaUWuU4,1005
|
|
12
|
+
linkfetch/parse/awards.py,sha256=bj59lqNrh2rMzyxbVvBDOHNXRLyW5fcXg8FnV-WImyM,1779
|
|
13
|
+
linkfetch/parse/basics.py,sha256=qSSeqtTNM21yHGlYJ7BK_drNI3c5RMpKvuipmGU1Rn0,2924
|
|
14
|
+
linkfetch/parse/certifications.py,sha256=fg05AEjmgNBfRLj9keu7JML4HrfDVKUNQA0yK-RZ9vU,2157
|
|
15
|
+
linkfetch/parse/common.py,sha256=jJ-FXZHLsevJV8sNVWP5acu8PqA94wAN9O3JC44SNh8,9766
|
|
16
|
+
linkfetch/parse/courses.py,sha256=Uci0wnUssXB5Sl1fGs7HxZ3PrBOzBKBnj8-FYEaPi8A,1059
|
|
17
|
+
linkfetch/parse/education.py,sha256=ur7RkyNl1YYP9vFl8vQht7IHQnzkOi3qdIPNpm8FxBw,1557
|
|
18
|
+
linkfetch/parse/experience.py,sha256=4cVExE5D3DOVGP249q4JGGRXOwpKT_QrDuNbNE1eh2w,3843
|
|
19
|
+
linkfetch/parse/honors.py,sha256=F_PnXumIMbjggSUHz0_FMZQlmOyWWzrXcoGK9TeSr_Y,968
|
|
20
|
+
linkfetch/parse/languages.py,sha256=R6j1KZLbHKGtBZ5HMvt_1vpP9DAB32bCZlQI2Bs7n3A,697
|
|
21
|
+
linkfetch/parse/patents.py,sha256=bNME4AilwMX6GVqaIpzjTLqsNg7KRIFE-914yZq0_Nk,1974
|
|
22
|
+
linkfetch/parse/projects.py,sha256=UTO7vLsy4QtH1N2Kv1wxwWVlwg1aJJzbQR-TWVRTU-I,1952
|
|
23
|
+
linkfetch/parse/recommendations.py,sha256=pWsGELNIzmnMqQUQW-Wzzddipafi8MV7tY4jkSCDpd8,2114
|
|
24
|
+
linkfetch/parse/skills.py,sha256=Z5YtfCgmq80L17D60ss-Uq1NC-LoUHcbdHWHUxn7Zkc,940
|
|
25
|
+
linkfetch/parse/volunteering.py,sha256=43BbC_mPttys-zCFirf-0jHhQaLBLoHzV4djhgdopPE,1269
|
|
26
|
+
linkfetch-0.1.0.dist-info/METADATA,sha256=eJLJy5MMimoMxmPMFS5NCsu8ZvRqiR3yzhFtLM16s7A,6441
|
|
27
|
+
linkfetch-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
28
|
+
linkfetch-0.1.0.dist-info/entry_points.txt,sha256=qDUzht2Wb4BB4Fenz7HghIUn8gIW6twu4DZviJA-5EY,48
|
|
29
|
+
linkfetch-0.1.0.dist-info/licenses/LICENSE,sha256=LykqJy8VKC9CS8l0vtqS6LZhkQ0fWdUI302VkOARMs4,1066
|
|
30
|
+
linkfetch-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Prosperis
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|