linkfetch 0.1.0__tar.gz → 0.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. {linkfetch-0.1.0 → linkfetch-0.1.2}/PKG-INFO +13 -5
  2. {linkfetch-0.1.0 → linkfetch-0.1.2}/README.md +12 -4
  3. {linkfetch-0.1.0 → linkfetch-0.1.2}/pyproject.toml +1 -1
  4. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/capture/browser.py +61 -18
  5. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/capture/expand.py +30 -4
  6. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/cli.py +14 -1
  7. linkfetch-0.1.2/tests/test_capture_retry.py +68 -0
  8. {linkfetch-0.1.0 → linkfetch-0.1.2}/uv.lock +1 -1
  9. {linkfetch-0.1.0 → linkfetch-0.1.2}/.github/workflows/release.yml +0 -0
  10. {linkfetch-0.1.0 → linkfetch-0.1.2}/.github/workflows/test.yml +0 -0
  11. {linkfetch-0.1.0 → linkfetch-0.1.2}/.gitignore +0 -0
  12. {linkfetch-0.1.0 → linkfetch-0.1.2}/LICENSE +0 -0
  13. {linkfetch-0.1.0 → linkfetch-0.1.2}/config.toml +0 -0
  14. {linkfetch-0.1.0 → linkfetch-0.1.2}/docs/superpowers/specs/2026-06-15-linkfetch-design.md +0 -0
  15. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/__init__.py +0 -0
  16. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/capture/__init__.py +0 -0
  17. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/config.py +0 -0
  18. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/emit.py +0 -0
  19. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/models.py +0 -0
  20. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/parse/__init__.py +0 -0
  21. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/parse/awards.py +0 -0
  22. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/parse/basics.py +0 -0
  23. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/parse/certifications.py +0 -0
  24. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/parse/common.py +0 -0
  25. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/parse/courses.py +0 -0
  26. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/parse/education.py +0 -0
  27. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/parse/experience.py +0 -0
  28. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/parse/honors.py +0 -0
  29. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/parse/languages.py +0 -0
  30. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/parse/patents.py +0 -0
  31. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/parse/projects.py +0 -0
  32. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/parse/recommendations.py +0 -0
  33. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/parse/skills.py +0 -0
  34. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/parse/volunteering.py +0 -0
  35. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/sections.py +0 -0
  36. {linkfetch-0.1.0 → linkfetch-0.1.2}/src/linkfetch/text.py +0 -0
  37. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/conftest.py +0 -0
  38. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/fixtures/basics.html +0 -0
  39. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/fixtures/certifications.html +0 -0
  40. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/fixtures/education.html +0 -0
  41. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/fixtures/experience.html +0 -0
  42. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/fixtures/honors.html +0 -0
  43. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/fixtures/languages.html +0 -0
  44. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/fixtures/patents.html +0 -0
  45. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/fixtures/projects.html +0 -0
  46. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/fixtures/recommendations.html +0 -0
  47. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/fixtures/skills.html +0 -0
  48. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/fixtures/volunteering.html +0 -0
  49. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/test_config.py +0 -0
  50. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/test_emit.py +0 -0
  51. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/test_experience.py +0 -0
  52. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/test_release.py +0 -0
  53. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/test_sections.py +0 -0
  54. {linkfetch-0.1.0 → linkfetch-0.1.2}/tests/test_text.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: linkfetch
3
- Version: 0.1.0
3
+ Version: 0.1.2
4
4
  Summary: Capture your own LinkedIn profile on your own computer and turn it into structured profile files
5
5
  Project-URL: Homepage, https://github.com/Prosperis/linkfetch
6
6
  Project-URL: Issues, https://github.com/Prosperis/linkfetch/issues
@@ -34,6 +34,11 @@ Description-Content-Type: text/markdown
34
34
 
35
35
  # linkfetch
36
36
 
37
+ [![PyPI](https://img.shields.io/pypi/v/linkfetch)](https://pypi.org/project/linkfetch/)
38
+ [![Python](https://img.shields.io/pypi/pyversions/linkfetch)](https://pypi.org/project/linkfetch/)
39
+ [![Tests](https://github.com/Prosperis/linkfetch/actions/workflows/test.yml/badge.svg)](https://github.com/Prosperis/linkfetch/actions/workflows/test.yml)
40
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue)](LICENSE)
41
+
37
42
  Capture **your own** LinkedIn profile on **your own computer** and turn it into
38
43
  structured profile files — every role, description, skill, project,
39
44
  certification and more — ready to import into
@@ -61,7 +66,7 @@ You need Python 3.10 or newer and [uv](https://docs.astral.sh/uv/) (or
61
66
  ```bash
62
67
  uv tool install linkfetch # or: pipx install linkfetch
63
68
  linkfetch doctor # shows where data is kept and whether Playwright and a login exist
64
- playwright install chromium # one-time download of the browser linkfetch drives
69
+ linkfetch install-browser # one-time download of the browser linkfetch drives (~150 MB)
65
70
  ```
66
71
 
67
72
  ## Use
@@ -108,6 +113,7 @@ Delete the folder to remove everything linkfetch stored.
108
113
 
109
114
  | Command | What it does |
110
115
  |---|---|
116
+ | `linkfetch install-browser` | Download the browser linkfetch drives (one time, and again if an upgrade asks) |
111
117
  | `linkfetch doctor` | Show the data folders and check Playwright and your login |
112
118
  | `linkfetch login` | Open a browser window to log in to LinkedIn once |
113
119
  | `linkfetch capture --vanity <you>` | Save your profile pages (needs the browser) |
@@ -124,8 +130,10 @@ One file per section, each a list under a top-level key:
124
130
  `projects.yaml`, `certifications.yaml`, `awards.yaml`, `languages.yaml`,
125
131
  `courses.yaml`, `patents.yaml`, `recommendations.yaml`, `volunteering.yaml`,
126
132
  plus `honors.yaml` (a richer copy of awards, not included in the ZIP).
127
- Experiences keep their full description as a summary plus bullets, and roles
128
- at the same company stay grouped.
133
+ Experiences keep their full description as bullets. Several roles at one
134
+ company become separate experiences — each with its own title, dates and
135
+ bullets — all under the company's name, because a role's dates and
136
+ achievements are what tailoring selects from.
129
137
 
130
138
  ## How it works
131
139
 
@@ -143,7 +151,7 @@ at the same company stay grouped.
143
151
  git clone https://github.com/Prosperis/linkfetch
144
152
  cd linkfetch
145
153
  uv venv && uv pip install -e ".[dev]"
146
- playwright install chromium
154
+ uv run linkfetch install-browser
147
155
  uv run pytest
148
156
  ```
149
157
 
@@ -1,5 +1,10 @@
1
1
  # linkfetch
2
2
 
3
+ [![PyPI](https://img.shields.io/pypi/v/linkfetch)](https://pypi.org/project/linkfetch/)
4
+ [![Python](https://img.shields.io/pypi/pyversions/linkfetch)](https://pypi.org/project/linkfetch/)
5
+ [![Tests](https://github.com/Prosperis/linkfetch/actions/workflows/test.yml/badge.svg)](https://github.com/Prosperis/linkfetch/actions/workflows/test.yml)
6
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue)](LICENSE)
7
+
3
8
  Capture **your own** LinkedIn profile on **your own computer** and turn it into
4
9
  structured profile files — every role, description, skill, project,
5
10
  certification and more — ready to import into
@@ -27,7 +32,7 @@ You need Python 3.10 or newer and [uv](https://docs.astral.sh/uv/) (or
27
32
  ```bash
28
33
  uv tool install linkfetch # or: pipx install linkfetch
29
34
  linkfetch doctor # shows where data is kept and whether Playwright and a login exist
30
- playwright install chromium # one-time download of the browser linkfetch drives
35
+ linkfetch install-browser # one-time download of the browser linkfetch drives (~150 MB)
31
36
  ```
32
37
 
33
38
  ## Use
@@ -74,6 +79,7 @@ Delete the folder to remove everything linkfetch stored.
74
79
 
75
80
  | Command | What it does |
76
81
  |---|---|
82
+ | `linkfetch install-browser` | Download the browser linkfetch drives (one time, and again if an upgrade asks) |
77
83
  | `linkfetch doctor` | Show the data folders and check Playwright and your login |
78
84
  | `linkfetch login` | Open a browser window to log in to LinkedIn once |
79
85
  | `linkfetch capture --vanity <you>` | Save your profile pages (needs the browser) |
@@ -90,8 +96,10 @@ One file per section, each a list under a top-level key:
90
96
  `projects.yaml`, `certifications.yaml`, `awards.yaml`, `languages.yaml`,
91
97
  `courses.yaml`, `patents.yaml`, `recommendations.yaml`, `volunteering.yaml`,
92
98
  plus `honors.yaml` (a richer copy of awards, not included in the ZIP).
93
- Experiences keep their full description as a summary plus bullets, and roles
94
- at the same company stay grouped.
99
+ Experiences keep their full description as bullets. Several roles at one
100
+ company become separate experiences — each with its own title, dates and
101
+ bullets — all under the company's name, because a role's dates and
102
+ achievements are what tailoring selects from.
95
103
 
96
104
  ## How it works
97
105
 
@@ -109,7 +117,7 @@ at the same company stay grouped.
109
117
  git clone https://github.com/Prosperis/linkfetch
110
118
  cd linkfetch
111
119
  uv venv && uv pip install -e ".[dev]"
112
- playwright install chromium
120
+ uv run linkfetch install-browser
113
121
  uv run pytest
114
122
  ```
115
123
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "linkfetch"
3
- version = "0.1.0"
3
+ version = "0.1.2"
4
4
  description = "Capture your own LinkedIn profile on your own computer and turn it into structured profile files"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -19,8 +19,12 @@ from linkfetch.config import Config
19
19
 
20
20
  _PLAYWRIGHT_HINT = (
21
21
  "Playwright is required to capture from LinkedIn but is not installed.\n"
22
- "Install it with:\n"
23
- " uv pip install playwright && playwright install chromium"
22
+ "Reinstall linkfetch (Playwright is a dependency): uv tool install --force linkfetch"
23
+ )
24
+
25
+ _BROWSER_HINT = (
26
+ "The browser linkfetch drives is not downloaded yet (or Playwright was updated).\n"
27
+ "Run: linkfetch install-browser"
24
28
  )
25
29
 
26
30
 
@@ -43,10 +47,15 @@ def browser_session(cfg: Config):
43
47
  sync_playwright = _require_playwright()
44
48
  cfg.user_data_dir.mkdir(parents=True, exist_ok=True)
45
49
  with sync_playwright() as p:
46
- context = p.chromium.launch_persistent_context(
47
- user_data_dir=str(cfg.user_data_dir),
48
- headless=cfg.browser.headless,
49
- )
50
+ try:
51
+ context = p.chromium.launch_persistent_context(
52
+ user_data_dir=str(cfg.user_data_dir),
53
+ headless=cfg.browser.headless,
54
+ )
55
+ except Exception as exc:
56
+ if "executable doesn't exist" in str(exc).lower():
57
+ raise CaptureError(_BROWSER_HINT) from None
58
+ raise
50
59
  try:
51
60
  page = context.pages[0] if context.pages else context.new_page()
52
61
  yield Session(cfg, context, page)
@@ -90,19 +99,53 @@ class Session:
90
99
 
91
100
  # -- capture -----------------------------------------------------------
92
101
 
93
- def capture(self, url: str) -> str:
94
- """Navigate to ``url``, fully expand the page, and return rendered HTML."""
95
- self.page.goto(
96
- url,
97
- wait_until="domcontentloaded",
98
- timeout=self.cfg.browser.nav_timeout * 1000,
99
- )
100
- expand_page(
101
- self.page,
102
- scroll_pause_ms=self.cfg.browser.scroll_pause_ms,
103
- max_scrolls=self.cfg.browser.max_scrolls,
102
+ def capture(self, url: str, *, attempts: int = 3) -> str:
103
+ """Navigate to ``url``, fully expand the page, and return rendered HTML.
104
+
105
+ LinkedIn sometimes navigates a details page while it is being read — a
106
+ late client-side redirect, or a control that turns out to be a link —
107
+ which destroys the page's JavaScript context mid-scroll. That is
108
+ retried: wait for the page to settle, load ``url`` again, start over.
109
+ """
110
+ last_error: Exception | None = None
111
+ for _ in range(attempts):
112
+ try:
113
+ self.page.goto(
114
+ url,
115
+ wait_until="domcontentloaded",
116
+ timeout=self.cfg.browser.nav_timeout * 1000,
117
+ )
118
+ try:
119
+ self.page.wait_for_load_state(
120
+ "load", timeout=self.cfg.browser.nav_timeout * 1000
121
+ )
122
+ except Exception:
123
+ pass # a slow "load" event is not a reason to give up
124
+ expand_page(
125
+ self.page,
126
+ scroll_pause_ms=self.cfg.browser.scroll_pause_ms,
127
+ max_scrolls=self.cfg.browser.max_scrolls,
128
+ )
129
+ return self.page.content()
130
+ except Exception as error:
131
+ if not is_navigation_race(error):
132
+ raise
133
+ last_error = error
134
+ self.page.wait_for_timeout(2000)
135
+ raise CaptureError(
136
+ f"{url} kept navigating away while it was being read "
137
+ f"({attempts} attempts). Try again in a minute. Last error: {last_error}"
104
138
  )
105
- return self.page.content()
139
+
140
+
141
+ def is_navigation_race(error: Exception) -> bool:
142
+ """True for errors caused by the page navigating while a script ran."""
143
+ text = str(error).lower()
144
+ return (
145
+ "execution context was destroyed" in text
146
+ or "page is navigating" in text
147
+ or "frame was detached" in text
148
+ )
106
149
 
107
150
 
108
151
  def save_capture(captures_dir: Path, name: str, html: str) -> Path:
@@ -26,17 +26,43 @@ def expand_page(page, *, scroll_pause_ms: int, max_scrolls: int) -> None:
26
26
  _scroll_to_bottom(page, scroll_pause_ms=scroll_pause_ms, max_scrolls=max_scrolls)
27
27
 
28
28
 
29
+ # LinkedIn's current layout does not scroll the window: the window is exactly
30
+ # one screen tall and the page scrolls inside a panel (`<main id="workspace">`).
31
+ # Scrolling the window there loads nothing, and every list stopped at its
32
+ # first batch (10 honors, 20 skills…). So scroll whichever element actually
33
+ # scrolls — the one with the most hidden content — falling back to the window.
34
+ _SCROLL_TARGET = """() => {
35
+ const scrollable = [...document.querySelectorAll('*')].filter((el) => {
36
+ const style = getComputedStyle(el);
37
+ return /(auto|scroll)/.test(style.overflowY) && el.scrollHeight > el.clientHeight + 50;
38
+ });
39
+ scrollable.sort((a, b) => (b.scrollHeight - b.clientHeight) - (a.scrollHeight - a.clientHeight));
40
+ return scrollable[0] || document.scrollingElement || document.documentElement;
41
+ }"""
42
+
43
+ _SCROLL_HEIGHT = f"() => ({_SCROLL_TARGET})().scrollHeight"
44
+ _SCROLL_DOWN = f"() => {{ const el = ({_SCROLL_TARGET})(); el.scrollTop = el.scrollHeight; window.scrollTo(0, document.body.scrollHeight); }}"
45
+ _SCROLL_UP = f"() => {{ const el = ({_SCROLL_TARGET})(); el.scrollTop = 0; window.scrollTo(0, 0); }}"
46
+
47
+
29
48
  def _scroll_to_bottom(page, *, scroll_pause_ms: int, max_scrolls: int) -> None:
30
49
  last_height = -1
50
+ unchanged = 0
31
51
  for _ in range(max_scrolls):
32
- height = page.evaluate("document.body.scrollHeight")
52
+ height = page.evaluate(_SCROLL_HEIGHT)
33
53
  if height == last_height:
34
- break
54
+ # LinkedIn fetches the next batch asynchronously; give it one more
55
+ # pause before deciding the list is complete.
56
+ unchanged += 1
57
+ if unchanged >= 2:
58
+ break
59
+ else:
60
+ unchanged = 0
35
61
  last_height = height
36
- page.evaluate("window.scrollTo(0, document.body.scrollHeight)")
62
+ page.evaluate(_SCROLL_DOWN)
37
63
  page.wait_for_timeout(scroll_pause_ms)
38
64
  # Return to top so subsequent content is laid out.
39
- page.evaluate("window.scrollTo(0, 0)")
65
+ page.evaluate(_SCROLL_UP)
40
66
 
41
67
 
42
68
  def _click_all_expanders(page, *, scroll_pause_ms: int, max_rounds: int = 6) -> None:
@@ -79,7 +79,7 @@ def doctor() -> None:
79
79
  else:
80
80
  console.print(
81
81
  "[yellow]![/] Playwright not installed — capture/login unavailable.\n"
82
- " Reinstall linkfetch (Playwright is a dependency), then run: playwright install chromium"
82
+ " Reinstall linkfetch (Playwright is a dependency), then run: linkfetch install-browser"
83
83
  )
84
84
 
85
85
  has_session = cfg.user_data_dir.exists() and any(cfg.user_data_dir.iterdir())
@@ -96,6 +96,19 @@ def doctor() -> None:
96
96
  # ----------------------------------------------------------------------------
97
97
 
98
98
 
99
+ @app.command("install-browser")
100
+ def install_browser() -> None:
101
+ """Download the Chromium build linkfetch drives (one time, ~150 MB)."""
102
+ import subprocess
103
+
104
+ # Run Playwright through linkfetch's own Python: an installed tool
105
+ # (uv tool / pipx) exposes only the `linkfetch` command, not `playwright`.
106
+ result = subprocess.run([sys.executable, "-m", "playwright", "install", "chromium"])
107
+ if result.returncode != 0:
108
+ _fail("The browser download failed. Check your connection and run it again.")
109
+ console.print("[green]✓[/] Browser ready. Next: linkfetch login")
110
+
111
+
99
112
  @app.command()
100
113
  def login() -> None:
101
114
  """Open a browser window so you can log in to LinkedIn once."""
@@ -0,0 +1,68 @@
1
+ """Capture recovers when LinkedIn navigates a page while it is being read."""
2
+
3
+ import pytest
4
+
5
+ from linkfetch.capture.browser import CaptureError, Session, is_navigation_race
6
+ from linkfetch.config import BrowserConfig, Config
7
+
8
+
9
+ class FakePage:
10
+ def __init__(self, failures: int, error: str = "Page.evaluate: Execution context was destroyed"):
11
+ self.failures = failures
12
+ self.error = error
13
+ self.gotos = 0
14
+
15
+ def goto(self, url, **_):
16
+ self.gotos += 1
17
+
18
+ def wait_for_load_state(self, *_args, **_kwargs):
19
+ pass
20
+
21
+ def wait_for_timeout(self, _ms):
22
+ pass
23
+
24
+ def evaluate(self, script):
25
+ if self.failures:
26
+ self.failures -= 1
27
+ raise RuntimeError(self.error)
28
+ return 100
29
+
30
+ def query_selector_all(self, _selector):
31
+ return []
32
+
33
+ def content(self):
34
+ return "<html>ok</html>"
35
+
36
+
37
+ def session(page):
38
+ cfg = Config(
39
+ root=None, base_url="https://www.linkedin.com",
40
+ browser=BrowserConfig(headless=True, nav_timeout=1, scroll_pause_ms=0, max_scrolls=2),
41
+ captures_dir=None, output_dir=None, user_data_dir=None,
42
+ )
43
+ return Session(cfg, context=None, page=page)
44
+
45
+
46
+ def test_retries_after_a_navigation_race():
47
+ page = FakePage(failures=1)
48
+ assert session(page).capture("https://www.linkedin.com/in/x/details/education/") == "<html>ok</html>"
49
+ assert page.gotos == 2
50
+
51
+
52
+ def test_gives_up_with_a_clear_error_after_repeated_races():
53
+ page = FakePage(failures=10)
54
+ with pytest.raises(CaptureError, match="kept navigating away"):
55
+ session(page).capture("https://www.linkedin.com/in/x/details/education/")
56
+ assert page.gotos == 3
57
+
58
+
59
+ def test_other_errors_are_not_retried():
60
+ page = FakePage(failures=1, error="net::ERR_NAME_NOT_RESOLVED")
61
+ with pytest.raises(RuntimeError, match="ERR_NAME_NOT_RESOLVED"):
62
+ session(page).capture("https://www.linkedin.com/in/x/details/education/")
63
+ assert page.gotos == 1
64
+
65
+
66
+ def test_recognizes_navigation_errors():
67
+ assert is_navigation_race(RuntimeError("Execution context was destroyed, most likely because of a navigation"))
68
+ assert not is_navigation_race(RuntimeError("Timeout 30000ms exceeded"))
@@ -138,7 +138,7 @@ wheels = [
138
138
 
139
139
  [[package]]
140
140
  name = "linkfetch"
141
- version = "0.1.0"
141
+ version = "0.1.2"
142
142
  source = { editable = "." }
143
143
  dependencies = [
144
144
  { name = "lxml" },
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes