linkfetch 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- linkfetch/__init__.py +3 -0
- linkfetch/capture/__init__.py +1 -0
- linkfetch/capture/browser.py +112 -0
- linkfetch/capture/expand.py +61 -0
- linkfetch/cli.py +274 -0
- linkfetch/config.py +105 -0
- linkfetch/emit.py +57 -0
- linkfetch/models.py +196 -0
- linkfetch/parse/__init__.py +43 -0
- linkfetch/parse/awards.py +57 -0
- linkfetch/parse/basics.py +89 -0
- linkfetch/parse/certifications.py +70 -0
- linkfetch/parse/common.py +251 -0
- linkfetch/parse/courses.py +38 -0
- linkfetch/parse/education.py +51 -0
- linkfetch/parse/experience.py +129 -0
- linkfetch/parse/honors.py +32 -0
- linkfetch/parse/languages.py +23 -0
- linkfetch/parse/patents.py +60 -0
- linkfetch/parse/projects.py +67 -0
- linkfetch/parse/recommendations.py +64 -0
- linkfetch/parse/skills.py +30 -0
- linkfetch/parse/volunteering.py +47 -0
- linkfetch/sections.py +84 -0
- linkfetch/text.py +78 -0
- linkfetch-0.1.0.dist-info/METADATA +174 -0
- linkfetch-0.1.0.dist-info/RECORD +30 -0
- linkfetch-0.1.0.dist-info/WHEEL +4 -0
- linkfetch-0.1.0.dist-info/entry_points.txt +2 -0
- linkfetch-0.1.0.dist-info/licenses/LICENSE +21 -0
linkfetch/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Phase 1: capture rendered HTML from LinkedIn via a real browser session."""
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""Playwright session management for capturing LinkedIn HTML.
|
|
2
|
+
|
|
3
|
+
Uses a **persistent browser context** rooted at ``user_data_dir`` so the login
|
|
4
|
+
cookie survives between runs — you log in once via ``linkfetch login`` and later
|
|
5
|
+
``capture`` runs reuse the session.
|
|
6
|
+
|
|
7
|
+
Playwright is an optional dependency. Importing this module is fine without it;
|
|
8
|
+
the ImportError only surfaces when you actually open a browser, with the same
|
|
9
|
+
install hint tailor uses.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from contextlib import contextmanager
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
from linkfetch.capture.expand import expand_page
|
|
18
|
+
from linkfetch.config import Config
|
|
19
|
+
|
|
20
|
+
_PLAYWRIGHT_HINT = (
|
|
21
|
+
"Playwright is required to capture from LinkedIn but is not installed.\n"
|
|
22
|
+
"Install it with:\n"
|
|
23
|
+
" uv pip install playwright && playwright install chromium"
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class CaptureError(RuntimeError):
|
|
28
|
+
pass
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _require_playwright():
|
|
32
|
+
try:
|
|
33
|
+
from playwright.sync_api import sync_playwright
|
|
34
|
+
|
|
35
|
+
return sync_playwright
|
|
36
|
+
except ImportError as exc: # pragma: no cover - environment dependent
|
|
37
|
+
raise CaptureError(_PLAYWRIGHT_HINT) from exc
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@contextmanager
|
|
41
|
+
def browser_session(cfg: Config):
|
|
42
|
+
"""Yield a logged-in-capable :class:`Session` backed by a persistent context."""
|
|
43
|
+
sync_playwright = _require_playwright()
|
|
44
|
+
cfg.user_data_dir.mkdir(parents=True, exist_ok=True)
|
|
45
|
+
with sync_playwright() as p:
|
|
46
|
+
context = p.chromium.launch_persistent_context(
|
|
47
|
+
user_data_dir=str(cfg.user_data_dir),
|
|
48
|
+
headless=cfg.browser.headless,
|
|
49
|
+
)
|
|
50
|
+
try:
|
|
51
|
+
page = context.pages[0] if context.pages else context.new_page()
|
|
52
|
+
yield Session(cfg, context, page)
|
|
53
|
+
finally:
|
|
54
|
+
context.close()
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class Session:
|
|
58
|
+
def __init__(self, cfg: Config, context, page) -> None:
|
|
59
|
+
self.cfg = cfg
|
|
60
|
+
self.context = context
|
|
61
|
+
self.page = page
|
|
62
|
+
|
|
63
|
+
# -- auth --------------------------------------------------------------
|
|
64
|
+
|
|
65
|
+
def is_logged_in(self) -> bool:
|
|
66
|
+
"""Heuristic: the feed page redirects to /login when not authenticated."""
|
|
67
|
+
self.page.goto(
|
|
68
|
+
f"{self.cfg.base_url}/feed/",
|
|
69
|
+
wait_until="domcontentloaded",
|
|
70
|
+
timeout=self.cfg.browser.nav_timeout * 1000,
|
|
71
|
+
)
|
|
72
|
+
url = self.page.url.lower()
|
|
73
|
+
return "/login" not in url and "/authwall" not in url and "/checkpoint" not in url
|
|
74
|
+
|
|
75
|
+
def wait_for_login(self, *, poll_seconds: int = 3, max_wait_seconds: int = 900) -> bool:
|
|
76
|
+
"""Block until the user completes login in the visible window, or time out."""
|
|
77
|
+
self.page.goto(
|
|
78
|
+
f"{self.cfg.base_url}/login",
|
|
79
|
+
wait_until="domcontentloaded",
|
|
80
|
+
timeout=self.cfg.browser.nav_timeout * 1000,
|
|
81
|
+
)
|
|
82
|
+
waited = 0
|
|
83
|
+
while waited < max_wait_seconds:
|
|
84
|
+
self.page.wait_for_timeout(poll_seconds * 1000)
|
|
85
|
+
waited += poll_seconds
|
|
86
|
+
url = self.page.url.lower()
|
|
87
|
+
if all(tok not in url for tok in ("/login", "/authwall", "/checkpoint")):
|
|
88
|
+
return True
|
|
89
|
+
return False
|
|
90
|
+
|
|
91
|
+
# -- capture -----------------------------------------------------------
|
|
92
|
+
|
|
93
|
+
def capture(self, url: str) -> str:
|
|
94
|
+
"""Navigate to ``url``, fully expand the page, and return rendered HTML."""
|
|
95
|
+
self.page.goto(
|
|
96
|
+
url,
|
|
97
|
+
wait_until="domcontentloaded",
|
|
98
|
+
timeout=self.cfg.browser.nav_timeout * 1000,
|
|
99
|
+
)
|
|
100
|
+
expand_page(
|
|
101
|
+
self.page,
|
|
102
|
+
scroll_pause_ms=self.cfg.browser.scroll_pause_ms,
|
|
103
|
+
max_scrolls=self.cfg.browser.max_scrolls,
|
|
104
|
+
)
|
|
105
|
+
return self.page.content()
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def save_capture(captures_dir: Path, name: str, html: str) -> Path:
|
|
109
|
+
captures_dir.mkdir(parents=True, exist_ok=True)
|
|
110
|
+
path = captures_dir / f"{name}.html"
|
|
111
|
+
path.write_text(html, encoding="utf-8")
|
|
112
|
+
return path
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""Deterministic page-expansion routine, shared by all sections.
|
|
2
|
+
|
|
3
|
+
A LinkedIn details page lazy-loads list items as you scroll and truncates long
|
|
4
|
+
text behind "…see more" / "Show all" controls. To capture *everything* we:
|
|
5
|
+
|
|
6
|
+
1. Scroll to the bottom in steps until the page height stops growing
|
|
7
|
+
(bounded by ``max_scrolls``).
|
|
8
|
+
2. Repeatedly click every expansion control that's still visible, until none
|
|
9
|
+
remain (bounded to avoid loops).
|
|
10
|
+
|
|
11
|
+
This is browser-bound and side-effecting; it deliberately contains no
|
|
12
|
+
section-specific logic. Takes a Playwright ``Page``.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
# Text on the controls that reveal hidden content. Matched case-insensitively
|
|
18
|
+
# against button labels / inner text.
|
|
19
|
+
_EXPAND_LABELS = ("see more", "show more", "…more", "see all", "show all")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def expand_page(page, *, scroll_pause_ms: int, max_scrolls: int) -> None:
|
|
23
|
+
_scroll_to_bottom(page, scroll_pause_ms=scroll_pause_ms, max_scrolls=max_scrolls)
|
|
24
|
+
_click_all_expanders(page, scroll_pause_ms=scroll_pause_ms)
|
|
25
|
+
# A second scroll pass: expanding text can reveal more lazy content.
|
|
26
|
+
_scroll_to_bottom(page, scroll_pause_ms=scroll_pause_ms, max_scrolls=max_scrolls)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _scroll_to_bottom(page, *, scroll_pause_ms: int, max_scrolls: int) -> None:
|
|
30
|
+
last_height = -1
|
|
31
|
+
for _ in range(max_scrolls):
|
|
32
|
+
height = page.evaluate("document.body.scrollHeight")
|
|
33
|
+
if height == last_height:
|
|
34
|
+
break
|
|
35
|
+
last_height = height
|
|
36
|
+
page.evaluate("window.scrollTo(0, document.body.scrollHeight)")
|
|
37
|
+
page.wait_for_timeout(scroll_pause_ms)
|
|
38
|
+
# Return to top so subsequent content is laid out.
|
|
39
|
+
page.evaluate("window.scrollTo(0, 0)")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _click_all_expanders(page, *, scroll_pause_ms: int, max_rounds: int = 6) -> None:
|
|
43
|
+
for _ in range(max_rounds):
|
|
44
|
+
clicked = 0
|
|
45
|
+
for button in page.query_selector_all("button"):
|
|
46
|
+
try:
|
|
47
|
+
if not button.is_visible():
|
|
48
|
+
continue
|
|
49
|
+
label = (button.inner_text() or "").strip().lower()
|
|
50
|
+
aria = (button.get_attribute("aria-label") or "").lower()
|
|
51
|
+
blob = f"{label} {aria}"
|
|
52
|
+
if any(token in blob for token in _EXPAND_LABELS):
|
|
53
|
+
button.click(timeout=1500)
|
|
54
|
+
clicked += 1
|
|
55
|
+
page.wait_for_timeout(150)
|
|
56
|
+
except Exception:
|
|
57
|
+
# Buttons can detach as the DOM updates; ignore and continue.
|
|
58
|
+
continue
|
|
59
|
+
if clicked == 0:
|
|
60
|
+
break
|
|
61
|
+
page.wait_for_timeout(scroll_pause_ms)
|
linkfetch/cli.py
ADDED
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
"""linkfetch CLI.
|
|
2
|
+
|
|
3
|
+
Commands:
|
|
4
|
+
linkfetch doctor — show config, Playwright availability, session state
|
|
5
|
+
linkfetch login — open the browser and establish/refresh the session
|
|
6
|
+
linkfetch capture [opts] — phase 1: save section HTML to data/captures
|
|
7
|
+
linkfetch parse [opts] — phase 2: HTML -> profile YAML (+ ZIP for THRIVE) in data/output
|
|
8
|
+
linkfetch run [opts] — capture then parse
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import sys
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
import typer
|
|
18
|
+
from rich.console import Console
|
|
19
|
+
from rich.table import Table
|
|
20
|
+
|
|
21
|
+
from linkfetch.config import Config, load_config
|
|
22
|
+
from linkfetch.emit import dump_yaml
|
|
23
|
+
from linkfetch.sections import Section, resolve
|
|
24
|
+
|
|
25
|
+
# A Windows console on a legacy code page (cp1252, gbk, …) cannot encode "✓"
|
|
26
|
+
# or "→"; printing one raised UnicodeEncodeError and killed the command.
|
|
27
|
+
# Substitute what the console cannot show instead of crashing.
|
|
28
|
+
for _stream in (sys.stdout, sys.stderr):
|
|
29
|
+
if hasattr(_stream, "reconfigure"):
|
|
30
|
+
_stream.reconfigure(errors="replace")
|
|
31
|
+
|
|
32
|
+
app = typer.Typer(
|
|
33
|
+
help="Capture your own LinkedIn profile locally and turn it into structured profile files.",
|
|
34
|
+
no_args_is_help=True,
|
|
35
|
+
)
|
|
36
|
+
console = Console()
|
|
37
|
+
err_console = Console(stderr=True)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _cfg() -> Config:
|
|
41
|
+
return load_config()
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _fail(msg: str) -> None:
|
|
45
|
+
err_console.print(f"[bold red]Error:[/] {msg}")
|
|
46
|
+
raise typer.Exit(1)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _playwright_installed() -> bool:
|
|
50
|
+
import importlib.util
|
|
51
|
+
|
|
52
|
+
return importlib.util.find_spec("playwright") is not None
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _section_arg(sections: list[str] | None) -> list[Section]:
|
|
56
|
+
try:
|
|
57
|
+
return resolve(sections)
|
|
58
|
+
except ValueError as exc:
|
|
59
|
+
_fail(str(exc))
|
|
60
|
+
raise # unreachable; for type-checkers
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
# ----------------------------------------------------------------------------
|
|
64
|
+
# doctor
|
|
65
|
+
# ----------------------------------------------------------------------------
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@app.command()
|
|
69
|
+
def doctor() -> None:
|
|
70
|
+
"""Check configuration, Playwright availability, and login session."""
|
|
71
|
+
cfg = _cfg()
|
|
72
|
+
console.print(f"[bold]Data folder:[/] {cfg.root}")
|
|
73
|
+
console.print(f"[bold]Captures dir:[/] {cfg.captures_dir}")
|
|
74
|
+
console.print(f"[bold]Output dir:[/] {cfg.output_dir}")
|
|
75
|
+
console.print(f"[bold]Browser profile:[/] {cfg.user_data_dir}")
|
|
76
|
+
|
|
77
|
+
if _playwright_installed():
|
|
78
|
+
console.print("[green]✓[/] Playwright is installed")
|
|
79
|
+
else:
|
|
80
|
+
console.print(
|
|
81
|
+
"[yellow]![/] Playwright not installed — capture/login unavailable.\n"
|
|
82
|
+
" Reinstall linkfetch (Playwright is a dependency), then run: playwright install chromium"
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
has_session = cfg.user_data_dir.exists() and any(cfg.user_data_dir.iterdir())
|
|
86
|
+
if has_session:
|
|
87
|
+
console.print("[green]✓[/] A browser profile exists (you may already be logged in)")
|
|
88
|
+
else:
|
|
89
|
+
console.print("[yellow]![/] No browser profile yet — run `linkfetch login`")
|
|
90
|
+
|
|
91
|
+
console.print(f"[bold]Sections:[/] {', '.join(s.slug for s in resolve(None))}")
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
# ----------------------------------------------------------------------------
|
|
95
|
+
# login
|
|
96
|
+
# ----------------------------------------------------------------------------
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
@app.command()
|
|
100
|
+
def login() -> None:
|
|
101
|
+
"""Open a browser window so you can log in to LinkedIn once."""
|
|
102
|
+
cfg = _cfg()
|
|
103
|
+
from linkfetch.capture.browser import CaptureError, browser_session
|
|
104
|
+
|
|
105
|
+
if cfg.browser.headless:
|
|
106
|
+
_fail("Set browser.headless = false in config.toml to log in interactively.")
|
|
107
|
+
|
|
108
|
+
try:
|
|
109
|
+
with browser_session(cfg) as session:
|
|
110
|
+
if session.is_logged_in():
|
|
111
|
+
console.print("[green]✓[/] Already logged in. Session is ready.")
|
|
112
|
+
return
|
|
113
|
+
console.print(
|
|
114
|
+
"A browser window is open. Please log in to LinkedIn there.\n"
|
|
115
|
+
"Waiting up to 15 minutes for you to finish…"
|
|
116
|
+
)
|
|
117
|
+
if session.wait_for_login():
|
|
118
|
+
console.print("[green]✓[/] Login detected. Session saved.")
|
|
119
|
+
else:
|
|
120
|
+
_fail("Timed out waiting for login. Try `linkfetch login` again.")
|
|
121
|
+
except CaptureError as exc:
|
|
122
|
+
_fail(str(exc))
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
# ----------------------------------------------------------------------------
|
|
126
|
+
# capture (phase 1)
|
|
127
|
+
# ----------------------------------------------------------------------------
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
@app.command()
|
|
131
|
+
def capture(
|
|
132
|
+
vanity: str = typer.Option(..., "--vanity", "-v", help="Your profile slug: linkedin.com/in/<vanity>"),
|
|
133
|
+
section: list[str] = typer.Option(None, "--section", "-s", help="Section slug(s); default all"),
|
|
134
|
+
) -> None:
|
|
135
|
+
"""Phase 1: log in (if needed) and save each section's rendered HTML."""
|
|
136
|
+
cfg = _cfg()
|
|
137
|
+
sections = _section_arg(section)
|
|
138
|
+
from linkfetch.capture.browser import CaptureError, browser_session, save_capture
|
|
139
|
+
|
|
140
|
+
try:
|
|
141
|
+
with browser_session(cfg) as session:
|
|
142
|
+
if not session.is_logged_in():
|
|
143
|
+
_fail("Not logged in. Run `linkfetch login` first.")
|
|
144
|
+
|
|
145
|
+
captured: dict[str, str] = {}
|
|
146
|
+
done_sources: set[str] = set()
|
|
147
|
+
for sec in sections:
|
|
148
|
+
if sec.capture_name in done_sources:
|
|
149
|
+
continue # awards + honors share one source page
|
|
150
|
+
url = _section_url(cfg, sec, vanity)
|
|
151
|
+
console.print(f"Capturing [bold]{sec.capture_name}[/] …")
|
|
152
|
+
html = session.capture(url)
|
|
153
|
+
path = save_capture(cfg.captures_dir, sec.capture_name, html)
|
|
154
|
+
captured[sec.capture_name] = str(path)
|
|
155
|
+
done_sources.add(sec.capture_name)
|
|
156
|
+
|
|
157
|
+
meta = {"vanity": vanity, "captured": captured}
|
|
158
|
+
(cfg.captures_dir / "meta.json").write_text(
|
|
159
|
+
json.dumps(meta, indent=2), encoding="utf-8"
|
|
160
|
+
)
|
|
161
|
+
console.print(
|
|
162
|
+
f"[green]✓[/] Saved {len(captured)} page(s) to {cfg.captures_dir}"
|
|
163
|
+
)
|
|
164
|
+
except CaptureError as exc:
|
|
165
|
+
_fail(str(exc))
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _section_url(cfg: Config, sec: Section, vanity: str) -> str:
|
|
169
|
+
if sec.detail_path is None: # basics: main profile page
|
|
170
|
+
return f"{cfg.base_url}/in/{vanity}/"
|
|
171
|
+
return f"{cfg.base_url}{sec.detail_path.format(vanity=vanity)}"
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
# ----------------------------------------------------------------------------
|
|
175
|
+
# parse (phase 2)
|
|
176
|
+
# ----------------------------------------------------------------------------
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
@app.command()
|
|
180
|
+
def parse(
|
|
181
|
+
section: list[str] = typer.Option(None, "--section", "-s", help="Section slug(s); default all"),
|
|
182
|
+
zip_output: bool = typer.Option(
|
|
183
|
+
True, "--zip/--no-zip", help="Also bundle the profile files into one ZIP for THRIVE"
|
|
184
|
+
),
|
|
185
|
+
) -> None:
|
|
186
|
+
"""Phase 2: parse captured HTML into YAML. No browser needed."""
|
|
187
|
+
cfg = _cfg()
|
|
188
|
+
sections = _section_arg(section)
|
|
189
|
+
if not cfg.captures_dir.exists():
|
|
190
|
+
_fail(f"No captures found at {cfg.captures_dir}. Run `linkfetch capture` first.")
|
|
191
|
+
|
|
192
|
+
table = Table(title="Parsed sections")
|
|
193
|
+
table.add_column("section")
|
|
194
|
+
table.add_column("items", justify="right")
|
|
195
|
+
table.add_column("output")
|
|
196
|
+
table.add_column("profile", justify="center")
|
|
197
|
+
|
|
198
|
+
any_written = False
|
|
199
|
+
profile_files: list[Path] = []
|
|
200
|
+
for sec in sections:
|
|
201
|
+
src = cfg.captures_dir / f"{sec.capture_name}.html"
|
|
202
|
+
if not src.is_file():
|
|
203
|
+
table.add_row(sec.slug, "—", "[dim]no capture[/]", "")
|
|
204
|
+
continue
|
|
205
|
+
html = src.read_text(encoding="utf-8")
|
|
206
|
+
try:
|
|
207
|
+
result = sec.parser(html)
|
|
208
|
+
except Exception as exc: # one bad section must not kill the run
|
|
209
|
+
err_console.print(f"[yellow]war: {sec.slug} parser failed: {exc}[/]")
|
|
210
|
+
table.add_row(sec.slug, "[red]error[/]", "[dim]skipped[/]", "")
|
|
211
|
+
continue
|
|
212
|
+
|
|
213
|
+
out_path = cfg.output_dir / sec.out_file
|
|
214
|
+
count = 1 if not isinstance(result, list) else len(result)
|
|
215
|
+
dump_yaml(out_path, sec.top_key, result)
|
|
216
|
+
any_written = True
|
|
217
|
+
if sec.mapped:
|
|
218
|
+
profile_files.append(out_path)
|
|
219
|
+
table.add_row(
|
|
220
|
+
sec.slug,
|
|
221
|
+
str(count),
|
|
222
|
+
str(out_path.relative_to(cfg.root)) if _within(out_path, cfg.root) else str(out_path),
|
|
223
|
+
"✓" if sec.mapped else "",
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
console.print(table)
|
|
227
|
+
if any_written:
|
|
228
|
+
console.print(f"\n[green]✓[/] YAML written to {cfg.output_dir}")
|
|
229
|
+
if zip_output and profile_files:
|
|
230
|
+
bundle = write_profile_zip(cfg.output_dir / PROFILE_ZIP, profile_files)
|
|
231
|
+
console.print(
|
|
232
|
+
f"[green]✓[/] Profile bundle: {bundle}\n"
|
|
233
|
+
"Import it in THRIVE: Tailor → Profile → Import → choose this file."
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
PROFILE_ZIP = "linkfetch-profile.zip"
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def write_profile_zip(target: Path, files: list[Path]) -> Path:
|
|
241
|
+
"""Bundle profile YAML files flat into one ZIP, the form THRIVE imports."""
|
|
242
|
+
import zipfile
|
|
243
|
+
|
|
244
|
+
with zipfile.ZipFile(target, "w", compression=zipfile.ZIP_DEFLATED) as bundle:
|
|
245
|
+
for path in files:
|
|
246
|
+
bundle.write(path, arcname=path.name)
|
|
247
|
+
return target
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _within(path: Path, root: Path) -> bool:
|
|
251
|
+
try:
|
|
252
|
+
path.relative_to(root)
|
|
253
|
+
return True
|
|
254
|
+
except ValueError:
|
|
255
|
+
return False
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
# ----------------------------------------------------------------------------
|
|
259
|
+
# run (capture + parse)
|
|
260
|
+
# ----------------------------------------------------------------------------
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
@app.command()
|
|
264
|
+
def run(
|
|
265
|
+
vanity: str = typer.Option(..., "--vanity", "-v", help="Your profile slug: linkedin.com/in/<vanity>"),
|
|
266
|
+
section: list[str] = typer.Option(None, "--section", "-s", help="Section slug(s); default all"),
|
|
267
|
+
) -> None:
|
|
268
|
+
"""Convenience: capture then parse in one go."""
|
|
269
|
+
capture(vanity=vanity, section=section)
|
|
270
|
+
parse(section=section, zip_output=True)
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
if __name__ == "__main__":
|
|
274
|
+
app()
|
linkfetch/config.py
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""Configuration loading.
|
|
2
|
+
|
|
3
|
+
Reads ``config.toml`` from the project root and exposes a typed ``Config``.
|
|
4
|
+
The project root is discovered by walking upward from the current working
|
|
5
|
+
directory looking for ``config.toml``, so a checkout works from any
|
|
6
|
+
subdirectory.
|
|
7
|
+
|
|
8
|
+
An installed copy (``uv tool install`` / ``pipx``) usually finds no
|
|
9
|
+
``config.toml``. It then keeps everything — captured pages, output, and the
|
|
10
|
+
logged-in browser profile — in the per-user data folder for the OS (see
|
|
11
|
+
``user_data_root``), never inside the Python installation.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import os
|
|
17
|
+
import sys
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
try: # Python 3.11+
|
|
22
|
+
import tomllib as _toml
|
|
23
|
+
except ModuleNotFoundError: # Python 3.10
|
|
24
|
+
import tomli as _toml # type: ignore[no-redef]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass(frozen=True)
|
|
28
|
+
class BrowserConfig:
|
|
29
|
+
headless: bool
|
|
30
|
+
nav_timeout: int # seconds
|
|
31
|
+
scroll_pause_ms: int
|
|
32
|
+
max_scrolls: int
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True)
|
|
36
|
+
class Config:
|
|
37
|
+
root: Path
|
|
38
|
+
base_url: str
|
|
39
|
+
browser: BrowserConfig
|
|
40
|
+
captures_dir: Path
|
|
41
|
+
output_dir: Path
|
|
42
|
+
user_data_dir: Path
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def user_data_root() -> Path:
|
|
46
|
+
r"""The per-user data folder for linkfetch on this OS.
|
|
47
|
+
|
|
48
|
+
Windows: ``%LOCALAPPDATA%\linkfetch``; macOS:
|
|
49
|
+
``~/Library/Application Support/linkfetch``; elsewhere:
|
|
50
|
+
``$XDG_DATA_HOME/linkfetch`` or ``~/.local/share/linkfetch``.
|
|
51
|
+
``LINKFETCH_HOME`` overrides all of them.
|
|
52
|
+
"""
|
|
53
|
+
override = os.environ.get("LINKFETCH_HOME")
|
|
54
|
+
if override:
|
|
55
|
+
return Path(override).expanduser()
|
|
56
|
+
if sys.platform == "win32":
|
|
57
|
+
base = os.environ.get("LOCALAPPDATA") or str(Path.home() / "AppData" / "Local")
|
|
58
|
+
return Path(base) / "linkfetch"
|
|
59
|
+
if sys.platform == "darwin":
|
|
60
|
+
return Path.home() / "Library" / "Application Support" / "linkfetch"
|
|
61
|
+
base = os.environ.get("XDG_DATA_HOME") or str(Path.home() / ".local" / "share")
|
|
62
|
+
return Path(base) / "linkfetch"
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def find_project_root(start: Path | None = None) -> Path:
|
|
66
|
+
"""The dir containing ``config.toml`` above ``start``, else the user data folder."""
|
|
67
|
+
start = (start or Path.cwd()).resolve()
|
|
68
|
+
for candidate in [start, *start.parents]:
|
|
69
|
+
if (candidate / "config.toml").is_file():
|
|
70
|
+
return candidate
|
|
71
|
+
return user_data_root()
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def load_config(root: Path | None = None) -> Config:
|
|
75
|
+
root = find_project_root(root)
|
|
76
|
+
config_path = root / "config.toml"
|
|
77
|
+
data: dict = {}
|
|
78
|
+
if config_path.is_file():
|
|
79
|
+
with config_path.open("rb") as fh:
|
|
80
|
+
data = _toml.load(fh)
|
|
81
|
+
|
|
82
|
+
browser_raw = data.get("browser", {})
|
|
83
|
+
paths_raw = data.get("paths", {})
|
|
84
|
+
linkedin_raw = data.get("linkedin", {})
|
|
85
|
+
|
|
86
|
+
browser = BrowserConfig(
|
|
87
|
+
headless=bool(browser_raw.get("headless", False)),
|
|
88
|
+
nav_timeout=int(browser_raw.get("nav_timeout", 30)),
|
|
89
|
+
scroll_pause_ms=int(browser_raw.get("scroll_pause_ms", 600)),
|
|
90
|
+
max_scrolls=int(browser_raw.get("max_scrolls", 40)),
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
def resolve(key: str, default: str) -> Path:
|
|
94
|
+
raw = paths_raw.get(key, default)
|
|
95
|
+
p = Path(raw)
|
|
96
|
+
return p if p.is_absolute() else root / p
|
|
97
|
+
|
|
98
|
+
return Config(
|
|
99
|
+
root=root,
|
|
100
|
+
base_url=linkedin_raw.get("base_url", "https://www.linkedin.com"),
|
|
101
|
+
browser=browser,
|
|
102
|
+
captures_dir=resolve("captures", "data/captures"),
|
|
103
|
+
output_dir=resolve("output", "data/output"),
|
|
104
|
+
user_data_dir=resolve("user_data_dir", "data/browser"),
|
|
105
|
+
)
|
linkfetch/emit.py
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""Serialize parsed models to YAML.
|
|
2
|
+
|
|
3
|
+
Uses the same dumper settings as tailor's profile store so the mapped output
|
|
4
|
+
files are stylistically identical to tailor's own YAML (and therefore drop in
|
|
5
|
+
cleanly): ``sort_keys=False``, ``allow_unicode=True``, block style, width 100.
|
|
6
|
+
|
|
7
|
+
Each output file wraps its list under a top-level key (``experiences:``,
|
|
8
|
+
``skills:``, …), matching tailor's per-category file shape. ``basics`` is
|
|
9
|
+
wrapped under ``basics:``.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
import yaml
|
|
17
|
+
from pydantic import BaseModel
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _to_plain(models: list[BaseModel] | BaseModel) -> object:
|
|
21
|
+
"""Convert pydantic models to plain dicts, dropping None/empty values.
|
|
22
|
+
|
|
23
|
+
Dropping empties keeps the YAML clean and human-editable: a scraped item
|
|
24
|
+
with no projects/tags doesn't litter the file with ``tags: []``.
|
|
25
|
+
"""
|
|
26
|
+
if isinstance(models, BaseModel):
|
|
27
|
+
return _prune(models.model_dump())
|
|
28
|
+
return [_prune(m.model_dump()) for m in models]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _prune(value: object) -> object:
|
|
32
|
+
if isinstance(value, dict):
|
|
33
|
+
pruned = {}
|
|
34
|
+
for k, v in value.items():
|
|
35
|
+
pv = _prune(v)
|
|
36
|
+
if pv is None or pv == [] or pv == {}:
|
|
37
|
+
continue
|
|
38
|
+
pruned[k] = pv
|
|
39
|
+
return pruned
|
|
40
|
+
if isinstance(value, list):
|
|
41
|
+
return [_prune(v) for v in value]
|
|
42
|
+
return value
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def dump_yaml(path: Path, top_key: str, models: list[BaseModel] | BaseModel) -> None:
|
|
46
|
+
"""Write ``{top_key: <plain data>}`` to ``path`` as tailor-style YAML."""
|
|
47
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
48
|
+
payload = {top_key: _to_plain(models)}
|
|
49
|
+
with path.open("w", encoding="utf-8") as fh:
|
|
50
|
+
yaml.safe_dump(
|
|
51
|
+
payload,
|
|
52
|
+
fh,
|
|
53
|
+
sort_keys=False,
|
|
54
|
+
allow_unicode=True,
|
|
55
|
+
default_flow_style=False,
|
|
56
|
+
width=100,
|
|
57
|
+
)
|