linkfetch 0.1.2__tar.gz → 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. {linkfetch-0.1.2 → linkfetch-1.0.0}/PKG-INFO +49 -20
  2. {linkfetch-0.1.2 → linkfetch-1.0.0}/README.md +47 -18
  3. {linkfetch-0.1.2 → linkfetch-1.0.0}/pyproject.toml +2 -2
  4. linkfetch-1.0.0/src/linkfetch/__init__.py +3 -0
  5. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/capture/browser.py +2 -3
  6. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/cli.py +22 -4
  7. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/emit.py +30 -8
  8. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/models.py +13 -14
  9. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/parse/awards.py +1 -1
  10. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/parse/common.py +3 -0
  11. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/parse/honors.py +2 -2
  12. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/parse/skills.py +1 -1
  13. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/sections.py +4 -4
  14. linkfetch-1.0.0/tests/test_emit.py +37 -0
  15. linkfetch-1.0.0/tests/test_regressions.py +34 -0
  16. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/test_release.py +3 -3
  17. {linkfetch-0.1.2 → linkfetch-1.0.0}/uv.lock +1 -1
  18. linkfetch-0.1.2/docs/superpowers/specs/2026-06-15-linkfetch-design.md +0 -184
  19. linkfetch-0.1.2/src/linkfetch/__init__.py +0 -3
  20. linkfetch-0.1.2/tests/test_emit.py +0 -78
  21. {linkfetch-0.1.2 → linkfetch-1.0.0}/.github/workflows/release.yml +0 -0
  22. {linkfetch-0.1.2 → linkfetch-1.0.0}/.github/workflows/test.yml +0 -0
  23. {linkfetch-0.1.2 → linkfetch-1.0.0}/.gitignore +0 -0
  24. {linkfetch-0.1.2 → linkfetch-1.0.0}/LICENSE +0 -0
  25. {linkfetch-0.1.2 → linkfetch-1.0.0}/config.toml +0 -0
  26. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/capture/__init__.py +0 -0
  27. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/capture/expand.py +0 -0
  28. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/config.py +0 -0
  29. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/parse/__init__.py +0 -0
  30. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/parse/basics.py +0 -0
  31. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/parse/certifications.py +0 -0
  32. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/parse/courses.py +0 -0
  33. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/parse/education.py +0 -0
  34. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/parse/experience.py +0 -0
  35. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/parse/languages.py +0 -0
  36. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/parse/patents.py +0 -0
  37. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/parse/projects.py +0 -0
  38. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/parse/recommendations.py +0 -0
  39. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/parse/volunteering.py +0 -0
  40. {linkfetch-0.1.2 → linkfetch-1.0.0}/src/linkfetch/text.py +0 -0
  41. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/conftest.py +0 -0
  42. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/fixtures/basics.html +0 -0
  43. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/fixtures/certifications.html +0 -0
  44. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/fixtures/education.html +0 -0
  45. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/fixtures/experience.html +0 -0
  46. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/fixtures/honors.html +0 -0
  47. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/fixtures/languages.html +0 -0
  48. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/fixtures/patents.html +0 -0
  49. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/fixtures/projects.html +0 -0
  50. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/fixtures/recommendations.html +0 -0
  51. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/fixtures/skills.html +0 -0
  52. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/fixtures/volunteering.html +0 -0
  53. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/test_capture_retry.py +0 -0
  54. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/test_config.py +0 -0
  55. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/test_experience.py +0 -0
  56. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/test_sections.py +0 -0
  57. {linkfetch-0.1.2 → linkfetch-1.0.0}/tests/test_text.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: linkfetch
3
- Version: 0.1.2
3
+ Version: 1.0.0
4
4
  Summary: Capture your own LinkedIn profile on your own computer and turn it into structured profile files
5
5
  Project-URL: Homepage, https://github.com/Prosperis/linkfetch
6
6
  Project-URL: Issues, https://github.com/Prosperis/linkfetch/issues
@@ -8,7 +8,7 @@ Author: Prosperis
8
8
  License-Expression: MIT
9
9
  License-File: LICENSE
10
10
  Keywords: cv,export,linkedin,playwright,profile,resume
11
- Classifier: Development Status :: 4 - Beta
11
+ Classifier: Development Status :: 5 - Production/Stable
12
12
  Classifier: Environment :: Console
13
13
  Classifier: Intended Audience :: End Users/Desktop
14
14
  Classifier: Operating System :: OS Independent
@@ -41,8 +41,11 @@ Description-Content-Type: text/markdown
41
41
 
42
42
  Capture **your own** LinkedIn profile on **your own computer** and turn it into
43
43
  structured profile files — every role, description, skill, project,
44
- certification and more — ready to import into
45
- [THRIVE](https://prosperis-thrive.vercel.app) or any tool that reads YAML.
44
+ certification and more — as plain YAML you can keep, edit, version or feed to
45
+ any other tool.
46
+
47
+ **In:** your LinkedIn profile address. **Out:** one YAML file per profile
48
+ section, plus `linkfetch-profile.zip` bundling them.
46
49
 
47
50
  LinkedIn's official data export leaves out details the profile page shows.
48
51
  linkfetch reads the rendered pages instead, in a real browser window that you
@@ -72,7 +75,7 @@ linkfetch install-browser # one-time download of the browser linkfetch dr
72
75
  ## Use
73
76
 
74
77
  ```bash
75
- linkfetch login # a browser window opens: log in to LinkedIn, then close it
78
+ linkfetch login # a browser window opens; log in to LinkedIn and it closes by itself
76
79
  linkfetch run --vanity your-name # capture your profile and build the files
77
80
  ```
78
81
 
@@ -82,12 +85,7 @@ linkfetch run --vanity your-name # capture your profile and build the files
82
85
  `run` visits each section of your profile (experience, education, skills, …),
83
86
  expands every "see more", saves the pages, and turns them into one YAML file
84
87
  per section plus **`linkfetch-profile.zip`**. `linkfetch` prints where they
85
- are.
86
-
87
- ### Import into THRIVE
88
-
89
- In THRIVE, open **Tailor → Profile → Import** and choose
90
- `linkfetch-profile.zip`. Review each section before saving.
88
+ are. See [Output](#output) for what they contain.
91
89
 
92
90
  ## Where your data is kept
93
91
 
@@ -125,15 +123,46 @@ and `parse --no-zip` to skip the ZIP.
125
123
 
126
124
  ## Output
127
125
 
128
- One file per section, each a list under a top-level key:
129
- `basics.yaml`, `experiences.yaml`, `education.yaml`, `skills.yaml`,
130
- `projects.yaml`, `certifications.yaml`, `awards.yaml`, `languages.yaml`,
131
- `courses.yaml`, `patents.yaml`, `recommendations.yaml`, `volunteering.yaml`,
132
- plus `honors.yaml` (a richer copy of awards, not included in the ZIP).
133
- Experiences keep their full description as bullets. Several roles at one
134
- company become separate experiences — each with its own title, dates and
135
- bullets — all under the company's name, because a role's dates and
136
- achievements are what tailoring selects from.
126
+ Written to `data/output/` inside the data folder.
127
+
128
+ | File | Contains |
129
+ |---|---|
130
+ | `basics.yaml` | name, headline, location, about, contact links |
131
+ | `experiences.yaml` | one entry per role: company, title, dates, location, description bullets, skills used |
132
+ | `education.yaml` | school, degree, field, dates, activities |
133
+ | `skills.yaml` | skill name, and where you used it |
134
+ | `projects.yaml` | name, role, dates, description, links |
135
+ | `certifications.yaml` | name, issuer, issue/expiry dates, credential id and link |
136
+ | `awards.yaml` | honors & awards: name, issuer, date |
137
+ | `languages.yaml` | language and proficiency |
138
+ | `courses.yaml` | course name and number |
139
+ | `patents.yaml` | title, number, status, date, summary |
140
+ | `recommendations.yaml` | author, their title, relationship, text |
141
+ | `volunteering.yaml` | organization, role, cause, dates, summary |
142
+ | `honors.yaml` | a fuller copy of honors & awards, with descriptions |
143
+ | `linkfetch-profile.zip` | every file above except `honors.yaml`, in one archive |
144
+
145
+ Each file holds its section under a top-level key named after the file:
146
+
147
+ ```yaml
148
+ experiences:
149
+ - id: example-co-senior-engineer
150
+ org: Example Co
151
+ location: Austin, Texas, United States
152
+ positions:
153
+ - title: Senior Engineer
154
+ timeline:
155
+ start: Mar 2022
156
+ end: Present
157
+ bullets:
158
+ - Led the migration of the billing platform
159
+ tech:
160
+ - TypeScript
161
+ ```
162
+
163
+ Several roles at one company become separate entries — each with its own
164
+ title, dates and bullets — under the company's name. Empty fields are left out.
165
+ `id` values are stable slugs, so re-running produces small, reviewable diffs.
137
166
 
138
167
  ## How it works
139
168
 
@@ -7,8 +7,11 @@
7
7
 
8
8
  Capture **your own** LinkedIn profile on **your own computer** and turn it into
9
9
  structured profile files — every role, description, skill, project,
10
- certification and more — ready to import into
11
- [THRIVE](https://prosperis-thrive.vercel.app) or any tool that reads YAML.
10
+ certification and more — as plain YAML you can keep, edit, version or feed to
11
+ any other tool.
12
+
13
+ **In:** your LinkedIn profile address. **Out:** one YAML file per profile
14
+ section, plus `linkfetch-profile.zip` bundling them.
12
15
 
13
16
  LinkedIn's official data export leaves out details the profile page shows.
14
17
  linkfetch reads the rendered pages instead, in a real browser window that you
@@ -38,7 +41,7 @@ linkfetch install-browser # one-time download of the browser linkfetch dr
38
41
  ## Use
39
42
 
40
43
  ```bash
41
- linkfetch login # a browser window opens: log in to LinkedIn, then close it
44
+ linkfetch login # a browser window opens; log in to LinkedIn and it closes by itself
42
45
  linkfetch run --vanity your-name # capture your profile and build the files
43
46
  ```
44
47
 
@@ -48,12 +51,7 @@ linkfetch run --vanity your-name # capture your profile and build the files
48
51
  `run` visits each section of your profile (experience, education, skills, …),
49
52
  expands every "see more", saves the pages, and turns them into one YAML file
50
53
  per section plus **`linkfetch-profile.zip`**. `linkfetch` prints where they
51
- are.
52
-
53
- ### Import into THRIVE
54
-
55
- In THRIVE, open **Tailor → Profile → Import** and choose
56
- `linkfetch-profile.zip`. Review each section before saving.
54
+ are. See [Output](#output) for what they contain.
57
55
 
58
56
  ## Where your data is kept
59
57
 
@@ -91,15 +89,46 @@ and `parse --no-zip` to skip the ZIP.
91
89
 
92
90
  ## Output
93
91
 
94
- One file per section, each a list under a top-level key:
95
- `basics.yaml`, `experiences.yaml`, `education.yaml`, `skills.yaml`,
96
- `projects.yaml`, `certifications.yaml`, `awards.yaml`, `languages.yaml`,
97
- `courses.yaml`, `patents.yaml`, `recommendations.yaml`, `volunteering.yaml`,
98
- plus `honors.yaml` (a richer copy of awards, not included in the ZIP).
99
- Experiences keep their full description as bullets. Several roles at one
100
- company become separate experiences — each with its own title, dates and
101
- bullets — all under the company's name, because a role's dates and
102
- achievements are what tailoring selects from.
92
+ Written to `data/output/` inside the data folder.
93
+
94
+ | File | Contains |
95
+ |---|---|
96
+ | `basics.yaml` | name, headline, location, about, contact links |
97
+ | `experiences.yaml` | one entry per role: company, title, dates, location, description bullets, skills used |
98
+ | `education.yaml` | school, degree, field, dates, activities |
99
+ | `skills.yaml` | skill name, and where you used it |
100
+ | `projects.yaml` | name, role, dates, description, links |
101
+ | `certifications.yaml` | name, issuer, issue/expiry dates, credential id and link |
102
+ | `awards.yaml` | honors & awards: name, issuer, date |
103
+ | `languages.yaml` | language and proficiency |
104
+ | `courses.yaml` | course name and number |
105
+ | `patents.yaml` | title, number, status, date, summary |
106
+ | `recommendations.yaml` | author, their title, relationship, text |
107
+ | `volunteering.yaml` | organization, role, cause, dates, summary |
108
+ | `honors.yaml` | a fuller copy of honors & awards, with descriptions |
109
+ | `linkfetch-profile.zip` | every file above except `honors.yaml`, in one archive |
110
+
111
+ Each file holds its section under a top-level key named after the file:
112
+
113
+ ```yaml
114
+ experiences:
115
+ - id: example-co-senior-engineer
116
+ org: Example Co
117
+ location: Austin, Texas, United States
118
+ positions:
119
+ - title: Senior Engineer
120
+ timeline:
121
+ start: Mar 2022
122
+ end: Present
123
+ bullets:
124
+ - Led the migration of the billing platform
125
+ tech:
126
+ - TypeScript
127
+ ```
128
+
129
+ Several roles at one company become separate entries — each with its own
130
+ title, dates and bullets — under the company's name. Empty fields are left out.
131
+ `id` values are stable slugs, so re-running produces small, reviewable diffs.
103
132
 
104
133
  ## How it works
105
134
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "linkfetch"
3
- version = "0.1.2"
3
+ version = "1.0.0"
4
4
  description = "Capture your own LinkedIn profile on your own computer and turn it into structured profile files"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -9,7 +9,7 @@ license-files = ["LICENSE"]
9
9
  authors = [{ name = "Prosperis" }]
10
10
  keywords = ["linkedin", "profile", "resume", "cv", "export", "playwright"]
11
11
  classifiers = [
12
- "Development Status :: 4 - Beta",
12
+ "Development Status :: 5 - Production/Stable",
13
13
  "Environment :: Console",
14
14
  "Intended Audience :: End Users/Desktop",
15
15
  "Operating System :: OS Independent",
@@ -0,0 +1,3 @@
1
+ """linkfetch — capture your own LinkedIn profile locally as structured YAML."""
2
+
3
+ __version__ = "0.1.0"
@@ -4,9 +4,8 @@ Uses a **persistent browser context** rooted at ``user_data_dir`` so the login
4
4
  cookie survives between runs — you log in once via ``linkfetch login`` and later
5
5
  ``capture`` runs reuse the session.
6
6
 
7
- Playwright is an optional dependency. Importing this module is fine without it;
8
- the ImportError only surfaces when you actually open a browser, with the same
9
- install hint tailor uses.
7
+ Importing this module never needs Playwright; a missing Playwright or browser
8
+ only surfaces when a browser is actually opened, with a hint on how to fix it.
10
9
  """
11
10
 
12
11
  from __future__ import annotations
@@ -4,7 +4,7 @@ Commands:
4
4
  linkfetch doctor — show config, Playwright availability, session state
5
5
  linkfetch login — open the browser and establish/refresh the session
6
6
  linkfetch capture [opts] — phase 1: save section HTML to data/captures
7
- linkfetch parse [opts] — phase 2: HTML -> profile YAML (+ ZIP for THRIVE) in data/output
7
+ linkfetch parse [opts] — phase 2: HTML -> profile YAML (+ linkfetch-profile.zip) in data/output
8
8
  linkfetch run [opts] — capture then parse
9
9
  """
10
10
 
@@ -52,6 +52,20 @@ def _playwright_installed() -> bool:
52
52
  return importlib.util.find_spec("playwright") is not None
53
53
 
54
54
 
55
+ def _browser_downloaded() -> bool:
56
+ """True if the Chromium build this Playwright version drives is on disk.
57
+
58
+ Starts Playwright's driver only to ask for the path; no browser opens.
59
+ """
60
+ try:
61
+ from playwright.sync_api import sync_playwright
62
+
63
+ with sync_playwright() as p:
64
+ return Path(p.chromium.executable_path).is_file()
65
+ except Exception:
66
+ return False
67
+
68
+
55
69
  def _section_arg(sections: list[str] | None) -> list[Section]:
56
70
  try:
57
71
  return resolve(sections)
@@ -76,6 +90,10 @@ def doctor() -> None:
76
90
 
77
91
  if _playwright_installed():
78
92
  console.print("[green]✓[/] Playwright is installed")
93
+ if _browser_downloaded():
94
+ console.print("[green]✓[/] The browser is downloaded")
95
+ else:
96
+ console.print("[yellow]![/] The browser is not downloaded yet — run `linkfetch install-browser`")
79
97
  else:
80
98
  console.print(
81
99
  "[yellow]![/] Playwright not installed — capture/login unavailable.\n"
@@ -193,7 +211,7 @@ def _section_url(cfg: Config, sec: Section, vanity: str) -> str:
193
211
  def parse(
194
212
  section: list[str] = typer.Option(None, "--section", "-s", help="Section slug(s); default all"),
195
213
  zip_output: bool = typer.Option(
196
- True, "--zip/--no-zip", help="Also bundle the profile files into one ZIP for THRIVE"
214
+ True, "--zip/--no-zip", help="Also bundle the profile files into linkfetch-profile.zip"
197
215
  ),
198
216
  ) -> None:
199
217
  """Phase 2: parse captured HTML into YAML. No browser needed."""
@@ -243,7 +261,7 @@ def parse(
243
261
  bundle = write_profile_zip(cfg.output_dir / PROFILE_ZIP, profile_files)
244
262
  console.print(
245
263
  f"[green]✓[/] Profile bundle: {bundle}\n"
246
- "Import it in THRIVE: Tailor → Profile → Import → choose this file."
264
+ "One file with every profile section, ready to import or share."
247
265
  )
248
266
 
249
267
 
@@ -251,7 +269,7 @@ PROFILE_ZIP = "linkfetch-profile.zip"
251
269
 
252
270
 
253
271
  def write_profile_zip(target: Path, files: list[Path]) -> Path:
254
- """Bundle profile YAML files flat into one ZIP, the form THRIVE imports."""
272
+ """Bundle profile YAML files flat into one ZIP (one YAML file per section)."""
255
273
  import zipfile
256
274
 
257
275
  with zipfile.ZipFile(target, "w", compression=zipfile.ZIP_DEFLATED) as bundle:
@@ -1,12 +1,11 @@
1
1
  """Serialize parsed models to YAML.
2
2
 
3
- Uses the same dumper settings as tailor's profile store so the mapped output
4
- files are stylistically identical to tailor's own YAML (and therefore drop in
5
- cleanly): ``sort_keys=False``, ``allow_unicode=True``, block style, width 100.
3
+ Output is stable and readable: ``sort_keys=False``, ``allow_unicode=True``,
4
+ block style, width 100, so re-runs produce small, reviewable diffs.
6
5
 
7
- Each output file wraps its list under a top-level key (``experiences:``,
8
- ``skills:``, …), matching tailor's per-category file shape. ``basics`` is
9
- wrapped under ``basics:``.
6
+ Each output file wraps its list under a top-level key named after the file
7
+ (``experiences:``, ``skills:``, …). ``basics`` is a single mapping under
8
+ ``basics:``.
10
9
  """
11
10
 
12
11
  from __future__ import annotations
@@ -25,7 +24,30 @@ def _to_plain(models: list[BaseModel] | BaseModel) -> object:
25
24
  """
26
25
  if isinstance(models, BaseModel):
27
26
  return _prune(models.model_dump())
28
- return [_prune(m.model_dump()) for m in models]
27
+ return _unique([_prune(m.model_dump()) for m in models])
28
+
29
+
30
+ def _unique(items: list[dict]) -> list[dict]:
31
+ """Drop exact duplicates and keep ``id`` values unique.
32
+
33
+ LinkedIn can render the same entry twice (a lazy-loaded list re-rendering,
34
+ or an entry genuinely listed twice); an identical copy adds nothing. Two
35
+ *different* entries that slug to the same ``id`` both stay, the later one
36
+ suffixed ``-2``, ``-3`` … so ids remain unique within a file.
37
+ """
38
+ out: list[dict] = []
39
+ seen_ids: dict[str, int] = {}
40
+ for item in items:
41
+ if item in out:
42
+ continue
43
+ base = item.get("id")
44
+ if isinstance(base, str):
45
+ count = seen_ids.get(base, 0) + 1
46
+ seen_ids[base] = count
47
+ if count > 1:
48
+ item = {**item, "id": f"{base}-{count}"}
49
+ out.append(item)
50
+ return out
29
51
 
30
52
 
31
53
  def _prune(value: object) -> object:
@@ -43,7 +65,7 @@ def _prune(value: object) -> object:
43
65
 
44
66
 
45
67
  def dump_yaml(path: Path, top_key: str, models: list[BaseModel] | BaseModel) -> None:
46
- """Write ``{top_key: <plain data>}`` to ``path`` as tailor-style YAML."""
68
+ """Write ``{top_key: <plain data>}`` to ``path`` as YAML."""
47
69
  path.parent.mkdir(parents=True, exist_ok=True)
48
70
  payload = {top_key: _to_plain(models)}
49
71
  with path.open("w", encoding="utf-8") as fh:
@@ -2,15 +2,14 @@
2
2
 
3
3
  Two families:
4
4
 
5
- 1. **tailor-mapped** models — a *subset* of tailor's profile models using the
6
- **same field names** (``id``, ``summary``, ``bullets``, ``tags``, ``org``,
7
- ``positions``, ``timeline``, …). YAML emitted from these drops straight into
8
- tailor's profile store. We intentionally omit fields tailor fills in by hand
9
- (e.g. multiple position framings, emphasis) — the scrape is deterministic and
5
+ 1. **profile** models — one per profile section, with consistent field names
6
+ (``id``, ``summary``, ``bullets``, ``tags``, ``org``, ``positions``,
7
+ ``timeline``, …). Fields that only a person can fill in (alternative title
8
+ framings, emphasis) exist but stay empty: the scrape is deterministic and
10
9
  leaves enrichment to the user.
11
10
 
12
- 2. **extra** models — small purpose-built models for LinkedIn sections tailor
13
- has no home for (education, volunteering, recommendations, …).
11
+ 2. **extra** models — richer copies of a section kept alongside the profile
12
+ files (currently ``honors``).
14
13
 
15
14
  All ``id`` values are stable slugs derived from the item's identity so re-runs
16
15
  produce diffable YAML.
@@ -28,7 +27,7 @@ from pydantic import BaseModel, Field
28
27
  class Timeline(BaseModel):
29
28
  """A start/end span. ``end`` omitted or 'present' means ongoing.
30
29
 
31
- Mirrors tailor.profile.models.Timeline (field-compatible).
30
+ ``start``/``end`` keep LinkedIn's wording; a missing ``end`` means current.
32
31
  """
33
32
 
34
33
  start: str | None = None
@@ -37,10 +36,10 @@ class Timeline(BaseModel):
37
36
 
38
37
 
39
38
  class Position(BaseModel):
40
- """One title framing for a role. Mirrors tailor's Position.
39
+ """One title framing for a role.
41
40
 
42
41
  The scrape produces exactly one framing (the real LinkedIn title); the user
43
- adds alternative framings later in tailor.
42
+ can add alternative framings later.
44
43
  """
45
44
 
46
45
  title: str
@@ -55,7 +54,7 @@ class Link(BaseModel):
55
54
 
56
55
 
57
56
  # ---------------------------------------------------------------------------
58
- # tailor-mapped models (field names match tailor.profile.models)
57
+ # profile models
59
58
  # ---------------------------------------------------------------------------
60
59
 
61
60
 
@@ -129,7 +128,7 @@ class Award(BaseModel):
129
128
 
130
129
 
131
130
  # ---------------------------------------------------------------------------
132
- # extra models (linkfetch-local schema; not in tailor)
131
+ # extra models
133
132
  # ---------------------------------------------------------------------------
134
133
 
135
134
 
@@ -185,8 +184,8 @@ class Language(BaseModel):
185
184
  class Honor(BaseModel):
186
185
  """Honors & awards as a standalone extra section.
187
186
 
188
- Note: tailor's Award is also populated from this section; this richer extra
189
- file preserves everything (issuer, date, description) for the user.
187
+ The trimmed Award (awards.yaml) comes from the same section; this richer
188
+ file preserves everything (issuer, date, description).
190
189
  """
191
190
 
192
191
  id: str
@@ -1,4 +1,4 @@
1
- """Parse Honors & awards into tailor's Award model.
1
+ """Parse Honors & awards into the trimmed Award model (awards.yaml).
2
2
 
3
3
  Ordered fragments per entry::
4
4
 
@@ -44,6 +44,9 @@ _CHROME = (
44
44
  "select language", "profile language", "accessibility", "talent solutions",
45
45
  "community guidelines", "marketing solutions", "privacy & terms",
46
46
  "ad choices", "sales solutions", "small business", "safety center",
47
+ # In-section notices, e.g. "Projects from connected apps sync
48
+ # automatically and can't be edited on LinkedIn."
49
+ "sync automatically", "be edited on linkedin",
47
50
  )
48
51
 
49
52
  # Section-title fragments that appear as the first child block of the content
@@ -1,8 +1,8 @@
1
1
  """Parse Honors & awards into the richer extra Honor model.
2
2
 
3
3
  Same source section as awards.py (same fragment order), but kept as a standalone
4
- extra file so the full description survives in ``summary``. tailor's Award
5
- (awards.py) is the trimmed, mappable view.
4
+ extra file so the full description survives in ``summary``. The Award
5
+ (awards.py) is the trimmed view included in the profile bundle.
6
6
  """
7
7
 
8
8
  from __future__ import annotations
@@ -3,7 +3,7 @@
3
3
  Each entry's first visible fragment is the skill name; a second fragment gives
4
4
  the context where it was used ("Senior Software Development Engineer at Roche"),
5
5
  which we keep as the skill ``summary``. Duplicates are dropped; categorization is
6
- left to the user in tailor.
6
+ left to the user.
7
7
  """
8
8
 
9
9
  from __future__ import annotations
@@ -6,12 +6,12 @@ Each Section ties together:
6
6
  None for ``basics`` which comes from the main profile page
7
7
  - ``parser``: the pure ``parse(html)`` callable
8
8
  - ``out_file`` / ``top_key``: where parsed YAML is written and under what key
9
- - ``mapped``: True if the output file is one of the profile files THRIVE and
10
- tailor read (``data/profile/<name>.yaml``); ``honors`` is an extra,
11
- richer copy of awards and is not
9
+ - ``mapped``: True for the standard profile files that go into
10
+ ``linkfetch-profile.zip``; ``honors`` is an extra, richer copy of awards
11
+ and is not
12
12
 
13
13
  Note ``honors`` and ``awards`` share the same source page ('honors') but produce
14
- two different output files (the trimmed tailor Award + the richer extra Honor).
14
+ two different output files (the trimmed Award + the richer extra Honor).
15
15
  """
16
16
 
17
17
  from __future__ import annotations
@@ -0,0 +1,37 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+
5
+ import yaml
6
+
7
+ from linkfetch.emit import dump_yaml
8
+ from linkfetch.models import Experience, Position, Timeline
9
+
10
+
11
+ def _sample_experiences() -> list[Experience]:
12
+ return [
13
+ Experience(
14
+ id="acme-corp-senior-backend-engineer",
15
+ org="Acme Corp",
16
+ location="Berlin, Germany",
17
+ positions=[Position(title="Senior Backend Engineer")],
18
+ timeline=Timeline(start="Mar 2021", end="Present"),
19
+ bullets=["Led the payments platform team.", "Cut p99 latency by 40%."],
20
+ )
21
+ ]
22
+
23
+
24
+ def test_round_trip_and_pruning(tmp_path):
25
+ path = tmp_path / "experiences.yaml"
26
+ dump_yaml(path, "experiences", _sample_experiences())
27
+ loaded = yaml.safe_load(path.read_text())
28
+
29
+ assert "experiences" in loaded
30
+ exp = loaded["experiences"][0]
31
+ assert exp["org"] == "Acme Corp"
32
+ assert exp["positions"][0]["title"] == "Senior Backend Engineer"
33
+ assert exp["timeline"]["start"] == "Mar 2021"
34
+ # Empty fields (tags, tech, links) are pruned, not emitted as [].
35
+ assert "tags" not in exp
36
+ assert "tech" not in exp
37
+ assert "links" not in exp
@@ -0,0 +1,34 @@
1
+ """Bugs found in a real capture, kept fixed."""
2
+
3
+ from linkfetch.emit import _to_plain
4
+ from linkfetch.models import Course
5
+ from linkfetch.parse import projects
6
+
7
+
8
+ def _page(*entries: str) -> str:
9
+ return (
10
+ "<html><body><main><section aria-label='Primary content'>"
11
+ "<div><p>Projects</p></div><div>" + "".join(entries) + "</div>"
12
+ "</section></main></body></html>"
13
+ )
14
+
15
+
16
+ def test_linkedin_notice_is_not_a_project():
17
+ html = _page(
18
+ "<div><p>Projects from connected apps sync automatically and can’t be edited on LinkedIn.</p>"
19
+ "<span>Learn more</span></div>",
20
+ "<div><p>Example Viewer</p><p>Nov 2025 – Present</p></div>",
21
+ "<div><p>Example CLI</p><p>Jan 2024 – Mar 2024</p></div>",
22
+ )
23
+ assert [p.name for p in projects.parse(html)] == ["Example Viewer", "Example CLI"]
24
+
25
+
26
+ def test_identical_entries_are_written_once_and_ids_stay_unique():
27
+ data = _to_plain([
28
+ Course(id="stats-math-32", name="Probability and Statistics", number="MATH 32"),
29
+ Course(id="stats-math-32", name="Probability and Statistics", number="MATH 32"),
30
+ Course(id="intro", name="Intro", associated_with="School A"),
31
+ Course(id="intro", name="Intro", associated_with="School B"),
32
+ ])
33
+ assert [c["id"] for c in data] == ["stats-math-32", "intro", "intro-2"]
34
+ assert data[2]["associated_with"] == "School B"
@@ -41,11 +41,11 @@ def test_profile_bundle_is_flat_and_has_only_profile_files(tmp_path):
41
41
  assert sorted(archive.namelist()) == ["basics.yaml", "experiences.yaml"]
42
42
 
43
43
 
44
- def test_every_profile_section_maps_to_a_thrive_category_file():
45
- thrive = {
44
+ def test_every_standard_section_goes_into_the_profile_bundle():
45
+ standard = {
46
46
  "basics", "experiences", "skills", "projects", "certifications", "awards",
47
47
  "education", "languages", "courses", "patents", "recommendations", "volunteering",
48
48
  }
49
49
  mapped = {Path(s.out_file).stem for s in SECTIONS if s.mapped}
50
- assert mapped == thrive
50
+ assert mapped == standard
51
51
  assert {s.slug for s in SECTIONS if not s.mapped} == {"honors"}
@@ -138,7 +138,7 @@ wheels = [
138
138
 
139
139
  [[package]]
140
140
  name = "linkfetch"
141
- version = "0.1.2"
141
+ version = "1.0.0"
142
142
  source = { editable = "." }
143
143
  dependencies = [
144
144
  { name = "lxml" },
@@ -1,184 +0,0 @@
1
- # linkfetch — Local LinkedIn Profile Scraper
2
-
3
- **Date:** 2026-06-15
4
- **Status:** Approved design, building
5
-
6
- ## Purpose
7
-
8
- A **local CLI tool** that scrapes *your own* LinkedIn profile via browser
9
- automation and emits **`tailor`-compatible YAML**, so you can bootstrap or
10
- refresh your `tailor` profile from LinkedIn instead of typing it by hand.
11
-
12
- It is a sibling project under `prosperis`, decoupled from `tailor`: it knows
13
- `tailor`'s profile schema and writes matching YAML, but does not import or
14
- modify `tailor`. You review the emitted YAML and copy it into `tailor/data/profile/`.
15
-
16
- This is **not** a web service. Nothing is hosted; it is a command run on your
17
- machine that drives a real browser you log into.
18
-
19
- ## Scope
20
-
21
- ### Sections captured
22
-
23
- LinkedIn exposes far more than `tailor` models. We capture all of the
24
- following. Sections that map onto `tailor`'s schema are emitted as
25
- `tailor`-ready files; the rest are emitted as extra YAML for the user to handle.
26
-
27
- | LinkedIn section | Output file | Maps to tailor? |
28
- |-----------------------------|--------------------------|-----------------|
29
- | Name / headline / about / location | `basics.yaml` | yes (`Basics`) |
30
- | Experience | `experiences.yaml` | yes (`Experience`) |
31
- | Skills | `skills.yaml` | yes (`Skill`) |
32
- | Projects | `projects.yaml` | yes (`Project`) |
33
- | Licenses & certifications | `certifications.yaml` | yes (`Certification`) |
34
- | Honors & awards | `awards.yaml` | yes (`Award`) |
35
- | Education | `education.yaml` | no — extra |
36
- | Volunteering | `volunteering.yaml` | no — extra |
37
- | Recommendations | `recommendations.yaml` | no — extra |
38
- | Patents | `patents.yaml` | no — extra |
39
- | Courses | `courses.yaml` | no — extra |
40
- | Languages | `languages.yaml` | no — extra |
41
-
42
- "Extra" files use a simple, readable schema of our own (defined in this repo),
43
- not `tailor`'s.
44
-
45
- ### Out of scope
46
-
47
- - Scraping other people / companies / job postings (job postings are `tailor`'s job).
48
- - AI / Ollama normalization. **Pure deterministic scrape** — fields come straight
49
- from the page. `summary` is raw text; `bullets` split on the page's own line
50
- breaks; the user enriches tags / position framings later in `tailor`.
51
- - Official-export ZIP parsing.
52
- - Auto-writing into `tailor`'s profile dir (user reviews and copies manually).
53
-
54
- ## Architecture — two-phase: capture HTML → parse offline
55
-
56
- The fragile part (LinkedIn's DOM, login, lazy loading) is **quarantined** in a
57
- capture phase that runs rarely and needs a live session. Parsing is a **pure
58
- function over saved HTML**, fully unit-testable against fixtures and re-runnable
59
- with no browser.
60
-
61
- ```
62
- phase 1: capture (Playwright, live session)
63
- LinkedIn ──► login (persistent context) ──► visit each section's
64
- /details/<section>/ page
65
- ──► expand "see more", scroll
66
- ──► save rendered HTML
67
- to captures/<section>.html
68
- ─────────────────────────────────────────────────────────────
69
- phase 2: parse (pure, no browser)
70
- captures/*.html ──► per-section lxml parser ──► pydantic models
71
- ──► YAML files in out/
72
- ```
73
-
74
- ### Components
75
-
76
- Package `linkfetch` under `src/linkfetch/` (mirrors `tailor`'s `src/` layout).
77
-
78
- - **`config.py`** — `Config` dataclass + `load_config()`, walking up for
79
- `config.toml`, exactly like `tailor`. Paths: `captures_dir` (default
80
- `data/captures`), `output_dir` (default `data/output`). Browser settings:
81
- `user_data_dir` (persistent Chromium profile so login sticks),
82
- `headless` (default false for login), `nav_timeout`.
83
-
84
- - **`capture/browser.py`** — Playwright session management. Launches a
85
- **persistent context** (`launch_persistent_context(user_data_dir=...)`) so the
86
- login cookie survives between runs. Exposes:
87
- - `ensure_logged_in()` — open `linkedin.com/feed`, detect whether logged in;
88
- if not, print instructions and wait for the user to log in manually, then
89
- continue.
90
- - `capture_section(section)` — navigate to the section's detail URL, run the
91
- auto-expand routine, return rendered HTML.
92
-
93
- - **`capture/expand.py`** — the scroll + "…see more" / "Show all" expansion
94
- routine, shared by all sections. Deterministic: scroll to bottom in steps
95
- until height stops changing; click every visible expand control until none
96
- remain. No section-specific logic here.
97
-
98
- - **`capture/sections.py`** — the registry of sections: slug, detail-page URL
99
- template (e.g. `/in/{vanity}/details/experience/`), and which parser handles it.
100
- The single source of truth for "what sections exist".
101
-
102
- - **`parse/` (one module per section)** — `experience.py`, `skills.py`,
103
- `education.py`, etc. Each exposes `parse(html: str) -> list[Model]` (or a single
104
- model for basics). Pure: `lxml` in, pydantic out. LinkedIn's randomized class
105
- names are avoided by anchoring on **stable structures**: the visually-hidden
106
- `<span aria-hidden="true">` text, `aria-label`s, and the section's anchor `id`.
107
-
108
- - **`models.py`** — pydantic models. For mapped sections, fields are a **subset**
109
- of `tailor`'s models using the **same field names** (`id`, `summary`, `bullets`,
110
- `tags`, `org`, `positions`, `timeline`, …) so emitted YAML drops into `tailor`.
111
- For extra sections, small purpose-built models.
112
-
113
- - **`emit.py`** — serialize models to YAML using the same dumper settings as
114
- `tailor`'s `store.py` (`sort_keys=False`, `allow_unicode=True`,
115
- `default_flow_style=False`, `width=100`) so output is human-editable and
116
- diffable, and identical in style to `tailor`'s files.
117
-
118
- - **`cli.py`** — `typer` app mirroring `tailor`'s style. Commands:
119
- - `linkfetch doctor` — show config, whether Playwright is installed, whether a
120
- login session exists.
121
- - `linkfetch login` — open the browser and establish/refresh the session.
122
- - `linkfetch capture [--section X ...] [--vanity NAME]` — phase 1; saves HTML
123
- to `captures/`. Default: all sections.
124
- - `linkfetch parse [--section X ...]` — phase 2; reads `captures/`, writes
125
- YAML to `out/`. No browser needed.
126
- - `linkfetch run` — convenience: `capture` then `parse`.
127
-
128
- ### Data flow
129
-
130
- 1. `login` → persistent Chromium profile stores the LinkedIn cookie under
131
- `user_data_dir`.
132
- 2. `capture` → for each section, navigate to its `/details/<section>/` page
133
- (these are paginated and far simpler than the monolithic profile page),
134
- expand everything, write `captures/<section>.html` + a `captures/meta.json`
135
- (timestamp, vanity, section list).
136
- 3. `parse` → for each captured section, run its parser, collect models, emit
137
- `out/<file>.yaml`. Mapped files are byte-compatible with `tailor`; extra files
138
- use local schemas.
139
- 4. User reviews `out/`, copies the mapped YAML into `tailor/data/profile/`,
140
- runs `tailor profile validate`.
141
-
142
- ### Error handling
143
-
144
- - **Playwright missing** → `capture`/`login` fail with the same install hint
145
- style `tailor` uses (`uv pip install playwright && playwright install chromium`).
146
- `parse` and `doctor` still work without it.
147
- - **Not logged in** → `capture` detects the login wall and tells the user to run
148
- `linkfetch login` (or logs in inline and waits).
149
- - **Section markup not recognized** → parser returns `[]` for that section and
150
- records a warning in the run summary rather than crashing the whole run; the
151
- raw HTML stays on disk so parsing can be retried after a selector fix.
152
- - **Partial captures** → `parse` only processes sections present in `captures/`;
153
- missing ones are skipped with a note.
154
-
155
- ### Testing
156
-
157
- - **Parsers (primary):** unit tests over **saved HTML fixtures** in
158
- `tests/fixtures/<section>.html` (sanitized, anonymized snippets). Each parser
159
- has a test asserting the structured output. This is where correctness lives and
160
- is fully offline / CI-friendly.
161
- - **expand.py:** logic is browser-bound; covered by a thin smoke path, not unit
162
- tested against live LinkedIn.
163
- - **emit.py:** round-trip test — models → YAML → re-load → equal; and a test that
164
- mapped YAML validates against `tailor`'s pydantic models (import `tailor` as a
165
- test-only dependency *if available*, else skip).
166
- - **config.py:** test root discovery and path resolution, mirroring `tailor`.
167
-
168
- TDD: write the failing parser test against a fixture first, then the parser.
169
-
170
- ## Stack & conventions
171
-
172
- - Python ≥ 3.10, `uv`-managed venv, `pyproject.toml` with `hatchling`,
173
- `src/linkfetch/` layout, `typer` + `rich` CLI, `pydantic` v2 models,
174
- `pyyaml`, `lxml`. `playwright` is an **optional** extra (`[browser]`), exactly
175
- like `tailor`. `pytest` + `pytest-mock` dev extra.
176
- - `config.toml` at project root; `data/` gitignored.
177
- - Console-script entry point `linkfetch = "linkfetch.cli:app"`.
178
-
179
- ## Legal / safety note
180
-
181
- For **personal use on the user's own profile**. LinkedIn's ToS restricts
182
- automated access; this tool uses a real logged-in browser session at human-ish
183
- pace, captures only the authenticated user's own data, and stores nothing
184
- remotely. The README states this plainly.
@@ -1,3 +0,0 @@
1
- """linkfetch — local LinkedIn profile scraper emitting tailor-compatible YAML."""
2
-
3
- __version__ = "0.1.0"
@@ -1,78 +0,0 @@
1
- from __future__ import annotations
2
-
3
- import importlib.util
4
- import sys
5
- from pathlib import Path
6
-
7
- import yaml
8
-
9
- from linkfetch.emit import dump_yaml
10
- from linkfetch.models import Experience, Position, Timeline
11
-
12
-
13
- def _sample_experiences() -> list[Experience]:
14
- return [
15
- Experience(
16
- id="acme-corp-senior-backend-engineer",
17
- org="Acme Corp",
18
- location="Berlin, Germany",
19
- positions=[Position(title="Senior Backend Engineer")],
20
- timeline=Timeline(start="Mar 2021", end="Present"),
21
- bullets=["Led the payments platform team.", "Cut p99 latency by 40%."],
22
- )
23
- ]
24
-
25
-
26
- def test_round_trip_and_pruning(tmp_path):
27
- path = tmp_path / "experiences.yaml"
28
- dump_yaml(path, "experiences", _sample_experiences())
29
- loaded = yaml.safe_load(path.read_text())
30
-
31
- assert "experiences" in loaded
32
- exp = loaded["experiences"][0]
33
- assert exp["org"] == "Acme Corp"
34
- assert exp["positions"][0]["title"] == "Senior Backend Engineer"
35
- assert exp["timeline"]["start"] == "Mar 2021"
36
- # Empty fields (tags, tech, links) are pruned, not emitted as [].
37
- assert "tags" not in exp
38
- assert "tech" not in exp
39
- assert "links" not in exp
40
-
41
-
42
- _TAILOR_MODELS = (
43
- Path(__file__).resolve().parents[2]
44
- / "tailor"
45
- / "src"
46
- / "tailor"
47
- / "profile"
48
- / "models.py"
49
- )
50
-
51
-
52
- def _load_tailor_models():
53
- """Import tailor's models module directly from the sibling project, if present."""
54
- if not _TAILOR_MODELS.is_file():
55
- return None
56
- spec = importlib.util.spec_from_file_location("tailor_profile_models", _TAILOR_MODELS)
57
- module = importlib.util.module_from_spec(spec)
58
- sys.modules[spec.name] = module
59
- spec.loader.exec_module(module)
60
- return module
61
-
62
-
63
- def test_emitted_yaml_validates_against_tailor_schema(tmp_path):
64
- """The mapped experiences.yaml must load into tailor's own pydantic model."""
65
- tailor = _load_tailor_models()
66
- if tailor is None:
67
- import pytest
68
-
69
- pytest.skip("tailor project not available next to linkfetch")
70
-
71
- path = tmp_path / "experiences.yaml"
72
- dump_yaml(path, "experiences", _sample_experiences())
73
- raw = yaml.safe_load(path.read_text())
74
-
75
- # Should validate without error.
76
- wrapper = tailor.ExperiencesFile.model_validate(raw)
77
- assert wrapper.experiences[0].org == "Acme Corp"
78
- assert wrapper.experiences[0].positions[0].title == "Senior Backend Engineer"
File without changes
File without changes
File without changes
File without changes
File without changes