linkfetch 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. linkfetch-0.1.0/.github/workflows/release.yml +63 -0
  2. linkfetch-0.1.0/.github/workflows/test.yml +30 -0
  3. linkfetch-0.1.0/.gitignore +14 -0
  4. linkfetch-0.1.0/LICENSE +21 -0
  5. linkfetch-0.1.0/PKG-INFO +174 -0
  6. linkfetch-0.1.0/README.md +140 -0
  7. linkfetch-0.1.0/config.toml +16 -0
  8. linkfetch-0.1.0/docs/superpowers/specs/2026-06-15-linkfetch-design.md +184 -0
  9. linkfetch-0.1.0/pyproject.toml +54 -0
  10. linkfetch-0.1.0/src/linkfetch/__init__.py +3 -0
  11. linkfetch-0.1.0/src/linkfetch/capture/__init__.py +1 -0
  12. linkfetch-0.1.0/src/linkfetch/capture/browser.py +112 -0
  13. linkfetch-0.1.0/src/linkfetch/capture/expand.py +61 -0
  14. linkfetch-0.1.0/src/linkfetch/cli.py +274 -0
  15. linkfetch-0.1.0/src/linkfetch/config.py +105 -0
  16. linkfetch-0.1.0/src/linkfetch/emit.py +57 -0
  17. linkfetch-0.1.0/src/linkfetch/models.py +196 -0
  18. linkfetch-0.1.0/src/linkfetch/parse/__init__.py +43 -0
  19. linkfetch-0.1.0/src/linkfetch/parse/awards.py +57 -0
  20. linkfetch-0.1.0/src/linkfetch/parse/basics.py +89 -0
  21. linkfetch-0.1.0/src/linkfetch/parse/certifications.py +70 -0
  22. linkfetch-0.1.0/src/linkfetch/parse/common.py +251 -0
  23. linkfetch-0.1.0/src/linkfetch/parse/courses.py +38 -0
  24. linkfetch-0.1.0/src/linkfetch/parse/education.py +51 -0
  25. linkfetch-0.1.0/src/linkfetch/parse/experience.py +129 -0
  26. linkfetch-0.1.0/src/linkfetch/parse/honors.py +32 -0
  27. linkfetch-0.1.0/src/linkfetch/parse/languages.py +23 -0
  28. linkfetch-0.1.0/src/linkfetch/parse/patents.py +60 -0
  29. linkfetch-0.1.0/src/linkfetch/parse/projects.py +67 -0
  30. linkfetch-0.1.0/src/linkfetch/parse/recommendations.py +64 -0
  31. linkfetch-0.1.0/src/linkfetch/parse/skills.py +30 -0
  32. linkfetch-0.1.0/src/linkfetch/parse/volunteering.py +47 -0
  33. linkfetch-0.1.0/src/linkfetch/sections.py +84 -0
  34. linkfetch-0.1.0/src/linkfetch/text.py +78 -0
  35. linkfetch-0.1.0/tests/conftest.py +19 -0
  36. linkfetch-0.1.0/tests/fixtures/basics.html +38 -0
  37. linkfetch-0.1.0/tests/fixtures/certifications.html +26 -0
  38. linkfetch-0.1.0/tests/fixtures/education.html +25 -0
  39. linkfetch-0.1.0/tests/fixtures/experience.html +67 -0
  40. linkfetch-0.1.0/tests/fixtures/honors.html +23 -0
  41. linkfetch-0.1.0/tests/fixtures/languages.html +11 -0
  42. linkfetch-0.1.0/tests/fixtures/patents.html +18 -0
  43. linkfetch-0.1.0/tests/fixtures/projects.html +23 -0
  44. linkfetch-0.1.0/tests/fixtures/recommendations.html +30 -0
  45. linkfetch-0.1.0/tests/fixtures/skills.html +15 -0
  46. linkfetch-0.1.0/tests/fixtures/volunteering.html +23 -0
  47. linkfetch-0.1.0/tests/test_config.py +26 -0
  48. linkfetch-0.1.0/tests/test_emit.py +78 -0
  49. linkfetch-0.1.0/tests/test_experience.py +65 -0
  50. linkfetch-0.1.0/tests/test_release.py +51 -0
  51. linkfetch-0.1.0/tests/test_sections.py +198 -0
  52. linkfetch-0.1.0/tests/test_text.py +70 -0
  53. linkfetch-0.1.0/uv.lock +705 -0
@@ -0,0 +1,63 @@
1
+ name: Release
2
+
3
+ # Push a tag like v0.1.1 to publish that version to PyPI. Publishing uses
4
+ # PyPI trusted publishing (OpenID Connect): no API token is stored anywhere.
5
+ on:
6
+ push:
7
+ tags: ["v*"]
8
+
9
+ permissions:
10
+ contents: read
11
+
12
+ jobs:
13
+ build:
14
+ runs-on: ubuntu-latest
15
+ steps:
16
+ - uses: actions/checkout@v4
17
+ - uses: astral-sh/setup-uv@v6
18
+ with:
19
+ python-version: "3.13"
20
+ - name: Tag matches the package version
21
+ run: |
22
+ version=$(uv version --short)
23
+ if [ "v$version" != "$GITHUB_REF_NAME" ]; then
24
+ echo "Tag $GITHUB_REF_NAME does not match pyproject version $version" >&2
25
+ exit 1
26
+ fi
27
+ - run: uv sync --extra dev --locked
28
+ - run: uv run pytest -q
29
+ - run: uv build
30
+ - uses: actions/upload-artifact@v4
31
+ with:
32
+ name: dist
33
+ path: dist/
34
+
35
+ publish:
36
+ needs: build
37
+ runs-on: ubuntu-latest
38
+ environment:
39
+ name: pypi
40
+ url: https://pypi.org/project/linkfetch/
41
+ permissions:
42
+ id-token: write
43
+ steps:
44
+ - uses: actions/download-artifact@v4
45
+ with:
46
+ name: dist
47
+ path: dist/
48
+ - uses: pypa/gh-action-pypi-publish@release/v1
49
+
50
+ github-release:
51
+ needs: publish
52
+ runs-on: ubuntu-latest
53
+ permissions:
54
+ contents: write
55
+ steps:
56
+ - uses: actions/download-artifact@v4
57
+ with:
58
+ name: dist
59
+ path: dist/
60
+ - name: Create the GitHub release
61
+ env:
62
+ GH_TOKEN: ${{ github.token }}
63
+ run: gh release create "$GITHUB_REF_NAME" dist/* --repo "$GITHUB_REPOSITORY" --title "$GITHUB_REF_NAME" --generate-notes
@@ -0,0 +1,30 @@
1
+ name: Test
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+
8
+ permissions:
9
+ contents: read
10
+
11
+ jobs:
12
+ test:
13
+ name: Python ${{ matrix.python }} on ${{ matrix.os }}
14
+ runs-on: ${{ matrix.os }}
15
+ strategy:
16
+ fail-fast: false
17
+ matrix:
18
+ os: [ubuntu-latest, windows-latest, macos-latest]
19
+ python: ["3.10", "3.13"]
20
+ steps:
21
+ - uses: actions/checkout@v4
22
+ - uses: astral-sh/setup-uv@v6
23
+ with:
24
+ python-version: ${{ matrix.python }}
25
+ # Tests exercise parsing and config only, so the Chromium download that
26
+ # `linkfetch capture` needs is skipped.
27
+ - run: uv sync --extra dev --locked
28
+ - run: uv run pytest -q
29
+ - name: Build the package
30
+ run: uv build
@@ -0,0 +1,14 @@
1
+ # data: captured HTML, emitted YAML, browser profile
2
+ /data/
3
+
4
+ # python
5
+ __pycache__/
6
+ *.py[cod]
7
+ .venv/
8
+ *.egg-info/
9
+ .pytest_cache/
10
+ dist/
11
+ build/
12
+
13
+ # os
14
+ .DS_Store
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Prosperis
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,174 @@
1
+ Metadata-Version: 2.5
2
+ Name: linkfetch
3
+ Version: 0.1.0
4
+ Summary: Capture your own LinkedIn profile on your own computer and turn it into structured profile files
5
+ Project-URL: Homepage, https://github.com/Prosperis/linkfetch
6
+ Project-URL: Issues, https://github.com/Prosperis/linkfetch/issues
7
+ Author: Prosperis
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: cv,export,linkedin,playwright,profile,resume
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Environment :: Console
13
+ Classifier: Intended Audience :: End Users/Desktop
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Topic :: Office/Business
21
+ Requires-Python: >=3.10
22
+ Requires-Dist: lxml>=5.0
23
+ Requires-Dist: playwright>=1.44
24
+ Requires-Dist: pydantic>=2.6
25
+ Requires-Dist: pyyaml>=6.0
26
+ Requires-Dist: rich>=13.7
27
+ Requires-Dist: tomli>=2.0; python_version < '3.11'
28
+ Requires-Dist: typer>=0.12
29
+ Provides-Extra: browser
30
+ Provides-Extra: dev
31
+ Requires-Dist: pytest-mock>=3.12; extra == 'dev'
32
+ Requires-Dist: pytest>=8.0; extra == 'dev'
33
+ Description-Content-Type: text/markdown
34
+
35
+ # linkfetch
36
+
37
+ Capture **your own** LinkedIn profile on **your own computer** and turn it into
38
+ structured profile files — every role, description, skill, project,
39
+ certification and more — ready to import into
40
+ [THRIVE](https://prosperis-thrive.vercel.app) or any tool that reads YAML.
41
+
42
+ LinkedIn's official data export leaves out details the profile page shows.
43
+ linkfetch reads the rendered pages instead, in a real browser window that you
44
+ log into yourself.
45
+
46
+ ## What it does and does not do
47
+
48
+ - **Runs only on your machine.** linkfetch is not a website or a service. It
49
+ sends your data nowhere; the files it writes stay in a folder on your
50
+ computer until you choose to upload them.
51
+ - **Reads only your own profile**, in a browser you log into. It never sees
52
+ your password: you type it into LinkedIn's own login page.
53
+ - **Deterministic, no AI.** Every field comes straight from the page.
54
+ - **Open source** (MIT), so you can read exactly what it does.
55
+
56
+ ## Install
57
+
58
+ You need Python 3.10 or newer and [uv](https://docs.astral.sh/uv/) (or
59
+ [pipx](https://pipx.pypa.io/)).
60
+
61
+ ```bash
62
+ uv tool install linkfetch # or: pipx install linkfetch
63
+ linkfetch doctor # shows where data is kept and whether Playwright and a login exist
64
+ playwright install chromium # one-time download of the browser linkfetch drives
65
+ ```
66
+
67
+ ## Use
68
+
69
+ ```bash
70
+ linkfetch login # a browser window opens: log in to LinkedIn, then close it
71
+ linkfetch run --vanity your-name # capture your profile and build the files
72
+ ```
73
+
74
+ `your-name` is the part after `/in/` in your profile address:
75
+ `linkedin.com/in/your-name`.
76
+
77
+ `run` visits each section of your profile (experience, education, skills, …),
78
+ expands every "see more", saves the pages, and turns them into one YAML file
79
+ per section plus **`linkfetch-profile.zip`**. `linkfetch` prints where they
80
+ are.
81
+
82
+ ### Import into THRIVE
83
+
84
+ In THRIVE, open **Tailor → Profile → Import** and choose
85
+ `linkfetch-profile.zip`. Review each section before saving.
86
+
87
+ ## Where your data is kept
88
+
89
+ Installed with `uv tool` or `pipx`, linkfetch keeps everything in your user
90
+ data folder:
91
+
92
+ | OS | Folder |
93
+ |---|---|
94
+ | Windows | `%LOCALAPPDATA%\linkfetch` |
95
+ | macOS | `~/Library/Application Support/linkfetch` |
96
+ | Linux | `$XDG_DATA_HOME/linkfetch` (usually `~/.local/share/linkfetch`) |
97
+
98
+ Set `LINKFETCH_HOME` to use a different folder. Inside it:
99
+
100
+ - `data/browser/` — the browser profile that keeps you logged in to LinkedIn.
101
+ Treat it like a password; delete it to log out.
102
+ - `data/captures/` — the saved profile pages.
103
+ - `data/output/` — the YAML files and `linkfetch-profile.zip`.
104
+
105
+ Delete the folder to remove everything linkfetch stored.
106
+
107
+ ## Commands
108
+
109
+ | Command | What it does |
110
+ |---|---|
111
+ | `linkfetch doctor` | Show the data folders and check Playwright and your login |
112
+ | `linkfetch login` | Open a browser window to log in to LinkedIn once |
113
+ | `linkfetch capture --vanity <you>` | Save your profile pages (needs the browser) |
114
+ | `linkfetch parse` | Turn saved pages into YAML and the ZIP (no browser; re-runnable) |
115
+ | `linkfetch run --vanity <you>` | `capture` then `parse` |
116
+
117
+ Use `--section experience --section skills` to limit a run to some sections,
118
+ and `parse --no-zip` to skip the ZIP.
119
+
120
+ ## Output
121
+
122
+ One file per section, each a list under a top-level key:
123
+ `basics.yaml`, `experiences.yaml`, `education.yaml`, `skills.yaml`,
124
+ `projects.yaml`, `certifications.yaml`, `awards.yaml`, `languages.yaml`,
125
+ `courses.yaml`, `patents.yaml`, `recommendations.yaml`, `volunteering.yaml`,
126
+ plus `honors.yaml` (a richer copy of awards, not included in the ZIP).
127
+ Experiences keep their full description as a summary plus bullets, and roles
128
+ at the same company stay grouped.
129
+
130
+ ## How it works
131
+
132
+ 1. **Capture** (needs the browser, runs rarely): a persistent Chromium profile
133
+ you logged into visits each section's details page
134
+ (`/in/<you>/details/experience/`, …), expands collapsed content, and saves
135
+ the rendered HTML.
136
+ 2. **Parse** (pure, offline): reads the saved HTML with `lxml` and writes YAML.
137
+ If LinkedIn changes its pages, a parser fix plus `linkfetch parse` is enough
138
+ — no need to capture again.
139
+
140
+ ## Development
141
+
142
+ ```bash
143
+ git clone https://github.com/Prosperis/linkfetch
144
+ cd linkfetch
145
+ uv venv && uv pip install -e ".[dev]"
146
+ playwright install chromium
147
+ uv run pytest
148
+ ```
149
+
150
+ A checkout keeps its data in `data/` next to `config.toml`, which also holds
151
+ browser settings (headless mode, timeouts, scroll pacing). Test fixtures are
152
+ anonymized excerpts of LinkedIn's page structure.
153
+
154
+ ## Releasing
155
+
156
+ Bump `version` in `pyproject.toml`, commit, then push a matching tag:
157
+
158
+ ```bash
159
+ git tag v0.1.1 && git push origin v0.1.1
160
+ ```
161
+
162
+ The Release workflow tests, builds and publishes to PyPI through trusted
163
+ publishing (no stored token), then creates the GitHub release.
164
+
165
+ ## Legal note
166
+
167
+ LinkedIn's User Agreement restricts automated access. linkfetch is meant for
168
+ **personal use on your own profile**: it uses a browser you logged into, moves
169
+ at a human pace, reads only your own data and stores nothing remotely. Use it
170
+ on your own account and at your own discretion.
171
+
172
+ ## License
173
+
174
+ [MIT](LICENSE) © Prosperis
@@ -0,0 +1,140 @@
1
+ # linkfetch
2
+
3
+ Capture **your own** LinkedIn profile on **your own computer** and turn it into
4
+ structured profile files — every role, description, skill, project,
5
+ certification and more — ready to import into
6
+ [THRIVE](https://prosperis-thrive.vercel.app) or any tool that reads YAML.
7
+
8
+ LinkedIn's official data export leaves out details the profile page shows.
9
+ linkfetch reads the rendered pages instead, in a real browser window that you
10
+ log into yourself.
11
+
12
+ ## What it does and does not do
13
+
14
+ - **Runs only on your machine.** linkfetch is not a website or a service. It
15
+ sends your data nowhere; the files it writes stay in a folder on your
16
+ computer until you choose to upload them.
17
+ - **Reads only your own profile**, in a browser you log into. It never sees
18
+ your password: you type it into LinkedIn's own login page.
19
+ - **Deterministic, no AI.** Every field comes straight from the page.
20
+ - **Open source** (MIT), so you can read exactly what it does.
21
+
22
+ ## Install
23
+
24
+ You need Python 3.10 or newer and [uv](https://docs.astral.sh/uv/) (or
25
+ [pipx](https://pipx.pypa.io/)).
26
+
27
+ ```bash
28
+ uv tool install linkfetch # or: pipx install linkfetch
29
+ linkfetch doctor # shows where data is kept and whether Playwright and a login exist
30
+ playwright install chromium # one-time download of the browser linkfetch drives
31
+ ```
32
+
33
+ ## Use
34
+
35
+ ```bash
36
+ linkfetch login # a browser window opens: log in to LinkedIn, then close it
37
+ linkfetch run --vanity your-name # capture your profile and build the files
38
+ ```
39
+
40
+ `your-name` is the part after `/in/` in your profile address:
41
+ `linkedin.com/in/your-name`.
42
+
43
+ `run` visits each section of your profile (experience, education, skills, …),
44
+ expands every "see more", saves the pages, and turns them into one YAML file
45
+ per section plus **`linkfetch-profile.zip`**. `linkfetch` prints where they
46
+ are.
47
+
48
+ ### Import into THRIVE
49
+
50
+ In THRIVE, open **Tailor → Profile → Import** and choose
51
+ `linkfetch-profile.zip`. Review each section before saving.
52
+
53
+ ## Where your data is kept
54
+
55
+ Installed with `uv tool` or `pipx`, linkfetch keeps everything in your user
56
+ data folder:
57
+
58
+ | OS | Folder |
59
+ |---|---|
60
+ | Windows | `%LOCALAPPDATA%\linkfetch` |
61
+ | macOS | `~/Library/Application Support/linkfetch` |
62
+ | Linux | `$XDG_DATA_HOME/linkfetch` (usually `~/.local/share/linkfetch`) |
63
+
64
+ Set `LINKFETCH_HOME` to use a different folder. Inside it:
65
+
66
+ - `data/browser/` — the browser profile that keeps you logged in to LinkedIn.
67
+ Treat it like a password; delete it to log out.
68
+ - `data/captures/` — the saved profile pages.
69
+ - `data/output/` — the YAML files and `linkfetch-profile.zip`.
70
+
71
+ Delete the folder to remove everything linkfetch stored.
72
+
73
+ ## Commands
74
+
75
+ | Command | What it does |
76
+ |---|---|
77
+ | `linkfetch doctor` | Show the data folders and check Playwright and your login |
78
+ | `linkfetch login` | Open a browser window to log in to LinkedIn once |
79
+ | `linkfetch capture --vanity <you>` | Save your profile pages (needs the browser) |
80
+ | `linkfetch parse` | Turn saved pages into YAML and the ZIP (no browser; re-runnable) |
81
+ | `linkfetch run --vanity <you>` | `capture` then `parse` |
82
+
83
+ Use `--section experience --section skills` to limit a run to some sections,
84
+ and `parse --no-zip` to skip the ZIP.
85
+
86
+ ## Output
87
+
88
+ One file per section, each a list under a top-level key:
89
+ `basics.yaml`, `experiences.yaml`, `education.yaml`, `skills.yaml`,
90
+ `projects.yaml`, `certifications.yaml`, `awards.yaml`, `languages.yaml`,
91
+ `courses.yaml`, `patents.yaml`, `recommendations.yaml`, `volunteering.yaml`,
92
+ plus `honors.yaml` (a richer copy of awards, not included in the ZIP).
93
+ Experiences keep their full description as a summary plus bullets, and roles
94
+ at the same company stay grouped.
95
+
96
+ ## How it works
97
+
98
+ 1. **Capture** (needs the browser, runs rarely): a persistent Chromium profile
99
+ you logged into visits each section's details page
100
+ (`/in/<you>/details/experience/`, …), expands collapsed content, and saves
101
+ the rendered HTML.
102
+ 2. **Parse** (pure, offline): reads the saved HTML with `lxml` and writes YAML.
103
+ If LinkedIn changes its pages, a parser fix plus `linkfetch parse` is enough
104
+ — no need to capture again.
105
+
106
+ ## Development
107
+
108
+ ```bash
109
+ git clone https://github.com/Prosperis/linkfetch
110
+ cd linkfetch
111
+ uv venv && uv pip install -e ".[dev]"
112
+ playwright install chromium
113
+ uv run pytest
114
+ ```
115
+
116
+ A checkout keeps its data in `data/` next to `config.toml`, which also holds
117
+ browser settings (headless mode, timeouts, scroll pacing). Test fixtures are
118
+ anonymized excerpts of LinkedIn's page structure.
119
+
120
+ ## Releasing
121
+
122
+ Bump `version` in `pyproject.toml`, commit, then push a matching tag:
123
+
124
+ ```bash
125
+ git tag v0.1.1 && git push origin v0.1.1
126
+ ```
127
+
128
+ The Release workflow tests, builds and publishes to PyPI through trusted
129
+ publishing (no stored token), then creates the GitHub release.
130
+
131
+ ## Legal note
132
+
133
+ LinkedIn's User Agreement restricts automated access. linkfetch is meant for
134
+ **personal use on your own profile**: it uses a browser you logged into, moves
135
+ at a human pace, reads only your own data and stores nothing remotely. Use it
136
+ on your own account and at your own discretion.
137
+
138
+ ## License
139
+
140
+ [MIT](LICENSE) © Prosperis
@@ -0,0 +1,16 @@
1
+ # linkfetch configuration.
2
+ # Paths are relative to this file's directory unless absolute.
3
+
4
+ [paths]
5
+ captures = "data/captures" # rendered HTML saved by `linkfetch capture`
6
+ output = "data/output" # YAML emitted by `linkfetch parse`
7
+ user_data_dir = "data/browser" # persistent Chromium profile (keeps you logged in)
8
+
9
+ [browser]
10
+ headless = false # show the window (needed for manual login)
11
+ nav_timeout = 30 # seconds per navigation
12
+ scroll_pause_ms = 600 # pause between scroll steps while loading lazy content
13
+ max_scrolls = 40 # safety cap on scroll-to-bottom loops
14
+
15
+ [linkedin]
16
+ base_url = "https://www.linkedin.com"
@@ -0,0 +1,184 @@
1
+ # linkfetch — Local LinkedIn Profile Scraper
2
+
3
+ **Date:** 2026-06-15
4
+ **Status:** Approved design, building
5
+
6
+ ## Purpose
7
+
8
+ A **local CLI tool** that scrapes *your own* LinkedIn profile via browser
9
+ automation and emits **`tailor`-compatible YAML**, so you can bootstrap or
10
+ refresh your `tailor` profile from LinkedIn instead of typing it by hand.
11
+
12
+ It is a sibling project under `prosperis`, decoupled from `tailor`: it knows
13
+ `tailor`'s profile schema and writes matching YAML, but does not import or
14
+ modify `tailor`. You review the emitted YAML and copy it into `tailor/data/profile/`.
15
+
16
+ This is **not** a web service. Nothing is hosted; it is a command run on your
17
+ machine that drives a real browser you log into.
18
+
19
+ ## Scope
20
+
21
+ ### Sections captured
22
+
23
+ LinkedIn exposes far more than `tailor` models. We capture all of the
24
+ following. Sections that map onto `tailor`'s schema are emitted as
25
+ `tailor`-ready files; the rest are emitted as extra YAML for the user to handle.
26
+
27
+ | LinkedIn section | Output file | Maps to tailor? |
28
+ |-----------------------------|--------------------------|-----------------|
29
+ | Name / headline / about / location | `basics.yaml` | yes (`Basics`) |
30
+ | Experience | `experiences.yaml` | yes (`Experience`) |
31
+ | Skills | `skills.yaml` | yes (`Skill`) |
32
+ | Projects | `projects.yaml` | yes (`Project`) |
33
+ | Licenses & certifications | `certifications.yaml` | yes (`Certification`) |
34
+ | Honors & awards | `awards.yaml` | yes (`Award`) |
35
+ | Education | `education.yaml` | no — extra |
36
+ | Volunteering | `volunteering.yaml` | no — extra |
37
+ | Recommendations | `recommendations.yaml` | no — extra |
38
+ | Patents | `patents.yaml` | no — extra |
39
+ | Courses | `courses.yaml` | no — extra |
40
+ | Languages | `languages.yaml` | no — extra |
41
+
42
+ "Extra" files use a simple, readable schema of our own (defined in this repo),
43
+ not `tailor`'s.
44
+
45
+ ### Out of scope
46
+
47
+ - Scraping other people / companies / job postings (job postings are `tailor`'s job).
48
+ - AI / Ollama normalization. **Pure deterministic scrape** — fields come straight
49
+ from the page. `summary` is raw text; `bullets` split on the page's own line
50
+ breaks; the user enriches tags / position framings later in `tailor`.
51
+ - Official-export ZIP parsing.
52
+ - Auto-writing into `tailor`'s profile dir (user reviews and copies manually).
53
+
54
+ ## Architecture — two-phase: capture HTML → parse offline
55
+
56
+ The fragile part (LinkedIn's DOM, login, lazy loading) is **quarantined** in a
57
+ capture phase that runs rarely and needs a live session. Parsing is a **pure
58
+ function over saved HTML**, fully unit-testable against fixtures and re-runnable
59
+ with no browser.
60
+
61
+ ```
62
+ phase 1: capture (Playwright, live session)
63
+ LinkedIn ──► login (persistent context) ──► visit each section's
64
+ /details/<section>/ page
65
+ ──► expand "see more", scroll
66
+ ──► save rendered HTML
67
+ to captures/<section>.html
68
+ ─────────────────────────────────────────────────────────────
69
+ phase 2: parse (pure, no browser)
70
+ captures/*.html ──► per-section lxml parser ──► pydantic models
71
+ ──► YAML files in out/
72
+ ```
73
+
74
+ ### Components
75
+
76
+ Package `linkfetch` under `src/linkfetch/` (mirrors `tailor`'s `src/` layout).
77
+
78
+ - **`config.py`** — `Config` dataclass + `load_config()`, walking up for
79
+ `config.toml`, exactly like `tailor`. Paths: `captures_dir` (default
80
+ `data/captures`), `output_dir` (default `data/output`). Browser settings:
81
+ `user_data_dir` (persistent Chromium profile so login sticks),
82
+ `headless` (default false for login), `nav_timeout`.
83
+
84
+ - **`capture/browser.py`** — Playwright session management. Launches a
85
+ **persistent context** (`launch_persistent_context(user_data_dir=...)`) so the
86
+ login cookie survives between runs. Exposes:
87
+ - `ensure_logged_in()` — open `linkedin.com/feed`, detect whether logged in;
88
+ if not, print instructions and wait for the user to log in manually, then
89
+ continue.
90
+ - `capture_section(section)` — navigate to the section's detail URL, run the
91
+ auto-expand routine, return rendered HTML.
92
+
93
+ - **`capture/expand.py`** — the scroll + "…see more" / "Show all" expansion
94
+ routine, shared by all sections. Deterministic: scroll to bottom in steps
95
+ until height stops changing; click every visible expand control until none
96
+ remain. No section-specific logic here.
97
+
98
+ - **`capture/sections.py`** — the registry of sections: slug, detail-page URL
99
+ template (e.g. `/in/{vanity}/details/experience/`), and which parser handles it.
100
+ The single source of truth for "what sections exist".
101
+
102
+ - **`parse/` (one module per section)** — `experience.py`, `skills.py`,
103
+ `education.py`, etc. Each exposes `parse(html: str) -> list[Model]` (or a single
104
+ model for basics). Pure: `lxml` in, pydantic out. LinkedIn's randomized class
105
+ names are avoided by anchoring on **stable structures**: the visually-hidden
106
+ `<span aria-hidden="true">` text, `aria-label`s, and the section's anchor `id`.
107
+
108
+ - **`models.py`** — pydantic models. For mapped sections, fields are a **subset**
109
+ of `tailor`'s models using the **same field names** (`id`, `summary`, `bullets`,
110
+ `tags`, `org`, `positions`, `timeline`, …) so emitted YAML drops into `tailor`.
111
+ For extra sections, small purpose-built models.
112
+
113
+ - **`emit.py`** — serialize models to YAML using the same dumper settings as
114
+ `tailor`'s `store.py` (`sort_keys=False`, `allow_unicode=True`,
115
+ `default_flow_style=False`, `width=100`) so output is human-editable and
116
+ diffable, and identical in style to `tailor`'s files.
117
+
118
+ - **`cli.py`** — `typer` app mirroring `tailor`'s style. Commands:
119
+ - `linkfetch doctor` — show config, whether Playwright is installed, whether a
120
+ login session exists.
121
+ - `linkfetch login` — open the browser and establish/refresh the session.
122
+ - `linkfetch capture [--section X ...] [--vanity NAME]` — phase 1; saves HTML
123
+ to `captures/`. Default: all sections.
124
+ - `linkfetch parse [--section X ...]` — phase 2; reads `captures/`, writes
125
+ YAML to `out/`. No browser needed.
126
+ - `linkfetch run` — convenience: `capture` then `parse`.
127
+
128
+ ### Data flow
129
+
130
+ 1. `login` → persistent Chromium profile stores the LinkedIn cookie under
131
+ `user_data_dir`.
132
+ 2. `capture` → for each section, navigate to its `/details/<section>/` page
133
+ (these are paginated and far simpler than the monolithic profile page),
134
+ expand everything, write `captures/<section>.html` + a `captures/meta.json`
135
+ (timestamp, vanity, section list).
136
+ 3. `parse` → for each captured section, run its parser, collect models, emit
137
+ `out/<file>.yaml`. Mapped files are byte-compatible with `tailor`; extra files
138
+ use local schemas.
139
+ 4. User reviews `out/`, copies the mapped YAML into `tailor/data/profile/`,
140
+ runs `tailor profile validate`.
141
+
142
+ ### Error handling
143
+
144
+ - **Playwright missing** → `capture`/`login` fail with the same install hint
145
+ style `tailor` uses (`uv pip install playwright && playwright install chromium`).
146
+ `parse` and `doctor` still work without it.
147
+ - **Not logged in** → `capture` detects the login wall and tells the user to run
148
+ `linkfetch login` (or logs in inline and waits).
149
+ - **Section markup not recognized** → parser returns `[]` for that section and
150
+ records a warning in the run summary rather than crashing the whole run; the
151
+ raw HTML stays on disk so parsing can be retried after a selector fix.
152
+ - **Partial captures** → `parse` only processes sections present in `captures/`;
153
+ missing ones are skipped with a note.
154
+
155
+ ### Testing
156
+
157
+ - **Parsers (primary):** unit tests over **saved HTML fixtures** in
158
+ `tests/fixtures/<section>.html` (sanitized, anonymized snippets). Each parser
159
+ has a test asserting the structured output. This is where correctness lives and
160
+ is fully offline / CI-friendly.
161
+ - **expand.py:** logic is browser-bound; covered by a thin smoke path, not unit
162
+ tested against live LinkedIn.
163
+ - **emit.py:** round-trip test — models → YAML → re-load → equal; and a test that
164
+ mapped YAML validates against `tailor`'s pydantic models (import `tailor` as a
165
+ test-only dependency *if available*, else skip).
166
+ - **config.py:** test root discovery and path resolution, mirroring `tailor`.
167
+
168
+ TDD: write the failing parser test against a fixture first, then the parser.
169
+
170
+ ## Stack & conventions
171
+
172
+ - Python ≥ 3.10, `uv`-managed venv, `pyproject.toml` with `hatchling`,
173
+ `src/linkfetch/` layout, `typer` + `rich` CLI, `pydantic` v2 models,
174
+ `pyyaml`, `lxml`. `playwright` is an **optional** extra (`[browser]`), exactly
175
+ like `tailor`. `pytest` + `pytest-mock` dev extra.
176
+ - `config.toml` at project root; `data/` gitignored.
177
+ - Console-script entry point `linkfetch = "linkfetch.cli:app"`.
178
+
179
+ ## Legal / safety note
180
+
181
+ For **personal use on the user's own profile**. LinkedIn's ToS restricts
182
+ automated access; this tool uses a real logged-in browser session at human-ish
183
+ pace, captures only the authenticated user's own data, and stores nothing
184
+ remotely. The README states this plainly.