linkfetch 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- linkfetch-0.1.0/.github/workflows/release.yml +63 -0
- linkfetch-0.1.0/.github/workflows/test.yml +30 -0
- linkfetch-0.1.0/.gitignore +14 -0
- linkfetch-0.1.0/LICENSE +21 -0
- linkfetch-0.1.0/PKG-INFO +174 -0
- linkfetch-0.1.0/README.md +140 -0
- linkfetch-0.1.0/config.toml +16 -0
- linkfetch-0.1.0/docs/superpowers/specs/2026-06-15-linkfetch-design.md +184 -0
- linkfetch-0.1.0/pyproject.toml +54 -0
- linkfetch-0.1.0/src/linkfetch/__init__.py +3 -0
- linkfetch-0.1.0/src/linkfetch/capture/__init__.py +1 -0
- linkfetch-0.1.0/src/linkfetch/capture/browser.py +112 -0
- linkfetch-0.1.0/src/linkfetch/capture/expand.py +61 -0
- linkfetch-0.1.0/src/linkfetch/cli.py +274 -0
- linkfetch-0.1.0/src/linkfetch/config.py +105 -0
- linkfetch-0.1.0/src/linkfetch/emit.py +57 -0
- linkfetch-0.1.0/src/linkfetch/models.py +196 -0
- linkfetch-0.1.0/src/linkfetch/parse/__init__.py +43 -0
- linkfetch-0.1.0/src/linkfetch/parse/awards.py +57 -0
- linkfetch-0.1.0/src/linkfetch/parse/basics.py +89 -0
- linkfetch-0.1.0/src/linkfetch/parse/certifications.py +70 -0
- linkfetch-0.1.0/src/linkfetch/parse/common.py +251 -0
- linkfetch-0.1.0/src/linkfetch/parse/courses.py +38 -0
- linkfetch-0.1.0/src/linkfetch/parse/education.py +51 -0
- linkfetch-0.1.0/src/linkfetch/parse/experience.py +129 -0
- linkfetch-0.1.0/src/linkfetch/parse/honors.py +32 -0
- linkfetch-0.1.0/src/linkfetch/parse/languages.py +23 -0
- linkfetch-0.1.0/src/linkfetch/parse/patents.py +60 -0
- linkfetch-0.1.0/src/linkfetch/parse/projects.py +67 -0
- linkfetch-0.1.0/src/linkfetch/parse/recommendations.py +64 -0
- linkfetch-0.1.0/src/linkfetch/parse/skills.py +30 -0
- linkfetch-0.1.0/src/linkfetch/parse/volunteering.py +47 -0
- linkfetch-0.1.0/src/linkfetch/sections.py +84 -0
- linkfetch-0.1.0/src/linkfetch/text.py +78 -0
- linkfetch-0.1.0/tests/conftest.py +19 -0
- linkfetch-0.1.0/tests/fixtures/basics.html +38 -0
- linkfetch-0.1.0/tests/fixtures/certifications.html +26 -0
- linkfetch-0.1.0/tests/fixtures/education.html +25 -0
- linkfetch-0.1.0/tests/fixtures/experience.html +67 -0
- linkfetch-0.1.0/tests/fixtures/honors.html +23 -0
- linkfetch-0.1.0/tests/fixtures/languages.html +11 -0
- linkfetch-0.1.0/tests/fixtures/patents.html +18 -0
- linkfetch-0.1.0/tests/fixtures/projects.html +23 -0
- linkfetch-0.1.0/tests/fixtures/recommendations.html +30 -0
- linkfetch-0.1.0/tests/fixtures/skills.html +15 -0
- linkfetch-0.1.0/tests/fixtures/volunteering.html +23 -0
- linkfetch-0.1.0/tests/test_config.py +26 -0
- linkfetch-0.1.0/tests/test_emit.py +78 -0
- linkfetch-0.1.0/tests/test_experience.py +65 -0
- linkfetch-0.1.0/tests/test_release.py +51 -0
- linkfetch-0.1.0/tests/test_sections.py +198 -0
- linkfetch-0.1.0/tests/test_text.py +70 -0
- linkfetch-0.1.0/uv.lock +705 -0
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
name: Release
|
|
2
|
+
|
|
3
|
+
# Push a tag like v0.1.1 to publish that version to PyPI. Publishing uses
|
|
4
|
+
# PyPI trusted publishing (OpenID Connect): no API token is stored anywhere.
|
|
5
|
+
on:
|
|
6
|
+
push:
|
|
7
|
+
tags: ["v*"]
|
|
8
|
+
|
|
9
|
+
permissions:
|
|
10
|
+
contents: read
|
|
11
|
+
|
|
12
|
+
jobs:
|
|
13
|
+
build:
|
|
14
|
+
runs-on: ubuntu-latest
|
|
15
|
+
steps:
|
|
16
|
+
- uses: actions/checkout@v4
|
|
17
|
+
- uses: astral-sh/setup-uv@v6
|
|
18
|
+
with:
|
|
19
|
+
python-version: "3.13"
|
|
20
|
+
- name: Tag matches the package version
|
|
21
|
+
run: |
|
|
22
|
+
version=$(uv version --short)
|
|
23
|
+
if [ "v$version" != "$GITHUB_REF_NAME" ]; then
|
|
24
|
+
echo "Tag $GITHUB_REF_NAME does not match pyproject version $version" >&2
|
|
25
|
+
exit 1
|
|
26
|
+
fi
|
|
27
|
+
- run: uv sync --extra dev --locked
|
|
28
|
+
- run: uv run pytest -q
|
|
29
|
+
- run: uv build
|
|
30
|
+
- uses: actions/upload-artifact@v4
|
|
31
|
+
with:
|
|
32
|
+
name: dist
|
|
33
|
+
path: dist/
|
|
34
|
+
|
|
35
|
+
publish:
|
|
36
|
+
needs: build
|
|
37
|
+
runs-on: ubuntu-latest
|
|
38
|
+
environment:
|
|
39
|
+
name: pypi
|
|
40
|
+
url: https://pypi.org/project/linkfetch/
|
|
41
|
+
permissions:
|
|
42
|
+
id-token: write
|
|
43
|
+
steps:
|
|
44
|
+
- uses: actions/download-artifact@v4
|
|
45
|
+
with:
|
|
46
|
+
name: dist
|
|
47
|
+
path: dist/
|
|
48
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
49
|
+
|
|
50
|
+
github-release:
|
|
51
|
+
needs: publish
|
|
52
|
+
runs-on: ubuntu-latest
|
|
53
|
+
permissions:
|
|
54
|
+
contents: write
|
|
55
|
+
steps:
|
|
56
|
+
- uses: actions/download-artifact@v4
|
|
57
|
+
with:
|
|
58
|
+
name: dist
|
|
59
|
+
path: dist/
|
|
60
|
+
- name: Create the GitHub release
|
|
61
|
+
env:
|
|
62
|
+
GH_TOKEN: ${{ github.token }}
|
|
63
|
+
run: gh release create "$GITHUB_REF_NAME" dist/* --repo "$GITHUB_REPOSITORY" --title "$GITHUB_REF_NAME" --generate-notes
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
name: Test
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
|
|
11
|
+
jobs:
|
|
12
|
+
test:
|
|
13
|
+
name: Python ${{ matrix.python }} on ${{ matrix.os }}
|
|
14
|
+
runs-on: ${{ matrix.os }}
|
|
15
|
+
strategy:
|
|
16
|
+
fail-fast: false
|
|
17
|
+
matrix:
|
|
18
|
+
os: [ubuntu-latest, windows-latest, macos-latest]
|
|
19
|
+
python: ["3.10", "3.13"]
|
|
20
|
+
steps:
|
|
21
|
+
- uses: actions/checkout@v4
|
|
22
|
+
- uses: astral-sh/setup-uv@v6
|
|
23
|
+
with:
|
|
24
|
+
python-version: ${{ matrix.python }}
|
|
25
|
+
# Tests exercise parsing and config only, so the Chromium download that
|
|
26
|
+
# `linkfetch capture` needs is skipped.
|
|
27
|
+
- run: uv sync --extra dev --locked
|
|
28
|
+
- run: uv run pytest -q
|
|
29
|
+
- name: Build the package
|
|
30
|
+
run: uv build
|
linkfetch-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Prosperis
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
linkfetch-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: linkfetch
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Capture your own LinkedIn profile on your own computer and turn it into structured profile files
|
|
5
|
+
Project-URL: Homepage, https://github.com/Prosperis/linkfetch
|
|
6
|
+
Project-URL: Issues, https://github.com/Prosperis/linkfetch/issues
|
|
7
|
+
Author: Prosperis
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: cv,export,linkedin,playwright,profile,resume
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: End Users/Desktop
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Topic :: Office/Business
|
|
21
|
+
Requires-Python: >=3.10
|
|
22
|
+
Requires-Dist: lxml>=5.0
|
|
23
|
+
Requires-Dist: playwright>=1.44
|
|
24
|
+
Requires-Dist: pydantic>=2.6
|
|
25
|
+
Requires-Dist: pyyaml>=6.0
|
|
26
|
+
Requires-Dist: rich>=13.7
|
|
27
|
+
Requires-Dist: tomli>=2.0; python_version < '3.11'
|
|
28
|
+
Requires-Dist: typer>=0.12
|
|
29
|
+
Provides-Extra: browser
|
|
30
|
+
Provides-Extra: dev
|
|
31
|
+
Requires-Dist: pytest-mock>=3.12; extra == 'dev'
|
|
32
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
33
|
+
Description-Content-Type: text/markdown
|
|
34
|
+
|
|
35
|
+
# linkfetch
|
|
36
|
+
|
|
37
|
+
Capture **your own** LinkedIn profile on **your own computer** and turn it into
|
|
38
|
+
structured profile files — every role, description, skill, project,
|
|
39
|
+
certification and more — ready to import into
|
|
40
|
+
[THRIVE](https://prosperis-thrive.vercel.app) or any tool that reads YAML.
|
|
41
|
+
|
|
42
|
+
LinkedIn's official data export leaves out details the profile page shows.
|
|
43
|
+
linkfetch reads the rendered pages instead, in a real browser window that you
|
|
44
|
+
log into yourself.
|
|
45
|
+
|
|
46
|
+
## What it does and does not do
|
|
47
|
+
|
|
48
|
+
- **Runs only on your machine.** linkfetch is not a website or a service. It
|
|
49
|
+
sends your data nowhere; the files it writes stay in a folder on your
|
|
50
|
+
computer until you choose to upload them.
|
|
51
|
+
- **Reads only your own profile**, in a browser you log into. It never sees
|
|
52
|
+
your password: you type it into LinkedIn's own login page.
|
|
53
|
+
- **Deterministic, no AI.** Every field comes straight from the page.
|
|
54
|
+
- **Open source** (MIT), so you can read exactly what it does.
|
|
55
|
+
|
|
56
|
+
## Install
|
|
57
|
+
|
|
58
|
+
You need Python 3.10 or newer and [uv](https://docs.astral.sh/uv/) (or
|
|
59
|
+
[pipx](https://pipx.pypa.io/)).
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
uv tool install linkfetch # or: pipx install linkfetch
|
|
63
|
+
linkfetch doctor # shows where data is kept and whether Playwright and a login exist
|
|
64
|
+
playwright install chromium # one-time download of the browser linkfetch drives
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
## Use
|
|
68
|
+
|
|
69
|
+
```bash
|
|
70
|
+
linkfetch login # a browser window opens: log in to LinkedIn, then close it
|
|
71
|
+
linkfetch run --vanity your-name # capture your profile and build the files
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
`your-name` is the part after `/in/` in your profile address:
|
|
75
|
+
`linkedin.com/in/your-name`.
|
|
76
|
+
|
|
77
|
+
`run` visits each section of your profile (experience, education, skills, …),
|
|
78
|
+
expands every "see more", saves the pages, and turns them into one YAML file
|
|
79
|
+
per section plus **`linkfetch-profile.zip`**. `linkfetch` prints where they
|
|
80
|
+
are.
|
|
81
|
+
|
|
82
|
+
### Import into THRIVE
|
|
83
|
+
|
|
84
|
+
In THRIVE, open **Tailor → Profile → Import** and choose
|
|
85
|
+
`linkfetch-profile.zip`. Review each section before saving.
|
|
86
|
+
|
|
87
|
+
## Where your data is kept
|
|
88
|
+
|
|
89
|
+
Installed with `uv tool` or `pipx`, linkfetch keeps everything in your user
|
|
90
|
+
data folder:
|
|
91
|
+
|
|
92
|
+
| OS | Folder |
|
|
93
|
+
|---|---|
|
|
94
|
+
| Windows | `%LOCALAPPDATA%\linkfetch` |
|
|
95
|
+
| macOS | `~/Library/Application Support/linkfetch` |
|
|
96
|
+
| Linux | `$XDG_DATA_HOME/linkfetch` (usually `~/.local/share/linkfetch`) |
|
|
97
|
+
|
|
98
|
+
Set `LINKFETCH_HOME` to use a different folder. Inside it:
|
|
99
|
+
|
|
100
|
+
- `data/browser/` — the browser profile that keeps you logged in to LinkedIn.
|
|
101
|
+
Treat it like a password; delete it to log out.
|
|
102
|
+
- `data/captures/` — the saved profile pages.
|
|
103
|
+
- `data/output/` — the YAML files and `linkfetch-profile.zip`.
|
|
104
|
+
|
|
105
|
+
Delete the folder to remove everything linkfetch stored.
|
|
106
|
+
|
|
107
|
+
## Commands
|
|
108
|
+
|
|
109
|
+
| Command | What it does |
|
|
110
|
+
|---|---|
|
|
111
|
+
| `linkfetch doctor` | Show the data folders and check Playwright and your login |
|
|
112
|
+
| `linkfetch login` | Open a browser window to log in to LinkedIn once |
|
|
113
|
+
| `linkfetch capture --vanity <you>` | Save your profile pages (needs the browser) |
|
|
114
|
+
| `linkfetch parse` | Turn saved pages into YAML and the ZIP (no browser; re-runnable) |
|
|
115
|
+
| `linkfetch run --vanity <you>` | `capture` then `parse` |
|
|
116
|
+
|
|
117
|
+
Use `--section experience --section skills` to limit a run to some sections,
|
|
118
|
+
and `parse --no-zip` to skip the ZIP.
|
|
119
|
+
|
|
120
|
+
## Output
|
|
121
|
+
|
|
122
|
+
One file per section, each a list under a top-level key:
|
|
123
|
+
`basics.yaml`, `experiences.yaml`, `education.yaml`, `skills.yaml`,
|
|
124
|
+
`projects.yaml`, `certifications.yaml`, `awards.yaml`, `languages.yaml`,
|
|
125
|
+
`courses.yaml`, `patents.yaml`, `recommendations.yaml`, `volunteering.yaml`,
|
|
126
|
+
plus `honors.yaml` (a richer copy of awards, not included in the ZIP).
|
|
127
|
+
Experiences keep their full description as a summary plus bullets, and roles
|
|
128
|
+
at the same company stay grouped.
|
|
129
|
+
|
|
130
|
+
## How it works
|
|
131
|
+
|
|
132
|
+
1. **Capture** (needs the browser, runs rarely): a persistent Chromium profile
|
|
133
|
+
you logged into visits each section's details page
|
|
134
|
+
(`/in/<you>/details/experience/`, …), expands collapsed content, and saves
|
|
135
|
+
the rendered HTML.
|
|
136
|
+
2. **Parse** (pure, offline): reads the saved HTML with `lxml` and writes YAML.
|
|
137
|
+
If LinkedIn changes its pages, a parser fix plus `linkfetch parse` is enough
|
|
138
|
+
— no need to capture again.
|
|
139
|
+
|
|
140
|
+
## Development
|
|
141
|
+
|
|
142
|
+
```bash
|
|
143
|
+
git clone https://github.com/Prosperis/linkfetch
|
|
144
|
+
cd linkfetch
|
|
145
|
+
uv venv && uv pip install -e ".[dev]"
|
|
146
|
+
playwright install chromium
|
|
147
|
+
uv run pytest
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
A checkout keeps its data in `data/` next to `config.toml`, which also holds
|
|
151
|
+
browser settings (headless mode, timeouts, scroll pacing). Test fixtures are
|
|
152
|
+
anonymized excerpts of LinkedIn's page structure.
|
|
153
|
+
|
|
154
|
+
## Releasing
|
|
155
|
+
|
|
156
|
+
Bump `version` in `pyproject.toml`, commit, then push a matching tag:
|
|
157
|
+
|
|
158
|
+
```bash
|
|
159
|
+
git tag v0.1.1 && git push origin v0.1.1
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
The Release workflow tests, builds and publishes to PyPI through trusted
|
|
163
|
+
publishing (no stored token), then creates the GitHub release.
|
|
164
|
+
|
|
165
|
+
## Legal note
|
|
166
|
+
|
|
167
|
+
LinkedIn's User Agreement restricts automated access. linkfetch is meant for
|
|
168
|
+
**personal use on your own profile**: it uses a browser you logged into, moves
|
|
169
|
+
at a human pace, reads only your own data and stores nothing remotely. Use it
|
|
170
|
+
on your own account and at your own discretion.
|
|
171
|
+
|
|
172
|
+
## License
|
|
173
|
+
|
|
174
|
+
[MIT](LICENSE) © Prosperis
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
# linkfetch
|
|
2
|
+
|
|
3
|
+
Capture **your own** LinkedIn profile on **your own computer** and turn it into
|
|
4
|
+
structured profile files — every role, description, skill, project,
|
|
5
|
+
certification and more — ready to import into
|
|
6
|
+
[THRIVE](https://prosperis-thrive.vercel.app) or any tool that reads YAML.
|
|
7
|
+
|
|
8
|
+
LinkedIn's official data export leaves out details the profile page shows.
|
|
9
|
+
linkfetch reads the rendered pages instead, in a real browser window that you
|
|
10
|
+
log into yourself.
|
|
11
|
+
|
|
12
|
+
## What it does and does not do
|
|
13
|
+
|
|
14
|
+
- **Runs only on your machine.** linkfetch is not a website or a service. It
|
|
15
|
+
sends your data nowhere; the files it writes stay in a folder on your
|
|
16
|
+
computer until you choose to upload them.
|
|
17
|
+
- **Reads only your own profile**, in a browser you log into. It never sees
|
|
18
|
+
your password: you type it into LinkedIn's own login page.
|
|
19
|
+
- **Deterministic, no AI.** Every field comes straight from the page.
|
|
20
|
+
- **Open source** (MIT), so you can read exactly what it does.
|
|
21
|
+
|
|
22
|
+
## Install
|
|
23
|
+
|
|
24
|
+
You need Python 3.10 or newer and [uv](https://docs.astral.sh/uv/) (or
|
|
25
|
+
[pipx](https://pipx.pypa.io/)).
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
uv tool install linkfetch # or: pipx install linkfetch
|
|
29
|
+
linkfetch doctor # shows where data is kept and whether Playwright and a login exist
|
|
30
|
+
playwright install chromium # one-time download of the browser linkfetch drives
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
## Use
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
linkfetch login # a browser window opens: log in to LinkedIn, then close it
|
|
37
|
+
linkfetch run --vanity your-name # capture your profile and build the files
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
`your-name` is the part after `/in/` in your profile address:
|
|
41
|
+
`linkedin.com/in/your-name`.
|
|
42
|
+
|
|
43
|
+
`run` visits each section of your profile (experience, education, skills, …),
|
|
44
|
+
expands every "see more", saves the pages, and turns them into one YAML file
|
|
45
|
+
per section plus **`linkfetch-profile.zip`**. `linkfetch` prints where they
|
|
46
|
+
are.
|
|
47
|
+
|
|
48
|
+
### Import into THRIVE
|
|
49
|
+
|
|
50
|
+
In THRIVE, open **Tailor → Profile → Import** and choose
|
|
51
|
+
`linkfetch-profile.zip`. Review each section before saving.
|
|
52
|
+
|
|
53
|
+
## Where your data is kept
|
|
54
|
+
|
|
55
|
+
Installed with `uv tool` or `pipx`, linkfetch keeps everything in your user
|
|
56
|
+
data folder:
|
|
57
|
+
|
|
58
|
+
| OS | Folder |
|
|
59
|
+
|---|---|
|
|
60
|
+
| Windows | `%LOCALAPPDATA%\linkfetch` |
|
|
61
|
+
| macOS | `~/Library/Application Support/linkfetch` |
|
|
62
|
+
| Linux | `$XDG_DATA_HOME/linkfetch` (usually `~/.local/share/linkfetch`) |
|
|
63
|
+
|
|
64
|
+
Set `LINKFETCH_HOME` to use a different folder. Inside it:
|
|
65
|
+
|
|
66
|
+
- `data/browser/` — the browser profile that keeps you logged in to LinkedIn.
|
|
67
|
+
Treat it like a password; delete it to log out.
|
|
68
|
+
- `data/captures/` — the saved profile pages.
|
|
69
|
+
- `data/output/` — the YAML files and `linkfetch-profile.zip`.
|
|
70
|
+
|
|
71
|
+
Delete the folder to remove everything linkfetch stored.
|
|
72
|
+
|
|
73
|
+
## Commands
|
|
74
|
+
|
|
75
|
+
| Command | What it does |
|
|
76
|
+
|---|---|
|
|
77
|
+
| `linkfetch doctor` | Show the data folders and check Playwright and your login |
|
|
78
|
+
| `linkfetch login` | Open a browser window to log in to LinkedIn once |
|
|
79
|
+
| `linkfetch capture --vanity <you>` | Save your profile pages (needs the browser) |
|
|
80
|
+
| `linkfetch parse` | Turn saved pages into YAML and the ZIP (no browser; re-runnable) |
|
|
81
|
+
| `linkfetch run --vanity <you>` | `capture` then `parse` |
|
|
82
|
+
|
|
83
|
+
Use `--section experience --section skills` to limit a run to some sections,
|
|
84
|
+
and `parse --no-zip` to skip the ZIP.
|
|
85
|
+
|
|
86
|
+
## Output
|
|
87
|
+
|
|
88
|
+
One file per section, each a list under a top-level key:
|
|
89
|
+
`basics.yaml`, `experiences.yaml`, `education.yaml`, `skills.yaml`,
|
|
90
|
+
`projects.yaml`, `certifications.yaml`, `awards.yaml`, `languages.yaml`,
|
|
91
|
+
`courses.yaml`, `patents.yaml`, `recommendations.yaml`, `volunteering.yaml`,
|
|
92
|
+
plus `honors.yaml` (a richer copy of awards, not included in the ZIP).
|
|
93
|
+
Experiences keep their full description as a summary plus bullets, and roles
|
|
94
|
+
at the same company stay grouped.
|
|
95
|
+
|
|
96
|
+
## How it works
|
|
97
|
+
|
|
98
|
+
1. **Capture** (needs the browser, runs rarely): a persistent Chromium profile
|
|
99
|
+
you logged into visits each section's details page
|
|
100
|
+
(`/in/<you>/details/experience/`, …), expands collapsed content, and saves
|
|
101
|
+
the rendered HTML.
|
|
102
|
+
2. **Parse** (pure, offline): reads the saved HTML with `lxml` and writes YAML.
|
|
103
|
+
If LinkedIn changes its pages, a parser fix plus `linkfetch parse` is enough
|
|
104
|
+
— no need to capture again.
|
|
105
|
+
|
|
106
|
+
## Development
|
|
107
|
+
|
|
108
|
+
```bash
|
|
109
|
+
git clone https://github.com/Prosperis/linkfetch
|
|
110
|
+
cd linkfetch
|
|
111
|
+
uv venv && uv pip install -e ".[dev]"
|
|
112
|
+
playwright install chromium
|
|
113
|
+
uv run pytest
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
A checkout keeps its data in `data/` next to `config.toml`, which also holds
|
|
117
|
+
browser settings (headless mode, timeouts, scroll pacing). Test fixtures are
|
|
118
|
+
anonymized excerpts of LinkedIn's page structure.
|
|
119
|
+
|
|
120
|
+
## Releasing
|
|
121
|
+
|
|
122
|
+
Bump `version` in `pyproject.toml`, commit, then push a matching tag:
|
|
123
|
+
|
|
124
|
+
```bash
|
|
125
|
+
git tag v0.1.1 && git push origin v0.1.1
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
The Release workflow tests, builds and publishes to PyPI through trusted
|
|
129
|
+
publishing (no stored token), then creates the GitHub release.
|
|
130
|
+
|
|
131
|
+
## Legal note
|
|
132
|
+
|
|
133
|
+
LinkedIn's User Agreement restricts automated access. linkfetch is meant for
|
|
134
|
+
**personal use on your own profile**: it uses a browser you logged into, moves
|
|
135
|
+
at a human pace, reads only your own data and stores nothing remotely. Use it
|
|
136
|
+
on your own account and at your own discretion.
|
|
137
|
+
|
|
138
|
+
## License
|
|
139
|
+
|
|
140
|
+
[MIT](LICENSE) © Prosperis
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
# linkfetch configuration.
|
|
2
|
+
# Paths are relative to this file's directory unless absolute.
|
|
3
|
+
|
|
4
|
+
[paths]
|
|
5
|
+
captures = "data/captures" # rendered HTML saved by `linkfetch capture`
|
|
6
|
+
output = "data/output" # YAML emitted by `linkfetch parse`
|
|
7
|
+
user_data_dir = "data/browser" # persistent Chromium profile (keeps you logged in)
|
|
8
|
+
|
|
9
|
+
[browser]
|
|
10
|
+
headless = false # show the window (needed for manual login)
|
|
11
|
+
nav_timeout = 30 # seconds per navigation
|
|
12
|
+
scroll_pause_ms = 600 # pause between scroll steps while loading lazy content
|
|
13
|
+
max_scrolls = 40 # safety cap on scroll-to-bottom loops
|
|
14
|
+
|
|
15
|
+
[linkedin]
|
|
16
|
+
base_url = "https://www.linkedin.com"
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
# linkfetch — Local LinkedIn Profile Scraper
|
|
2
|
+
|
|
3
|
+
**Date:** 2026-06-15
|
|
4
|
+
**Status:** Approved design, building
|
|
5
|
+
|
|
6
|
+
## Purpose
|
|
7
|
+
|
|
8
|
+
A **local CLI tool** that scrapes *your own* LinkedIn profile via browser
|
|
9
|
+
automation and emits **`tailor`-compatible YAML**, so you can bootstrap or
|
|
10
|
+
refresh your `tailor` profile from LinkedIn instead of typing it by hand.
|
|
11
|
+
|
|
12
|
+
It is a sibling project under `prosperis`, decoupled from `tailor`: it knows
|
|
13
|
+
`tailor`'s profile schema and writes matching YAML, but does not import or
|
|
14
|
+
modify `tailor`. You review the emitted YAML and copy it into `tailor/data/profile/`.
|
|
15
|
+
|
|
16
|
+
This is **not** a web service. Nothing is hosted; it is a command run on your
|
|
17
|
+
machine that drives a real browser you log into.
|
|
18
|
+
|
|
19
|
+
## Scope
|
|
20
|
+
|
|
21
|
+
### Sections captured
|
|
22
|
+
|
|
23
|
+
LinkedIn exposes far more than `tailor` models. We capture all of the
|
|
24
|
+
following. Sections that map onto `tailor`'s schema are emitted as
|
|
25
|
+
`tailor`-ready files; the rest are emitted as extra YAML for the user to handle.
|
|
26
|
+
|
|
27
|
+
| LinkedIn section | Output file | Maps to tailor? |
|
|
28
|
+
|-----------------------------|--------------------------|-----------------|
|
|
29
|
+
| Name / headline / about / location | `basics.yaml` | yes (`Basics`) |
|
|
30
|
+
| Experience | `experiences.yaml` | yes (`Experience`) |
|
|
31
|
+
| Skills | `skills.yaml` | yes (`Skill`) |
|
|
32
|
+
| Projects | `projects.yaml` | yes (`Project`) |
|
|
33
|
+
| Licenses & certifications | `certifications.yaml` | yes (`Certification`) |
|
|
34
|
+
| Honors & awards | `awards.yaml` | yes (`Award`) |
|
|
35
|
+
| Education | `education.yaml` | no — extra |
|
|
36
|
+
| Volunteering | `volunteering.yaml` | no — extra |
|
|
37
|
+
| Recommendations | `recommendations.yaml` | no — extra |
|
|
38
|
+
| Patents | `patents.yaml` | no — extra |
|
|
39
|
+
| Courses | `courses.yaml` | no — extra |
|
|
40
|
+
| Languages | `languages.yaml` | no — extra |
|
|
41
|
+
|
|
42
|
+
"Extra" files use a simple, readable schema of our own (defined in this repo),
|
|
43
|
+
not `tailor`'s.
|
|
44
|
+
|
|
45
|
+
### Out of scope
|
|
46
|
+
|
|
47
|
+
- Scraping other people / companies / job postings (job postings are `tailor`'s job).
|
|
48
|
+
- AI / Ollama normalization. **Pure deterministic scrape** — fields come straight
|
|
49
|
+
from the page. `summary` is raw text; `bullets` split on the page's own line
|
|
50
|
+
breaks; the user enriches tags / position framings later in `tailor`.
|
|
51
|
+
- Official-export ZIP parsing.
|
|
52
|
+
- Auto-writing into `tailor`'s profile dir (user reviews and copies manually).
|
|
53
|
+
|
|
54
|
+
## Architecture — two-phase: capture HTML → parse offline
|
|
55
|
+
|
|
56
|
+
The fragile part (LinkedIn's DOM, login, lazy loading) is **quarantined** in a
|
|
57
|
+
capture phase that runs rarely and needs a live session. Parsing is a **pure
|
|
58
|
+
function over saved HTML**, fully unit-testable against fixtures and re-runnable
|
|
59
|
+
with no browser.
|
|
60
|
+
|
|
61
|
+
```
|
|
62
|
+
phase 1: capture (Playwright, live session)
|
|
63
|
+
LinkedIn ──► login (persistent context) ──► visit each section's
|
|
64
|
+
/details/<section>/ page
|
|
65
|
+
──► expand "see more", scroll
|
|
66
|
+
──► save rendered HTML
|
|
67
|
+
to captures/<section>.html
|
|
68
|
+
─────────────────────────────────────────────────────────────
|
|
69
|
+
phase 2: parse (pure, no browser)
|
|
70
|
+
captures/*.html ──► per-section lxml parser ──► pydantic models
|
|
71
|
+
──► YAML files in out/
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
### Components
|
|
75
|
+
|
|
76
|
+
Package `linkfetch` under `src/linkfetch/` (mirrors `tailor`'s `src/` layout).
|
|
77
|
+
|
|
78
|
+
- **`config.py`** — `Config` dataclass + `load_config()`, walking up for
|
|
79
|
+
`config.toml`, exactly like `tailor`. Paths: `captures_dir` (default
|
|
80
|
+
`data/captures`), `output_dir` (default `data/output`). Browser settings:
|
|
81
|
+
`user_data_dir` (persistent Chromium profile so login sticks),
|
|
82
|
+
`headless` (default false for login), `nav_timeout`.
|
|
83
|
+
|
|
84
|
+
- **`capture/browser.py`** — Playwright session management. Launches a
|
|
85
|
+
**persistent context** (`launch_persistent_context(user_data_dir=...)`) so the
|
|
86
|
+
login cookie survives between runs. Exposes:
|
|
87
|
+
- `ensure_logged_in()` — open `linkedin.com/feed`, detect whether logged in;
|
|
88
|
+
if not, print instructions and wait for the user to log in manually, then
|
|
89
|
+
continue.
|
|
90
|
+
- `capture_section(section)` — navigate to the section's detail URL, run the
|
|
91
|
+
auto-expand routine, return rendered HTML.
|
|
92
|
+
|
|
93
|
+
- **`capture/expand.py`** — the scroll + "…see more" / "Show all" expansion
|
|
94
|
+
routine, shared by all sections. Deterministic: scroll to bottom in steps
|
|
95
|
+
until height stops changing; click every visible expand control until none
|
|
96
|
+
remain. No section-specific logic here.
|
|
97
|
+
|
|
98
|
+
- **`capture/sections.py`** — the registry of sections: slug, detail-page URL
|
|
99
|
+
template (e.g. `/in/{vanity}/details/experience/`), and which parser handles it.
|
|
100
|
+
The single source of truth for "what sections exist".
|
|
101
|
+
|
|
102
|
+
- **`parse/` (one module per section)** — `experience.py`, `skills.py`,
|
|
103
|
+
`education.py`, etc. Each exposes `parse(html: str) -> list[Model]` (or a single
|
|
104
|
+
model for basics). Pure: `lxml` in, pydantic out. LinkedIn's randomized class
|
|
105
|
+
names are avoided by anchoring on **stable structures**: the visually-hidden
|
|
106
|
+
`<span aria-hidden="true">` text, `aria-label`s, and the section's anchor `id`.
|
|
107
|
+
|
|
108
|
+
- **`models.py`** — pydantic models. For mapped sections, fields are a **subset**
|
|
109
|
+
of `tailor`'s models using the **same field names** (`id`, `summary`, `bullets`,
|
|
110
|
+
`tags`, `org`, `positions`, `timeline`, …) so emitted YAML drops into `tailor`.
|
|
111
|
+
For extra sections, small purpose-built models.
|
|
112
|
+
|
|
113
|
+
- **`emit.py`** — serialize models to YAML using the same dumper settings as
|
|
114
|
+
`tailor`'s `store.py` (`sort_keys=False`, `allow_unicode=True`,
|
|
115
|
+
`default_flow_style=False`, `width=100`) so output is human-editable and
|
|
116
|
+
diffable, and identical in style to `tailor`'s files.
|
|
117
|
+
|
|
118
|
+
- **`cli.py`** — `typer` app mirroring `tailor`'s style. Commands:
|
|
119
|
+
- `linkfetch doctor` — show config, whether Playwright is installed, whether a
|
|
120
|
+
login session exists.
|
|
121
|
+
- `linkfetch login` — open the browser and establish/refresh the session.
|
|
122
|
+
- `linkfetch capture [--section X ...] [--vanity NAME]` — phase 1; saves HTML
|
|
123
|
+
to `captures/`. Default: all sections.
|
|
124
|
+
- `linkfetch parse [--section X ...]` — phase 2; reads `captures/`, writes
|
|
125
|
+
YAML to `out/`. No browser needed.
|
|
126
|
+
- `linkfetch run` — convenience: `capture` then `parse`.
|
|
127
|
+
|
|
128
|
+
### Data flow
|
|
129
|
+
|
|
130
|
+
1. `login` → persistent Chromium profile stores the LinkedIn cookie under
|
|
131
|
+
`user_data_dir`.
|
|
132
|
+
2. `capture` → for each section, navigate to its `/details/<section>/` page
|
|
133
|
+
(these are paginated and far simpler than the monolithic profile page),
|
|
134
|
+
expand everything, write `captures/<section>.html` + a `captures/meta.json`
|
|
135
|
+
(timestamp, vanity, section list).
|
|
136
|
+
3. `parse` → for each captured section, run its parser, collect models, emit
|
|
137
|
+
`out/<file>.yaml`. Mapped files are byte-compatible with `tailor`; extra files
|
|
138
|
+
use local schemas.
|
|
139
|
+
4. User reviews `out/`, copies the mapped YAML into `tailor/data/profile/`,
|
|
140
|
+
runs `tailor profile validate`.
|
|
141
|
+
|
|
142
|
+
### Error handling
|
|
143
|
+
|
|
144
|
+
- **Playwright missing** → `capture`/`login` fail with the same install hint
|
|
145
|
+
style `tailor` uses (`uv pip install playwright && playwright install chromium`).
|
|
146
|
+
`parse` and `doctor` still work without it.
|
|
147
|
+
- **Not logged in** → `capture` detects the login wall and tells the user to run
|
|
148
|
+
`linkfetch login` (or logs in inline and waits).
|
|
149
|
+
- **Section markup not recognized** → parser returns `[]` for that section and
|
|
150
|
+
records a warning in the run summary rather than crashing the whole run; the
|
|
151
|
+
raw HTML stays on disk so parsing can be retried after a selector fix.
|
|
152
|
+
- **Partial captures** → `parse` only processes sections present in `captures/`;
|
|
153
|
+
missing ones are skipped with a note.
|
|
154
|
+
|
|
155
|
+
### Testing
|
|
156
|
+
|
|
157
|
+
- **Parsers (primary):** unit tests over **saved HTML fixtures** in
|
|
158
|
+
`tests/fixtures/<section>.html` (sanitized, anonymized snippets). Each parser
|
|
159
|
+
has a test asserting the structured output. This is where correctness lives and
|
|
160
|
+
is fully offline / CI-friendly.
|
|
161
|
+
- **expand.py:** logic is browser-bound; covered by a thin smoke path, not unit
|
|
162
|
+
tested against live LinkedIn.
|
|
163
|
+
- **emit.py:** round-trip test — models → YAML → re-load → equal; and a test that
|
|
164
|
+
mapped YAML validates against `tailor`'s pydantic models (import `tailor` as a
|
|
165
|
+
test-only dependency *if available*, else skip).
|
|
166
|
+
- **config.py:** test root discovery and path resolution, mirroring `tailor`.
|
|
167
|
+
|
|
168
|
+
TDD: write the failing parser test against a fixture first, then the parser.
|
|
169
|
+
|
|
170
|
+
## Stack & conventions
|
|
171
|
+
|
|
172
|
+
- Python ≥ 3.10, `uv`-managed venv, `pyproject.toml` with `hatchling`,
|
|
173
|
+
`src/linkfetch/` layout, `typer` + `rich` CLI, `pydantic` v2 models,
|
|
174
|
+
`pyyaml`, `lxml`. `playwright` is an **optional** extra (`[browser]`), exactly
|
|
175
|
+
like `tailor`. `pytest` + `pytest-mock` dev extra.
|
|
176
|
+
- `config.toml` at project root; `data/` gitignored.
|
|
177
|
+
- Console-script entry point `linkfetch = "linkfetch.cli:app"`.
|
|
178
|
+
|
|
179
|
+
## Legal / safety note
|
|
180
|
+
|
|
181
|
+
For **personal use on the user's own profile**. LinkedIn's ToS restricts
|
|
182
|
+
automated access; this tool uses a real logged-in browser session at human-ish
|
|
183
|
+
pace, captures only the authenticated user's own data, and stores nothing
|
|
184
|
+
remotely. The README states this plainly.
|