linkedin-playwright-scraper 4.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- linkedin_playwright_scraper-4.0.0.dist-info/METADATA +502 -0
- linkedin_playwright_scraper-4.0.0.dist-info/RECORD +26 -0
- linkedin_playwright_scraper-4.0.0.dist-info/WHEEL +4 -0
- linkedin_playwright_scraper-4.0.0.dist-info/licenses/LICENSE +674 -0
- linkedin_scraper/__init__.py +106 -0
- linkedin_scraper/callbacks.py +162 -0
- linkedin_scraper/core/__init__.py +78 -0
- linkedin_scraper/core/auth.py +314 -0
- linkedin_scraper/core/browser.py +265 -0
- linkedin_scraper/core/exceptions.py +39 -0
- linkedin_scraper/core/permalink_cache.py +49 -0
- linkedin_scraper/core/rate_limit_guard.py +122 -0
- linkedin_scraper/core/utils.py +299 -0
- linkedin_scraper/models/__init__.py +20 -0
- linkedin_scraper/models/company.py +80 -0
- linkedin_scraper/models/job.py +59 -0
- linkedin_scraper/models/person.py +133 -0
- linkedin_scraper/models/post.py +54 -0
- linkedin_scraper/scrapers/__init__.py +19 -0
- linkedin_scraper/scrapers/base.py +271 -0
- linkedin_scraper/scrapers/company.py +208 -0
- linkedin_scraper/scrapers/company_posts.py +346 -0
- linkedin_scraper/scrapers/feed.py +1760 -0
- linkedin_scraper/scrapers/job.py +205 -0
- linkedin_scraper/scrapers/job_search.py +145 -0
- linkedin_scraper/scrapers/person.py +720 -0
|
@@ -0,0 +1,502 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: linkedin-playwright-scraper
|
|
3
|
+
Version: 4.0.0
|
|
4
|
+
Summary: Async LinkedIn scraper for profiles, companies, jobs and feed — Playwright-based
|
|
5
|
+
Project-URL: Homepage, https://github.com/vinzlac/linkedin_scraper
|
|
6
|
+
Project-URL: Documentation, https://github.com/vinzlac/linkedin_scraper#readme
|
|
7
|
+
Project-URL: Repository, https://github.com/vinzlac/linkedin_scraper
|
|
8
|
+
Project-URL: Issues, https://github.com/vinzlac/linkedin_scraper/issues
|
|
9
|
+
Author: Vincent Lacombe
|
|
10
|
+
License: Apache 2.0
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: async,feed,linkedin,mcp,playwright,profiles,scraper,scraping
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Internet :: WWW/HTTP :: Dynamic Content
|
|
24
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
25
|
+
Requires-Python: >=3.9
|
|
26
|
+
Requires-Dist: aiofiles>=23.0.0
|
|
27
|
+
Requires-Dist: lxml>=5.0.0
|
|
28
|
+
Requires-Dist: playwright>=1.40.0
|
|
29
|
+
Requires-Dist: pydantic>=2.0.0
|
|
30
|
+
Requires-Dist: python-dotenv>=1.0.0
|
|
31
|
+
Requires-Dist: requests>=2.31.0
|
|
32
|
+
Description-Content-Type: text/markdown
|
|
33
|
+
|
|
34
|
+
# LinkedIn Scraper
|
|
35
|
+
|
|
36
|
+
[](https://badge.fury.io/py/linkedin-scraper)
|
|
37
|
+
[](https://www.python.org/downloads/)
|
|
38
|
+
[](https://opensource.org/licenses/Apache-2.0)
|
|
39
|
+
|
|
40
|
+
Async LinkedIn scraper built with Playwright for extracting profile, company, and job data from LinkedIn.
|
|
41
|
+
|
|
42
|
+
## ⚠️ Breaking Changes in v3.0.0
|
|
43
|
+
|
|
44
|
+
**Version 3.0.0 introduces breaking changes and is NOT backwards compatible with previous versions.**
|
|
45
|
+
|
|
46
|
+
### What Changed:
|
|
47
|
+
- **Playwright instead of Selenium** - Complete rewrite using Playwright for better performance and reliability
|
|
48
|
+
- **Async/await throughout** - All methods are now async and require `await`
|
|
49
|
+
- **New package structure** - Imports have changed (e.g., `from linkedin_scraper import PersonScraper`)
|
|
50
|
+
- **Updated data models** - Using Pydantic models instead of simple objects
|
|
51
|
+
- **Different API** - Method signatures and return types have changed
|
|
52
|
+
|
|
53
|
+
### Migration Guide:
|
|
54
|
+
|
|
55
|
+
**Before (v2.x with Selenium):**
|
|
56
|
+
```python
|
|
57
|
+
from linkedin_scraper import Person
|
|
58
|
+
|
|
59
|
+
person = Person("https://linkedin.com/in/username", driver=driver)
|
|
60
|
+
print(person.name)
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
**After (v3.0+ with Playwright):**
|
|
64
|
+
```python
|
|
65
|
+
import asyncio
|
|
66
|
+
from linkedin_scraper import BrowserManager, PersonScraper
|
|
67
|
+
|
|
68
|
+
async def main():
|
|
69
|
+
async with BrowserManager() as browser:
|
|
70
|
+
await browser.load_session("session.json")
|
|
71
|
+
scraper = PersonScraper(browser.page)
|
|
72
|
+
person = await scraper.scrape("https://linkedin.com/in/username")
|
|
73
|
+
print(person.name)
|
|
74
|
+
|
|
75
|
+
asyncio.run(main())
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
**If you need the old Selenium-based version:**
|
|
79
|
+
```bash
|
|
80
|
+
pip install linkedin-scraper==2.11.2
|
|
81
|
+
```
|
|
82
|
+
## Quick Start (development)
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
git clone https://github.com/joeyism/linkedin_scraper.git
|
|
86
|
+
cd linkedin_scraper
|
|
87
|
+
|
|
88
|
+
# Install dependencies + Playwright browser
|
|
89
|
+
just install
|
|
90
|
+
|
|
91
|
+
# Create your LinkedIn session (manual login in browser)
|
|
92
|
+
just session
|
|
93
|
+
|
|
94
|
+
# Scrape a profile (slug or full URL)
|
|
95
|
+
just run-person your-linkedin-slug
|
|
96
|
+
|
|
97
|
+
# Scrape your feed
|
|
98
|
+
just run-feed
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
Run `just` with no arguments to list all available commands.
|
|
102
|
+
|
|
103
|
+
---
|
|
104
|
+
|
|
105
|
+
---
|
|
106
|
+
|
|
107
|
+
## Features
|
|
108
|
+
|
|
109
|
+
- **Person Profiles** - Scrape comprehensive profile information
|
|
110
|
+
- Basic info (name, headline, location, about)
|
|
111
|
+
- Work experience with details
|
|
112
|
+
- Education history
|
|
113
|
+
- Skills and accomplishments
|
|
114
|
+
|
|
115
|
+
- **Company Pages** - Extract company information
|
|
116
|
+
- Company overview and details
|
|
117
|
+
- Industry and size
|
|
118
|
+
- Headquarters location
|
|
119
|
+
|
|
120
|
+
- **Company Posts** - Scrape posts from company pages
|
|
121
|
+
- Post content and text
|
|
122
|
+
- Reactions, comments, reposts counts
|
|
123
|
+
- Posted date and images
|
|
124
|
+
|
|
125
|
+
- **Feed** - Scrape your authenticated LinkedIn feed
|
|
126
|
+
- Author name and profile URL
|
|
127
|
+
- Post content, date, reactions, comments, reposts
|
|
128
|
+
- Sponsored/promoted posts automatically filtered out
|
|
129
|
+
|
|
130
|
+
- **Job Listings** - Scrape job postings
|
|
131
|
+
- Job details and requirements
|
|
132
|
+
- Company information
|
|
133
|
+
- Application links
|
|
134
|
+
|
|
135
|
+
- **Async/Await** - Modern async Python with Playwright
|
|
136
|
+
- **Type Safety** - Full Pydantic models for all data
|
|
137
|
+
- **Progress Callbacks** - Track scraping progress
|
|
138
|
+
- **Session Management** - Reuse authenticated sessions
|
|
139
|
+
|
|
140
|
+
## Installation
|
|
141
|
+
|
|
142
|
+
**From PyPI:**
|
|
143
|
+
```bash
|
|
144
|
+
pip install linkedin-playwright-scraper
|
|
145
|
+
playwright install chromium
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
**From source (recommended for development):**
|
|
149
|
+
```bash
|
|
150
|
+
# Requires uv and just
|
|
151
|
+
just install
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
## Quick Start
|
|
155
|
+
|
|
156
|
+
### Basic Usage
|
|
157
|
+
|
|
158
|
+
```python
|
|
159
|
+
import asyncio
|
|
160
|
+
from linkedin_scraper import BrowserManager, PersonScraper
|
|
161
|
+
|
|
162
|
+
async def main():
|
|
163
|
+
# Initialize browser
|
|
164
|
+
async with BrowserManager(headless=False) as browser:
|
|
165
|
+
# Load authenticated session
|
|
166
|
+
await browser.load_session("session.json")
|
|
167
|
+
|
|
168
|
+
# Create scraper
|
|
169
|
+
scraper = PersonScraper(browser.page)
|
|
170
|
+
|
|
171
|
+
# Scrape a profile
|
|
172
|
+
person = await scraper.scrape("https://linkedin.com/in/williamhgates/")
|
|
173
|
+
|
|
174
|
+
# Access data
|
|
175
|
+
print(f"Name: {person.name}")
|
|
176
|
+
print(f"Headline: {person.headline}")
|
|
177
|
+
print(f"Location: {person.location}")
|
|
178
|
+
print(f"Experiences: {len(person.experiences)}")
|
|
179
|
+
print(f"Education: {len(person.educations)}")
|
|
180
|
+
|
|
181
|
+
asyncio.run(main())
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
### Company Scraping
|
|
185
|
+
|
|
186
|
+
```python
|
|
187
|
+
from linkedin_scraper import CompanyScraper
|
|
188
|
+
|
|
189
|
+
async def scrape_company():
|
|
190
|
+
async with BrowserManager(headless=False) as browser:
|
|
191
|
+
await browser.load_session("session.json")
|
|
192
|
+
|
|
193
|
+
scraper = CompanyScraper(browser.page)
|
|
194
|
+
company = await scraper.scrape("https://linkedin.com/company/microsoft/")
|
|
195
|
+
|
|
196
|
+
print(f"Company: {company.name}")
|
|
197
|
+
print(f"Industry: {company.industry}")
|
|
198
|
+
print(f"Size: {company.company_size}")
|
|
199
|
+
print(f"About: {company.about_us[:200]}...")
|
|
200
|
+
|
|
201
|
+
asyncio.run(scrape_company())
|
|
202
|
+
```
|
|
203
|
+
|
|
204
|
+
### Job Scraping
|
|
205
|
+
|
|
206
|
+
```python
|
|
207
|
+
from linkedin_scraper import JobSearchScraper
|
|
208
|
+
|
|
209
|
+
async def search_jobs():
|
|
210
|
+
async with BrowserManager(headless=False) as browser:
|
|
211
|
+
await browser.load_session("session.json")
|
|
212
|
+
|
|
213
|
+
scraper = JobSearchScraper(browser.page)
|
|
214
|
+
jobs = await scraper.search(
|
|
215
|
+
keywords="Python Developer",
|
|
216
|
+
location="San Francisco",
|
|
217
|
+
limit=10
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
for job in jobs:
|
|
221
|
+
print(f"{job.title} at {job.company}")
|
|
222
|
+
print(f"Location: {job.location}")
|
|
223
|
+
print(f"Link: {job.linkedin_url}")
|
|
224
|
+
print("---")
|
|
225
|
+
|
|
226
|
+
asyncio.run(search_jobs())
|
|
227
|
+
```
|
|
228
|
+
|
|
229
|
+
### Company Posts Scraping
|
|
230
|
+
|
|
231
|
+
```python
|
|
232
|
+
from linkedin_scraper import BrowserManager, CompanyPostsScraper
|
|
233
|
+
|
|
234
|
+
async def scrape_company_posts():
|
|
235
|
+
async with BrowserManager(headless=False) as browser:
|
|
236
|
+
await browser.load_session("session.json")
|
|
237
|
+
|
|
238
|
+
scraper = CompanyPostsScraper(browser.page)
|
|
239
|
+
posts = await scraper.scrape(
|
|
240
|
+
"https://linkedin.com/company/microsoft/",
|
|
241
|
+
limit=10
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
for post in posts:
|
|
245
|
+
print(f"Posted: {post.posted_date}")
|
|
246
|
+
print(f"Text: {post.text[:200]}...")
|
|
247
|
+
print(f"Reactions: {post.reactions_count}")
|
|
248
|
+
print(f"Comments: {post.comments_count}")
|
|
249
|
+
print(f"URL: {post.linkedin_url}")
|
|
250
|
+
print("---")
|
|
251
|
+
|
|
252
|
+
asyncio.run(scrape_company_posts())
|
|
253
|
+
```
|
|
254
|
+
|
|
255
|
+
### Feed Scraping
|
|
256
|
+
|
|
257
|
+
```python
|
|
258
|
+
from linkedin_scraper import BrowserManager, FeedScraper
|
|
259
|
+
|
|
260
|
+
async def scrape_feed():
|
|
261
|
+
async with BrowserManager(headless=False) as browser:
|
|
262
|
+
await browser.load_session("session.json")
|
|
263
|
+
|
|
264
|
+
scraper = FeedScraper(browser.page)
|
|
265
|
+
posts = await scraper.scrape(limit=20)
|
|
266
|
+
|
|
267
|
+
for post in posts:
|
|
268
|
+
print(f"Author: {post.author_name} ({post.author_url})")
|
|
269
|
+
print(f"Posted: {post.posted_date}")
|
|
270
|
+
print(f"Text: {post.text[:200]}...")
|
|
271
|
+
print(f"Reactions: {post.reactions_count}")
|
|
272
|
+
print("---")
|
|
273
|
+
|
|
274
|
+
asyncio.run(scrape_feed())
|
|
275
|
+
```
|
|
276
|
+
|
|
277
|
+
Or via the command line:
|
|
278
|
+
|
|
279
|
+
```bash
|
|
280
|
+
just run-feed
|
|
281
|
+
```
|
|
282
|
+
|
|
283
|
+
## Authentication
|
|
284
|
+
|
|
285
|
+
LinkedIn requires authentication. You need to create a session file first:
|
|
286
|
+
|
|
287
|
+
### Option 1: Manual Login Script
|
|
288
|
+
|
|
289
|
+
```python
|
|
290
|
+
from linkedin_scraper import BrowserManager, wait_for_manual_login
|
|
291
|
+
|
|
292
|
+
async def create_session():
|
|
293
|
+
async with BrowserManager(headless=False) as browser:
|
|
294
|
+
# Navigate to LinkedIn
|
|
295
|
+
await browser.page.goto("https://www.linkedin.com/login")
|
|
296
|
+
|
|
297
|
+
# Wait for manual login (opens browser)
|
|
298
|
+
print("Please log in to LinkedIn...")
|
|
299
|
+
await wait_for_manual_login(browser.page, timeout=300)
|
|
300
|
+
|
|
301
|
+
# Save session
|
|
302
|
+
await browser.save_session("session.json")
|
|
303
|
+
print("✓ Session saved!")
|
|
304
|
+
|
|
305
|
+
asyncio.run(create_session())
|
|
306
|
+
```
|
|
307
|
+
|
|
308
|
+
### Option 2: Programmatic Login
|
|
309
|
+
|
|
310
|
+
```python
|
|
311
|
+
from linkedin_scraper import BrowserManager, login_with_credentials
|
|
312
|
+
import os
|
|
313
|
+
|
|
314
|
+
async def login():
|
|
315
|
+
async with BrowserManager(headless=False) as browser:
|
|
316
|
+
# Login with credentials
|
|
317
|
+
await login_with_credentials(
|
|
318
|
+
browser.page,
|
|
319
|
+
username=os.getenv("LINKEDIN_EMAIL"),
|
|
320
|
+
password=os.getenv("LINKEDIN_PASSWORD")
|
|
321
|
+
)
|
|
322
|
+
|
|
323
|
+
# Save session for reuse
|
|
324
|
+
await browser.save_session("session.json")
|
|
325
|
+
|
|
326
|
+
asyncio.run(login())
|
|
327
|
+
```
|
|
328
|
+
|
|
329
|
+
## Progress Tracking
|
|
330
|
+
|
|
331
|
+
Track scraping progress with callbacks:
|
|
332
|
+
|
|
333
|
+
```python
|
|
334
|
+
from linkedin_scraper import ConsoleCallback, PersonScraper
|
|
335
|
+
|
|
336
|
+
async def scrape_with_progress():
|
|
337
|
+
callback = ConsoleCallback() # Prints progress to console
|
|
338
|
+
|
|
339
|
+
async with BrowserManager(headless=False) as browser:
|
|
340
|
+
await browser.load_session("session.json")
|
|
341
|
+
|
|
342
|
+
scraper = PersonScraper(browser.page, callback=callback)
|
|
343
|
+
person = await scraper.scrape("https://linkedin.com/in/williamhgates/")
|
|
344
|
+
|
|
345
|
+
asyncio.run(scrape_with_progress())
|
|
346
|
+
```
|
|
347
|
+
|
|
348
|
+
### Custom Callbacks
|
|
349
|
+
|
|
350
|
+
```python
|
|
351
|
+
from linkedin_scraper import ProgressCallback
|
|
352
|
+
|
|
353
|
+
class MyCallback(ProgressCallback):
|
|
354
|
+
async def on_start(self, scraper_type: str, url: str):
|
|
355
|
+
print(f"Starting {scraper_type} scraping: {url}")
|
|
356
|
+
|
|
357
|
+
async def on_progress(self, message: str, percent: int):
|
|
358
|
+
print(f"[{percent}%] {message}")
|
|
359
|
+
|
|
360
|
+
async def on_complete(self, scraper_type: str, url: str):
|
|
361
|
+
print(f"Completed {scraper_type}: {url}")
|
|
362
|
+
|
|
363
|
+
async def on_error(self, error: Exception):
|
|
364
|
+
print(f"Error: {error}")
|
|
365
|
+
```
|
|
366
|
+
|
|
367
|
+
## Data Models
|
|
368
|
+
|
|
369
|
+
All scraped data is returned as Pydantic models:
|
|
370
|
+
|
|
371
|
+
### Person
|
|
372
|
+
|
|
373
|
+
```python
|
|
374
|
+
class Person(BaseModel):
|
|
375
|
+
name: str
|
|
376
|
+
headline: Optional[str]
|
|
377
|
+
location: Optional[str]
|
|
378
|
+
about: Optional[str]
|
|
379
|
+
linkedin_url: str
|
|
380
|
+
experiences: List[Experience]
|
|
381
|
+
educations: List[Education]
|
|
382
|
+
skills: List[str]
|
|
383
|
+
accomplishments: Optional[Accomplishment]
|
|
384
|
+
```
|
|
385
|
+
|
|
386
|
+
### Company
|
|
387
|
+
|
|
388
|
+
```python
|
|
389
|
+
class Company(BaseModel):
|
|
390
|
+
name: str
|
|
391
|
+
industry: Optional[str]
|
|
392
|
+
company_size: Optional[str]
|
|
393
|
+
headquarters: Optional[str]
|
|
394
|
+
founded: Optional[str]
|
|
395
|
+
specialties: List[str]
|
|
396
|
+
about: Optional[str]
|
|
397
|
+
linkedin_url: str
|
|
398
|
+
```
|
|
399
|
+
|
|
400
|
+
### Job
|
|
401
|
+
|
|
402
|
+
```python
|
|
403
|
+
class Job(BaseModel):
|
|
404
|
+
title: str
|
|
405
|
+
company: str
|
|
406
|
+
location: Optional[str]
|
|
407
|
+
description: Optional[str]
|
|
408
|
+
employment_type: Optional[str]
|
|
409
|
+
seniority_level: Optional[str]
|
|
410
|
+
linkedin_url: str
|
|
411
|
+
```
|
|
412
|
+
|
|
413
|
+
### Post
|
|
414
|
+
|
|
415
|
+
```python
|
|
416
|
+
class Post(BaseModel):
|
|
417
|
+
linkedin_url: Optional[str]
|
|
418
|
+
urn: Optional[str]
|
|
419
|
+
author_name: Optional[str]
|
|
420
|
+
author_url: Optional[str]
|
|
421
|
+
text: Optional[str]
|
|
422
|
+
posted_date: Optional[str]
|
|
423
|
+
reactions_count: Optional[int]
|
|
424
|
+
comments_count: Optional[int]
|
|
425
|
+
reposts_count: Optional[int]
|
|
426
|
+
image_urls: List[str]
|
|
427
|
+
video_url: Optional[str]
|
|
428
|
+
article_url: Optional[str]
|
|
429
|
+
```
|
|
430
|
+
|
|
431
|
+
## Advanced Usage
|
|
432
|
+
|
|
433
|
+
### Browser Configuration
|
|
434
|
+
|
|
435
|
+
```python
|
|
436
|
+
browser = BrowserManager(
|
|
437
|
+
headless=False, # Show browser window
|
|
438
|
+
slow_mo=100, # Slow down operations (ms)
|
|
439
|
+
viewport={"width": 1920, "height": 1080},
|
|
440
|
+
user_agent="Custom User Agent"
|
|
441
|
+
)
|
|
442
|
+
```
|
|
443
|
+
|
|
444
|
+
### Error Handling
|
|
445
|
+
|
|
446
|
+
```python
|
|
447
|
+
from linkedin_scraper import (
|
|
448
|
+
AuthenticationError,
|
|
449
|
+
RateLimitError,
|
|
450
|
+
ProfileNotFoundError
|
|
451
|
+
)
|
|
452
|
+
|
|
453
|
+
try:
|
|
454
|
+
person = await scraper.scrape(url)
|
|
455
|
+
except AuthenticationError:
|
|
456
|
+
print("Not logged in - session expired")
|
|
457
|
+
except RateLimitError:
|
|
458
|
+
print("Rate limited by LinkedIn")
|
|
459
|
+
except ProfileNotFoundError:
|
|
460
|
+
print("Profile not found or private")
|
|
461
|
+
```
|
|
462
|
+
|
|
463
|
+
## Best Practices
|
|
464
|
+
|
|
465
|
+
1. **Rate Limiting** - Add delays between requests
|
|
466
|
+
```python
|
|
467
|
+
import asyncio
|
|
468
|
+
await asyncio.sleep(2) # 2 second delay
|
|
469
|
+
```
|
|
470
|
+
|
|
471
|
+
2. **Session Reuse** - Save and reuse sessions to avoid frequent logins
|
|
472
|
+
|
|
473
|
+
3. **Error Handling** - Always handle exceptions (rate limits, auth errors, etc.)
|
|
474
|
+
|
|
475
|
+
4. **Headless Mode** - Use `headless=False` during development, `True` for production
|
|
476
|
+
|
|
477
|
+
5. **Respect LinkedIn** - Don't scrape aggressively, respect rate limits
|
|
478
|
+
|
|
479
|
+
## Requirements
|
|
480
|
+
|
|
481
|
+
- Python 3.9+
|
|
482
|
+
- [uv](https://docs.astral.sh/uv/) (package manager)
|
|
483
|
+
- [just](https://just.systems/) (task runner)
|
|
484
|
+
- Playwright (installed automatically via `just install`)
|
|
485
|
+
|
|
486
|
+
## License
|
|
487
|
+
|
|
488
|
+
Apache License 2.0 - see [LICENSE](LICENSE) file for details.
|
|
489
|
+
|
|
490
|
+
## Contributing
|
|
491
|
+
|
|
492
|
+
Contributions are welcome! Please feel free to submit a Pull Request.
|
|
493
|
+
|
|
494
|
+
## Disclaimer
|
|
495
|
+
|
|
496
|
+
This tool is for educational purposes only. Make sure to comply with LinkedIn's Terms of Service and use responsibly. The authors are not responsible for any misuse of this tool.
|
|
497
|
+
|
|
498
|
+
## Links
|
|
499
|
+
|
|
500
|
+
- [GitHub Repository](https://github.com/joeyism/linkedin_scraper)
|
|
501
|
+
- [Issue Tracker](https://github.com/joeyism/linkedin_scraper/issues)
|
|
502
|
+
- [PyPI Package](https://pypi.org/project/linkedin-scraper/)
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
linkedin_scraper/__init__.py,sha256=PiEZ1emaIBqXEyGZM0SPXFK1pUUZuverw09CJzFRa9U,2031
|
|
2
|
+
linkedin_scraper/callbacks.py,sha256=5QynTionrIL-HWr9PvfHZjhOipx5tPDE7qXIXqzW29I,5119
|
|
3
|
+
linkedin_scraper/core/__init__.py,sha256=4RX5Us0AEctZckkQH0dhW2oA6RxEjFwZKldTMy_sMJ0,1784
|
|
4
|
+
linkedin_scraper/core/auth.py,sha256=g9B0kcgxW61f8sxhl5EqNswjWUSKpnUP4IPXt15fBqM,10652
|
|
5
|
+
linkedin_scraper/core/browser.py,sha256=VoM2knGosmUSsAPQS82lyVUh_RIoASo-mLoxSV2mTy0,8715
|
|
6
|
+
linkedin_scraper/core/exceptions.py,sha256=W_bFWxJOgmK_1Ug_jATwfN7DQzUVyx5407fH4VzUu00,979
|
|
7
|
+
linkedin_scraper/core/permalink_cache.py,sha256=3bYGuKp4TeuvslAYO20RO2kaUXSxcIyvKgeby-Zr0-c,1624
|
|
8
|
+
linkedin_scraper/core/rate_limit_guard.py,sha256=IadnB5AzRKIxCAp1fh6047sZVE41JNiDEUrJ0yywiT4,4240
|
|
9
|
+
linkedin_scraper/core/utils.py,sha256=vdtVNbhM0t5_zpEFLotLeZzm8jbVZNmkalR2_opux-8,9629
|
|
10
|
+
linkedin_scraper/models/__init__.py,sha256=C7LuNmnK1WpiIwrHq7A0-9b9LU3Xy_EiVI0nKMLPNKs,427
|
|
11
|
+
linkedin_scraper/models/company.py,sha256=d-XMY3YaDk8Y4Tx8-g8Sp9UumBV5CVbfOvt4S9c73ZE,2500
|
|
12
|
+
linkedin_scraper/models/job.py,sha256=_Z8XT8-OyIju0YbYesMjr_2hAmBSu_GjlUmsCobtJq0,1751
|
|
13
|
+
linkedin_scraper/models/person.py,sha256=TfYmvefVNQS-o2wwy5_7o9tmG2uNZHUjhUAEv2qG4Pg,3641
|
|
14
|
+
linkedin_scraper/models/post.py,sha256=cDNFUiwJlZbsC_6sF_fYIOHRRYuhzr9flkT62VmUWDc,2150
|
|
15
|
+
linkedin_scraper/scrapers/__init__.py,sha256=3Ilgx4KnmY8YSwUK3slamuhkCmw7EbBvAmpcK-XxxBs,448
|
|
16
|
+
linkedin_scraper/scrapers/base.py,sha256=wd6IHbYnenpLlMPdg2-vfDx7HHMIAnCCx5Jmpq7D980,8509
|
|
17
|
+
linkedin_scraper/scrapers/company.py,sha256=ulHdUP-CrWiAfqaQ0GYE1OHhDmODwPMlFN2tT9g2gLA,8372
|
|
18
|
+
linkedin_scraper/scrapers/company_posts.py,sha256=AibnAtKRp_CGgOEskrRkrrgrT0wvyUpwg2slownPyEk,13962
|
|
19
|
+
linkedin_scraper/scrapers/feed.py,sha256=4IS6tdNBNOnTfmw7BND95RRc_gRw73-CoflpI-5yV_k,85000
|
|
20
|
+
linkedin_scraper/scrapers/job.py,sha256=VylRclztwwURAXbqx8ZsNuWM5VV14YKQnV6os4BprS0,7631
|
|
21
|
+
linkedin_scraper/scrapers/job_search.py,sha256=5aAPnPdLOAAJvwrb1M7IPOKmNP-loWKcmjjrvbsuGJA,4804
|
|
22
|
+
linkedin_scraper/scrapers/person.py,sha256=MIbcDf75bORDmGeYhB_SAkL3SkoIP9BC7E1-CrVWujM,30047
|
|
23
|
+
linkedin_playwright_scraper-4.0.0.dist-info/METADATA,sha256=DeoyQgFaXU7eupJrTWveyiNihxtsvcWEQHalgrZTHGU,13855
|
|
24
|
+
linkedin_playwright_scraper-4.0.0.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
25
|
+
linkedin_playwright_scraper-4.0.0.dist-info/licenses/LICENSE,sha256=OXLcl0T2SZ8Pmy2_dmlvKuetivmyPd5m1q-Gyd-zaYY,35149
|
|
26
|
+
linkedin_playwright_scraper-4.0.0.dist-info/RECORD,,
|