linkedin-pulse-search-posts 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- linkedin_pulse_search_posts-0.1.0/.gitignore +39 -0
- linkedin_pulse_search_posts-0.1.0/LICENSE +3 -0
- linkedin_pulse_search_posts-0.1.0/PKG-INFO +72 -0
- linkedin_pulse_search_posts-0.1.0/README.md +44 -0
- linkedin_pulse_search_posts-0.1.0/pyproject.toml +52 -0
- linkedin_pulse_search_posts-0.1.0/src/linkedin_search_posts/__init__.py +6 -0
- linkedin_pulse_search_posts-0.1.0/src/linkedin_search_posts/loading/__init__.py +3 -0
- linkedin_pulse_search_posts-0.1.0/src/linkedin_search_posts/loading/waiter.py +183 -0
- linkedin_pulse_search_posts-0.1.0/src/linkedin_search_posts/models/__init__.py +3 -0
- linkedin_pulse_search_posts-0.1.0/src/linkedin_search_posts/models/post.py +70 -0
- linkedin_pulse_search_posts-0.1.0/src/linkedin_search_posts/navigation/__init__.py +3 -0
- linkedin_pulse_search_posts-0.1.0/src/linkedin_search_posts/navigation/feed.py +158 -0
- linkedin_pulse_search_posts-0.1.0/src/linkedin_search_posts/parsing/__init__.py +3 -0
- linkedin_pulse_search_posts-0.1.0/src/linkedin_search_posts/parsing/author_extract.py +398 -0
- linkedin_pulse_search_posts-0.1.0/src/linkedin_search_posts/parsing/post_parser.py +325 -0
- linkedin_pulse_search_posts-0.1.0/src/linkedin_search_posts/parsing/post_url.py +487 -0
- linkedin_pulse_search_posts-0.1.0/src/linkedin_search_posts/parsing/read_posts.py +251 -0
- linkedin_pulse_search_posts-0.1.0/src/linkedin_search_posts/search.py +320 -0
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
cookies.json
|
|
2
|
+
cookies*.json
|
|
3
|
+
storage_state.json
|
|
4
|
+
session_snapshot.json
|
|
5
|
+
*.session
|
|
6
|
+
linkedin-session*
|
|
7
|
+
credentials.json
|
|
8
|
+
secrets.json
|
|
9
|
+
.secrets
|
|
10
|
+
posts.json
|
|
11
|
+
selected_posts.json
|
|
12
|
+
worker/results/*.json
|
|
13
|
+
.env
|
|
14
|
+
.venv/
|
|
15
|
+
venv/
|
|
16
|
+
__pycache__/
|
|
17
|
+
*.pyc
|
|
18
|
+
*.pyo
|
|
19
|
+
*.egg-info/
|
|
20
|
+
*.egg
|
|
21
|
+
dist/
|
|
22
|
+
build/
|
|
23
|
+
**/dist/
|
|
24
|
+
**/build/
|
|
25
|
+
**/*.egg-info/
|
|
26
|
+
.pytest_cache/
|
|
27
|
+
|
|
28
|
+
# Cursor debug session logs
|
|
29
|
+
**/.cursor/debug-*.log
|
|
30
|
+
|
|
31
|
+
# Frontend (Next.js)
|
|
32
|
+
frontend/node_modules/
|
|
33
|
+
frontend/.next/
|
|
34
|
+
frontend/.vercel/
|
|
35
|
+
frontend/.env.local
|
|
36
|
+
frontend/.env.*.local
|
|
37
|
+
|
|
38
|
+
# Local tools (portable Node, etc.)
|
|
39
|
+
.tools/
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: linkedin-pulse-search-posts
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: LinkedIn post search built on linkedin-search-core.
|
|
5
|
+
Project-URL: Homepage, https://github.com/vkudymov/linkedin-pulse-reader-v.3
|
|
6
|
+
Project-URL: Repository, https://github.com/vkudymov/linkedin-pulse-reader-v.3
|
|
7
|
+
Project-URL: Issues, https://github.com/vkudymov/linkedin-pulse-reader-v.3/issues
|
|
8
|
+
Author: LinkedIn Pulse Reader
|
|
9
|
+
Maintainer: LinkedIn Pulse Reader
|
|
10
|
+
License: Proprietary
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: linkedin,playwright,posts
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: License :: Other/Proprietary License
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Typing :: Typed
|
|
22
|
+
Requires-Python: >=3.11
|
|
23
|
+
Requires-Dist: linkedin-search-core>=0.1.0
|
|
24
|
+
Requires-Dist: playwright>=1.49.0
|
|
25
|
+
Provides-Extra: dev
|
|
26
|
+
Requires-Dist: pytest>=8.0.0; extra == 'dev'
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
|
|
29
|
+
# linkedin-search-posts
|
|
30
|
+
|
|
31
|
+
## Description
|
|
32
|
+
|
|
33
|
+
LinkedIn feed post search. Depends on `linkedin-search-core` and does not import the jobs library.
|
|
34
|
+
|
|
35
|
+
Requires Python 3.11+.
|
|
36
|
+
|
|
37
|
+
## Installation
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
pip install linkedin-search-posts
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
That also installs `linkedin-search-core`.
|
|
44
|
+
|
|
45
|
+
Local editable install (core first):
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
pip install -e packages/linkedin-search-core
|
|
49
|
+
pip install -e packages/linkedin-search-posts
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## Dependencies
|
|
53
|
+
|
|
54
|
+
- `linkedin-search-core>=0.1.0`
|
|
55
|
+
- `playwright>=1.49.0`
|
|
56
|
+
|
|
57
|
+
## Usage
|
|
58
|
+
|
|
59
|
+
```python
|
|
60
|
+
from linkedin_search_core import LinkedInSession
|
|
61
|
+
from linkedin_search_posts import PostSearch
|
|
62
|
+
|
|
63
|
+
with LinkedInSession(cookies=cookies) as session:
|
|
64
|
+
posts = PostSearch(session).fetch(limit=10)
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
## API
|
|
68
|
+
|
|
69
|
+
- `PostSearch.fetch(limit=10)` — open the feed, scroll, and parse posts
|
|
70
|
+
- `PostSearch.read(limit=10)` — parse posts already on the current page
|
|
71
|
+
- `PostSearch.search(limit=10)` — alias of `fetch`
|
|
72
|
+
- `Author`, `Post`
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# linkedin-search-posts
|
|
2
|
+
|
|
3
|
+
## Description
|
|
4
|
+
|
|
5
|
+
LinkedIn feed post search. Depends on `linkedin-search-core` and does not import the jobs library.
|
|
6
|
+
|
|
7
|
+
Requires Python 3.11+.
|
|
8
|
+
|
|
9
|
+
## Installation
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
pip install linkedin-search-posts
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
That also installs `linkedin-search-core`.
|
|
16
|
+
|
|
17
|
+
Local editable install (core first):
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
pip install -e packages/linkedin-search-core
|
|
21
|
+
pip install -e packages/linkedin-search-posts
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
## Dependencies
|
|
25
|
+
|
|
26
|
+
- `linkedin-search-core>=0.1.0`
|
|
27
|
+
- `playwright>=1.49.0`
|
|
28
|
+
|
|
29
|
+
## Usage
|
|
30
|
+
|
|
31
|
+
```python
|
|
32
|
+
from linkedin_search_core import LinkedInSession
|
|
33
|
+
from linkedin_search_posts import PostSearch
|
|
34
|
+
|
|
35
|
+
with LinkedInSession(cookies=cookies) as session:
|
|
36
|
+
posts = PostSearch(session).fetch(limit=10)
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
## API
|
|
40
|
+
|
|
41
|
+
- `PostSearch.fetch(limit=10)` — open the feed, scroll, and parse posts
|
|
42
|
+
- `PostSearch.read(limit=10)` — parse posts already on the current page
|
|
43
|
+
- `PostSearch.search(limit=10)` — alias of `fetch`
|
|
44
|
+
- `Author`, `Post`
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.21.0"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "linkedin-pulse-search-posts"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "LinkedIn post search built on linkedin-search-core."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = { text = "Proprietary" }
|
|
12
|
+
authors = [{ name = "LinkedIn Pulse Reader" }]
|
|
13
|
+
maintainers = [{ name = "LinkedIn Pulse Reader" }]
|
|
14
|
+
keywords = ["linkedin", "posts", "playwright"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 3 - Alpha",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"License :: Other/Proprietary License",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Programming Language :: Python :: 3.11",
|
|
22
|
+
"Programming Language :: Python :: 3.12",
|
|
23
|
+
"Programming Language :: Python :: 3.13",
|
|
24
|
+
"Typing :: Typed",
|
|
25
|
+
]
|
|
26
|
+
dependencies = [
|
|
27
|
+
"linkedin-search-core>=0.1.0",
|
|
28
|
+
"playwright>=1.49.0",
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
[project.urls]
|
|
32
|
+
Homepage = "https://github.com/vkudymov/linkedin-pulse-reader-v.3"
|
|
33
|
+
Repository = "https://github.com/vkudymov/linkedin-pulse-reader-v.3"
|
|
34
|
+
Issues = "https://github.com/vkudymov/linkedin-pulse-reader-v.3/issues"
|
|
35
|
+
|
|
36
|
+
[project.optional-dependencies]
|
|
37
|
+
dev = [
|
|
38
|
+
"pytest>=8.0.0",
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
[tool.hatch.build.targets.sdist]
|
|
42
|
+
include = [
|
|
43
|
+
"/src",
|
|
44
|
+
"/README.md",
|
|
45
|
+
"/LICENSE",
|
|
46
|
+
]
|
|
47
|
+
|
|
48
|
+
[tool.hatch.build.targets.wheel]
|
|
49
|
+
packages = ["src/linkedin_search_posts"]
|
|
50
|
+
|
|
51
|
+
[tool.pytest.ini_options]
|
|
52
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
"""
|
|
4
|
+
RU: Ожидания и стабилизация динамической ленты.
|
|
5
|
+
Определяет “готовность” страницы для дальнейших шагов (скролл/парсинг) через наблюдаемые сигналы
|
|
6
|
+
DOM, чтобы снизить флейки в SPA.
|
|
7
|
+
|
|
8
|
+
EN: Waiting and stabilization for a dynamic feed.
|
|
9
|
+
Defines “page readiness” for subsequent steps (scrolling/parsing) using observable DOM signals to
|
|
10
|
+
reduce SPA flakiness.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import time
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from typing import Literal
|
|
16
|
+
|
|
17
|
+
from playwright.sync_api import Page, TimeoutError as PlaywrightTimeoutError
|
|
18
|
+
|
|
19
|
+
from linkedin_search_core.exceptions import FeedLoadError
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(frozen=True, slots=True)
|
|
23
|
+
class FeedWaiter:
|
|
24
|
+
"""
|
|
25
|
+
RU: Политика готовности feed.
|
|
26
|
+
Используется между навигацией и скроллом/парсингом; при проблемах поднимает `FeedLoadError`,
|
|
27
|
+
чтобы оркестратор мог выполнить retry/reload.
|
|
28
|
+
|
|
29
|
+
EN: Feed readiness policy.
|
|
30
|
+
Used between navigation and scrolling/parsing; raises `FeedLoadError` so the orchestrator can
|
|
31
|
+
retry/reload.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
timeout_ms: int
|
|
35
|
+
|
|
36
|
+
feed_container_selectors: tuple[str, ...] = (
|
|
37
|
+
"main[role='main']",
|
|
38
|
+
"div[role='main']",
|
|
39
|
+
"div.application-outlet",
|
|
40
|
+
"div#react-root",
|
|
41
|
+
"div#content",
|
|
42
|
+
"div#main",
|
|
43
|
+
"main",
|
|
44
|
+
"div.scaffold-layout",
|
|
45
|
+
"div.scaffold-layout__content",
|
|
46
|
+
"div.scaffold-layout__inner",
|
|
47
|
+
"div.scaffold-layout__main",
|
|
48
|
+
"div.scaffold-finite-scroll",
|
|
49
|
+
"div.scaffold-finite-scroll__content",
|
|
50
|
+
# Last-resort: always present once DOM is there.
|
|
51
|
+
"body",
|
|
52
|
+
)
|
|
53
|
+
# post_container_selector_group: str = (
|
|
54
|
+
# "div.feed-shared-update-v2, "
|
|
55
|
+
# "div.occludable-update, "
|
|
56
|
+
# "article[data-urn], "
|
|
57
|
+
# "div[data-urn], "
|
|
58
|
+
# "div[data-id], "
|
|
59
|
+
# "div[data-urn*='urn:li:activity'], "
|
|
60
|
+
# "div[data-urn*='urn:li:ugcPost'], "
|
|
61
|
+
# "div[data-urn*='urn:li:share']"
|
|
62
|
+
# )
|
|
63
|
+
post_container_selector_group: str = "[data-testid='expandable-text-box']"
|
|
64
|
+
text_selector = "[data-testid='expandable-text-box']"
|
|
65
|
+
# Fallback anchors for accounts/variants where data-testid is absent.
|
|
66
|
+
text_selector_fallbacks: tuple[str, ...] = (
|
|
67
|
+
"div.update-components-text",
|
|
68
|
+
"div.feed-shared-update-v2__description",
|
|
69
|
+
"div.feed-shared-update-v2",
|
|
70
|
+
"div.occludable-update",
|
|
71
|
+
"article[data-urn]",
|
|
72
|
+
"div[data-urn*='urn:li:activity']",
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
def wait_for_feed_ready(self, page: Page) -> None:
|
|
76
|
+
deadline = time.time() + (self.timeout_ms / 1000.0)
|
|
77
|
+
|
|
78
|
+
# print(
|
|
79
|
+
# "POST CONTAINERS:", self.page.locator("div.feed-shared-update-v2").count()
|
|
80
|
+
# )
|
|
81
|
+
# print(
|
|
82
|
+
# "TEXT BLOCKS:",
|
|
83
|
+
# self.page.locator("[data-testid='expandable-text-box']").count(),
|
|
84
|
+
# )
|
|
85
|
+
|
|
86
|
+
# 1) Ensure primary layout exists.
|
|
87
|
+
# Use state="attached" here: LinkedIn containers may exist but not be considered "visible"
|
|
88
|
+
# due to overlays/transitions; we only need the layout to be present before waiting for posts.
|
|
89
|
+
self._wait_for_any_selector(
|
|
90
|
+
page, self.feed_container_selectors, deadline, state="attached"
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
# 2) Best-effort: wait for at least one content anchor, but do not fail the whole flow.
|
|
94
|
+
# The selector-agnostic scroll loop can still make progress and parsing can be retried later.
|
|
95
|
+
remaining_ms = max(0, int((deadline - time.time()) * 1000))
|
|
96
|
+
try:
|
|
97
|
+
timeout_ms = min(remaining_ms, 6000)
|
|
98
|
+
|
|
99
|
+
selector_used: str | None = None
|
|
100
|
+
selectors = (self.text_selector, *self.text_selector_fallbacks)
|
|
101
|
+
start = time.time()
|
|
102
|
+
for sel in selectors:
|
|
103
|
+
budget_ms = max(
|
|
104
|
+
1, int(timeout_ms * (1.0 - min(0.9, (time.time() - start) / 6.0)))
|
|
105
|
+
)
|
|
106
|
+
try:
|
|
107
|
+
page.wait_for_selector(sel, timeout=budget_ms, state="attached")
|
|
108
|
+
selector_used = sel
|
|
109
|
+
break
|
|
110
|
+
except Exception: # noqa: BLE001 - best-effort across fallbacks
|
|
111
|
+
continue
|
|
112
|
+
|
|
113
|
+
if not selector_used:
|
|
114
|
+
raise PlaywrightTimeoutError("No feed content anchor found.")
|
|
115
|
+
except PlaywrightTimeoutError:
|
|
116
|
+
# Best-effort: do not fail the whole run here.
|
|
117
|
+
# The scroll loop + parsers can still make progress once LinkedIn finishes rendering.
|
|
118
|
+
return
|
|
119
|
+
# except PlaywrightTimeoutError:
|
|
120
|
+
# return
|
|
121
|
+
|
|
122
|
+
# 3) Best-effort stabilization to reduce races while parsing.
|
|
123
|
+
try:
|
|
124
|
+
self._wait_for_stable_post_count(page, deadline)
|
|
125
|
+
except FeedLoadError:
|
|
126
|
+
return
|
|
127
|
+
|
|
128
|
+
def wait_for_new_posts(
|
|
129
|
+
self, page: Page, previous_count: int, timeout_ms: int
|
|
130
|
+
) -> int:
|
|
131
|
+
deadline = time.time() + (timeout_ms / 1000.0)
|
|
132
|
+
locator = page.locator(self.post_container_selector_group)
|
|
133
|
+
while time.time() < deadline:
|
|
134
|
+
count = locator.count()
|
|
135
|
+
if count > previous_count:
|
|
136
|
+
return count
|
|
137
|
+
page.wait_for_timeout(250)
|
|
138
|
+
return locator.count()
|
|
139
|
+
|
|
140
|
+
def current_post_count(self, page: Page) -> int:
|
|
141
|
+
return page.locator(self.post_container_selector_group).count()
|
|
142
|
+
|
|
143
|
+
def _wait_for_any_selector(
|
|
144
|
+
self,
|
|
145
|
+
page: Page,
|
|
146
|
+
selectors: tuple[str, ...],
|
|
147
|
+
deadline: float,
|
|
148
|
+
*,
|
|
149
|
+
state: Literal["attached", "detached", "hidden", "visible"] = "visible",
|
|
150
|
+
) -> None:
|
|
151
|
+
last_error: Exception | None = None
|
|
152
|
+
for idx, sel in enumerate(selectors):
|
|
153
|
+
remaining_ms = max(0, int((deadline - time.time()) * 1000))
|
|
154
|
+
if remaining_ms <= 0:
|
|
155
|
+
break
|
|
156
|
+
try:
|
|
157
|
+
# Budget remaining time across fallback selectors.
|
|
158
|
+
selectors_left = max(1, len(selectors) - idx)
|
|
159
|
+
used_timeout = max(1, int(remaining_ms / selectors_left))
|
|
160
|
+
page.wait_for_selector(sel, timeout=used_timeout, state=state)
|
|
161
|
+
return
|
|
162
|
+
except Exception as e: # noqa: BLE001 - best-effort across fallbacks
|
|
163
|
+
last_error = e
|
|
164
|
+
continue
|
|
165
|
+
raise FeedLoadError(
|
|
166
|
+
"Timed out waiting for LinkedIn feed container to load."
|
|
167
|
+
) from last_error
|
|
168
|
+
|
|
169
|
+
def _wait_for_stable_post_count(self, page: Page, deadline: float) -> None:
|
|
170
|
+
locator = page.locator(self.post_container_selector_group)
|
|
171
|
+
stable_samples = 0
|
|
172
|
+
last_count = -1
|
|
173
|
+
while time.time() < deadline:
|
|
174
|
+
count = locator.count()
|
|
175
|
+
if count == last_count and count > 0:
|
|
176
|
+
stable_samples += 1
|
|
177
|
+
if stable_samples >= 2:
|
|
178
|
+
return
|
|
179
|
+
else:
|
|
180
|
+
stable_samples = 0
|
|
181
|
+
last_count = count
|
|
182
|
+
page.wait_for_timeout(350)
|
|
183
|
+
raise FeedLoadError("Feed did not stabilize in time (post list kept changing).")
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from types import MappingProxyType
|
|
5
|
+
from typing import Any, Mapping
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass(frozen=True, slots=True)
|
|
9
|
+
class Author:
|
|
10
|
+
"""
|
|
11
|
+
RU: Доменная модель автора поста (boundary-объект).
|
|
12
|
+
Возвращается наружу как стабильный контракт, независимо от того, как меняется DOM LinkedIn.
|
|
13
|
+
|
|
14
|
+
EN: Post author domain model (boundary object).
|
|
15
|
+
Returned as a stable contract, independent of LinkedIn DOM changes.
|
|
16
|
+
"""
|
|
17
|
+
name: str
|
|
18
|
+
headline: str | None = None
|
|
19
|
+
profile_url: str | None = None
|
|
20
|
+
urn: str | None = None
|
|
21
|
+
avatar_url: str | None = None
|
|
22
|
+
|
|
23
|
+
extra: Mapping[str, Any] = field(default_factory=dict, repr=False)
|
|
24
|
+
|
|
25
|
+
def __post_init__(self) -> None:
|
|
26
|
+
object.__setattr__(self, "extra", MappingProxyType(dict(self.extra)))
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def merge_authors(a: Author | None, b: Author | None) -> Author | None:
|
|
30
|
+
if a is None:
|
|
31
|
+
return b
|
|
32
|
+
if b is None:
|
|
33
|
+
return a
|
|
34
|
+
return Author(
|
|
35
|
+
name=a.name or b.name,
|
|
36
|
+
headline=a.headline or b.headline,
|
|
37
|
+
profile_url=a.profile_url or b.profile_url,
|
|
38
|
+
urn=a.urn or b.urn,
|
|
39
|
+
avatar_url=a.avatar_url or b.avatar_url,
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass(frozen=True, slots=True)
|
|
44
|
+
class Post:
|
|
45
|
+
"""
|
|
46
|
+
RU: Доменная модель поста из ленты (публичный контракт).
|
|
47
|
+
Поля optional по дизайну: LinkedIn UI/типы постов различаются, а парсер работает best-effort.
|
|
48
|
+
|
|
49
|
+
EN: Feed post domain model (public contract).
|
|
50
|
+
Optional fields are intentional: LinkedIn UI/post types vary and parsing is best-effort.
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
id: str | None = None
|
|
54
|
+
urn: str | None = None
|
|
55
|
+
post_url: str | None = None
|
|
56
|
+
|
|
57
|
+
author: Author | None = None
|
|
58
|
+
content: str | None = None
|
|
59
|
+
published_at_text: str | None = None
|
|
60
|
+
|
|
61
|
+
reactions_count: int | None = None
|
|
62
|
+
comments_count: int | None = None
|
|
63
|
+
|
|
64
|
+
media_urls: tuple[str, ...] = ()
|
|
65
|
+
|
|
66
|
+
extra: Mapping[str, Any] = field(default_factory=dict, repr=False)
|
|
67
|
+
|
|
68
|
+
def __post_init__(self) -> None:
|
|
69
|
+
object.__setattr__(self, "extra", MappingProxyType(dict(self.extra)))
|
|
70
|
+
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
"""
|
|
4
|
+
RU: Навигация к LinkedIn feed и детекция auth-редиректов.
|
|
5
|
+
Модуль отделяет “куда перейти” от ожиданий/скролла/парсинга и явно сигнализирует об отсутствии
|
|
6
|
+
авторизации через `LoginRequiredError`.
|
|
7
|
+
|
|
8
|
+
EN: Navigation to LinkedIn feed and auth-redirect detection.
|
|
9
|
+
Separates “where to go” from waiting/scrolling/parsing and explicitly signals missing auth via
|
|
10
|
+
`LoginRequiredError`.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from contextlib import suppress
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from playwright.sync_api import Error as PlaywrightError
|
|
18
|
+
from playwright.sync_api import Page, TimeoutError as PlaywrightTimeoutError
|
|
19
|
+
|
|
20
|
+
from linkedin_search_core.exceptions import FeedLoadError, LoginRequiredError
|
|
21
|
+
|
|
22
|
+
_session_is_logged_out: Any
|
|
23
|
+
try:
|
|
24
|
+
from session_snapshot import is_logged_out as _session_is_logged_out
|
|
25
|
+
except ImportError: # pragma: no cover
|
|
26
|
+
_session_is_logged_out = None
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass(frozen=True, slots=True)
|
|
30
|
+
class FeedNavigator:
|
|
31
|
+
"""
|
|
32
|
+
RU: Роутер фичи “feed”.
|
|
33
|
+
Используется оркестратором перед ожиданиями/скроллом/парсингом, чтобы гарантировать корректную
|
|
34
|
+
целевую страницу.
|
|
35
|
+
|
|
36
|
+
EN: Feed feature router.
|
|
37
|
+
Used by the orchestrator before waiting/scrolling/parsing to ensure the correct target page.
|
|
38
|
+
"""
|
|
39
|
+
feed_url: str = "https://www.linkedin.com/feed/"
|
|
40
|
+
|
|
41
|
+
def goto_feed(self, page: Page) -> None:
|
|
42
|
+
# If we're already on /feed, avoid re-navigation (LinkedIn can be slow/flaky on reload).
|
|
43
|
+
try:
|
|
44
|
+
current = (page.url or "").lower()
|
|
45
|
+
except Exception:
|
|
46
|
+
current = ""
|
|
47
|
+
|
|
48
|
+
if "linkedin.com/feed" not in current:
|
|
49
|
+
try:
|
|
50
|
+
page.goto(self.feed_url, wait_until="domcontentloaded")
|
|
51
|
+
except PlaywrightTimeoutError:
|
|
52
|
+
# Fallback: LinkedIn may keep the document "loading" for a long time.
|
|
53
|
+
# We only need the navigation to commit; readiness is handled by waiter/selectors.
|
|
54
|
+
try:
|
|
55
|
+
page.goto(self.feed_url, wait_until="commit")
|
|
56
|
+
except PlaywrightError as e:
|
|
57
|
+
raise FeedLoadError(_feed_open_error_message(e)) from e
|
|
58
|
+
except PlaywrightError as e:
|
|
59
|
+
raise FeedLoadError(_feed_open_error_message(e)) from e
|
|
60
|
+
|
|
61
|
+
# LinkedIn may serve authwall/login UI for /feed without changing the URL immediately.
|
|
62
|
+
# Wait briefly for either feed layout or login markers, then decide.
|
|
63
|
+
with suppress(Exception):
|
|
64
|
+
page.wait_for_selector(
|
|
65
|
+
"main[role='main'], div.application-outlet, "
|
|
66
|
+
"input#username, input[name='session_key'], input[name='session_password']",
|
|
67
|
+
timeout=10_000,
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
if self._looks_like_login_page(page) or self._looks_like_checkpoint_page(page):
|
|
71
|
+
raise LoginRequiredError(
|
|
72
|
+
"LinkedIn returned a login/authwall page for the feed URL. "
|
|
73
|
+
"Provide valid cookies or complete manual login first."
|
|
74
|
+
)
|
|
75
|
+
if self._is_auth_redirect(page.url):
|
|
76
|
+
raise LoginRequiredError(
|
|
77
|
+
"LinkedIn redirected to an authentication/checkpoint page. "
|
|
78
|
+
"Provide valid cookies or complete manual login first."
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
@staticmethod
|
|
82
|
+
def _is_auth_redirect(url: str) -> bool:
|
|
83
|
+
if _session_is_logged_out is not None:
|
|
84
|
+
return bool(_session_is_logged_out(url or ""))
|
|
85
|
+
lowered = (url or "").lower()
|
|
86
|
+
return any(
|
|
87
|
+
marker in lowered
|
|
88
|
+
for marker in (
|
|
89
|
+
"/login",
|
|
90
|
+
"/checkpoint",
|
|
91
|
+
"/authwall",
|
|
92
|
+
"linkedin.com/uas/",
|
|
93
|
+
)
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
@staticmethod
|
|
97
|
+
def _looks_like_login_page(page: Page) -> bool:
|
|
98
|
+
"""
|
|
99
|
+
RU: LinkedIn может отдавать страницу логина даже на /feed (URL не всегда меняется).
|
|
100
|
+
EN: LinkedIn may serve a login page even on /feed (URL does not always change).
|
|
101
|
+
"""
|
|
102
|
+
with suppress(Exception):
|
|
103
|
+
title = (page.title() or "").lower()
|
|
104
|
+
if "sign in" in title or "login" in title:
|
|
105
|
+
return True
|
|
106
|
+
|
|
107
|
+
try:
|
|
108
|
+
# Login form markers (avoid user content / PII).
|
|
109
|
+
return page.locator(
|
|
110
|
+
"input#username, input[name='session_key'], input[name='session_password']"
|
|
111
|
+
).count() > 0
|
|
112
|
+
except Exception:
|
|
113
|
+
return False
|
|
114
|
+
|
|
115
|
+
@staticmethod
|
|
116
|
+
def _looks_like_checkpoint_page(page: Page) -> bool:
|
|
117
|
+
"""
|
|
118
|
+
RU: Иногда LinkedIn показывает checkpoint/verification UI, при этом URL может оставаться /feed.
|
|
119
|
+
EN: LinkedIn may show checkpoint/verification UI while the URL still looks like /feed.
|
|
120
|
+
"""
|
|
121
|
+
try:
|
|
122
|
+
return (
|
|
123
|
+
page.locator(
|
|
124
|
+
"form[action*='checkpoint'], input[name='challengeId'], input[name='pin']"
|
|
125
|
+
).count()
|
|
126
|
+
> 0
|
|
127
|
+
)
|
|
128
|
+
except Exception:
|
|
129
|
+
return False
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _safe_url(url: str) -> str:
|
|
133
|
+
# Avoid logging full URLs with params/fragments.
|
|
134
|
+
try:
|
|
135
|
+
from urllib.parse import urlparse
|
|
136
|
+
|
|
137
|
+
p = urlparse(url)
|
|
138
|
+
return f"{p.scheme}://{p.netloc}{p.path}"
|
|
139
|
+
except Exception:
|
|
140
|
+
return url
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _feed_open_error_message(error: Exception) -> str:
|
|
144
|
+
detail = str(error).splitlines()[0] if str(error).strip() else type(error).__name__
|
|
145
|
+
return (
|
|
146
|
+
"LinkedIn feed did not open: https://www.linkedin.com/feed/. "
|
|
147
|
+
"Check internet connection, VPN/proxy, and that LinkedIn opens in a regular browser. "
|
|
148
|
+
f"Playwright error: {detail}"
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _safe_title(page: Page) -> str:
|
|
153
|
+
try:
|
|
154
|
+
t = page.title()
|
|
155
|
+
return t[:120]
|
|
156
|
+
except Exception:
|
|
157
|
+
return ""
|
|
158
|
+
|