stage-cli 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stage/__init__.py +1 -0
- stage/__main__.py +8 -0
- stage/banner.py +32 -0
- stage/bootstrap/__init__.py +0 -0
- stage/bootstrap/openjobs.py +392 -0
- stage/classify/__init__.py +29 -0
- stage/classify/eligibility.py +115 -0
- stage/classify/internship.py +64 -0
- stage/classify/role.py +91 -0
- stage/classify/scope.py +47 -0
- stage/cli/__init__.py +0 -0
- stage/cli/app.py +4 -0
- stage/cli/commands/__init__.py +8 -0
- stage/cli/commands/discovery.py +294 -0
- stage/cli/commands/insight.py +494 -0
- stage/cli/commands/pipeline.py +337 -0
- stage/cli/commands/postings.py +473 -0
- stage/cli/commands/schedule.py +171 -0
- stage/cli/housekeeping.py +64 -0
- stage/cli/logfile.py +56 -0
- stage/cli/notify.py +170 -0
- stage/cli/options.py +678 -0
- stage/cli/render.py +1398 -0
- stage/cli/runlock.py +74 -0
- stage/cli/schedule.py +702 -0
- stage/cli/schedule_state.py +363 -0
- stage/cli/selection.py +83 -0
- stage/cli/serialize.py +196 -0
- stage/companies.py +542 -0
- stage/data/companies/a.yaml +1289 -0
- stage/data/companies/b.yaml +900 -0
- stage/data/companies/c.yaml +1377 -0
- stage/data/companies/d.yaml +497 -0
- stage/data/companies/e.yaml +519 -0
- stage/data/companies/f.yaml +454 -0
- stage/data/companies/g.yaml +601 -0
- stage/data/companies/h.yaml +446 -0
- stage/data/companies/i.yaml +503 -0
- stage/data/companies/j.yaml +138 -0
- stage/data/companies/k.yaml +278 -0
- stage/data/companies/l.yaml +402 -0
- stage/data/companies/m.yaml +937 -0
- stage/data/companies/n.yaml +549 -0
- stage/data/companies/o.yaml +371 -0
- stage/data/companies/other.yaml +58 -0
- stage/data/companies/p.yaml +825 -0
- stage/data/companies/q.yaml +121 -0
- stage/data/companies/r.yaml +583 -0
- stage/data/companies/s.yaml +1140 -0
- stage/data/companies/t.yaml +817 -0
- stage/data/companies/u.yaml +196 -0
- stage/data/companies/v.yaml +325 -0
- stage/data/companies/w.yaml +353 -0
- stage/data/companies/x.yaml +67 -0
- stage/data/companies/y.yaml +36 -0
- stage/data/companies/z.yaml +146 -0
- stage/data/fonts/DejaVuSans.LICENSE.txt +99 -0
- stage/data/fonts/DejaVuSans.ttf +0 -0
- stage/data/lexicon/company_tokens.yaml +228 -0
- stage/data/lexicon/eligibility.yaml +455 -0
- stage/data/lexicon/inclusive_suffixes.yaml +37 -0
- stage/data/lexicon/internship.yaml +187 -0
- stage/data/lexicon/language.yaml +226 -0
- stage/data/lexicon/locations.yaml +1159 -0
- stage/data/lexicon/roles.yaml +2012 -0
- stage/data/lexicon/terms.yaml +76 -0
- stage/data/lexicon/workday_facets.yaml +27 -0
- stage/data/seed_companies.yaml +198 -0
- stage/dedup/__init__.py +19 -0
- stage/dedup/identity.py +113 -0
- stage/dedup/resolve.py +97 -0
- stage/domain/__init__.py +244 -0
- stage/domain/company.py +49 -0
- stage/domain/coverage.py +86 -0
- stage/domain/custom_board.py +92 -0
- stage/domain/discovery.py +94 -0
- stage/domain/enums.py +114 -0
- stage/domain/events.py +204 -0
- stage/domain/filters.py +27 -0
- stage/domain/health.py +169 -0
- stage/domain/ids.py +48 -0
- stage/domain/job.py +47 -0
- stage/domain/matching.py +15 -0
- stage/domain/priority.py +34 -0
- stage/domain/quarantine.py +39 -0
- stage/domain/rate_state.py +78 -0
- stage/domain/retention.py +20 -0
- stage/domain/rotation.py +46 -0
- stage/domain/signals.py +12 -0
- stage/domain/sync_run.py +35 -0
- stage/domain/text.py +113 -0
- stage/domain/validator.py +14 -0
- stage/domain/visits.py +60 -0
- stage/domain/workday.py +38 -0
- stage/http/__init__.py +58 -0
- stage/http/breaker.py +53 -0
- stage/http/cache.py +44 -0
- stage/http/client.py +725 -0
- stage/http/profiles.py +101 -0
- stage/lexicon.py +370 -0
- stage/normalize/__init__.py +16 -0
- stage/normalize/language.py +47 -0
- stage/normalize/location.py +271 -0
- stage/normalize/terms.py +153 -0
- stage/normalize/urls.py +122 -0
- stage/paths.py +86 -0
- stage/py.typed +0 -0
- stage/services/__init__.py +0 -0
- stage/services/canary.py +120 -0
- stage/services/coverage.py +231 -0
- stage/services/discover.py +747 -0
- stage/services/export.py +274 -0
- stage/services/health.py +237 -0
- stage/services/maintenance.py +225 -0
- stage/services/quarantine.py +20 -0
- stage/services/query.py +86 -0
- stage/services/sync.py +1257 -0
- stage/sources/__init__.py +82 -0
- stage/sources/_text.py +79 -0
- stage/sources/ashby.py +93 -0
- stage/sources/bamboohr.py +80 -0
- stage/sources/base.py +225 -0
- stage/sources/breezy.py +90 -0
- stage/sources/collage.py +60 -0
- stage/sources/community_feeds.py +142 -0
- stage/sources/curated_markdown.py +289 -0
- stage/sources/custom_json.py +610 -0
- stage/sources/espresso.py +154 -0
- stage/sources/feed.py +44 -0
- stage/sources/greenhouse.py +104 -0
- stage/sources/jobbank.py +147 -0
- stage/sources/jobvite.py +133 -0
- stage/sources/lever.py +76 -0
- stage/sources/oracle_cloud.py +187 -0
- stage/sources/platforms.py +609 -0
- stage/sources/quebec_emploi.py +146 -0
- stage/sources/recruitee.py +96 -0
- stage/sources/simplify.py +110 -0
- stage/sources/smartrecruiters.py +216 -0
- stage/sources/speedyapply.py +200 -0
- stage/sources/themuse.py +157 -0
- stage/sources/workable.py +83 -0
- stage/sources/workday.py +524 -0
- stage/sources/zshah.py +99 -0
- stage/storage/__init__.py +29 -0
- stage/storage/migrations/0001_initial.sql +239 -0
- stage/storage/migrations/__init__.py +135 -0
- stage/storage/repository.py +213 -0
- stage/storage/search.py +28 -0
- stage/storage/sqlite_repo.py +1586 -0
- stage/storage/writer.py +249 -0
- stage/tui/__init__.py +0 -0
- stage/tui/app.py +82 -0
- stage/tui/help.py +26 -0
- stage/tui/safe.py +21 -0
- stage/tui/screens/__init__.py +0 -0
- stage/tui/screens/boards.py +186 -0
- stage/tui/screens/postings.py +509 -0
- stage/tui/screens/review.py +209 -0
- stage/tui/screens/splash.py +37 -0
- stage/tui/screens/stats.py +124 -0
- stage/tui/screens/sync.py +194 -0
- stage/tui/state.py +160 -0
- stage/tui/theme.tcss +205 -0
- stage/tui/widgets/__init__.py +0 -0
- stage_cli-1.0.0.dist-info/METADATA +379 -0
- stage_cli-1.0.0.dist-info/RECORD +170 -0
- stage_cli-1.0.0.dist-info/WHEEL +4 -0
- stage_cli-1.0.0.dist-info/entry_points.txt +2 -0
- stage_cli-1.0.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
from datetime import datetime
|
|
2
|
+
from typing import ClassVar
|
|
3
|
+
|
|
4
|
+
from bs4 import BeautifulSoup
|
|
5
|
+
from bs4.element import Tag
|
|
6
|
+
|
|
7
|
+
from stage.domain import Job, SourceSignals, job_id
|
|
8
|
+
from stage.http import HttpClient, HttpStatusError
|
|
9
|
+
from stage.sources import register_feed
|
|
10
|
+
from stage.sources._text import collapse_whitespace
|
|
11
|
+
from stage.sources.base import FetchResult, PayloadValidationError, capture_payload, malformed_note
|
|
12
|
+
|
|
13
|
+
HOST = "www.espresso-jobs.com"
|
|
14
|
+
SEARCH = f"https://{HOST}/emploi"
|
|
15
|
+
POSTING = f"https://{HOST}/emploi/{{id}}/{{slug}}"
|
|
16
|
+
TERMS = ("stage", "stagiaire", "intern", "internship")
|
|
17
|
+
PAGE_CAP = 3
|
|
18
|
+
PAGE_SIZE = 21
|
|
19
|
+
MAX_ROWS = 1000
|
|
20
|
+
INTERNSHIP_BADGE = "stage"
|
|
21
|
+
ROW_SELECTOR = "div.job_index-content_list_item"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def badge(row: Tag) -> str:
|
|
25
|
+
found = row.select_one("p.job_index-content_list_item_infos-type")
|
|
26
|
+
if found is None:
|
|
27
|
+
return ""
|
|
28
|
+
return collapse_whitespace(found.get_text(" ", strip=True))
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def field(row: Tag, selector: str) -> str:
|
|
32
|
+
found = row.select_one(selector)
|
|
33
|
+
if found is None:
|
|
34
|
+
return ""
|
|
35
|
+
return collapse_whitespace(found.get_text(" ", strip=True))
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def where(row: Tag) -> str:
|
|
39
|
+
found = row.select_one("div.job-location-info")
|
|
40
|
+
if found is None:
|
|
41
|
+
return "Québec, Canada"
|
|
42
|
+
city = collapse_whitespace(str(found.get("data-city") or ""))
|
|
43
|
+
province = collapse_whitespace(str(found.get("data-province") or ""))
|
|
44
|
+
parts = [part for part in (city, province, "Canada") if part]
|
|
45
|
+
return ", ".join(parts)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@register_feed
|
|
49
|
+
class EspressoJobsFeed:
|
|
50
|
+
name: ClassVar[str] = "espresso-jobs"
|
|
51
|
+
rate_profile: ClassVar[str] = "feeds"
|
|
52
|
+
hosts: ClassVar[frozenset[str]] = frozenset({HOST})
|
|
53
|
+
bucket_key: ClassVar[str] = "espresso-jobs"
|
|
54
|
+
|
|
55
|
+
def season_year(self, now: datetime) -> int:
|
|
56
|
+
return now.year
|
|
57
|
+
|
|
58
|
+
def plan(self, now: datetime) -> tuple[str, ...]:
|
|
59
|
+
return tuple(f"{SEARCH}?keyword={term}&distance=all&page_no=1" for term in TERMS)
|
|
60
|
+
|
|
61
|
+
async def fetch(self, client: HttpClient, now: datetime) -> FetchResult:
|
|
62
|
+
seen: dict[str, Job] = {}
|
|
63
|
+
malformed = 0
|
|
64
|
+
truncated = False
|
|
65
|
+
searched = False
|
|
66
|
+
empty: list[str] = []
|
|
67
|
+
last_empty = ""
|
|
68
|
+
|
|
69
|
+
exhausted: list[str] = []
|
|
70
|
+
|
|
71
|
+
for term in TERMS:
|
|
72
|
+
for page in range(1, PAGE_CAP + 1):
|
|
73
|
+
url = f"{SEARCH}?keyword={term}&distance=all&page_no={page}"
|
|
74
|
+
try:
|
|
75
|
+
text = await client.get_text(url, revalidate=page > 1)
|
|
76
|
+
except HttpStatusError as exc:
|
|
77
|
+
if page > 1 and exc.status == 404:
|
|
78
|
+
exhausted.append(term)
|
|
79
|
+
break
|
|
80
|
+
raise
|
|
81
|
+
if text.not_modified:
|
|
82
|
+
break
|
|
83
|
+
rows = self._rows(text.text)
|
|
84
|
+
if not rows:
|
|
85
|
+
if page == 1:
|
|
86
|
+
empty.append(term)
|
|
87
|
+
last_empty = text.text
|
|
88
|
+
break
|
|
89
|
+
searched = True
|
|
90
|
+
for row in rows:
|
|
91
|
+
job, dropped = self._to_job(row, now)
|
|
92
|
+
malformed += dropped
|
|
93
|
+
if job is not None:
|
|
94
|
+
seen.setdefault(job.id, job)
|
|
95
|
+
if len(seen) >= MAX_ROWS:
|
|
96
|
+
truncated = True
|
|
97
|
+
break
|
|
98
|
+
if len(rows) < PAGE_SIZE:
|
|
99
|
+
break
|
|
100
|
+
if truncated:
|
|
101
|
+
break
|
|
102
|
+
|
|
103
|
+
if not searched:
|
|
104
|
+
captured = capture_payload(self.name, "search", {"head": last_empty[:4000]})
|
|
105
|
+
raise PayloadValidationError(
|
|
106
|
+
f"{self.name}: no query returned a listing row, so the results page changed "
|
|
107
|
+
f"shape rather than matching nothing; captured at {captured}"
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
notes = []
|
|
111
|
+
if empty:
|
|
112
|
+
notes.append(f"{len(empty)} of {len(TERMS)} searches matched nothing")
|
|
113
|
+
if exhausted:
|
|
114
|
+
notes.append(f"{len(exhausted)} search(es) ran past their last page, which answers 404")
|
|
115
|
+
if truncated:
|
|
116
|
+
notes.append(f"stopped at the {MAX_ROWS}-posting cap for one run")
|
|
117
|
+
if malformed:
|
|
118
|
+
notes.append(malformed_note(malformed))
|
|
119
|
+
return FetchResult(
|
|
120
|
+
jobs=tuple(seen.values()),
|
|
121
|
+
authoritative=False,
|
|
122
|
+
degraded="; ".join(notes),
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
@staticmethod
|
|
126
|
+
def _rows(text: str) -> list[Tag]:
|
|
127
|
+
return BeautifulSoup(text, "html.parser").select(ROW_SELECTOR)
|
|
128
|
+
|
|
129
|
+
def _to_job(self, row: Tag, now: datetime) -> tuple[Job | None, int]:
|
|
130
|
+
identifier = collapse_whitespace(str(row.get("id") or ""))
|
|
131
|
+
slug = collapse_whitespace(str(row.get("data-slug") or ""))
|
|
132
|
+
title = field(row, "h2.job_index-content_list_item-title")
|
|
133
|
+
if not identifier or not slug or not title:
|
|
134
|
+
return None, 1
|
|
135
|
+
declared = badge(row)
|
|
136
|
+
if declared.lower() != INTERNSHIP_BADGE:
|
|
137
|
+
return None, 0
|
|
138
|
+
company = field(row, "p.job_index-content_list_item-company") or "Espresso-Jobs"
|
|
139
|
+
return (
|
|
140
|
+
Job(
|
|
141
|
+
id=job_id(self.name, "search", identifier),
|
|
142
|
+
source=self.name,
|
|
143
|
+
company=company,
|
|
144
|
+
title_raw=title,
|
|
145
|
+
title_normalized=title.lower(),
|
|
146
|
+
apply_url_raw=POSTING.format(id=identifier, slug=slug),
|
|
147
|
+
description="",
|
|
148
|
+
location_raw=where(row),
|
|
149
|
+
first_seen=now,
|
|
150
|
+
last_seen=now,
|
|
151
|
+
signals=SourceSignals(employment_type=declared.lower()),
|
|
152
|
+
),
|
|
153
|
+
0,
|
|
154
|
+
)
|
stage/sources/feed.py
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
from datetime import datetime
|
|
2
|
+
from typing import ClassVar, Protocol, runtime_checkable
|
|
3
|
+
|
|
4
|
+
from stage.http import HttpClient
|
|
5
|
+
from stage.sources.base import FetchResult
|
|
6
|
+
|
|
7
|
+
_FEEDS: dict[str, "FeedAdapter"] = {}
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@runtime_checkable
|
|
11
|
+
class FeedAdapter(Protocol):
|
|
12
|
+
name: ClassVar[str]
|
|
13
|
+
rate_profile: ClassVar[str]
|
|
14
|
+
hosts: ClassVar[frozenset[str]]
|
|
15
|
+
bucket_key: ClassVar[str]
|
|
16
|
+
|
|
17
|
+
def season_year(self, now: datetime) -> int:
|
|
18
|
+
pass
|
|
19
|
+
|
|
20
|
+
def plan(self, now: datetime) -> tuple[str, ...]:
|
|
21
|
+
pass
|
|
22
|
+
|
|
23
|
+
async def fetch(self, client: HttpClient, now: datetime) -> FetchResult:
|
|
24
|
+
pass
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def register_feed[F: FeedAdapter](cls: type[F]) -> type[F]:
|
|
28
|
+
adapter = cls()
|
|
29
|
+
existing = _FEEDS.get(adapter.name)
|
|
30
|
+
if existing is not None and type(existing) is not cls:
|
|
31
|
+
raise ValueError(f"two feeds claim the name {adapter.name!r}")
|
|
32
|
+
_FEEDS[adapter.name] = adapter
|
|
33
|
+
return cls
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def get_feeds() -> dict[str, FeedAdapter]:
|
|
37
|
+
from stage.sources import load_builtins
|
|
38
|
+
|
|
39
|
+
load_builtins()
|
|
40
|
+
return dict(_FEEDS)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def upcoming_season_year(now: datetime, rolls_in_month: int = 8) -> int:
|
|
44
|
+
return now.year + 1 if now.month >= rolls_in_month else now.year
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
from collections.abc import Sequence
|
|
2
|
+
from datetime import datetime
|
|
3
|
+
from typing import Any, ClassVar
|
|
4
|
+
|
|
5
|
+
from pydantic import BaseModel, ConfigDict
|
|
6
|
+
|
|
7
|
+
from stage.domain import Company, Job, Platform, job_id
|
|
8
|
+
from stage.http import HttpClient, ResponseTooLargeError
|
|
9
|
+
from stage.sources import register
|
|
10
|
+
from stage.sources._text import collapse_whitespace, strip_html
|
|
11
|
+
from stage.sources.base import BoardAdapter, FetchResult, malformed_note
|
|
12
|
+
|
|
13
|
+
BASE_URL = "https://boards-api.greenhouse.io/v1/boards/{slug}/jobs"
|
|
14
|
+
HOST = "boards-api.greenhouse.io"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class GreenhouseLocation(BaseModel):
|
|
18
|
+
model_config = ConfigDict(extra="ignore")
|
|
19
|
+
|
|
20
|
+
name: str | None = ""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class GreenhouseJob(BaseModel):
|
|
24
|
+
model_config = ConfigDict(extra="ignore")
|
|
25
|
+
|
|
26
|
+
id: int
|
|
27
|
+
title: str
|
|
28
|
+
absolute_url: str
|
|
29
|
+
updated_at: datetime | None = None
|
|
30
|
+
location: GreenhouseLocation | None = None
|
|
31
|
+
content: str = ""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class GreenhouseBoard(BaseModel):
|
|
35
|
+
model_config = ConfigDict(extra="ignore")
|
|
36
|
+
|
|
37
|
+
jobs: list[Any]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@register
|
|
41
|
+
class GreenhouseAdapter(BoardAdapter):
|
|
42
|
+
name: ClassVar[str] = "greenhouse"
|
|
43
|
+
platform: ClassVar[Platform] = Platform.GREENHOUSE
|
|
44
|
+
rate_profile: ClassVar[str] = "broad"
|
|
45
|
+
hosts: ClassVar[frozenset[str]] = frozenset({HOST})
|
|
46
|
+
detail_budget: ClassVar[int] = 0
|
|
47
|
+
max_requests_per_company: ClassVar[int] = 2
|
|
48
|
+
|
|
49
|
+
base_url: ClassVar[str] = BASE_URL
|
|
50
|
+
query: ClassVar[tuple[tuple[str, str], ...]] = (("content", "true"),)
|
|
51
|
+
root_model: ClassVar[type[BaseModel] | None] = GreenhouseBoard
|
|
52
|
+
rows_field: ClassVar[str] = "jobs"
|
|
53
|
+
row_model: ClassVar[type[BaseModel]] = GreenhouseJob
|
|
54
|
+
|
|
55
|
+
async def fetch(
|
|
56
|
+
self,
|
|
57
|
+
company: Company,
|
|
58
|
+
client: HttpClient,
|
|
59
|
+
now: datetime,
|
|
60
|
+
facets: object = None,
|
|
61
|
+
details: Sequence[str] = (),
|
|
62
|
+
) -> FetchResult:
|
|
63
|
+
url = self.url_for(company)
|
|
64
|
+
try:
|
|
65
|
+
response = await client.get_json(url, params={"content": "true"})
|
|
66
|
+
except ResponseTooLargeError:
|
|
67
|
+
return await self._without_descriptions(company, client, now, url)
|
|
68
|
+
if response.not_modified:
|
|
69
|
+
return FetchResult(not_modified=True)
|
|
70
|
+
return self.result(company, response.payload, now)
|
|
71
|
+
|
|
72
|
+
async def _without_descriptions(
|
|
73
|
+
self, company: Company, client: HttpClient, now: datetime, url: str
|
|
74
|
+
) -> FetchResult:
|
|
75
|
+
response = await client.get_json(url, params={"content": "false"})
|
|
76
|
+
if response.not_modified:
|
|
77
|
+
return FetchResult(not_modified=True)
|
|
78
|
+
postings, dropped = self.validate(company, response.payload)
|
|
79
|
+
notes = ["board exceeds the response cap with content=true; fetched without descriptions"]
|
|
80
|
+
if dropped:
|
|
81
|
+
notes.append(malformed_note(dropped))
|
|
82
|
+
return FetchResult(
|
|
83
|
+
jobs=tuple(self.to_job(company, posting, now) for posting in postings),
|
|
84
|
+
degraded="; ".join(notes),
|
|
85
|
+
authoritative=not dropped,
|
|
86
|
+
stale_urls=(f"{url}?content=false",),
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
def to_job(self, company: Company, row: Any, now: datetime) -> Job:
|
|
90
|
+
return Job(
|
|
91
|
+
id=job_id(self.name, company.slug, str(row.id)),
|
|
92
|
+
source=self.name,
|
|
93
|
+
company=company.name,
|
|
94
|
+
title_raw=row.title,
|
|
95
|
+
title_normalized=collapse_whitespace(row.title),
|
|
96
|
+
apply_url_raw=row.absolute_url,
|
|
97
|
+
description=strip_html(row.content),
|
|
98
|
+
location_raw=collapse_whitespace(
|
|
99
|
+
row.location.name if row.location and row.location.name else ""
|
|
100
|
+
),
|
|
101
|
+
first_seen=now,
|
|
102
|
+
last_seen=now,
|
|
103
|
+
source_posted_at=row.updated_at,
|
|
104
|
+
)
|
stage/sources/jobbank.py
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
import re
|
|
2
|
+
from datetime import datetime
|
|
3
|
+
from typing import ClassVar
|
|
4
|
+
|
|
5
|
+
from bs4 import BeautifulSoup
|
|
6
|
+
from bs4.element import Tag
|
|
7
|
+
|
|
8
|
+
from stage.domain import Job, SourceSignals, job_id
|
|
9
|
+
from stage.http import HttpClient
|
|
10
|
+
from stage.sources import register_feed
|
|
11
|
+
from stage.sources._text import collapse_whitespace
|
|
12
|
+
from stage.sources.base import FetchResult, PayloadValidationError, capture_payload, malformed_note
|
|
13
|
+
|
|
14
|
+
HOST = "www.jobbank.gc.ca"
|
|
15
|
+
SEARCH = f"https://{HOST}/jobsearch/jobsearch"
|
|
16
|
+
POSTING = f"https://{HOST}/jobsearch/jobposting/{{id}}"
|
|
17
|
+
TERMS = ("programmer", "developer", "software", "informatique", "programmeur", "développeur")
|
|
18
|
+
PROVINCES = ("QC", "ON", "BC", "AB")
|
|
19
|
+
PAGE_CAP = 2
|
|
20
|
+
MAX_ROWS = 2000
|
|
21
|
+
_ARTICLE_ID = re.compile(r"^article-(\d+)$")
|
|
22
|
+
KEPT_FLAGS = ("jobinternshipflag", "jobstudentflag")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def posting_id(article: Tag) -> str:
|
|
26
|
+
found = _ARTICLE_ID.match(str(article.get("id") or ""))
|
|
27
|
+
return found.group(1) if found else ""
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def declared_term(article: Tag) -> str:
|
|
31
|
+
for flag in KEPT_FLAGS:
|
|
32
|
+
found = article.select_one(f".{flag}")
|
|
33
|
+
if found is not None:
|
|
34
|
+
return collapse_whitespace(found.get_text(" ", strip=True))
|
|
35
|
+
return ""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def field(article: Tag, selector: str) -> str:
|
|
39
|
+
found = article.select_one(selector)
|
|
40
|
+
if found is None:
|
|
41
|
+
return ""
|
|
42
|
+
for hidden in found.select(".wb-inv"):
|
|
43
|
+
hidden.decompose()
|
|
44
|
+
return collapse_whitespace(found.get_text(" ", strip=True))
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@register_feed
|
|
48
|
+
class JobBankFeed:
|
|
49
|
+
name: ClassVar[str] = "jobbank"
|
|
50
|
+
rate_profile: ClassVar[str] = "jobbank"
|
|
51
|
+
hosts: ClassVar[frozenset[str]] = frozenset({HOST})
|
|
52
|
+
bucket_key: ClassVar[str] = "jobbank"
|
|
53
|
+
|
|
54
|
+
def season_year(self, now: datetime) -> int:
|
|
55
|
+
return now.year
|
|
56
|
+
|
|
57
|
+
def plan(self, now: datetime) -> tuple[str, ...]:
|
|
58
|
+
return tuple(
|
|
59
|
+
f"{SEARCH}?searchstring={term}&fprov={province}"
|
|
60
|
+
for province in PROVINCES
|
|
61
|
+
for term in TERMS
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
async def fetch(self, client: HttpClient, now: datetime) -> FetchResult:
|
|
65
|
+
seen: dict[str, Job] = {}
|
|
66
|
+
malformed = 0
|
|
67
|
+
truncated = False
|
|
68
|
+
searched = False
|
|
69
|
+
empty: list[str] = []
|
|
70
|
+
last_empty = ""
|
|
71
|
+
for province in PROVINCES:
|
|
72
|
+
for term in TERMS:
|
|
73
|
+
for page in range(1, PAGE_CAP + 1):
|
|
74
|
+
url = f"{SEARCH}?searchstring={term}&fprov={province}&page={page}"
|
|
75
|
+
text = await client.get_text(url, revalidate=page > 1)
|
|
76
|
+
if text.not_modified:
|
|
77
|
+
break
|
|
78
|
+
rows = self._articles(text.text)
|
|
79
|
+
if not rows:
|
|
80
|
+
if page == 1:
|
|
81
|
+
empty.append(f"{province}/{term}")
|
|
82
|
+
last_empty = text.text
|
|
83
|
+
break
|
|
84
|
+
searched = True
|
|
85
|
+
for article in rows:
|
|
86
|
+
job, dropped = self._to_job(article, now)
|
|
87
|
+
malformed += dropped
|
|
88
|
+
if job is not None:
|
|
89
|
+
seen.setdefault(job.id, job)
|
|
90
|
+
if len(seen) >= MAX_ROWS:
|
|
91
|
+
truncated = True
|
|
92
|
+
break
|
|
93
|
+
if truncated:
|
|
94
|
+
break
|
|
95
|
+
if truncated:
|
|
96
|
+
break
|
|
97
|
+
|
|
98
|
+
if not searched:
|
|
99
|
+
captured = capture_payload(self.name, "search", {"head": last_empty[:4000]})
|
|
100
|
+
raise PayloadValidationError(
|
|
101
|
+
f"{self.name}: no query returned an <article> row, so the results page changed "
|
|
102
|
+
f"shape rather than matching nothing; captured at {captured}"
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
notes = []
|
|
106
|
+
if empty:
|
|
107
|
+
notes.append(f"{len(empty)} of {len(PROVINCES) * len(TERMS)} searches matched nothing")
|
|
108
|
+
if truncated:
|
|
109
|
+
notes.append(f"stopped at the {MAX_ROWS}-posting cap for one run")
|
|
110
|
+
if malformed:
|
|
111
|
+
notes.append(malformed_note(malformed))
|
|
112
|
+
return FetchResult(
|
|
113
|
+
jobs=tuple(seen.values()),
|
|
114
|
+
authoritative=False,
|
|
115
|
+
degraded="; ".join(notes),
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
@staticmethod
|
|
119
|
+
def _articles(text: str) -> list[Tag]:
|
|
120
|
+
return BeautifulSoup(text, "html.parser").select("article")
|
|
121
|
+
|
|
122
|
+
def _to_job(self, article: Tag, now: datetime) -> tuple[Job | None, int]:
|
|
123
|
+
identifier = posting_id(article)
|
|
124
|
+
title = field(article, ".noctitle")
|
|
125
|
+
term = declared_term(article)
|
|
126
|
+
if not identifier or not title:
|
|
127
|
+
return None, 1
|
|
128
|
+
if not term:
|
|
129
|
+
return None, 0
|
|
130
|
+
company = field(article, ".business") or "Job Bank"
|
|
131
|
+
city = field(article, ".location")
|
|
132
|
+
return (
|
|
133
|
+
Job(
|
|
134
|
+
id=job_id(self.name, "search", identifier),
|
|
135
|
+
source=self.name,
|
|
136
|
+
company=company,
|
|
137
|
+
title_raw=title,
|
|
138
|
+
title_normalized=title.lower(),
|
|
139
|
+
apply_url_raw=POSTING.format(id=identifier),
|
|
140
|
+
description="",
|
|
141
|
+
location_raw=city or "Canada",
|
|
142
|
+
first_seen=now,
|
|
143
|
+
last_seen=now,
|
|
144
|
+
signals=SourceSignals(employment_type=term.lower()),
|
|
145
|
+
),
|
|
146
|
+
0,
|
|
147
|
+
)
|
stage/sources/jobvite.py
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
import re
|
|
2
|
+
from collections.abc import Sequence
|
|
3
|
+
from datetime import datetime
|
|
4
|
+
from typing import Any, ClassVar
|
|
5
|
+
from urllib.parse import urljoin
|
|
6
|
+
|
|
7
|
+
from bs4 import Tag
|
|
8
|
+
from pydantic import BaseModel, ConfigDict
|
|
9
|
+
|
|
10
|
+
from stage.domain import Company, Job, Platform, job_id
|
|
11
|
+
from stage.http import HttpClient
|
|
12
|
+
from stage.sources import register
|
|
13
|
+
from stage.sources._text import collapse_whitespace
|
|
14
|
+
from stage.sources.base import (
|
|
15
|
+
BoardAdapter,
|
|
16
|
+
FetchResult,
|
|
17
|
+
NonEmptyStr,
|
|
18
|
+
PayloadValidationError,
|
|
19
|
+
capture_payload,
|
|
20
|
+
malformed_note,
|
|
21
|
+
validate_rows,
|
|
22
|
+
)
|
|
23
|
+
from stage.sources.custom_json import html_rows
|
|
24
|
+
|
|
25
|
+
HOST = "jobs.jobvite.com"
|
|
26
|
+
BASE_URL = "https://jobs.jobvite.com/{slug}/jobs"
|
|
27
|
+
LISTING = "table.jv-job-list"
|
|
28
|
+
ROW = "table.jv-job-list tbody tr"
|
|
29
|
+
NAME_CELL = "td.jv-job-list-name"
|
|
30
|
+
LOCATION_CELL = "td.jv-job-list-location"
|
|
31
|
+
ONCLICK = re.compile(r"""location\.href\s*=\s*['"](?P<href>[^'"]{1,300})['"]""")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class JobviteRow(BaseModel):
|
|
35
|
+
model_config = ConfigDict(extra="ignore")
|
|
36
|
+
|
|
37
|
+
title: NonEmptyStr
|
|
38
|
+
url: NonEmptyStr
|
|
39
|
+
location: str = ""
|
|
40
|
+
|
|
41
|
+
def posting_id(self) -> str:
|
|
42
|
+
return self.url.split("?", 1)[0].rstrip("/").rsplit("/", 1)[-1]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _text(block: Tag, selector: str) -> str:
|
|
46
|
+
found = block.select_one(selector)
|
|
47
|
+
return collapse_whitespace(found.get_text(" ", strip=True)) if found is not None else ""
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _href(block: Tag) -> str:
|
|
51
|
+
link = block.select_one(f"{NAME_CELL} a")
|
|
52
|
+
if link is not None:
|
|
53
|
+
return str(link.get("href", ""))
|
|
54
|
+
found = ONCLICK.search(str(block.get("onclick", "")))
|
|
55
|
+
return found.group("href") if found is not None else ""
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _row(block: Tag) -> dict[str, str]:
|
|
59
|
+
href = _href(block)
|
|
60
|
+
return {
|
|
61
|
+
"title": _text(block, NAME_CELL),
|
|
62
|
+
"url": urljoin(f"https://{HOST}/", href) if href else "",
|
|
63
|
+
"location": _text(block, LOCATION_CELL),
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@register
|
|
68
|
+
class JobviteAdapter(BoardAdapter):
|
|
69
|
+
name: ClassVar[str] = "jobvite"
|
|
70
|
+
platform: ClassVar[Platform] = Platform.JOBVITE
|
|
71
|
+
rate_profile: ClassVar[str] = "moderate"
|
|
72
|
+
bucket_key: ClassVar[str] = "jobvite"
|
|
73
|
+
hosts: ClassVar[frozenset[str]] = frozenset({HOST})
|
|
74
|
+
detail_budget: ClassVar[int] = 0
|
|
75
|
+
max_requests_per_company: ClassVar[int] = 1
|
|
76
|
+
|
|
77
|
+
base_url: ClassVar[str] = BASE_URL
|
|
78
|
+
row_model: ClassVar[type[BaseModel]] = JobviteRow
|
|
79
|
+
|
|
80
|
+
async def fetch(
|
|
81
|
+
self,
|
|
82
|
+
company: Company,
|
|
83
|
+
client: HttpClient,
|
|
84
|
+
now: datetime,
|
|
85
|
+
facets: object = None,
|
|
86
|
+
details: Sequence[str] = (),
|
|
87
|
+
) -> FetchResult:
|
|
88
|
+
response = await client.get_text(self.url_for(company))
|
|
89
|
+
if response.not_modified:
|
|
90
|
+
return FetchResult(not_modified=True)
|
|
91
|
+
return self.result(company, response.text, now)
|
|
92
|
+
|
|
93
|
+
def result(self, company: Company, payload: Any, now: datetime) -> FetchResult:
|
|
94
|
+
blocks = self._blocks(company, str(payload))
|
|
95
|
+
listed = [block for block in blocks if block.select_one(NAME_CELL) is not None]
|
|
96
|
+
rows, dropped = validate_rows(
|
|
97
|
+
self.row_model, [_row(block) for block in listed], source=self.name, slug=company.slug
|
|
98
|
+
)
|
|
99
|
+
truncated = len(listed) != len(blocks)
|
|
100
|
+
notes = [malformed_note(dropped)] if dropped else []
|
|
101
|
+
if truncated:
|
|
102
|
+
notes.append(
|
|
103
|
+
"the board hides the rest of some categories behind a Show More search page, "
|
|
104
|
+
"so this listing closes nothing"
|
|
105
|
+
)
|
|
106
|
+
return FetchResult(
|
|
107
|
+
jobs=tuple(self.to_job(company, row, now) for row in rows),
|
|
108
|
+
degraded="; ".join(note for note in notes if note),
|
|
109
|
+
authoritative=not dropped and not truncated,
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
def _blocks(self, company: Company, text: str) -> list[Tag]:
|
|
113
|
+
if not html_rows(text, LISTING):
|
|
114
|
+
captured = capture_payload(self.name, company.slug, {"text": text})
|
|
115
|
+
raise PayloadValidationError(
|
|
116
|
+
f"{self.name}/{company.slug}: the board page carries no {LISTING!r} listing, "
|
|
117
|
+
f"so this is drift or a retired tenant; raw page captured at {captured}"
|
|
118
|
+
)
|
|
119
|
+
return html_rows(text, ROW)
|
|
120
|
+
|
|
121
|
+
def to_job(self, company: Company, row: Any, now: datetime) -> Job:
|
|
122
|
+
return Job(
|
|
123
|
+
id=job_id(self.name, company.slug, row.posting_id()),
|
|
124
|
+
source=self.name,
|
|
125
|
+
company=company.name,
|
|
126
|
+
title_raw=row.title,
|
|
127
|
+
title_normalized=row.title.lower(),
|
|
128
|
+
apply_url_raw=row.url,
|
|
129
|
+
description="",
|
|
130
|
+
location_raw=row.location,
|
|
131
|
+
first_seen=now,
|
|
132
|
+
last_seen=now,
|
|
133
|
+
)
|
stage/sources/lever.py
ADDED
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
from datetime import UTC, datetime
|
|
2
|
+
from typing import Any, ClassVar
|
|
3
|
+
|
|
4
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
5
|
+
|
|
6
|
+
from stage.domain import Company, Job, Platform, SourceSignals, job_id
|
|
7
|
+
from stage.sources import register
|
|
8
|
+
from stage.sources._text import collapse_whitespace
|
|
9
|
+
from stage.sources.base import BoardAdapter
|
|
10
|
+
from stage.sources.platforms import safe_cased_slug
|
|
11
|
+
|
|
12
|
+
BASE_URL = "https://api.lever.co/v0/postings/{slug}"
|
|
13
|
+
HOST = "api.lever.co"
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class LeverCategories(BaseModel):
|
|
17
|
+
model_config = ConfigDict(extra="ignore")
|
|
18
|
+
|
|
19
|
+
location: str = ""
|
|
20
|
+
allLocations: list[str] = Field(default_factory=list)
|
|
21
|
+
commitment: str = ""
|
|
22
|
+
team: str = ""
|
|
23
|
+
department: str = ""
|
|
24
|
+
|
|
25
|
+
def label(self) -> str:
|
|
26
|
+
if self.allLocations:
|
|
27
|
+
return " / ".join(dict.fromkeys(self.allLocations))
|
|
28
|
+
return self.location
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class LeverPosting(BaseModel):
|
|
32
|
+
model_config = ConfigDict(extra="ignore")
|
|
33
|
+
|
|
34
|
+
id: str
|
|
35
|
+
text: str
|
|
36
|
+
hostedUrl: str = ""
|
|
37
|
+
applyUrl: str = ""
|
|
38
|
+
createdAt: int | None = None
|
|
39
|
+
categories: LeverCategories | None = None
|
|
40
|
+
descriptionPlain: str = ""
|
|
41
|
+
additionalPlain: str = ""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@register
|
|
45
|
+
class LeverAdapter(BoardAdapter):
|
|
46
|
+
name: ClassVar[str] = "lever"
|
|
47
|
+
platform: ClassVar[Platform] = Platform.LEVER
|
|
48
|
+
rate_profile: ClassVar[str] = "standard"
|
|
49
|
+
hosts: ClassVar[frozenset[str]] = frozenset({HOST})
|
|
50
|
+
detail_budget: ClassVar[int] = 0
|
|
51
|
+
max_requests_per_company: ClassVar[int] = 1
|
|
52
|
+
|
|
53
|
+
base_url: ClassVar[str] = BASE_URL
|
|
54
|
+
query: ClassVar[tuple[tuple[str, str], ...]] = (("mode", "json"),)
|
|
55
|
+
row_model: ClassVar[type[BaseModel]] = LeverPosting
|
|
56
|
+
slug_validator = safe_cased_slug
|
|
57
|
+
|
|
58
|
+
def to_job(self, company: Company, row: Any, now: datetime) -> Job:
|
|
59
|
+
posted = datetime.fromtimestamp(row.createdAt / 1000, tz=UTC) if row.createdAt else None
|
|
60
|
+
parts = (row.descriptionPlain, row.additionalPlain)
|
|
61
|
+
return Job(
|
|
62
|
+
id=job_id(self.name, company.slug, row.id),
|
|
63
|
+
source=self.name,
|
|
64
|
+
company=company.name,
|
|
65
|
+
title_raw=row.text,
|
|
66
|
+
title_normalized=collapse_whitespace(row.text),
|
|
67
|
+
apply_url_raw=row.hostedUrl or row.applyUrl,
|
|
68
|
+
description="\n\n".join(part for part in parts if part),
|
|
69
|
+
location_raw=collapse_whitespace(row.categories.label() if row.categories else ""),
|
|
70
|
+
first_seen=now,
|
|
71
|
+
last_seen=now,
|
|
72
|
+
source_posted_at=posted,
|
|
73
|
+
signals=SourceSignals(
|
|
74
|
+
employment_type=row.categories.commitment if row.categories else ""
|
|
75
|
+
),
|
|
76
|
+
)
|