stage-cli 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stage/__init__.py +1 -0
- stage/__main__.py +8 -0
- stage/banner.py +32 -0
- stage/bootstrap/__init__.py +0 -0
- stage/bootstrap/openjobs.py +392 -0
- stage/classify/__init__.py +29 -0
- stage/classify/eligibility.py +115 -0
- stage/classify/internship.py +64 -0
- stage/classify/role.py +91 -0
- stage/classify/scope.py +47 -0
- stage/cli/__init__.py +0 -0
- stage/cli/app.py +4 -0
- stage/cli/commands/__init__.py +8 -0
- stage/cli/commands/discovery.py +294 -0
- stage/cli/commands/insight.py +494 -0
- stage/cli/commands/pipeline.py +337 -0
- stage/cli/commands/postings.py +473 -0
- stage/cli/commands/schedule.py +171 -0
- stage/cli/housekeeping.py +64 -0
- stage/cli/logfile.py +56 -0
- stage/cli/notify.py +170 -0
- stage/cli/options.py +678 -0
- stage/cli/render.py +1398 -0
- stage/cli/runlock.py +74 -0
- stage/cli/schedule.py +702 -0
- stage/cli/schedule_state.py +363 -0
- stage/cli/selection.py +83 -0
- stage/cli/serialize.py +196 -0
- stage/companies.py +542 -0
- stage/data/companies/a.yaml +1289 -0
- stage/data/companies/b.yaml +900 -0
- stage/data/companies/c.yaml +1377 -0
- stage/data/companies/d.yaml +497 -0
- stage/data/companies/e.yaml +519 -0
- stage/data/companies/f.yaml +454 -0
- stage/data/companies/g.yaml +601 -0
- stage/data/companies/h.yaml +446 -0
- stage/data/companies/i.yaml +503 -0
- stage/data/companies/j.yaml +138 -0
- stage/data/companies/k.yaml +278 -0
- stage/data/companies/l.yaml +402 -0
- stage/data/companies/m.yaml +937 -0
- stage/data/companies/n.yaml +549 -0
- stage/data/companies/o.yaml +371 -0
- stage/data/companies/other.yaml +58 -0
- stage/data/companies/p.yaml +825 -0
- stage/data/companies/q.yaml +121 -0
- stage/data/companies/r.yaml +583 -0
- stage/data/companies/s.yaml +1140 -0
- stage/data/companies/t.yaml +817 -0
- stage/data/companies/u.yaml +196 -0
- stage/data/companies/v.yaml +325 -0
- stage/data/companies/w.yaml +353 -0
- stage/data/companies/x.yaml +67 -0
- stage/data/companies/y.yaml +36 -0
- stage/data/companies/z.yaml +146 -0
- stage/data/fonts/DejaVuSans.LICENSE.txt +99 -0
- stage/data/fonts/DejaVuSans.ttf +0 -0
- stage/data/lexicon/company_tokens.yaml +228 -0
- stage/data/lexicon/eligibility.yaml +455 -0
- stage/data/lexicon/inclusive_suffixes.yaml +37 -0
- stage/data/lexicon/internship.yaml +187 -0
- stage/data/lexicon/language.yaml +226 -0
- stage/data/lexicon/locations.yaml +1159 -0
- stage/data/lexicon/roles.yaml +2012 -0
- stage/data/lexicon/terms.yaml +76 -0
- stage/data/lexicon/workday_facets.yaml +27 -0
- stage/data/seed_companies.yaml +198 -0
- stage/dedup/__init__.py +19 -0
- stage/dedup/identity.py +113 -0
- stage/dedup/resolve.py +97 -0
- stage/domain/__init__.py +244 -0
- stage/domain/company.py +49 -0
- stage/domain/coverage.py +86 -0
- stage/domain/custom_board.py +92 -0
- stage/domain/discovery.py +94 -0
- stage/domain/enums.py +114 -0
- stage/domain/events.py +204 -0
- stage/domain/filters.py +27 -0
- stage/domain/health.py +169 -0
- stage/domain/ids.py +48 -0
- stage/domain/job.py +47 -0
- stage/domain/matching.py +15 -0
- stage/domain/priority.py +34 -0
- stage/domain/quarantine.py +39 -0
- stage/domain/rate_state.py +78 -0
- stage/domain/retention.py +20 -0
- stage/domain/rotation.py +46 -0
- stage/domain/signals.py +12 -0
- stage/domain/sync_run.py +35 -0
- stage/domain/text.py +113 -0
- stage/domain/validator.py +14 -0
- stage/domain/visits.py +60 -0
- stage/domain/workday.py +38 -0
- stage/http/__init__.py +58 -0
- stage/http/breaker.py +53 -0
- stage/http/cache.py +44 -0
- stage/http/client.py +725 -0
- stage/http/profiles.py +101 -0
- stage/lexicon.py +370 -0
- stage/normalize/__init__.py +16 -0
- stage/normalize/language.py +47 -0
- stage/normalize/location.py +271 -0
- stage/normalize/terms.py +153 -0
- stage/normalize/urls.py +122 -0
- stage/paths.py +86 -0
- stage/py.typed +0 -0
- stage/services/__init__.py +0 -0
- stage/services/canary.py +120 -0
- stage/services/coverage.py +231 -0
- stage/services/discover.py +747 -0
- stage/services/export.py +274 -0
- stage/services/health.py +237 -0
- stage/services/maintenance.py +225 -0
- stage/services/quarantine.py +20 -0
- stage/services/query.py +86 -0
- stage/services/sync.py +1257 -0
- stage/sources/__init__.py +82 -0
- stage/sources/_text.py +79 -0
- stage/sources/ashby.py +93 -0
- stage/sources/bamboohr.py +80 -0
- stage/sources/base.py +225 -0
- stage/sources/breezy.py +90 -0
- stage/sources/collage.py +60 -0
- stage/sources/community_feeds.py +142 -0
- stage/sources/curated_markdown.py +289 -0
- stage/sources/custom_json.py +610 -0
- stage/sources/espresso.py +154 -0
- stage/sources/feed.py +44 -0
- stage/sources/greenhouse.py +104 -0
- stage/sources/jobbank.py +147 -0
- stage/sources/jobvite.py +133 -0
- stage/sources/lever.py +76 -0
- stage/sources/oracle_cloud.py +187 -0
- stage/sources/platforms.py +609 -0
- stage/sources/quebec_emploi.py +146 -0
- stage/sources/recruitee.py +96 -0
- stage/sources/simplify.py +110 -0
- stage/sources/smartrecruiters.py +216 -0
- stage/sources/speedyapply.py +200 -0
- stage/sources/themuse.py +157 -0
- stage/sources/workable.py +83 -0
- stage/sources/workday.py +524 -0
- stage/sources/zshah.py +99 -0
- stage/storage/__init__.py +29 -0
- stage/storage/migrations/0001_initial.sql +239 -0
- stage/storage/migrations/__init__.py +135 -0
- stage/storage/repository.py +213 -0
- stage/storage/search.py +28 -0
- stage/storage/sqlite_repo.py +1586 -0
- stage/storage/writer.py +249 -0
- stage/tui/__init__.py +0 -0
- stage/tui/app.py +82 -0
- stage/tui/help.py +26 -0
- stage/tui/safe.py +21 -0
- stage/tui/screens/__init__.py +0 -0
- stage/tui/screens/boards.py +186 -0
- stage/tui/screens/postings.py +509 -0
- stage/tui/screens/review.py +209 -0
- stage/tui/screens/splash.py +37 -0
- stage/tui/screens/stats.py +124 -0
- stage/tui/screens/sync.py +194 -0
- stage/tui/state.py +160 -0
- stage/tui/theme.tcss +205 -0
- stage/tui/widgets/__init__.py +0 -0
- stage_cli-1.0.0.dist-info/METADATA +379 -0
- stage_cli-1.0.0.dist-info/RECORD +170 -0
- stage_cli-1.0.0.dist-info/WHEEL +4 -0
- stage_cli-1.0.0.dist-info/entry_points.txt +2 -0
- stage_cli-1.0.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
from datetime import datetime
|
|
2
|
+
from typing import Any, ClassVar
|
|
3
|
+
|
|
4
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
5
|
+
|
|
6
|
+
from stage.domain import Job, SourceSignals, job_id
|
|
7
|
+
from stage.http import HttpClient
|
|
8
|
+
from stage.sources import register_feed
|
|
9
|
+
from stage.sources._text import collapse_whitespace
|
|
10
|
+
from stage.sources.base import (
|
|
11
|
+
FetchResult,
|
|
12
|
+
NonEmptyStr,
|
|
13
|
+
PayloadValidationError,
|
|
14
|
+
capture_payload,
|
|
15
|
+
malformed_note,
|
|
16
|
+
validate_rows,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
HOST = "www.quebecemploi.gouv.qc.ca"
|
|
20
|
+
URL = f"https://{HOST}/search/postingFilteredAI"
|
|
21
|
+
PAGE_CAP = 4
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class QuebecEmploiListing(BaseModel):
|
|
25
|
+
model_config = ConfigDict(extra="ignore")
|
|
26
|
+
|
|
27
|
+
ide_affch: int = Field(gt=0)
|
|
28
|
+
titre: NonEmptyStr
|
|
29
|
+
employeur: str = ""
|
|
30
|
+
nom_ville: str = ""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class QuebecEmploiMeta(BaseModel):
|
|
34
|
+
model_config = ConfigDict(extra="ignore")
|
|
35
|
+
|
|
36
|
+
total_hits: int = Field(ge=0)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class QuebecEmploiPage(BaseModel):
|
|
40
|
+
model_config = ConfigDict(extra="ignore")
|
|
41
|
+
|
|
42
|
+
items: list[Any] = Field(default_factory=list)
|
|
43
|
+
meta: QuebecEmploiMeta
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@register_feed
|
|
47
|
+
class QuebecEmploiFeed:
|
|
48
|
+
name: ClassVar[str] = "quebec-emploi"
|
|
49
|
+
rate_profile: ClassVar[str] = "paginated"
|
|
50
|
+
hosts: ClassVar[frozenset[str]] = frozenset({HOST})
|
|
51
|
+
bucket_key: ClassVar[str] = "quebec-emploi"
|
|
52
|
+
|
|
53
|
+
def season_year(self, now: datetime) -> int:
|
|
54
|
+
return now.year
|
|
55
|
+
|
|
56
|
+
def plan(self, now: datetime) -> tuple[str, ...]:
|
|
57
|
+
return (URL,)
|
|
58
|
+
|
|
59
|
+
async def fetch(self, client: HttpClient, now: datetime) -> FetchResult:
|
|
60
|
+
listings: list[QuebecEmploiListing] = []
|
|
61
|
+
malformed = 0
|
|
62
|
+
truncated = False
|
|
63
|
+
|
|
64
|
+
for page in range(1, PAGE_CAP + 1):
|
|
65
|
+
response = await client.post_json(URL, body=self._request(page))
|
|
66
|
+
rows, dropped, total = self._validate(response.payload)
|
|
67
|
+
listings.extend(rows)
|
|
68
|
+
malformed += dropped
|
|
69
|
+
if len(listings) + malformed >= total:
|
|
70
|
+
break
|
|
71
|
+
else:
|
|
72
|
+
truncated = True
|
|
73
|
+
|
|
74
|
+
jobs = tuple(self._to_job(listing, now) for listing in listings)
|
|
75
|
+
notes = []
|
|
76
|
+
if truncated:
|
|
77
|
+
notes.append(f"stopped at Québec Emploi's {PAGE_CAP}-page public-search cap")
|
|
78
|
+
if malformed:
|
|
79
|
+
notes.append(malformed_note(malformed))
|
|
80
|
+
return FetchResult(
|
|
81
|
+
jobs=jobs,
|
|
82
|
+
authoritative=not (truncated or malformed),
|
|
83
|
+
degraded="; ".join(notes),
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
def _validate(self, payload: Any) -> tuple[list[QuebecEmploiListing], int, int]:
|
|
87
|
+
try:
|
|
88
|
+
page = QuebecEmploiPage.model_validate(payload)
|
|
89
|
+
except Exception as exc:
|
|
90
|
+
captured = capture_payload(self.name, "stages-students", payload)
|
|
91
|
+
raise PayloadValidationError(
|
|
92
|
+
f"{self.name}: public search response failed validation; raw payload captured at "
|
|
93
|
+
f"{captured}"
|
|
94
|
+
) from exc
|
|
95
|
+
rows, dropped = validate_rows(
|
|
96
|
+
QuebecEmploiListing, page.items, source=self.name, slug="stages-students"
|
|
97
|
+
)
|
|
98
|
+
return rows, dropped, page.meta.total_hits
|
|
99
|
+
|
|
100
|
+
def _to_job(self, listing: QuebecEmploiListing, now: datetime) -> Job:
|
|
101
|
+
title = collapse_whitespace(listing.titre)
|
|
102
|
+
company = collapse_whitespace(listing.employeur) or "Québec Emploi"
|
|
103
|
+
city = collapse_whitespace(listing.nom_ville)
|
|
104
|
+
location = f"{city}, Québec, Canada" if city else "Québec, Canada"
|
|
105
|
+
return Job(
|
|
106
|
+
id=job_id(self.name, "stages-students", str(listing.ide_affch)),
|
|
107
|
+
source=self.name,
|
|
108
|
+
company=company,
|
|
109
|
+
title_raw=title,
|
|
110
|
+
title_normalized=title.lower(),
|
|
111
|
+
apply_url_raw=(f"https://{HOST}/plateforme-emploi/poste/{listing.ide_affch}"),
|
|
112
|
+
description="",
|
|
113
|
+
location_raw=location,
|
|
114
|
+
first_seen=now,
|
|
115
|
+
last_seen=now,
|
|
116
|
+
signals=SourceSignals(employment_type="student stage"),
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
@staticmethod
|
|
120
|
+
def _request(page: int) -> dict[str, object]:
|
|
121
|
+
return {
|
|
122
|
+
"sort": {"type": "AUTO"},
|
|
123
|
+
"langue": "fr",
|
|
124
|
+
"page": page,
|
|
125
|
+
"identAWS": "stage-public-feed",
|
|
126
|
+
"filter": {
|
|
127
|
+
"inputSearch": "",
|
|
128
|
+
"address": "",
|
|
129
|
+
"localisation": {"longitude": "", "latitude": "", "distance": 20},
|
|
130
|
+
"adminRegion": [],
|
|
131
|
+
"offerType": ["2", "3"],
|
|
132
|
+
"commitment": [],
|
|
133
|
+
"jobDuration": [],
|
|
134
|
+
"levelEducation": [],
|
|
135
|
+
"studyDiscipline": [],
|
|
136
|
+
"mrc": [],
|
|
137
|
+
"bsq": [],
|
|
138
|
+
"scian": [],
|
|
139
|
+
"postedSince": "",
|
|
140
|
+
"excludeAgencies": False,
|
|
141
|
+
"isUkrainian": False,
|
|
142
|
+
"isExperimente": False,
|
|
143
|
+
"isSubsidized": False,
|
|
144
|
+
"isTrainingProgram": False,
|
|
145
|
+
},
|
|
146
|
+
}
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
from datetime import UTC, datetime
|
|
2
|
+
from typing import Any, ClassVar
|
|
3
|
+
|
|
4
|
+
from pydantic import BaseModel, ConfigDict
|
|
5
|
+
|
|
6
|
+
from stage.domain import Company, Job, Platform, SourceSignals, job_id
|
|
7
|
+
from stage.sources import register
|
|
8
|
+
from stage.sources._text import collapse_whitespace, strip_html
|
|
9
|
+
from stage.sources.base import BoardAdapter, NullableBool, NullableStr
|
|
10
|
+
|
|
11
|
+
HOST_TEMPLATE = "{slug}.recruitee.com"
|
|
12
|
+
PATH = "/api/offers/"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class RecruiteeOffer(BaseModel):
|
|
16
|
+
model_config = ConfigDict(extra="ignore")
|
|
17
|
+
|
|
18
|
+
id: int
|
|
19
|
+
title: str
|
|
20
|
+
slug: NullableStr = ""
|
|
21
|
+
description: NullableStr = ""
|
|
22
|
+
requirements: NullableStr = ""
|
|
23
|
+
location: NullableStr = ""
|
|
24
|
+
city: NullableStr = ""
|
|
25
|
+
country_code: NullableStr = ""
|
|
26
|
+
department: NullableStr = ""
|
|
27
|
+
employment_type_code: NullableStr = ""
|
|
28
|
+
careers_url: NullableStr = ""
|
|
29
|
+
careers_apply_url: NullableStr = ""
|
|
30
|
+
remote: NullableBool = False
|
|
31
|
+
published_at: NullableStr = ""
|
|
32
|
+
|
|
33
|
+
def where(self) -> str:
|
|
34
|
+
if self.location:
|
|
35
|
+
return self.location
|
|
36
|
+
parts = [part for part in (self.city, self.country_code) if part]
|
|
37
|
+
if not parts and self.remote:
|
|
38
|
+
return "Remote"
|
|
39
|
+
return ", ".join(parts)
|
|
40
|
+
|
|
41
|
+
def posted(self) -> datetime | None:
|
|
42
|
+
raw = self.published_at.strip()
|
|
43
|
+
if not raw:
|
|
44
|
+
return None
|
|
45
|
+
for pattern in ("%Y-%m-%d %H:%M:%S %Z", "%Y-%m-%d %H:%M:%S"):
|
|
46
|
+
try:
|
|
47
|
+
return datetime.strptime(raw, pattern).replace(tzinfo=UTC)
|
|
48
|
+
except ValueError:
|
|
49
|
+
continue
|
|
50
|
+
try:
|
|
51
|
+
return datetime.fromisoformat(raw)
|
|
52
|
+
except ValueError:
|
|
53
|
+
return None
|
|
54
|
+
|
|
55
|
+
def body(self) -> str:
|
|
56
|
+
joined = "\n\n".join(part for part in (self.description, self.requirements) if part)
|
|
57
|
+
return collapse_whitespace(strip_html(joined))
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class RecruiteeBoard(BaseModel):
|
|
61
|
+
model_config = ConfigDict(extra="ignore")
|
|
62
|
+
|
|
63
|
+
offers: list[Any]
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
@register
|
|
67
|
+
class RecruiteeAdapter(BoardAdapter):
|
|
68
|
+
name: ClassVar[str] = "recruitee"
|
|
69
|
+
platform: ClassVar[Platform] = Platform.RECRUITEE
|
|
70
|
+
rate_profile: ClassVar[str] = "moderate"
|
|
71
|
+
bucket_key: ClassVar[str] = "recruitee"
|
|
72
|
+
detail_budget: ClassVar[int] = 0
|
|
73
|
+
max_requests_per_company: ClassVar[int] = 1
|
|
74
|
+
|
|
75
|
+
host_template: ClassVar[str] = HOST_TEMPLATE
|
|
76
|
+
path: ClassVar[str] = PATH
|
|
77
|
+
root_model: ClassVar[type[BaseModel] | None] = RecruiteeBoard
|
|
78
|
+
rows_field: ClassVar[str] = "offers"
|
|
79
|
+
row_model: ClassVar[type[BaseModel]] = RecruiteeOffer
|
|
80
|
+
|
|
81
|
+
def to_job(self, company: Company, row: Any, now: datetime) -> Job:
|
|
82
|
+
title = collapse_whitespace(row.title)
|
|
83
|
+
return Job(
|
|
84
|
+
id=job_id(self.name, company.slug, str(row.id)),
|
|
85
|
+
source=self.name,
|
|
86
|
+
company=company.name,
|
|
87
|
+
title_raw=title,
|
|
88
|
+
title_normalized=title.lower(),
|
|
89
|
+
apply_url_raw=row.careers_url or row.careers_apply_url,
|
|
90
|
+
description=row.body(),
|
|
91
|
+
location_raw=collapse_whitespace(row.where()),
|
|
92
|
+
first_seen=now,
|
|
93
|
+
last_seen=now,
|
|
94
|
+
source_posted_at=row.posted(),
|
|
95
|
+
signals=SourceSignals(employment_type=row.employment_type_code),
|
|
96
|
+
)
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
from datetime import datetime
|
|
2
|
+
from typing import Any, ClassVar
|
|
3
|
+
|
|
4
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
5
|
+
|
|
6
|
+
from stage.domain import Job, SourceSignals, job_id
|
|
7
|
+
from stage.http import HttpClient
|
|
8
|
+
from stage.sources._text import collapse_whitespace, strip_html
|
|
9
|
+
from stage.sources.base import (
|
|
10
|
+
FetchResult,
|
|
11
|
+
NonEmptyStr,
|
|
12
|
+
PayloadValidationError,
|
|
13
|
+
capture_payload,
|
|
14
|
+
convert_rows,
|
|
15
|
+
malformed_note,
|
|
16
|
+
validate_rows,
|
|
17
|
+
)
|
|
18
|
+
from stage.sources.feed import register_feed, upcoming_season_year
|
|
19
|
+
|
|
20
|
+
HOST = "raw.githubusercontent.com"
|
|
21
|
+
LISTINGS_URL = (
|
|
22
|
+
"https://raw.githubusercontent.com/SimplifyJobs/Summer{year}-Internships/dev/.github/"
|
|
23
|
+
"scripts/listings.json"
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class SimplifyListing(BaseModel):
|
|
28
|
+
model_config = ConfigDict(extra="ignore")
|
|
29
|
+
|
|
30
|
+
id: NonEmptyStr
|
|
31
|
+
company_name: NonEmptyStr
|
|
32
|
+
title: NonEmptyStr
|
|
33
|
+
url: str = ""
|
|
34
|
+
locations: list[str] = Field(default_factory=list)
|
|
35
|
+
active: bool = True
|
|
36
|
+
is_visible: bool = True
|
|
37
|
+
date_posted: int | None = None
|
|
38
|
+
terms: list[str] = Field(default_factory=list)
|
|
39
|
+
sponsorship: str = ""
|
|
40
|
+
degrees: list[str] = Field(default_factory=list)
|
|
41
|
+
category: str = ""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@register_feed
|
|
45
|
+
class SimplifyFeed:
|
|
46
|
+
name: ClassVar[str] = "simplify"
|
|
47
|
+
rate_profile: ClassVar[str] = "feeds"
|
|
48
|
+
hosts: ClassVar[frozenset[str]] = frozenset({HOST})
|
|
49
|
+
bucket_key: ClassVar[str] = ""
|
|
50
|
+
|
|
51
|
+
def season_year(self, now: datetime) -> int:
|
|
52
|
+
return upcoming_season_year(now)
|
|
53
|
+
|
|
54
|
+
def plan(self, now: datetime) -> tuple[str, ...]:
|
|
55
|
+
return (LISTINGS_URL.format(year=self.season_year(now)),)
|
|
56
|
+
|
|
57
|
+
async def fetch(self, client: HttpClient, now: datetime) -> FetchResult:
|
|
58
|
+
response = await client.get_json(self.plan(now)[0])
|
|
59
|
+
if response.not_modified:
|
|
60
|
+
return FetchResult(not_modified=True)
|
|
61
|
+
listings, dropped = self._validate(response.payload, now)
|
|
62
|
+
jobs, unconvertible = convert_rows(
|
|
63
|
+
lambda listing: self._to_job(listing, now),
|
|
64
|
+
[listing for listing in listings if listing.active and listing.is_visible],
|
|
65
|
+
source=self.name,
|
|
66
|
+
slug=str(self.season_year(now)),
|
|
67
|
+
)
|
|
68
|
+
dropped += unconvertible
|
|
69
|
+
return FetchResult(
|
|
70
|
+
jobs=tuple(jobs),
|
|
71
|
+
degraded=malformed_note(dropped),
|
|
72
|
+
authoritative=not dropped,
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
def _validate(self, payload: Any, now: datetime) -> tuple[list[SimplifyListing], int]:
|
|
76
|
+
if not isinstance(payload, list):
|
|
77
|
+
captured = capture_payload(self.name, str(self.season_year(now)), payload)
|
|
78
|
+
raise PayloadValidationError(
|
|
79
|
+
f"simplify/{self.season_year(now)}: field '<root>' failed validation "
|
|
80
|
+
f"(expected a JSON list of listings); raw payload captured at {captured}"
|
|
81
|
+
)
|
|
82
|
+
return validate_rows(
|
|
83
|
+
SimplifyListing, payload, source=self.name, slug=str(self.season_year(now))
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
def _to_job(self, listing: SimplifyListing, now: datetime) -> Job:
|
|
87
|
+
posted = (
|
|
88
|
+
datetime.fromtimestamp(listing.date_posted, tz=now.tzinfo)
|
|
89
|
+
if listing.date_posted
|
|
90
|
+
else None
|
|
91
|
+
)
|
|
92
|
+
return Job(
|
|
93
|
+
id=job_id(self.name, listing.company_name, listing.id),
|
|
94
|
+
source=self.name,
|
|
95
|
+
company=collapse_whitespace(listing.company_name),
|
|
96
|
+
title_raw=listing.title,
|
|
97
|
+
title_normalized=collapse_whitespace(listing.title),
|
|
98
|
+
apply_url_raw=listing.url,
|
|
99
|
+
description=strip_html(""),
|
|
100
|
+
location_raw=collapse_whitespace(" / ".join(listing.locations)),
|
|
101
|
+
first_seen=now,
|
|
102
|
+
last_seen=now,
|
|
103
|
+
source_posted_at=posted,
|
|
104
|
+
signals=SourceSignals(
|
|
105
|
+
terms=tuple(listing.terms),
|
|
106
|
+
sponsorship=listing.sponsorship,
|
|
107
|
+
degrees=tuple(listing.degrees),
|
|
108
|
+
category=listing.category,
|
|
109
|
+
),
|
|
110
|
+
)
|
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
from collections.abc import Sequence
|
|
2
|
+
from dataclasses import replace
|
|
3
|
+
from datetime import datetime
|
|
4
|
+
from typing import Any, ClassVar
|
|
5
|
+
|
|
6
|
+
from pydantic import BaseModel, ConfigDict, ValidationError
|
|
7
|
+
|
|
8
|
+
from stage.domain import Company, DetailFetch, Job, Platform, board_key, job_id
|
|
9
|
+
from stage.http import HttpClient, HttpError
|
|
10
|
+
from stage.sources import register
|
|
11
|
+
from stage.sources._text import collapse_whitespace, strip_html
|
|
12
|
+
from stage.sources.base import (
|
|
13
|
+
FetchResult,
|
|
14
|
+
PayloadValidationError,
|
|
15
|
+
capture_payload,
|
|
16
|
+
malformed_note,
|
|
17
|
+
validate_rows,
|
|
18
|
+
)
|
|
19
|
+
from stage.sources.platforms import safe_slug
|
|
20
|
+
|
|
21
|
+
BASE_URL = "https://api.smartrecruiters.com/v1/companies/{slug}/postings"
|
|
22
|
+
DETAIL_URL = "https://api.smartrecruiters.com/v1/companies/{slug}/postings/{posting}"
|
|
23
|
+
HOST = "api.smartrecruiters.com"
|
|
24
|
+
PAGE_SIZE = 100
|
|
25
|
+
MAX_PAGES = 30
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class SmartRecruitersLocation(BaseModel):
|
|
29
|
+
model_config = ConfigDict(extra="ignore")
|
|
30
|
+
|
|
31
|
+
city: str = ""
|
|
32
|
+
region: str = ""
|
|
33
|
+
country: str = ""
|
|
34
|
+
remote: bool = False
|
|
35
|
+
fullLocation: str = ""
|
|
36
|
+
|
|
37
|
+
def label(self) -> str:
|
|
38
|
+
if self.fullLocation:
|
|
39
|
+
return self.fullLocation
|
|
40
|
+
parts = [part for part in (self.city, self.region, self.country) if part]
|
|
41
|
+
return ", ".join(parts) or ("Remote" if self.remote else "")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class SmartRecruitersPosting(BaseModel):
|
|
45
|
+
model_config = ConfigDict(extra="ignore")
|
|
46
|
+
|
|
47
|
+
id: str
|
|
48
|
+
name: str
|
|
49
|
+
releasedDate: datetime | None = None
|
|
50
|
+
location: SmartRecruitersLocation | None = None
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class SmartRecruitersPage(BaseModel):
|
|
54
|
+
model_config = ConfigDict(extra="ignore")
|
|
55
|
+
|
|
56
|
+
totalFound: int
|
|
57
|
+
content: list[Any]
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@register
|
|
61
|
+
class SmartRecruitersAdapter:
|
|
62
|
+
name: ClassVar[str] = "smartrecruiters"
|
|
63
|
+
platform: ClassVar[Platform] = Platform.SMARTRECRUITERS
|
|
64
|
+
rate_profile: ClassVar[str] = "paginated"
|
|
65
|
+
hosts: ClassVar[frozenset[str]] = frozenset({HOST})
|
|
66
|
+
bucket_key: ClassVar[str] = ""
|
|
67
|
+
detail_budget: ClassVar[int] = 120
|
|
68
|
+
rotation_slice: ClassVar[int] = 0
|
|
69
|
+
|
|
70
|
+
max_requests_per_company: ClassVar[int] = MAX_PAGES
|
|
71
|
+
|
|
72
|
+
def hosts_for(self, companies: Sequence[Company]) -> frozenset[str]:
|
|
73
|
+
return self.hosts
|
|
74
|
+
|
|
75
|
+
def board_key(self, company: Company) -> str:
|
|
76
|
+
return board_key(self.name, company.slug)
|
|
77
|
+
|
|
78
|
+
def plan(self, company: Company) -> tuple[str, ...]:
|
|
79
|
+
return (f"{BASE_URL.format(slug=safe_slug(company.slug))}?limit={PAGE_SIZE}&offset=0",)
|
|
80
|
+
|
|
81
|
+
async def fetch(
|
|
82
|
+
self,
|
|
83
|
+
company: Company,
|
|
84
|
+
client: HttpClient,
|
|
85
|
+
now: datetime,
|
|
86
|
+
facets: object = None,
|
|
87
|
+
details: Sequence[str] = (),
|
|
88
|
+
) -> FetchResult:
|
|
89
|
+
url = BASE_URL.format(slug=safe_slug(company.slug))
|
|
90
|
+
postings: list[SmartRecruitersPosting] = []
|
|
91
|
+
truncated = False
|
|
92
|
+
stale_page = False
|
|
93
|
+
malformed = 0
|
|
94
|
+
|
|
95
|
+
wanted = {job for job in details if job.startswith(f"{self.board_key(company)}:")}
|
|
96
|
+
|
|
97
|
+
for page in range(MAX_PAGES):
|
|
98
|
+
response = await client.get_json(
|
|
99
|
+
url,
|
|
100
|
+
params={"limit": str(PAGE_SIZE), "offset": str(page * PAGE_SIZE)},
|
|
101
|
+
revalidate=bool(wanted),
|
|
102
|
+
)
|
|
103
|
+
if response.not_modified:
|
|
104
|
+
if page == 0:
|
|
105
|
+
return FetchResult(not_modified=True)
|
|
106
|
+
stale_page = True
|
|
107
|
+
break
|
|
108
|
+
rows, dropped, total = self._validate(company, response.payload)
|
|
109
|
+
malformed += dropped
|
|
110
|
+
if not rows and not dropped:
|
|
111
|
+
break
|
|
112
|
+
postings.extend(rows)
|
|
113
|
+
if len(postings) + malformed >= total:
|
|
114
|
+
break
|
|
115
|
+
else:
|
|
116
|
+
truncated = True
|
|
117
|
+
|
|
118
|
+
notes = []
|
|
119
|
+
if truncated:
|
|
120
|
+
notes.append(f"stopped at the {MAX_PAGES}-page cap")
|
|
121
|
+
if stale_page:
|
|
122
|
+
notes.append(
|
|
123
|
+
"a later page answered 304, so the walk ended early on an unchanged page "
|
|
124
|
+
"rather than on the end of the list"
|
|
125
|
+
)
|
|
126
|
+
if malformed:
|
|
127
|
+
notes.append(malformed_note(malformed))
|
|
128
|
+
paired = [(posting, self._to_job(company, posting, now)) for posting in postings]
|
|
129
|
+
fetched: list[DetailFetch] = []
|
|
130
|
+
if wanted:
|
|
131
|
+
paired, fetched = await self._attach_descriptions(company, client, paired, wanted)
|
|
132
|
+
jobs = [job for _, job in paired]
|
|
133
|
+
|
|
134
|
+
return FetchResult(
|
|
135
|
+
jobs=tuple(jobs),
|
|
136
|
+
degraded="; ".join(notes),
|
|
137
|
+
authoritative=not (truncated or stale_page or malformed),
|
|
138
|
+
detail_fetches=tuple(fetched),
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
async def _attach_descriptions(
|
|
142
|
+
self,
|
|
143
|
+
company: Company,
|
|
144
|
+
client: HttpClient,
|
|
145
|
+
paired: list[tuple[SmartRecruitersPosting, Job]],
|
|
146
|
+
wanted: set[str],
|
|
147
|
+
) -> tuple[list[tuple[SmartRecruitersPosting, Job]], list[DetailFetch]]:
|
|
148
|
+
outcomes: list[DetailFetch] = []
|
|
149
|
+
merged: list[tuple[SmartRecruitersPosting, Job]] = []
|
|
150
|
+
for posting, job in paired:
|
|
151
|
+
if job.id not in wanted:
|
|
152
|
+
merged.append((posting, job))
|
|
153
|
+
continue
|
|
154
|
+
url = DETAIL_URL.format(slug=safe_slug(company.slug), posting=posting.id)
|
|
155
|
+
try:
|
|
156
|
+
response = await client.get_json(url)
|
|
157
|
+
except HttpError:
|
|
158
|
+
outcomes.append(DetailFetch(id=job.id, resolved=False, failed=True))
|
|
159
|
+
merged.append((posting, job))
|
|
160
|
+
continue
|
|
161
|
+
body = _description_from(response.payload)
|
|
162
|
+
outcomes.append(DetailFetch(id=job.id, resolved=bool(body)))
|
|
163
|
+
merged.append((posting, replace(job, description=body) if body else job))
|
|
164
|
+
return merged, outcomes
|
|
165
|
+
|
|
166
|
+
def _validate(
|
|
167
|
+
self, company: Company, payload: Any
|
|
168
|
+
) -> tuple[list[SmartRecruitersPosting], int, int]:
|
|
169
|
+
page = self._validate_page(company, payload)
|
|
170
|
+
rows, dropped = validate_rows(
|
|
171
|
+
SmartRecruitersPosting, page.content, source=self.name, slug=company.slug
|
|
172
|
+
)
|
|
173
|
+
return rows, dropped, page.totalFound
|
|
174
|
+
|
|
175
|
+
def _validate_page(self, company: Company, payload: Any) -> SmartRecruitersPage:
|
|
176
|
+
try:
|
|
177
|
+
return SmartRecruitersPage.model_validate(payload)
|
|
178
|
+
except ValidationError as exc:
|
|
179
|
+
captured = capture_payload(self.name, company.slug, payload)
|
|
180
|
+
first = exc.errors()[0]
|
|
181
|
+
field = ".".join(str(part) for part in first["loc"]) or "<root>"
|
|
182
|
+
raise PayloadValidationError(
|
|
183
|
+
f"smartrecruiters/{company.slug}: field {field!r} failed validation "
|
|
184
|
+
f"({first['msg']}); raw payload captured at {captured}"
|
|
185
|
+
) from exc
|
|
186
|
+
|
|
187
|
+
def _to_job(self, company: Company, posting: SmartRecruitersPosting, now: datetime) -> Job:
|
|
188
|
+
return Job(
|
|
189
|
+
id=job_id(self.name, company.slug, posting.id),
|
|
190
|
+
source=self.name,
|
|
191
|
+
company=company.name,
|
|
192
|
+
title_raw=posting.name,
|
|
193
|
+
title_normalized=collapse_whitespace(posting.name),
|
|
194
|
+
apply_url_raw=f"https://jobs.smartrecruiters.com/{company.slug}/{posting.id}",
|
|
195
|
+
description="",
|
|
196
|
+
location_raw=collapse_whitespace(posting.location.label() if posting.location else ""),
|
|
197
|
+
first_seen=now,
|
|
198
|
+
last_seen=now,
|
|
199
|
+
source_posted_at=posting.releasedDate,
|
|
200
|
+
)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _description_from(payload: Any) -> str:
|
|
204
|
+
if not isinstance(payload, dict):
|
|
205
|
+
return ""
|
|
206
|
+
ad = payload.get("jobAd")
|
|
207
|
+
sections = ad.get("sections") if isinstance(ad, dict) else None
|
|
208
|
+
if not isinstance(sections, dict):
|
|
209
|
+
return ""
|
|
210
|
+
parts: list[str] = []
|
|
211
|
+
for section in sections.values():
|
|
212
|
+
if isinstance(section, dict):
|
|
213
|
+
text = section.get("text")
|
|
214
|
+
if isinstance(text, str) and text.strip():
|
|
215
|
+
parts.append(strip_html(text))
|
|
216
|
+
return collapse_whitespace(" ".join(parts))
|