stage-cli 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stage/__init__.py +1 -0
- stage/__main__.py +8 -0
- stage/banner.py +32 -0
- stage/bootstrap/__init__.py +0 -0
- stage/bootstrap/openjobs.py +392 -0
- stage/classify/__init__.py +29 -0
- stage/classify/eligibility.py +115 -0
- stage/classify/internship.py +64 -0
- stage/classify/role.py +91 -0
- stage/classify/scope.py +47 -0
- stage/cli/__init__.py +0 -0
- stage/cli/app.py +4 -0
- stage/cli/commands/__init__.py +8 -0
- stage/cli/commands/discovery.py +294 -0
- stage/cli/commands/insight.py +494 -0
- stage/cli/commands/pipeline.py +337 -0
- stage/cli/commands/postings.py +473 -0
- stage/cli/commands/schedule.py +171 -0
- stage/cli/housekeeping.py +64 -0
- stage/cli/logfile.py +56 -0
- stage/cli/notify.py +170 -0
- stage/cli/options.py +678 -0
- stage/cli/render.py +1398 -0
- stage/cli/runlock.py +74 -0
- stage/cli/schedule.py +702 -0
- stage/cli/schedule_state.py +363 -0
- stage/cli/selection.py +83 -0
- stage/cli/serialize.py +196 -0
- stage/companies.py +542 -0
- stage/data/companies/a.yaml +1289 -0
- stage/data/companies/b.yaml +900 -0
- stage/data/companies/c.yaml +1377 -0
- stage/data/companies/d.yaml +497 -0
- stage/data/companies/e.yaml +519 -0
- stage/data/companies/f.yaml +454 -0
- stage/data/companies/g.yaml +601 -0
- stage/data/companies/h.yaml +446 -0
- stage/data/companies/i.yaml +503 -0
- stage/data/companies/j.yaml +138 -0
- stage/data/companies/k.yaml +278 -0
- stage/data/companies/l.yaml +402 -0
- stage/data/companies/m.yaml +937 -0
- stage/data/companies/n.yaml +549 -0
- stage/data/companies/o.yaml +371 -0
- stage/data/companies/other.yaml +58 -0
- stage/data/companies/p.yaml +825 -0
- stage/data/companies/q.yaml +121 -0
- stage/data/companies/r.yaml +583 -0
- stage/data/companies/s.yaml +1140 -0
- stage/data/companies/t.yaml +817 -0
- stage/data/companies/u.yaml +196 -0
- stage/data/companies/v.yaml +325 -0
- stage/data/companies/w.yaml +353 -0
- stage/data/companies/x.yaml +67 -0
- stage/data/companies/y.yaml +36 -0
- stage/data/companies/z.yaml +146 -0
- stage/data/fonts/DejaVuSans.LICENSE.txt +99 -0
- stage/data/fonts/DejaVuSans.ttf +0 -0
- stage/data/lexicon/company_tokens.yaml +228 -0
- stage/data/lexicon/eligibility.yaml +455 -0
- stage/data/lexicon/inclusive_suffixes.yaml +37 -0
- stage/data/lexicon/internship.yaml +187 -0
- stage/data/lexicon/language.yaml +226 -0
- stage/data/lexicon/locations.yaml +1159 -0
- stage/data/lexicon/roles.yaml +2012 -0
- stage/data/lexicon/terms.yaml +76 -0
- stage/data/lexicon/workday_facets.yaml +27 -0
- stage/data/seed_companies.yaml +198 -0
- stage/dedup/__init__.py +19 -0
- stage/dedup/identity.py +113 -0
- stage/dedup/resolve.py +97 -0
- stage/domain/__init__.py +244 -0
- stage/domain/company.py +49 -0
- stage/domain/coverage.py +86 -0
- stage/domain/custom_board.py +92 -0
- stage/domain/discovery.py +94 -0
- stage/domain/enums.py +114 -0
- stage/domain/events.py +204 -0
- stage/domain/filters.py +27 -0
- stage/domain/health.py +169 -0
- stage/domain/ids.py +48 -0
- stage/domain/job.py +47 -0
- stage/domain/matching.py +15 -0
- stage/domain/priority.py +34 -0
- stage/domain/quarantine.py +39 -0
- stage/domain/rate_state.py +78 -0
- stage/domain/retention.py +20 -0
- stage/domain/rotation.py +46 -0
- stage/domain/signals.py +12 -0
- stage/domain/sync_run.py +35 -0
- stage/domain/text.py +113 -0
- stage/domain/validator.py +14 -0
- stage/domain/visits.py +60 -0
- stage/domain/workday.py +38 -0
- stage/http/__init__.py +58 -0
- stage/http/breaker.py +53 -0
- stage/http/cache.py +44 -0
- stage/http/client.py +725 -0
- stage/http/profiles.py +101 -0
- stage/lexicon.py +370 -0
- stage/normalize/__init__.py +16 -0
- stage/normalize/language.py +47 -0
- stage/normalize/location.py +271 -0
- stage/normalize/terms.py +153 -0
- stage/normalize/urls.py +122 -0
- stage/paths.py +86 -0
- stage/py.typed +0 -0
- stage/services/__init__.py +0 -0
- stage/services/canary.py +120 -0
- stage/services/coverage.py +231 -0
- stage/services/discover.py +747 -0
- stage/services/export.py +274 -0
- stage/services/health.py +237 -0
- stage/services/maintenance.py +225 -0
- stage/services/quarantine.py +20 -0
- stage/services/query.py +86 -0
- stage/services/sync.py +1257 -0
- stage/sources/__init__.py +82 -0
- stage/sources/_text.py +79 -0
- stage/sources/ashby.py +93 -0
- stage/sources/bamboohr.py +80 -0
- stage/sources/base.py +225 -0
- stage/sources/breezy.py +90 -0
- stage/sources/collage.py +60 -0
- stage/sources/community_feeds.py +142 -0
- stage/sources/curated_markdown.py +289 -0
- stage/sources/custom_json.py +610 -0
- stage/sources/espresso.py +154 -0
- stage/sources/feed.py +44 -0
- stage/sources/greenhouse.py +104 -0
- stage/sources/jobbank.py +147 -0
- stage/sources/jobvite.py +133 -0
- stage/sources/lever.py +76 -0
- stage/sources/oracle_cloud.py +187 -0
- stage/sources/platforms.py +609 -0
- stage/sources/quebec_emploi.py +146 -0
- stage/sources/recruitee.py +96 -0
- stage/sources/simplify.py +110 -0
- stage/sources/smartrecruiters.py +216 -0
- stage/sources/speedyapply.py +200 -0
- stage/sources/themuse.py +157 -0
- stage/sources/workable.py +83 -0
- stage/sources/workday.py +524 -0
- stage/sources/zshah.py +99 -0
- stage/storage/__init__.py +29 -0
- stage/storage/migrations/0001_initial.sql +239 -0
- stage/storage/migrations/__init__.py +135 -0
- stage/storage/repository.py +213 -0
- stage/storage/search.py +28 -0
- stage/storage/sqlite_repo.py +1586 -0
- stage/storage/writer.py +249 -0
- stage/tui/__init__.py +0 -0
- stage/tui/app.py +82 -0
- stage/tui/help.py +26 -0
- stage/tui/safe.py +21 -0
- stage/tui/screens/__init__.py +0 -0
- stage/tui/screens/boards.py +186 -0
- stage/tui/screens/postings.py +509 -0
- stage/tui/screens/review.py +209 -0
- stage/tui/screens/splash.py +37 -0
- stage/tui/screens/stats.py +124 -0
- stage/tui/screens/sync.py +194 -0
- stage/tui/state.py +160 -0
- stage/tui/theme.tcss +205 -0
- stage/tui/widgets/__init__.py +0 -0
- stage_cli-1.0.0.dist-info/METADATA +379 -0
- stage_cli-1.0.0.dist-info/RECORD +170 -0
- stage_cli-1.0.0.dist-info/WHEEL +4 -0
- stage_cli-1.0.0.dist-info/entry_points.txt +2 -0
- stage_cli-1.0.0.dist-info/licenses/LICENSE +21 -0
stage/sources/workday.py
ADDED
|
@@ -0,0 +1,524 @@
|
|
|
1
|
+
from collections.abc import Mapping, Sequence
|
|
2
|
+
from dataclasses import replace
|
|
3
|
+
from datetime import datetime
|
|
4
|
+
from typing import Any, ClassVar
|
|
5
|
+
|
|
6
|
+
from pydantic import BaseModel, ConfigDict, Field, ValidationError
|
|
7
|
+
|
|
8
|
+
from stage.domain import (
|
|
9
|
+
Company,
|
|
10
|
+
DetailFetch,
|
|
11
|
+
Job,
|
|
12
|
+
Platform,
|
|
13
|
+
SourceSignals,
|
|
14
|
+
WorkdayCrawl,
|
|
15
|
+
WorkdayCrawlStep,
|
|
16
|
+
WorkdayFacet,
|
|
17
|
+
board_key,
|
|
18
|
+
job_id,
|
|
19
|
+
)
|
|
20
|
+
from stage.http import HttpClient, HttpError
|
|
21
|
+
from stage.lexicon import fold, internship_lexicon, workday_facet_lexicon
|
|
22
|
+
from stage.sources import register
|
|
23
|
+
from stage.sources._text import collapse_whitespace, strip_html
|
|
24
|
+
from stage.sources.base import (
|
|
25
|
+
FetchResult,
|
|
26
|
+
PayloadValidationError,
|
|
27
|
+
capture_payload,
|
|
28
|
+
malformed_note,
|
|
29
|
+
validate_rows,
|
|
30
|
+
)
|
|
31
|
+
from stage.sources.platforms import SlugRejectedError, workday_target
|
|
32
|
+
|
|
33
|
+
PAGE_SIZE = 20
|
|
34
|
+
MAX_PAGES = 25
|
|
35
|
+
MAX_CRAWL_PAGES = 100
|
|
36
|
+
RESULT_CAP = 10_000
|
|
37
|
+
CRAWL_PAGE_CAP = 6
|
|
38
|
+
RETRY_RESERVE = 40
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class WorkdayPosting(BaseModel):
|
|
42
|
+
model_config = ConfigDict(extra="ignore")
|
|
43
|
+
|
|
44
|
+
title: str
|
|
45
|
+
externalPath: str = ""
|
|
46
|
+
locationsText: str = ""
|
|
47
|
+
postedOn: str = ""
|
|
48
|
+
bulletFields: list[str] = Field(default_factory=list)
|
|
49
|
+
|
|
50
|
+
def requisition(self) -> str:
|
|
51
|
+
for field in self.bulletFields:
|
|
52
|
+
cleaned = field.strip()
|
|
53
|
+
if cleaned:
|
|
54
|
+
return cleaned
|
|
55
|
+
tail = self.externalPath.rstrip("/").rsplit("_", 1)
|
|
56
|
+
return tail[-1] if len(tail) == 2 and tail[-1] else self.externalPath.rstrip("/")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class WorkdayFacetValue(BaseModel):
|
|
60
|
+
model_config = ConfigDict(extra="ignore")
|
|
61
|
+
|
|
62
|
+
id: str = ""
|
|
63
|
+
descriptor: str = ""
|
|
64
|
+
count: int = 0
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class WorkdayFacetGroup(BaseModel):
|
|
68
|
+
model_config = ConfigDict(extra="ignore")
|
|
69
|
+
|
|
70
|
+
facetParameter: str = ""
|
|
71
|
+
descriptor: str = ""
|
|
72
|
+
values: list[WorkdayFacetValue] = Field(default_factory=list)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class WorkdayPage(BaseModel):
|
|
76
|
+
model_config = ConfigDict(extra="ignore")
|
|
77
|
+
|
|
78
|
+
total: int = 0
|
|
79
|
+
jobPostings: list[WorkdayPosting]
|
|
80
|
+
facets: list[WorkdayFacetGroup] | None = None
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class WorkdayRawPage(BaseModel):
|
|
84
|
+
model_config = ConfigDict(extra="ignore")
|
|
85
|
+
|
|
86
|
+
total: int = 0
|
|
87
|
+
jobPostings: list[dict[str, Any]]
|
|
88
|
+
facets: list[WorkdayFacetGroup] | None = None
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _matches(folded: str, descriptors: frozenset[str]) -> bool:
|
|
92
|
+
padded = f" {folded} "
|
|
93
|
+
if any(f" {phrase} " in padded for phrase in internship_lexicon().blocked_bigrams):
|
|
94
|
+
return False
|
|
95
|
+
return any(f" {phrase} " in padded for phrase in descriptors)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def resolve_facet(page: WorkdayPage, tenant: str, site: str, now: datetime) -> WorkdayFacet | None:
|
|
99
|
+
parameters, descriptors = workday_facet_lexicon()
|
|
100
|
+
groups = {group.facetParameter: group for group in page.facets or ()}
|
|
101
|
+
for parameter in parameters:
|
|
102
|
+
group = groups.get(parameter)
|
|
103
|
+
if group is None:
|
|
104
|
+
continue
|
|
105
|
+
matched = [
|
|
106
|
+
value
|
|
107
|
+
for value in group.values
|
|
108
|
+
if value.id and _matches(fold(value.descriptor), descriptors)
|
|
109
|
+
]
|
|
110
|
+
if matched:
|
|
111
|
+
return WorkdayFacet(
|
|
112
|
+
tenant=tenant,
|
|
113
|
+
site=site,
|
|
114
|
+
parameter=parameter,
|
|
115
|
+
facet_ids=tuple(value.id for value in matched),
|
|
116
|
+
descriptor=", ".join(value.descriptor for value in matched),
|
|
117
|
+
resolved_at=now,
|
|
118
|
+
)
|
|
119
|
+
return None
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def facet_still_offered(page: WorkdayPage, facet: WorkdayFacet) -> bool:
|
|
123
|
+
if page.facets is None:
|
|
124
|
+
return True
|
|
125
|
+
for group in page.facets:
|
|
126
|
+
if group.facetParameter != facet.parameter:
|
|
127
|
+
continue
|
|
128
|
+
offered = {value.id for value in group.values}
|
|
129
|
+
if all(facet_id in offered for facet_id in facet.facet_ids):
|
|
130
|
+
return True
|
|
131
|
+
return False
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
@register
|
|
135
|
+
class WorkdayAdapter:
|
|
136
|
+
name: ClassVar[str] = "workday"
|
|
137
|
+
platform: ClassVar[Platform] = Platform.WORKDAY
|
|
138
|
+
rate_profile: ClassVar[str] = "workday"
|
|
139
|
+
hosts: ClassVar[frozenset[str]] = frozenset()
|
|
140
|
+
bucket_key: ClassVar[str] = "workday"
|
|
141
|
+
|
|
142
|
+
detail_budget: ClassVar[int] = 60
|
|
143
|
+
|
|
144
|
+
rotation_slice: ClassVar[int] = 300
|
|
145
|
+
|
|
146
|
+
max_requests_per_company: ClassVar[int] = MAX_PAGES
|
|
147
|
+
|
|
148
|
+
crawl_page_cap: ClassVar[int] = CRAWL_PAGE_CAP
|
|
149
|
+
retry_reserve: ClassVar[int] = RETRY_RESERVE
|
|
150
|
+
|
|
151
|
+
@classmethod
|
|
152
|
+
def crawl_budgets(
|
|
153
|
+
cls,
|
|
154
|
+
companies: Sequence[Company],
|
|
155
|
+
crawls: Mapping[str, WorkdayCrawl],
|
|
156
|
+
facets: Mapping[tuple[str, str], WorkdayFacet],
|
|
157
|
+
ceiling: int,
|
|
158
|
+
) -> tuple[dict[str, int], int]:
|
|
159
|
+
if not companies:
|
|
160
|
+
raise ValueError("a Workday crawl needs at least one board")
|
|
161
|
+
available = max(1, ceiling - cls.retry_reserve)
|
|
162
|
+
budgets = {company.registry_key: 1 for company in companies}
|
|
163
|
+
remaining = max(0, available - len(budgets))
|
|
164
|
+
demands: list[tuple[int, str]] = []
|
|
165
|
+
for company in companies:
|
|
166
|
+
crawl = crawls.get(board_key(cls.name, _board(company)))
|
|
167
|
+
facet = _pinned_facet(company) or facets.get(
|
|
168
|
+
(company.workday_tenant or "", company.workday_site or "")
|
|
169
|
+
)
|
|
170
|
+
if crawl is None or crawl.total is None:
|
|
171
|
+
continue
|
|
172
|
+
if crawl.facet_parameter != (facet.parameter if facet is not None else ""):
|
|
173
|
+
continue
|
|
174
|
+
if crawl.facet_ids != (facet.facet_ids if facet is not None else ()):
|
|
175
|
+
continue
|
|
176
|
+
start = max(0, crawl.next_offset - PAGE_SIZE)
|
|
177
|
+
pages = max(1, (max(0, crawl.total - start) + PAGE_SIZE - 1) // PAGE_SIZE)
|
|
178
|
+
demands.append((min(MAX_CRAWL_PAGES, pages), company.registry_key))
|
|
179
|
+
for pages, key in sorted(demands, key=lambda item: (-item[0], item[1])):
|
|
180
|
+
extra = max(0, min(remaining, pages - budgets[key]))
|
|
181
|
+
budgets[key] += extra
|
|
182
|
+
remaining -= extra
|
|
183
|
+
for company in companies:
|
|
184
|
+
key = company.registry_key
|
|
185
|
+
extra = max(0, min(remaining, cls.crawl_page_cap - budgets[key]))
|
|
186
|
+
budgets[key] += extra
|
|
187
|
+
remaining -= extra
|
|
188
|
+
details = min(cls.detail_budget, max(0, remaining))
|
|
189
|
+
return budgets, details
|
|
190
|
+
|
|
191
|
+
def hosts_for(self, companies: Sequence[Company]) -> frozenset[str]:
|
|
192
|
+
allowed: set[str] = set()
|
|
193
|
+
for company in companies:
|
|
194
|
+
try:
|
|
195
|
+
host, _ = workday_target(
|
|
196
|
+
company.workday_tenant or "",
|
|
197
|
+
company.workday_site or "",
|
|
198
|
+
company.workday_dc or "",
|
|
199
|
+
)
|
|
200
|
+
except SlugRejectedError:
|
|
201
|
+
continue
|
|
202
|
+
allowed.add(host)
|
|
203
|
+
return frozenset(allowed)
|
|
204
|
+
|
|
205
|
+
def board_key(self, company: Company) -> str:
|
|
206
|
+
return board_key(self.name, _board(company))
|
|
207
|
+
|
|
208
|
+
def plan(self, company: Company) -> tuple[str, ...]:
|
|
209
|
+
host, path = workday_target(
|
|
210
|
+
company.workday_tenant or "",
|
|
211
|
+
company.workday_site or "",
|
|
212
|
+
company.workday_dc or "",
|
|
213
|
+
)
|
|
214
|
+
return (f"https://{host}{path}",)
|
|
215
|
+
|
|
216
|
+
async def fetch(
|
|
217
|
+
self,
|
|
218
|
+
company: Company,
|
|
219
|
+
client: HttpClient,
|
|
220
|
+
now: datetime,
|
|
221
|
+
facets: Mapping[tuple[str, str], WorkdayFacet] | None = None,
|
|
222
|
+
details: Sequence[str] = (),
|
|
223
|
+
crawl: WorkdayCrawl | None = None,
|
|
224
|
+
page_budget: int = MAX_PAGES,
|
|
225
|
+
) -> FetchResult:
|
|
226
|
+
if page_budget < 1:
|
|
227
|
+
raise ValueError("Workday page budget must be positive")
|
|
228
|
+
if page_budget > MAX_CRAWL_PAGES:
|
|
229
|
+
raise ValueError("Workday page budget exceeds its crawl cap")
|
|
230
|
+
url = self.plan(company)[0]
|
|
231
|
+
tenant = company.workday_tenant or ""
|
|
232
|
+
site = company.workday_site or ""
|
|
233
|
+
facet = _pinned_facet(company) or (facets or {}).get((tenant, site))
|
|
234
|
+
applied = {facet.parameter: list(facet.facet_ids)} if facet is not None else {}
|
|
235
|
+
crawl_reset = crawl is not None and (
|
|
236
|
+
crawl.facet_parameter != (facet.parameter if facet is not None else "")
|
|
237
|
+
or crawl.facet_ids != (facet.facet_ids if facet is not None else ())
|
|
238
|
+
)
|
|
239
|
+
persisted = None if crawl_reset else crawl
|
|
240
|
+
degraded = ""
|
|
241
|
+
|
|
242
|
+
discovered: WorkdayFacet | None = None
|
|
243
|
+
forgotten: WorkdayFacet | None = None
|
|
244
|
+
restarted = False
|
|
245
|
+
malformed = 0
|
|
246
|
+
fell_back = False
|
|
247
|
+
stale_facet = False
|
|
248
|
+
drifted = False
|
|
249
|
+
postings: list[WorkdayPosting] = []
|
|
250
|
+
reached_end = False
|
|
251
|
+
offset = 0
|
|
252
|
+
if persisted is not None:
|
|
253
|
+
offset = persisted.next_offset
|
|
254
|
+
if page_budget > 1:
|
|
255
|
+
offset = max(0, offset - PAGE_SIZE)
|
|
256
|
+
pages = 0
|
|
257
|
+
total = persisted.total if persisted is not None else None
|
|
258
|
+
total_changed = False
|
|
259
|
+
|
|
260
|
+
while pages < page_budget:
|
|
261
|
+
body: dict[str, Any] = {
|
|
262
|
+
"appliedFacets": applied,
|
|
263
|
+
"limit": PAGE_SIZE,
|
|
264
|
+
"offset": offset,
|
|
265
|
+
"searchText": "",
|
|
266
|
+
}
|
|
267
|
+
response = await client.post_json(url, body=body)
|
|
268
|
+
page, dropped = _validate(response.payload, company)
|
|
269
|
+
malformed += dropped
|
|
270
|
+
pages += 1
|
|
271
|
+
|
|
272
|
+
if facet is not None and page.facets is None:
|
|
273
|
+
if not drifted:
|
|
274
|
+
drifted = True
|
|
275
|
+
captured = capture_payload("workday-nofacets", company.slug, response.payload)
|
|
276
|
+
degraded = (
|
|
277
|
+
f"no facet list at all while facet {facet.facet_ids!r} applied, "
|
|
278
|
+
f"so staleness is undecided; facet kept, nothing closed. "
|
|
279
|
+
f"Payload captured at {captured}"
|
|
280
|
+
)
|
|
281
|
+
elif facet is not None and not facet_still_offered(page, facet):
|
|
282
|
+
if facet.pinned:
|
|
283
|
+
stale_facet = True
|
|
284
|
+
degraded = (
|
|
285
|
+
f"pinned facet {facet.facet_ids!r} is no longer offered under "
|
|
286
|
+
f"{facet.parameter!r}; honoured anyway. Re-pin with "
|
|
287
|
+
"`stage discover --url` or clear `workday_facet`"
|
|
288
|
+
)
|
|
289
|
+
else:
|
|
290
|
+
probe_body: dict[str, Any] = {
|
|
291
|
+
"appliedFacets": {},
|
|
292
|
+
"limit": PAGE_SIZE,
|
|
293
|
+
"offset": 0,
|
|
294
|
+
"searchText": "",
|
|
295
|
+
}
|
|
296
|
+
probe_response = await client.post_json(url, body=probe_body)
|
|
297
|
+
probe, probe_dropped = _validate(probe_response.payload, company)
|
|
298
|
+
malformed += probe_dropped
|
|
299
|
+
pages += 1
|
|
300
|
+
if not facet_still_offered(probe, facet):
|
|
301
|
+
stale_facet = True
|
|
302
|
+
degraded = (
|
|
303
|
+
f"cached facet {facet.facet_ids!r} is no longer offered under "
|
|
304
|
+
f"{facet.parameter!r}; re-resolving from this tenant's own facet list"
|
|
305
|
+
)
|
|
306
|
+
forgotten = facet
|
|
307
|
+
facet = None
|
|
308
|
+
applied = {}
|
|
309
|
+
postings.clear()
|
|
310
|
+
offset = 0
|
|
311
|
+
total = None
|
|
312
|
+
crawl_reset = True
|
|
313
|
+
page = probe
|
|
314
|
+
dropped = probe_dropped
|
|
315
|
+
response = probe_response
|
|
316
|
+
|
|
317
|
+
if facet is None and not applied:
|
|
318
|
+
resolved = resolve_facet(page, tenant, site, now)
|
|
319
|
+
if resolved is not None:
|
|
320
|
+
discovered = resolved
|
|
321
|
+
forgotten = None
|
|
322
|
+
else:
|
|
323
|
+
fell_back = True
|
|
324
|
+
degraded = _fallback_reason(page, company, response.payload)
|
|
325
|
+
|
|
326
|
+
if discovered is not None and not applied and not restarted:
|
|
327
|
+
applied = {discovered.parameter: list(discovered.facet_ids)}
|
|
328
|
+
restarted = True
|
|
329
|
+
postings.clear()
|
|
330
|
+
offset = 0
|
|
331
|
+
total = None
|
|
332
|
+
crawl_reset = True
|
|
333
|
+
continue
|
|
334
|
+
|
|
335
|
+
postings.extend(page.jobPostings)
|
|
336
|
+
if page.total > 0:
|
|
337
|
+
if total is None:
|
|
338
|
+
total = page.total
|
|
339
|
+
elif total != page.total:
|
|
340
|
+
total_changed = True
|
|
341
|
+
total = page.total
|
|
342
|
+
|
|
343
|
+
returned = len(page.jobPostings) + dropped
|
|
344
|
+
if returned < PAGE_SIZE:
|
|
345
|
+
reached_end = True
|
|
346
|
+
break
|
|
347
|
+
offset += PAGE_SIZE
|
|
348
|
+
if offset >= (RESULT_CAP if total is None else min(total, RESULT_CAP)):
|
|
349
|
+
reached_end = True
|
|
350
|
+
break
|
|
351
|
+
|
|
352
|
+
capped = not reached_end
|
|
353
|
+
if capped:
|
|
354
|
+
if page_budget != MAX_PAGES and page_budget < MAX_CRAWL_PAGES:
|
|
355
|
+
degraded = (
|
|
356
|
+
f"resumable crawl paused after {pages} page(s); resumes from offset {offset} "
|
|
357
|
+
"on the next sync and closes nothing yet"
|
|
358
|
+
)
|
|
359
|
+
else:
|
|
360
|
+
degraded = f"stopped at the {page_budget}-page cap; the board may be truncated"
|
|
361
|
+
if total_changed:
|
|
362
|
+
degraded = (
|
|
363
|
+
"reported total changed before the board ended; progress was retained and "
|
|
364
|
+
f"resumes from offset {offset} on the next sync"
|
|
365
|
+
if not reached_end
|
|
366
|
+
else "reported total changed during the completed crawl; jobs were refreshed, "
|
|
367
|
+
"but nothing closed and the next sync starts a fresh pass"
|
|
368
|
+
)
|
|
369
|
+
if malformed:
|
|
370
|
+
degraded = malformed_note(malformed) + (f" ({degraded})" if degraded else "")
|
|
371
|
+
|
|
372
|
+
faceted = "internship" if applied else ""
|
|
373
|
+
paired = [(posting, _to_job(company, posting, now, faceted)) for posting in postings]
|
|
374
|
+
wanted = set(details)
|
|
375
|
+
fetched: list[DetailFetch] = []
|
|
376
|
+
if wanted:
|
|
377
|
+
paired, fetched = await _attach_descriptions(company, client, paired, wanted)
|
|
378
|
+
|
|
379
|
+
crawl_step = None
|
|
380
|
+
crawling = page_budget < MAX_PAGES or crawl is not None
|
|
381
|
+
safe = not (malformed or fell_back or stale_facet or drifted or total_changed)
|
|
382
|
+
if crawling:
|
|
383
|
+
crawl_step = WorkdayCrawlStep(
|
|
384
|
+
board=self.board_key(company),
|
|
385
|
+
next_offset=0 if reached_end else offset,
|
|
386
|
+
total=total,
|
|
387
|
+
facet_parameter=next(iter(applied), ""),
|
|
388
|
+
facet_ids=tuple(next(iter(applied.values()), ())),
|
|
389
|
+
seen_ids=tuple(dict.fromkeys(job.id for _, job in paired)),
|
|
390
|
+
complete=reached_end and safe,
|
|
391
|
+
reset=crawl_reset,
|
|
392
|
+
discard=bool(malformed) or (reached_end and not safe),
|
|
393
|
+
)
|
|
394
|
+
|
|
395
|
+
return FetchResult(
|
|
396
|
+
jobs=tuple(job for _, job in paired),
|
|
397
|
+
degraded=degraded,
|
|
398
|
+
authoritative=reached_end and safe,
|
|
399
|
+
facets=(discovered,) if discovered is not None else (),
|
|
400
|
+
forgotten_facets=(forgotten,) if forgotten is not None else (),
|
|
401
|
+
detail_fetches=tuple(fetched),
|
|
402
|
+
workday_crawl=crawl_step,
|
|
403
|
+
)
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
def _fallback_reason(page: WorkdayPage, company: Company, payload: Any) -> str:
|
|
407
|
+
if page.facets:
|
|
408
|
+
names = {group.facetParameter for group in page.facets if group.facetParameter}
|
|
409
|
+
advertised = ", ".join(sorted(names))
|
|
410
|
+
return (
|
|
411
|
+
f"no internship facet among the values this tenant advertises ({advertised}); "
|
|
412
|
+
"walking the whole board instead, which the bilingual classifier then filters"
|
|
413
|
+
)
|
|
414
|
+
if page.facets == []:
|
|
415
|
+
return (
|
|
416
|
+
"this tenant advertises an empty facet list, so there is no internship facet "
|
|
417
|
+
"to resolve; walking the whole board instead"
|
|
418
|
+
)
|
|
419
|
+
captured = capture_payload("workday-nofacets", company.slug, payload)
|
|
420
|
+
return f"no facet list at all, so resolution could not run; payload captured at {captured}"
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
def _pinned_facet(company: Company) -> WorkdayFacet | None:
|
|
424
|
+
if not company.workday_facet:
|
|
425
|
+
return None
|
|
426
|
+
parameter, _, value = company.workday_facet.partition(":")
|
|
427
|
+
if not value:
|
|
428
|
+
return WorkdayFacet(
|
|
429
|
+
tenant=company.workday_tenant or "",
|
|
430
|
+
site=company.workday_site or "",
|
|
431
|
+
parameter=workday_facet_lexicon()[0][0],
|
|
432
|
+
facet_ids=(parameter,),
|
|
433
|
+
pinned=True,
|
|
434
|
+
)
|
|
435
|
+
return WorkdayFacet(
|
|
436
|
+
tenant=company.workday_tenant or "",
|
|
437
|
+
site=company.workday_site or "",
|
|
438
|
+
parameter=parameter,
|
|
439
|
+
facet_ids=tuple(value.split(",")),
|
|
440
|
+
pinned=True,
|
|
441
|
+
)
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def _validate(payload: Any, company: Company) -> tuple[WorkdayPage, int]:
|
|
445
|
+
try:
|
|
446
|
+
raw = WorkdayRawPage.model_validate(payload)
|
|
447
|
+
except ValidationError as exc:
|
|
448
|
+
captured = capture_payload("workday", company.slug, payload)
|
|
449
|
+
raise PayloadValidationError(
|
|
450
|
+
f"workday payload for {company.name} failed validation: {exc} (captured {captured})"
|
|
451
|
+
) from exc
|
|
452
|
+
|
|
453
|
+
kept, dropped = validate_rows(
|
|
454
|
+
WorkdayPosting, raw.jobPostings, source="workday", slug=company.slug
|
|
455
|
+
)
|
|
456
|
+
return WorkdayPage(total=raw.total, jobPostings=kept, facets=raw.facets), dropped
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
async def _attach_descriptions(
|
|
460
|
+
company: Company,
|
|
461
|
+
client: HttpClient,
|
|
462
|
+
paired: list[tuple[WorkdayPosting, Job]],
|
|
463
|
+
wanted: set[str],
|
|
464
|
+
) -> tuple[list[tuple[WorkdayPosting, Job]], list[DetailFetch]]:
|
|
465
|
+
host, _ = workday_target(
|
|
466
|
+
company.workday_tenant or "", company.workday_site or "", company.workday_dc or ""
|
|
467
|
+
)
|
|
468
|
+
outcomes: list[DetailFetch] = []
|
|
469
|
+
merged: list[tuple[WorkdayPosting, Job]] = []
|
|
470
|
+
for posting, job in paired:
|
|
471
|
+
if job.id not in wanted or not posting.externalPath:
|
|
472
|
+
merged.append((posting, job))
|
|
473
|
+
continue
|
|
474
|
+
path = posting.externalPath
|
|
475
|
+
url = f"https://{host}/wday/cxs/{company.workday_tenant}/{company.workday_site}{path}"
|
|
476
|
+
try:
|
|
477
|
+
response = await client.get_json(url)
|
|
478
|
+
except HttpError:
|
|
479
|
+
outcomes.append(DetailFetch(id=job.id, resolved=False, failed=True))
|
|
480
|
+
merged.append((posting, job))
|
|
481
|
+
continue
|
|
482
|
+
body = _description_from(response.payload)
|
|
483
|
+
outcomes.append(DetailFetch(id=job.id, resolved=bool(body)))
|
|
484
|
+
merged.append((posting, replace(job, description=body) if body else job))
|
|
485
|
+
return merged, outcomes
|
|
486
|
+
|
|
487
|
+
|
|
488
|
+
def _description_from(payload: Any) -> str:
|
|
489
|
+
if not isinstance(payload, dict):
|
|
490
|
+
return ""
|
|
491
|
+
info = payload.get("jobPostingInfo")
|
|
492
|
+
if not isinstance(info, dict):
|
|
493
|
+
return ""
|
|
494
|
+
body = info.get("jobDescription")
|
|
495
|
+
return collapse_whitespace(strip_html(body)) if isinstance(body, str) else ""
|
|
496
|
+
|
|
497
|
+
|
|
498
|
+
def _board(company: Company) -> str:
|
|
499
|
+
return f"{company.workday_tenant or company.slug}-{company.workday_site or ''}"
|
|
500
|
+
|
|
501
|
+
|
|
502
|
+
def _to_job(
|
|
503
|
+
company: Company, posting: WorkdayPosting, now: datetime, employment_type: str = ""
|
|
504
|
+
) -> Job:
|
|
505
|
+
host, _ = workday_target(
|
|
506
|
+
company.workday_tenant or "", company.workday_site or "", company.workday_dc or ""
|
|
507
|
+
)
|
|
508
|
+
raw_path = posting.externalPath
|
|
509
|
+
path = raw_path if raw_path.startswith("/") else f"/{raw_path}"
|
|
510
|
+
apply_url = f"https://{host}/{company.workday_site}{path}" if raw_path else ""
|
|
511
|
+
title = collapse_whitespace(posting.title)
|
|
512
|
+
return Job(
|
|
513
|
+
id=job_id("workday", _board(company), posting.requisition()),
|
|
514
|
+
source="workday",
|
|
515
|
+
company=company.name,
|
|
516
|
+
title_raw=title,
|
|
517
|
+
title_normalized=title.lower(),
|
|
518
|
+
apply_url_raw=apply_url,
|
|
519
|
+
description="",
|
|
520
|
+
location_raw=collapse_whitespace(posting.locationsText),
|
|
521
|
+
first_seen=now,
|
|
522
|
+
last_seen=now,
|
|
523
|
+
signals=SourceSignals(employment_type=employment_type),
|
|
524
|
+
)
|
stage/sources/zshah.py
ADDED
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
from datetime import datetime
|
|
2
|
+
from typing import Any, ClassVar
|
|
3
|
+
|
|
4
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
5
|
+
|
|
6
|
+
from stage.domain import Job, SourceSignals, job_id
|
|
7
|
+
from stage.http import HttpClient
|
|
8
|
+
from stage.sources._text import collapse_whitespace
|
|
9
|
+
from stage.sources.base import (
|
|
10
|
+
FetchResult,
|
|
11
|
+
NonEmptyStr,
|
|
12
|
+
PayloadValidationError,
|
|
13
|
+
capture_payload,
|
|
14
|
+
validate_rows,
|
|
15
|
+
)
|
|
16
|
+
from stage.sources.feed import register_feed, upcoming_season_year
|
|
17
|
+
|
|
18
|
+
URL = "https://zshah101.github.io/Automated-List-Of-Summer-{year}-and-Fall-2026-Tech-Internships/api/jobs.json"
|
|
19
|
+
HOST = "zshah101.github.io"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class ZshahListing(BaseModel):
|
|
23
|
+
model_config = ConfigDict(extra="ignore")
|
|
24
|
+
|
|
25
|
+
id: NonEmptyStr
|
|
26
|
+
company: NonEmptyStr
|
|
27
|
+
title: NonEmptyStr
|
|
28
|
+
location: str = ""
|
|
29
|
+
url: NonEmptyStr
|
|
30
|
+
program: str = ""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class ZshahEnvelope(BaseModel):
|
|
34
|
+
model_config = ConfigDict(extra="ignore")
|
|
35
|
+
|
|
36
|
+
jobs: list[Any] = Field(default_factory=list)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@register_feed
|
|
40
|
+
class ZshahFeed:
|
|
41
|
+
name: ClassVar[str] = "zshah101"
|
|
42
|
+
rate_profile: ClassVar[str] = "feeds"
|
|
43
|
+
hosts: ClassVar[frozenset[str]] = frozenset({HOST})
|
|
44
|
+
bucket_key: ClassVar[str] = ""
|
|
45
|
+
|
|
46
|
+
def season_year(self, now: datetime) -> int:
|
|
47
|
+
return upcoming_season_year(now)
|
|
48
|
+
|
|
49
|
+
def plan(self, now: datetime) -> tuple[str, ...]:
|
|
50
|
+
return (URL.format(year=self.season_year(now)),)
|
|
51
|
+
|
|
52
|
+
async def fetch(self, client: HttpClient, now: datetime) -> FetchResult:
|
|
53
|
+
url = self.plan(now)[0]
|
|
54
|
+
response = await client.get_json(url)
|
|
55
|
+
if response.not_modified:
|
|
56
|
+
return FetchResult(not_modified=True)
|
|
57
|
+
try:
|
|
58
|
+
envelope = ZshahEnvelope.model_validate(response.payload)
|
|
59
|
+
except Exception as exc:
|
|
60
|
+
captured = capture_payload(self.name, str(self.season_year(now)), response.payload)
|
|
61
|
+
raise PayloadValidationError(
|
|
62
|
+
f"{self.name}/{url}: payload envelope failed validation (captured {captured})"
|
|
63
|
+
) from exc
|
|
64
|
+
listings, malformed = validate_rows(
|
|
65
|
+
ZshahListing, envelope.jobs, source=self.name, slug=str(self.season_year(now))
|
|
66
|
+
)
|
|
67
|
+
internships = [
|
|
68
|
+
listing for listing in listings if listing.program.casefold() == "internship"
|
|
69
|
+
]
|
|
70
|
+
if not internships:
|
|
71
|
+
captured = capture_payload(self.name, str(self.season_year(now)), response.payload)
|
|
72
|
+
raise PayloadValidationError(
|
|
73
|
+
f"{self.name}/{url}: no internship records were found (captured {captured})"
|
|
74
|
+
)
|
|
75
|
+
jobs = tuple(
|
|
76
|
+
Job(
|
|
77
|
+
id=job_id(self.name, listing.company, listing.id),
|
|
78
|
+
source=self.name,
|
|
79
|
+
company=collapse_whitespace(listing.company),
|
|
80
|
+
title_raw=collapse_whitespace(listing.title),
|
|
81
|
+
title_normalized=collapse_whitespace(listing.title),
|
|
82
|
+
apply_url_raw=listing.url,
|
|
83
|
+
description="",
|
|
84
|
+
location_raw=collapse_whitespace(listing.location),
|
|
85
|
+
first_seen=now,
|
|
86
|
+
last_seen=now,
|
|
87
|
+
signals=SourceSignals(employment_type="internship"),
|
|
88
|
+
)
|
|
89
|
+
for listing in internships
|
|
90
|
+
)
|
|
91
|
+
return FetchResult(
|
|
92
|
+
jobs=jobs,
|
|
93
|
+
degraded=(
|
|
94
|
+
f"{malformed} malformed posting(s) were dropped; the feed closes nothing"
|
|
95
|
+
if malformed
|
|
96
|
+
else ""
|
|
97
|
+
),
|
|
98
|
+
authoritative=not malformed,
|
|
99
|
+
)
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
from collections.abc import AsyncIterator
|
|
2
|
+
from contextlib import asynccontextmanager
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from stage.storage.repository import Repository, SourceBatch, SourceBatchResult
|
|
6
|
+
from stage.storage.sqlite_repo import SqliteRepository
|
|
7
|
+
from stage.storage.writer import AsyncRepository, DatabaseWriter, WriterNotStartedError
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@asynccontextmanager
|
|
11
|
+
async def open_repository(db_path: Path) -> AsyncIterator[AsyncRepository]:
|
|
12
|
+
writer = DatabaseWriter(db_path)
|
|
13
|
+
await writer.start()
|
|
14
|
+
try:
|
|
15
|
+
yield AsyncRepository(writer)
|
|
16
|
+
finally:
|
|
17
|
+
await writer.aclose()
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
__all__ = [
|
|
21
|
+
"AsyncRepository",
|
|
22
|
+
"DatabaseWriter",
|
|
23
|
+
"Repository",
|
|
24
|
+
"SourceBatch",
|
|
25
|
+
"SourceBatchResult",
|
|
26
|
+
"SqliteRepository",
|
|
27
|
+
"WriterNotStartedError",
|
|
28
|
+
"open_repository",
|
|
29
|
+
]
|