stage-cli 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. stage/__init__.py +1 -0
  2. stage/__main__.py +8 -0
  3. stage/banner.py +32 -0
  4. stage/bootstrap/__init__.py +0 -0
  5. stage/bootstrap/openjobs.py +392 -0
  6. stage/classify/__init__.py +29 -0
  7. stage/classify/eligibility.py +115 -0
  8. stage/classify/internship.py +64 -0
  9. stage/classify/role.py +91 -0
  10. stage/classify/scope.py +47 -0
  11. stage/cli/__init__.py +0 -0
  12. stage/cli/app.py +4 -0
  13. stage/cli/commands/__init__.py +8 -0
  14. stage/cli/commands/discovery.py +294 -0
  15. stage/cli/commands/insight.py +494 -0
  16. stage/cli/commands/pipeline.py +337 -0
  17. stage/cli/commands/postings.py +473 -0
  18. stage/cli/commands/schedule.py +171 -0
  19. stage/cli/housekeeping.py +64 -0
  20. stage/cli/logfile.py +56 -0
  21. stage/cli/notify.py +170 -0
  22. stage/cli/options.py +678 -0
  23. stage/cli/render.py +1398 -0
  24. stage/cli/runlock.py +74 -0
  25. stage/cli/schedule.py +702 -0
  26. stage/cli/schedule_state.py +363 -0
  27. stage/cli/selection.py +83 -0
  28. stage/cli/serialize.py +196 -0
  29. stage/companies.py +542 -0
  30. stage/data/companies/a.yaml +1289 -0
  31. stage/data/companies/b.yaml +900 -0
  32. stage/data/companies/c.yaml +1377 -0
  33. stage/data/companies/d.yaml +497 -0
  34. stage/data/companies/e.yaml +519 -0
  35. stage/data/companies/f.yaml +454 -0
  36. stage/data/companies/g.yaml +601 -0
  37. stage/data/companies/h.yaml +446 -0
  38. stage/data/companies/i.yaml +503 -0
  39. stage/data/companies/j.yaml +138 -0
  40. stage/data/companies/k.yaml +278 -0
  41. stage/data/companies/l.yaml +402 -0
  42. stage/data/companies/m.yaml +937 -0
  43. stage/data/companies/n.yaml +549 -0
  44. stage/data/companies/o.yaml +371 -0
  45. stage/data/companies/other.yaml +58 -0
  46. stage/data/companies/p.yaml +825 -0
  47. stage/data/companies/q.yaml +121 -0
  48. stage/data/companies/r.yaml +583 -0
  49. stage/data/companies/s.yaml +1140 -0
  50. stage/data/companies/t.yaml +817 -0
  51. stage/data/companies/u.yaml +196 -0
  52. stage/data/companies/v.yaml +325 -0
  53. stage/data/companies/w.yaml +353 -0
  54. stage/data/companies/x.yaml +67 -0
  55. stage/data/companies/y.yaml +36 -0
  56. stage/data/companies/z.yaml +146 -0
  57. stage/data/fonts/DejaVuSans.LICENSE.txt +99 -0
  58. stage/data/fonts/DejaVuSans.ttf +0 -0
  59. stage/data/lexicon/company_tokens.yaml +228 -0
  60. stage/data/lexicon/eligibility.yaml +455 -0
  61. stage/data/lexicon/inclusive_suffixes.yaml +37 -0
  62. stage/data/lexicon/internship.yaml +187 -0
  63. stage/data/lexicon/language.yaml +226 -0
  64. stage/data/lexicon/locations.yaml +1159 -0
  65. stage/data/lexicon/roles.yaml +2012 -0
  66. stage/data/lexicon/terms.yaml +76 -0
  67. stage/data/lexicon/workday_facets.yaml +27 -0
  68. stage/data/seed_companies.yaml +198 -0
  69. stage/dedup/__init__.py +19 -0
  70. stage/dedup/identity.py +113 -0
  71. stage/dedup/resolve.py +97 -0
  72. stage/domain/__init__.py +244 -0
  73. stage/domain/company.py +49 -0
  74. stage/domain/coverage.py +86 -0
  75. stage/domain/custom_board.py +92 -0
  76. stage/domain/discovery.py +94 -0
  77. stage/domain/enums.py +114 -0
  78. stage/domain/events.py +204 -0
  79. stage/domain/filters.py +27 -0
  80. stage/domain/health.py +169 -0
  81. stage/domain/ids.py +48 -0
  82. stage/domain/job.py +47 -0
  83. stage/domain/matching.py +15 -0
  84. stage/domain/priority.py +34 -0
  85. stage/domain/quarantine.py +39 -0
  86. stage/domain/rate_state.py +78 -0
  87. stage/domain/retention.py +20 -0
  88. stage/domain/rotation.py +46 -0
  89. stage/domain/signals.py +12 -0
  90. stage/domain/sync_run.py +35 -0
  91. stage/domain/text.py +113 -0
  92. stage/domain/validator.py +14 -0
  93. stage/domain/visits.py +60 -0
  94. stage/domain/workday.py +38 -0
  95. stage/http/__init__.py +58 -0
  96. stage/http/breaker.py +53 -0
  97. stage/http/cache.py +44 -0
  98. stage/http/client.py +725 -0
  99. stage/http/profiles.py +101 -0
  100. stage/lexicon.py +370 -0
  101. stage/normalize/__init__.py +16 -0
  102. stage/normalize/language.py +47 -0
  103. stage/normalize/location.py +271 -0
  104. stage/normalize/terms.py +153 -0
  105. stage/normalize/urls.py +122 -0
  106. stage/paths.py +86 -0
  107. stage/py.typed +0 -0
  108. stage/services/__init__.py +0 -0
  109. stage/services/canary.py +120 -0
  110. stage/services/coverage.py +231 -0
  111. stage/services/discover.py +747 -0
  112. stage/services/export.py +274 -0
  113. stage/services/health.py +237 -0
  114. stage/services/maintenance.py +225 -0
  115. stage/services/quarantine.py +20 -0
  116. stage/services/query.py +86 -0
  117. stage/services/sync.py +1257 -0
  118. stage/sources/__init__.py +82 -0
  119. stage/sources/_text.py +79 -0
  120. stage/sources/ashby.py +93 -0
  121. stage/sources/bamboohr.py +80 -0
  122. stage/sources/base.py +225 -0
  123. stage/sources/breezy.py +90 -0
  124. stage/sources/collage.py +60 -0
  125. stage/sources/community_feeds.py +142 -0
  126. stage/sources/curated_markdown.py +289 -0
  127. stage/sources/custom_json.py +610 -0
  128. stage/sources/espresso.py +154 -0
  129. stage/sources/feed.py +44 -0
  130. stage/sources/greenhouse.py +104 -0
  131. stage/sources/jobbank.py +147 -0
  132. stage/sources/jobvite.py +133 -0
  133. stage/sources/lever.py +76 -0
  134. stage/sources/oracle_cloud.py +187 -0
  135. stage/sources/platforms.py +609 -0
  136. stage/sources/quebec_emploi.py +146 -0
  137. stage/sources/recruitee.py +96 -0
  138. stage/sources/simplify.py +110 -0
  139. stage/sources/smartrecruiters.py +216 -0
  140. stage/sources/speedyapply.py +200 -0
  141. stage/sources/themuse.py +157 -0
  142. stage/sources/workable.py +83 -0
  143. stage/sources/workday.py +524 -0
  144. stage/sources/zshah.py +99 -0
  145. stage/storage/__init__.py +29 -0
  146. stage/storage/migrations/0001_initial.sql +239 -0
  147. stage/storage/migrations/__init__.py +135 -0
  148. stage/storage/repository.py +213 -0
  149. stage/storage/search.py +28 -0
  150. stage/storage/sqlite_repo.py +1586 -0
  151. stage/storage/writer.py +249 -0
  152. stage/tui/__init__.py +0 -0
  153. stage/tui/app.py +82 -0
  154. stage/tui/help.py +26 -0
  155. stage/tui/safe.py +21 -0
  156. stage/tui/screens/__init__.py +0 -0
  157. stage/tui/screens/boards.py +186 -0
  158. stage/tui/screens/postings.py +509 -0
  159. stage/tui/screens/review.py +209 -0
  160. stage/tui/screens/splash.py +37 -0
  161. stage/tui/screens/stats.py +124 -0
  162. stage/tui/screens/sync.py +194 -0
  163. stage/tui/state.py +160 -0
  164. stage/tui/theme.tcss +205 -0
  165. stage/tui/widgets/__init__.py +0 -0
  166. stage_cli-1.0.0.dist-info/METADATA +379 -0
  167. stage_cli-1.0.0.dist-info/RECORD +170 -0
  168. stage_cli-1.0.0.dist-info/WHEEL +4 -0
  169. stage_cli-1.0.0.dist-info/entry_points.txt +2 -0
  170. stage_cli-1.0.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,524 @@
1
+ from collections.abc import Mapping, Sequence
2
+ from dataclasses import replace
3
+ from datetime import datetime
4
+ from typing import Any, ClassVar
5
+
6
+ from pydantic import BaseModel, ConfigDict, Field, ValidationError
7
+
8
+ from stage.domain import (
9
+ Company,
10
+ DetailFetch,
11
+ Job,
12
+ Platform,
13
+ SourceSignals,
14
+ WorkdayCrawl,
15
+ WorkdayCrawlStep,
16
+ WorkdayFacet,
17
+ board_key,
18
+ job_id,
19
+ )
20
+ from stage.http import HttpClient, HttpError
21
+ from stage.lexicon import fold, internship_lexicon, workday_facet_lexicon
22
+ from stage.sources import register
23
+ from stage.sources._text import collapse_whitespace, strip_html
24
+ from stage.sources.base import (
25
+ FetchResult,
26
+ PayloadValidationError,
27
+ capture_payload,
28
+ malformed_note,
29
+ validate_rows,
30
+ )
31
+ from stage.sources.platforms import SlugRejectedError, workday_target
32
+
33
+ PAGE_SIZE = 20
34
+ MAX_PAGES = 25
35
+ MAX_CRAWL_PAGES = 100
36
+ RESULT_CAP = 10_000
37
+ CRAWL_PAGE_CAP = 6
38
+ RETRY_RESERVE = 40
39
+
40
+
41
+ class WorkdayPosting(BaseModel):
42
+ model_config = ConfigDict(extra="ignore")
43
+
44
+ title: str
45
+ externalPath: str = ""
46
+ locationsText: str = ""
47
+ postedOn: str = ""
48
+ bulletFields: list[str] = Field(default_factory=list)
49
+
50
+ def requisition(self) -> str:
51
+ for field in self.bulletFields:
52
+ cleaned = field.strip()
53
+ if cleaned:
54
+ return cleaned
55
+ tail = self.externalPath.rstrip("/").rsplit("_", 1)
56
+ return tail[-1] if len(tail) == 2 and tail[-1] else self.externalPath.rstrip("/")
57
+
58
+
59
+ class WorkdayFacetValue(BaseModel):
60
+ model_config = ConfigDict(extra="ignore")
61
+
62
+ id: str = ""
63
+ descriptor: str = ""
64
+ count: int = 0
65
+
66
+
67
+ class WorkdayFacetGroup(BaseModel):
68
+ model_config = ConfigDict(extra="ignore")
69
+
70
+ facetParameter: str = ""
71
+ descriptor: str = ""
72
+ values: list[WorkdayFacetValue] = Field(default_factory=list)
73
+
74
+
75
+ class WorkdayPage(BaseModel):
76
+ model_config = ConfigDict(extra="ignore")
77
+
78
+ total: int = 0
79
+ jobPostings: list[WorkdayPosting]
80
+ facets: list[WorkdayFacetGroup] | None = None
81
+
82
+
83
+ class WorkdayRawPage(BaseModel):
84
+ model_config = ConfigDict(extra="ignore")
85
+
86
+ total: int = 0
87
+ jobPostings: list[dict[str, Any]]
88
+ facets: list[WorkdayFacetGroup] | None = None
89
+
90
+
91
+ def _matches(folded: str, descriptors: frozenset[str]) -> bool:
92
+ padded = f" {folded} "
93
+ if any(f" {phrase} " in padded for phrase in internship_lexicon().blocked_bigrams):
94
+ return False
95
+ return any(f" {phrase} " in padded for phrase in descriptors)
96
+
97
+
98
+ def resolve_facet(page: WorkdayPage, tenant: str, site: str, now: datetime) -> WorkdayFacet | None:
99
+ parameters, descriptors = workday_facet_lexicon()
100
+ groups = {group.facetParameter: group for group in page.facets or ()}
101
+ for parameter in parameters:
102
+ group = groups.get(parameter)
103
+ if group is None:
104
+ continue
105
+ matched = [
106
+ value
107
+ for value in group.values
108
+ if value.id and _matches(fold(value.descriptor), descriptors)
109
+ ]
110
+ if matched:
111
+ return WorkdayFacet(
112
+ tenant=tenant,
113
+ site=site,
114
+ parameter=parameter,
115
+ facet_ids=tuple(value.id for value in matched),
116
+ descriptor=", ".join(value.descriptor for value in matched),
117
+ resolved_at=now,
118
+ )
119
+ return None
120
+
121
+
122
+ def facet_still_offered(page: WorkdayPage, facet: WorkdayFacet) -> bool:
123
+ if page.facets is None:
124
+ return True
125
+ for group in page.facets:
126
+ if group.facetParameter != facet.parameter:
127
+ continue
128
+ offered = {value.id for value in group.values}
129
+ if all(facet_id in offered for facet_id in facet.facet_ids):
130
+ return True
131
+ return False
132
+
133
+
134
+ @register
135
+ class WorkdayAdapter:
136
+ name: ClassVar[str] = "workday"
137
+ platform: ClassVar[Platform] = Platform.WORKDAY
138
+ rate_profile: ClassVar[str] = "workday"
139
+ hosts: ClassVar[frozenset[str]] = frozenset()
140
+ bucket_key: ClassVar[str] = "workday"
141
+
142
+ detail_budget: ClassVar[int] = 60
143
+
144
+ rotation_slice: ClassVar[int] = 300
145
+
146
+ max_requests_per_company: ClassVar[int] = MAX_PAGES
147
+
148
+ crawl_page_cap: ClassVar[int] = CRAWL_PAGE_CAP
149
+ retry_reserve: ClassVar[int] = RETRY_RESERVE
150
+
151
+ @classmethod
152
+ def crawl_budgets(
153
+ cls,
154
+ companies: Sequence[Company],
155
+ crawls: Mapping[str, WorkdayCrawl],
156
+ facets: Mapping[tuple[str, str], WorkdayFacet],
157
+ ceiling: int,
158
+ ) -> tuple[dict[str, int], int]:
159
+ if not companies:
160
+ raise ValueError("a Workday crawl needs at least one board")
161
+ available = max(1, ceiling - cls.retry_reserve)
162
+ budgets = {company.registry_key: 1 for company in companies}
163
+ remaining = max(0, available - len(budgets))
164
+ demands: list[tuple[int, str]] = []
165
+ for company in companies:
166
+ crawl = crawls.get(board_key(cls.name, _board(company)))
167
+ facet = _pinned_facet(company) or facets.get(
168
+ (company.workday_tenant or "", company.workday_site or "")
169
+ )
170
+ if crawl is None or crawl.total is None:
171
+ continue
172
+ if crawl.facet_parameter != (facet.parameter if facet is not None else ""):
173
+ continue
174
+ if crawl.facet_ids != (facet.facet_ids if facet is not None else ()):
175
+ continue
176
+ start = max(0, crawl.next_offset - PAGE_SIZE)
177
+ pages = max(1, (max(0, crawl.total - start) + PAGE_SIZE - 1) // PAGE_SIZE)
178
+ demands.append((min(MAX_CRAWL_PAGES, pages), company.registry_key))
179
+ for pages, key in sorted(demands, key=lambda item: (-item[0], item[1])):
180
+ extra = max(0, min(remaining, pages - budgets[key]))
181
+ budgets[key] += extra
182
+ remaining -= extra
183
+ for company in companies:
184
+ key = company.registry_key
185
+ extra = max(0, min(remaining, cls.crawl_page_cap - budgets[key]))
186
+ budgets[key] += extra
187
+ remaining -= extra
188
+ details = min(cls.detail_budget, max(0, remaining))
189
+ return budgets, details
190
+
191
+ def hosts_for(self, companies: Sequence[Company]) -> frozenset[str]:
192
+ allowed: set[str] = set()
193
+ for company in companies:
194
+ try:
195
+ host, _ = workday_target(
196
+ company.workday_tenant or "",
197
+ company.workday_site or "",
198
+ company.workday_dc or "",
199
+ )
200
+ except SlugRejectedError:
201
+ continue
202
+ allowed.add(host)
203
+ return frozenset(allowed)
204
+
205
+ def board_key(self, company: Company) -> str:
206
+ return board_key(self.name, _board(company))
207
+
208
+ def plan(self, company: Company) -> tuple[str, ...]:
209
+ host, path = workday_target(
210
+ company.workday_tenant or "",
211
+ company.workday_site or "",
212
+ company.workday_dc or "",
213
+ )
214
+ return (f"https://{host}{path}",)
215
+
216
+ async def fetch(
217
+ self,
218
+ company: Company,
219
+ client: HttpClient,
220
+ now: datetime,
221
+ facets: Mapping[tuple[str, str], WorkdayFacet] | None = None,
222
+ details: Sequence[str] = (),
223
+ crawl: WorkdayCrawl | None = None,
224
+ page_budget: int = MAX_PAGES,
225
+ ) -> FetchResult:
226
+ if page_budget < 1:
227
+ raise ValueError("Workday page budget must be positive")
228
+ if page_budget > MAX_CRAWL_PAGES:
229
+ raise ValueError("Workday page budget exceeds its crawl cap")
230
+ url = self.plan(company)[0]
231
+ tenant = company.workday_tenant or ""
232
+ site = company.workday_site or ""
233
+ facet = _pinned_facet(company) or (facets or {}).get((tenant, site))
234
+ applied = {facet.parameter: list(facet.facet_ids)} if facet is not None else {}
235
+ crawl_reset = crawl is not None and (
236
+ crawl.facet_parameter != (facet.parameter if facet is not None else "")
237
+ or crawl.facet_ids != (facet.facet_ids if facet is not None else ())
238
+ )
239
+ persisted = None if crawl_reset else crawl
240
+ degraded = ""
241
+
242
+ discovered: WorkdayFacet | None = None
243
+ forgotten: WorkdayFacet | None = None
244
+ restarted = False
245
+ malformed = 0
246
+ fell_back = False
247
+ stale_facet = False
248
+ drifted = False
249
+ postings: list[WorkdayPosting] = []
250
+ reached_end = False
251
+ offset = 0
252
+ if persisted is not None:
253
+ offset = persisted.next_offset
254
+ if page_budget > 1:
255
+ offset = max(0, offset - PAGE_SIZE)
256
+ pages = 0
257
+ total = persisted.total if persisted is not None else None
258
+ total_changed = False
259
+
260
+ while pages < page_budget:
261
+ body: dict[str, Any] = {
262
+ "appliedFacets": applied,
263
+ "limit": PAGE_SIZE,
264
+ "offset": offset,
265
+ "searchText": "",
266
+ }
267
+ response = await client.post_json(url, body=body)
268
+ page, dropped = _validate(response.payload, company)
269
+ malformed += dropped
270
+ pages += 1
271
+
272
+ if facet is not None and page.facets is None:
273
+ if not drifted:
274
+ drifted = True
275
+ captured = capture_payload("workday-nofacets", company.slug, response.payload)
276
+ degraded = (
277
+ f"no facet list at all while facet {facet.facet_ids!r} applied, "
278
+ f"so staleness is undecided; facet kept, nothing closed. "
279
+ f"Payload captured at {captured}"
280
+ )
281
+ elif facet is not None and not facet_still_offered(page, facet):
282
+ if facet.pinned:
283
+ stale_facet = True
284
+ degraded = (
285
+ f"pinned facet {facet.facet_ids!r} is no longer offered under "
286
+ f"{facet.parameter!r}; honoured anyway. Re-pin with "
287
+ "`stage discover --url` or clear `workday_facet`"
288
+ )
289
+ else:
290
+ probe_body: dict[str, Any] = {
291
+ "appliedFacets": {},
292
+ "limit": PAGE_SIZE,
293
+ "offset": 0,
294
+ "searchText": "",
295
+ }
296
+ probe_response = await client.post_json(url, body=probe_body)
297
+ probe, probe_dropped = _validate(probe_response.payload, company)
298
+ malformed += probe_dropped
299
+ pages += 1
300
+ if not facet_still_offered(probe, facet):
301
+ stale_facet = True
302
+ degraded = (
303
+ f"cached facet {facet.facet_ids!r} is no longer offered under "
304
+ f"{facet.parameter!r}; re-resolving from this tenant's own facet list"
305
+ )
306
+ forgotten = facet
307
+ facet = None
308
+ applied = {}
309
+ postings.clear()
310
+ offset = 0
311
+ total = None
312
+ crawl_reset = True
313
+ page = probe
314
+ dropped = probe_dropped
315
+ response = probe_response
316
+
317
+ if facet is None and not applied:
318
+ resolved = resolve_facet(page, tenant, site, now)
319
+ if resolved is not None:
320
+ discovered = resolved
321
+ forgotten = None
322
+ else:
323
+ fell_back = True
324
+ degraded = _fallback_reason(page, company, response.payload)
325
+
326
+ if discovered is not None and not applied and not restarted:
327
+ applied = {discovered.parameter: list(discovered.facet_ids)}
328
+ restarted = True
329
+ postings.clear()
330
+ offset = 0
331
+ total = None
332
+ crawl_reset = True
333
+ continue
334
+
335
+ postings.extend(page.jobPostings)
336
+ if page.total > 0:
337
+ if total is None:
338
+ total = page.total
339
+ elif total != page.total:
340
+ total_changed = True
341
+ total = page.total
342
+
343
+ returned = len(page.jobPostings) + dropped
344
+ if returned < PAGE_SIZE:
345
+ reached_end = True
346
+ break
347
+ offset += PAGE_SIZE
348
+ if offset >= (RESULT_CAP if total is None else min(total, RESULT_CAP)):
349
+ reached_end = True
350
+ break
351
+
352
+ capped = not reached_end
353
+ if capped:
354
+ if page_budget != MAX_PAGES and page_budget < MAX_CRAWL_PAGES:
355
+ degraded = (
356
+ f"resumable crawl paused after {pages} page(s); resumes from offset {offset} "
357
+ "on the next sync and closes nothing yet"
358
+ )
359
+ else:
360
+ degraded = f"stopped at the {page_budget}-page cap; the board may be truncated"
361
+ if total_changed:
362
+ degraded = (
363
+ "reported total changed before the board ended; progress was retained and "
364
+ f"resumes from offset {offset} on the next sync"
365
+ if not reached_end
366
+ else "reported total changed during the completed crawl; jobs were refreshed, "
367
+ "but nothing closed and the next sync starts a fresh pass"
368
+ )
369
+ if malformed:
370
+ degraded = malformed_note(malformed) + (f" ({degraded})" if degraded else "")
371
+
372
+ faceted = "internship" if applied else ""
373
+ paired = [(posting, _to_job(company, posting, now, faceted)) for posting in postings]
374
+ wanted = set(details)
375
+ fetched: list[DetailFetch] = []
376
+ if wanted:
377
+ paired, fetched = await _attach_descriptions(company, client, paired, wanted)
378
+
379
+ crawl_step = None
380
+ crawling = page_budget < MAX_PAGES or crawl is not None
381
+ safe = not (malformed or fell_back or stale_facet or drifted or total_changed)
382
+ if crawling:
383
+ crawl_step = WorkdayCrawlStep(
384
+ board=self.board_key(company),
385
+ next_offset=0 if reached_end else offset,
386
+ total=total,
387
+ facet_parameter=next(iter(applied), ""),
388
+ facet_ids=tuple(next(iter(applied.values()), ())),
389
+ seen_ids=tuple(dict.fromkeys(job.id for _, job in paired)),
390
+ complete=reached_end and safe,
391
+ reset=crawl_reset,
392
+ discard=bool(malformed) or (reached_end and not safe),
393
+ )
394
+
395
+ return FetchResult(
396
+ jobs=tuple(job for _, job in paired),
397
+ degraded=degraded,
398
+ authoritative=reached_end and safe,
399
+ facets=(discovered,) if discovered is not None else (),
400
+ forgotten_facets=(forgotten,) if forgotten is not None else (),
401
+ detail_fetches=tuple(fetched),
402
+ workday_crawl=crawl_step,
403
+ )
404
+
405
+
406
+ def _fallback_reason(page: WorkdayPage, company: Company, payload: Any) -> str:
407
+ if page.facets:
408
+ names = {group.facetParameter for group in page.facets if group.facetParameter}
409
+ advertised = ", ".join(sorted(names))
410
+ return (
411
+ f"no internship facet among the values this tenant advertises ({advertised}); "
412
+ "walking the whole board instead, which the bilingual classifier then filters"
413
+ )
414
+ if page.facets == []:
415
+ return (
416
+ "this tenant advertises an empty facet list, so there is no internship facet "
417
+ "to resolve; walking the whole board instead"
418
+ )
419
+ captured = capture_payload("workday-nofacets", company.slug, payload)
420
+ return f"no facet list at all, so resolution could not run; payload captured at {captured}"
421
+
422
+
423
+ def _pinned_facet(company: Company) -> WorkdayFacet | None:
424
+ if not company.workday_facet:
425
+ return None
426
+ parameter, _, value = company.workday_facet.partition(":")
427
+ if not value:
428
+ return WorkdayFacet(
429
+ tenant=company.workday_tenant or "",
430
+ site=company.workday_site or "",
431
+ parameter=workday_facet_lexicon()[0][0],
432
+ facet_ids=(parameter,),
433
+ pinned=True,
434
+ )
435
+ return WorkdayFacet(
436
+ tenant=company.workday_tenant or "",
437
+ site=company.workday_site or "",
438
+ parameter=parameter,
439
+ facet_ids=tuple(value.split(",")),
440
+ pinned=True,
441
+ )
442
+
443
+
444
+ def _validate(payload: Any, company: Company) -> tuple[WorkdayPage, int]:
445
+ try:
446
+ raw = WorkdayRawPage.model_validate(payload)
447
+ except ValidationError as exc:
448
+ captured = capture_payload("workday", company.slug, payload)
449
+ raise PayloadValidationError(
450
+ f"workday payload for {company.name} failed validation: {exc} (captured {captured})"
451
+ ) from exc
452
+
453
+ kept, dropped = validate_rows(
454
+ WorkdayPosting, raw.jobPostings, source="workday", slug=company.slug
455
+ )
456
+ return WorkdayPage(total=raw.total, jobPostings=kept, facets=raw.facets), dropped
457
+
458
+
459
+ async def _attach_descriptions(
460
+ company: Company,
461
+ client: HttpClient,
462
+ paired: list[tuple[WorkdayPosting, Job]],
463
+ wanted: set[str],
464
+ ) -> tuple[list[tuple[WorkdayPosting, Job]], list[DetailFetch]]:
465
+ host, _ = workday_target(
466
+ company.workday_tenant or "", company.workday_site or "", company.workday_dc or ""
467
+ )
468
+ outcomes: list[DetailFetch] = []
469
+ merged: list[tuple[WorkdayPosting, Job]] = []
470
+ for posting, job in paired:
471
+ if job.id not in wanted or not posting.externalPath:
472
+ merged.append((posting, job))
473
+ continue
474
+ path = posting.externalPath
475
+ url = f"https://{host}/wday/cxs/{company.workday_tenant}/{company.workday_site}{path}"
476
+ try:
477
+ response = await client.get_json(url)
478
+ except HttpError:
479
+ outcomes.append(DetailFetch(id=job.id, resolved=False, failed=True))
480
+ merged.append((posting, job))
481
+ continue
482
+ body = _description_from(response.payload)
483
+ outcomes.append(DetailFetch(id=job.id, resolved=bool(body)))
484
+ merged.append((posting, replace(job, description=body) if body else job))
485
+ return merged, outcomes
486
+
487
+
488
+ def _description_from(payload: Any) -> str:
489
+ if not isinstance(payload, dict):
490
+ return ""
491
+ info = payload.get("jobPostingInfo")
492
+ if not isinstance(info, dict):
493
+ return ""
494
+ body = info.get("jobDescription")
495
+ return collapse_whitespace(strip_html(body)) if isinstance(body, str) else ""
496
+
497
+
498
+ def _board(company: Company) -> str:
499
+ return f"{company.workday_tenant or company.slug}-{company.workday_site or ''}"
500
+
501
+
502
+ def _to_job(
503
+ company: Company, posting: WorkdayPosting, now: datetime, employment_type: str = ""
504
+ ) -> Job:
505
+ host, _ = workday_target(
506
+ company.workday_tenant or "", company.workday_site or "", company.workday_dc or ""
507
+ )
508
+ raw_path = posting.externalPath
509
+ path = raw_path if raw_path.startswith("/") else f"/{raw_path}"
510
+ apply_url = f"https://{host}/{company.workday_site}{path}" if raw_path else ""
511
+ title = collapse_whitespace(posting.title)
512
+ return Job(
513
+ id=job_id("workday", _board(company), posting.requisition()),
514
+ source="workday",
515
+ company=company.name,
516
+ title_raw=title,
517
+ title_normalized=title.lower(),
518
+ apply_url_raw=apply_url,
519
+ description="",
520
+ location_raw=collapse_whitespace(posting.locationsText),
521
+ first_seen=now,
522
+ last_seen=now,
523
+ signals=SourceSignals(employment_type=employment_type),
524
+ )
stage/sources/zshah.py ADDED
@@ -0,0 +1,99 @@
1
+ from datetime import datetime
2
+ from typing import Any, ClassVar
3
+
4
+ from pydantic import BaseModel, ConfigDict, Field
5
+
6
+ from stage.domain import Job, SourceSignals, job_id
7
+ from stage.http import HttpClient
8
+ from stage.sources._text import collapse_whitespace
9
+ from stage.sources.base import (
10
+ FetchResult,
11
+ NonEmptyStr,
12
+ PayloadValidationError,
13
+ capture_payload,
14
+ validate_rows,
15
+ )
16
+ from stage.sources.feed import register_feed, upcoming_season_year
17
+
18
+ URL = "https://zshah101.github.io/Automated-List-Of-Summer-{year}-and-Fall-2026-Tech-Internships/api/jobs.json"
19
+ HOST = "zshah101.github.io"
20
+
21
+
22
+ class ZshahListing(BaseModel):
23
+ model_config = ConfigDict(extra="ignore")
24
+
25
+ id: NonEmptyStr
26
+ company: NonEmptyStr
27
+ title: NonEmptyStr
28
+ location: str = ""
29
+ url: NonEmptyStr
30
+ program: str = ""
31
+
32
+
33
+ class ZshahEnvelope(BaseModel):
34
+ model_config = ConfigDict(extra="ignore")
35
+
36
+ jobs: list[Any] = Field(default_factory=list)
37
+
38
+
39
+ @register_feed
40
+ class ZshahFeed:
41
+ name: ClassVar[str] = "zshah101"
42
+ rate_profile: ClassVar[str] = "feeds"
43
+ hosts: ClassVar[frozenset[str]] = frozenset({HOST})
44
+ bucket_key: ClassVar[str] = ""
45
+
46
+ def season_year(self, now: datetime) -> int:
47
+ return upcoming_season_year(now)
48
+
49
+ def plan(self, now: datetime) -> tuple[str, ...]:
50
+ return (URL.format(year=self.season_year(now)),)
51
+
52
+ async def fetch(self, client: HttpClient, now: datetime) -> FetchResult:
53
+ url = self.plan(now)[0]
54
+ response = await client.get_json(url)
55
+ if response.not_modified:
56
+ return FetchResult(not_modified=True)
57
+ try:
58
+ envelope = ZshahEnvelope.model_validate(response.payload)
59
+ except Exception as exc:
60
+ captured = capture_payload(self.name, str(self.season_year(now)), response.payload)
61
+ raise PayloadValidationError(
62
+ f"{self.name}/{url}: payload envelope failed validation (captured {captured})"
63
+ ) from exc
64
+ listings, malformed = validate_rows(
65
+ ZshahListing, envelope.jobs, source=self.name, slug=str(self.season_year(now))
66
+ )
67
+ internships = [
68
+ listing for listing in listings if listing.program.casefold() == "internship"
69
+ ]
70
+ if not internships:
71
+ captured = capture_payload(self.name, str(self.season_year(now)), response.payload)
72
+ raise PayloadValidationError(
73
+ f"{self.name}/{url}: no internship records were found (captured {captured})"
74
+ )
75
+ jobs = tuple(
76
+ Job(
77
+ id=job_id(self.name, listing.company, listing.id),
78
+ source=self.name,
79
+ company=collapse_whitespace(listing.company),
80
+ title_raw=collapse_whitespace(listing.title),
81
+ title_normalized=collapse_whitespace(listing.title),
82
+ apply_url_raw=listing.url,
83
+ description="",
84
+ location_raw=collapse_whitespace(listing.location),
85
+ first_seen=now,
86
+ last_seen=now,
87
+ signals=SourceSignals(employment_type="internship"),
88
+ )
89
+ for listing in internships
90
+ )
91
+ return FetchResult(
92
+ jobs=jobs,
93
+ degraded=(
94
+ f"{malformed} malformed posting(s) were dropped; the feed closes nothing"
95
+ if malformed
96
+ else ""
97
+ ),
98
+ authoritative=not malformed,
99
+ )
@@ -0,0 +1,29 @@
1
+ from collections.abc import AsyncIterator
2
+ from contextlib import asynccontextmanager
3
+ from pathlib import Path
4
+
5
+ from stage.storage.repository import Repository, SourceBatch, SourceBatchResult
6
+ from stage.storage.sqlite_repo import SqliteRepository
7
+ from stage.storage.writer import AsyncRepository, DatabaseWriter, WriterNotStartedError
8
+
9
+
10
+ @asynccontextmanager
11
+ async def open_repository(db_path: Path) -> AsyncIterator[AsyncRepository]:
12
+ writer = DatabaseWriter(db_path)
13
+ await writer.start()
14
+ try:
15
+ yield AsyncRepository(writer)
16
+ finally:
17
+ await writer.aclose()
18
+
19
+
20
+ __all__ = [
21
+ "AsyncRepository",
22
+ "DatabaseWriter",
23
+ "Repository",
24
+ "SourceBatch",
25
+ "SourceBatchResult",
26
+ "SqliteRepository",
27
+ "WriterNotStartedError",
28
+ "open_repository",
29
+ ]