stage-cli 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. stage/__init__.py +1 -0
  2. stage/__main__.py +8 -0
  3. stage/banner.py +32 -0
  4. stage/bootstrap/__init__.py +0 -0
  5. stage/bootstrap/openjobs.py +392 -0
  6. stage/classify/__init__.py +29 -0
  7. stage/classify/eligibility.py +115 -0
  8. stage/classify/internship.py +64 -0
  9. stage/classify/role.py +91 -0
  10. stage/classify/scope.py +47 -0
  11. stage/cli/__init__.py +0 -0
  12. stage/cli/app.py +4 -0
  13. stage/cli/commands/__init__.py +8 -0
  14. stage/cli/commands/discovery.py +294 -0
  15. stage/cli/commands/insight.py +494 -0
  16. stage/cli/commands/pipeline.py +337 -0
  17. stage/cli/commands/postings.py +473 -0
  18. stage/cli/commands/schedule.py +171 -0
  19. stage/cli/housekeeping.py +64 -0
  20. stage/cli/logfile.py +56 -0
  21. stage/cli/notify.py +170 -0
  22. stage/cli/options.py +678 -0
  23. stage/cli/render.py +1398 -0
  24. stage/cli/runlock.py +74 -0
  25. stage/cli/schedule.py +702 -0
  26. stage/cli/schedule_state.py +363 -0
  27. stage/cli/selection.py +83 -0
  28. stage/cli/serialize.py +196 -0
  29. stage/companies.py +542 -0
  30. stage/data/companies/a.yaml +1289 -0
  31. stage/data/companies/b.yaml +900 -0
  32. stage/data/companies/c.yaml +1377 -0
  33. stage/data/companies/d.yaml +497 -0
  34. stage/data/companies/e.yaml +519 -0
  35. stage/data/companies/f.yaml +454 -0
  36. stage/data/companies/g.yaml +601 -0
  37. stage/data/companies/h.yaml +446 -0
  38. stage/data/companies/i.yaml +503 -0
  39. stage/data/companies/j.yaml +138 -0
  40. stage/data/companies/k.yaml +278 -0
  41. stage/data/companies/l.yaml +402 -0
  42. stage/data/companies/m.yaml +937 -0
  43. stage/data/companies/n.yaml +549 -0
  44. stage/data/companies/o.yaml +371 -0
  45. stage/data/companies/other.yaml +58 -0
  46. stage/data/companies/p.yaml +825 -0
  47. stage/data/companies/q.yaml +121 -0
  48. stage/data/companies/r.yaml +583 -0
  49. stage/data/companies/s.yaml +1140 -0
  50. stage/data/companies/t.yaml +817 -0
  51. stage/data/companies/u.yaml +196 -0
  52. stage/data/companies/v.yaml +325 -0
  53. stage/data/companies/w.yaml +353 -0
  54. stage/data/companies/x.yaml +67 -0
  55. stage/data/companies/y.yaml +36 -0
  56. stage/data/companies/z.yaml +146 -0
  57. stage/data/fonts/DejaVuSans.LICENSE.txt +99 -0
  58. stage/data/fonts/DejaVuSans.ttf +0 -0
  59. stage/data/lexicon/company_tokens.yaml +228 -0
  60. stage/data/lexicon/eligibility.yaml +455 -0
  61. stage/data/lexicon/inclusive_suffixes.yaml +37 -0
  62. stage/data/lexicon/internship.yaml +187 -0
  63. stage/data/lexicon/language.yaml +226 -0
  64. stage/data/lexicon/locations.yaml +1159 -0
  65. stage/data/lexicon/roles.yaml +2012 -0
  66. stage/data/lexicon/terms.yaml +76 -0
  67. stage/data/lexicon/workday_facets.yaml +27 -0
  68. stage/data/seed_companies.yaml +198 -0
  69. stage/dedup/__init__.py +19 -0
  70. stage/dedup/identity.py +113 -0
  71. stage/dedup/resolve.py +97 -0
  72. stage/domain/__init__.py +244 -0
  73. stage/domain/company.py +49 -0
  74. stage/domain/coverage.py +86 -0
  75. stage/domain/custom_board.py +92 -0
  76. stage/domain/discovery.py +94 -0
  77. stage/domain/enums.py +114 -0
  78. stage/domain/events.py +204 -0
  79. stage/domain/filters.py +27 -0
  80. stage/domain/health.py +169 -0
  81. stage/domain/ids.py +48 -0
  82. stage/domain/job.py +47 -0
  83. stage/domain/matching.py +15 -0
  84. stage/domain/priority.py +34 -0
  85. stage/domain/quarantine.py +39 -0
  86. stage/domain/rate_state.py +78 -0
  87. stage/domain/retention.py +20 -0
  88. stage/domain/rotation.py +46 -0
  89. stage/domain/signals.py +12 -0
  90. stage/domain/sync_run.py +35 -0
  91. stage/domain/text.py +113 -0
  92. stage/domain/validator.py +14 -0
  93. stage/domain/visits.py +60 -0
  94. stage/domain/workday.py +38 -0
  95. stage/http/__init__.py +58 -0
  96. stage/http/breaker.py +53 -0
  97. stage/http/cache.py +44 -0
  98. stage/http/client.py +725 -0
  99. stage/http/profiles.py +101 -0
  100. stage/lexicon.py +370 -0
  101. stage/normalize/__init__.py +16 -0
  102. stage/normalize/language.py +47 -0
  103. stage/normalize/location.py +271 -0
  104. stage/normalize/terms.py +153 -0
  105. stage/normalize/urls.py +122 -0
  106. stage/paths.py +86 -0
  107. stage/py.typed +0 -0
  108. stage/services/__init__.py +0 -0
  109. stage/services/canary.py +120 -0
  110. stage/services/coverage.py +231 -0
  111. stage/services/discover.py +747 -0
  112. stage/services/export.py +274 -0
  113. stage/services/health.py +237 -0
  114. stage/services/maintenance.py +225 -0
  115. stage/services/quarantine.py +20 -0
  116. stage/services/query.py +86 -0
  117. stage/services/sync.py +1257 -0
  118. stage/sources/__init__.py +82 -0
  119. stage/sources/_text.py +79 -0
  120. stage/sources/ashby.py +93 -0
  121. stage/sources/bamboohr.py +80 -0
  122. stage/sources/base.py +225 -0
  123. stage/sources/breezy.py +90 -0
  124. stage/sources/collage.py +60 -0
  125. stage/sources/community_feeds.py +142 -0
  126. stage/sources/curated_markdown.py +289 -0
  127. stage/sources/custom_json.py +610 -0
  128. stage/sources/espresso.py +154 -0
  129. stage/sources/feed.py +44 -0
  130. stage/sources/greenhouse.py +104 -0
  131. stage/sources/jobbank.py +147 -0
  132. stage/sources/jobvite.py +133 -0
  133. stage/sources/lever.py +76 -0
  134. stage/sources/oracle_cloud.py +187 -0
  135. stage/sources/platforms.py +609 -0
  136. stage/sources/quebec_emploi.py +146 -0
  137. stage/sources/recruitee.py +96 -0
  138. stage/sources/simplify.py +110 -0
  139. stage/sources/smartrecruiters.py +216 -0
  140. stage/sources/speedyapply.py +200 -0
  141. stage/sources/themuse.py +157 -0
  142. stage/sources/workable.py +83 -0
  143. stage/sources/workday.py +524 -0
  144. stage/sources/zshah.py +99 -0
  145. stage/storage/__init__.py +29 -0
  146. stage/storage/migrations/0001_initial.sql +239 -0
  147. stage/storage/migrations/__init__.py +135 -0
  148. stage/storage/repository.py +213 -0
  149. stage/storage/search.py +28 -0
  150. stage/storage/sqlite_repo.py +1586 -0
  151. stage/storage/writer.py +249 -0
  152. stage/tui/__init__.py +0 -0
  153. stage/tui/app.py +82 -0
  154. stage/tui/help.py +26 -0
  155. stage/tui/safe.py +21 -0
  156. stage/tui/screens/__init__.py +0 -0
  157. stage/tui/screens/boards.py +186 -0
  158. stage/tui/screens/postings.py +509 -0
  159. stage/tui/screens/review.py +209 -0
  160. stage/tui/screens/splash.py +37 -0
  161. stage/tui/screens/stats.py +124 -0
  162. stage/tui/screens/sync.py +194 -0
  163. stage/tui/state.py +160 -0
  164. stage/tui/theme.tcss +205 -0
  165. stage/tui/widgets/__init__.py +0 -0
  166. stage_cli-1.0.0.dist-info/METADATA +379 -0
  167. stage_cli-1.0.0.dist-info/RECORD +170 -0
  168. stage_cli-1.0.0.dist-info/WHEEL +4 -0
  169. stage_cli-1.0.0.dist-info/entry_points.txt +2 -0
  170. stage_cli-1.0.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,154 @@
1
+ from datetime import datetime
2
+ from typing import ClassVar
3
+
4
+ from bs4 import BeautifulSoup
5
+ from bs4.element import Tag
6
+
7
+ from stage.domain import Job, SourceSignals, job_id
8
+ from stage.http import HttpClient, HttpStatusError
9
+ from stage.sources import register_feed
10
+ from stage.sources._text import collapse_whitespace
11
+ from stage.sources.base import FetchResult, PayloadValidationError, capture_payload, malformed_note
12
+
13
+ HOST = "www.espresso-jobs.com"
14
+ SEARCH = f"https://{HOST}/emploi"
15
+ POSTING = f"https://{HOST}/emploi/{{id}}/{{slug}}"
16
+ TERMS = ("stage", "stagiaire", "intern", "internship")
17
+ PAGE_CAP = 3
18
+ PAGE_SIZE = 21
19
+ MAX_ROWS = 1000
20
+ INTERNSHIP_BADGE = "stage"
21
+ ROW_SELECTOR = "div.job_index-content_list_item"
22
+
23
+
24
+ def badge(row: Tag) -> str:
25
+ found = row.select_one("p.job_index-content_list_item_infos-type")
26
+ if found is None:
27
+ return ""
28
+ return collapse_whitespace(found.get_text(" ", strip=True))
29
+
30
+
31
+ def field(row: Tag, selector: str) -> str:
32
+ found = row.select_one(selector)
33
+ if found is None:
34
+ return ""
35
+ return collapse_whitespace(found.get_text(" ", strip=True))
36
+
37
+
38
+ def where(row: Tag) -> str:
39
+ found = row.select_one("div.job-location-info")
40
+ if found is None:
41
+ return "Québec, Canada"
42
+ city = collapse_whitespace(str(found.get("data-city") or ""))
43
+ province = collapse_whitespace(str(found.get("data-province") or ""))
44
+ parts = [part for part in (city, province, "Canada") if part]
45
+ return ", ".join(parts)
46
+
47
+
48
+ @register_feed
49
+ class EspressoJobsFeed:
50
+ name: ClassVar[str] = "espresso-jobs"
51
+ rate_profile: ClassVar[str] = "feeds"
52
+ hosts: ClassVar[frozenset[str]] = frozenset({HOST})
53
+ bucket_key: ClassVar[str] = "espresso-jobs"
54
+
55
+ def season_year(self, now: datetime) -> int:
56
+ return now.year
57
+
58
+ def plan(self, now: datetime) -> tuple[str, ...]:
59
+ return tuple(f"{SEARCH}?keyword={term}&distance=all&page_no=1" for term in TERMS)
60
+
61
+ async def fetch(self, client: HttpClient, now: datetime) -> FetchResult:
62
+ seen: dict[str, Job] = {}
63
+ malformed = 0
64
+ truncated = False
65
+ searched = False
66
+ empty: list[str] = []
67
+ last_empty = ""
68
+
69
+ exhausted: list[str] = []
70
+
71
+ for term in TERMS:
72
+ for page in range(1, PAGE_CAP + 1):
73
+ url = f"{SEARCH}?keyword={term}&distance=all&page_no={page}"
74
+ try:
75
+ text = await client.get_text(url, revalidate=page > 1)
76
+ except HttpStatusError as exc:
77
+ if page > 1 and exc.status == 404:
78
+ exhausted.append(term)
79
+ break
80
+ raise
81
+ if text.not_modified:
82
+ break
83
+ rows = self._rows(text.text)
84
+ if not rows:
85
+ if page == 1:
86
+ empty.append(term)
87
+ last_empty = text.text
88
+ break
89
+ searched = True
90
+ for row in rows:
91
+ job, dropped = self._to_job(row, now)
92
+ malformed += dropped
93
+ if job is not None:
94
+ seen.setdefault(job.id, job)
95
+ if len(seen) >= MAX_ROWS:
96
+ truncated = True
97
+ break
98
+ if len(rows) < PAGE_SIZE:
99
+ break
100
+ if truncated:
101
+ break
102
+
103
+ if not searched:
104
+ captured = capture_payload(self.name, "search", {"head": last_empty[:4000]})
105
+ raise PayloadValidationError(
106
+ f"{self.name}: no query returned a listing row, so the results page changed "
107
+ f"shape rather than matching nothing; captured at {captured}"
108
+ )
109
+
110
+ notes = []
111
+ if empty:
112
+ notes.append(f"{len(empty)} of {len(TERMS)} searches matched nothing")
113
+ if exhausted:
114
+ notes.append(f"{len(exhausted)} search(es) ran past their last page, which answers 404")
115
+ if truncated:
116
+ notes.append(f"stopped at the {MAX_ROWS}-posting cap for one run")
117
+ if malformed:
118
+ notes.append(malformed_note(malformed))
119
+ return FetchResult(
120
+ jobs=tuple(seen.values()),
121
+ authoritative=False,
122
+ degraded="; ".join(notes),
123
+ )
124
+
125
+ @staticmethod
126
+ def _rows(text: str) -> list[Tag]:
127
+ return BeautifulSoup(text, "html.parser").select(ROW_SELECTOR)
128
+
129
+ def _to_job(self, row: Tag, now: datetime) -> tuple[Job | None, int]:
130
+ identifier = collapse_whitespace(str(row.get("id") or ""))
131
+ slug = collapse_whitespace(str(row.get("data-slug") or ""))
132
+ title = field(row, "h2.job_index-content_list_item-title")
133
+ if not identifier or not slug or not title:
134
+ return None, 1
135
+ declared = badge(row)
136
+ if declared.lower() != INTERNSHIP_BADGE:
137
+ return None, 0
138
+ company = field(row, "p.job_index-content_list_item-company") or "Espresso-Jobs"
139
+ return (
140
+ Job(
141
+ id=job_id(self.name, "search", identifier),
142
+ source=self.name,
143
+ company=company,
144
+ title_raw=title,
145
+ title_normalized=title.lower(),
146
+ apply_url_raw=POSTING.format(id=identifier, slug=slug),
147
+ description="",
148
+ location_raw=where(row),
149
+ first_seen=now,
150
+ last_seen=now,
151
+ signals=SourceSignals(employment_type=declared.lower()),
152
+ ),
153
+ 0,
154
+ )
stage/sources/feed.py ADDED
@@ -0,0 +1,44 @@
1
+ from datetime import datetime
2
+ from typing import ClassVar, Protocol, runtime_checkable
3
+
4
+ from stage.http import HttpClient
5
+ from stage.sources.base import FetchResult
6
+
7
+ _FEEDS: dict[str, "FeedAdapter"] = {}
8
+
9
+
10
+ @runtime_checkable
11
+ class FeedAdapter(Protocol):
12
+ name: ClassVar[str]
13
+ rate_profile: ClassVar[str]
14
+ hosts: ClassVar[frozenset[str]]
15
+ bucket_key: ClassVar[str]
16
+
17
+ def season_year(self, now: datetime) -> int:
18
+ pass
19
+
20
+ def plan(self, now: datetime) -> tuple[str, ...]:
21
+ pass
22
+
23
+ async def fetch(self, client: HttpClient, now: datetime) -> FetchResult:
24
+ pass
25
+
26
+
27
+ def register_feed[F: FeedAdapter](cls: type[F]) -> type[F]:
28
+ adapter = cls()
29
+ existing = _FEEDS.get(adapter.name)
30
+ if existing is not None and type(existing) is not cls:
31
+ raise ValueError(f"two feeds claim the name {adapter.name!r}")
32
+ _FEEDS[adapter.name] = adapter
33
+ return cls
34
+
35
+
36
+ def get_feeds() -> dict[str, FeedAdapter]:
37
+ from stage.sources import load_builtins
38
+
39
+ load_builtins()
40
+ return dict(_FEEDS)
41
+
42
+
43
+ def upcoming_season_year(now: datetime, rolls_in_month: int = 8) -> int:
44
+ return now.year + 1 if now.month >= rolls_in_month else now.year
@@ -0,0 +1,104 @@
1
+ from collections.abc import Sequence
2
+ from datetime import datetime
3
+ from typing import Any, ClassVar
4
+
5
+ from pydantic import BaseModel, ConfigDict
6
+
7
+ from stage.domain import Company, Job, Platform, job_id
8
+ from stage.http import HttpClient, ResponseTooLargeError
9
+ from stage.sources import register
10
+ from stage.sources._text import collapse_whitespace, strip_html
11
+ from stage.sources.base import BoardAdapter, FetchResult, malformed_note
12
+
13
+ BASE_URL = "https://boards-api.greenhouse.io/v1/boards/{slug}/jobs"
14
+ HOST = "boards-api.greenhouse.io"
15
+
16
+
17
+ class GreenhouseLocation(BaseModel):
18
+ model_config = ConfigDict(extra="ignore")
19
+
20
+ name: str | None = ""
21
+
22
+
23
+ class GreenhouseJob(BaseModel):
24
+ model_config = ConfigDict(extra="ignore")
25
+
26
+ id: int
27
+ title: str
28
+ absolute_url: str
29
+ updated_at: datetime | None = None
30
+ location: GreenhouseLocation | None = None
31
+ content: str = ""
32
+
33
+
34
+ class GreenhouseBoard(BaseModel):
35
+ model_config = ConfigDict(extra="ignore")
36
+
37
+ jobs: list[Any]
38
+
39
+
40
+ @register
41
+ class GreenhouseAdapter(BoardAdapter):
42
+ name: ClassVar[str] = "greenhouse"
43
+ platform: ClassVar[Platform] = Platform.GREENHOUSE
44
+ rate_profile: ClassVar[str] = "broad"
45
+ hosts: ClassVar[frozenset[str]] = frozenset({HOST})
46
+ detail_budget: ClassVar[int] = 0
47
+ max_requests_per_company: ClassVar[int] = 2
48
+
49
+ base_url: ClassVar[str] = BASE_URL
50
+ query: ClassVar[tuple[tuple[str, str], ...]] = (("content", "true"),)
51
+ root_model: ClassVar[type[BaseModel] | None] = GreenhouseBoard
52
+ rows_field: ClassVar[str] = "jobs"
53
+ row_model: ClassVar[type[BaseModel]] = GreenhouseJob
54
+
55
+ async def fetch(
56
+ self,
57
+ company: Company,
58
+ client: HttpClient,
59
+ now: datetime,
60
+ facets: object = None,
61
+ details: Sequence[str] = (),
62
+ ) -> FetchResult:
63
+ url = self.url_for(company)
64
+ try:
65
+ response = await client.get_json(url, params={"content": "true"})
66
+ except ResponseTooLargeError:
67
+ return await self._without_descriptions(company, client, now, url)
68
+ if response.not_modified:
69
+ return FetchResult(not_modified=True)
70
+ return self.result(company, response.payload, now)
71
+
72
+ async def _without_descriptions(
73
+ self, company: Company, client: HttpClient, now: datetime, url: str
74
+ ) -> FetchResult:
75
+ response = await client.get_json(url, params={"content": "false"})
76
+ if response.not_modified:
77
+ return FetchResult(not_modified=True)
78
+ postings, dropped = self.validate(company, response.payload)
79
+ notes = ["board exceeds the response cap with content=true; fetched without descriptions"]
80
+ if dropped:
81
+ notes.append(malformed_note(dropped))
82
+ return FetchResult(
83
+ jobs=tuple(self.to_job(company, posting, now) for posting in postings),
84
+ degraded="; ".join(notes),
85
+ authoritative=not dropped,
86
+ stale_urls=(f"{url}?content=false",),
87
+ )
88
+
89
+ def to_job(self, company: Company, row: Any, now: datetime) -> Job:
90
+ return Job(
91
+ id=job_id(self.name, company.slug, str(row.id)),
92
+ source=self.name,
93
+ company=company.name,
94
+ title_raw=row.title,
95
+ title_normalized=collapse_whitespace(row.title),
96
+ apply_url_raw=row.absolute_url,
97
+ description=strip_html(row.content),
98
+ location_raw=collapse_whitespace(
99
+ row.location.name if row.location and row.location.name else ""
100
+ ),
101
+ first_seen=now,
102
+ last_seen=now,
103
+ source_posted_at=row.updated_at,
104
+ )
@@ -0,0 +1,147 @@
1
+ import re
2
+ from datetime import datetime
3
+ from typing import ClassVar
4
+
5
+ from bs4 import BeautifulSoup
6
+ from bs4.element import Tag
7
+
8
+ from stage.domain import Job, SourceSignals, job_id
9
+ from stage.http import HttpClient
10
+ from stage.sources import register_feed
11
+ from stage.sources._text import collapse_whitespace
12
+ from stage.sources.base import FetchResult, PayloadValidationError, capture_payload, malformed_note
13
+
14
+ HOST = "www.jobbank.gc.ca"
15
+ SEARCH = f"https://{HOST}/jobsearch/jobsearch"
16
+ POSTING = f"https://{HOST}/jobsearch/jobposting/{{id}}"
17
+ TERMS = ("programmer", "developer", "software", "informatique", "programmeur", "développeur")
18
+ PROVINCES = ("QC", "ON", "BC", "AB")
19
+ PAGE_CAP = 2
20
+ MAX_ROWS = 2000
21
+ _ARTICLE_ID = re.compile(r"^article-(\d+)$")
22
+ KEPT_FLAGS = ("jobinternshipflag", "jobstudentflag")
23
+
24
+
25
+ def posting_id(article: Tag) -> str:
26
+ found = _ARTICLE_ID.match(str(article.get("id") or ""))
27
+ return found.group(1) if found else ""
28
+
29
+
30
+ def declared_term(article: Tag) -> str:
31
+ for flag in KEPT_FLAGS:
32
+ found = article.select_one(f".{flag}")
33
+ if found is not None:
34
+ return collapse_whitespace(found.get_text(" ", strip=True))
35
+ return ""
36
+
37
+
38
+ def field(article: Tag, selector: str) -> str:
39
+ found = article.select_one(selector)
40
+ if found is None:
41
+ return ""
42
+ for hidden in found.select(".wb-inv"):
43
+ hidden.decompose()
44
+ return collapse_whitespace(found.get_text(" ", strip=True))
45
+
46
+
47
+ @register_feed
48
+ class JobBankFeed:
49
+ name: ClassVar[str] = "jobbank"
50
+ rate_profile: ClassVar[str] = "jobbank"
51
+ hosts: ClassVar[frozenset[str]] = frozenset({HOST})
52
+ bucket_key: ClassVar[str] = "jobbank"
53
+
54
+ def season_year(self, now: datetime) -> int:
55
+ return now.year
56
+
57
+ def plan(self, now: datetime) -> tuple[str, ...]:
58
+ return tuple(
59
+ f"{SEARCH}?searchstring={term}&fprov={province}"
60
+ for province in PROVINCES
61
+ for term in TERMS
62
+ )
63
+
64
+ async def fetch(self, client: HttpClient, now: datetime) -> FetchResult:
65
+ seen: dict[str, Job] = {}
66
+ malformed = 0
67
+ truncated = False
68
+ searched = False
69
+ empty: list[str] = []
70
+ last_empty = ""
71
+ for province in PROVINCES:
72
+ for term in TERMS:
73
+ for page in range(1, PAGE_CAP + 1):
74
+ url = f"{SEARCH}?searchstring={term}&fprov={province}&page={page}"
75
+ text = await client.get_text(url, revalidate=page > 1)
76
+ if text.not_modified:
77
+ break
78
+ rows = self._articles(text.text)
79
+ if not rows:
80
+ if page == 1:
81
+ empty.append(f"{province}/{term}")
82
+ last_empty = text.text
83
+ break
84
+ searched = True
85
+ for article in rows:
86
+ job, dropped = self._to_job(article, now)
87
+ malformed += dropped
88
+ if job is not None:
89
+ seen.setdefault(job.id, job)
90
+ if len(seen) >= MAX_ROWS:
91
+ truncated = True
92
+ break
93
+ if truncated:
94
+ break
95
+ if truncated:
96
+ break
97
+
98
+ if not searched:
99
+ captured = capture_payload(self.name, "search", {"head": last_empty[:4000]})
100
+ raise PayloadValidationError(
101
+ f"{self.name}: no query returned an <article> row, so the results page changed "
102
+ f"shape rather than matching nothing; captured at {captured}"
103
+ )
104
+
105
+ notes = []
106
+ if empty:
107
+ notes.append(f"{len(empty)} of {len(PROVINCES) * len(TERMS)} searches matched nothing")
108
+ if truncated:
109
+ notes.append(f"stopped at the {MAX_ROWS}-posting cap for one run")
110
+ if malformed:
111
+ notes.append(malformed_note(malformed))
112
+ return FetchResult(
113
+ jobs=tuple(seen.values()),
114
+ authoritative=False,
115
+ degraded="; ".join(notes),
116
+ )
117
+
118
+ @staticmethod
119
+ def _articles(text: str) -> list[Tag]:
120
+ return BeautifulSoup(text, "html.parser").select("article")
121
+
122
+ def _to_job(self, article: Tag, now: datetime) -> tuple[Job | None, int]:
123
+ identifier = posting_id(article)
124
+ title = field(article, ".noctitle")
125
+ term = declared_term(article)
126
+ if not identifier or not title:
127
+ return None, 1
128
+ if not term:
129
+ return None, 0
130
+ company = field(article, ".business") or "Job Bank"
131
+ city = field(article, ".location")
132
+ return (
133
+ Job(
134
+ id=job_id(self.name, "search", identifier),
135
+ source=self.name,
136
+ company=company,
137
+ title_raw=title,
138
+ title_normalized=title.lower(),
139
+ apply_url_raw=POSTING.format(id=identifier),
140
+ description="",
141
+ location_raw=city or "Canada",
142
+ first_seen=now,
143
+ last_seen=now,
144
+ signals=SourceSignals(employment_type=term.lower()),
145
+ ),
146
+ 0,
147
+ )
@@ -0,0 +1,133 @@
1
+ import re
2
+ from collections.abc import Sequence
3
+ from datetime import datetime
4
+ from typing import Any, ClassVar
5
+ from urllib.parse import urljoin
6
+
7
+ from bs4 import Tag
8
+ from pydantic import BaseModel, ConfigDict
9
+
10
+ from stage.domain import Company, Job, Platform, job_id
11
+ from stage.http import HttpClient
12
+ from stage.sources import register
13
+ from stage.sources._text import collapse_whitespace
14
+ from stage.sources.base import (
15
+ BoardAdapter,
16
+ FetchResult,
17
+ NonEmptyStr,
18
+ PayloadValidationError,
19
+ capture_payload,
20
+ malformed_note,
21
+ validate_rows,
22
+ )
23
+ from stage.sources.custom_json import html_rows
24
+
25
+ HOST = "jobs.jobvite.com"
26
+ BASE_URL = "https://jobs.jobvite.com/{slug}/jobs"
27
+ LISTING = "table.jv-job-list"
28
+ ROW = "table.jv-job-list tbody tr"
29
+ NAME_CELL = "td.jv-job-list-name"
30
+ LOCATION_CELL = "td.jv-job-list-location"
31
+ ONCLICK = re.compile(r"""location\.href\s*=\s*['"](?P<href>[^'"]{1,300})['"]""")
32
+
33
+
34
+ class JobviteRow(BaseModel):
35
+ model_config = ConfigDict(extra="ignore")
36
+
37
+ title: NonEmptyStr
38
+ url: NonEmptyStr
39
+ location: str = ""
40
+
41
+ def posting_id(self) -> str:
42
+ return self.url.split("?", 1)[0].rstrip("/").rsplit("/", 1)[-1]
43
+
44
+
45
+ def _text(block: Tag, selector: str) -> str:
46
+ found = block.select_one(selector)
47
+ return collapse_whitespace(found.get_text(" ", strip=True)) if found is not None else ""
48
+
49
+
50
+ def _href(block: Tag) -> str:
51
+ link = block.select_one(f"{NAME_CELL} a")
52
+ if link is not None:
53
+ return str(link.get("href", ""))
54
+ found = ONCLICK.search(str(block.get("onclick", "")))
55
+ return found.group("href") if found is not None else ""
56
+
57
+
58
+ def _row(block: Tag) -> dict[str, str]:
59
+ href = _href(block)
60
+ return {
61
+ "title": _text(block, NAME_CELL),
62
+ "url": urljoin(f"https://{HOST}/", href) if href else "",
63
+ "location": _text(block, LOCATION_CELL),
64
+ }
65
+
66
+
67
+ @register
68
+ class JobviteAdapter(BoardAdapter):
69
+ name: ClassVar[str] = "jobvite"
70
+ platform: ClassVar[Platform] = Platform.JOBVITE
71
+ rate_profile: ClassVar[str] = "moderate"
72
+ bucket_key: ClassVar[str] = "jobvite"
73
+ hosts: ClassVar[frozenset[str]] = frozenset({HOST})
74
+ detail_budget: ClassVar[int] = 0
75
+ max_requests_per_company: ClassVar[int] = 1
76
+
77
+ base_url: ClassVar[str] = BASE_URL
78
+ row_model: ClassVar[type[BaseModel]] = JobviteRow
79
+
80
+ async def fetch(
81
+ self,
82
+ company: Company,
83
+ client: HttpClient,
84
+ now: datetime,
85
+ facets: object = None,
86
+ details: Sequence[str] = (),
87
+ ) -> FetchResult:
88
+ response = await client.get_text(self.url_for(company))
89
+ if response.not_modified:
90
+ return FetchResult(not_modified=True)
91
+ return self.result(company, response.text, now)
92
+
93
+ def result(self, company: Company, payload: Any, now: datetime) -> FetchResult:
94
+ blocks = self._blocks(company, str(payload))
95
+ listed = [block for block in blocks if block.select_one(NAME_CELL) is not None]
96
+ rows, dropped = validate_rows(
97
+ self.row_model, [_row(block) for block in listed], source=self.name, slug=company.slug
98
+ )
99
+ truncated = len(listed) != len(blocks)
100
+ notes = [malformed_note(dropped)] if dropped else []
101
+ if truncated:
102
+ notes.append(
103
+ "the board hides the rest of some categories behind a Show More search page, "
104
+ "so this listing closes nothing"
105
+ )
106
+ return FetchResult(
107
+ jobs=tuple(self.to_job(company, row, now) for row in rows),
108
+ degraded="; ".join(note for note in notes if note),
109
+ authoritative=not dropped and not truncated,
110
+ )
111
+
112
+ def _blocks(self, company: Company, text: str) -> list[Tag]:
113
+ if not html_rows(text, LISTING):
114
+ captured = capture_payload(self.name, company.slug, {"text": text})
115
+ raise PayloadValidationError(
116
+ f"{self.name}/{company.slug}: the board page carries no {LISTING!r} listing, "
117
+ f"so this is drift or a retired tenant; raw page captured at {captured}"
118
+ )
119
+ return html_rows(text, ROW)
120
+
121
+ def to_job(self, company: Company, row: Any, now: datetime) -> Job:
122
+ return Job(
123
+ id=job_id(self.name, company.slug, row.posting_id()),
124
+ source=self.name,
125
+ company=company.name,
126
+ title_raw=row.title,
127
+ title_normalized=row.title.lower(),
128
+ apply_url_raw=row.url,
129
+ description="",
130
+ location_raw=row.location,
131
+ first_seen=now,
132
+ last_seen=now,
133
+ )
stage/sources/lever.py ADDED
@@ -0,0 +1,76 @@
1
+ from datetime import UTC, datetime
2
+ from typing import Any, ClassVar
3
+
4
+ from pydantic import BaseModel, ConfigDict, Field
5
+
6
+ from stage.domain import Company, Job, Platform, SourceSignals, job_id
7
+ from stage.sources import register
8
+ from stage.sources._text import collapse_whitespace
9
+ from stage.sources.base import BoardAdapter
10
+ from stage.sources.platforms import safe_cased_slug
11
+
12
+ BASE_URL = "https://api.lever.co/v0/postings/{slug}"
13
+ HOST = "api.lever.co"
14
+
15
+
16
+ class LeverCategories(BaseModel):
17
+ model_config = ConfigDict(extra="ignore")
18
+
19
+ location: str = ""
20
+ allLocations: list[str] = Field(default_factory=list)
21
+ commitment: str = ""
22
+ team: str = ""
23
+ department: str = ""
24
+
25
+ def label(self) -> str:
26
+ if self.allLocations:
27
+ return " / ".join(dict.fromkeys(self.allLocations))
28
+ return self.location
29
+
30
+
31
+ class LeverPosting(BaseModel):
32
+ model_config = ConfigDict(extra="ignore")
33
+
34
+ id: str
35
+ text: str
36
+ hostedUrl: str = ""
37
+ applyUrl: str = ""
38
+ createdAt: int | None = None
39
+ categories: LeverCategories | None = None
40
+ descriptionPlain: str = ""
41
+ additionalPlain: str = ""
42
+
43
+
44
+ @register
45
+ class LeverAdapter(BoardAdapter):
46
+ name: ClassVar[str] = "lever"
47
+ platform: ClassVar[Platform] = Platform.LEVER
48
+ rate_profile: ClassVar[str] = "standard"
49
+ hosts: ClassVar[frozenset[str]] = frozenset({HOST})
50
+ detail_budget: ClassVar[int] = 0
51
+ max_requests_per_company: ClassVar[int] = 1
52
+
53
+ base_url: ClassVar[str] = BASE_URL
54
+ query: ClassVar[tuple[tuple[str, str], ...]] = (("mode", "json"),)
55
+ row_model: ClassVar[type[BaseModel]] = LeverPosting
56
+ slug_validator = safe_cased_slug
57
+
58
+ def to_job(self, company: Company, row: Any, now: datetime) -> Job:
59
+ posted = datetime.fromtimestamp(row.createdAt / 1000, tz=UTC) if row.createdAt else None
60
+ parts = (row.descriptionPlain, row.additionalPlain)
61
+ return Job(
62
+ id=job_id(self.name, company.slug, row.id),
63
+ source=self.name,
64
+ company=company.name,
65
+ title_raw=row.text,
66
+ title_normalized=collapse_whitespace(row.text),
67
+ apply_url_raw=row.hostedUrl or row.applyUrl,
68
+ description="\n\n".join(part for part in parts if part),
69
+ location_raw=collapse_whitespace(row.categories.label() if row.categories else ""),
70
+ first_seen=now,
71
+ last_seen=now,
72
+ source_posted_at=posted,
73
+ signals=SourceSignals(
74
+ employment_type=row.categories.commitment if row.categories else ""
75
+ ),
76
+ )