stage-cli 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. stage/__init__.py +1 -0
  2. stage/__main__.py +8 -0
  3. stage/banner.py +32 -0
  4. stage/bootstrap/__init__.py +0 -0
  5. stage/bootstrap/openjobs.py +392 -0
  6. stage/classify/__init__.py +29 -0
  7. stage/classify/eligibility.py +115 -0
  8. stage/classify/internship.py +64 -0
  9. stage/classify/role.py +91 -0
  10. stage/classify/scope.py +47 -0
  11. stage/cli/__init__.py +0 -0
  12. stage/cli/app.py +4 -0
  13. stage/cli/commands/__init__.py +8 -0
  14. stage/cli/commands/discovery.py +294 -0
  15. stage/cli/commands/insight.py +494 -0
  16. stage/cli/commands/pipeline.py +337 -0
  17. stage/cli/commands/postings.py +473 -0
  18. stage/cli/commands/schedule.py +171 -0
  19. stage/cli/housekeeping.py +64 -0
  20. stage/cli/logfile.py +56 -0
  21. stage/cli/notify.py +170 -0
  22. stage/cli/options.py +678 -0
  23. stage/cli/render.py +1398 -0
  24. stage/cli/runlock.py +74 -0
  25. stage/cli/schedule.py +702 -0
  26. stage/cli/schedule_state.py +363 -0
  27. stage/cli/selection.py +83 -0
  28. stage/cli/serialize.py +196 -0
  29. stage/companies.py +542 -0
  30. stage/data/companies/a.yaml +1289 -0
  31. stage/data/companies/b.yaml +900 -0
  32. stage/data/companies/c.yaml +1377 -0
  33. stage/data/companies/d.yaml +497 -0
  34. stage/data/companies/e.yaml +519 -0
  35. stage/data/companies/f.yaml +454 -0
  36. stage/data/companies/g.yaml +601 -0
  37. stage/data/companies/h.yaml +446 -0
  38. stage/data/companies/i.yaml +503 -0
  39. stage/data/companies/j.yaml +138 -0
  40. stage/data/companies/k.yaml +278 -0
  41. stage/data/companies/l.yaml +402 -0
  42. stage/data/companies/m.yaml +937 -0
  43. stage/data/companies/n.yaml +549 -0
  44. stage/data/companies/o.yaml +371 -0
  45. stage/data/companies/other.yaml +58 -0
  46. stage/data/companies/p.yaml +825 -0
  47. stage/data/companies/q.yaml +121 -0
  48. stage/data/companies/r.yaml +583 -0
  49. stage/data/companies/s.yaml +1140 -0
  50. stage/data/companies/t.yaml +817 -0
  51. stage/data/companies/u.yaml +196 -0
  52. stage/data/companies/v.yaml +325 -0
  53. stage/data/companies/w.yaml +353 -0
  54. stage/data/companies/x.yaml +67 -0
  55. stage/data/companies/y.yaml +36 -0
  56. stage/data/companies/z.yaml +146 -0
  57. stage/data/fonts/DejaVuSans.LICENSE.txt +99 -0
  58. stage/data/fonts/DejaVuSans.ttf +0 -0
  59. stage/data/lexicon/company_tokens.yaml +228 -0
  60. stage/data/lexicon/eligibility.yaml +455 -0
  61. stage/data/lexicon/inclusive_suffixes.yaml +37 -0
  62. stage/data/lexicon/internship.yaml +187 -0
  63. stage/data/lexicon/language.yaml +226 -0
  64. stage/data/lexicon/locations.yaml +1159 -0
  65. stage/data/lexicon/roles.yaml +2012 -0
  66. stage/data/lexicon/terms.yaml +76 -0
  67. stage/data/lexicon/workday_facets.yaml +27 -0
  68. stage/data/seed_companies.yaml +198 -0
  69. stage/dedup/__init__.py +19 -0
  70. stage/dedup/identity.py +113 -0
  71. stage/dedup/resolve.py +97 -0
  72. stage/domain/__init__.py +244 -0
  73. stage/domain/company.py +49 -0
  74. stage/domain/coverage.py +86 -0
  75. stage/domain/custom_board.py +92 -0
  76. stage/domain/discovery.py +94 -0
  77. stage/domain/enums.py +114 -0
  78. stage/domain/events.py +204 -0
  79. stage/domain/filters.py +27 -0
  80. stage/domain/health.py +169 -0
  81. stage/domain/ids.py +48 -0
  82. stage/domain/job.py +47 -0
  83. stage/domain/matching.py +15 -0
  84. stage/domain/priority.py +34 -0
  85. stage/domain/quarantine.py +39 -0
  86. stage/domain/rate_state.py +78 -0
  87. stage/domain/retention.py +20 -0
  88. stage/domain/rotation.py +46 -0
  89. stage/domain/signals.py +12 -0
  90. stage/domain/sync_run.py +35 -0
  91. stage/domain/text.py +113 -0
  92. stage/domain/validator.py +14 -0
  93. stage/domain/visits.py +60 -0
  94. stage/domain/workday.py +38 -0
  95. stage/http/__init__.py +58 -0
  96. stage/http/breaker.py +53 -0
  97. stage/http/cache.py +44 -0
  98. stage/http/client.py +725 -0
  99. stage/http/profiles.py +101 -0
  100. stage/lexicon.py +370 -0
  101. stage/normalize/__init__.py +16 -0
  102. stage/normalize/language.py +47 -0
  103. stage/normalize/location.py +271 -0
  104. stage/normalize/terms.py +153 -0
  105. stage/normalize/urls.py +122 -0
  106. stage/paths.py +86 -0
  107. stage/py.typed +0 -0
  108. stage/services/__init__.py +0 -0
  109. stage/services/canary.py +120 -0
  110. stage/services/coverage.py +231 -0
  111. stage/services/discover.py +747 -0
  112. stage/services/export.py +274 -0
  113. stage/services/health.py +237 -0
  114. stage/services/maintenance.py +225 -0
  115. stage/services/quarantine.py +20 -0
  116. stage/services/query.py +86 -0
  117. stage/services/sync.py +1257 -0
  118. stage/sources/__init__.py +82 -0
  119. stage/sources/_text.py +79 -0
  120. stage/sources/ashby.py +93 -0
  121. stage/sources/bamboohr.py +80 -0
  122. stage/sources/base.py +225 -0
  123. stage/sources/breezy.py +90 -0
  124. stage/sources/collage.py +60 -0
  125. stage/sources/community_feeds.py +142 -0
  126. stage/sources/curated_markdown.py +289 -0
  127. stage/sources/custom_json.py +610 -0
  128. stage/sources/espresso.py +154 -0
  129. stage/sources/feed.py +44 -0
  130. stage/sources/greenhouse.py +104 -0
  131. stage/sources/jobbank.py +147 -0
  132. stage/sources/jobvite.py +133 -0
  133. stage/sources/lever.py +76 -0
  134. stage/sources/oracle_cloud.py +187 -0
  135. stage/sources/platforms.py +609 -0
  136. stage/sources/quebec_emploi.py +146 -0
  137. stage/sources/recruitee.py +96 -0
  138. stage/sources/simplify.py +110 -0
  139. stage/sources/smartrecruiters.py +216 -0
  140. stage/sources/speedyapply.py +200 -0
  141. stage/sources/themuse.py +157 -0
  142. stage/sources/workable.py +83 -0
  143. stage/sources/workday.py +524 -0
  144. stage/sources/zshah.py +99 -0
  145. stage/storage/__init__.py +29 -0
  146. stage/storage/migrations/0001_initial.sql +239 -0
  147. stage/storage/migrations/__init__.py +135 -0
  148. stage/storage/repository.py +213 -0
  149. stage/storage/search.py +28 -0
  150. stage/storage/sqlite_repo.py +1586 -0
  151. stage/storage/writer.py +249 -0
  152. stage/tui/__init__.py +0 -0
  153. stage/tui/app.py +82 -0
  154. stage/tui/help.py +26 -0
  155. stage/tui/safe.py +21 -0
  156. stage/tui/screens/__init__.py +0 -0
  157. stage/tui/screens/boards.py +186 -0
  158. stage/tui/screens/postings.py +509 -0
  159. stage/tui/screens/review.py +209 -0
  160. stage/tui/screens/splash.py +37 -0
  161. stage/tui/screens/stats.py +124 -0
  162. stage/tui/screens/sync.py +194 -0
  163. stage/tui/state.py +160 -0
  164. stage/tui/theme.tcss +205 -0
  165. stage/tui/widgets/__init__.py +0 -0
  166. stage_cli-1.0.0.dist-info/METADATA +379 -0
  167. stage_cli-1.0.0.dist-info/RECORD +170 -0
  168. stage_cli-1.0.0.dist-info/WHEEL +4 -0
  169. stage_cli-1.0.0.dist-info/entry_points.txt +2 -0
  170. stage_cli-1.0.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,271 @@
1
+ import re
2
+ from dataclasses import dataclass
3
+ from functools import lru_cache
4
+
5
+ from stage.domain import LocationBucket, RemoteScope
6
+ from stage.lexicon import LocationLexicon, fold, location_lexicon
7
+
8
+ _SEGMENT_SPLIT = re.compile(r"[;/•|\n]+")
9
+ _FIELD_SPLIT = re.compile(r"[,\-]+")
10
+ _SUBDIVISION_CODE = re.compile(r"[A-Z]{1,3}")
11
+ _COUNTRY_PREFIXES = frozenset({"can", "us", "usa"})
12
+
13
+
14
+ @dataclass(frozen=True, slots=True)
15
+ class ResolvedLocation:
16
+ bucket: LocationBucket = LocationBucket.UNKNOWN
17
+ remote_scope: RemoteScope | None = None
18
+
19
+
20
+ @dataclass(frozen=True, slots=True)
21
+ class _Segment:
22
+ montreal: bool
23
+ canada: bool
24
+ usa: bool
25
+ international: bool
26
+ remote: bool
27
+
28
+
29
+ class _PhraseIndex:
30
+ __slots__ = ("_by_first_token",)
31
+
32
+ def __init__(self, categories: dict[str, frozenset[str]]) -> None:
33
+ index: dict[str, list[tuple[str, str]]] = {}
34
+ for category, phrases in categories.items():
35
+ for phrase in phrases:
36
+ index.setdefault(phrase.split(" ", 1)[0], []).append((category, phrase))
37
+ self._by_first_token = {token: tuple(entries) for token, entries in index.items()}
38
+
39
+ def hits(self, folded: str) -> set[tuple[str, str]]:
40
+ padded = f" {folded} "
41
+ found: set[tuple[str, str]] = set()
42
+ for token in folded.split():
43
+ for entry in self._by_first_token.get(token, ()):
44
+ if f" {entry[1]} " in padded:
45
+ found.add(entry)
46
+ return found
47
+
48
+
49
+ @lru_cache(maxsize=1)
50
+ def _index() -> _PhraseIndex:
51
+ lexicon = location_lexicon()
52
+ return _PhraseIndex(
53
+ {
54
+ "montreal": lexicon.montreal,
55
+ "montreal_ambiguous": lexicon.montreal_ambiguous,
56
+ "canada_cities": lexicon.canada_cities,
57
+ "canada_ambiguous": lexicon.canada_ambiguous,
58
+ "canada_regions": lexicon.canada_regions,
59
+ "canada_country": lexicon.canada_country,
60
+ "canada_overrides": lexicon.canada_overrides,
61
+ "usa_ambiguous": lexicon.usa_ambiguous,
62
+ "usa_cities": lexicon.usa_cities,
63
+ "usa_regions": lexicon.usa_regions,
64
+ "usa_country": lexicon.usa_country,
65
+ "international": lexicon.international,
66
+ "international_cities": lexicon.international_cities,
67
+ "remote": lexicon.remote,
68
+ }
69
+ )
70
+
71
+
72
+ def _foreign_subdivision_before(fields: list[str], index: int, known: frozenset[str]) -> bool:
73
+ for previous in reversed(fields[:index]):
74
+ stripped = previous.strip()
75
+ if not stripped:
76
+ continue
77
+ if not _SUBDIVISION_CODE.fullmatch(stripped):
78
+ return False
79
+ folded = fold(stripped)
80
+ return folded not in known and folded not in _COUNTRY_PREFIXES
81
+ return False
82
+
83
+
84
+ def _last_alphabetic_field(fields: list[str]) -> int:
85
+ for index in reversed(range(len(fields))):
86
+ if any(character.isalpha() for character in fields[index]):
87
+ return index
88
+ return -1
89
+
90
+
91
+ def _reads_as_country(fields: list[str], index: int, token: str, known: frozenset[str]) -> bool:
92
+ if index != _last_alphabetic_field(fields):
93
+ return False
94
+ if fold(fields[index].strip()) != token:
95
+ return False
96
+ return _foreign_subdivision_before(fields, index, known)
97
+
98
+
99
+ def _code_hits(segment: str, lexicon: LocationLexicon) -> tuple[bool, bool]:
100
+ canada = usa = False
101
+ fields = _FIELD_SPLIT.split(segment)
102
+ known = lexicon.canada_codes | lexicon.usa_codes
103
+ for index, field in enumerate(fields):
104
+ for token in fold(field).split():
105
+ if token not in known:
106
+ continue
107
+ if not re.search(rf"\b{token.upper()}\b", segment):
108
+ continue
109
+ if _reads_as_country(fields, index, token, known):
110
+ continue
111
+ if token in lexicon.canada_codes:
112
+ canada = True
113
+ else:
114
+ usa = True
115
+ return canada, usa
116
+
117
+
118
+ _FOREIGN_ONLY_CODES = frozenset({"de", "es", "ie", "in", "it", "fr", "uk", "nl", "pl", "br", "cn"})
119
+
120
+
121
+ def _trails_a_foreign_code(segment: str) -> bool:
122
+ lexicon = location_lexicon()
123
+ fields = [field.strip() for field in _FIELD_SPLIT.split(segment) if field.strip()]
124
+ if len(fields) < 2:
125
+ return False
126
+ known = lexicon.canada_codes | lexicon.usa_codes
127
+ for field in fields[1:]:
128
+ folded = fold(field)
129
+ if folded in _FOREIGN_ONLY_CODES:
130
+ return True
131
+ if folded and len(folded) <= 3 and folded not in known and not folded.isdigit():
132
+ return True
133
+ return False
134
+
135
+
136
+ def _resolve_segment(segment: str, lexicon: LocationLexicon) -> _Segment:
137
+ folded = fold(segment)
138
+ if not folded:
139
+ return _Segment(False, False, False, False, False)
140
+ hits = _index().hits(folded)
141
+ found = {category for category, _ in hits}
142
+ international_phrases = {
143
+ phrase for category, phrase in hits if category in {"international", "international_cities"}
144
+ }
145
+ domestic_phrases = {
146
+ phrase
147
+ for category, phrase in hits
148
+ if category in {"canada_cities", "canada_regions", "usa_cities", "usa_regions"}
149
+ }
150
+ international = any(
151
+ not any(
152
+ len(domestic) > len(phrase) and f" {phrase} " in f" {domestic} "
153
+ for domestic in domestic_phrases
154
+ )
155
+ for phrase in international_phrases
156
+ )
157
+ code_canada, code_usa = _code_hits(segment, lexicon)
158
+
159
+ override_phrases = {phrase for category, phrase in hits if category == "canada_overrides"}
160
+ named_canada = {
161
+ phrase
162
+ for category, phrase in hits
163
+ if category in {"canada_cities", "canada_ambiguous", "montreal", "montreal_ambiguous"}
164
+ }
165
+ overridden = bool(override_phrases) and all(
166
+ any(f" {phrase} " in f" {override} " for override in override_phrases)
167
+ for phrase in named_canada
168
+ )
169
+ ambiguous_here = bool(found & {"canada_ambiguous", "montreal_ambiguous", "usa_ambiguous"})
170
+ coded = (
171
+ (code_canada or code_usa)
172
+ and ambiguous_here
173
+ and not _trails_a_foreign_code(segment)
174
+ and not found & {"international"}
175
+ )
176
+ canada_context = (
177
+ "canada_country" in found
178
+ or "canada_regions" in found
179
+ or (code_canada and (not international or coded))
180
+ ) and not overridden
181
+
182
+ montreal = ("montreal" in found and not overridden) or (
183
+ "montreal_ambiguous" in found and canada_context
184
+ )
185
+ canada = (
186
+ montreal
187
+ or canada_context
188
+ or ("canada_cities" in found and not overridden and not international)
189
+ or ("canada_ambiguous" in found and canada_context)
190
+ )
191
+ usa = "usa_country" in found or (
192
+ ("usa_regions" in found or "usa_cities" in found or code_usa)
193
+ and (not international or (code_usa and coded))
194
+ )
195
+ return _Segment(
196
+ montreal=montreal,
197
+ canada=canada,
198
+ usa=usa,
199
+ international=international and not (coded and (canada or usa)),
200
+ remote="remote" in found,
201
+ )
202
+
203
+
204
+ def _scope(segments: list[_Segment]) -> RemoteScope | None:
205
+ scopes = {
206
+ RemoteScope.CANADA
207
+ if segment.canada
208
+ else RemoteScope.US
209
+ if segment.usa
210
+ else RemoteScope.UNSPECIFIED
211
+ for segment in segments
212
+ if segment.remote
213
+ }
214
+ if not scopes:
215
+ return None
216
+ for candidate in (RemoteScope.CANADA, RemoteScope.UNSPECIFIED, RemoteScope.US):
217
+ if candidate in scopes:
218
+ return candidate
219
+ return RemoteScope.UNSPECIFIED
220
+
221
+
222
+ _MULTI_SITE_SUFFIX = re.compile(r"\s*\+\s*\d+\s*$")
223
+
224
+
225
+ def display_location(raw: str) -> str:
226
+ if not raw or not raw.strip():
227
+ return ""
228
+
229
+ aliases = location_lexicon().city_aliases
230
+ first = _MULTI_SITE_SUFFIX.sub("", _SEGMENT_SPLIT.split(raw)[0].strip())
231
+ if not first:
232
+ return raw.strip()
233
+
234
+ others = len([part for part in _SEGMENT_SPLIT.split(raw)[1:] if part.strip()])
235
+ more = f" +{others}" if others else ""
236
+
237
+ fields = [field.strip() for field in first.split(",") if field.strip()]
238
+ if not fields:
239
+ return raw.strip()
240
+
241
+ canonical = aliases.get(fold(fields[0]))
242
+ if canonical is None:
243
+ return f"{first}{more}"
244
+ return ", ".join([canonical, *fields[1:]]) + more
245
+
246
+
247
+ def resolve_location(raw: str) -> ResolvedLocation:
248
+ if not raw or not raw.strip():
249
+ return ResolvedLocation()
250
+
251
+ lexicon = location_lexicon()
252
+ segments = [
253
+ _resolve_segment(part, lexicon) for part in _SEGMENT_SPLIT.split(raw) if part.strip()
254
+ ]
255
+ if not segments:
256
+ return ResolvedLocation()
257
+
258
+ scope = _scope(segments)
259
+
260
+ if any(segment.montreal for segment in segments):
261
+ bucket = LocationBucket.MONTREAL
262
+ elif any(segment.canada for segment in segments):
263
+ bucket = LocationBucket.CANADA
264
+ elif any(segment.usa for segment in segments):
265
+ bucket = LocationBucket.USA
266
+ elif any(segment.international for segment in segments):
267
+ bucket = LocationBucket.INTERNATIONAL
268
+ else:
269
+ bucket = LocationBucket.UNKNOWN
270
+
271
+ return ResolvedLocation(bucket=bucket, remote_scope=scope)
@@ -0,0 +1,153 @@
1
+ import re
2
+ from collections.abc import Sequence
3
+ from dataclasses import dataclass
4
+
5
+ from stage.domain import UNKNOWN_TERM
6
+ from stage.lexicon import fold, term_lexicon
7
+
8
+ _YEAR = re.compile(r"^(20[2-3]\d)$")
9
+ _SHORT_YEAR = re.compile(r"^(\d{2})$")
10
+ _PROXIMITY = 2
11
+
12
+
13
+ @dataclass(frozen=True, slots=True)
14
+ class ResolvedTerm:
15
+ term: str = UNKNOWN_TERM
16
+ season: str = ""
17
+ year: int | None = None
18
+ evidence: tuple[str, ...] = ()
19
+ conflict: bool = False
20
+
21
+
22
+ def _year_from(token: str, pivot_year: int | None) -> int | None:
23
+ full = _YEAR.match(token)
24
+ if full:
25
+ return int(full.group(1))
26
+ short = _SHORT_YEAR.match(token)
27
+ if short and pivot_year is not None:
28
+ candidate = pivot_year - pivot_year % 100 + int(short.group(1))
29
+ if abs(candidate - pivot_year) <= 5:
30
+ return candidate
31
+ return None
32
+
33
+
34
+ def _blocked(tokens: Sequence[str], index: int, blocked: frozenset[str]) -> bool:
35
+ if index + 1 < len(tokens) and f"{tokens[index]} {tokens[index + 1]}" in blocked:
36
+ return True
37
+ return index > 0 and f"{tokens[index - 1]} {tokens[index]}" in blocked
38
+
39
+
40
+ def scan(text: str, pivot_year: int | None = None) -> list[tuple[str, int | None, str]]:
41
+ lexicon = term_lexicon()
42
+ tokens = fold(text).split()
43
+ if not tokens:
44
+ return []
45
+
46
+ surface: dict[str, str] = {}
47
+ for name, phrases in lexicon.seasons.items():
48
+ for phrase in phrases:
49
+ surface[phrase] = name
50
+
51
+ found: list[tuple[str, int | None, str]] = []
52
+ for index, token in enumerate(tokens):
53
+ matched = surface.get(token)
54
+ if matched is None or _blocked(tokens, index, lexicon.blocked_bigrams):
55
+ continue
56
+ season = matched
57
+ year: int | None = None
58
+ phrase = token
59
+ for offset in range(1, _PROXIMITY + 2):
60
+ ahead = index + offset
61
+ if ahead >= len(tokens):
62
+ break
63
+ candidate = _year_from(tokens[ahead], pivot_year)
64
+ if candidate is not None:
65
+ year, phrase = candidate, " ".join(tokens[index : ahead + 1])
66
+ break
67
+ if tokens[ahead] not in lexicon.fillers:
68
+ break
69
+ if year is None:
70
+ behind = index - 1
71
+ if behind >= 0:
72
+ candidate = _year_from(tokens[behind], pivot_year)
73
+ if candidate is not None:
74
+ year, phrase = candidate, f"{tokens[behind]} {token}"
75
+ found.append((season, year, phrase))
76
+ return found
77
+
78
+
79
+ def _terms_in(text: str, pivot_year: int | None) -> tuple[set[str], set[str], list[str]]:
80
+ terms: set[str] = set()
81
+ seasons: set[str] = set()
82
+ evidence: list[str] = []
83
+ for season, year, phrase in scan(text, pivot_year):
84
+ seasons.add(season)
85
+ evidence.append(phrase)
86
+ if year is not None:
87
+ terms.add(f"{season}-{year}")
88
+ return terms, seasons, evidence
89
+
90
+
91
+ def _structured(
92
+ values: Sequence[str], season: str, pivot_year: int | None
93
+ ) -> tuple[set[str], set[str], list[str]]:
94
+ terms: set[str] = set()
95
+ seasons: set[str] = set()
96
+ evidence: list[str] = []
97
+ for value in [*values, season]:
98
+ if not value or value.strip().lower() in {"n/a", "na", "none", "null", "unknown"}:
99
+ continue
100
+ found, found_seasons, phrases = _terms_in(value, pivot_year)
101
+ terms |= found
102
+ seasons |= found_seasons
103
+ evidence.extend(phrases)
104
+ return terms, seasons, evidence
105
+
106
+
107
+ def resolve_term(
108
+ *,
109
+ title: str = "",
110
+ description: str = "",
111
+ structured_terms: Sequence[str] = (),
112
+ structured_season: str = "",
113
+ pivot_year: int | None = None,
114
+ ) -> ResolvedTerm:
115
+ authored = (
116
+ _structured(structured_terms, structured_season, pivot_year),
117
+ _terms_in(title, pivot_year),
118
+ )
119
+ evidence: list[str] = []
120
+ seasons: set[str] = set()
121
+ speaking: list[set[str]] = []
122
+ for terms, found_seasons, phrases in authored:
123
+ evidence.extend(phrases)
124
+ seasons |= found_seasons
125
+ if terms:
126
+ speaking.append(terms)
127
+
128
+ if not speaking:
129
+ body_terms, body_seasons, body_phrases = _terms_in(description, pivot_year)
130
+ evidence.extend(body_phrases)
131
+ seasons |= body_seasons
132
+ if body_terms:
133
+ speaking.append(body_terms)
134
+
135
+ unique = tuple(dict.fromkeys(evidence))
136
+ season = next(iter(sorted(seasons))) if len(seasons) == 1 else ""
137
+
138
+ if not speaking:
139
+ return ResolvedTerm(season=season, evidence=unique)
140
+
141
+ conflict = len({frozenset(terms) for terms in speaking}) > 1
142
+ chosen = speaking[0]
143
+ if conflict or len(chosen) != 1:
144
+ return ResolvedTerm(season=season, evidence=unique, conflict=conflict)
145
+
146
+ term = next(iter(chosen))
147
+ resolved_season, _, year = term.partition("-")
148
+ return ResolvedTerm(
149
+ term=term,
150
+ season=resolved_season,
151
+ year=int(year),
152
+ evidence=unique,
153
+ )
@@ -0,0 +1,122 @@
1
+ import re
2
+ from urllib.parse import SplitResult, parse_qsl, urlsplit, urlunsplit
3
+
4
+ TRACKER_DOMAINS: frozenset[str] = frozenset(
5
+ {
6
+ "simplify.jobs",
7
+ "click.appcast.io",
8
+ "trk.simplify.jobs",
9
+ "jobright.ai",
10
+ "app.otta.com",
11
+ "get.hiring.cafe",
12
+ "grnh.se",
13
+ "boards.greenhouse.io.simplify.jobs",
14
+ "l.workwithus.io",
15
+ "track.rippling-ats.com",
16
+ }
17
+ )
18
+
19
+ LOCALE_LANGUAGES: frozenset[str] = frozenset(
20
+ {
21
+ "ar",
22
+ "cs",
23
+ "da",
24
+ "de",
25
+ "el",
26
+ "en",
27
+ "es",
28
+ "fi",
29
+ "fr",
30
+ "he",
31
+ "hu",
32
+ "ja",
33
+ "ko",
34
+ "nl",
35
+ "pl",
36
+ "pt",
37
+ "ro",
38
+ "ru",
39
+ "sv",
40
+ "th",
41
+ "tr",
42
+ "vi",
43
+ "zh",
44
+ }
45
+ )
46
+
47
+ IDENTITY_PARAMS: frozenset[str] = frozenset(
48
+ {
49
+ "gh_jid",
50
+ "id",
51
+ "jid",
52
+ "job",
53
+ "jobid",
54
+ "job_id",
55
+ "jobnumber",
56
+ "jobreqid",
57
+ "opportunityid",
58
+ "pid",
59
+ "positionid",
60
+ "postingid",
61
+ "req",
62
+ "reqid",
63
+ "req_id",
64
+ "requisitionid",
65
+ "rid",
66
+ "vacancyid",
67
+ }
68
+ )
69
+
70
+ _LOCALE_REGION = re.compile(r"^[a-z]{2}[-_][a-zA-Z]{2}$")
71
+ _IDENTITY_VALUE = re.compile(r"^[A-Za-z0-9._~-]{1,128}$")
72
+ _IDENTITY_KEYS: frozenset[str] = frozenset(name.replace("_", "") for name in IDENTITY_PARAMS)
73
+
74
+
75
+ def _is_locale_segment(segment: str) -> bool:
76
+ return bool(_LOCALE_REGION.match(segment)) or segment.lower() in LOCALE_LANGUAGES
77
+
78
+
79
+ def _split(url: str) -> SplitResult | None:
80
+ try:
81
+ return urlsplit(url)
82
+ except ValueError:
83
+ return None
84
+
85
+
86
+ def _host(url: str) -> str:
87
+ parts = _split(url)
88
+ return "" if parts is None else (parts.hostname or "").lower()
89
+
90
+
91
+ def is_tracker_url(url: str) -> bool:
92
+ host = _host(url)
93
+ if not host:
94
+ return False
95
+ return any(host == domain or host.endswith(f".{domain}") for domain in TRACKER_DOMAINS)
96
+
97
+
98
+ def _identity_query(query: str) -> str:
99
+ if not query:
100
+ return ""
101
+ kept: list[str] = []
102
+ for name, value in parse_qsl(query, keep_blank_values=False):
103
+ if name.lower().replace("-", "_").replace("_", "") not in _IDENTITY_KEYS:
104
+ continue
105
+ if _IDENTITY_VALUE.match(value):
106
+ kept.append(f"{name.lower()}={value}")
107
+ return "&".join(sorted(kept))
108
+
109
+
110
+ def canonical_apply_url(raw: str) -> str:
111
+ if not raw or not raw.strip():
112
+ return ""
113
+ parts = _split(raw.strip())
114
+ if parts is None or parts.scheme not in ("http", "https"):
115
+ return ""
116
+ host = (parts.hostname or "").lower()
117
+ if not host or is_tracker_url(raw):
118
+ return ""
119
+ segments = [segment for segment in parts.path.split("/") if segment]
120
+ kept = [segment for segment in segments if not _is_locale_segment(segment)]
121
+ path = "/" + "/".join(kept)
122
+ return urlunsplit(("https", host, path.rstrip("/") or "/", _identity_query(parts.query), ""))
stage/paths.py ADDED
@@ -0,0 +1,86 @@
1
+ import getpass
2
+ import os
3
+ import stat
4
+ import subprocess
5
+ from pathlib import Path
6
+
7
+ from platformdirs import PlatformDirs
8
+
9
+ APP_NAME = "stage"
10
+ _DIRS = PlatformDirs(appname=APP_NAME, appauthor=False, roaming=False)
11
+
12
+
13
+ def data_dir() -> Path:
14
+ path = Path(_DIRS.user_data_dir)
15
+ path.mkdir(parents=True, exist_ok=True)
16
+ return path
17
+
18
+
19
+ def config_dir() -> Path:
20
+ path = Path(_DIRS.user_config_dir)
21
+ path.mkdir(parents=True, exist_ok=True)
22
+ return path
23
+
24
+
25
+ def database_path() -> Path:
26
+ override = os.environ.get("STAGE_DB")
27
+ if override:
28
+ return Path(override).expanduser().resolve()
29
+ return data_dir() / "stage.db"
30
+
31
+
32
+ def registry_path() -> Path:
33
+ override = os.environ.get("STAGE_REGISTRY")
34
+ if override:
35
+ return Path(override).expanduser().resolve()
36
+ data = Path(__file__).resolve().parent / "data"
37
+ for packaged in (data / "companies", data / "companies.yaml"):
38
+ if packaged.exists():
39
+ return packaged
40
+ return config_dir() / "companies.yaml"
41
+
42
+
43
+ def lexicon_dir() -> Path:
44
+ override = os.environ.get("STAGE_LEXICON")
45
+ if override:
46
+ return Path(override).expanduser().resolve()
47
+ return Path(__file__).resolve().parent / "data" / "lexicon"
48
+
49
+
50
+ def font_path() -> Path:
51
+ packaged = Path(__file__).resolve().parent / "data" / "fonts" / "DejaVuSans.ttf"
52
+ if not packaged.exists():
53
+ raise FileNotFoundError(
54
+ f"the embedded PDF font is missing from {packaged.parent} — "
55
+ "export --format csv needs no font"
56
+ )
57
+ return packaged
58
+
59
+
60
+ def capture_dir() -> Path:
61
+ override = os.environ.get("STAGE_CAPTURE_DIR")
62
+ root = Path(override).expanduser() if override else data_dir() / "captured"
63
+ root.mkdir(parents=True, exist_ok=True)
64
+ return root
65
+
66
+
67
+ def restrict_permissions(path: Path) -> None:
68
+ if os.name == "nt":
69
+ _restrict_windows_permissions(path)
70
+ return
71
+ path.chmod(stat.S_IRUSR | stat.S_IWUSR)
72
+
73
+
74
+ def _restrict_windows_permissions(path: Path) -> None:
75
+ root = os.environ.get("SYSTEMROOT", r"C:\Windows").rstrip("\\/")
76
+ executable = f"{root}\\System32\\icacls.exe"
77
+ principal = f"{os.environ.get('USERDOMAIN', '.')}\\{getpass.getuser()}"
78
+ result = subprocess.run(
79
+ (executable, str(path), "/inheritance:r", "/grant:r", f"{principal}:(F)"),
80
+ check=False,
81
+ capture_output=True,
82
+ text=True,
83
+ )
84
+ if result.returncode != 0:
85
+ detail = (result.stderr or result.stdout).strip()
86
+ raise PermissionError(f"could not restrict permissions on {path}: {detail}")
stage/py.typed ADDED
File without changes
File without changes