stage-cli 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stage/__init__.py +1 -0
- stage/__main__.py +8 -0
- stage/banner.py +32 -0
- stage/bootstrap/__init__.py +0 -0
- stage/bootstrap/openjobs.py +392 -0
- stage/classify/__init__.py +29 -0
- stage/classify/eligibility.py +115 -0
- stage/classify/internship.py +64 -0
- stage/classify/role.py +91 -0
- stage/classify/scope.py +47 -0
- stage/cli/__init__.py +0 -0
- stage/cli/app.py +4 -0
- stage/cli/commands/__init__.py +8 -0
- stage/cli/commands/discovery.py +294 -0
- stage/cli/commands/insight.py +494 -0
- stage/cli/commands/pipeline.py +337 -0
- stage/cli/commands/postings.py +473 -0
- stage/cli/commands/schedule.py +171 -0
- stage/cli/housekeeping.py +64 -0
- stage/cli/logfile.py +56 -0
- stage/cli/notify.py +170 -0
- stage/cli/options.py +678 -0
- stage/cli/render.py +1398 -0
- stage/cli/runlock.py +74 -0
- stage/cli/schedule.py +702 -0
- stage/cli/schedule_state.py +363 -0
- stage/cli/selection.py +83 -0
- stage/cli/serialize.py +196 -0
- stage/companies.py +542 -0
- stage/data/companies/a.yaml +1289 -0
- stage/data/companies/b.yaml +900 -0
- stage/data/companies/c.yaml +1377 -0
- stage/data/companies/d.yaml +497 -0
- stage/data/companies/e.yaml +519 -0
- stage/data/companies/f.yaml +454 -0
- stage/data/companies/g.yaml +601 -0
- stage/data/companies/h.yaml +446 -0
- stage/data/companies/i.yaml +503 -0
- stage/data/companies/j.yaml +138 -0
- stage/data/companies/k.yaml +278 -0
- stage/data/companies/l.yaml +402 -0
- stage/data/companies/m.yaml +937 -0
- stage/data/companies/n.yaml +549 -0
- stage/data/companies/o.yaml +371 -0
- stage/data/companies/other.yaml +58 -0
- stage/data/companies/p.yaml +825 -0
- stage/data/companies/q.yaml +121 -0
- stage/data/companies/r.yaml +583 -0
- stage/data/companies/s.yaml +1140 -0
- stage/data/companies/t.yaml +817 -0
- stage/data/companies/u.yaml +196 -0
- stage/data/companies/v.yaml +325 -0
- stage/data/companies/w.yaml +353 -0
- stage/data/companies/x.yaml +67 -0
- stage/data/companies/y.yaml +36 -0
- stage/data/companies/z.yaml +146 -0
- stage/data/fonts/DejaVuSans.LICENSE.txt +99 -0
- stage/data/fonts/DejaVuSans.ttf +0 -0
- stage/data/lexicon/company_tokens.yaml +228 -0
- stage/data/lexicon/eligibility.yaml +455 -0
- stage/data/lexicon/inclusive_suffixes.yaml +37 -0
- stage/data/lexicon/internship.yaml +187 -0
- stage/data/lexicon/language.yaml +226 -0
- stage/data/lexicon/locations.yaml +1159 -0
- stage/data/lexicon/roles.yaml +2012 -0
- stage/data/lexicon/terms.yaml +76 -0
- stage/data/lexicon/workday_facets.yaml +27 -0
- stage/data/seed_companies.yaml +198 -0
- stage/dedup/__init__.py +19 -0
- stage/dedup/identity.py +113 -0
- stage/dedup/resolve.py +97 -0
- stage/domain/__init__.py +244 -0
- stage/domain/company.py +49 -0
- stage/domain/coverage.py +86 -0
- stage/domain/custom_board.py +92 -0
- stage/domain/discovery.py +94 -0
- stage/domain/enums.py +114 -0
- stage/domain/events.py +204 -0
- stage/domain/filters.py +27 -0
- stage/domain/health.py +169 -0
- stage/domain/ids.py +48 -0
- stage/domain/job.py +47 -0
- stage/domain/matching.py +15 -0
- stage/domain/priority.py +34 -0
- stage/domain/quarantine.py +39 -0
- stage/domain/rate_state.py +78 -0
- stage/domain/retention.py +20 -0
- stage/domain/rotation.py +46 -0
- stage/domain/signals.py +12 -0
- stage/domain/sync_run.py +35 -0
- stage/domain/text.py +113 -0
- stage/domain/validator.py +14 -0
- stage/domain/visits.py +60 -0
- stage/domain/workday.py +38 -0
- stage/http/__init__.py +58 -0
- stage/http/breaker.py +53 -0
- stage/http/cache.py +44 -0
- stage/http/client.py +725 -0
- stage/http/profiles.py +101 -0
- stage/lexicon.py +370 -0
- stage/normalize/__init__.py +16 -0
- stage/normalize/language.py +47 -0
- stage/normalize/location.py +271 -0
- stage/normalize/terms.py +153 -0
- stage/normalize/urls.py +122 -0
- stage/paths.py +86 -0
- stage/py.typed +0 -0
- stage/services/__init__.py +0 -0
- stage/services/canary.py +120 -0
- stage/services/coverage.py +231 -0
- stage/services/discover.py +747 -0
- stage/services/export.py +274 -0
- stage/services/health.py +237 -0
- stage/services/maintenance.py +225 -0
- stage/services/quarantine.py +20 -0
- stage/services/query.py +86 -0
- stage/services/sync.py +1257 -0
- stage/sources/__init__.py +82 -0
- stage/sources/_text.py +79 -0
- stage/sources/ashby.py +93 -0
- stage/sources/bamboohr.py +80 -0
- stage/sources/base.py +225 -0
- stage/sources/breezy.py +90 -0
- stage/sources/collage.py +60 -0
- stage/sources/community_feeds.py +142 -0
- stage/sources/curated_markdown.py +289 -0
- stage/sources/custom_json.py +610 -0
- stage/sources/espresso.py +154 -0
- stage/sources/feed.py +44 -0
- stage/sources/greenhouse.py +104 -0
- stage/sources/jobbank.py +147 -0
- stage/sources/jobvite.py +133 -0
- stage/sources/lever.py +76 -0
- stage/sources/oracle_cloud.py +187 -0
- stage/sources/platforms.py +609 -0
- stage/sources/quebec_emploi.py +146 -0
- stage/sources/recruitee.py +96 -0
- stage/sources/simplify.py +110 -0
- stage/sources/smartrecruiters.py +216 -0
- stage/sources/speedyapply.py +200 -0
- stage/sources/themuse.py +157 -0
- stage/sources/workable.py +83 -0
- stage/sources/workday.py +524 -0
- stage/sources/zshah.py +99 -0
- stage/storage/__init__.py +29 -0
- stage/storage/migrations/0001_initial.sql +239 -0
- stage/storage/migrations/__init__.py +135 -0
- stage/storage/repository.py +213 -0
- stage/storage/search.py +28 -0
- stage/storage/sqlite_repo.py +1586 -0
- stage/storage/writer.py +249 -0
- stage/tui/__init__.py +0 -0
- stage/tui/app.py +82 -0
- stage/tui/help.py +26 -0
- stage/tui/safe.py +21 -0
- stage/tui/screens/__init__.py +0 -0
- stage/tui/screens/boards.py +186 -0
- stage/tui/screens/postings.py +509 -0
- stage/tui/screens/review.py +209 -0
- stage/tui/screens/splash.py +37 -0
- stage/tui/screens/stats.py +124 -0
- stage/tui/screens/sync.py +194 -0
- stage/tui/state.py +160 -0
- stage/tui/theme.tcss +205 -0
- stage/tui/widgets/__init__.py +0 -0
- stage_cli-1.0.0.dist-info/METADATA +379 -0
- stage_cli-1.0.0.dist-info/RECORD +170 -0
- stage_cli-1.0.0.dist-info/WHEEL +4 -0
- stage_cli-1.0.0.dist-info/entry_points.txt +2 -0
- stage_cli-1.0.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,271 @@
|
|
|
1
|
+
import re
|
|
2
|
+
from dataclasses import dataclass
|
|
3
|
+
from functools import lru_cache
|
|
4
|
+
|
|
5
|
+
from stage.domain import LocationBucket, RemoteScope
|
|
6
|
+
from stage.lexicon import LocationLexicon, fold, location_lexicon
|
|
7
|
+
|
|
8
|
+
_SEGMENT_SPLIT = re.compile(r"[;/•|\n]+")
|
|
9
|
+
_FIELD_SPLIT = re.compile(r"[,\-]+")
|
|
10
|
+
_SUBDIVISION_CODE = re.compile(r"[A-Z]{1,3}")
|
|
11
|
+
_COUNTRY_PREFIXES = frozenset({"can", "us", "usa"})
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True, slots=True)
|
|
15
|
+
class ResolvedLocation:
|
|
16
|
+
bucket: LocationBucket = LocationBucket.UNKNOWN
|
|
17
|
+
remote_scope: RemoteScope | None = None
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True, slots=True)
|
|
21
|
+
class _Segment:
|
|
22
|
+
montreal: bool
|
|
23
|
+
canada: bool
|
|
24
|
+
usa: bool
|
|
25
|
+
international: bool
|
|
26
|
+
remote: bool
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class _PhraseIndex:
|
|
30
|
+
__slots__ = ("_by_first_token",)
|
|
31
|
+
|
|
32
|
+
def __init__(self, categories: dict[str, frozenset[str]]) -> None:
|
|
33
|
+
index: dict[str, list[tuple[str, str]]] = {}
|
|
34
|
+
for category, phrases in categories.items():
|
|
35
|
+
for phrase in phrases:
|
|
36
|
+
index.setdefault(phrase.split(" ", 1)[0], []).append((category, phrase))
|
|
37
|
+
self._by_first_token = {token: tuple(entries) for token, entries in index.items()}
|
|
38
|
+
|
|
39
|
+
def hits(self, folded: str) -> set[tuple[str, str]]:
|
|
40
|
+
padded = f" {folded} "
|
|
41
|
+
found: set[tuple[str, str]] = set()
|
|
42
|
+
for token in folded.split():
|
|
43
|
+
for entry in self._by_first_token.get(token, ()):
|
|
44
|
+
if f" {entry[1]} " in padded:
|
|
45
|
+
found.add(entry)
|
|
46
|
+
return found
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@lru_cache(maxsize=1)
|
|
50
|
+
def _index() -> _PhraseIndex:
|
|
51
|
+
lexicon = location_lexicon()
|
|
52
|
+
return _PhraseIndex(
|
|
53
|
+
{
|
|
54
|
+
"montreal": lexicon.montreal,
|
|
55
|
+
"montreal_ambiguous": lexicon.montreal_ambiguous,
|
|
56
|
+
"canada_cities": lexicon.canada_cities,
|
|
57
|
+
"canada_ambiguous": lexicon.canada_ambiguous,
|
|
58
|
+
"canada_regions": lexicon.canada_regions,
|
|
59
|
+
"canada_country": lexicon.canada_country,
|
|
60
|
+
"canada_overrides": lexicon.canada_overrides,
|
|
61
|
+
"usa_ambiguous": lexicon.usa_ambiguous,
|
|
62
|
+
"usa_cities": lexicon.usa_cities,
|
|
63
|
+
"usa_regions": lexicon.usa_regions,
|
|
64
|
+
"usa_country": lexicon.usa_country,
|
|
65
|
+
"international": lexicon.international,
|
|
66
|
+
"international_cities": lexicon.international_cities,
|
|
67
|
+
"remote": lexicon.remote,
|
|
68
|
+
}
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _foreign_subdivision_before(fields: list[str], index: int, known: frozenset[str]) -> bool:
|
|
73
|
+
for previous in reversed(fields[:index]):
|
|
74
|
+
stripped = previous.strip()
|
|
75
|
+
if not stripped:
|
|
76
|
+
continue
|
|
77
|
+
if not _SUBDIVISION_CODE.fullmatch(stripped):
|
|
78
|
+
return False
|
|
79
|
+
folded = fold(stripped)
|
|
80
|
+
return folded not in known and folded not in _COUNTRY_PREFIXES
|
|
81
|
+
return False
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _last_alphabetic_field(fields: list[str]) -> int:
|
|
85
|
+
for index in reversed(range(len(fields))):
|
|
86
|
+
if any(character.isalpha() for character in fields[index]):
|
|
87
|
+
return index
|
|
88
|
+
return -1
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _reads_as_country(fields: list[str], index: int, token: str, known: frozenset[str]) -> bool:
|
|
92
|
+
if index != _last_alphabetic_field(fields):
|
|
93
|
+
return False
|
|
94
|
+
if fold(fields[index].strip()) != token:
|
|
95
|
+
return False
|
|
96
|
+
return _foreign_subdivision_before(fields, index, known)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _code_hits(segment: str, lexicon: LocationLexicon) -> tuple[bool, bool]:
|
|
100
|
+
canada = usa = False
|
|
101
|
+
fields = _FIELD_SPLIT.split(segment)
|
|
102
|
+
known = lexicon.canada_codes | lexicon.usa_codes
|
|
103
|
+
for index, field in enumerate(fields):
|
|
104
|
+
for token in fold(field).split():
|
|
105
|
+
if token not in known:
|
|
106
|
+
continue
|
|
107
|
+
if not re.search(rf"\b{token.upper()}\b", segment):
|
|
108
|
+
continue
|
|
109
|
+
if _reads_as_country(fields, index, token, known):
|
|
110
|
+
continue
|
|
111
|
+
if token in lexicon.canada_codes:
|
|
112
|
+
canada = True
|
|
113
|
+
else:
|
|
114
|
+
usa = True
|
|
115
|
+
return canada, usa
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
_FOREIGN_ONLY_CODES = frozenset({"de", "es", "ie", "in", "it", "fr", "uk", "nl", "pl", "br", "cn"})
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _trails_a_foreign_code(segment: str) -> bool:
|
|
122
|
+
lexicon = location_lexicon()
|
|
123
|
+
fields = [field.strip() for field in _FIELD_SPLIT.split(segment) if field.strip()]
|
|
124
|
+
if len(fields) < 2:
|
|
125
|
+
return False
|
|
126
|
+
known = lexicon.canada_codes | lexicon.usa_codes
|
|
127
|
+
for field in fields[1:]:
|
|
128
|
+
folded = fold(field)
|
|
129
|
+
if folded in _FOREIGN_ONLY_CODES:
|
|
130
|
+
return True
|
|
131
|
+
if folded and len(folded) <= 3 and folded not in known and not folded.isdigit():
|
|
132
|
+
return True
|
|
133
|
+
return False
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _resolve_segment(segment: str, lexicon: LocationLexicon) -> _Segment:
|
|
137
|
+
folded = fold(segment)
|
|
138
|
+
if not folded:
|
|
139
|
+
return _Segment(False, False, False, False, False)
|
|
140
|
+
hits = _index().hits(folded)
|
|
141
|
+
found = {category for category, _ in hits}
|
|
142
|
+
international_phrases = {
|
|
143
|
+
phrase for category, phrase in hits if category in {"international", "international_cities"}
|
|
144
|
+
}
|
|
145
|
+
domestic_phrases = {
|
|
146
|
+
phrase
|
|
147
|
+
for category, phrase in hits
|
|
148
|
+
if category in {"canada_cities", "canada_regions", "usa_cities", "usa_regions"}
|
|
149
|
+
}
|
|
150
|
+
international = any(
|
|
151
|
+
not any(
|
|
152
|
+
len(domestic) > len(phrase) and f" {phrase} " in f" {domestic} "
|
|
153
|
+
for domestic in domestic_phrases
|
|
154
|
+
)
|
|
155
|
+
for phrase in international_phrases
|
|
156
|
+
)
|
|
157
|
+
code_canada, code_usa = _code_hits(segment, lexicon)
|
|
158
|
+
|
|
159
|
+
override_phrases = {phrase for category, phrase in hits if category == "canada_overrides"}
|
|
160
|
+
named_canada = {
|
|
161
|
+
phrase
|
|
162
|
+
for category, phrase in hits
|
|
163
|
+
if category in {"canada_cities", "canada_ambiguous", "montreal", "montreal_ambiguous"}
|
|
164
|
+
}
|
|
165
|
+
overridden = bool(override_phrases) and all(
|
|
166
|
+
any(f" {phrase} " in f" {override} " for override in override_phrases)
|
|
167
|
+
for phrase in named_canada
|
|
168
|
+
)
|
|
169
|
+
ambiguous_here = bool(found & {"canada_ambiguous", "montreal_ambiguous", "usa_ambiguous"})
|
|
170
|
+
coded = (
|
|
171
|
+
(code_canada or code_usa)
|
|
172
|
+
and ambiguous_here
|
|
173
|
+
and not _trails_a_foreign_code(segment)
|
|
174
|
+
and not found & {"international"}
|
|
175
|
+
)
|
|
176
|
+
canada_context = (
|
|
177
|
+
"canada_country" in found
|
|
178
|
+
or "canada_regions" in found
|
|
179
|
+
or (code_canada and (not international or coded))
|
|
180
|
+
) and not overridden
|
|
181
|
+
|
|
182
|
+
montreal = ("montreal" in found and not overridden) or (
|
|
183
|
+
"montreal_ambiguous" in found and canada_context
|
|
184
|
+
)
|
|
185
|
+
canada = (
|
|
186
|
+
montreal
|
|
187
|
+
or canada_context
|
|
188
|
+
or ("canada_cities" in found and not overridden and not international)
|
|
189
|
+
or ("canada_ambiguous" in found and canada_context)
|
|
190
|
+
)
|
|
191
|
+
usa = "usa_country" in found or (
|
|
192
|
+
("usa_regions" in found or "usa_cities" in found or code_usa)
|
|
193
|
+
and (not international or (code_usa and coded))
|
|
194
|
+
)
|
|
195
|
+
return _Segment(
|
|
196
|
+
montreal=montreal,
|
|
197
|
+
canada=canada,
|
|
198
|
+
usa=usa,
|
|
199
|
+
international=international and not (coded and (canada or usa)),
|
|
200
|
+
remote="remote" in found,
|
|
201
|
+
)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _scope(segments: list[_Segment]) -> RemoteScope | None:
|
|
205
|
+
scopes = {
|
|
206
|
+
RemoteScope.CANADA
|
|
207
|
+
if segment.canada
|
|
208
|
+
else RemoteScope.US
|
|
209
|
+
if segment.usa
|
|
210
|
+
else RemoteScope.UNSPECIFIED
|
|
211
|
+
for segment in segments
|
|
212
|
+
if segment.remote
|
|
213
|
+
}
|
|
214
|
+
if not scopes:
|
|
215
|
+
return None
|
|
216
|
+
for candidate in (RemoteScope.CANADA, RemoteScope.UNSPECIFIED, RemoteScope.US):
|
|
217
|
+
if candidate in scopes:
|
|
218
|
+
return candidate
|
|
219
|
+
return RemoteScope.UNSPECIFIED
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
_MULTI_SITE_SUFFIX = re.compile(r"\s*\+\s*\d+\s*$")
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def display_location(raw: str) -> str:
|
|
226
|
+
if not raw or not raw.strip():
|
|
227
|
+
return ""
|
|
228
|
+
|
|
229
|
+
aliases = location_lexicon().city_aliases
|
|
230
|
+
first = _MULTI_SITE_SUFFIX.sub("", _SEGMENT_SPLIT.split(raw)[0].strip())
|
|
231
|
+
if not first:
|
|
232
|
+
return raw.strip()
|
|
233
|
+
|
|
234
|
+
others = len([part for part in _SEGMENT_SPLIT.split(raw)[1:] if part.strip()])
|
|
235
|
+
more = f" +{others}" if others else ""
|
|
236
|
+
|
|
237
|
+
fields = [field.strip() for field in first.split(",") if field.strip()]
|
|
238
|
+
if not fields:
|
|
239
|
+
return raw.strip()
|
|
240
|
+
|
|
241
|
+
canonical = aliases.get(fold(fields[0]))
|
|
242
|
+
if canonical is None:
|
|
243
|
+
return f"{first}{more}"
|
|
244
|
+
return ", ".join([canonical, *fields[1:]]) + more
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def resolve_location(raw: str) -> ResolvedLocation:
|
|
248
|
+
if not raw or not raw.strip():
|
|
249
|
+
return ResolvedLocation()
|
|
250
|
+
|
|
251
|
+
lexicon = location_lexicon()
|
|
252
|
+
segments = [
|
|
253
|
+
_resolve_segment(part, lexicon) for part in _SEGMENT_SPLIT.split(raw) if part.strip()
|
|
254
|
+
]
|
|
255
|
+
if not segments:
|
|
256
|
+
return ResolvedLocation()
|
|
257
|
+
|
|
258
|
+
scope = _scope(segments)
|
|
259
|
+
|
|
260
|
+
if any(segment.montreal for segment in segments):
|
|
261
|
+
bucket = LocationBucket.MONTREAL
|
|
262
|
+
elif any(segment.canada for segment in segments):
|
|
263
|
+
bucket = LocationBucket.CANADA
|
|
264
|
+
elif any(segment.usa for segment in segments):
|
|
265
|
+
bucket = LocationBucket.USA
|
|
266
|
+
elif any(segment.international for segment in segments):
|
|
267
|
+
bucket = LocationBucket.INTERNATIONAL
|
|
268
|
+
else:
|
|
269
|
+
bucket = LocationBucket.UNKNOWN
|
|
270
|
+
|
|
271
|
+
return ResolvedLocation(bucket=bucket, remote_scope=scope)
|
stage/normalize/terms.py
ADDED
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
import re
|
|
2
|
+
from collections.abc import Sequence
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
|
|
5
|
+
from stage.domain import UNKNOWN_TERM
|
|
6
|
+
from stage.lexicon import fold, term_lexicon
|
|
7
|
+
|
|
8
|
+
_YEAR = re.compile(r"^(20[2-3]\d)$")
|
|
9
|
+
_SHORT_YEAR = re.compile(r"^(\d{2})$")
|
|
10
|
+
_PROXIMITY = 2
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass(frozen=True, slots=True)
|
|
14
|
+
class ResolvedTerm:
|
|
15
|
+
term: str = UNKNOWN_TERM
|
|
16
|
+
season: str = ""
|
|
17
|
+
year: int | None = None
|
|
18
|
+
evidence: tuple[str, ...] = ()
|
|
19
|
+
conflict: bool = False
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _year_from(token: str, pivot_year: int | None) -> int | None:
|
|
23
|
+
full = _YEAR.match(token)
|
|
24
|
+
if full:
|
|
25
|
+
return int(full.group(1))
|
|
26
|
+
short = _SHORT_YEAR.match(token)
|
|
27
|
+
if short and pivot_year is not None:
|
|
28
|
+
candidate = pivot_year - pivot_year % 100 + int(short.group(1))
|
|
29
|
+
if abs(candidate - pivot_year) <= 5:
|
|
30
|
+
return candidate
|
|
31
|
+
return None
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _blocked(tokens: Sequence[str], index: int, blocked: frozenset[str]) -> bool:
|
|
35
|
+
if index + 1 < len(tokens) and f"{tokens[index]} {tokens[index + 1]}" in blocked:
|
|
36
|
+
return True
|
|
37
|
+
return index > 0 and f"{tokens[index - 1]} {tokens[index]}" in blocked
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def scan(text: str, pivot_year: int | None = None) -> list[tuple[str, int | None, str]]:
|
|
41
|
+
lexicon = term_lexicon()
|
|
42
|
+
tokens = fold(text).split()
|
|
43
|
+
if not tokens:
|
|
44
|
+
return []
|
|
45
|
+
|
|
46
|
+
surface: dict[str, str] = {}
|
|
47
|
+
for name, phrases in lexicon.seasons.items():
|
|
48
|
+
for phrase in phrases:
|
|
49
|
+
surface[phrase] = name
|
|
50
|
+
|
|
51
|
+
found: list[tuple[str, int | None, str]] = []
|
|
52
|
+
for index, token in enumerate(tokens):
|
|
53
|
+
matched = surface.get(token)
|
|
54
|
+
if matched is None or _blocked(tokens, index, lexicon.blocked_bigrams):
|
|
55
|
+
continue
|
|
56
|
+
season = matched
|
|
57
|
+
year: int | None = None
|
|
58
|
+
phrase = token
|
|
59
|
+
for offset in range(1, _PROXIMITY + 2):
|
|
60
|
+
ahead = index + offset
|
|
61
|
+
if ahead >= len(tokens):
|
|
62
|
+
break
|
|
63
|
+
candidate = _year_from(tokens[ahead], pivot_year)
|
|
64
|
+
if candidate is not None:
|
|
65
|
+
year, phrase = candidate, " ".join(tokens[index : ahead + 1])
|
|
66
|
+
break
|
|
67
|
+
if tokens[ahead] not in lexicon.fillers:
|
|
68
|
+
break
|
|
69
|
+
if year is None:
|
|
70
|
+
behind = index - 1
|
|
71
|
+
if behind >= 0:
|
|
72
|
+
candidate = _year_from(tokens[behind], pivot_year)
|
|
73
|
+
if candidate is not None:
|
|
74
|
+
year, phrase = candidate, f"{tokens[behind]} {token}"
|
|
75
|
+
found.append((season, year, phrase))
|
|
76
|
+
return found
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _terms_in(text: str, pivot_year: int | None) -> tuple[set[str], set[str], list[str]]:
|
|
80
|
+
terms: set[str] = set()
|
|
81
|
+
seasons: set[str] = set()
|
|
82
|
+
evidence: list[str] = []
|
|
83
|
+
for season, year, phrase in scan(text, pivot_year):
|
|
84
|
+
seasons.add(season)
|
|
85
|
+
evidence.append(phrase)
|
|
86
|
+
if year is not None:
|
|
87
|
+
terms.add(f"{season}-{year}")
|
|
88
|
+
return terms, seasons, evidence
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _structured(
|
|
92
|
+
values: Sequence[str], season: str, pivot_year: int | None
|
|
93
|
+
) -> tuple[set[str], set[str], list[str]]:
|
|
94
|
+
terms: set[str] = set()
|
|
95
|
+
seasons: set[str] = set()
|
|
96
|
+
evidence: list[str] = []
|
|
97
|
+
for value in [*values, season]:
|
|
98
|
+
if not value or value.strip().lower() in {"n/a", "na", "none", "null", "unknown"}:
|
|
99
|
+
continue
|
|
100
|
+
found, found_seasons, phrases = _terms_in(value, pivot_year)
|
|
101
|
+
terms |= found
|
|
102
|
+
seasons |= found_seasons
|
|
103
|
+
evidence.extend(phrases)
|
|
104
|
+
return terms, seasons, evidence
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def resolve_term(
|
|
108
|
+
*,
|
|
109
|
+
title: str = "",
|
|
110
|
+
description: str = "",
|
|
111
|
+
structured_terms: Sequence[str] = (),
|
|
112
|
+
structured_season: str = "",
|
|
113
|
+
pivot_year: int | None = None,
|
|
114
|
+
) -> ResolvedTerm:
|
|
115
|
+
authored = (
|
|
116
|
+
_structured(structured_terms, structured_season, pivot_year),
|
|
117
|
+
_terms_in(title, pivot_year),
|
|
118
|
+
)
|
|
119
|
+
evidence: list[str] = []
|
|
120
|
+
seasons: set[str] = set()
|
|
121
|
+
speaking: list[set[str]] = []
|
|
122
|
+
for terms, found_seasons, phrases in authored:
|
|
123
|
+
evidence.extend(phrases)
|
|
124
|
+
seasons |= found_seasons
|
|
125
|
+
if terms:
|
|
126
|
+
speaking.append(terms)
|
|
127
|
+
|
|
128
|
+
if not speaking:
|
|
129
|
+
body_terms, body_seasons, body_phrases = _terms_in(description, pivot_year)
|
|
130
|
+
evidence.extend(body_phrases)
|
|
131
|
+
seasons |= body_seasons
|
|
132
|
+
if body_terms:
|
|
133
|
+
speaking.append(body_terms)
|
|
134
|
+
|
|
135
|
+
unique = tuple(dict.fromkeys(evidence))
|
|
136
|
+
season = next(iter(sorted(seasons))) if len(seasons) == 1 else ""
|
|
137
|
+
|
|
138
|
+
if not speaking:
|
|
139
|
+
return ResolvedTerm(season=season, evidence=unique)
|
|
140
|
+
|
|
141
|
+
conflict = len({frozenset(terms) for terms in speaking}) > 1
|
|
142
|
+
chosen = speaking[0]
|
|
143
|
+
if conflict or len(chosen) != 1:
|
|
144
|
+
return ResolvedTerm(season=season, evidence=unique, conflict=conflict)
|
|
145
|
+
|
|
146
|
+
term = next(iter(chosen))
|
|
147
|
+
resolved_season, _, year = term.partition("-")
|
|
148
|
+
return ResolvedTerm(
|
|
149
|
+
term=term,
|
|
150
|
+
season=resolved_season,
|
|
151
|
+
year=int(year),
|
|
152
|
+
evidence=unique,
|
|
153
|
+
)
|
stage/normalize/urls.py
ADDED
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
import re
|
|
2
|
+
from urllib.parse import SplitResult, parse_qsl, urlsplit, urlunsplit
|
|
3
|
+
|
|
4
|
+
TRACKER_DOMAINS: frozenset[str] = frozenset(
|
|
5
|
+
{
|
|
6
|
+
"simplify.jobs",
|
|
7
|
+
"click.appcast.io",
|
|
8
|
+
"trk.simplify.jobs",
|
|
9
|
+
"jobright.ai",
|
|
10
|
+
"app.otta.com",
|
|
11
|
+
"get.hiring.cafe",
|
|
12
|
+
"grnh.se",
|
|
13
|
+
"boards.greenhouse.io.simplify.jobs",
|
|
14
|
+
"l.workwithus.io",
|
|
15
|
+
"track.rippling-ats.com",
|
|
16
|
+
}
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
LOCALE_LANGUAGES: frozenset[str] = frozenset(
|
|
20
|
+
{
|
|
21
|
+
"ar",
|
|
22
|
+
"cs",
|
|
23
|
+
"da",
|
|
24
|
+
"de",
|
|
25
|
+
"el",
|
|
26
|
+
"en",
|
|
27
|
+
"es",
|
|
28
|
+
"fi",
|
|
29
|
+
"fr",
|
|
30
|
+
"he",
|
|
31
|
+
"hu",
|
|
32
|
+
"ja",
|
|
33
|
+
"ko",
|
|
34
|
+
"nl",
|
|
35
|
+
"pl",
|
|
36
|
+
"pt",
|
|
37
|
+
"ro",
|
|
38
|
+
"ru",
|
|
39
|
+
"sv",
|
|
40
|
+
"th",
|
|
41
|
+
"tr",
|
|
42
|
+
"vi",
|
|
43
|
+
"zh",
|
|
44
|
+
}
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
IDENTITY_PARAMS: frozenset[str] = frozenset(
|
|
48
|
+
{
|
|
49
|
+
"gh_jid",
|
|
50
|
+
"id",
|
|
51
|
+
"jid",
|
|
52
|
+
"job",
|
|
53
|
+
"jobid",
|
|
54
|
+
"job_id",
|
|
55
|
+
"jobnumber",
|
|
56
|
+
"jobreqid",
|
|
57
|
+
"opportunityid",
|
|
58
|
+
"pid",
|
|
59
|
+
"positionid",
|
|
60
|
+
"postingid",
|
|
61
|
+
"req",
|
|
62
|
+
"reqid",
|
|
63
|
+
"req_id",
|
|
64
|
+
"requisitionid",
|
|
65
|
+
"rid",
|
|
66
|
+
"vacancyid",
|
|
67
|
+
}
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
_LOCALE_REGION = re.compile(r"^[a-z]{2}[-_][a-zA-Z]{2}$")
|
|
71
|
+
_IDENTITY_VALUE = re.compile(r"^[A-Za-z0-9._~-]{1,128}$")
|
|
72
|
+
_IDENTITY_KEYS: frozenset[str] = frozenset(name.replace("_", "") for name in IDENTITY_PARAMS)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _is_locale_segment(segment: str) -> bool:
|
|
76
|
+
return bool(_LOCALE_REGION.match(segment)) or segment.lower() in LOCALE_LANGUAGES
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _split(url: str) -> SplitResult | None:
|
|
80
|
+
try:
|
|
81
|
+
return urlsplit(url)
|
|
82
|
+
except ValueError:
|
|
83
|
+
return None
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _host(url: str) -> str:
|
|
87
|
+
parts = _split(url)
|
|
88
|
+
return "" if parts is None else (parts.hostname or "").lower()
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def is_tracker_url(url: str) -> bool:
|
|
92
|
+
host = _host(url)
|
|
93
|
+
if not host:
|
|
94
|
+
return False
|
|
95
|
+
return any(host == domain or host.endswith(f".{domain}") for domain in TRACKER_DOMAINS)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _identity_query(query: str) -> str:
|
|
99
|
+
if not query:
|
|
100
|
+
return ""
|
|
101
|
+
kept: list[str] = []
|
|
102
|
+
for name, value in parse_qsl(query, keep_blank_values=False):
|
|
103
|
+
if name.lower().replace("-", "_").replace("_", "") not in _IDENTITY_KEYS:
|
|
104
|
+
continue
|
|
105
|
+
if _IDENTITY_VALUE.match(value):
|
|
106
|
+
kept.append(f"{name.lower()}={value}")
|
|
107
|
+
return "&".join(sorted(kept))
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def canonical_apply_url(raw: str) -> str:
|
|
111
|
+
if not raw or not raw.strip():
|
|
112
|
+
return ""
|
|
113
|
+
parts = _split(raw.strip())
|
|
114
|
+
if parts is None or parts.scheme not in ("http", "https"):
|
|
115
|
+
return ""
|
|
116
|
+
host = (parts.hostname or "").lower()
|
|
117
|
+
if not host or is_tracker_url(raw):
|
|
118
|
+
return ""
|
|
119
|
+
segments = [segment for segment in parts.path.split("/") if segment]
|
|
120
|
+
kept = [segment for segment in segments if not _is_locale_segment(segment)]
|
|
121
|
+
path = "/" + "/".join(kept)
|
|
122
|
+
return urlunsplit(("https", host, path.rstrip("/") or "/", _identity_query(parts.query), ""))
|
stage/paths.py
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
import getpass
|
|
2
|
+
import os
|
|
3
|
+
import stat
|
|
4
|
+
import subprocess
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from platformdirs import PlatformDirs
|
|
8
|
+
|
|
9
|
+
APP_NAME = "stage"
|
|
10
|
+
_DIRS = PlatformDirs(appname=APP_NAME, appauthor=False, roaming=False)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def data_dir() -> Path:
|
|
14
|
+
path = Path(_DIRS.user_data_dir)
|
|
15
|
+
path.mkdir(parents=True, exist_ok=True)
|
|
16
|
+
return path
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def config_dir() -> Path:
|
|
20
|
+
path = Path(_DIRS.user_config_dir)
|
|
21
|
+
path.mkdir(parents=True, exist_ok=True)
|
|
22
|
+
return path
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def database_path() -> Path:
|
|
26
|
+
override = os.environ.get("STAGE_DB")
|
|
27
|
+
if override:
|
|
28
|
+
return Path(override).expanduser().resolve()
|
|
29
|
+
return data_dir() / "stage.db"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def registry_path() -> Path:
|
|
33
|
+
override = os.environ.get("STAGE_REGISTRY")
|
|
34
|
+
if override:
|
|
35
|
+
return Path(override).expanduser().resolve()
|
|
36
|
+
data = Path(__file__).resolve().parent / "data"
|
|
37
|
+
for packaged in (data / "companies", data / "companies.yaml"):
|
|
38
|
+
if packaged.exists():
|
|
39
|
+
return packaged
|
|
40
|
+
return config_dir() / "companies.yaml"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def lexicon_dir() -> Path:
|
|
44
|
+
override = os.environ.get("STAGE_LEXICON")
|
|
45
|
+
if override:
|
|
46
|
+
return Path(override).expanduser().resolve()
|
|
47
|
+
return Path(__file__).resolve().parent / "data" / "lexicon"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def font_path() -> Path:
|
|
51
|
+
packaged = Path(__file__).resolve().parent / "data" / "fonts" / "DejaVuSans.ttf"
|
|
52
|
+
if not packaged.exists():
|
|
53
|
+
raise FileNotFoundError(
|
|
54
|
+
f"the embedded PDF font is missing from {packaged.parent} — "
|
|
55
|
+
"export --format csv needs no font"
|
|
56
|
+
)
|
|
57
|
+
return packaged
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def capture_dir() -> Path:
|
|
61
|
+
override = os.environ.get("STAGE_CAPTURE_DIR")
|
|
62
|
+
root = Path(override).expanduser() if override else data_dir() / "captured"
|
|
63
|
+
root.mkdir(parents=True, exist_ok=True)
|
|
64
|
+
return root
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def restrict_permissions(path: Path) -> None:
|
|
68
|
+
if os.name == "nt":
|
|
69
|
+
_restrict_windows_permissions(path)
|
|
70
|
+
return
|
|
71
|
+
path.chmod(stat.S_IRUSR | stat.S_IWUSR)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _restrict_windows_permissions(path: Path) -> None:
|
|
75
|
+
root = os.environ.get("SYSTEMROOT", r"C:\Windows").rstrip("\\/")
|
|
76
|
+
executable = f"{root}\\System32\\icacls.exe"
|
|
77
|
+
principal = f"{os.environ.get('USERDOMAIN', '.')}\\{getpass.getuser()}"
|
|
78
|
+
result = subprocess.run(
|
|
79
|
+
(executable, str(path), "/inheritance:r", "/grant:r", f"{principal}:(F)"),
|
|
80
|
+
check=False,
|
|
81
|
+
capture_output=True,
|
|
82
|
+
text=True,
|
|
83
|
+
)
|
|
84
|
+
if result.returncode != 0:
|
|
85
|
+
detail = (result.stderr or result.stdout).strip()
|
|
86
|
+
raise PermissionError(f"could not restrict permissions on {path}: {detail}")
|
stage/py.typed
ADDED
|
File without changes
|
|
File without changes
|