stage-cli 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stage/__init__.py +1 -0
- stage/__main__.py +8 -0
- stage/banner.py +32 -0
- stage/bootstrap/__init__.py +0 -0
- stage/bootstrap/openjobs.py +392 -0
- stage/classify/__init__.py +29 -0
- stage/classify/eligibility.py +115 -0
- stage/classify/internship.py +64 -0
- stage/classify/role.py +91 -0
- stage/classify/scope.py +47 -0
- stage/cli/__init__.py +0 -0
- stage/cli/app.py +4 -0
- stage/cli/commands/__init__.py +8 -0
- stage/cli/commands/discovery.py +294 -0
- stage/cli/commands/insight.py +494 -0
- stage/cli/commands/pipeline.py +337 -0
- stage/cli/commands/postings.py +473 -0
- stage/cli/commands/schedule.py +171 -0
- stage/cli/housekeeping.py +64 -0
- stage/cli/logfile.py +56 -0
- stage/cli/notify.py +170 -0
- stage/cli/options.py +678 -0
- stage/cli/render.py +1398 -0
- stage/cli/runlock.py +74 -0
- stage/cli/schedule.py +702 -0
- stage/cli/schedule_state.py +363 -0
- stage/cli/selection.py +83 -0
- stage/cli/serialize.py +196 -0
- stage/companies.py +542 -0
- stage/data/companies/a.yaml +1289 -0
- stage/data/companies/b.yaml +900 -0
- stage/data/companies/c.yaml +1377 -0
- stage/data/companies/d.yaml +497 -0
- stage/data/companies/e.yaml +519 -0
- stage/data/companies/f.yaml +454 -0
- stage/data/companies/g.yaml +601 -0
- stage/data/companies/h.yaml +446 -0
- stage/data/companies/i.yaml +503 -0
- stage/data/companies/j.yaml +138 -0
- stage/data/companies/k.yaml +278 -0
- stage/data/companies/l.yaml +402 -0
- stage/data/companies/m.yaml +937 -0
- stage/data/companies/n.yaml +549 -0
- stage/data/companies/o.yaml +371 -0
- stage/data/companies/other.yaml +58 -0
- stage/data/companies/p.yaml +825 -0
- stage/data/companies/q.yaml +121 -0
- stage/data/companies/r.yaml +583 -0
- stage/data/companies/s.yaml +1140 -0
- stage/data/companies/t.yaml +817 -0
- stage/data/companies/u.yaml +196 -0
- stage/data/companies/v.yaml +325 -0
- stage/data/companies/w.yaml +353 -0
- stage/data/companies/x.yaml +67 -0
- stage/data/companies/y.yaml +36 -0
- stage/data/companies/z.yaml +146 -0
- stage/data/fonts/DejaVuSans.LICENSE.txt +99 -0
- stage/data/fonts/DejaVuSans.ttf +0 -0
- stage/data/lexicon/company_tokens.yaml +228 -0
- stage/data/lexicon/eligibility.yaml +455 -0
- stage/data/lexicon/inclusive_suffixes.yaml +37 -0
- stage/data/lexicon/internship.yaml +187 -0
- stage/data/lexicon/language.yaml +226 -0
- stage/data/lexicon/locations.yaml +1159 -0
- stage/data/lexicon/roles.yaml +2012 -0
- stage/data/lexicon/terms.yaml +76 -0
- stage/data/lexicon/workday_facets.yaml +27 -0
- stage/data/seed_companies.yaml +198 -0
- stage/dedup/__init__.py +19 -0
- stage/dedup/identity.py +113 -0
- stage/dedup/resolve.py +97 -0
- stage/domain/__init__.py +244 -0
- stage/domain/company.py +49 -0
- stage/domain/coverage.py +86 -0
- stage/domain/custom_board.py +92 -0
- stage/domain/discovery.py +94 -0
- stage/domain/enums.py +114 -0
- stage/domain/events.py +204 -0
- stage/domain/filters.py +27 -0
- stage/domain/health.py +169 -0
- stage/domain/ids.py +48 -0
- stage/domain/job.py +47 -0
- stage/domain/matching.py +15 -0
- stage/domain/priority.py +34 -0
- stage/domain/quarantine.py +39 -0
- stage/domain/rate_state.py +78 -0
- stage/domain/retention.py +20 -0
- stage/domain/rotation.py +46 -0
- stage/domain/signals.py +12 -0
- stage/domain/sync_run.py +35 -0
- stage/domain/text.py +113 -0
- stage/domain/validator.py +14 -0
- stage/domain/visits.py +60 -0
- stage/domain/workday.py +38 -0
- stage/http/__init__.py +58 -0
- stage/http/breaker.py +53 -0
- stage/http/cache.py +44 -0
- stage/http/client.py +725 -0
- stage/http/profiles.py +101 -0
- stage/lexicon.py +370 -0
- stage/normalize/__init__.py +16 -0
- stage/normalize/language.py +47 -0
- stage/normalize/location.py +271 -0
- stage/normalize/terms.py +153 -0
- stage/normalize/urls.py +122 -0
- stage/paths.py +86 -0
- stage/py.typed +0 -0
- stage/services/__init__.py +0 -0
- stage/services/canary.py +120 -0
- stage/services/coverage.py +231 -0
- stage/services/discover.py +747 -0
- stage/services/export.py +274 -0
- stage/services/health.py +237 -0
- stage/services/maintenance.py +225 -0
- stage/services/quarantine.py +20 -0
- stage/services/query.py +86 -0
- stage/services/sync.py +1257 -0
- stage/sources/__init__.py +82 -0
- stage/sources/_text.py +79 -0
- stage/sources/ashby.py +93 -0
- stage/sources/bamboohr.py +80 -0
- stage/sources/base.py +225 -0
- stage/sources/breezy.py +90 -0
- stage/sources/collage.py +60 -0
- stage/sources/community_feeds.py +142 -0
- stage/sources/curated_markdown.py +289 -0
- stage/sources/custom_json.py +610 -0
- stage/sources/espresso.py +154 -0
- stage/sources/feed.py +44 -0
- stage/sources/greenhouse.py +104 -0
- stage/sources/jobbank.py +147 -0
- stage/sources/jobvite.py +133 -0
- stage/sources/lever.py +76 -0
- stage/sources/oracle_cloud.py +187 -0
- stage/sources/platforms.py +609 -0
- stage/sources/quebec_emploi.py +146 -0
- stage/sources/recruitee.py +96 -0
- stage/sources/simplify.py +110 -0
- stage/sources/smartrecruiters.py +216 -0
- stage/sources/speedyapply.py +200 -0
- stage/sources/themuse.py +157 -0
- stage/sources/workable.py +83 -0
- stage/sources/workday.py +524 -0
- stage/sources/zshah.py +99 -0
- stage/storage/__init__.py +29 -0
- stage/storage/migrations/0001_initial.sql +239 -0
- stage/storage/migrations/__init__.py +135 -0
- stage/storage/repository.py +213 -0
- stage/storage/search.py +28 -0
- stage/storage/sqlite_repo.py +1586 -0
- stage/storage/writer.py +249 -0
- stage/tui/__init__.py +0 -0
- stage/tui/app.py +82 -0
- stage/tui/help.py +26 -0
- stage/tui/safe.py +21 -0
- stage/tui/screens/__init__.py +0 -0
- stage/tui/screens/boards.py +186 -0
- stage/tui/screens/postings.py +509 -0
- stage/tui/screens/review.py +209 -0
- stage/tui/screens/splash.py +37 -0
- stage/tui/screens/stats.py +124 -0
- stage/tui/screens/sync.py +194 -0
- stage/tui/state.py +160 -0
- stage/tui/theme.tcss +205 -0
- stage/tui/widgets/__init__.py +0 -0
- stage_cli-1.0.0.dist-info/METADATA +379 -0
- stage_cli-1.0.0.dist-info/RECORD +170 -0
- stage_cli-1.0.0.dist-info/WHEEL +4 -0
- stage_cli-1.0.0.dist-info/entry_points.txt +2 -0
- stage_cli-1.0.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
CREATE TABLE jobs (
|
|
2
|
+
id TEXT PRIMARY KEY,
|
|
3
|
+
source TEXT NOT NULL,
|
|
4
|
+
company TEXT NOT NULL,
|
|
5
|
+
company_fold TEXT NOT NULL DEFAULT '',
|
|
6
|
+
title_raw TEXT NOT NULL,
|
|
7
|
+
title_normalized TEXT NOT NULL,
|
|
8
|
+
title_canonical TEXT NOT NULL DEFAULT '',
|
|
9
|
+
apply_url_raw TEXT NOT NULL,
|
|
10
|
+
apply_url_canonical TEXT NOT NULL DEFAULT '',
|
|
11
|
+
description TEXT NOT NULL DEFAULT '',
|
|
12
|
+
location_raw TEXT NOT NULL DEFAULT '',
|
|
13
|
+
location TEXT NOT NULL DEFAULT 'unknown',
|
|
14
|
+
remote_scope TEXT,
|
|
15
|
+
language TEXT NOT NULL DEFAULT 'unknown',
|
|
16
|
+
term TEXT NOT NULL DEFAULT 'unknown',
|
|
17
|
+
role TEXT NOT NULL DEFAULT 'unknown',
|
|
18
|
+
degree_requirement TEXT NOT NULL DEFAULT 'unknown',
|
|
19
|
+
work_auth_flag INTEGER NOT NULL DEFAULT 0,
|
|
20
|
+
compensation TEXT,
|
|
21
|
+
employment_type TEXT NOT NULL DEFAULT '',
|
|
22
|
+
source_category TEXT NOT NULL DEFAULT '',
|
|
23
|
+
status TEXT NOT NULL DEFAULT 'open',
|
|
24
|
+
duplicate_of TEXT,
|
|
25
|
+
first_seen TEXT NOT NULL,
|
|
26
|
+
last_seen TEXT NOT NULL,
|
|
27
|
+
source_posted_at TEXT
|
|
28
|
+
);
|
|
29
|
+
|
|
30
|
+
CREATE INDEX idx_jobs_first_seen ON jobs (first_seen DESC);
|
|
31
|
+
CREATE INDEX idx_jobs_status_first_seen ON jobs (status, first_seen DESC);
|
|
32
|
+
CREATE INDEX idx_jobs_source_company ON jobs (source, company);
|
|
33
|
+
CREATE INDEX idx_jobs_duplicate_of ON jobs (duplicate_of);
|
|
34
|
+
CREATE INDEX idx_jobs_degree ON jobs (degree_requirement);
|
|
35
|
+
CREATE INDEX idx_jobs_company_canonical ON jobs (company) WHERE duplicate_of IS NULL;
|
|
36
|
+
CREATE INDEX idx_jobs_company_fold ON jobs (company_fold);
|
|
37
|
+
CREATE INDEX idx_jobs_apply_url_canonical ON jobs (apply_url_canonical);
|
|
38
|
+
|
|
39
|
+
CREATE VIRTUAL TABLE jobs_fts USING fts5 (
|
|
40
|
+
company,
|
|
41
|
+
title_raw,
|
|
42
|
+
title_normalized,
|
|
43
|
+
title_canonical,
|
|
44
|
+
location_raw,
|
|
45
|
+
description,
|
|
46
|
+
content = 'jobs',
|
|
47
|
+
content_rowid = 'rowid',
|
|
48
|
+
tokenize = 'unicode61 remove_diacritics 2'
|
|
49
|
+
);
|
|
50
|
+
|
|
51
|
+
CREATE TRIGGER jobs_fts_after_insert AFTER INSERT ON jobs BEGIN
|
|
52
|
+
INSERT INTO jobs_fts (
|
|
53
|
+
rowid, company, title_raw, title_normalized, title_canonical,
|
|
54
|
+
location_raw, description
|
|
55
|
+
)
|
|
56
|
+
VALUES (
|
|
57
|
+
new.rowid, new.company, new.title_raw, new.title_normalized,
|
|
58
|
+
new.title_canonical, new.location_raw, new.description
|
|
59
|
+
);
|
|
60
|
+
END;
|
|
61
|
+
|
|
62
|
+
CREATE TRIGGER jobs_fts_after_delete AFTER DELETE ON jobs BEGIN
|
|
63
|
+
INSERT INTO jobs_fts (
|
|
64
|
+
jobs_fts, rowid, company, title_raw, title_normalized, title_canonical,
|
|
65
|
+
location_raw, description
|
|
66
|
+
)
|
|
67
|
+
VALUES (
|
|
68
|
+
'delete', old.rowid, old.company, old.title_raw, old.title_normalized,
|
|
69
|
+
old.title_canonical, old.location_raw, old.description
|
|
70
|
+
);
|
|
71
|
+
END;
|
|
72
|
+
|
|
73
|
+
CREATE TRIGGER jobs_fts_after_update AFTER UPDATE ON jobs
|
|
74
|
+
WHEN new.company IS NOT old.company
|
|
75
|
+
OR new.title_raw IS NOT old.title_raw
|
|
76
|
+
OR new.title_normalized IS NOT old.title_normalized
|
|
77
|
+
OR new.title_canonical IS NOT old.title_canonical
|
|
78
|
+
OR new.location_raw IS NOT old.location_raw
|
|
79
|
+
OR new.description IS NOT old.description
|
|
80
|
+
BEGIN
|
|
81
|
+
INSERT INTO jobs_fts (
|
|
82
|
+
jobs_fts, rowid, company, title_raw, title_normalized, title_canonical,
|
|
83
|
+
location_raw, description
|
|
84
|
+
)
|
|
85
|
+
VALUES (
|
|
86
|
+
'delete', old.rowid, old.company, old.title_raw, old.title_normalized,
|
|
87
|
+
old.title_canonical, old.location_raw, old.description
|
|
88
|
+
);
|
|
89
|
+
INSERT INTO jobs_fts (
|
|
90
|
+
rowid, company, title_raw, title_normalized, title_canonical,
|
|
91
|
+
location_raw, description
|
|
92
|
+
)
|
|
93
|
+
VALUES (
|
|
94
|
+
new.rowid, new.company, new.title_raw, new.title_normalized,
|
|
95
|
+
new.title_canonical, new.location_raw, new.description
|
|
96
|
+
);
|
|
97
|
+
END;
|
|
98
|
+
|
|
99
|
+
CREATE TABLE quarantine (
|
|
100
|
+
id TEXT PRIMARY KEY,
|
|
101
|
+
source TEXT NOT NULL,
|
|
102
|
+
company TEXT NOT NULL,
|
|
103
|
+
title_raw TEXT NOT NULL,
|
|
104
|
+
apply_url_raw TEXT NOT NULL DEFAULT '',
|
|
105
|
+
location_raw TEXT NOT NULL DEFAULT '',
|
|
106
|
+
location TEXT NOT NULL DEFAULT 'unknown',
|
|
107
|
+
remote_scope TEXT,
|
|
108
|
+
reason TEXT NOT NULL,
|
|
109
|
+
matched_phrase TEXT NOT NULL DEFAULT '',
|
|
110
|
+
first_seen TEXT NOT NULL,
|
|
111
|
+
last_seen TEXT NOT NULL
|
|
112
|
+
);
|
|
113
|
+
|
|
114
|
+
CREATE INDEX idx_quarantine_first_seen ON quarantine (first_seen DESC);
|
|
115
|
+
CREATE INDEX idx_quarantine_reason_first_seen ON quarantine (reason, first_seen DESC);
|
|
116
|
+
CREATE INDEX idx_quarantine_company ON quarantine (company, last_seen DESC);
|
|
117
|
+
CREATE INDEX idx_quarantine_source_company ON quarantine (source, company);
|
|
118
|
+
|
|
119
|
+
CREATE TABLE tombstones (
|
|
120
|
+
id TEXT PRIMARY KEY,
|
|
121
|
+
source TEXT NOT NULL,
|
|
122
|
+
first_seen TEXT NOT NULL,
|
|
123
|
+
purged_at TEXT NOT NULL
|
|
124
|
+
);
|
|
125
|
+
|
|
126
|
+
CREATE INDEX idx_tombstones_purged_at ON tombstones (purged_at DESC);
|
|
127
|
+
|
|
128
|
+
CREATE TABLE http_cache (
|
|
129
|
+
url TEXT PRIMARY KEY,
|
|
130
|
+
source TEXT NOT NULL,
|
|
131
|
+
etag TEXT,
|
|
132
|
+
last_modified TEXT,
|
|
133
|
+
fetched_at TEXT NOT NULL
|
|
134
|
+
);
|
|
135
|
+
|
|
136
|
+
CREATE INDEX idx_http_cache_source ON http_cache (source);
|
|
137
|
+
|
|
138
|
+
CREATE TABLE rate_state (
|
|
139
|
+
bucket TEXT PRIMARY KEY,
|
|
140
|
+
blocked_until TEXT,
|
|
141
|
+
min_interval_override REAL,
|
|
142
|
+
consecutive_failures INTEGER NOT NULL DEFAULT 0,
|
|
143
|
+
last_failure_at TEXT,
|
|
144
|
+
reason TEXT NOT NULL DEFAULT '',
|
|
145
|
+
rotation_cursor TEXT NOT NULL DEFAULT '',
|
|
146
|
+
updated_at TEXT NOT NULL
|
|
147
|
+
);
|
|
148
|
+
|
|
149
|
+
CREATE TABLE source_visits (
|
|
150
|
+
source TEXT NOT NULL,
|
|
151
|
+
board TEXT NOT NULL,
|
|
152
|
+
label TEXT NOT NULL DEFAULT '',
|
|
153
|
+
last_attempt_at TEXT NOT NULL,
|
|
154
|
+
last_success_at TEXT,
|
|
155
|
+
consecutive_failures INTEGER NOT NULL DEFAULT 0,
|
|
156
|
+
last_error TEXT NOT NULL DEFAULT '',
|
|
157
|
+
PRIMARY KEY (source, board)
|
|
158
|
+
);
|
|
159
|
+
|
|
160
|
+
CREATE INDEX idx_source_visits_success ON source_visits (source, last_success_at);
|
|
161
|
+
|
|
162
|
+
CREATE TABLE coverage_classifications (
|
|
163
|
+
company TEXT NOT NULL CHECK (length(trim(company)) > 0),
|
|
164
|
+
company_fold TEXT PRIMARY KEY,
|
|
165
|
+
disposition TEXT NOT NULL CHECK (disposition IN (
|
|
166
|
+
'feed-only', 'unavailable', 'custom-json-candidate', 'adapter-candidate', 'deferred'
|
|
167
|
+
)),
|
|
168
|
+
note TEXT NOT NULL CHECK (length(trim(note)) > 0),
|
|
169
|
+
checked_on TEXT NOT NULL,
|
|
170
|
+
url TEXT
|
|
171
|
+
);
|
|
172
|
+
|
|
173
|
+
CREATE TABLE workday_facets (
|
|
174
|
+
tenant TEXT NOT NULL,
|
|
175
|
+
site TEXT NOT NULL,
|
|
176
|
+
parameter TEXT NOT NULL,
|
|
177
|
+
facet_id TEXT NOT NULL,
|
|
178
|
+
descriptor TEXT NOT NULL DEFAULT '',
|
|
179
|
+
resolved_at TEXT NOT NULL,
|
|
180
|
+
PRIMARY KEY (tenant, site)
|
|
181
|
+
);
|
|
182
|
+
|
|
183
|
+
CREATE TABLE workday_crawls (
|
|
184
|
+
board TEXT PRIMARY KEY,
|
|
185
|
+
next_offset INTEGER NOT NULL CHECK (next_offset >= 0),
|
|
186
|
+
total INTEGER,
|
|
187
|
+
facet_parameter TEXT NOT NULL DEFAULT '',
|
|
188
|
+
facet_ids TEXT NOT NULL DEFAULT ''
|
|
189
|
+
);
|
|
190
|
+
|
|
191
|
+
CREATE TABLE workday_crawl_seen (
|
|
192
|
+
board TEXT NOT NULL REFERENCES workday_crawls (board) ON DELETE CASCADE,
|
|
193
|
+
id TEXT NOT NULL,
|
|
194
|
+
PRIMARY KEY (board, id)
|
|
195
|
+
);
|
|
196
|
+
|
|
197
|
+
CREATE INDEX idx_workday_crawl_seen_id ON workday_crawl_seen (id);
|
|
198
|
+
|
|
199
|
+
CREATE TABLE detail_fetches (
|
|
200
|
+
id TEXT PRIMARY KEY,
|
|
201
|
+
source TEXT NOT NULL,
|
|
202
|
+
fetched_at TEXT NOT NULL,
|
|
203
|
+
resolved INTEGER NOT NULL DEFAULT 0,
|
|
204
|
+
attempts INTEGER NOT NULL DEFAULT 1,
|
|
205
|
+
failed INTEGER NOT NULL DEFAULT 0
|
|
206
|
+
);
|
|
207
|
+
|
|
208
|
+
CREATE INDEX idx_detail_fetches_source ON detail_fetches (source, resolved, failed);
|
|
209
|
+
|
|
210
|
+
CREATE TABLE sync_runs (
|
|
211
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
212
|
+
started_at TEXT NOT NULL,
|
|
213
|
+
finished_at TEXT NOT NULL,
|
|
214
|
+
outcome TEXT NOT NULL
|
|
215
|
+
);
|
|
216
|
+
|
|
217
|
+
CREATE INDEX idx_sync_runs_started_at ON sync_runs (started_at DESC);
|
|
218
|
+
|
|
219
|
+
CREATE TABLE sync_run_sources (
|
|
220
|
+
run_id INTEGER NOT NULL REFERENCES sync_runs (id) ON DELETE CASCADE,
|
|
221
|
+
source TEXT NOT NULL,
|
|
222
|
+
fetched INTEGER NOT NULL DEFAULT 0,
|
|
223
|
+
added INTEGER NOT NULL DEFAULT 0,
|
|
224
|
+
updated INTEGER NOT NULL DEFAULT 0,
|
|
225
|
+
closed INTEGER NOT NULL DEFAULT 0,
|
|
226
|
+
errors INTEGER NOT NULL DEFAULT 0,
|
|
227
|
+
requests INTEGER NOT NULL DEFAULT 0,
|
|
228
|
+
not_modified INTEGER NOT NULL DEFAULT 0,
|
|
229
|
+
retries INTEGER NOT NULL DEFAULT 0,
|
|
230
|
+
tightenings INTEGER NOT NULL DEFAULT 0,
|
|
231
|
+
quarantined INTEGER NOT NULL DEFAULT 0,
|
|
232
|
+
deferred INTEGER NOT NULL DEFAULT 0,
|
|
233
|
+
blocked INTEGER NOT NULL DEFAULT 0,
|
|
234
|
+
stored INTEGER NOT NULL DEFAULT -1,
|
|
235
|
+
latency_p50_ms REAL NOT NULL DEFAULT 0,
|
|
236
|
+
latency_p95_ms REAL NOT NULL DEFAULT 0,
|
|
237
|
+
elapsed_ms REAL NOT NULL DEFAULT 0,
|
|
238
|
+
PRIMARY KEY (run_id, source)
|
|
239
|
+
);
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
import re
|
|
2
|
+
import sqlite3
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from datetime import UTC, datetime
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
_FILENAME = re.compile(r"^(\d{4})_[a-z0-9_]+\.sql$")
|
|
8
|
+
_MIGRATIONS_DIR = Path(__file__).resolve().parent
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class SchemaVersionError(Exception):
|
|
12
|
+
pass
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True, slots=True)
|
|
16
|
+
class Migration:
|
|
17
|
+
version: int
|
|
18
|
+
name: str
|
|
19
|
+
path: Path
|
|
20
|
+
|
|
21
|
+
def read(self) -> str:
|
|
22
|
+
return self.path.read_text(encoding="utf-8")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def discover() -> tuple[Migration, ...]:
|
|
26
|
+
found: list[Migration] = []
|
|
27
|
+
for path in sorted(_MIGRATIONS_DIR.glob("*.sql")):
|
|
28
|
+
match = _FILENAME.match(path.name)
|
|
29
|
+
if match is None:
|
|
30
|
+
raise SchemaVersionError(f"migration {path.name!r} does not follow NNNN_name.sql")
|
|
31
|
+
found.append(Migration(version=int(match.group(1)), name=path.stem, path=path))
|
|
32
|
+
versions = [migration.version for migration in found]
|
|
33
|
+
if len(set(versions)) != len(versions):
|
|
34
|
+
raise SchemaVersionError("duplicate migration version numbers")
|
|
35
|
+
return tuple(found)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def latest_version() -> int:
|
|
39
|
+
migrations = discover()
|
|
40
|
+
return migrations[-1].version if migrations else 0
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _ensure_migrations_table(conn: sqlite3.Connection) -> None:
|
|
44
|
+
conn.execute(
|
|
45
|
+
"CREATE TABLE IF NOT EXISTS schema_migrations ("
|
|
46
|
+
"version INTEGER PRIMARY KEY, name TEXT NOT NULL, applied_at TEXT NOT NULL)"
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def applied_versions(conn: sqlite3.Connection) -> tuple[int, ...]:
|
|
51
|
+
_ensure_migrations_table(conn)
|
|
52
|
+
rows = conn.execute("SELECT version FROM schema_migrations ORDER BY version").fetchall()
|
|
53
|
+
return tuple(int(row[0]) for row in rows)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
BASELINE_COLUMNS = frozenset({"company_fold", "employment_type", "source_category"})
|
|
57
|
+
BASELINE_TABLES = frozenset({"coverage_classifications", "workday_crawls", "workday_crawl_seen"})
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _refuse_a_pre_baseline_database(conn: sqlite3.Connection, db_path: Path) -> None:
|
|
61
|
+
columns = {str(row[1]) for row in conn.execute("PRAGMA table_info(jobs)").fetchall()}
|
|
62
|
+
tables = {
|
|
63
|
+
str(row[0])
|
|
64
|
+
for row in conn.execute("SELECT name FROM sqlite_master WHERE type = 'table'").fetchall()
|
|
65
|
+
}
|
|
66
|
+
missing = sorted(BASELINE_COLUMNS - columns) + sorted(BASELINE_TABLES - tables)
|
|
67
|
+
if not missing:
|
|
68
|
+
return
|
|
69
|
+
raise SchemaVersionError(
|
|
70
|
+
f"database at {db_path} records the current schema version but is missing "
|
|
71
|
+
f"{', '.join(missing)} — it predates this baseline. Back it up or delete it, then run "
|
|
72
|
+
"stage sync; the corpus is reproducible in under a minute"
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def snapshot_path(db_path: Path, when: datetime) -> Path:
|
|
77
|
+
return db_path.with_name(f"{db_path.name}.bak-{when.strftime('%Y%m%dT%H%M%S')}")
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _snapshot(conn: sqlite3.Connection, db_path: Path) -> Path:
|
|
81
|
+
target = snapshot_path(db_path, datetime.now(UTC))
|
|
82
|
+
backup = sqlite3.connect(target)
|
|
83
|
+
try:
|
|
84
|
+
conn.backup(backup)
|
|
85
|
+
finally:
|
|
86
|
+
backup.close()
|
|
87
|
+
return target
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def migrate(conn: sqlite3.Connection, db_path: Path) -> tuple[int, ...]:
|
|
91
|
+
migrations = discover()
|
|
92
|
+
known = {migration.version for migration in migrations}
|
|
93
|
+
applied = applied_versions(conn)
|
|
94
|
+
|
|
95
|
+
unknown = [version for version in applied if version not in known]
|
|
96
|
+
if unknown:
|
|
97
|
+
raise SchemaVersionError(
|
|
98
|
+
f"database at {db_path} has schema version {max(unknown)}, newer than this build "
|
|
99
|
+
f"understands ({latest_version()}) — upgrade stage-cli"
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
pending = [migration for migration in migrations if migration.version not in applied]
|
|
103
|
+
if not pending:
|
|
104
|
+
_refuse_a_pre_baseline_database(conn, db_path)
|
|
105
|
+
return ()
|
|
106
|
+
|
|
107
|
+
if applied:
|
|
108
|
+
_snapshot(conn, db_path)
|
|
109
|
+
|
|
110
|
+
for migration in pending:
|
|
111
|
+
try:
|
|
112
|
+
conn.executescript(f"BEGIN IMMEDIATE;\n{migration.read()}")
|
|
113
|
+
conn.execute(
|
|
114
|
+
"INSERT INTO schema_migrations (version, name, applied_at) VALUES (?, ?, ?)",
|
|
115
|
+
(migration.version, migration.name, datetime.now(UTC).isoformat()),
|
|
116
|
+
)
|
|
117
|
+
except Exception:
|
|
118
|
+
conn.rollback()
|
|
119
|
+
raise
|
|
120
|
+
conn.commit()
|
|
121
|
+
|
|
122
|
+
return tuple(migration.version for migration in pending)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
__all__ = [
|
|
126
|
+
"BASELINE_COLUMNS",
|
|
127
|
+
"BASELINE_TABLES",
|
|
128
|
+
"Migration",
|
|
129
|
+
"SchemaVersionError",
|
|
130
|
+
"applied_versions",
|
|
131
|
+
"discover",
|
|
132
|
+
"latest_version",
|
|
133
|
+
"migrate",
|
|
134
|
+
"snapshot_path",
|
|
135
|
+
]
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
from collections.abc import Callable, Mapping, Sequence
|
|
2
|
+
from dataclasses import dataclass, field
|
|
3
|
+
from datetime import datetime
|
|
4
|
+
from typing import Protocol
|
|
5
|
+
|
|
6
|
+
from stage.domain import (
|
|
7
|
+
CompanyVisit,
|
|
8
|
+
CoverageClassification,
|
|
9
|
+
DetailFetch,
|
|
10
|
+
HttpValidator,
|
|
11
|
+
IntegrityFinding,
|
|
12
|
+
IntegrityRepair,
|
|
13
|
+
Job,
|
|
14
|
+
JobFilters,
|
|
15
|
+
PurgeResult,
|
|
16
|
+
QuarantinedJob,
|
|
17
|
+
QuarantineFilters,
|
|
18
|
+
RateState,
|
|
19
|
+
SourceVisit,
|
|
20
|
+
SyncRun,
|
|
21
|
+
VolumePoint,
|
|
22
|
+
WorkdayCrawl,
|
|
23
|
+
WorkdayCrawlStep,
|
|
24
|
+
WorkdayFacet,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass(frozen=True, slots=True)
|
|
29
|
+
class SourceBatch:
|
|
30
|
+
source: str
|
|
31
|
+
run_started_at: datetime
|
|
32
|
+
jobs: tuple[Job, ...] = field(default_factory=tuple)
|
|
33
|
+
closable_boards: tuple[str, ...] = field(default_factory=tuple)
|
|
34
|
+
unchanged_boards: tuple[str, ...] = field(default_factory=tuple)
|
|
35
|
+
validators: tuple[HttpValidator, ...] = field(default_factory=tuple)
|
|
36
|
+
rate_state: tuple[RateState, ...] = field(default_factory=tuple)
|
|
37
|
+
workday_facets: tuple[WorkdayFacet, ...] = field(default_factory=tuple)
|
|
38
|
+
forgotten_facets: tuple[WorkdayFacet, ...] = field(default_factory=tuple)
|
|
39
|
+
workday_crawls: tuple[WorkdayCrawlStep, ...] = field(default_factory=tuple)
|
|
40
|
+
detail_fetches: tuple[DetailFetch, ...] = field(default_factory=tuple)
|
|
41
|
+
visits: tuple[CompanyVisit, ...] = field(default_factory=tuple)
|
|
42
|
+
quarantined: tuple[QuarantinedJob, ...] = field(default_factory=tuple)
|
|
43
|
+
resolve_duplicates: "Callable[[Sequence[Job], Sequence[Job]], Sequence[object]] | None" = None
|
|
44
|
+
|
|
45
|
+
closes_whole_source: bool = False
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass(frozen=True, slots=True)
|
|
49
|
+
class SourceBatchResult:
|
|
50
|
+
source: str
|
|
51
|
+
fetched: int
|
|
52
|
+
added: int
|
|
53
|
+
updated: int
|
|
54
|
+
closed: int
|
|
55
|
+
touched: int = 0
|
|
56
|
+
quarantined: int = 0
|
|
57
|
+
duplicates: int = 0
|
|
58
|
+
stored: int = 0
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class Repository(Protocol):
|
|
62
|
+
def apply_source_batch(self, batch: SourceBatch) -> SourceBatchResult:
|
|
63
|
+
pass
|
|
64
|
+
|
|
65
|
+
def load_validators(self, source: str) -> Mapping[str, HttpValidator]:
|
|
66
|
+
pass
|
|
67
|
+
|
|
68
|
+
def load_rate_state(self) -> Mapping[str, RateState]:
|
|
69
|
+
pass
|
|
70
|
+
|
|
71
|
+
def clear_rate_state(self, bucket: str | None = None) -> int:
|
|
72
|
+
pass
|
|
73
|
+
|
|
74
|
+
def stale_members(self, source: str, before: datetime) -> list[SourceVisit]:
|
|
75
|
+
pass
|
|
76
|
+
|
|
77
|
+
def detail_queue(self, source: str, limit: int) -> list[str]:
|
|
78
|
+
pass
|
|
79
|
+
|
|
80
|
+
def detail_queue_size(self, source: str) -> int:
|
|
81
|
+
pass
|
|
82
|
+
|
|
83
|
+
def load_workday_facets(self) -> Mapping[tuple[str, str], WorkdayFacet]:
|
|
84
|
+
pass
|
|
85
|
+
|
|
86
|
+
def load_workday_crawls(self) -> Mapping[str, WorkdayCrawl]:
|
|
87
|
+
pass
|
|
88
|
+
|
|
89
|
+
def list_quarantined(self, filters: QuarantineFilters) -> list[QuarantinedJob]:
|
|
90
|
+
pass
|
|
91
|
+
|
|
92
|
+
def count_duplicates(self) -> int:
|
|
93
|
+
pass
|
|
94
|
+
|
|
95
|
+
def purge(self, now: datetime) -> PurgeResult:
|
|
96
|
+
pass
|
|
97
|
+
|
|
98
|
+
def close_orphan_boards(self, sources: Sequence[str], boards: Sequence[str]) -> int:
|
|
99
|
+
pass
|
|
100
|
+
|
|
101
|
+
def preview_purge(self, now: datetime) -> PurgeResult:
|
|
102
|
+
pass
|
|
103
|
+
|
|
104
|
+
def tombstone_count(self) -> int:
|
|
105
|
+
pass
|
|
106
|
+
|
|
107
|
+
def count_quarantined(self, filters: QuarantineFilters) -> int:
|
|
108
|
+
pass
|
|
109
|
+
|
|
110
|
+
def relabel_quarantine(self, entries: Sequence[QuarantinedJob]) -> int:
|
|
111
|
+
pass
|
|
112
|
+
|
|
113
|
+
def refresh_quarantine_locations(self, resolve: Callable[[str], tuple[str, str | None]]) -> int:
|
|
114
|
+
pass
|
|
115
|
+
|
|
116
|
+
def quarantine_reason_counts(self) -> dict[str, int]:
|
|
117
|
+
pass
|
|
118
|
+
|
|
119
|
+
def list_jobs(self, filters: JobFilters) -> list[Job]:
|
|
120
|
+
pass
|
|
121
|
+
|
|
122
|
+
def get_job(self, job_id: str) -> Job | None:
|
|
123
|
+
pass
|
|
124
|
+
|
|
125
|
+
def duplicates_of(self, job_id: str) -> list[Job]:
|
|
126
|
+
pass
|
|
127
|
+
|
|
128
|
+
def search_jobs(self, query: str, filters: JobFilters) -> list[Job]:
|
|
129
|
+
pass
|
|
130
|
+
|
|
131
|
+
def count_search(self, query: str, filters: JobFilters) -> int:
|
|
132
|
+
pass
|
|
133
|
+
|
|
134
|
+
def count_jobs(self, filters: JobFilters) -> int:
|
|
135
|
+
pass
|
|
136
|
+
|
|
137
|
+
def company_names(self) -> list[str]:
|
|
138
|
+
pass
|
|
139
|
+
|
|
140
|
+
def previous_sync_at(self) -> datetime | None:
|
|
141
|
+
pass
|
|
142
|
+
|
|
143
|
+
def requests_since(self, since: datetime) -> tuple[dict[str, int], bool]:
|
|
144
|
+
pass
|
|
145
|
+
|
|
146
|
+
def closed_among(self, job_ids: Sequence[str]) -> int:
|
|
147
|
+
pass
|
|
148
|
+
|
|
149
|
+
def board_counts(self) -> dict[str, int]:
|
|
150
|
+
pass
|
|
151
|
+
|
|
152
|
+
def company_counts(self) -> dict[str, dict[str, int]]:
|
|
153
|
+
pass
|
|
154
|
+
|
|
155
|
+
def quarantine_company_counts(self) -> dict[str, dict[str, int]]:
|
|
156
|
+
pass
|
|
157
|
+
|
|
158
|
+
def quarantine_company_reasons(self) -> dict[str, dict[str, int]]:
|
|
159
|
+
pass
|
|
160
|
+
|
|
161
|
+
def company_apply_urls(self, companies: Sequence[str]) -> dict[str, tuple[str, ...]]:
|
|
162
|
+
pass
|
|
163
|
+
|
|
164
|
+
def coverage_classifications(self) -> list[CoverageClassification]:
|
|
165
|
+
pass
|
|
166
|
+
|
|
167
|
+
def record_coverage_classification(self, entry: CoverageClassification) -> bool:
|
|
168
|
+
pass
|
|
169
|
+
|
|
170
|
+
def clear_coverage_classification(self, company: str) -> bool:
|
|
171
|
+
pass
|
|
172
|
+
|
|
173
|
+
def record_sync_run(self, run: SyncRun) -> None:
|
|
174
|
+
pass
|
|
175
|
+
|
|
176
|
+
def last_sync_at(self) -> datetime | None:
|
|
177
|
+
pass
|
|
178
|
+
|
|
179
|
+
def clear_validators(self, source: str | None = None) -> int:
|
|
180
|
+
pass
|
|
181
|
+
|
|
182
|
+
def cached_url_count(self) -> int:
|
|
183
|
+
pass
|
|
184
|
+
|
|
185
|
+
def volume_history(self, limit: int) -> Mapping[str, list[VolumePoint]]:
|
|
186
|
+
pass
|
|
187
|
+
|
|
188
|
+
def run_history(self, limit: int) -> list[SyncRun]:
|
|
189
|
+
pass
|
|
190
|
+
|
|
191
|
+
def all_visits(self) -> list[SourceVisit]:
|
|
192
|
+
pass
|
|
193
|
+
|
|
194
|
+
def repair_integrity(self) -> list[IntegrityRepair]:
|
|
195
|
+
pass
|
|
196
|
+
|
|
197
|
+
def integrity_findings(self) -> list[IntegrityFinding]:
|
|
198
|
+
pass
|
|
199
|
+
|
|
200
|
+
def composition(self, column: str) -> dict[str, int]:
|
|
201
|
+
pass
|
|
202
|
+
|
|
203
|
+
def stored_counts(self) -> dict[str, int]:
|
|
204
|
+
pass
|
|
205
|
+
|
|
206
|
+
def schema_version(self) -> int:
|
|
207
|
+
pass
|
|
208
|
+
|
|
209
|
+
def close(self) -> None:
|
|
210
|
+
pass
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
__all__ = ["Repository", "SourceBatch", "SourceBatchResult"]
|
stage/storage/search.py
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
import re
|
|
2
|
+
|
|
3
|
+
_TERM = re.compile(r"^[0-9a-z]+$")
|
|
4
|
+
|
|
5
|
+
MAX_TERM_LENGTH = 48
|
|
6
|
+
MAX_TERMS = 32
|
|
7
|
+
|
|
8
|
+
FTS_COLUMN_WEIGHTS = (5.0, 10.0, 8.0, 8.0, 2.0, 1.0)
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def search_terms(query: str) -> tuple[str, ...]:
|
|
12
|
+
from stage.lexicon import fold
|
|
13
|
+
|
|
14
|
+
found = [term[:MAX_TERM_LENGTH] for term in fold(query).split() if _TERM.match(term)]
|
|
15
|
+
return tuple(found[:MAX_TERMS])
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def match_expression(terms: tuple[str, ...]) -> str:
|
|
19
|
+
return " ".join(f'"{term}"*' for term in terms)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
"FTS_COLUMN_WEIGHTS",
|
|
24
|
+
"MAX_TERMS",
|
|
25
|
+
"MAX_TERM_LENGTH",
|
|
26
|
+
"match_expression",
|
|
27
|
+
"search_terms",
|
|
28
|
+
]
|