stage-cli 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. stage/__init__.py +1 -0
  2. stage/__main__.py +8 -0
  3. stage/banner.py +32 -0
  4. stage/bootstrap/__init__.py +0 -0
  5. stage/bootstrap/openjobs.py +392 -0
  6. stage/classify/__init__.py +29 -0
  7. stage/classify/eligibility.py +115 -0
  8. stage/classify/internship.py +64 -0
  9. stage/classify/role.py +91 -0
  10. stage/classify/scope.py +47 -0
  11. stage/cli/__init__.py +0 -0
  12. stage/cli/app.py +4 -0
  13. stage/cli/commands/__init__.py +8 -0
  14. stage/cli/commands/discovery.py +294 -0
  15. stage/cli/commands/insight.py +494 -0
  16. stage/cli/commands/pipeline.py +337 -0
  17. stage/cli/commands/postings.py +473 -0
  18. stage/cli/commands/schedule.py +171 -0
  19. stage/cli/housekeeping.py +64 -0
  20. stage/cli/logfile.py +56 -0
  21. stage/cli/notify.py +170 -0
  22. stage/cli/options.py +678 -0
  23. stage/cli/render.py +1398 -0
  24. stage/cli/runlock.py +74 -0
  25. stage/cli/schedule.py +702 -0
  26. stage/cli/schedule_state.py +363 -0
  27. stage/cli/selection.py +83 -0
  28. stage/cli/serialize.py +196 -0
  29. stage/companies.py +542 -0
  30. stage/data/companies/a.yaml +1289 -0
  31. stage/data/companies/b.yaml +900 -0
  32. stage/data/companies/c.yaml +1377 -0
  33. stage/data/companies/d.yaml +497 -0
  34. stage/data/companies/e.yaml +519 -0
  35. stage/data/companies/f.yaml +454 -0
  36. stage/data/companies/g.yaml +601 -0
  37. stage/data/companies/h.yaml +446 -0
  38. stage/data/companies/i.yaml +503 -0
  39. stage/data/companies/j.yaml +138 -0
  40. stage/data/companies/k.yaml +278 -0
  41. stage/data/companies/l.yaml +402 -0
  42. stage/data/companies/m.yaml +937 -0
  43. stage/data/companies/n.yaml +549 -0
  44. stage/data/companies/o.yaml +371 -0
  45. stage/data/companies/other.yaml +58 -0
  46. stage/data/companies/p.yaml +825 -0
  47. stage/data/companies/q.yaml +121 -0
  48. stage/data/companies/r.yaml +583 -0
  49. stage/data/companies/s.yaml +1140 -0
  50. stage/data/companies/t.yaml +817 -0
  51. stage/data/companies/u.yaml +196 -0
  52. stage/data/companies/v.yaml +325 -0
  53. stage/data/companies/w.yaml +353 -0
  54. stage/data/companies/x.yaml +67 -0
  55. stage/data/companies/y.yaml +36 -0
  56. stage/data/companies/z.yaml +146 -0
  57. stage/data/fonts/DejaVuSans.LICENSE.txt +99 -0
  58. stage/data/fonts/DejaVuSans.ttf +0 -0
  59. stage/data/lexicon/company_tokens.yaml +228 -0
  60. stage/data/lexicon/eligibility.yaml +455 -0
  61. stage/data/lexicon/inclusive_suffixes.yaml +37 -0
  62. stage/data/lexicon/internship.yaml +187 -0
  63. stage/data/lexicon/language.yaml +226 -0
  64. stage/data/lexicon/locations.yaml +1159 -0
  65. stage/data/lexicon/roles.yaml +2012 -0
  66. stage/data/lexicon/terms.yaml +76 -0
  67. stage/data/lexicon/workday_facets.yaml +27 -0
  68. stage/data/seed_companies.yaml +198 -0
  69. stage/dedup/__init__.py +19 -0
  70. stage/dedup/identity.py +113 -0
  71. stage/dedup/resolve.py +97 -0
  72. stage/domain/__init__.py +244 -0
  73. stage/domain/company.py +49 -0
  74. stage/domain/coverage.py +86 -0
  75. stage/domain/custom_board.py +92 -0
  76. stage/domain/discovery.py +94 -0
  77. stage/domain/enums.py +114 -0
  78. stage/domain/events.py +204 -0
  79. stage/domain/filters.py +27 -0
  80. stage/domain/health.py +169 -0
  81. stage/domain/ids.py +48 -0
  82. stage/domain/job.py +47 -0
  83. stage/domain/matching.py +15 -0
  84. stage/domain/priority.py +34 -0
  85. stage/domain/quarantine.py +39 -0
  86. stage/domain/rate_state.py +78 -0
  87. stage/domain/retention.py +20 -0
  88. stage/domain/rotation.py +46 -0
  89. stage/domain/signals.py +12 -0
  90. stage/domain/sync_run.py +35 -0
  91. stage/domain/text.py +113 -0
  92. stage/domain/validator.py +14 -0
  93. stage/domain/visits.py +60 -0
  94. stage/domain/workday.py +38 -0
  95. stage/http/__init__.py +58 -0
  96. stage/http/breaker.py +53 -0
  97. stage/http/cache.py +44 -0
  98. stage/http/client.py +725 -0
  99. stage/http/profiles.py +101 -0
  100. stage/lexicon.py +370 -0
  101. stage/normalize/__init__.py +16 -0
  102. stage/normalize/language.py +47 -0
  103. stage/normalize/location.py +271 -0
  104. stage/normalize/terms.py +153 -0
  105. stage/normalize/urls.py +122 -0
  106. stage/paths.py +86 -0
  107. stage/py.typed +0 -0
  108. stage/services/__init__.py +0 -0
  109. stage/services/canary.py +120 -0
  110. stage/services/coverage.py +231 -0
  111. stage/services/discover.py +747 -0
  112. stage/services/export.py +274 -0
  113. stage/services/health.py +237 -0
  114. stage/services/maintenance.py +225 -0
  115. stage/services/quarantine.py +20 -0
  116. stage/services/query.py +86 -0
  117. stage/services/sync.py +1257 -0
  118. stage/sources/__init__.py +82 -0
  119. stage/sources/_text.py +79 -0
  120. stage/sources/ashby.py +93 -0
  121. stage/sources/bamboohr.py +80 -0
  122. stage/sources/base.py +225 -0
  123. stage/sources/breezy.py +90 -0
  124. stage/sources/collage.py +60 -0
  125. stage/sources/community_feeds.py +142 -0
  126. stage/sources/curated_markdown.py +289 -0
  127. stage/sources/custom_json.py +610 -0
  128. stage/sources/espresso.py +154 -0
  129. stage/sources/feed.py +44 -0
  130. stage/sources/greenhouse.py +104 -0
  131. stage/sources/jobbank.py +147 -0
  132. stage/sources/jobvite.py +133 -0
  133. stage/sources/lever.py +76 -0
  134. stage/sources/oracle_cloud.py +187 -0
  135. stage/sources/platforms.py +609 -0
  136. stage/sources/quebec_emploi.py +146 -0
  137. stage/sources/recruitee.py +96 -0
  138. stage/sources/simplify.py +110 -0
  139. stage/sources/smartrecruiters.py +216 -0
  140. stage/sources/speedyapply.py +200 -0
  141. stage/sources/themuse.py +157 -0
  142. stage/sources/workable.py +83 -0
  143. stage/sources/workday.py +524 -0
  144. stage/sources/zshah.py +99 -0
  145. stage/storage/__init__.py +29 -0
  146. stage/storage/migrations/0001_initial.sql +239 -0
  147. stage/storage/migrations/__init__.py +135 -0
  148. stage/storage/repository.py +213 -0
  149. stage/storage/search.py +28 -0
  150. stage/storage/sqlite_repo.py +1586 -0
  151. stage/storage/writer.py +249 -0
  152. stage/tui/__init__.py +0 -0
  153. stage/tui/app.py +82 -0
  154. stage/tui/help.py +26 -0
  155. stage/tui/safe.py +21 -0
  156. stage/tui/screens/__init__.py +0 -0
  157. stage/tui/screens/boards.py +186 -0
  158. stage/tui/screens/postings.py +509 -0
  159. stage/tui/screens/review.py +209 -0
  160. stage/tui/screens/splash.py +37 -0
  161. stage/tui/screens/stats.py +124 -0
  162. stage/tui/screens/sync.py +194 -0
  163. stage/tui/state.py +160 -0
  164. stage/tui/theme.tcss +205 -0
  165. stage/tui/widgets/__init__.py +0 -0
  166. stage_cli-1.0.0.dist-info/METADATA +379 -0
  167. stage_cli-1.0.0.dist-info/RECORD +170 -0
  168. stage_cli-1.0.0.dist-info/WHEEL +4 -0
  169. stage_cli-1.0.0.dist-info/entry_points.txt +2 -0
  170. stage_cli-1.0.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,239 @@
1
+ CREATE TABLE jobs (
2
+ id TEXT PRIMARY KEY,
3
+ source TEXT NOT NULL,
4
+ company TEXT NOT NULL,
5
+ company_fold TEXT NOT NULL DEFAULT '',
6
+ title_raw TEXT NOT NULL,
7
+ title_normalized TEXT NOT NULL,
8
+ title_canonical TEXT NOT NULL DEFAULT '',
9
+ apply_url_raw TEXT NOT NULL,
10
+ apply_url_canonical TEXT NOT NULL DEFAULT '',
11
+ description TEXT NOT NULL DEFAULT '',
12
+ location_raw TEXT NOT NULL DEFAULT '',
13
+ location TEXT NOT NULL DEFAULT 'unknown',
14
+ remote_scope TEXT,
15
+ language TEXT NOT NULL DEFAULT 'unknown',
16
+ term TEXT NOT NULL DEFAULT 'unknown',
17
+ role TEXT NOT NULL DEFAULT 'unknown',
18
+ degree_requirement TEXT NOT NULL DEFAULT 'unknown',
19
+ work_auth_flag INTEGER NOT NULL DEFAULT 0,
20
+ compensation TEXT,
21
+ employment_type TEXT NOT NULL DEFAULT '',
22
+ source_category TEXT NOT NULL DEFAULT '',
23
+ status TEXT NOT NULL DEFAULT 'open',
24
+ duplicate_of TEXT,
25
+ first_seen TEXT NOT NULL,
26
+ last_seen TEXT NOT NULL,
27
+ source_posted_at TEXT
28
+ );
29
+
30
+ CREATE INDEX idx_jobs_first_seen ON jobs (first_seen DESC);
31
+ CREATE INDEX idx_jobs_status_first_seen ON jobs (status, first_seen DESC);
32
+ CREATE INDEX idx_jobs_source_company ON jobs (source, company);
33
+ CREATE INDEX idx_jobs_duplicate_of ON jobs (duplicate_of);
34
+ CREATE INDEX idx_jobs_degree ON jobs (degree_requirement);
35
+ CREATE INDEX idx_jobs_company_canonical ON jobs (company) WHERE duplicate_of IS NULL;
36
+ CREATE INDEX idx_jobs_company_fold ON jobs (company_fold);
37
+ CREATE INDEX idx_jobs_apply_url_canonical ON jobs (apply_url_canonical);
38
+
39
+ CREATE VIRTUAL TABLE jobs_fts USING fts5 (
40
+ company,
41
+ title_raw,
42
+ title_normalized,
43
+ title_canonical,
44
+ location_raw,
45
+ description,
46
+ content = 'jobs',
47
+ content_rowid = 'rowid',
48
+ tokenize = 'unicode61 remove_diacritics 2'
49
+ );
50
+
51
+ CREATE TRIGGER jobs_fts_after_insert AFTER INSERT ON jobs BEGIN
52
+ INSERT INTO jobs_fts (
53
+ rowid, company, title_raw, title_normalized, title_canonical,
54
+ location_raw, description
55
+ )
56
+ VALUES (
57
+ new.rowid, new.company, new.title_raw, new.title_normalized,
58
+ new.title_canonical, new.location_raw, new.description
59
+ );
60
+ END;
61
+
62
+ CREATE TRIGGER jobs_fts_after_delete AFTER DELETE ON jobs BEGIN
63
+ INSERT INTO jobs_fts (
64
+ jobs_fts, rowid, company, title_raw, title_normalized, title_canonical,
65
+ location_raw, description
66
+ )
67
+ VALUES (
68
+ 'delete', old.rowid, old.company, old.title_raw, old.title_normalized,
69
+ old.title_canonical, old.location_raw, old.description
70
+ );
71
+ END;
72
+
73
+ CREATE TRIGGER jobs_fts_after_update AFTER UPDATE ON jobs
74
+ WHEN new.company IS NOT old.company
75
+ OR new.title_raw IS NOT old.title_raw
76
+ OR new.title_normalized IS NOT old.title_normalized
77
+ OR new.title_canonical IS NOT old.title_canonical
78
+ OR new.location_raw IS NOT old.location_raw
79
+ OR new.description IS NOT old.description
80
+ BEGIN
81
+ INSERT INTO jobs_fts (
82
+ jobs_fts, rowid, company, title_raw, title_normalized, title_canonical,
83
+ location_raw, description
84
+ )
85
+ VALUES (
86
+ 'delete', old.rowid, old.company, old.title_raw, old.title_normalized,
87
+ old.title_canonical, old.location_raw, old.description
88
+ );
89
+ INSERT INTO jobs_fts (
90
+ rowid, company, title_raw, title_normalized, title_canonical,
91
+ location_raw, description
92
+ )
93
+ VALUES (
94
+ new.rowid, new.company, new.title_raw, new.title_normalized,
95
+ new.title_canonical, new.location_raw, new.description
96
+ );
97
+ END;
98
+
99
+ CREATE TABLE quarantine (
100
+ id TEXT PRIMARY KEY,
101
+ source TEXT NOT NULL,
102
+ company TEXT NOT NULL,
103
+ title_raw TEXT NOT NULL,
104
+ apply_url_raw TEXT NOT NULL DEFAULT '',
105
+ location_raw TEXT NOT NULL DEFAULT '',
106
+ location TEXT NOT NULL DEFAULT 'unknown',
107
+ remote_scope TEXT,
108
+ reason TEXT NOT NULL,
109
+ matched_phrase TEXT NOT NULL DEFAULT '',
110
+ first_seen TEXT NOT NULL,
111
+ last_seen TEXT NOT NULL
112
+ );
113
+
114
+ CREATE INDEX idx_quarantine_first_seen ON quarantine (first_seen DESC);
115
+ CREATE INDEX idx_quarantine_reason_first_seen ON quarantine (reason, first_seen DESC);
116
+ CREATE INDEX idx_quarantine_company ON quarantine (company, last_seen DESC);
117
+ CREATE INDEX idx_quarantine_source_company ON quarantine (source, company);
118
+
119
+ CREATE TABLE tombstones (
120
+ id TEXT PRIMARY KEY,
121
+ source TEXT NOT NULL,
122
+ first_seen TEXT NOT NULL,
123
+ purged_at TEXT NOT NULL
124
+ );
125
+
126
+ CREATE INDEX idx_tombstones_purged_at ON tombstones (purged_at DESC);
127
+
128
+ CREATE TABLE http_cache (
129
+ url TEXT PRIMARY KEY,
130
+ source TEXT NOT NULL,
131
+ etag TEXT,
132
+ last_modified TEXT,
133
+ fetched_at TEXT NOT NULL
134
+ );
135
+
136
+ CREATE INDEX idx_http_cache_source ON http_cache (source);
137
+
138
+ CREATE TABLE rate_state (
139
+ bucket TEXT PRIMARY KEY,
140
+ blocked_until TEXT,
141
+ min_interval_override REAL,
142
+ consecutive_failures INTEGER NOT NULL DEFAULT 0,
143
+ last_failure_at TEXT,
144
+ reason TEXT NOT NULL DEFAULT '',
145
+ rotation_cursor TEXT NOT NULL DEFAULT '',
146
+ updated_at TEXT NOT NULL
147
+ );
148
+
149
+ CREATE TABLE source_visits (
150
+ source TEXT NOT NULL,
151
+ board TEXT NOT NULL,
152
+ label TEXT NOT NULL DEFAULT '',
153
+ last_attempt_at TEXT NOT NULL,
154
+ last_success_at TEXT,
155
+ consecutive_failures INTEGER NOT NULL DEFAULT 0,
156
+ last_error TEXT NOT NULL DEFAULT '',
157
+ PRIMARY KEY (source, board)
158
+ );
159
+
160
+ CREATE INDEX idx_source_visits_success ON source_visits (source, last_success_at);
161
+
162
+ CREATE TABLE coverage_classifications (
163
+ company TEXT NOT NULL CHECK (length(trim(company)) > 0),
164
+ company_fold TEXT PRIMARY KEY,
165
+ disposition TEXT NOT NULL CHECK (disposition IN (
166
+ 'feed-only', 'unavailable', 'custom-json-candidate', 'adapter-candidate', 'deferred'
167
+ )),
168
+ note TEXT NOT NULL CHECK (length(trim(note)) > 0),
169
+ checked_on TEXT NOT NULL,
170
+ url TEXT
171
+ );
172
+
173
+ CREATE TABLE workday_facets (
174
+ tenant TEXT NOT NULL,
175
+ site TEXT NOT NULL,
176
+ parameter TEXT NOT NULL,
177
+ facet_id TEXT NOT NULL,
178
+ descriptor TEXT NOT NULL DEFAULT '',
179
+ resolved_at TEXT NOT NULL,
180
+ PRIMARY KEY (tenant, site)
181
+ );
182
+
183
+ CREATE TABLE workday_crawls (
184
+ board TEXT PRIMARY KEY,
185
+ next_offset INTEGER NOT NULL CHECK (next_offset >= 0),
186
+ total INTEGER,
187
+ facet_parameter TEXT NOT NULL DEFAULT '',
188
+ facet_ids TEXT NOT NULL DEFAULT ''
189
+ );
190
+
191
+ CREATE TABLE workday_crawl_seen (
192
+ board TEXT NOT NULL REFERENCES workday_crawls (board) ON DELETE CASCADE,
193
+ id TEXT NOT NULL,
194
+ PRIMARY KEY (board, id)
195
+ );
196
+
197
+ CREATE INDEX idx_workday_crawl_seen_id ON workday_crawl_seen (id);
198
+
199
+ CREATE TABLE detail_fetches (
200
+ id TEXT PRIMARY KEY,
201
+ source TEXT NOT NULL,
202
+ fetched_at TEXT NOT NULL,
203
+ resolved INTEGER NOT NULL DEFAULT 0,
204
+ attempts INTEGER NOT NULL DEFAULT 1,
205
+ failed INTEGER NOT NULL DEFAULT 0
206
+ );
207
+
208
+ CREATE INDEX idx_detail_fetches_source ON detail_fetches (source, resolved, failed);
209
+
210
+ CREATE TABLE sync_runs (
211
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
212
+ started_at TEXT NOT NULL,
213
+ finished_at TEXT NOT NULL,
214
+ outcome TEXT NOT NULL
215
+ );
216
+
217
+ CREATE INDEX idx_sync_runs_started_at ON sync_runs (started_at DESC);
218
+
219
+ CREATE TABLE sync_run_sources (
220
+ run_id INTEGER NOT NULL REFERENCES sync_runs (id) ON DELETE CASCADE,
221
+ source TEXT NOT NULL,
222
+ fetched INTEGER NOT NULL DEFAULT 0,
223
+ added INTEGER NOT NULL DEFAULT 0,
224
+ updated INTEGER NOT NULL DEFAULT 0,
225
+ closed INTEGER NOT NULL DEFAULT 0,
226
+ errors INTEGER NOT NULL DEFAULT 0,
227
+ requests INTEGER NOT NULL DEFAULT 0,
228
+ not_modified INTEGER NOT NULL DEFAULT 0,
229
+ retries INTEGER NOT NULL DEFAULT 0,
230
+ tightenings INTEGER NOT NULL DEFAULT 0,
231
+ quarantined INTEGER NOT NULL DEFAULT 0,
232
+ deferred INTEGER NOT NULL DEFAULT 0,
233
+ blocked INTEGER NOT NULL DEFAULT 0,
234
+ stored INTEGER NOT NULL DEFAULT -1,
235
+ latency_p50_ms REAL NOT NULL DEFAULT 0,
236
+ latency_p95_ms REAL NOT NULL DEFAULT 0,
237
+ elapsed_ms REAL NOT NULL DEFAULT 0,
238
+ PRIMARY KEY (run_id, source)
239
+ );
@@ -0,0 +1,135 @@
1
+ import re
2
+ import sqlite3
3
+ from dataclasses import dataclass
4
+ from datetime import UTC, datetime
5
+ from pathlib import Path
6
+
7
+ _FILENAME = re.compile(r"^(\d{4})_[a-z0-9_]+\.sql$")
8
+ _MIGRATIONS_DIR = Path(__file__).resolve().parent
9
+
10
+
11
+ class SchemaVersionError(Exception):
12
+ pass
13
+
14
+
15
+ @dataclass(frozen=True, slots=True)
16
+ class Migration:
17
+ version: int
18
+ name: str
19
+ path: Path
20
+
21
+ def read(self) -> str:
22
+ return self.path.read_text(encoding="utf-8")
23
+
24
+
25
+ def discover() -> tuple[Migration, ...]:
26
+ found: list[Migration] = []
27
+ for path in sorted(_MIGRATIONS_DIR.glob("*.sql")):
28
+ match = _FILENAME.match(path.name)
29
+ if match is None:
30
+ raise SchemaVersionError(f"migration {path.name!r} does not follow NNNN_name.sql")
31
+ found.append(Migration(version=int(match.group(1)), name=path.stem, path=path))
32
+ versions = [migration.version for migration in found]
33
+ if len(set(versions)) != len(versions):
34
+ raise SchemaVersionError("duplicate migration version numbers")
35
+ return tuple(found)
36
+
37
+
38
+ def latest_version() -> int:
39
+ migrations = discover()
40
+ return migrations[-1].version if migrations else 0
41
+
42
+
43
+ def _ensure_migrations_table(conn: sqlite3.Connection) -> None:
44
+ conn.execute(
45
+ "CREATE TABLE IF NOT EXISTS schema_migrations ("
46
+ "version INTEGER PRIMARY KEY, name TEXT NOT NULL, applied_at TEXT NOT NULL)"
47
+ )
48
+
49
+
50
+ def applied_versions(conn: sqlite3.Connection) -> tuple[int, ...]:
51
+ _ensure_migrations_table(conn)
52
+ rows = conn.execute("SELECT version FROM schema_migrations ORDER BY version").fetchall()
53
+ return tuple(int(row[0]) for row in rows)
54
+
55
+
56
+ BASELINE_COLUMNS = frozenset({"company_fold", "employment_type", "source_category"})
57
+ BASELINE_TABLES = frozenset({"coverage_classifications", "workday_crawls", "workday_crawl_seen"})
58
+
59
+
60
+ def _refuse_a_pre_baseline_database(conn: sqlite3.Connection, db_path: Path) -> None:
61
+ columns = {str(row[1]) for row in conn.execute("PRAGMA table_info(jobs)").fetchall()}
62
+ tables = {
63
+ str(row[0])
64
+ for row in conn.execute("SELECT name FROM sqlite_master WHERE type = 'table'").fetchall()
65
+ }
66
+ missing = sorted(BASELINE_COLUMNS - columns) + sorted(BASELINE_TABLES - tables)
67
+ if not missing:
68
+ return
69
+ raise SchemaVersionError(
70
+ f"database at {db_path} records the current schema version but is missing "
71
+ f"{', '.join(missing)} — it predates this baseline. Back it up or delete it, then run "
72
+ "stage sync; the corpus is reproducible in under a minute"
73
+ )
74
+
75
+
76
+ def snapshot_path(db_path: Path, when: datetime) -> Path:
77
+ return db_path.with_name(f"{db_path.name}.bak-{when.strftime('%Y%m%dT%H%M%S')}")
78
+
79
+
80
+ def _snapshot(conn: sqlite3.Connection, db_path: Path) -> Path:
81
+ target = snapshot_path(db_path, datetime.now(UTC))
82
+ backup = sqlite3.connect(target)
83
+ try:
84
+ conn.backup(backup)
85
+ finally:
86
+ backup.close()
87
+ return target
88
+
89
+
90
+ def migrate(conn: sqlite3.Connection, db_path: Path) -> tuple[int, ...]:
91
+ migrations = discover()
92
+ known = {migration.version for migration in migrations}
93
+ applied = applied_versions(conn)
94
+
95
+ unknown = [version for version in applied if version not in known]
96
+ if unknown:
97
+ raise SchemaVersionError(
98
+ f"database at {db_path} has schema version {max(unknown)}, newer than this build "
99
+ f"understands ({latest_version()}) — upgrade stage-cli"
100
+ )
101
+
102
+ pending = [migration for migration in migrations if migration.version not in applied]
103
+ if not pending:
104
+ _refuse_a_pre_baseline_database(conn, db_path)
105
+ return ()
106
+
107
+ if applied:
108
+ _snapshot(conn, db_path)
109
+
110
+ for migration in pending:
111
+ try:
112
+ conn.executescript(f"BEGIN IMMEDIATE;\n{migration.read()}")
113
+ conn.execute(
114
+ "INSERT INTO schema_migrations (version, name, applied_at) VALUES (?, ?, ?)",
115
+ (migration.version, migration.name, datetime.now(UTC).isoformat()),
116
+ )
117
+ except Exception:
118
+ conn.rollback()
119
+ raise
120
+ conn.commit()
121
+
122
+ return tuple(migration.version for migration in pending)
123
+
124
+
125
+ __all__ = [
126
+ "BASELINE_COLUMNS",
127
+ "BASELINE_TABLES",
128
+ "Migration",
129
+ "SchemaVersionError",
130
+ "applied_versions",
131
+ "discover",
132
+ "latest_version",
133
+ "migrate",
134
+ "snapshot_path",
135
+ ]
@@ -0,0 +1,213 @@
1
+ from collections.abc import Callable, Mapping, Sequence
2
+ from dataclasses import dataclass, field
3
+ from datetime import datetime
4
+ from typing import Protocol
5
+
6
+ from stage.domain import (
7
+ CompanyVisit,
8
+ CoverageClassification,
9
+ DetailFetch,
10
+ HttpValidator,
11
+ IntegrityFinding,
12
+ IntegrityRepair,
13
+ Job,
14
+ JobFilters,
15
+ PurgeResult,
16
+ QuarantinedJob,
17
+ QuarantineFilters,
18
+ RateState,
19
+ SourceVisit,
20
+ SyncRun,
21
+ VolumePoint,
22
+ WorkdayCrawl,
23
+ WorkdayCrawlStep,
24
+ WorkdayFacet,
25
+ )
26
+
27
+
28
+ @dataclass(frozen=True, slots=True)
29
+ class SourceBatch:
30
+ source: str
31
+ run_started_at: datetime
32
+ jobs: tuple[Job, ...] = field(default_factory=tuple)
33
+ closable_boards: tuple[str, ...] = field(default_factory=tuple)
34
+ unchanged_boards: tuple[str, ...] = field(default_factory=tuple)
35
+ validators: tuple[HttpValidator, ...] = field(default_factory=tuple)
36
+ rate_state: tuple[RateState, ...] = field(default_factory=tuple)
37
+ workday_facets: tuple[WorkdayFacet, ...] = field(default_factory=tuple)
38
+ forgotten_facets: tuple[WorkdayFacet, ...] = field(default_factory=tuple)
39
+ workday_crawls: tuple[WorkdayCrawlStep, ...] = field(default_factory=tuple)
40
+ detail_fetches: tuple[DetailFetch, ...] = field(default_factory=tuple)
41
+ visits: tuple[CompanyVisit, ...] = field(default_factory=tuple)
42
+ quarantined: tuple[QuarantinedJob, ...] = field(default_factory=tuple)
43
+ resolve_duplicates: "Callable[[Sequence[Job], Sequence[Job]], Sequence[object]] | None" = None
44
+
45
+ closes_whole_source: bool = False
46
+
47
+
48
+ @dataclass(frozen=True, slots=True)
49
+ class SourceBatchResult:
50
+ source: str
51
+ fetched: int
52
+ added: int
53
+ updated: int
54
+ closed: int
55
+ touched: int = 0
56
+ quarantined: int = 0
57
+ duplicates: int = 0
58
+ stored: int = 0
59
+
60
+
61
+ class Repository(Protocol):
62
+ def apply_source_batch(self, batch: SourceBatch) -> SourceBatchResult:
63
+ pass
64
+
65
+ def load_validators(self, source: str) -> Mapping[str, HttpValidator]:
66
+ pass
67
+
68
+ def load_rate_state(self) -> Mapping[str, RateState]:
69
+ pass
70
+
71
+ def clear_rate_state(self, bucket: str | None = None) -> int:
72
+ pass
73
+
74
+ def stale_members(self, source: str, before: datetime) -> list[SourceVisit]:
75
+ pass
76
+
77
+ def detail_queue(self, source: str, limit: int) -> list[str]:
78
+ pass
79
+
80
+ def detail_queue_size(self, source: str) -> int:
81
+ pass
82
+
83
+ def load_workday_facets(self) -> Mapping[tuple[str, str], WorkdayFacet]:
84
+ pass
85
+
86
+ def load_workday_crawls(self) -> Mapping[str, WorkdayCrawl]:
87
+ pass
88
+
89
+ def list_quarantined(self, filters: QuarantineFilters) -> list[QuarantinedJob]:
90
+ pass
91
+
92
+ def count_duplicates(self) -> int:
93
+ pass
94
+
95
+ def purge(self, now: datetime) -> PurgeResult:
96
+ pass
97
+
98
+ def close_orphan_boards(self, sources: Sequence[str], boards: Sequence[str]) -> int:
99
+ pass
100
+
101
+ def preview_purge(self, now: datetime) -> PurgeResult:
102
+ pass
103
+
104
+ def tombstone_count(self) -> int:
105
+ pass
106
+
107
+ def count_quarantined(self, filters: QuarantineFilters) -> int:
108
+ pass
109
+
110
+ def relabel_quarantine(self, entries: Sequence[QuarantinedJob]) -> int:
111
+ pass
112
+
113
+ def refresh_quarantine_locations(self, resolve: Callable[[str], tuple[str, str | None]]) -> int:
114
+ pass
115
+
116
+ def quarantine_reason_counts(self) -> dict[str, int]:
117
+ pass
118
+
119
+ def list_jobs(self, filters: JobFilters) -> list[Job]:
120
+ pass
121
+
122
+ def get_job(self, job_id: str) -> Job | None:
123
+ pass
124
+
125
+ def duplicates_of(self, job_id: str) -> list[Job]:
126
+ pass
127
+
128
+ def search_jobs(self, query: str, filters: JobFilters) -> list[Job]:
129
+ pass
130
+
131
+ def count_search(self, query: str, filters: JobFilters) -> int:
132
+ pass
133
+
134
+ def count_jobs(self, filters: JobFilters) -> int:
135
+ pass
136
+
137
+ def company_names(self) -> list[str]:
138
+ pass
139
+
140
+ def previous_sync_at(self) -> datetime | None:
141
+ pass
142
+
143
+ def requests_since(self, since: datetime) -> tuple[dict[str, int], bool]:
144
+ pass
145
+
146
+ def closed_among(self, job_ids: Sequence[str]) -> int:
147
+ pass
148
+
149
+ def board_counts(self) -> dict[str, int]:
150
+ pass
151
+
152
+ def company_counts(self) -> dict[str, dict[str, int]]:
153
+ pass
154
+
155
+ def quarantine_company_counts(self) -> dict[str, dict[str, int]]:
156
+ pass
157
+
158
+ def quarantine_company_reasons(self) -> dict[str, dict[str, int]]:
159
+ pass
160
+
161
+ def company_apply_urls(self, companies: Sequence[str]) -> dict[str, tuple[str, ...]]:
162
+ pass
163
+
164
+ def coverage_classifications(self) -> list[CoverageClassification]:
165
+ pass
166
+
167
+ def record_coverage_classification(self, entry: CoverageClassification) -> bool:
168
+ pass
169
+
170
+ def clear_coverage_classification(self, company: str) -> bool:
171
+ pass
172
+
173
+ def record_sync_run(self, run: SyncRun) -> None:
174
+ pass
175
+
176
+ def last_sync_at(self) -> datetime | None:
177
+ pass
178
+
179
+ def clear_validators(self, source: str | None = None) -> int:
180
+ pass
181
+
182
+ def cached_url_count(self) -> int:
183
+ pass
184
+
185
+ def volume_history(self, limit: int) -> Mapping[str, list[VolumePoint]]:
186
+ pass
187
+
188
+ def run_history(self, limit: int) -> list[SyncRun]:
189
+ pass
190
+
191
+ def all_visits(self) -> list[SourceVisit]:
192
+ pass
193
+
194
+ def repair_integrity(self) -> list[IntegrityRepair]:
195
+ pass
196
+
197
+ def integrity_findings(self) -> list[IntegrityFinding]:
198
+ pass
199
+
200
+ def composition(self, column: str) -> dict[str, int]:
201
+ pass
202
+
203
+ def stored_counts(self) -> dict[str, int]:
204
+ pass
205
+
206
+ def schema_version(self) -> int:
207
+ pass
208
+
209
+ def close(self) -> None:
210
+ pass
211
+
212
+
213
+ __all__ = ["Repository", "SourceBatch", "SourceBatchResult"]
@@ -0,0 +1,28 @@
1
+ import re
2
+
3
+ _TERM = re.compile(r"^[0-9a-z]+$")
4
+
5
+ MAX_TERM_LENGTH = 48
6
+ MAX_TERMS = 32
7
+
8
+ FTS_COLUMN_WEIGHTS = (5.0, 10.0, 8.0, 8.0, 2.0, 1.0)
9
+
10
+
11
+ def search_terms(query: str) -> tuple[str, ...]:
12
+ from stage.lexicon import fold
13
+
14
+ found = [term[:MAX_TERM_LENGTH] for term in fold(query).split() if _TERM.match(term)]
15
+ return tuple(found[:MAX_TERMS])
16
+
17
+
18
+ def match_expression(terms: tuple[str, ...]) -> str:
19
+ return " ".join(f'"{term}"*' for term in terms)
20
+
21
+
22
+ __all__ = [
23
+ "FTS_COLUMN_WEIGHTS",
24
+ "MAX_TERMS",
25
+ "MAX_TERM_LENGTH",
26
+ "match_expression",
27
+ "search_terms",
28
+ ]