stage-cli 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. stage/__init__.py +1 -0
  2. stage/__main__.py +8 -0
  3. stage/banner.py +32 -0
  4. stage/bootstrap/__init__.py +0 -0
  5. stage/bootstrap/openjobs.py +392 -0
  6. stage/classify/__init__.py +29 -0
  7. stage/classify/eligibility.py +115 -0
  8. stage/classify/internship.py +64 -0
  9. stage/classify/role.py +91 -0
  10. stage/classify/scope.py +47 -0
  11. stage/cli/__init__.py +0 -0
  12. stage/cli/app.py +4 -0
  13. stage/cli/commands/__init__.py +8 -0
  14. stage/cli/commands/discovery.py +294 -0
  15. stage/cli/commands/insight.py +494 -0
  16. stage/cli/commands/pipeline.py +337 -0
  17. stage/cli/commands/postings.py +473 -0
  18. stage/cli/commands/schedule.py +171 -0
  19. stage/cli/housekeeping.py +64 -0
  20. stage/cli/logfile.py +56 -0
  21. stage/cli/notify.py +170 -0
  22. stage/cli/options.py +678 -0
  23. stage/cli/render.py +1398 -0
  24. stage/cli/runlock.py +74 -0
  25. stage/cli/schedule.py +702 -0
  26. stage/cli/schedule_state.py +363 -0
  27. stage/cli/selection.py +83 -0
  28. stage/cli/serialize.py +196 -0
  29. stage/companies.py +542 -0
  30. stage/data/companies/a.yaml +1289 -0
  31. stage/data/companies/b.yaml +900 -0
  32. stage/data/companies/c.yaml +1377 -0
  33. stage/data/companies/d.yaml +497 -0
  34. stage/data/companies/e.yaml +519 -0
  35. stage/data/companies/f.yaml +454 -0
  36. stage/data/companies/g.yaml +601 -0
  37. stage/data/companies/h.yaml +446 -0
  38. stage/data/companies/i.yaml +503 -0
  39. stage/data/companies/j.yaml +138 -0
  40. stage/data/companies/k.yaml +278 -0
  41. stage/data/companies/l.yaml +402 -0
  42. stage/data/companies/m.yaml +937 -0
  43. stage/data/companies/n.yaml +549 -0
  44. stage/data/companies/o.yaml +371 -0
  45. stage/data/companies/other.yaml +58 -0
  46. stage/data/companies/p.yaml +825 -0
  47. stage/data/companies/q.yaml +121 -0
  48. stage/data/companies/r.yaml +583 -0
  49. stage/data/companies/s.yaml +1140 -0
  50. stage/data/companies/t.yaml +817 -0
  51. stage/data/companies/u.yaml +196 -0
  52. stage/data/companies/v.yaml +325 -0
  53. stage/data/companies/w.yaml +353 -0
  54. stage/data/companies/x.yaml +67 -0
  55. stage/data/companies/y.yaml +36 -0
  56. stage/data/companies/z.yaml +146 -0
  57. stage/data/fonts/DejaVuSans.LICENSE.txt +99 -0
  58. stage/data/fonts/DejaVuSans.ttf +0 -0
  59. stage/data/lexicon/company_tokens.yaml +228 -0
  60. stage/data/lexicon/eligibility.yaml +455 -0
  61. stage/data/lexicon/inclusive_suffixes.yaml +37 -0
  62. stage/data/lexicon/internship.yaml +187 -0
  63. stage/data/lexicon/language.yaml +226 -0
  64. stage/data/lexicon/locations.yaml +1159 -0
  65. stage/data/lexicon/roles.yaml +2012 -0
  66. stage/data/lexicon/terms.yaml +76 -0
  67. stage/data/lexicon/workday_facets.yaml +27 -0
  68. stage/data/seed_companies.yaml +198 -0
  69. stage/dedup/__init__.py +19 -0
  70. stage/dedup/identity.py +113 -0
  71. stage/dedup/resolve.py +97 -0
  72. stage/domain/__init__.py +244 -0
  73. stage/domain/company.py +49 -0
  74. stage/domain/coverage.py +86 -0
  75. stage/domain/custom_board.py +92 -0
  76. stage/domain/discovery.py +94 -0
  77. stage/domain/enums.py +114 -0
  78. stage/domain/events.py +204 -0
  79. stage/domain/filters.py +27 -0
  80. stage/domain/health.py +169 -0
  81. stage/domain/ids.py +48 -0
  82. stage/domain/job.py +47 -0
  83. stage/domain/matching.py +15 -0
  84. stage/domain/priority.py +34 -0
  85. stage/domain/quarantine.py +39 -0
  86. stage/domain/rate_state.py +78 -0
  87. stage/domain/retention.py +20 -0
  88. stage/domain/rotation.py +46 -0
  89. stage/domain/signals.py +12 -0
  90. stage/domain/sync_run.py +35 -0
  91. stage/domain/text.py +113 -0
  92. stage/domain/validator.py +14 -0
  93. stage/domain/visits.py +60 -0
  94. stage/domain/workday.py +38 -0
  95. stage/http/__init__.py +58 -0
  96. stage/http/breaker.py +53 -0
  97. stage/http/cache.py +44 -0
  98. stage/http/client.py +725 -0
  99. stage/http/profiles.py +101 -0
  100. stage/lexicon.py +370 -0
  101. stage/normalize/__init__.py +16 -0
  102. stage/normalize/language.py +47 -0
  103. stage/normalize/location.py +271 -0
  104. stage/normalize/terms.py +153 -0
  105. stage/normalize/urls.py +122 -0
  106. stage/paths.py +86 -0
  107. stage/py.typed +0 -0
  108. stage/services/__init__.py +0 -0
  109. stage/services/canary.py +120 -0
  110. stage/services/coverage.py +231 -0
  111. stage/services/discover.py +747 -0
  112. stage/services/export.py +274 -0
  113. stage/services/health.py +237 -0
  114. stage/services/maintenance.py +225 -0
  115. stage/services/quarantine.py +20 -0
  116. stage/services/query.py +86 -0
  117. stage/services/sync.py +1257 -0
  118. stage/sources/__init__.py +82 -0
  119. stage/sources/_text.py +79 -0
  120. stage/sources/ashby.py +93 -0
  121. stage/sources/bamboohr.py +80 -0
  122. stage/sources/base.py +225 -0
  123. stage/sources/breezy.py +90 -0
  124. stage/sources/collage.py +60 -0
  125. stage/sources/community_feeds.py +142 -0
  126. stage/sources/curated_markdown.py +289 -0
  127. stage/sources/custom_json.py +610 -0
  128. stage/sources/espresso.py +154 -0
  129. stage/sources/feed.py +44 -0
  130. stage/sources/greenhouse.py +104 -0
  131. stage/sources/jobbank.py +147 -0
  132. stage/sources/jobvite.py +133 -0
  133. stage/sources/lever.py +76 -0
  134. stage/sources/oracle_cloud.py +187 -0
  135. stage/sources/platforms.py +609 -0
  136. stage/sources/quebec_emploi.py +146 -0
  137. stage/sources/recruitee.py +96 -0
  138. stage/sources/simplify.py +110 -0
  139. stage/sources/smartrecruiters.py +216 -0
  140. stage/sources/speedyapply.py +200 -0
  141. stage/sources/themuse.py +157 -0
  142. stage/sources/workable.py +83 -0
  143. stage/sources/workday.py +524 -0
  144. stage/sources/zshah.py +99 -0
  145. stage/storage/__init__.py +29 -0
  146. stage/storage/migrations/0001_initial.sql +239 -0
  147. stage/storage/migrations/__init__.py +135 -0
  148. stage/storage/repository.py +213 -0
  149. stage/storage/search.py +28 -0
  150. stage/storage/sqlite_repo.py +1586 -0
  151. stage/storage/writer.py +249 -0
  152. stage/tui/__init__.py +0 -0
  153. stage/tui/app.py +82 -0
  154. stage/tui/help.py +26 -0
  155. stage/tui/safe.py +21 -0
  156. stage/tui/screens/__init__.py +0 -0
  157. stage/tui/screens/boards.py +186 -0
  158. stage/tui/screens/postings.py +509 -0
  159. stage/tui/screens/review.py +209 -0
  160. stage/tui/screens/splash.py +37 -0
  161. stage/tui/screens/stats.py +124 -0
  162. stage/tui/screens/sync.py +194 -0
  163. stage/tui/state.py +160 -0
  164. stage/tui/theme.tcss +205 -0
  165. stage/tui/widgets/__init__.py +0 -0
  166. stage_cli-1.0.0.dist-info/METADATA +379 -0
  167. stage_cli-1.0.0.dist-info/RECORD +170 -0
  168. stage_cli-1.0.0.dist-info/WHEEL +4 -0
  169. stage_cli-1.0.0.dist-info/entry_points.txt +2 -0
  170. stage_cli-1.0.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,1586 @@
1
+ import json
2
+ import sqlite3
3
+ from collections.abc import Callable, Iterator, Mapping, Sequence
4
+ from contextlib import contextmanager
5
+ from dataclasses import replace
6
+ from datetime import UTC, datetime, timedelta
7
+ from pathlib import Path
8
+ from typing import Any
9
+
10
+ from stage.domain import (
11
+ CLOSED_RETENTION_DAYS,
12
+ OPEN_RETENTION_DAYS,
13
+ UNKNOWN_TERM,
14
+ CoverageClassification,
15
+ CoverageDisposition,
16
+ DegreeRequirement,
17
+ HttpValidator,
18
+ IntegrityFinding,
19
+ IntegrityRepair,
20
+ Job,
21
+ JobFilters,
22
+ JobStatus,
23
+ Language,
24
+ LocationBucket,
25
+ PurgeResult,
26
+ QuarantinedJob,
27
+ QuarantineFilters,
28
+ RateState,
29
+ RejectionReason,
30
+ RemoteScope,
31
+ RoleCategory,
32
+ SourceRunStats,
33
+ SourceSignals,
34
+ SourceVisit,
35
+ SyncOutcome,
36
+ SyncRun,
37
+ VolumePoint,
38
+ WorkdayCrawl,
39
+ WorkdayFacet,
40
+ board_of,
41
+ public_https_url,
42
+ source_rank,
43
+ )
44
+ from stage.paths import restrict_permissions
45
+ from stage.storage.migrations import migrate
46
+ from stage.storage.repository import SourceBatch, SourceBatchResult
47
+ from stage.storage.search import FTS_COLUMN_WEIGHTS, match_expression, search_terms
48
+
49
+ _BM25_WEIGHTS = ", ".join(f"{weight:.1f}" for weight in FTS_COLUMN_WEIGHTS)
50
+
51
+ _COMPOSITION_COLUMNS = frozenset(
52
+ {"source", "location", "role", "term", "language", "status", "degree_requirement"}
53
+ )
54
+ _LEGACY_LOCATION_BUCKETS = {"other": "international", "remote": "unknown"}
55
+ _STORED_LOCATION_VALUES = {
56
+ LocationBucket.INTERNATIONAL: ("international", "other"),
57
+ LocationBucket.UNKNOWN: ("unknown", "remote"),
58
+ }
59
+
60
+
61
+ def _location_bucket(value: str) -> LocationBucket:
62
+ return LocationBucket(_LEGACY_LOCATION_BUCKETS.get(value, value))
63
+
64
+
65
+ def _stored_location_values(bucket: LocationBucket) -> tuple[str, ...]:
66
+ return _STORED_LOCATION_VALUES.get(bucket, (bucket.value,))
67
+
68
+
69
+ def fold_company(value: str) -> str:
70
+ from stage.lexicon import fold
71
+
72
+ return fold(value)
73
+
74
+
75
+ _JOB_COLUMNS = (
76
+ "id",
77
+ "source",
78
+ "company",
79
+ "company_fold",
80
+ "title_raw",
81
+ "title_normalized",
82
+ "title_canonical",
83
+ "apply_url_raw",
84
+ "apply_url_canonical",
85
+ "description",
86
+ "location_raw",
87
+ "location",
88
+ "remote_scope",
89
+ "language",
90
+ "term",
91
+ "role",
92
+ "work_auth_flag",
93
+ "degree_requirement",
94
+ "compensation",
95
+ "employment_type",
96
+ "source_category",
97
+ "status",
98
+ "first_seen",
99
+ "last_seen",
100
+ "source_posted_at",
101
+ "duplicate_of",
102
+ )
103
+
104
+ _UPDATE_ON_CONFLICT = (
105
+ "source = excluded.source",
106
+ "company = excluded.company",
107
+ "company_fold = excluded.company_fold",
108
+ "title_raw = excluded.title_raw",
109
+ "title_normalized = excluded.title_normalized",
110
+ "title_canonical = excluded.title_canonical",
111
+ "apply_url_raw = excluded.apply_url_raw",
112
+ "apply_url_canonical = excluded.apply_url_canonical",
113
+ "description = excluded.description",
114
+ "location_raw = excluded.location_raw",
115
+ "location = excluded.location",
116
+ "remote_scope = excluded.remote_scope",
117
+ "language = excluded.language",
118
+ "term = excluded.term",
119
+ "role = excluded.role",
120
+ "work_auth_flag = excluded.work_auth_flag",
121
+ "degree_requirement = excluded.degree_requirement",
122
+ "compensation = excluded.compensation",
123
+ "employment_type = excluded.employment_type",
124
+ "source_category = excluded.source_category",
125
+ "status = excluded.status",
126
+ "last_seen = excluded.last_seen",
127
+ "source_posted_at = excluded.source_posted_at",
128
+ )
129
+
130
+ _UPSERT_SQL = (
131
+ f"INSERT INTO jobs ({', '.join(_JOB_COLUMNS)}) "
132
+ f"VALUES ({', '.join('?' * len(_JOB_COLUMNS))}) "
133
+ f"ON CONFLICT(id) DO UPDATE SET {', '.join(_UPDATE_ON_CONFLICT)}"
134
+ )
135
+
136
+
137
+ _QUARANTINE_COLUMNS = (
138
+ "id",
139
+ "source",
140
+ "company",
141
+ "title_raw",
142
+ "apply_url_raw",
143
+ "location_raw",
144
+ "location",
145
+ "remote_scope",
146
+ "reason",
147
+ "matched_phrase",
148
+ "first_seen",
149
+ "last_seen",
150
+ )
151
+
152
+ _QUARANTINE_UPSERT_SQL = (
153
+ f"INSERT INTO quarantine ({', '.join(_QUARANTINE_COLUMNS)}) "
154
+ f"VALUES ({', '.join('?' * len(_QUARANTINE_COLUMNS))}) "
155
+ "ON CONFLICT(id) DO UPDATE SET "
156
+ "source = excluded.source, company = excluded.company, "
157
+ "title_raw = excluded.title_raw, apply_url_raw = excluded.apply_url_raw, "
158
+ "location_raw = excluded.location_raw, location = excluded.location, "
159
+ "remote_scope = excluded.remote_scope, reason = excluded.reason, "
160
+ "matched_phrase = excluded.matched_phrase, last_seen = excluded.last_seen"
161
+ )
162
+
163
+
164
+ _RATE_STATE_COLUMNS = (
165
+ "bucket",
166
+ "blocked_until",
167
+ "min_interval_override",
168
+ "consecutive_failures",
169
+ "last_failure_at",
170
+ "reason",
171
+ "rotation_cursor",
172
+ "updated_at",
173
+ )
174
+
175
+ _RATE_STATE_UPSERT_SQL = (
176
+ f"INSERT INTO rate_state ({', '.join(_RATE_STATE_COLUMNS)}) "
177
+ f"VALUES ({', '.join('?' * len(_RATE_STATE_COLUMNS))}) "
178
+ "ON CONFLICT(bucket) DO UPDATE SET "
179
+ "blocked_until = NULLIF(MAX(COALESCE(excluded.blocked_until, ''), "
180
+ "COALESCE(rate_state.blocked_until, '')), ''), "
181
+ "reason = CASE WHEN COALESCE(rate_state.blocked_until, '') > "
182
+ "COALESCE(excluded.blocked_until, '') THEN rate_state.reason ELSE excluded.reason END, "
183
+ "min_interval_override = excluded.min_interval_override, "
184
+ "consecutive_failures = excluded.consecutive_failures, "
185
+ "last_failure_at = excluded.last_failure_at, "
186
+ "rotation_cursor = excluded.rotation_cursor, updated_at = excluded.updated_at"
187
+ )
188
+
189
+
190
+ _VISIT_UPSERT_SQL = (
191
+ "INSERT INTO source_visits "
192
+ "(source, board, label, last_attempt_at, last_success_at, consecutive_failures, last_error) "
193
+ "VALUES (?, ?, ?, ?, ?, ?, ?) "
194
+ "ON CONFLICT(source, board) DO UPDATE SET "
195
+ "label = excluded.label, "
196
+ "last_attempt_at = excluded.last_attempt_at, "
197
+ "last_success_at = COALESCE(excluded.last_success_at, source_visits.last_success_at), "
198
+ "consecutive_failures = CASE WHEN excluded.consecutive_failures = 0 THEN 0 "
199
+ "ELSE source_visits.consecutive_failures + 1 END, "
200
+ "last_error = excluded.last_error"
201
+ )
202
+
203
+
204
+ MAX_DETAIL_ATTEMPTS = 3
205
+ _ID_CHUNK = 500
206
+
207
+ _QUEUE_ELIGIBLE = "(d.id IS NULL OR (d.failed = 1 AND d.attempts < ?))"
208
+
209
+
210
+ def _board_glob(board: str) -> str:
211
+ return f"{board}:*"
212
+
213
+
214
+ def _limit_clause(limit: int | None) -> tuple[str, tuple[int, ...]]:
215
+ return ("", ()) if limit is None else (" LIMIT ?", (limit,))
216
+
217
+
218
+ def _to_text(value: datetime) -> str:
219
+ if value.tzinfo is None:
220
+ raise ValueError("naive datetimes are not stored; supply an aware datetime")
221
+ return value.astimezone(UTC).isoformat()
222
+
223
+
224
+ def _from_text(value: str | None) -> datetime | None:
225
+ return datetime.fromisoformat(value) if value else None
226
+
227
+
228
+ def _require_datetime(value: str | None, column: str) -> datetime:
229
+ parsed = _from_text(value)
230
+ if parsed is None:
231
+ raise ValueError(f"column {column!r} is unexpectedly null")
232
+ return parsed
233
+
234
+
235
+ def _classification_from_row(row: sqlite3.Row) -> CoverageClassification:
236
+ return CoverageClassification(
237
+ company=str(row["company"]),
238
+ disposition=CoverageDisposition(str(row["disposition"])),
239
+ note=str(row["note"]),
240
+ checked_on=_require_datetime(row["checked_on"], "checked_on"),
241
+ url=str(row["url"]) if row["url"] is not None else None,
242
+ )
243
+
244
+
245
+ def _classification_params(
246
+ entry: CoverageClassification,
247
+ ) -> tuple[str, str, str, str, str, str | None]:
248
+ company = entry.company.strip()
249
+ company_fold = fold_company(company)
250
+ if not company_fold:
251
+ raise ValueError("classification company must contain letters or numbers")
252
+ note = entry.note.strip()
253
+ if not note:
254
+ raise ValueError("classification note must be a non-empty string")
255
+ if not isinstance(entry.disposition, CoverageDisposition):
256
+ raise ValueError("classification disposition is invalid")
257
+ url = entry.url
258
+ if url is not None:
259
+ url = public_https_url(url)
260
+ if url is None:
261
+ raise ValueError(
262
+ "classification url must be a public https address without credentials"
263
+ )
264
+ return company, company_fold, entry.disposition.value, note, _to_text(entry.checked_on), url
265
+
266
+
267
+ def _row_to_job(row: sqlite3.Row) -> Job:
268
+ remote_scope = row["remote_scope"]
269
+ return Job(
270
+ id=row["id"],
271
+ source=row["source"],
272
+ company=row["company"],
273
+ title_raw=row["title_raw"],
274
+ title_normalized=row["title_normalized"],
275
+ title_canonical=row["title_canonical"],
276
+ apply_url_raw=row["apply_url_raw"],
277
+ apply_url_canonical=row["apply_url_canonical"],
278
+ description=row["description"],
279
+ location_raw=row["location_raw"],
280
+ location=_location_bucket(row["location"]),
281
+ remote_scope=RemoteScope(remote_scope) if remote_scope else None,
282
+ language=Language(row["language"]),
283
+ term=row["term"],
284
+ role=RoleCategory(row["role"]),
285
+ work_auth_flag=bool(row["work_auth_flag"]),
286
+ degree_requirement=DegreeRequirement(row["degree_requirement"]),
287
+ compensation=row["compensation"],
288
+ signals=SourceSignals(
289
+ employment_type=row["employment_type"],
290
+ category=row["source_category"],
291
+ ),
292
+ status=JobStatus(row["status"]),
293
+ first_seen=_require_datetime(row["first_seen"], "first_seen"),
294
+ last_seen=_require_datetime(row["last_seen"], "last_seen"),
295
+ source_posted_at=_from_text(row["source_posted_at"]),
296
+ duplicate_of=row["duplicate_of"],
297
+ )
298
+
299
+
300
+ def _job_to_params(job: Job) -> tuple[Any, ...]:
301
+ return (
302
+ job.id,
303
+ job.source,
304
+ job.company,
305
+ fold_company(job.company),
306
+ job.title_raw,
307
+ job.title_normalized,
308
+ job.title_canonical,
309
+ job.apply_url_raw,
310
+ job.apply_url_canonical,
311
+ job.description,
312
+ job.location_raw,
313
+ job.location.value,
314
+ job.remote_scope.value if job.remote_scope else None,
315
+ job.language.value,
316
+ job.term,
317
+ job.role.value,
318
+ int(job.work_auth_flag),
319
+ job.degree_requirement.value,
320
+ job.compensation,
321
+ job.signals.employment_type,
322
+ job.signals.category,
323
+ job.status.value,
324
+ _to_text(job.first_seen),
325
+ _to_text(job.last_seen),
326
+ _to_text(job.source_posted_at) if job.source_posted_at else None,
327
+ job.duplicate_of,
328
+ )
329
+
330
+
331
+ def _row_to_quarantined(row: sqlite3.Row) -> QuarantinedJob:
332
+ remote_scope = row["remote_scope"]
333
+ return QuarantinedJob(
334
+ id=row["id"],
335
+ source=row["source"],
336
+ company=row["company"],
337
+ title_raw=row["title_raw"],
338
+ reason=RejectionReason(row["reason"]),
339
+ first_seen=_require_datetime(row["first_seen"], "first_seen"),
340
+ last_seen=_require_datetime(row["last_seen"], "last_seen"),
341
+ apply_url_raw=row["apply_url_raw"],
342
+ location_raw=row["location_raw"],
343
+ location=_location_bucket(row["location"]),
344
+ remote_scope=RemoteScope(remote_scope) if remote_scope else None,
345
+ matched_phrase=row["matched_phrase"],
346
+ )
347
+
348
+
349
+ def _row_to_rate_state(row: sqlite3.Row) -> RateState:
350
+ return RateState(
351
+ bucket=str(row["bucket"]),
352
+ updated_at=_require_datetime(row["updated_at"], "updated_at"),
353
+ blocked_until=_from_text(row["blocked_until"]),
354
+ min_interval_override=(
355
+ float(row["min_interval_override"])
356
+ if row["min_interval_override"] is not None
357
+ else None
358
+ ),
359
+ consecutive_failures=int(row["consecutive_failures"]),
360
+ last_failure_at=_from_text(row["last_failure_at"]),
361
+ reason=str(row["reason"]),
362
+ rotation_cursor=str(row["rotation_cursor"]),
363
+ )
364
+
365
+
366
+ def _rate_state_to_params(state: RateState) -> tuple[Any, ...]:
367
+ return (
368
+ state.bucket,
369
+ _to_text(state.blocked_until) if state.blocked_until else None,
370
+ state.min_interval_override,
371
+ state.consecutive_failures,
372
+ _to_text(state.last_failure_at) if state.last_failure_at else None,
373
+ state.reason,
374
+ state.rotation_cursor,
375
+ _to_text(state.updated_at),
376
+ )
377
+
378
+
379
+ def _quarantined_to_params(entry: QuarantinedJob) -> tuple[Any, ...]:
380
+ return (
381
+ entry.id,
382
+ entry.source,
383
+ entry.company,
384
+ entry.title_raw,
385
+ entry.apply_url_raw,
386
+ entry.location_raw,
387
+ entry.location.value,
388
+ entry.remote_scope.value if entry.remote_scope else None,
389
+ entry.reason.value,
390
+ entry.matched_phrase,
391
+ _to_text(entry.first_seen),
392
+ _to_text(entry.last_seen),
393
+ )
394
+
395
+
396
+ MAX_CHAIN_REPAIR_PASSES = 8
397
+
398
+
399
+ class SqliteRepository:
400
+ def __init__(self, connection: sqlite3.Connection) -> None:
401
+ self._conn = connection
402
+ connection.create_function("stage_board_of", 2, board_of, deterministic=True)
403
+
404
+ @classmethod
405
+ def connect(cls, db_path: Path) -> "SqliteRepository":
406
+ is_new = not db_path.exists()
407
+ db_path.parent.mkdir(parents=True, exist_ok=True)
408
+ conn = sqlite3.connect(db_path, isolation_level=None)
409
+ try:
410
+ conn.row_factory = sqlite3.Row
411
+ conn.execute("PRAGMA journal_mode = WAL")
412
+ conn.execute("PRAGMA synchronous = NORMAL")
413
+ conn.execute("PRAGMA foreign_keys = ON")
414
+ conn.execute("PRAGMA busy_timeout = 5000")
415
+ migrate(conn, db_path)
416
+ if is_new:
417
+ restrict_permissions(db_path)
418
+ except Exception:
419
+ conn.close()
420
+ raise
421
+ return cls(conn)
422
+
423
+ def close(self) -> None:
424
+ self._conn.close()
425
+
426
+ @contextmanager
427
+ def _transaction(self) -> Iterator[sqlite3.Connection]:
428
+ self._conn.execute("BEGIN IMMEDIATE")
429
+ try:
430
+ yield self._conn
431
+ except Exception:
432
+ self._conn.rollback()
433
+ raise
434
+ self._conn.commit()
435
+
436
+ def _existing_ids(self, job_ids: Sequence[str]) -> set[str]:
437
+ existing: set[str] = set()
438
+ chunk_size = 400
439
+ for start in range(0, len(job_ids), chunk_size):
440
+ chunk = job_ids[start : start + chunk_size]
441
+ placeholders = ", ".join("?" * len(chunk))
442
+ rows = self._conn.execute(
443
+ f"SELECT id FROM jobs WHERE id IN ({placeholders})", tuple(chunk)
444
+ ).fetchall()
445
+ existing.update(str(row["id"]) for row in rows)
446
+ return existing
447
+
448
+ def _chunked(self, values: Sequence[str]) -> Iterator[Sequence[str]]:
449
+ chunk_size = 400
450
+ for start in range(0, len(values), chunk_size):
451
+ yield values[start : start + chunk_size]
452
+
453
+ def _preserved_first_seen(self, job_ids: Sequence[str]) -> dict[str, datetime]:
454
+ found: dict[str, datetime] = {}
455
+ for table in ("jobs", "quarantine", "tombstones"):
456
+ for chunk in self._chunked(job_ids):
457
+ placeholders = ", ".join("?" * len(chunk))
458
+ rows = self._conn.execute(
459
+ f"SELECT id, first_seen FROM {table} WHERE id IN ({placeholders})",
460
+ tuple(chunk),
461
+ ).fetchall()
462
+ for row in rows:
463
+ found.setdefault(
464
+ str(row["id"]), _require_datetime(row["first_seen"], "first_seen")
465
+ )
466
+ return found
467
+
468
+ def _duplicate_candidates(self, jobs: Sequence[Job]) -> list[Job]:
469
+ companies = sorted({folded for job in jobs if (folded := fold_company(job.company))})
470
+ urls = sorted({job.apply_url_canonical for job in jobs if job.apply_url_canonical})
471
+ found: dict[str, Job] = {}
472
+ for column, values in (("company_fold", companies), ("apply_url_canonical", urls)):
473
+ for start in range(0, len(values), 400):
474
+ chunk = values[start : start + 400]
475
+ if not chunk:
476
+ continue
477
+ placeholders = ", ".join("?" * len(chunk))
478
+ rows = self._conn.execute(
479
+ f"SELECT * FROM jobs WHERE {column} IN ({placeholders})", tuple(chunk)
480
+ ).fetchall()
481
+ for row in rows:
482
+ found[str(row["id"])] = _row_to_job(row)
483
+ return list(found.values())
484
+
485
+ def _apply_duplicate_links(self, batch: SourceBatch, jobs: Sequence[Job]) -> int:
486
+ if batch.resolve_duplicates is None or not jobs:
487
+ return 0
488
+ candidates = self._duplicate_candidates(jobs)
489
+ links = batch.resolve_duplicates(jobs, candidates)
490
+ if not links:
491
+ self._conn.execute(
492
+ "UPDATE jobs SET duplicate_of = NULL WHERE id IN "
493
+ f"({', '.join('?' * len(jobs))}) AND duplicate_of IS NOT NULL",
494
+ tuple(job.id for job in jobs),
495
+ )
496
+ return 0
497
+ pairs = [
498
+ (duplicate, canonical)
499
+ for duplicate, canonical in (
500
+ (str(link.duplicate_id), str(link.canonical_id)) # type: ignore[attr-defined]
501
+ for link in links
502
+ )
503
+ if board_of(duplicate, duplicate) != board_of(canonical, canonical)
504
+ ]
505
+ if not pairs:
506
+ return 0
507
+ duplicates = {duplicate for duplicate, _ in pairs}
508
+ self._clear_stale_links(jobs, duplicates)
509
+ canonicals = {canonical for _, canonical in pairs}
510
+ self._conn.executemany(
511
+ "UPDATE jobs SET duplicate_of = ? WHERE id = ?",
512
+ [(canonical, duplicate) for duplicate, canonical in pairs],
513
+ )
514
+ placeholders = ", ".join("?" * len(canonicals))
515
+ self._conn.execute(
516
+ f"UPDATE jobs SET duplicate_of = NULL WHERE id IN ({placeholders})",
517
+ tuple(sorted(canonicals)),
518
+ )
519
+ for duplicate, canonical in pairs:
520
+ self._conn.execute(
521
+ "UPDATE jobs SET duplicate_of = ? WHERE duplicate_of = ? AND id != ?",
522
+ (canonical, duplicate, canonical),
523
+ )
524
+ return len(duplicates)
525
+
526
+ def _clear_stale_links(self, jobs: Sequence[Job], linked: set[str]) -> None:
527
+ stale = [job.id for job in jobs if job.id not in linked]
528
+ for chunk in self._chunked(stale):
529
+ placeholders = ", ".join("?" * len(chunk))
530
+ self._conn.execute(
531
+ f"UPDATE jobs SET duplicate_of = NULL WHERE id IN ({placeholders}) "
532
+ "AND duplicate_of IS NOT NULL",
533
+ tuple(chunk),
534
+ )
535
+
536
+ def _reseat_inverted_links(self) -> int:
537
+ rows = self._conn.execute(
538
+ "SELECT d.id AS duplicate, d.source AS duplicate_source, c.id AS canonical, "
539
+ "c.source AS canonical_source FROM jobs d JOIN jobs c ON d.duplicate_of = c.id"
540
+ ).fetchall()
541
+ swapped = 0
542
+ for row in rows:
543
+ duplicate, canonical = str(row["duplicate"]), str(row["canonical"])
544
+ if source_rank(str(row["duplicate_source"]), duplicate) >= source_rank(
545
+ str(row["canonical_source"]), canonical
546
+ ):
547
+ continue
548
+ self._conn.execute(
549
+ "UPDATE jobs SET duplicate_of = ? WHERE duplicate_of = ? AND id != ?",
550
+ (duplicate, canonical, duplicate),
551
+ )
552
+ self._conn.execute("UPDATE jobs SET duplicate_of = NULL WHERE id = ?", (duplicate,))
553
+ self._conn.execute(
554
+ "UPDATE jobs SET duplicate_of = ? WHERE id = ?", (duplicate, canonical)
555
+ )
556
+ swapped += 1
557
+ return swapped
558
+
559
+ def _promote_orphaned_duplicates(self, purged: Sequence[str]) -> int:
560
+ doomed = set(purged)
561
+ clusters: dict[str, list[tuple[int, str]]] = {}
562
+ for chunk in self._chunked(purged):
563
+ placeholders = ", ".join("?" * len(chunk))
564
+ rows = self._conn.execute(
565
+ f"SELECT id, source, duplicate_of FROM jobs WHERE duplicate_of IN ({placeholders})",
566
+ tuple(chunk),
567
+ ).fetchall()
568
+ for row in rows:
569
+ if str(row["id"]) in doomed:
570
+ continue
571
+ clusters.setdefault(str(row["duplicate_of"]), []).append(
572
+ source_rank(str(row["source"]), str(row["id"]))
573
+ )
574
+ for row in self._conn.execute(
575
+ "SELECT a.id, a.source, a.duplicate_of FROM jobs a "
576
+ "LEFT JOIN jobs b ON a.duplicate_of = b.id "
577
+ "WHERE a.duplicate_of IS NOT NULL AND b.id IS NULL"
578
+ ).fetchall():
579
+ if str(row["id"]) in doomed:
580
+ continue
581
+ clusters.setdefault(str(row["duplicate_of"]), []).append(
582
+ source_rank(str(row["source"]), str(row["id"]))
583
+ )
584
+
585
+ promoted = 0
586
+ for members in clusters.values():
587
+ unique = sorted(set(members))
588
+ if not unique:
589
+ continue
590
+ winner = unique[0][1]
591
+ self._conn.execute("UPDATE jobs SET duplicate_of = NULL WHERE id = ?", (winner,))
592
+ for _, member in unique[1:]:
593
+ self._conn.execute(
594
+ "UPDATE jobs SET duplicate_of = ? WHERE id = ?", (winner, member)
595
+ )
596
+ promoted += 1
597
+ return promoted
598
+
599
+ def purge(
600
+ self,
601
+ now: datetime,
602
+ *,
603
+ open_days: int = OPEN_RETENTION_DAYS,
604
+ closed_days: int = CLOSED_RETENTION_DAYS,
605
+ ) -> PurgeResult:
606
+ open_cutoff = _to_text(now - timedelta(days=open_days))
607
+ closed_cutoff = _to_text(now - timedelta(days=closed_days))
608
+ with self._transaction() as conn:
609
+ rows = conn.execute(
610
+ "SELECT id, source, first_seen FROM jobs WHERE "
611
+ "(status = ? AND first_seen < ?) OR (status = ? AND first_seen < ?)",
612
+ (JobStatus.OPEN.value, open_cutoff, JobStatus.CLOSED.value, closed_cutoff),
613
+ ).fetchall()
614
+ ids = [str(row["id"]) for row in rows]
615
+ reseated = self._reseat_inverted_links()
616
+ promoted = self._promote_orphaned_duplicates(ids) + reseated
617
+ if not ids:
618
+ return PurgeResult(promoted=promoted)
619
+ conn.executemany(
620
+ "INSERT INTO tombstones (id, source, first_seen, purged_at) "
621
+ "VALUES (?, ?, ?, ?) ON CONFLICT(id) DO UPDATE SET "
622
+ "purged_at = excluded.purged_at",
623
+ [
624
+ (
625
+ str(row["id"]),
626
+ str(row["source"]),
627
+ str(row["first_seen"]),
628
+ _to_text(now),
629
+ )
630
+ for row in rows
631
+ ],
632
+ )
633
+ for chunk in self._chunked(ids):
634
+ placeholders = ", ".join("?" * len(chunk))
635
+ conn.execute(f"DELETE FROM jobs WHERE id IN ({placeholders})", tuple(chunk))
636
+ return PurgeResult(purged=len(ids), tombstoned=len(ids), promoted=promoted)
637
+
638
+ def preview_purge(
639
+ self,
640
+ now: datetime,
641
+ *,
642
+ open_days: int = OPEN_RETENTION_DAYS,
643
+ closed_days: int = CLOSED_RETENTION_DAYS,
644
+ ) -> PurgeResult:
645
+ open_cutoff = _to_text(now - timedelta(days=open_days))
646
+ closed_cutoff = _to_text(now - timedelta(days=closed_days))
647
+ row = self._conn.execute(
648
+ "SELECT COUNT(*) AS total FROM jobs WHERE "
649
+ "(status = ? AND first_seen < ?) OR (status = ? AND first_seen < ?)",
650
+ (JobStatus.OPEN.value, open_cutoff, JobStatus.CLOSED.value, closed_cutoff),
651
+ ).fetchone()
652
+ count = int(row["total"])
653
+ return PurgeResult(purged=count, tombstoned=count)
654
+
655
+ def tombstone_count(self) -> int:
656
+ row = self._conn.execute("SELECT COUNT(*) AS total FROM tombstones").fetchone()
657
+ return int(row["total"])
658
+
659
+ def count_duplicates(self) -> int:
660
+ row = self._conn.execute(
661
+ "SELECT COUNT(*) AS total FROM jobs WHERE duplicate_of IS NOT NULL"
662
+ ).fetchone()
663
+ return int(row["total"])
664
+
665
+ def apply_source_batch(self, batch: SourceBatch) -> SourceBatchResult:
666
+ stamp = _to_text(batch.run_started_at)
667
+ quarantined_ids = [entry.id for entry in batch.quarantined]
668
+
669
+ with self._transaction() as conn:
670
+ known = self._preserved_first_seen([job.id for job in batch.jobs] + quarantined_ids)
671
+ jobs = tuple(
672
+ replace(job, first_seen=known[job.id]) if job.id in known else job
673
+ for job in batch.jobs
674
+ )
675
+ quarantined = tuple(
676
+ replace(entry, first_seen=known[entry.id]) if entry.id in known else entry
677
+ for entry in batch.quarantined
678
+ )
679
+ existing = self._existing_ids([job.id for job in jobs])
680
+ conn.executemany(_UPSERT_SQL, [_job_to_params(job) for job in jobs])
681
+
682
+ if quarantined:
683
+ conn.executemany(
684
+ _QUARANTINE_UPSERT_SQL,
685
+ [_quarantined_to_params(entry) for entry in quarantined],
686
+ )
687
+ self._promote_orphaned_duplicates(quarantined_ids)
688
+ for chunk in self._chunked(quarantined_ids):
689
+ placeholders = ", ".join("?" * len(chunk))
690
+ conn.execute(f"DELETE FROM jobs WHERE id IN ({placeholders})", tuple(chunk))
691
+ released = {job.id for job in jobs} & set(known)
692
+ if released:
693
+ for chunk in self._chunked(sorted(released)):
694
+ placeholders = ", ".join("?" * len(chunk))
695
+ conn.execute(
696
+ f"DELETE FROM quarantine WHERE id IN ({placeholders})", tuple(chunk)
697
+ )
698
+
699
+ for crawl in batch.workday_crawls:
700
+ if crawl.reset or crawl.discard:
701
+ conn.execute("DELETE FROM workday_crawls WHERE board = ?", (crawl.board,))
702
+ if crawl.discard:
703
+ continue
704
+ conn.execute(
705
+ "INSERT INTO workday_crawls "
706
+ "(board, next_offset, total, facet_parameter, facet_ids) "
707
+ "VALUES (?, ?, ?, ?, ?) ON CONFLICT(board) DO UPDATE SET "
708
+ "next_offset = excluded.next_offset, total = excluded.total, "
709
+ "facet_parameter = excluded.facet_parameter, facet_ids = excluded.facet_ids",
710
+ (
711
+ crawl.board,
712
+ crawl.next_offset,
713
+ crawl.total,
714
+ crawl.facet_parameter,
715
+ json.dumps(crawl.facet_ids),
716
+ ),
717
+ )
718
+ if crawl.seen_ids:
719
+ conn.executemany(
720
+ "INSERT OR IGNORE INTO workday_crawl_seen (board, id) VALUES (?, ?)",
721
+ [(crawl.board, job_id) for job_id in crawl.seen_ids],
722
+ )
723
+ if crawl.complete:
724
+ conn.execute(
725
+ "UPDATE jobs SET last_seen = ? WHERE source = 'workday' "
726
+ "AND status = ? AND id IN ("
727
+ "SELECT id FROM workday_crawl_seen WHERE board = ?)",
728
+ (stamp, JobStatus.OPEN.value, crawl.board),
729
+ )
730
+ conn.execute("DELETE FROM workday_crawls WHERE board = ?", (crawl.board,))
731
+
732
+ touched = 0
733
+ for board in batch.unchanged_boards:
734
+ cursor = conn.execute(
735
+ "UPDATE jobs SET last_seen = ? WHERE source = ? AND id GLOB ? AND status = ?",
736
+ (stamp, batch.source, _board_glob(board), JobStatus.OPEN.value),
737
+ )
738
+ touched += cursor.rowcount
739
+
740
+ closed = 0
741
+ if batch.closes_whole_source:
742
+ cursor = conn.execute(
743
+ "UPDATE jobs SET status = ? WHERE source = ? AND status = ? AND last_seen < ?",
744
+ (JobStatus.CLOSED.value, batch.source, JobStatus.OPEN.value, stamp),
745
+ )
746
+ closed = cursor.rowcount
747
+ else:
748
+ for board in batch.closable_boards:
749
+ cursor = conn.execute(
750
+ "UPDATE jobs SET status = ? WHERE source = ? AND id GLOB ? "
751
+ "AND status = ? AND last_seen < ?",
752
+ (
753
+ JobStatus.CLOSED.value,
754
+ batch.source,
755
+ _board_glob(board),
756
+ JobStatus.OPEN.value,
757
+ stamp,
758
+ ),
759
+ )
760
+ closed += cursor.rowcount
761
+
762
+ linked = self._apply_duplicate_links(batch, jobs)
763
+
764
+ if batch.visits:
765
+ conn.executemany(
766
+ _VISIT_UPSERT_SQL,
767
+ [
768
+ (
769
+ batch.source,
770
+ visit.board,
771
+ visit.label,
772
+ stamp,
773
+ stamp if visit.succeeded else None,
774
+ 0 if visit.succeeded else 1,
775
+ "" if visit.succeeded else visit.error,
776
+ )
777
+ for visit in batch.visits
778
+ ],
779
+ )
780
+
781
+ if batch.detail_fetches:
782
+ conn.executemany(
783
+ "INSERT INTO detail_fetches (id, source, fetched_at, resolved, "
784
+ "attempts, failed) VALUES (?, ?, ?, ?, 1, ?) "
785
+ "ON CONFLICT(id) DO UPDATE SET "
786
+ "fetched_at = excluded.fetched_at, resolved = excluded.resolved, "
787
+ "attempts = detail_fetches.attempts + 1, failed = excluded.failed",
788
+ [
789
+ (entry.id, batch.source, stamp, int(entry.resolved), int(entry.failed))
790
+ for entry in batch.detail_fetches
791
+ ],
792
+ )
793
+
794
+ if batch.forgotten_facets:
795
+ conn.executemany(
796
+ "DELETE FROM workday_facets WHERE tenant = ? AND site = ?",
797
+ [(facet.tenant, facet.site) for facet in batch.forgotten_facets],
798
+ )
799
+
800
+ if batch.workday_facets:
801
+ conn.executemany(
802
+ "INSERT INTO workday_facets "
803
+ "(tenant, site, parameter, facet_id, descriptor, resolved_at) "
804
+ "VALUES (?, ?, ?, ?, ?, ?) ON CONFLICT(tenant, site) DO UPDATE SET "
805
+ "parameter = excluded.parameter, facet_id = excluded.facet_id, "
806
+ "descriptor = excluded.descriptor, resolved_at = excluded.resolved_at",
807
+ [
808
+ (
809
+ facet.tenant,
810
+ facet.site,
811
+ facet.parameter,
812
+ " ".join(facet.facet_ids),
813
+ facet.descriptor,
814
+ _to_text(facet.resolved_at or batch.run_started_at),
815
+ )
816
+ for facet in batch.workday_facets
817
+ if not facet.pinned
818
+ ],
819
+ )
820
+
821
+ if batch.rate_state:
822
+ conn.executemany(
823
+ _RATE_STATE_UPSERT_SQL,
824
+ [_rate_state_to_params(state) for state in batch.rate_state],
825
+ )
826
+
827
+ if batch.validators:
828
+ conn.executemany(
829
+ "INSERT INTO http_cache (url, source, etag, last_modified, fetched_at) "
830
+ "VALUES (?, ?, ?, ?, ?) ON CONFLICT(url) DO UPDATE SET "
831
+ "source = excluded.source, etag = excluded.etag, "
832
+ "last_modified = excluded.last_modified, fetched_at = excluded.fetched_at",
833
+ [
834
+ (
835
+ validator.url,
836
+ batch.source,
837
+ validator.etag,
838
+ validator.last_modified,
839
+ _to_text(validator.fetched_at or batch.run_started_at),
840
+ )
841
+ for validator in batch.validators
842
+ ],
843
+ )
844
+
845
+ updated = sum(1 for job in jobs if job.id in existing)
846
+ stored = conn.execute(
847
+ "SELECT COUNT(*) AS total FROM jobs WHERE source = ? AND status = ?",
848
+ (batch.source, JobStatus.OPEN.value),
849
+ ).fetchone()
850
+ return SourceBatchResult(
851
+ source=batch.source,
852
+ fetched=len(jobs) + len(batch.quarantined),
853
+ added=len(jobs) - updated,
854
+ updated=updated,
855
+ closed=closed,
856
+ touched=touched,
857
+ quarantined=len(batch.quarantined),
858
+ duplicates=linked,
859
+ stored=int(stored["total"]),
860
+ )
861
+
862
+ def load_validators(self, source: str) -> Mapping[str, HttpValidator]:
863
+ rows = self._conn.execute(
864
+ "SELECT url, etag, last_modified, fetched_at FROM http_cache WHERE source = ?",
865
+ (source,),
866
+ ).fetchall()
867
+ return {
868
+ str(row["url"]): HttpValidator(
869
+ url=str(row["url"]),
870
+ etag=row["etag"],
871
+ last_modified=row["last_modified"],
872
+ fetched_at=_from_text(row["fetched_at"]),
873
+ )
874
+ for row in rows
875
+ }
876
+
877
+ def clear_validators(self, source: str | None = None) -> int:
878
+ with self._transaction() as conn:
879
+ if source is None:
880
+ cursor = conn.execute("DELETE FROM http_cache")
881
+ else:
882
+ cursor = conn.execute("DELETE FROM http_cache WHERE source = ?", (source,))
883
+ return cursor.rowcount if cursor.rowcount > 0 else 0
884
+
885
+ def cached_url_count(self) -> int:
886
+ row = self._conn.execute("SELECT COUNT(*) AS total FROM http_cache").fetchone()
887
+ return int(row["total"])
888
+
889
+ def clear_rate_state(self, bucket: str | None = None) -> int:
890
+ assignments = (
891
+ "blocked_until = NULL, min_interval_override = NULL, "
892
+ "consecutive_failures = 0, last_failure_at = NULL, reason = ''"
893
+ )
894
+ with self._transaction() as conn:
895
+ if bucket is None:
896
+ cursor = conn.execute(f"UPDATE rate_state SET {assignments}")
897
+ else:
898
+ cursor = conn.execute(
899
+ f"UPDATE rate_state SET {assignments} WHERE bucket = ?", (bucket,)
900
+ )
901
+ return int(cursor.rowcount)
902
+
903
+ def load_rate_state(self) -> Mapping[str, RateState]:
904
+ rows = self._conn.execute(
905
+ "SELECT bucket, blocked_until, min_interval_override, consecutive_failures, "
906
+ "last_failure_at, reason, rotation_cursor, updated_at FROM rate_state"
907
+ ).fetchall()
908
+ return {str(row["bucket"]): _row_to_rate_state(row) for row in rows}
909
+
910
+ def stale_members(self, source: str, before: datetime) -> list[SourceVisit]:
911
+ rows = self._conn.execute(
912
+ "SELECT source, board, label, last_attempt_at, last_success_at, "
913
+ "consecutive_failures, last_error FROM source_visits WHERE source = ? "
914
+ "AND (last_success_at IS NULL OR last_success_at < ?) "
915
+ "ORDER BY last_success_at IS NOT NULL, last_success_at, board",
916
+ (source, _to_text(before)),
917
+ ).fetchall()
918
+ return [
919
+ SourceVisit(
920
+ source=str(row["source"]),
921
+ board=str(row["board"]),
922
+ label=str(row["label"]),
923
+ last_attempt_at=_require_datetime(row["last_attempt_at"], "last_attempt_at"),
924
+ last_success_at=_from_text(row["last_success_at"]),
925
+ consecutive_failures=int(row["consecutive_failures"]),
926
+ last_error=str(row["last_error"]),
927
+ )
928
+ for row in rows
929
+ ]
930
+
931
+ def detail_queue(self, source: str, limit: int) -> list[str]:
932
+ rows = self._conn.execute(
933
+ f"SELECT j.id FROM jobs j LEFT JOIN detail_fetches d ON d.id = j.id "
934
+ f"WHERE j.source = ? AND j.description = '' AND {_QUEUE_ELIGIBLE} "
935
+ f"AND (j.term = ? OR j.role = ?) "
936
+ f"ORDER BY j.first_seen DESC LIMIT ?",
937
+ (source, MAX_DETAIL_ATTEMPTS, UNKNOWN_TERM, RoleCategory.UNKNOWN.value, limit),
938
+ ).fetchall()
939
+ return [str(row["id"]) for row in rows]
940
+
941
+ def detail_queue_size(self, source: str) -> int:
942
+ row = self._conn.execute(
943
+ f"SELECT COUNT(*) AS total FROM jobs j LEFT JOIN detail_fetches d ON d.id = j.id "
944
+ f"WHERE j.source = ? AND j.description = '' AND {_QUEUE_ELIGIBLE} "
945
+ f"AND (j.term = ? OR j.role = ?)",
946
+ (source, MAX_DETAIL_ATTEMPTS, UNKNOWN_TERM, RoleCategory.UNKNOWN.value),
947
+ ).fetchone()
948
+ return int(row["total"])
949
+
950
+ def load_workday_facets(self) -> Mapping[tuple[str, str], WorkdayFacet]:
951
+ rows = self._conn.execute(
952
+ "SELECT tenant, site, parameter, facet_id, descriptor, resolved_at FROM workday_facets"
953
+ ).fetchall()
954
+ return {
955
+ (str(row["tenant"]), str(row["site"])): WorkdayFacet(
956
+ tenant=str(row["tenant"]),
957
+ site=str(row["site"]),
958
+ parameter=str(row["parameter"]),
959
+ facet_ids=tuple(str(row["facet_id"]).split()),
960
+ descriptor=str(row["descriptor"]),
961
+ resolved_at=_from_text(row["resolved_at"]),
962
+ )
963
+ for row in rows
964
+ }
965
+
966
+ def load_workday_crawls(self) -> Mapping[str, WorkdayCrawl]:
967
+ rows = self._conn.execute(
968
+ "SELECT board, next_offset, total, facet_parameter, facet_ids FROM workday_crawls"
969
+ ).fetchall()
970
+ return {
971
+ str(row["board"]): WorkdayCrawl(
972
+ board=str(row["board"]),
973
+ next_offset=int(row["next_offset"]),
974
+ total=int(row["total"]) if row["total"] is not None else None,
975
+ facet_parameter=str(row["facet_parameter"]),
976
+ facet_ids=tuple(json.loads(str(row["facet_ids"]))),
977
+ )
978
+ for row in rows
979
+ }
980
+
981
+ def _quarantine_where(self, filters: QuarantineFilters) -> tuple[str, list[Any]]:
982
+ clauses: list[str] = []
983
+ params: list[Any] = []
984
+ if filters.reason is not None:
985
+ clauses.append("reason = ?")
986
+ params.append(filters.reason.value)
987
+ if filters.source is not None:
988
+ clauses.append("source = ?")
989
+ params.append(filters.source)
990
+ if filters.company is not None:
991
+ clauses.append("company = ?")
992
+ params.append(filters.company)
993
+ where = f" WHERE {' AND '.join(clauses)}" if clauses else ""
994
+ return where, params
995
+
996
+ def list_quarantined(self, filters: QuarantineFilters) -> list[QuarantinedJob]:
997
+ where, params = self._quarantine_where(filters)
998
+ limit, limit_params = _limit_clause(filters.limit)
999
+ rows = self._conn.execute(
1000
+ f"SELECT * FROM quarantine{where} ORDER BY company COLLATE NOCASE, "
1001
+ f"first_seen DESC, title_raw COLLATE NOCASE{limit}",
1002
+ (*params, *limit_params),
1003
+ ).fetchall()
1004
+ return [_row_to_quarantined(row) for row in rows]
1005
+
1006
+ def count_quarantined(self, filters: QuarantineFilters) -> int:
1007
+ where, params = self._quarantine_where(filters)
1008
+ row = self._conn.execute(
1009
+ f"SELECT COUNT(*) AS total FROM quarantine{where}", tuple(params)
1010
+ ).fetchone()
1011
+ return int(row["total"])
1012
+
1013
+ def relabel_quarantine(self, entries: Sequence[QuarantinedJob]) -> int:
1014
+ if not entries:
1015
+ return 0
1016
+ relabelled = 0
1017
+ with self._transaction() as conn:
1018
+ for entry in entries:
1019
+ cursor = conn.execute(
1020
+ "UPDATE quarantine SET reason = ?, matched_phrase = ? "
1021
+ "WHERE id = ? AND (reason <> ? OR matched_phrase <> ?)",
1022
+ (
1023
+ entry.reason.value,
1024
+ entry.matched_phrase,
1025
+ entry.id,
1026
+ entry.reason.value,
1027
+ entry.matched_phrase,
1028
+ ),
1029
+ )
1030
+ relabelled += int(cursor.rowcount or 0)
1031
+ return relabelled
1032
+
1033
+ def refresh_quarantine_locations(self, resolve: Callable[[str], tuple[str, str | None]]) -> int:
1034
+ rows = self._conn.execute(
1035
+ "SELECT id, location_raw, location, remote_scope FROM quarantine "
1036
+ "WHERE location_raw IS NOT NULL AND location_raw <> ''"
1037
+ ).fetchall()
1038
+ updates = []
1039
+ for row in rows:
1040
+ bucket, scope = resolve(str(row["location_raw"]))
1041
+ if bucket == str(row["location"]) and scope == row["remote_scope"]:
1042
+ continue
1043
+ updates.append((bucket, scope, str(row["id"])))
1044
+ if not updates:
1045
+ return 0
1046
+ with self._transaction() as conn:
1047
+ conn.executemany(
1048
+ "UPDATE quarantine SET location = ?, remote_scope = ? WHERE id = ?", updates
1049
+ )
1050
+ return len(updates)
1051
+
1052
+ def quarantine_reason_counts(self) -> dict[str, int]:
1053
+ rows = self._conn.execute(
1054
+ "SELECT reason, COUNT(*) AS total FROM quarantine GROUP BY reason ORDER BY total DESC"
1055
+ ).fetchall()
1056
+ return {str(row["reason"]): int(row["total"]) for row in rows}
1057
+
1058
+ def _where(self, filters: JobFilters) -> tuple[str, list[Any]]:
1059
+ clauses, params = self._clauses(filters)
1060
+ return f" WHERE {' AND '.join(clauses)}", params
1061
+
1062
+ def _clauses(self, filters: JobFilters) -> tuple[list[str], list[Any]]:
1063
+ clauses: list[str] = ["jobs.duplicate_of IS NULL"]
1064
+ params: list[Any] = []
1065
+ if filters.status is not None:
1066
+ clauses.append("jobs.status = ?")
1067
+ params.append(filters.status.value)
1068
+ if filters.location is not None:
1069
+ locations = _stored_location_values(filters.location)
1070
+ clauses.append(f"jobs.location IN ({', '.join('?' * len(locations))})")
1071
+ params.extend(locations)
1072
+ if filters.term is not None:
1073
+ clauses.append("jobs.term = ?")
1074
+ params.append(filters.term)
1075
+ if filters.degree is not None:
1076
+ clauses.append("jobs.degree_requirement = ?")
1077
+ params.append(filters.degree.value)
1078
+ if filters.role is not None:
1079
+ clauses.append("jobs.role = ?")
1080
+ params.append(filters.role.value)
1081
+ if filters.language is not None:
1082
+ if filters.language in (Language.EN, Language.FR):
1083
+ clauses.append("jobs.language IN (?, ?)")
1084
+ params.extend((filters.language.value, Language.BILINGUAL.value))
1085
+ else:
1086
+ clauses.append("jobs.language = ?")
1087
+ params.append(filters.language.value)
1088
+ if filters.source is not None:
1089
+ clauses.append("jobs.source = ?")
1090
+ params.append(filters.source)
1091
+ if filters.company is not None:
1092
+ clauses.append("jobs.company = ?")
1093
+ params.append(filters.company)
1094
+ if filters.first_seen_after is not None:
1095
+ clauses.append("jobs.first_seen >= ?")
1096
+ params.append(_to_text(filters.first_seen_after))
1097
+ return clauses, params
1098
+
1099
+ def list_jobs(self, filters: JobFilters) -> list[Job]:
1100
+ where, params = self._where(filters)
1101
+ limit, limit_params = _limit_clause(filters.limit)
1102
+ rows = self._conn.execute(
1103
+ f"SELECT * FROM jobs{where} ORDER BY first_seen DESC, id ASC{limit}",
1104
+ (*params, *limit_params),
1105
+ ).fetchall()
1106
+ return [_row_to_job(row) for row in rows]
1107
+
1108
+ def count_jobs(self, filters: JobFilters) -> int:
1109
+ where, params = self._where(filters)
1110
+ row = self._conn.execute(
1111
+ f"SELECT COUNT(*) AS total FROM jobs{where}", tuple(params)
1112
+ ).fetchone()
1113
+ return int(row["total"])
1114
+
1115
+ def company_names(self) -> list[str]:
1116
+ rows = self._conn.execute(
1117
+ "SELECT DISTINCT company FROM jobs ORDER BY company COLLATE NOCASE"
1118
+ ).fetchall()
1119
+ return [str(row["company"]) for row in rows]
1120
+
1121
+ def get_job(self, job_id: str) -> Job | None:
1122
+ row = self._conn.execute("SELECT * FROM jobs WHERE id = ?", (job_id,)).fetchone()
1123
+ return _row_to_job(row) if row else None
1124
+
1125
+ def duplicates_of(self, job_id: str) -> list[Job]:
1126
+ rows = self._conn.execute(
1127
+ "SELECT * FROM jobs WHERE duplicate_of = ? ORDER BY source, id",
1128
+ (job_id,),
1129
+ ).fetchall()
1130
+ return [_row_to_job(row) for row in rows]
1131
+
1132
+ def _match(self, query: str, filters: JobFilters) -> tuple[str, str, list[Any]]:
1133
+ clauses, params = self._clauses(filters)
1134
+ clauses.insert(0, "jobs_fts MATCH ?")
1135
+ return (
1136
+ match_expression(search_terms(query)),
1137
+ f"FROM jobs JOIN jobs_fts ON jobs_fts.rowid = jobs.rowid WHERE {' AND '.join(clauses)}",
1138
+ params,
1139
+ )
1140
+
1141
+ def search_jobs(self, query: str, filters: JobFilters) -> list[Job]:
1142
+ expression, source, params = self._match(query, filters)
1143
+ if not expression:
1144
+ return []
1145
+ limit, limit_params = _limit_clause(filters.limit)
1146
+ rows = self._conn.execute(
1147
+ f"SELECT jobs.* {source} "
1148
+ f"ORDER BY bm25(jobs_fts, {_BM25_WEIGHTS}), jobs.first_seen DESC, jobs.id ASC"
1149
+ f"{limit}",
1150
+ (expression, *params, *limit_params),
1151
+ ).fetchall()
1152
+ return [_row_to_job(row) for row in rows]
1153
+
1154
+ def count_search(self, query: str, filters: JobFilters) -> int:
1155
+ expression, source, params = self._match(query, filters)
1156
+ if not expression:
1157
+ return 0
1158
+ row = self._conn.execute(
1159
+ f"SELECT COUNT(*) AS total {source}", (expression, *params)
1160
+ ).fetchone()
1161
+ return int(row["total"])
1162
+
1163
+ def record_sync_run(self, run: SyncRun) -> None:
1164
+ with self._transaction() as conn:
1165
+ cursor = conn.execute(
1166
+ "INSERT INTO sync_runs (started_at, finished_at, outcome) VALUES (?, ?, ?)",
1167
+ (_to_text(run.started_at), _to_text(run.finished_at), run.outcome.value),
1168
+ )
1169
+ run_id = cursor.lastrowid
1170
+ conn.executemany(
1171
+ "INSERT INTO sync_run_sources "
1172
+ "(run_id, source, fetched, added, updated, closed, quarantined, errors, "
1173
+ "requests, not_modified, retries, tightenings, latency_p50_ms, "
1174
+ "latency_p95_ms, elapsed_ms, deferred, blocked, stored) "
1175
+ "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
1176
+ [
1177
+ (
1178
+ run_id,
1179
+ stats.source,
1180
+ stats.fetched,
1181
+ stats.added,
1182
+ stats.updated,
1183
+ stats.closed,
1184
+ stats.quarantined,
1185
+ stats.errors,
1186
+ stats.requests,
1187
+ stats.not_modified,
1188
+ stats.retries,
1189
+ stats.tightenings,
1190
+ stats.latency_p50_ms,
1191
+ stats.latency_p95_ms,
1192
+ stats.elapsed_ms,
1193
+ stats.deferred,
1194
+ int(stats.blocked),
1195
+ stats.stored,
1196
+ )
1197
+ for stats in run.sources
1198
+ ],
1199
+ )
1200
+
1201
+ def last_sync_at(self) -> datetime | None:
1202
+ row = self._conn.execute(
1203
+ "SELECT finished_at FROM sync_runs ORDER BY finished_at DESC LIMIT 1"
1204
+ ).fetchone()
1205
+ return _from_text(row["finished_at"]) if row else None
1206
+
1207
+ def closed_among(self, job_ids: Sequence[str]) -> int:
1208
+ total = 0
1209
+ for start in range(0, len(job_ids), _ID_CHUNK):
1210
+ chunk = job_ids[start : start + _ID_CHUNK]
1211
+ marks = ", ".join("?" * len(chunk))
1212
+ row = self._conn.execute(
1213
+ f"SELECT COUNT(*) AS total FROM jobs WHERE status = ? AND id IN ({marks})",
1214
+ (JobStatus.CLOSED.value, *chunk),
1215
+ ).fetchone()
1216
+ total += int(row["total"])
1217
+ return total
1218
+
1219
+ def previous_sync_at(self) -> datetime | None:
1220
+ row = self._conn.execute(
1221
+ "SELECT finished_at FROM sync_runs ORDER BY finished_at DESC LIMIT 1 OFFSET 1"
1222
+ ).fetchone()
1223
+ return _from_text(row["finished_at"]) if row else None
1224
+
1225
+ def schema_version(self) -> int:
1226
+ row = self._conn.execute("SELECT MAX(version) AS version FROM schema_migrations").fetchone()
1227
+ return int(row["version"]) if row and row["version"] is not None else 0
1228
+
1229
+ def stored_counts(self) -> dict[str, int]:
1230
+ rows = self._conn.execute(
1231
+ "SELECT source, COUNT(*) AS total FROM jobs WHERE status = ? GROUP BY source",
1232
+ (JobStatus.OPEN.value,),
1233
+ ).fetchall()
1234
+ return {str(row["source"]): int(row["total"]) for row in rows}
1235
+
1236
+ def board_counts(self) -> dict[str, int]:
1237
+ rows = self._conn.execute(
1238
+ "SELECT id, source FROM jobs WHERE status = ?",
1239
+ (JobStatus.OPEN.value,),
1240
+ ).fetchall()
1241
+ counts: dict[str, int] = {}
1242
+ for row in rows:
1243
+ key = board_of(str(row["id"]), str(row["source"]))
1244
+ counts[key] = counts.get(key, 0) + 1
1245
+ return counts
1246
+
1247
+ def company_counts(self) -> dict[str, dict[str, int]]:
1248
+ rows = self._conn.execute(
1249
+ "SELECT company, source, COUNT(*) AS total FROM jobs "
1250
+ "WHERE duplicate_of IS NULL AND status = ? "
1251
+ "GROUP BY company, source",
1252
+ (JobStatus.OPEN.value,),
1253
+ ).fetchall()
1254
+ counts: dict[str, dict[str, int]] = {}
1255
+ for row in rows:
1256
+ counts.setdefault(str(row["company"]), {})[str(row["source"])] = int(row["total"])
1257
+ return counts
1258
+
1259
+ def quarantine_company_counts(self) -> dict[str, dict[str, int]]:
1260
+ rows = self._conn.execute(
1261
+ "SELECT company, source, COUNT(*) AS total FROM quarantine GROUP BY company, source"
1262
+ ).fetchall()
1263
+ counts: dict[str, dict[str, int]] = {}
1264
+ for row in rows:
1265
+ counts.setdefault(str(row["company"]), {})[str(row["source"])] = int(row["total"])
1266
+ return counts
1267
+
1268
+ def quarantine_company_reasons(self) -> dict[str, dict[str, int]]:
1269
+ rows = self._conn.execute(
1270
+ "SELECT company, reason, COUNT(*) AS total FROM quarantine GROUP BY company, reason"
1271
+ ).fetchall()
1272
+ reasons: dict[str, dict[str, int]] = {}
1273
+ for row in rows:
1274
+ reasons.setdefault(str(row["company"]), {})[str(row["reason"])] = int(row["total"])
1275
+ return reasons
1276
+
1277
+ def company_apply_urls(self, companies: Sequence[str]) -> dict[str, tuple[str, ...]]:
1278
+ wanted = {fold_company(company): company for company in companies}
1279
+ if not wanted:
1280
+ return {}
1281
+ collected: dict[str, list[str]] = {}
1282
+ folds = sorted(wanted)
1283
+ for start in range(0, len(folds), _ID_CHUNK):
1284
+ chunk = folds[start : start + _ID_CHUNK]
1285
+ marks = ", ".join("?" * len(chunk))
1286
+ kept = self._conn.execute(
1287
+ "SELECT company_fold, apply_url_raw FROM jobs "
1288
+ f"WHERE company_fold IN ({marks}) "
1289
+ "AND duplicate_of IS NULL AND status = ? AND apply_url_raw <> '' "
1290
+ "ORDER BY last_seen DESC, apply_url_raw",
1291
+ (*chunk, JobStatus.OPEN.value),
1292
+ ).fetchall()
1293
+ for row in kept:
1294
+ self._collect_apply_url(
1295
+ collected, wanted, str(row["company_fold"]), str(row["apply_url_raw"])
1296
+ )
1297
+
1298
+ names = sorted(set(wanted.values()))
1299
+ for start in range(0, len(names), _ID_CHUNK):
1300
+ chunk = names[start : start + _ID_CHUNK]
1301
+ marks = ", ".join("?" * len(chunk))
1302
+ rejected = self._conn.execute(
1303
+ "SELECT company, apply_url_raw FROM quarantine "
1304
+ f"WHERE company IN ({marks}) "
1305
+ "AND apply_url_raw IS NOT NULL AND apply_url_raw <> '' "
1306
+ "ORDER BY last_seen DESC, apply_url_raw",
1307
+ tuple(chunk),
1308
+ ).fetchall()
1309
+ for row in rejected:
1310
+ self._collect_apply_url(
1311
+ collected, wanted, fold_company(str(row["company"])), str(row["apply_url_raw"])
1312
+ )
1313
+ return {company: tuple(urls) for company, urls in collected.items()}
1314
+
1315
+ @staticmethod
1316
+ def _collect_apply_url(
1317
+ collected: dict[str, list[str]],
1318
+ wanted: dict[str, str],
1319
+ company_fold: str,
1320
+ apply_url: str,
1321
+ ) -> None:
1322
+ company = wanted.get(company_fold)
1323
+ url = apply_url.strip()
1324
+ if company is None or not url:
1325
+ return
1326
+ urls = collected.setdefault(company, [])
1327
+ if url not in urls and len(urls) < 5:
1328
+ urls.append(url)
1329
+
1330
+ def coverage_classifications(self) -> list[CoverageClassification]:
1331
+ rows = self._conn.execute(
1332
+ "SELECT company, disposition, note, checked_on, url FROM coverage_classifications "
1333
+ "ORDER BY company COLLATE NOCASE"
1334
+ ).fetchall()
1335
+ return [_classification_from_row(row) for row in rows]
1336
+
1337
+ def record_coverage_classification(self, entry: CoverageClassification) -> bool:
1338
+ company, company_fold, disposition, note, checked_on, url = _classification_params(entry)
1339
+ with self._transaction() as conn:
1340
+ existing = conn.execute(
1341
+ "SELECT 1 FROM coverage_classifications WHERE company_fold = ?", (company_fold,)
1342
+ ).fetchone()
1343
+ conn.execute(
1344
+ "INSERT INTO coverage_classifications "
1345
+ "(company, company_fold, disposition, note, checked_on, url) "
1346
+ "VALUES (?, ?, ?, ?, ?, ?) "
1347
+ "ON CONFLICT(company_fold) DO UPDATE SET company = excluded.company, "
1348
+ "disposition = excluded.disposition, note = excluded.note, "
1349
+ "checked_on = excluded.checked_on, url = excluded.url",
1350
+ (company, company_fold, disposition, note, checked_on, url),
1351
+ )
1352
+ return existing is not None
1353
+
1354
+ def clear_coverage_classification(self, company: str) -> bool:
1355
+ company_fold = fold_company(company)
1356
+ if not company_fold:
1357
+ raise ValueError("classification company must contain letters or numbers")
1358
+ with self._transaction() as conn:
1359
+ cursor = conn.execute(
1360
+ "DELETE FROM coverage_classifications WHERE company_fold = ?", (company_fold,)
1361
+ )
1362
+ return cursor.rowcount > 0
1363
+
1364
+ def composition(self, column: str) -> dict[str, int]:
1365
+ if column not in _COMPOSITION_COLUMNS:
1366
+ raise ValueError(
1367
+ f"{column!r} is not a composition column; "
1368
+ f"known: {', '.join(sorted(_COMPOSITION_COLUMNS))}"
1369
+ )
1370
+ rows = self._conn.execute(
1371
+ f"SELECT {column} AS bucket, COUNT(*) AS total FROM jobs "
1372
+ "WHERE duplicate_of IS NULL GROUP BY bucket ORDER BY total DESC"
1373
+ ).fetchall()
1374
+ counts: dict[str, int] = {}
1375
+ for row in rows:
1376
+ bucket = str(row["bucket"])
1377
+ if column == "location":
1378
+ bucket = _location_bucket(bucket).value
1379
+ counts[bucket] = counts.get(bucket, 0) + int(row["total"])
1380
+ return counts
1381
+
1382
+ def volume_history(self, limit: int) -> Mapping[str, list[VolumePoint]]:
1383
+ rows = self._conn.execute(
1384
+ "SELECT source, stored, deferred, blocked FROM sync_run_sources "
1385
+ "WHERE run_id IN (SELECT id FROM sync_runs ORDER BY id DESC LIMIT ?) "
1386
+ "ORDER BY run_id DESC",
1387
+ (limit,),
1388
+ ).fetchall()
1389
+ history: dict[str, list[VolumePoint]] = {}
1390
+ for row in rows:
1391
+ history.setdefault(str(row["source"]), []).append(
1392
+ VolumePoint(
1393
+ stored=int(row["stored"]),
1394
+ deferred=int(row["deferred"]),
1395
+ blocked=bool(row["blocked"]),
1396
+ )
1397
+ )
1398
+ return history
1399
+
1400
+ def requests_since(self, since: datetime) -> tuple[dict[str, int], bool]:
1401
+ rows = self._conn.execute(
1402
+ "SELECT s.source AS source, SUM(s.requests) AS spent "
1403
+ "FROM sync_run_sources s JOIN sync_runs r ON r.id = s.run_id "
1404
+ "WHERE r.started_at >= ? GROUP BY s.source",
1405
+ (_to_text(since),),
1406
+ ).fetchall()
1407
+ seen = self._conn.execute(
1408
+ "SELECT 1 FROM sync_runs WHERE started_at >= ? LIMIT 1", (_to_text(since),)
1409
+ ).fetchone()
1410
+ return {str(row["source"]): int(row["spent"] or 0) for row in rows}, seen is not None
1411
+
1412
+ def run_history(self, limit: int) -> list[SyncRun]:
1413
+ runs = self._conn.execute(
1414
+ "SELECT id, started_at, finished_at, outcome FROM sync_runs ORDER BY id DESC LIMIT ?",
1415
+ (limit,),
1416
+ ).fetchall()
1417
+ if not runs:
1418
+ return []
1419
+ placeholders = ", ".join("?" * len(runs))
1420
+ stats = self._conn.execute(
1421
+ "SELECT * FROM sync_run_sources "
1422
+ f"WHERE run_id IN ({placeholders}) ORDER BY run_id DESC, source",
1423
+ tuple(int(run["id"]) for run in runs),
1424
+ ).fetchall()
1425
+ by_run: dict[int, list[SourceRunStats]] = {}
1426
+ for row in stats:
1427
+ by_run.setdefault(int(row["run_id"]), []).append(
1428
+ SourceRunStats(
1429
+ source=str(row["source"]),
1430
+ fetched=int(row["fetched"]),
1431
+ added=int(row["added"]),
1432
+ updated=int(row["updated"]),
1433
+ closed=int(row["closed"]),
1434
+ quarantined=int(row["quarantined"]),
1435
+ errors=int(row["errors"]),
1436
+ requests=int(row["requests"]),
1437
+ not_modified=int(row["not_modified"]),
1438
+ retries=int(row["retries"]),
1439
+ tightenings=int(row["tightenings"]),
1440
+ latency_p50_ms=float(row["latency_p50_ms"]),
1441
+ latency_p95_ms=float(row["latency_p95_ms"]),
1442
+ elapsed_ms=float(row["elapsed_ms"]),
1443
+ deferred=int(row["deferred"]),
1444
+ blocked=bool(row["blocked"]),
1445
+ stored=int(row["stored"]),
1446
+ )
1447
+ )
1448
+ return [
1449
+ SyncRun(
1450
+ started_at=_require_datetime(run["started_at"], "started_at"),
1451
+ finished_at=_require_datetime(run["finished_at"], "finished_at"),
1452
+ outcome=SyncOutcome(str(run["outcome"])),
1453
+ sources=tuple(by_run.get(int(run["id"]), ())),
1454
+ )
1455
+ for run in runs
1456
+ ]
1457
+
1458
+ def all_visits(self) -> list[SourceVisit]:
1459
+ rows = self._conn.execute(
1460
+ "SELECT source, board, label, last_attempt_at, last_success_at, "
1461
+ "consecutive_failures, last_error FROM source_visits "
1462
+ "ORDER BY last_success_at IS NOT NULL, last_success_at, source, board"
1463
+ ).fetchall()
1464
+ return [
1465
+ SourceVisit(
1466
+ source=str(row["source"]),
1467
+ board=str(row["board"]),
1468
+ label=str(row["label"]),
1469
+ last_attempt_at=_require_datetime(row["last_attempt_at"], "last_attempt_at"),
1470
+ last_success_at=_from_text(row["last_success_at"]),
1471
+ consecutive_failures=int(row["consecutive_failures"]),
1472
+ last_error=str(row["last_error"]),
1473
+ )
1474
+ for row in rows
1475
+ ]
1476
+
1477
+ def repair_integrity(self) -> list[IntegrityRepair]:
1478
+ repairs = [
1479
+ IntegrityRepair(
1480
+ check="dangling duplicate links",
1481
+ repaired=self._execute_repair(
1482
+ "UPDATE jobs SET duplicate_of = NULL WHERE duplicate_of IS NOT NULL "
1483
+ "AND duplicate_of NOT IN (SELECT id FROM jobs)"
1484
+ ),
1485
+ detail="the posting is visible again on its own",
1486
+ ),
1487
+ IntegrityRepair(
1488
+ check="same-board merges",
1489
+ repaired=self._execute_repair(
1490
+ "UPDATE jobs SET duplicate_of = NULL WHERE id IN ("
1491
+ "SELECT a.id FROM jobs a JOIN jobs b ON a.duplicate_of = b.id "
1492
+ "WHERE stage_board_of(a.id, a.source) = stage_board_of(b.id, b.source))"
1493
+ ),
1494
+ detail="two rows from one board are two requisitions",
1495
+ ),
1496
+ IntegrityRepair(
1497
+ check="duplicate chains",
1498
+ repaired=self._repair_chains(),
1499
+ detail="followers now point at the survivor",
1500
+ ),
1501
+ IntegrityRepair(
1502
+ check="tombstoned rows re-ingested as new",
1503
+ repaired=self._execute_repair(
1504
+ "UPDATE jobs SET first_seen = ("
1505
+ "SELECT t.first_seen FROM tombstones t WHERE t.id = jobs.id) "
1506
+ "WHERE id IN (SELECT j.id FROM jobs j JOIN tombstones t ON t.id = j.id "
1507
+ "WHERE j.first_seen > t.first_seen)"
1508
+ ),
1509
+ detail="the original first_seen is restored from the tombstone",
1510
+ ),
1511
+ ]
1512
+ self._conn.commit()
1513
+ return [repair for repair in repairs if repair.repaired]
1514
+
1515
+ def close_orphan_boards(self, sources: Sequence[str], boards: Sequence[str]) -> int:
1516
+ if not sources or not boards:
1517
+ return 0
1518
+ source_slots = ", ".join("?" * len(sources))
1519
+ board_slots = ", ".join("?" * len(boards))
1520
+ with self._transaction() as conn:
1521
+ closed = conn.execute(
1522
+ f"UPDATE jobs SET status = ? WHERE status = ? AND source IN ({source_slots}) "
1523
+ "AND stage_board_of(id, source) <> source "
1524
+ f"AND stage_board_of(id, source) NOT IN ({board_slots})",
1525
+ (JobStatus.CLOSED.value, JobStatus.OPEN.value, *sources, *boards),
1526
+ ).rowcount
1527
+ return int(closed or 0)
1528
+
1529
+ def _execute_repair(self, sql: str) -> int:
1530
+ return int(self._conn.execute(sql).rowcount or 0)
1531
+
1532
+ def _repair_chains(self) -> int:
1533
+ repaired = 0
1534
+ for _ in range(MAX_CHAIN_REPAIR_PASSES):
1535
+ changed = self._execute_repair(
1536
+ "UPDATE jobs SET duplicate_of = ("
1537
+ "SELECT b.duplicate_of FROM jobs b WHERE b.id = jobs.duplicate_of) "
1538
+ "WHERE duplicate_of IN (SELECT id FROM jobs WHERE duplicate_of IS NOT NULL)"
1539
+ )
1540
+ repaired += changed
1541
+ if not changed:
1542
+ break
1543
+ return repaired
1544
+
1545
+ def integrity_findings(self) -> list[IntegrityFinding]:
1546
+ checks = (
1547
+ (
1548
+ "dangling duplicate links",
1549
+ "SELECT COUNT(*) AS total FROM jobs a LEFT JOIN jobs b ON a.duplicate_of = b.id "
1550
+ "WHERE a.duplicate_of IS NOT NULL AND b.id IS NULL",
1551
+ "the survivor is gone, so the posting is invisible",
1552
+ ),
1553
+ (
1554
+ "duplicate chains",
1555
+ "SELECT COUNT(*) AS total FROM jobs a JOIN jobs b ON a.duplicate_of = b.id "
1556
+ "WHERE b.duplicate_of IS NOT NULL",
1557
+ "followers must repoint at the survivor",
1558
+ ),
1559
+ (
1560
+ "same-board merges",
1561
+ "SELECT COUNT(*) AS total FROM jobs a JOIN jobs b ON a.duplicate_of = b.id "
1562
+ "WHERE stage_board_of(a.id, a.source) = stage_board_of(b.id, b.source)",
1563
+ "two rows from one board are two requisitions",
1564
+ ),
1565
+ (
1566
+ "postings in both tables",
1567
+ "SELECT COUNT(*) AS total FROM jobs j JOIN quarantine q ON q.id = j.id",
1568
+ "quarantine is a move, not a copy",
1569
+ ),
1570
+ (
1571
+ "tombstoned rows re-ingested as new",
1572
+ "SELECT COUNT(*) AS total FROM jobs j JOIN tombstones t ON t.id = j.id "
1573
+ "WHERE j.first_seen > t.first_seen",
1574
+ "a purged posting came back with a fresh date",
1575
+ ),
1576
+ (
1577
+ "open postings with no first_seen",
1578
+ "SELECT COUNT(*) AS total FROM jobs WHERE first_seen IS NULL OR first_seen = ''",
1579
+ "first_seen is the sort key and is assigned locally",
1580
+ ),
1581
+ )
1582
+ findings: list[IntegrityFinding] = []
1583
+ for check, sql, detail in checks:
1584
+ row = self._conn.execute(sql).fetchone()
1585
+ findings.append(IntegrityFinding(check=check, count=int(row["total"]), detail=detail))
1586
+ return findings