openings 0.1.37 → 0.1.39

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openings",
3
- "version": "0.1.37",
3
+ "version": "0.1.39",
4
4
  "description": "Find evidence-grounded jobs, including relevant roles you may not have searched for, without accounts or API keys.",
5
5
  "author": { "name": "Openings contributors" },
6
6
  "license": "MIT",
@@ -108370,5 +108370,49 @@
108370
108370
  "payloadVersion": "amazon-jobs-search:v1",
108371
108371
  "jobCount": 2368
108372
108372
  }
108373
+ },
108374
+ "jpmorganchase": {
108375
+ "name": "JPMorganChase",
108376
+ "ats": "oraclecloud",
108377
+ "token": "jpmc.fa.oraclecloud.com/CX_1",
108378
+ "cohorts": [
108379
+ "IN"
108380
+ ],
108381
+ "sourceUrl": "https://jpmc.fa.oraclecloud.com/hcmUI/CandidateExperience/en/sites/CX_1/requisitions",
108382
+ "discoveredFrom": {
108383
+ "channel": "company_site",
108384
+ "reference": "https://careers.jpmorgan.com/"
108385
+ },
108386
+ "verification": {
108387
+ "observedCompanyName": "JPMorganChase",
108388
+ "identityEvidence": "company_page_link",
108389
+ "contentType": "application/json",
108390
+ "payloadVersion": "oracle-recruiting-ce:v1",
108391
+ "jobCount": 7447,
108392
+ "checkedAt": "2026-09-17T10:47:14.367307Z",
108393
+ "canonicalSourceUrl": "https://jpmc.fa.oraclecloud.com/hcmUI/CandidateExperience/en/sites/CX_1/requisitions"
108394
+ }
108395
+ },
108396
+ "tatacapital": {
108397
+ "name": "Tata Capital",
108398
+ "ats": "oraclecloud",
108399
+ "token": "eofh.fa.em2.oraclecloud.com/CX",
108400
+ "cohorts": [
108401
+ "IN"
108402
+ ],
108403
+ "sourceUrl": "https://eofh.fa.em2.oraclecloud.com/hcmUI/CandidateExperience/en/sites/CX/requisitions",
108404
+ "discoveredFrom": {
108405
+ "channel": "company_site",
108406
+ "reference": "https://www.tatacapital.com/careers.html"
108407
+ },
108408
+ "verification": {
108409
+ "observedCompanyName": "Tata Capital",
108410
+ "identityEvidence": "company_page_link",
108411
+ "contentType": "application/json",
108412
+ "payloadVersion": "oracle-recruiting-ce:v1",
108413
+ "jobCount": 5278,
108414
+ "checkedAt": "2026-09-17T10:47:14.367307Z",
108415
+ "canonicalSourceUrl": "https://eofh.fa.em2.oraclecloud.com/hcmUI/CandidateExperience/en/sites/CX/requisitions"
108416
+ }
108373
108417
  }
108374
108418
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openings",
3
- "version": "0.1.37",
3
+ "version": "0.1.39",
4
4
  "description": "A free, candidate-safe job-search substrate for AI agents",
5
5
  "license": "MIT",
6
6
  "repository": { "type": "git", "url": "git+https://github.com/abhay-avagama/hiring-agent.git" },
@@ -55,7 +55,9 @@ export async function verifyBoards(registryPath: string, catalogPath: string, op
55
55
  for (const lead of registry.leads) {
56
56
  if (!BOARD_TIER_PROVIDERS.has(lead.ats) || !resolveSource(lead.sourceUrl)) { skipped.unsupported += 1; continue; }
57
57
  if (knownSources.has(`${lead.ats}:${lead.token.toLowerCase()}`)) { skipped.inCatalog += 1; continue; }
58
- if (looksLikeTestBoard(lead.token)) { skipped.unsupported += 1; continue; }
58
+ // Keka tokens contain tenant/UUID; the provider-issued routing UUID is not a company name.
59
+ const boardName = lead.ats === "keka" ? lead.token.split("/")[0]! : lead.token;
60
+ if (looksLikeTestBoard(boardName)) { skipped.unsupported += 1; continue; }
59
61
  if (coolingDown(lead, now(), cooldownMs)) { skipped.coolingDown += 1; continue; }
60
62
  eligible.push(lead);
61
63
  }
@@ -0,0 +1,228 @@
1
+ import { Database } from "bun:sqlite";
2
+ import { readFile } from "node:fs/promises";
3
+ import { resolve } from "node:path";
4
+ import { assertArtifactFile } from "./artifact-path.ts";
5
+ import { atomicJson } from "./atomic-file.ts";
6
+ import { commonCrawlPatterns } from "./common-crawl-discovery.ts";
7
+ import { withFileLock } from "./file-lock.ts";
8
+ import { resolveSource } from "./source-verification.ts";
9
+ import { ALL_PROVIDERS, type Ats } from "./types.ts";
10
+
11
+ interface Options {
12
+ state: string; indexes: string[]; providers: Ats[]; execute?: boolean;
13
+ requestBudget?: number; pageBudget?: number; targetBoards?: number; delayMs?: number;
14
+ catalogPath?: string; exportPath?: string;
15
+ }
16
+ interface Dependencies {
17
+ fetch?: (input: string | URL, init?: RequestInit) => Promise<Response>;
18
+ sleep?: (ms: number) => Promise<void>; now?: () => number;
19
+ }
20
+ interface Query { id: number; index_id: string; provider: Ats; pattern: string; pages: number | null; next_page: number }
21
+ interface Board { key: string; ats: Ats; token: string; url: string }
22
+ interface GatewayRetry { url: string; failures: number; nextAt: number }
23
+ interface FailureDetail { at: string; url: string; status: number; headers: Record<string, string>; bodyPreview: string }
24
+ const MAX_BODY_BYTES = 8 * 1024 * 1024;
25
+ const SCHEMA = 1;
26
+
27
+ /** Discovery only: fixed-host index requests, durable page checkpoints, no ATS probes or catalog writes. */
28
+ export async function discoverCatalog(options: Options, deps: Dependencies = {}) {
29
+ // Distinct suffixes prevent SQLite files from aliasing the export's .lock or .PID.tmp sidecars.
30
+ if (!options.state.endsWith(".sqlite")) throw Error("Discovery state must use a .sqlite filename");
31
+ if (options.exportPath && !options.exportPath.endsWith(".json")) throw Error("Discovery export must use a .json filename");
32
+ const indexes = [...new Set(options.indexes)].sort();
33
+ const providers = [...new Set(options.providers)].sort();
34
+ if (!indexes.length || indexes.length > 24 || indexes.some(id => !/^CC-MAIN-20\d{2}-\d{2}$/.test(id))) throw Error("Supply 1–24 pinned CC-MAIN-YYYY-NN index IDs");
35
+ if (!providers.length || providers.some(p => !ALL_PROVIDERS.includes(p) || !commonCrawlPatterns(p)?.length)) throw Error("Supply supported providers with discovery patterns");
36
+ const requests = integer(options.requestBudget ?? 100, 1, 1000, "requestBudget");
37
+ const pages = integer(options.pageBudget ?? 80, 1, 1000, "pageBudget");
38
+ const target = integer(options.targetBoards ?? 50000, 1, 1000000, "targetBoards");
39
+ const delay = integer(options.delayMs ?? 1500, 1000, 60000, "delayMs");
40
+ await assertArtifactFile(options.state);
41
+ if (options.exportPath) {
42
+ await assertArtifactFile(options.exportPath);
43
+ if ([options.state, `${options.state}.lock`, `${options.state}-wal`, `${options.state}-shm`, `${options.state}-journal`, options.catalogPath].some(p => p && resolve(p) === resolve(options.exportPath!))) throw Error("Export must not overwrite state, lock, or catalog");
44
+ if (await Bun.file(options.exportPath).exists()) throw Error("Export already exists; choose a new artifact path to preserve verification outcomes");
45
+ }
46
+ const known = new Set<string>();
47
+ if (options.catalogPath) {
48
+ const catalog = JSON.parse(await readFile(options.catalogPath, "utf8"));
49
+ if (!catalog || typeof catalog !== "object" || Array.isArray(catalog)) throw Error("Invalid catalog");
50
+ for (const entry of Object.values(catalog) as Array<{ ats: string; token: string }>) {
51
+ if (typeof entry.ats !== "string" || typeof entry.token !== "string") throw Error("Invalid catalog entry");
52
+ known.add(`${entry.ats}:${entry.token.toLowerCase()}`);
53
+ }
54
+ }
55
+ const config = JSON.stringify({ schema: SCHEMA, indexes, providers, pageSize: 1, patterns: providers.map(p => [p, commonCrawlPatterns(p)]) });
56
+ const now = deps.now ?? Date.now;
57
+ const sleep = deps.sleep ?? ((ms: number) => new Promise<void>(done => setTimeout(done, ms)));
58
+ const fetcher = deps.fetch ?? globalThis.fetch;
59
+ return withFileLock(options.state, async () => {
60
+ const db = new Database(options.state);
61
+ try {
62
+ db.exec(`PRAGMA foreign_keys=ON; PRAGMA busy_timeout=10000;
63
+ CREATE TABLE IF NOT EXISTS meta (key TEXT PRIMARY KEY, value TEXT NOT NULL);
64
+ CREATE TABLE IF NOT EXISTS queries (id INTEGER PRIMARY KEY, index_id TEXT NOT NULL, provider TEXT NOT NULL, pattern TEXT NOT NULL, pages INTEGER, next_page INTEGER NOT NULL DEFAULT 0, UNIQUE(index_id,pattern));
65
+ CREATE TABLE IF NOT EXISTS boards (key TEXT PRIMARY KEY, ats TEXT NOT NULL, token TEXT NOT NULL, url TEXT NOT NULL);
66
+ CREATE TABLE IF NOT EXISTS origins (board_key TEXT REFERENCES boards(key), query_id INTEGER REFERENCES queries(id), PRIMARY KEY(board_key,query_id));`);
67
+ const get = (key: string) => db.query<{ value: string }, [string]>("SELECT value FROM meta WHERE key=?").get(key)?.value;
68
+ const put = (key: string, value: string | number) => db.prepare("INSERT INTO meta VALUES (?,?) ON CONFLICT(key) DO UPDATE SET value=excluded.value").run(key, String(value));
69
+ if (get("config") && get("config") !== config) throw Error("Checkpoint configuration differs; use a new state file for different indexes/providers");
70
+ if (!get("config")) db.transaction(() => {
71
+ put("config", config);
72
+ for (const index of indexes) for (const provider of providers) for (const pattern of commonCrawlPatterns(provider)) db.prepare("INSERT INTO queries(index_id,provider,pattern) VALUES (?,?,?)").run(index, provider, pattern);
73
+ })();
74
+ const boardCount = () => db.query<{ n: number }, []>("SELECT count(*) AS n FROM boards").get()!.n;
75
+ let requestsThisRun = 0, pagesThisRun = 0;
76
+ let status = options.execute ? "complete" : "plan";
77
+ let error: string | undefined;
78
+ if (options.execute) while (true) {
79
+ const query = db.query<Query, []>("SELECT * FROM queries WHERE pages IS NULL OR next_page < pages ORDER BY next_page,id LIMIT 1").get();
80
+ if (!query) break;
81
+ if (Number(get("retryAt") ?? 0) > now()) { status = "cooling_down"; break; }
82
+ if (boardCount() >= target) { status = "target_reached"; break; }
83
+ if (requestsThisRun >= requests) { status = "request_budget"; break; }
84
+ if (pagesThisRun >= pages) { status = "page_budget"; break; }
85
+ const url = new URL(`https://index.commoncrawl.org/${query.index_id}-index`);
86
+ url.search = new URLSearchParams({ url: query.pattern, output: "json", filter: "=status:200", fl: "url", pageSize: "1", ...(query.pages === null ? { showNumPages: "true" } : { page: String(query.next_page) }) }).toString();
87
+ const gatewayRetry: GatewayRetry | null = JSON.parse(get("gatewayRetry") ?? "null");
88
+ // Persist request accounting before dispatch: a process crash cannot erase a consumed attempt.
89
+ const waitMs = Math.max(delay, Number(get("lastRequestAt") ?? 0) + delay - now(), gatewayRetry?.url === String(url) ? gatewayRetry.nextAt - now() : 0);
90
+ // Long Retry-After values yield control to the operator rather than holding a CLI open indefinitely.
91
+ if (waitMs > 60000) { status = "cooling_down"; break; }
92
+ await sleep(waitMs);
93
+ put("lastRequestAt", now()); put("requests", Number(get("requests") ?? 0) + 1); requestsThisRun++;
94
+ try {
95
+ const response = await fetcher(url, { redirect: "error", signal: AbortSignal.timeout(30000), headers: { "user-agent": "OpeningsCatalogDiscovery/1.0" } });
96
+ if (response.status === 502 || response.status === 504) {
97
+ const failures = (gatewayRetry?.url === String(url) ? gatewayRetry.failures : 0) + 1;
98
+ const nextAt = Math.max(now() + 10000 * 2 ** (failures - 1), retryAfter(response, now()));
99
+ put("lastFailure", JSON.stringify(await failureDetail(response, url, now())));
100
+ if (failures >= 3) {
101
+ put("retryAt", Math.max(now() + 15 * 60000, nextAt)); put("gatewayRetry", "null");
102
+ error = `Common Crawl HTTP ${response.status}; gateway retry allowance exhausted`; status = "error"; break;
103
+ }
104
+ put("gatewayRetry", JSON.stringify({ url: String(url), failures, nextAt } satisfies GatewayRetry));
105
+ continue;
106
+ }
107
+ if (response.status === 429 || response.status === 503) {
108
+ put("retryAt", Math.max(now() + 15 * 60000, retryAfter(response, now())));
109
+ put("lastFailure", JSON.stringify(await failureDetail(response, url, now())));
110
+ status = "throttled"; break;
111
+ }
112
+ if (response.status === 404) {
113
+ const text = await boundedText(response);
114
+ let message: unknown;
115
+ try { const value = JSON.parse(text); message = value?.message ?? value?.error; } catch { /* Unknown 404s must not advance. */ }
116
+ // CDX prefix queries may echo the prefix without its trailing wildcard.
117
+ const emptyMessages = [query.pattern, query.pattern.replace(/\*$/, "")].map(pattern => `No Captures found for: ${pattern}`.toLowerCase());
118
+ if (typeof message === "string" && emptyMessages.includes(message.toLowerCase())) {
119
+ if (query.pages === null) db.prepare("UPDATE queries SET pages=0 WHERE id=?").run(query.id);
120
+ else { db.prepare("UPDATE queries SET next_page=next_page+1 WHERE id=?").run(query.id); pagesThisRun++; }
121
+ put("gatewayRetry", "null");
122
+ continue;
123
+ }
124
+ put("lastFailure", JSON.stringify({ at: new Date(now()).toISOString(), url: String(url), status: 404, headers: diagnosticHeaders(response), bodyPreview: text.slice(0,2048) } satisfies FailureDetail));
125
+ throw Error("Common Crawl HTTP 404");
126
+ }
127
+ if (!response.ok) { put("lastFailure", JSON.stringify(await failureDetail(response, url, now()))); throw Error(`Common Crawl HTTP ${response.status}`); }
128
+ const body = await boundedText(response);
129
+ if (query.pages === null) {
130
+ const info = JSON.parse(body);
131
+ if (!Number.isSafeInteger(info.pages) || info.pages < 0 || info.pages > 1000000 || info.pageSize !== 1) throw Error("Invalid CDX page count or pageSize");
132
+ db.prepare("UPDATE queries SET pages=? WHERE id=?").run(info.pages, query.id);
133
+ } else {
134
+ const records = body.split(/\r?\n/).filter(line => line.trim());
135
+ const additions: Board[] = []; let rejected = 0;
136
+ for (const line of records) {
137
+ const record: unknown = JSON.parse(line);
138
+ if (!record || typeof record !== "object" || !("url" in record) || typeof record.url !== "string") throw Error("Malformed CDX URL record; page was not checkpointed");
139
+ const source = resolveSource(record.url);
140
+ if (!source || source.ats !== query.provider) { rejected++; continue; }
141
+ additions.push({ key: `${source.ats}:${source.token.toLowerCase()}`, ats: source.ats, token: source.token, url: source.canonicalSourceUrl });
142
+ }
143
+ db.transaction(() => {
144
+ for (const board of additions) {
145
+ db.prepare("INSERT OR IGNORE INTO boards VALUES (?,?,?,?)").run(board.key, board.ats, board.token, board.url);
146
+ db.prepare("INSERT OR IGNORE INTO origins VALUES (?,?)").run(board.key, query.id);
147
+ }
148
+ put("records", Number(get("records") ?? 0) + records.length);
149
+ put("rejected", Number(get("rejected") ?? 0) + rejected);
150
+ db.prepare("UPDATE queries SET next_page=next_page+1 WHERE id=?").run(query.id);
151
+ })();
152
+ pagesThisRun++;
153
+ }
154
+ put("gatewayRetry", "null");
155
+ } catch (cause) { error = cause instanceof Error ? cause.message : String(cause); status = "error"; break; }
156
+ }
157
+ const boards = db.query<Board, []>("SELECT * FROM boards ORDER BY key").all();
158
+ let exported: number | undefined;
159
+ if (options.exportPath) {
160
+ const origins = db.query<{ board_key: string; index_id: string }, []>("SELECT DISTINCT board_key,index_id FROM origins JOIN queries ON queries.id=origins.query_id ORDER BY board_key,index_id").all();
161
+ const byKey = new Map<string, string[]>();
162
+ for (const origin of origins) { const refs = byKey.get(origin.board_key) ?? []; refs.push(origin.index_id); byKey.set(origin.board_key, refs); }
163
+ const leads = boards.filter(b => !known.has(b.key)).map(b => ({ sourceKey: b.key, ats: b.ats, token: b.token, sourceUrl: b.url, discoveredFrom: (byKey.get(b.key) ?? []).map(id => ({ channel: "dataset", reference: `https://index.commoncrawl.org/${id}-index` })), companyMatches: [], identityEvidence: [], attempts: [] }));
164
+ await withFileLock(options.exportPath, async () => {
165
+ if (await Bun.file(options.exportPath!).exists()) throw Error("Export already exists; choose a new artifact path");
166
+ await atomicJson(options.exportPath!, { version: 1, updatedAt: new Date(now()).toISOString(), leads });
167
+ }, { operation: "export discovery leads" }); exported = leads.length;
168
+ }
169
+ return { status, error, state: options.state, indexes, providers, boards: boards.length, knownBoards: boards.filter(b => known.has(b.key)).length, newBoardLeads: boards.filter(b => !known.has(b.key)).length,
170
+ records: Number(get("records") ?? 0), rejectedRecords: Number(get("rejected") ?? 0), requestsTotal: Number(get("requests") ?? 0), requestsThisRun, pagesThisRun,
171
+ retryAt: nextRetryAt(get("retryAt"), get("gatewayRetry"), now()),
172
+ lastFailure: JSON.parse(get("lastFailure") ?? "null") as FailureDetail | null,
173
+ limits: { requestBudget: requests, pageBudget: pages, targetBoards: target, delayMs: delay, maxResponseBytes: MAX_BODY_BYTES },
174
+ queries: db.query<Query, []>("SELECT * FROM queries ORDER BY id").all(), exported,
175
+ note: "Unverified discovery leads, not active boards or eligible employers. No catalog or shared registry was changed." };
176
+ } finally { db.close(); }
177
+ }, { operation: "catalog discovery" });
178
+ }
179
+
180
+ function retryAfter(response: Response, now: number): number {
181
+ const header = response.headers.get("retry-after");
182
+ const value = header && /^\d+$/.test(header) ? now + Number(header) * 1000 : Date.parse(header ?? "");
183
+ return Number.isFinite(value) && value <= 8640000000000000 ? value : 0;
184
+ }
185
+ function nextRetryAt(cooldown: string | undefined, gateway: string | undefined, now: number) {
186
+ const retry: GatewayRetry | null = JSON.parse(gateway ?? "null");
187
+ const value = Math.max(Number(cooldown ?? 0), retry?.nextAt ?? 0);
188
+ return value > now ? new Date(value).toISOString() : undefined;
189
+ }
190
+ function diagnosticHeaders(response: Response) {
191
+ return Object.fromEntries(["server", "via", "cf-ray", "x-request-id", "x-cache", "retry-after", "content-type"].flatMap(name => {
192
+ const value = response.headers.get(name); return value === null ? [] : [[name, value.slice(0,512)]];
193
+ }));
194
+ }
195
+ async function failureDetail(response: Response, url: URL, now: number): Promise<FailureDetail> {
196
+ const detail: FailureDetail = { at: new Date(now).toISOString(), url: String(url), status: response.status, headers: diagnosticHeaders(response), bodyPreview: "" };
197
+ const reader = response.body?.getReader(); if (!reader) return detail;
198
+ const chunks: Uint8Array[] = []; let bytes = 0;
199
+ try {
200
+ while (bytes < 2048) {
201
+ const chunk = await reader.read(); if (chunk.done) break;
202
+ const part = chunk.value.subarray(0, 2048-bytes); chunks.push(part); bytes += part.length;
203
+ }
204
+ const buffer = new Uint8Array(bytes); let offset = 0;
205
+ for (const chunk of chunks) { buffer.set(chunk, offset); offset += chunk.length; }
206
+ detail.bodyPreview = new TextDecoder().decode(buffer);
207
+ } catch { detail.bodyPreview = "[response body unavailable]"; }
208
+ finally { await reader.cancel().catch(() => {}); reader.releaseLock(); }
209
+ return detail;
210
+ }
211
+
212
+ function integer(value: number, min: number, max: number, name: string) {
213
+ if (!Number.isSafeInteger(value) || value < min || value > max) throw Error(`${name} must be an integer from ${min} to ${max}`);
214
+ return value;
215
+ }
216
+ async function boundedText(response: Response): Promise<string> {
217
+ const reader = response.body?.getReader(); if (!reader) return "";
218
+ const decoder = new TextDecoder(); let bytes = 0, text = "";
219
+ try {
220
+ while (true) {
221
+ const chunk = await reader.read(); if (chunk.done) break;
222
+ bytes += chunk.value.byteLength;
223
+ if (bytes > MAX_BODY_BYTES) throw Error("CDX response exceeded byte limit; page was not checkpointed");
224
+ text += decoder.decode(chunk.value, { stream: true });
225
+ }
226
+ return text + decoder.decode();
227
+ } finally { await reader.cancel().catch(() => {}); reader.releaseLock(); }
228
+ }
package/src/catalog.ts CHANGED
@@ -199,7 +199,15 @@ const WORKDAY_LISTING_CAP = 2000;
199
199
  const WORKDAY_COUNTRY_NAMES: Record<string, string[]> = { IN: ["india"], US: ["united states of america", "united states", "usa"] };
200
200
  /** Countries whose multi-location postings ("2 Locations") are labelled from the tenant's own country filter, not only on capped tenants. */
201
201
  const WORKDAY_LABEL_COUNTRIES = new Set(["IN"]);
202
- function workdayJobKey(job: WorkdayJob): string { return job.bulletFields?.[0] ?? job.externalPath; }
202
+ function workdayJobKey(job: WorkdayJob): string {
203
+ // Bullets are tenant-defined display fields (e.g. "Regular"), not a requisition contract.
204
+ // Preserve existing requisition IDs only when the posting URL corroborates them.
205
+ const tail = job.externalPath.split("/").at(-1) ?? "";
206
+ const suffix = tail.includes("_") ? tail.slice(tail.lastIndexOf("_") + 1) : "";
207
+ const bullet = job.bulletFields?.[0]?.trim();
208
+ if (bullet && /\d/.test(bullet) && (suffix === bullet || (suffix.startsWith(`${bullet}-`) && /^\d+$/.test(suffix.slice(bullet.length + 1))))) return bullet;
209
+ return suffix && /^[a-z0-9.-]+$/i.test(suffix) && /\d/.test(suffix) ? suffix : job.externalPath;
210
+ }
203
211
 
204
212
  export interface WorkdayCountryFacet { parameter: string; id: string; count: number }
205
213
  /**
@@ -378,7 +386,7 @@ export function abortableDelay(ms: number, signal?: AbortSignal): Promise<void>
378
386
 
379
387
  function normalizeWorkday(company: Company, source: ReturnType<typeof parseWorkdayToken>, job: WorkdayJob): Job {
380
388
  const location = job.locationsText ?? "Unspecified";
381
- const requisition = job.bulletFields?.[0] ?? job.externalPath.split("_").at(-1) ?? job.externalPath;
389
+ const requisition = workdayJobKey(job);
382
390
  return classifyJob({
383
391
  id: `workday:${company.slug}:${requisition}`,
384
392
  company: company.name,
package/src/cli.ts CHANGED
@@ -11,6 +11,8 @@ import { runSourceDiscovery, runYcSourceDiscovery } from "./source-discovery.ts"
11
11
  import { discoverAndPromote } from "./source-discovery-pipeline.ts";
12
12
  import { traceCareerSources } from "./career-tracing.ts";
13
13
  import { discoverCommonCrawlSources } from "./common-crawl-discovery.ts";
14
+ import { discoverCatalog } from "./catalog-discovery.ts";
15
+ import { parseArgs } from "node:util";
14
16
  import { generateYcCompanySeeds } from "./company-seeds.ts";
15
17
  import { exportSnapshot } from "./snapshot-export.ts";
16
18
  import { enrichSourcesFromCompanies } from "./source-enrichment.ts";
@@ -39,6 +41,7 @@ Usage:
39
41
  openings sources discover-yc --country CODE [--registry FILE] [--output FILE] [--catalog FILE] [--report FILE]
40
42
  openings sources seed-companies-yc --country CODE [--output FILE]
41
43
  openings sources discover-common-crawl [--country CODE] [--provider NAME] [--index-record-limit N] [--sample-token-limit N] [--exclude-token TOKEN] [--report-only] [--registry FILE] [--output FILE] [--report FILE] [--index-url URL]
44
+ openings sources discover-catalog --index CC-MAIN-YYYY-NN --provider NAME [repeat index/provider] --state .openings/campaign.sqlite [--execute] [--request-budget N] [--page-budget N] [--target-boards N] [--delay-ms N] [--catalog FILE] [--export .openings/leads.json]
42
45
  openings sources enrich COMPANIES.json [--companies FILE]... [--evidence-kind authoritative_dataset|company_registry] [--registry FILE] [--output FILE] [--report FILE]
43
46
  openings sources trace-careers COMPANIES.json [--country CODE] [--registry FILE] [--common-crawl-report FILE] [--search-key-env NAME] [--output FILE] [--catalog FILE] [--report FILE]
44
47
  openings sources probe-jobposting COMPANIES.md [--catalog FILE] [--report FILE] [--company-limit 10|20]
@@ -223,6 +226,22 @@ export async function run(args: string[]): Promise<number> {
223
226
  console.log(JSON.stringify({ discovery: compactCareerTraceReport(discovery), promotion }, null, 2));
224
227
  return 0;
225
228
  }
229
+ if (rest[0] === "discover-catalog") {
230
+ const { values } = parseArgs({ args: rest.slice(1), strict: true, allowPositionals: false, options: {
231
+ index: { type: "string", multiple: true }, provider: { type: "string", multiple: true }, state: { type: "string" },
232
+ execute: { type: "boolean" }, "request-budget": { type: "string" }, "page-budget": { type: "string" },
233
+ "target-boards": { type: "string" }, "delay-ms": { type: "string" }, catalog: { type: "string" }, export: { type: "string" },
234
+ } });
235
+ if (!values.state) return fail("discover-catalog requires --state under .openings");
236
+ const report = await discoverCatalog({ state: values.state, indexes: values.index ?? [], providers: (values.provider ?? []) as Ats[], execute: values.execute,
237
+ requestBudget: values["request-budget"] === undefined ? undefined : Number(values["request-budget"]),
238
+ pageBudget: values["page-budget"] === undefined ? undefined : Number(values["page-budget"]),
239
+ targetBoards: values["target-boards"] === undefined ? undefined : Number(values["target-boards"]),
240
+ delayMs: values["delay-ms"] === undefined ? undefined : Number(values["delay-ms"]),
241
+ catalogPath: values.catalog ?? "data/companies.json", exportPath: values.export });
242
+ console.log(JSON.stringify(report, null, 2));
243
+ return ["error", "throttled", "cooling_down"].includes(report.status) ? 1 : 0;
244
+ }
226
245
  if (rest[0] === "discover-common-crawl") {
227
246
  const parsed = parseCommonCrawlDiscovery(rest.slice(1));
228
247
  if (typeof parsed === "string") return fail(parsed);
@@ -39,10 +39,11 @@ export interface CommonCrawlDiscoveryReport extends ReportMeta {
39
39
 
40
40
  const providerPatterns: Record<Ats, string[]> = {
41
41
  greenhouse: ["job-boards.greenhouse.io/*", "boards.greenhouse.io/*"], lever: ["jobs.lever.co/*"], ashby: ["jobs.ashbyhq.com/*"], workday: ["*.myworkdayjobs.com/*"], recruitee: ["*.recruitee.com/*"],
42
- ...Object.fromEntries(PROVIDERS.map((spec) => [spec.ats, spec.crawlPatterns])) as Record<"smartrecruiters" | "workable" | "breezy" | "freshteam" | "keka" | "zohorecruit" | "accenture" | "infosys" | "capgemini" | "amazon", string[]>,
42
+ ...Object.fromEntries(PROVIDERS.map((spec) => [spec.ats, spec.crawlPatterns])) as Record<"smartrecruiters" | "workable" | "breezy" | "freshteam" | "keka" | "zohorecruit" | "oraclecloud" | "accenture" | "infosys" | "capgemini" | "amazon", string[]>,
43
43
  jobposting: [], // company sites are found by probing seeds, never by URL pattern
44
44
  };
45
45
  const patterns = Object.values(providerPatterns).flat();
46
+ export function commonCrawlPatterns(provider: Ats): readonly string[] { return providerPatterns[provider]; }
46
47
  const recordsPerPattern = 10_000;
47
48
 
48
49
  export async function discoverCommonCrawlSources(candidatesPath: string, reportPath: string, options: Options = {}): Promise<CommonCrawlDiscoveryReport> {
package/src/locations.ts CHANGED
@@ -15,7 +15,7 @@ const INDIA_PLACES = [
15
15
  const INDIA_PATTERN = new RegExp(`\\b(${INDIA_PLACES.map(escapeRegExp).join("|")})\\b`, "i");
16
16
  const INDIA_INCLUSIVE_REGION_PATTERN = /\b(apac|asia|asia[ -]pacific|worldwide|anywhere|global)\b/i;
17
17
  const INDIA_EXCLUSION_PATTERN = /\b(not available|unavailable|excluding|except|cannot hire|can't hire|unable to hire|do not hire|does not hire)\b.{0,80}\b(india|apac|asia)\b|\b(india|apac|asia)\b.{0,40}\b(excluded|not eligible|not supported)\b/i;
18
- const INDIA_ELIGIBILITY_PATTERN = /\b(remote (?:in|from)|available (?:in|to)|open to|hiring (?:in|from)|candidates? (?:in|from)|applicants? (?:in|from)|work (?:in|from)|based in)\b.{0,80}\b(india|apac|asia)\b|\b(india|apac|asia)\b.{0,40}\b(remote|candidates?|applicants?|eligible|hiring)\b/i;
18
+ const INDIA_ELIGIBILITY_PATTERN = /\b(remote (?:in|from)|available (?:in|to)|open to|hiring (?:in|from)|candidates? (?:in|from)|applicants? (?:in|from)|based in)\b.{0,80}\b(india|apac|asia)\b|\bwork (?:in|from)\s+(?:anywhere\s+in\s+)?(?:india|apac|asia)\b|\b(india|apac|asia)\b.{0,40}\b(remote|candidates?|applicants?|eligible|hiring)\b/i;
19
19
 
20
20
  export function normalizeLocation(value: string): string {
21
21
  const aliases: Record<string, string> = {
@@ -135,7 +135,7 @@ function detectEligibleCountries(description: string): string[] {
135
135
  for (const rule of countryRules()) {
136
136
  if (rule.eligibility.test(description)) matches.push(rule.code);
137
137
  }
138
- if (/\b(open to|hiring|candidates?|applicants?|eligible|remote (?:in|from)|work (?:in|from)|based in|available (?:in|to))\b.{0,80}\bGeorgia\b/i.test(description)) matches.push("GE");
138
+ if (descriptionEligibilityPattern("\\bGeorgia\\b").test(description)) matches.push("GE");
139
139
  return matches;
140
140
  }
141
141
 
@@ -172,6 +172,10 @@ function countryMatchers(): Array<[string, string]> {
172
172
  }
173
173
 
174
174
  interface CountryRule { code: string; location: RegExp; codeLocation: RegExp; exclusion: RegExp; eligibility: RegExp }
175
+ /** "Work in tandem with our India team" describes colleagues, not an applicant's work location. */
176
+ function descriptionEligibilityPattern(country: string): RegExp {
177
+ return new RegExp(`\\b(open to|hiring|candidates?|applicants?|eligible|remote (?:in|from)|based in|available (?:in|to))\\b.{0,80}(?:${country})|\\bwork (?:in|from)\\s+(?:anywhere\\s+in\\s+)?(?:${country})`, "i");
178
+ }
175
179
  let cachedCountryRules: CountryRule[] | undefined;
176
180
  function countryRules(): CountryRule[] {
177
181
  if (cachedCountryRules) return cachedCountryRules;
@@ -180,7 +184,7 @@ function countryRules(): CountryRule[] {
180
184
  location: new RegExp(country, "i"),
181
185
  codeLocation: new RegExp(`(?:^|[,(/-]\\s*)${code === "GB" ? "(?:GB|UK)" : code}(?=\\s*(?:$|[,)/-]))`, "i"),
182
186
  exclusion: new RegExp(`\\b(not available|unavailable|excluding|except|cannot hire|can't hire|unable to hire|do not hire|does not hire)\\b.{0,80}(?:${country})|(?:${country}).{0,40}\\b(excluded|not eligible|not supported)\\b`, "i"),
183
- eligibility: new RegExp(`\\b(open to|hiring|candidates?|applicants?|eligible|remote (?:in|from)|work (?:in|from)|based in|available (?:in|to))\\b.{0,80}(?:${country})`, "i"),
187
+ eligibility: descriptionEligibilityPattern(country),
184
188
  }));
185
189
  return cachedCountryRules;
186
190
  }
package/src/providers.ts CHANGED
@@ -186,6 +186,80 @@ const freshteam: ProviderSpec = {
186
186
  };
187
187
 
188
188
  /** Keka (India HRMS): the careers portal calls an unauthenticated embed API. Token is `tenant/orgId`; the org id sits in the portal shell. */
189
+ /** Oracle Cloud Recruiting (the "CandidateExperience" career sites). Token: "<host>/<siteNumber>", e.g. "jpmc.fa.oraclecloud.com/CX_1". */
190
+ const ORACLE_HOST = /^[a-z0-9-]+\.fa(?:\.[a-z0-9-]+)*\.oraclecloud\.com$/i;
191
+ const ORACLE_PAGE = 200;
192
+ /** Newest first, so a large employer costs a few pages: stop once a whole page predates this. */
193
+ const ORACLE_MAX_AGE_DAYS = 45;
194
+ // Enterprise tenants run to thousands of open roles; this bounds a crawl at 20 requests while covering the window we care about.
195
+ const ORACLE_MAX_RECORDS = 4000;
196
+ const oracleHost = (token: string) => token.split("/")[0] ?? "";
197
+ const oracleSite = (token: string) => token.split("/")[1] ?? "CX_1";
198
+ function oracleList(token: string, offset: number): string {
199
+ return `https://${oracleHost(token)}/hcmRestApi/resources/latest/recruitingCEJobRequisitions?onlyData=true&expand=requisitionList.secondaryLocations` +
200
+ `&finder=findReqs;siteNumber=${encodeURIComponent(oracleSite(token))},limit=${ORACLE_PAGE},offset=${offset},sortBy=POSTING_DATES_DESC`;
201
+ }
202
+ function oracleRequisitions(body: unknown): Rec[] | null {
203
+ if (!isRecord(body)) return null;
204
+ const items = asRecords(body.items) ?? [];
205
+ const list = items.flatMap((item) => asRecords(item.requisitionList) ?? []);
206
+ return items.length ? list : null;
207
+ }
208
+
209
+ const oraclecloud: ProviderSpec = {
210
+ ats: "oraclecloud",
211
+ label: "Oracle Cloud Recruiting",
212
+ hosts: ["oraclecloud.com"],
213
+ crawlPatterns: ["*.fa.oraclecloud.com/hcmUI/CandidateExperience*"],
214
+ resolve(url) {
215
+ if (!ORACLE_HOST.test(url.hostname)) return null;
216
+ const site = /\/sites\/([A-Za-z0-9_]{2,32})/.exec(url.pathname)?.[1] ?? url.searchParams.get("siteNumber");
217
+ return site && /^[A-Za-z0-9_]{2,32}$/.test(site) ? `${url.hostname.toLowerCase()}/${site}` : null;
218
+ },
219
+ canonicalUrl: (token) => `https://${oracleHost(token)}/hcmUI/CandidateExperience/en/sites/${oracleSite(token)}/requisitions`,
220
+ endpoint: (token) => oracleList(token, 0),
221
+ jobsFromBody: (body) => oracleRequisitions(body),
222
+ providerName: () => "",
223
+ payloadVersion: () => "oracle-recruiting-ce:v1",
224
+ async fetchAll(token, get) {
225
+ const records: Rec[] = [];
226
+ const cutoff = Date.now() - ORACLE_MAX_AGE_DAYS * 86_400_000;
227
+ for (let offset = 0; offset < ORACLE_MAX_RECORDS; offset += ORACLE_PAGE) {
228
+ const page = oracleRequisitions(await get(oracleList(token, offset))) ?? [];
229
+ records.push(...page);
230
+ if (page.length < ORACLE_PAGE) break;
231
+ // Sorted newest first: once a whole page is older than the window, later pages are older still.
232
+ const dates = page.map((job) => Date.parse(str(job.PostedDate))).filter((time) => Number.isFinite(time));
233
+ if (dates.length && Math.max(...dates) < cutoff) break;
234
+ }
235
+ return records;
236
+ },
237
+ normalize(company, job) {
238
+ const secondary = (asRecords(job.secondaryLocations) ?? []).map((place) => str(place.Name)).filter(Boolean);
239
+ const country = countryLabel(str(job.PrimaryLocationCountry));
240
+ const primary = str(job.PrimaryLocation);
241
+ // The country name is appended so eligibility reads it: Oracle gives the code separately from the label.
242
+ const location = [primary, ...secondary].filter(Boolean).join("; ") + (country && !new RegExp(`\\b${country}\\b`, "i").test(primary) ? `, ${country}` : "");
243
+ const workplace = str(job.WorkplaceType) || str(job.WorkplaceTypeCode);
244
+ const remote = /remote|work from home/i.test(`${workplace} ${location}`);
245
+ return classifyJob({
246
+ id: `oraclecloud:${company.slug}:${str(job.Id)}`, company: company.name, title: str(job.Title),
247
+ location: location || "Unspecified", remote, workMode: remote ? "remote" : "unknown",
248
+ eligibleCountries: [], excludedCountries: [], eligibleRegions: [], eligibilityConfidence: "unknown",
249
+ url: `https://${oracleHost(company.token)}/hcmUI/CandidateExperience/en/sites/${oracleSite(company.token)}/job/${encodeURIComponent(str(job.Id))}`,
250
+ ...(str(job.PostedDate) ? { updatedAt: str(job.PostedDate) } : {}),
251
+ description: "", // the listing carries only a teaser; detail() reads the advert
252
+ });
253
+ },
254
+ async detail(company, job, get) {
255
+ const id = job.id.split(":").pop() ?? "";
256
+ const body = await get(`https://${oracleHost(company.token)}/hcmRestApi/resources/latest/recruitingCEJobRequisitionDetails?expand=all&onlyData=true&finder=ById;Id="${encodeURIComponent(id)}",siteNumber=${encodeURIComponent(oracleSite(company.token))}`);
257
+ const item = (asRecords(isRecord(body) ? body.items : null) ?? [])[0];
258
+ if (!item) return "";
259
+ return plainText([str(item.ExternalDescriptionStr), str(item.ExternalResponsibilitiesStr), str(item.ExternalQualificationsStr)].filter(Boolean).join("\n\n"));
260
+ },
261
+ };
262
+
189
263
  const keka: ProviderSpec = {
190
264
  ats: "keka",
191
265
  label: "Keka",
@@ -264,7 +338,7 @@ function tag(xml: string, name: string): string { return new RegExp(`<${name}(?:
264
338
  function cdata(value: string): string { return value.replace(/^<!\[CDATA\[([\s\S]*?)\]\]>$/, "$1").trim(); }
265
339
 
266
340
  import { PORTALS } from "./portals.ts";
267
- export const PROVIDERS: ReadonlyArray<ProviderSpec> = [smartrecruiters, workable, breezy, freshteam, keka, zohorecruit, ...PORTALS];
341
+ export const PROVIDERS: ReadonlyArray<ProviderSpec> = [smartrecruiters, workable, breezy, freshteam, keka, zohorecruit, oraclecloud, ...PORTALS];
268
342
 
269
343
  export function providerSpec(ats: string): ProviderSpec | undefined {
270
344
  return PROVIDERS.find((spec) => spec.ats === ats);
package/src/types.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  /** ATS boards plus "jobposting": a company's own career site read only through its schema.org JobPosting markup (token = careers URL). */
2
- export const ALL_PROVIDERS = ["greenhouse", "lever", "ashby", "workday", "recruitee", "smartrecruiters", "workable", "breezy", "freshteam", "keka", "zohorecruit", "jobposting", "accenture", "infosys", "capgemini", "amazon"] as const;
2
+ export const ALL_PROVIDERS = ["greenhouse", "lever", "ashby", "workday", "recruitee", "smartrecruiters", "workable", "breezy", "freshteam", "keka", "zohorecruit", "oraclecloud", "jobposting", "accenture", "infosys", "capgemini", "amazon"] as const;
3
3
  export type Ats = (typeof ALL_PROVIDERS)[number];
4
4
 
5
5
  export interface DomainEvidence {
package/src/version.ts CHANGED
@@ -1 +1 @@
1
- export const VERSION = "0.1.37";
1
+ export const VERSION = "0.1.39";