openings 0.1.18 → 0.1.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.codex-plugin/plugin.json +1 -1
- package/README.md +1 -1
- package/data/companies.json +6525 -0
- package/package.json +1 -1
- package/src/board-verification.ts +5 -4
- package/src/catalog.ts +9 -3
- package/src/cli.ts +36 -0
- package/src/common-crawl-discovery.ts +8 -2
- package/src/country-coverage.ts +1 -1
- package/src/jobposting-site.ts +162 -0
- package/src/keka-tenants.ts +46 -0
- package/src/locations.ts +4 -0
- package/src/providers.ts +84 -3
- package/src/site-admission.ts +39 -0
- package/src/site-probe.ts +43 -0
- package/src/source-enrichment.ts +1 -1
- package/src/source-verification.ts +15 -6
- package/src/types.ts +3 -2
- package/src/version.ts +1 -1
package/package.json
CHANGED
|
@@ -8,7 +8,7 @@ import type { Ats, Company, SourceVerification } from "./types.ts";
|
|
|
8
8
|
type Fetch = (input: string | URL, init?: RequestInit) => Promise<Response>;
|
|
9
9
|
|
|
10
10
|
/** Providers whose public board is accepted as identity on its own. Workday boards are identified by tenant; their crawls are heavier but they carry the large employers. */
|
|
11
|
-
export const BOARD_TIER_PROVIDERS: ReadonlySet<Ats> = new Set(["greenhouse", "lever", "ashby", "recruitee", "smartrecruiters", "workable", "breezy", "workday", "freshteam"]);
|
|
11
|
+
export const BOARD_TIER_PROVIDERS: ReadonlySet<Ats> = new Set(["greenhouse", "lever", "ashby", "recruitee", "smartrecruiters", "workable", "breezy", "workday", "freshteam", "keka", "zohorecruit"]);
|
|
12
12
|
|
|
13
13
|
export interface BoardVerificationOptions {
|
|
14
14
|
fetch?: Fetch;
|
|
@@ -129,14 +129,15 @@ async function probeBoard(lead: EnrichmentLead, fetcher: Fetch, timeoutMs: numbe
|
|
|
129
129
|
: { signal: controller.signal };
|
|
130
130
|
const response = await fetcher(source.structuredEndpoint, init);
|
|
131
131
|
if (!response.ok) { await response.body?.cancel().catch(() => undefined); throw new BoardError(response.status === 404 || response.status === 410 ? "invalid_payload" : "unreachable", `HTTP ${response.status}`); }
|
|
132
|
-
let body: unknown;
|
|
133
|
-
try { body = await response.json(); } catch { throw new BoardError("invalid_payload", "Endpoint did not return JSON"); }
|
|
134
132
|
const spec = providerSpec(source.ats);
|
|
133
|
+
let body: unknown;
|
|
134
|
+
try { body = spec?.bodyFormat === "text" ? await response.text() : await response.json(); } catch { throw new BoardError("invalid_payload", "Endpoint did not return JSON"); }
|
|
135
135
|
const jobs = spec ? spec.jobsFromBody(body) : source.ats === "lever" ? asArray(body) : asArray(isRecord(body) ? body[source.ats === "recruitee" ? "offers" : source.ats === "workday" ? "jobPostings" : "jobs"] : undefined);
|
|
136
136
|
if (!jobs) throw new BoardError("invalid_payload", "Payload does not contain the expected jobs array");
|
|
137
137
|
if (jobs.length === 0) throw new BoardError("empty_board", "Board has no jobs");
|
|
138
|
-
|
|
138
|
+
let providerName = spec ? spec.providerName(jobs, body) : source.ats === "greenhouse" ? majority(jobs.map((job) => typeof job.company_name === "string" ? job.company_name.trim() : "").filter(Boolean)) : source.ats === "workday" ? humanize(source.token.split("/")[1] ?? source.token) : "";
|
|
139
139
|
const payloadVersion = spec ? spec.payloadVersion(body) : source.ats === "greenhouse" ? "greenhouse-job-board:v1" : source.ats === "lever" ? "lever-postings:v0" : source.ats === "recruitee" ? "recruitee-careers:v1" : source.ats === "workday" ? "workday-cxs:v1" : `ashby-job-board:${isRecord(body) && typeof body.apiVersion === "string" ? body.apiVersion : "unknown"}`;
|
|
140
|
+
if (spec?.companyInfo && !providerName) providerName = (await spec.companyInfo(source.token, async (url, format = "json") => { const reply = await fetcher(url, { signal: controller.signal }); if (!reply.ok) { await reply.body?.cancel().catch(() => undefined); throw new BoardError("unreachable", `HTTP ${reply.status}`); } return format === "text" ? reply.text() : reply.json(); }).catch(() => ({ name: "" }))).name;
|
|
140
141
|
const jobCount = source.ats === "workday" && isRecord(body) && typeof body.total === "number" ? body.total : jobs.length;
|
|
141
142
|
return { providerName, contentType: response.headers.get("content-type") ?? "unknown", payloadVersion, jobCount };
|
|
142
143
|
} finally { clearTimeout(timer); }
|
package/src/catalog.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import type { Company, Job, JobSummary, SearchQuery } from "./types.ts";
|
|
2
2
|
import { providerSpec, type JsonGet } from "./providers.ts";
|
|
3
|
+
import { crawlSite, sitePostingsToJobs } from "./jobposting-site.ts";
|
|
3
4
|
import { classifyJob, isEligibleForCountry, normalizeLocation } from "./locations.ts";
|
|
4
5
|
|
|
5
6
|
type Fetch = (input: string | URL, init?: RequestInit) => Promise<Response>;
|
|
@@ -134,10 +135,15 @@ export interface FetchJobsObserver {
|
|
|
134
135
|
}
|
|
135
136
|
export async function fetchSourceJobs(company: Company, fetcher: Fetch = globalThis.fetch, signal?: AbortSignal, observer?: FetchJobsObserver): Promise<Job[]> {
|
|
136
137
|
if (company.ats === "workday") return fetchWorkdayJobs(company, fetcher, signal, observer);
|
|
138
|
+
if (company.ats === "jobposting") {
|
|
139
|
+
if (!company.companyDomain) throw new Error(`${company.name} company site source has no company domain`);
|
|
140
|
+
const site = await crawlSite({ companyName: company.name, companyDomain: company.companyDomain, careerUrl: company.token }, { maxPages: 60, delayMs: 300, signal });
|
|
141
|
+
return sitePostingsToJobs(company, site.postings);
|
|
142
|
+
}
|
|
137
143
|
const spec = providerSpec(company.ats);
|
|
138
144
|
if (spec) {
|
|
139
145
|
const get = jsonGetter(fetcher, company.name, signal, observer);
|
|
140
|
-
const records = spec.fetchAll ? await spec.fetchAll(company.token, get) : spec.jobsFromBody(await get(spec.endpoint(company.token)));
|
|
146
|
+
const records = spec.fetchAll ? await spec.fetchAll(company.token, get) : spec.jobsFromBody(await get(spec.endpoint(company.token), spec.bodyFormat ?? "json"));
|
|
141
147
|
if (!records) throw new Error(`${company.name} job board returned an invalid payload`);
|
|
142
148
|
return records.map((record) => spec.normalize(company, record));
|
|
143
149
|
}
|
|
@@ -243,10 +249,10 @@ async function fetchWorkdayJobs(company: Company, fetcher: Fetch, signal?: Abort
|
|
|
243
249
|
|
|
244
250
|
/** JSON fetch with the catalog's retry and backoff, shaped for the table-driven providers. */
|
|
245
251
|
function jsonGetter(fetcher: Fetch, companyName: string, signal?: AbortSignal, observer?: FetchJobsObserver): JsonGet {
|
|
246
|
-
return async (url) => {
|
|
252
|
+
return async (url, format = "json") => {
|
|
247
253
|
const response = await fetchWithRetry(fetcher, url, signal ? { signal } : undefined, companyName, observer);
|
|
248
254
|
if (!response.ok) { await response.body?.cancel().catch(() => undefined); throw new Error(`${companyName} job board returned HTTP ${response.status}`); }
|
|
249
|
-
return response.json();
|
|
255
|
+
return format === "text" ? response.text() : response.json();
|
|
250
256
|
};
|
|
251
257
|
}
|
|
252
258
|
|
package/src/cli.ts
CHANGED
|
@@ -17,6 +17,9 @@ import { enrichSourcesFromCompanies } from "./source-enrichment.ts";
|
|
|
17
17
|
import { forceReleaseFileLock, inspectFileLock } from "./file-lock.ts";
|
|
18
18
|
import { generateCountryCoverageReport } from "./country-coverage.ts";
|
|
19
19
|
import { probeJobPostingJsonLd } from "./jobposting-probe.ts";
|
|
20
|
+
import { probeSites } from "./site-probe.ts";
|
|
21
|
+
import { admitSites } from "./site-admission.ts";
|
|
22
|
+
import { kekaTenantCandidates } from "./keka-tenants.ts";
|
|
20
23
|
import { mergeAttemptedRoundLeads, prepareRecruiteeRoundArtifacts } from "./recruitee-round.ts";
|
|
21
24
|
|
|
22
25
|
const HELP = `Openings — search public company job boards
|
|
@@ -35,6 +38,9 @@ Usage:
|
|
|
35
38
|
openings sources enrich COMPANIES.json [--companies FILE]... [--evidence-kind authoritative_dataset|company_registry] [--registry FILE] [--output FILE] [--report FILE]
|
|
36
39
|
openings sources trace-careers COMPANIES.json [--country CODE] [--registry FILE] [--common-crawl-report FILE] [--search-key-env NAME] [--output FILE] [--catalog FILE] [--report FILE]
|
|
37
40
|
openings sources probe-jobposting COMPANIES.md [--catalog FILE] [--report FILE] [--company-limit 10|20]
|
|
41
|
+
openings sources probe-sites SEEDS.json [--sample N] [--limit N] [--concurrency N] [--max-pages N] [--delay-ms N] [--report FILE]
|
|
42
|
+
openings sources admit-sites REPORT.json [--catalog FILE] [--min-postings N]
|
|
43
|
+
openings sources keka-tenants HOSTS.txt [--output FILE] [--registry FILE] [--concurrency N]
|
|
38
44
|
openings sources prepare-recruitee-round IDENTITIES.json [--catalog FILE] [--artifacts DIR]
|
|
39
45
|
openings sources merge-attempted-round-leads ISOLATED_REGISTRY [--registry FILE]
|
|
40
46
|
openings search [words] [--country CODE|--india] [--location PLACE] [--remote|--onsite]
|
|
@@ -114,6 +120,36 @@ export async function run(args: string[]): Promise<number> {
|
|
|
114
120
|
console.log(JSON.stringify(await mergeAttemptedRoundLeads(parsed.isolatedRegistryPath, parsed.registry), null, 2));
|
|
115
121
|
return 0;
|
|
116
122
|
}
|
|
123
|
+
if (rest[0] === "probe-sites") {
|
|
124
|
+
const args = rest.slice(1);
|
|
125
|
+
const inputPath = args[0];
|
|
126
|
+
if (!inputPath || inputPath.startsWith("--")) return fail("sources probe-sites requires a seeds JSON file");
|
|
127
|
+
const num = (flag: string) => { const index = args.indexOf(flag); return index >= 0 ? Number(args[index + 1]) : undefined; };
|
|
128
|
+
const reportIndex = args.indexOf("--report");
|
|
129
|
+
const report = await probeSites(inputPath, reportIndex >= 0 ? args[reportIndex + 1]! : ".openings/site-probe.json", { sample: num("--sample"), limit: num("--limit"), concurrency: num("--concurrency"), maxPages: num("--max-pages"), delayMs: num("--delay-ms") });
|
|
130
|
+
const { sites, ...summary } = report;
|
|
131
|
+
console.log(JSON.stringify({ ...summary, topSites: sites.slice(0, 15) }, null, 2));
|
|
132
|
+
return 0;
|
|
133
|
+
}
|
|
134
|
+
if (rest[0] === "keka-tenants") {
|
|
135
|
+
const args = rest.slice(1);
|
|
136
|
+
const hostsPath = args[0];
|
|
137
|
+
if (!hostsPath || hostsPath.startsWith("--")) return fail("sources keka-tenants requires a hosts file, one tenant host per line");
|
|
138
|
+
const outputIndex = args.indexOf("--output");
|
|
139
|
+
const concurrencyIndex = args.indexOf("--concurrency");
|
|
140
|
+
const registryIndex = args.indexOf("--registry");
|
|
141
|
+
console.log(JSON.stringify(await kekaTenantCandidates(hostsPath, outputIndex >= 0 ? args[outputIndex + 1]! : "data/source-candidates.json", { concurrency: concurrencyIndex >= 0 ? Number(args[concurrencyIndex + 1]) : undefined, registryPath: registryIndex >= 0 ? args[registryIndex + 1] : undefined }), null, 2));
|
|
142
|
+
return 0;
|
|
143
|
+
}
|
|
144
|
+
if (rest[0] === "admit-sites") {
|
|
145
|
+
const args = rest.slice(1);
|
|
146
|
+
const reportPath = args[0];
|
|
147
|
+
if (!reportPath || reportPath.startsWith("--")) return fail("sources admit-sites requires a probe report JSON file");
|
|
148
|
+
const catalogIndex = args.indexOf("--catalog");
|
|
149
|
+
const minIndex = args.indexOf("--min-postings");
|
|
150
|
+
console.log(JSON.stringify(await admitSites(reportPath, catalogIndex >= 0 ? args[catalogIndex + 1]! : "data/companies.json", { minPostings: minIndex >= 0 ? Number(args[minIndex + 1]) : undefined }), null, 2));
|
|
151
|
+
return 0;
|
|
152
|
+
}
|
|
117
153
|
if (rest[0] === "probe-jobposting") {
|
|
118
154
|
const parsed = parseJobPostingProbe(rest.slice(1));
|
|
119
155
|
if (typeof parsed === "string") return fail(parsed);
|
|
@@ -39,7 +39,8 @@ export interface CommonCrawlDiscoveryReport extends ReportMeta {
|
|
|
39
39
|
|
|
40
40
|
const providerPatterns: Record<Ats, string[]> = {
|
|
41
41
|
greenhouse: ["job-boards.greenhouse.io/*", "boards.greenhouse.io/*"], lever: ["jobs.lever.co/*"], ashby: ["jobs.ashbyhq.com/*"], workday: ["*.myworkdayjobs.com/*"], recruitee: ["*.recruitee.com/*"],
|
|
42
|
-
...Object.fromEntries(PROVIDERS.map((spec) => [spec.ats, spec.crawlPatterns])) as Record<"smartrecruiters" | "workable" | "breezy" | "freshteam", string[]>,
|
|
42
|
+
...Object.fromEntries(PROVIDERS.map((spec) => [spec.ats, spec.crawlPatterns])) as Record<"smartrecruiters" | "workable" | "breezy" | "freshteam" | "keka" | "zohorecruit", string[]>,
|
|
43
|
+
jobposting: [], // company sites are found by probing seeds, never by URL pattern
|
|
43
44
|
};
|
|
44
45
|
const patterns = Object.values(providerPatterns).flat();
|
|
45
46
|
const recordsPerPattern = 10_000;
|
|
@@ -70,7 +71,12 @@ export async function discoverCommonCrawlSources(candidatesPath: string, reportP
|
|
|
70
71
|
query.searchParams.set("collapse", "urlkey");
|
|
71
72
|
query.searchParams.set("fl", "url");
|
|
72
73
|
query.searchParams.set("limit", String(indexRecordLimit));
|
|
73
|
-
|
|
74
|
+
let response = await fetcher(query);
|
|
75
|
+
for (let attempt = 1; !response.ok && response.status >= 500 && attempt <= 3; attempt += 1) { // the index server sheds load with 502s; back off and retry
|
|
76
|
+
await response.body?.cancel().catch(() => undefined);
|
|
77
|
+
await new Promise((resolve) => setTimeout(resolve, attempt * 5_000));
|
|
78
|
+
response = await fetcher(query);
|
|
79
|
+
}
|
|
74
80
|
if (!response.ok) throw new Error(`Common Crawl index returned HTTP ${response.status} for ${pattern}`);
|
|
75
81
|
const lines = (await response.text()).split(/\r?\n/).filter(Boolean).slice(0, indexRecordLimit);
|
|
76
82
|
indexRecordsExamined += lines.length;
|
package/src/country-coverage.ts
CHANGED
|
@@ -192,7 +192,7 @@ function validLastCrawl(value: Record<string, unknown>): boolean {
|
|
|
192
192
|
}
|
|
193
193
|
function validVerification(value: unknown): boolean {
|
|
194
194
|
return isRecord(value) && validDate(value.checkedAt) && typeof value.canonicalSourceUrl === "string" && typeof value.observedCompanyName === "string"
|
|
195
|
-
&& ["provider_company_name", "provider_tenant", "structured_domain_link", "company_redirect"].includes(String(value.identityEvidence))
|
|
195
|
+
&& ["provider_company_name", "provider_tenant", "structured_domain_link", "company_redirect", "company_page_link", "provider_board", "company_site"].includes(String(value.identityEvidence))
|
|
196
196
|
&& typeof value.contentType === "string" && typeof value.payloadVersion === "string" && nonNegativeInteger(value.jobCount);
|
|
197
197
|
}
|
|
198
198
|
function validDate(value: unknown): value is string { return typeof value === "string" && Number.isFinite(Date.parse(value)); }
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
import { classifyJob } from "./locations.ts";
|
|
2
|
+
import { countryLabel } from "./providers.ts";
|
|
3
|
+
import { extractLinks, fetchSafePage, robotsAllows, type PageTransport, type ResolveHost } from "./safe-head.ts";
|
|
4
|
+
import type { Job } from "./types.ts";
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Company-owned career sites: the only thing read is the schema.org JobPosting JSON-LD block that employers publish
|
|
8
|
+
* for search engines. Discovery is bounded (robots.txt honoured, same company domain, a page cap, a delay between
|
|
9
|
+
* requests) and page prose is never interpreted.
|
|
10
|
+
*/
|
|
11
|
+
export interface SitePosting { title: string; company?: string; location: string; url: string; identifier?: string; datePosted?: string; validThrough?: string; description: string; remote: boolean }
|
|
12
|
+
export interface SiteSeed { companyName: string; companyDomain: string; careerUrl?: string }
|
|
13
|
+
export interface SiteCrawlOptions { resolveHost?: ResolveHost; pageTransport?: PageTransport; timeoutMs?: number; maxPages?: number; delayMs?: number; sleep?: (ms: number) => Promise<void>; signal?: AbortSignal }
|
|
14
|
+
export interface SiteCrawlResult { careerUrl?: string; robotsBlocked: boolean; candidatePages: number; pagesFetched: number; postings: SitePosting[]; issues: string[] }
|
|
15
|
+
|
|
16
|
+
const JOB_PATH = /job|career|opening|vacanc|position|recruit|apply|hiring/i;
|
|
17
|
+
|
|
18
|
+
export function ownedBy(url: string, companyDomain: string): boolean {
|
|
19
|
+
try {
|
|
20
|
+
const host = new URL(url).hostname.toLowerCase().replace(/^www\./, "");
|
|
21
|
+
const domain = companyDomain.toLowerCase().replace(/^www\./, "");
|
|
22
|
+
return host === domain || host.endsWith(`.${domain}`);
|
|
23
|
+
} catch { return false; }
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/** JobPosting objects from every ld+json block on a page, including @graph, arrays, and ItemList wrappers. */
|
|
27
|
+
export function extractJobPostings(html: string, pageUrl: string): SitePosting[] {
|
|
28
|
+
const postings: SitePosting[] = [];
|
|
29
|
+
for (const match of html.matchAll(/<script[^>]*type\s*=\s*["']application\/ld\+json["'][^>]*>([\s\S]*?)<\/script>/gi)) {
|
|
30
|
+
let value: unknown;
|
|
31
|
+
try { value = JSON.parse(match[1]!.trim().replace(/^<!--|-->$/g, "")); } catch { continue; }
|
|
32
|
+
for (const node of walk(value)) {
|
|
33
|
+
const type = node["@type"];
|
|
34
|
+
const types = Array.isArray(type) ? type.map(String) : [String(type ?? "")];
|
|
35
|
+
if (!types.some((entry) => entry.toLowerCase() === "jobposting")) continue;
|
|
36
|
+
const posting = normalizePosting(node, pageUrl);
|
|
37
|
+
if (posting) postings.push(posting);
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
return postings;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
function* walk(value: unknown, depth = 0): Generator<Record<string, unknown>> {
|
|
44
|
+
if (depth > 6) return;
|
|
45
|
+
if (Array.isArray(value)) { for (const item of value) yield* walk(item, depth + 1); return; }
|
|
46
|
+
if (!isRecord(value)) return;
|
|
47
|
+
yield value;
|
|
48
|
+
for (const key of ["@graph", "itemListElement", "item", "mainEntity", "hasPart"]) if (key in value) yield* walk(value[key], depth + 1);
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
function normalizePosting(node: Record<string, unknown>, pageUrl: string): SitePosting | null {
|
|
52
|
+
const title = text(node.title) || text(node.name);
|
|
53
|
+
if (!title) return null;
|
|
54
|
+
const organization = isRecord(node.hiringOrganization) ? text(node.hiringOrganization.name) : text(node.hiringOrganization);
|
|
55
|
+
const locations = (Array.isArray(node.jobLocation) ? node.jobLocation : [node.jobLocation]).filter(isRecord).map(placeText).filter(Boolean);
|
|
56
|
+
const remote = String(node.jobLocationType ?? "").toUpperCase().includes("TELECOMMUTE");
|
|
57
|
+
const applicant = isRecord(node.applicantLocationRequirements) ? [node.applicantLocationRequirements] : Array.isArray(node.applicantLocationRequirements) ? node.applicantLocationRequirements.filter(isRecord) : [];
|
|
58
|
+
const applicantText = applicant.map((entry) => text(entry.name)).filter(Boolean).join("; ");
|
|
59
|
+
const location = locations.join("; ") || (remote ? `Remote${applicantText ? ` (${applicantText})` : ""}` : "Unspecified");
|
|
60
|
+
const identifier = isRecord(node.identifier) ? text(node.identifier.value) || text(node.identifier.name) : text(node.identifier);
|
|
61
|
+
let url = text(node.url) || (isRecord(node.mainEntityOfPage) ? text(node.mainEntityOfPage["@id"]) : text(node.mainEntityOfPage)) || pageUrl;
|
|
62
|
+
try { url = new URL(url, pageUrl).href; } catch { url = pageUrl; }
|
|
63
|
+
return {
|
|
64
|
+
title, company: organization || undefined, location, url, identifier: identifier || undefined,
|
|
65
|
+
datePosted: isoDate(node.datePosted), validThrough: isoDate(node.validThrough),
|
|
66
|
+
description: stripHtml(text(node.description)).slice(0, 20_000), remote,
|
|
67
|
+
};
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
function placeText(place: Record<string, unknown>): string {
|
|
71
|
+
const address = isRecord(place.address) ? place.address : place;
|
|
72
|
+
const parts = [text(address.addressLocality), text(address.addressRegion), text(address.addressCountry) ? countryLabel(text(address.addressCountry)) : ""].filter(Boolean);
|
|
73
|
+
if (parts.length) return parts.join(", ");
|
|
74
|
+
return text(place.name) || (typeof place.address === "string" ? place.address : "");
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/** Find the careers page, gather job-like links from it and from sitemaps, and read the JobPosting blocks on each. */
|
|
78
|
+
export async function crawlSite(seed: SiteSeed, options: SiteCrawlOptions = {}): Promise<SiteCrawlResult> {
|
|
79
|
+
const maxPages = options.maxPages ?? 25;
|
|
80
|
+
const sleep = options.sleep ?? ((ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms)));
|
|
81
|
+
const fetchOptions = { resolveHost: options.resolveHost, transport: options.pageTransport, timeoutMs: options.timeoutMs ?? 15_000 };
|
|
82
|
+
const result: SiteCrawlResult = { robotsBlocked: false, candidatePages: 0, pagesFetched: 0, postings: [], issues: [] };
|
|
83
|
+
let requests = 0;
|
|
84
|
+
const get = async (url: string) => {
|
|
85
|
+
if (options.signal?.aborted) throw options.signal.reason instanceof Error ? options.signal.reason : new Error("Crawl aborted");
|
|
86
|
+
if (requests > 0 && options.delayMs) await sleep(options.delayMs);
|
|
87
|
+
requests += 1;
|
|
88
|
+
if (!(await robotsAllows(url, fetchOptions))) { result.robotsBlocked = true; return null; }
|
|
89
|
+
const page = await fetchSafePage(url, fetchOptions);
|
|
90
|
+
if (page.status !== 200 || !ownedBy(page.finalUrl, seed.companyDomain)) return null;
|
|
91
|
+
return page;
|
|
92
|
+
};
|
|
93
|
+
const candidates = seed.careerUrl ? [seed.careerUrl] : ["careers", "career", "jobs"].map((path) => `https://${seed.companyDomain}/${path}`);
|
|
94
|
+
let landing: { finalUrl: string; html: string } | null = null;
|
|
95
|
+
for (const url of candidates) {
|
|
96
|
+
try { landing = await get(url); if (landing) { result.careerUrl = landing.finalUrl; break; } }
|
|
97
|
+
catch (error) { result.issues.push(`${url}: ${error instanceof Error ? error.message : String(error)}`); }
|
|
98
|
+
}
|
|
99
|
+
const seen = new Set<string>();
|
|
100
|
+
const pages: string[] = [];
|
|
101
|
+
const consider = (url: string) => { if (!seen.has(url) && ownedBy(url, seed.companyDomain) && JOB_PATH.test(new URL(url).pathname)) { seen.add(url); pages.push(url); } };
|
|
102
|
+
if (landing) {
|
|
103
|
+
result.postings.push(...extractJobPostings(landing.html, landing.finalUrl));
|
|
104
|
+
result.pagesFetched += 1;
|
|
105
|
+
for (const link of extractLinks(landing.html, landing.finalUrl)) consider(link);
|
|
106
|
+
}
|
|
107
|
+
// Sitemaps list the job-detail pages that carry the markup; read at most three of them.
|
|
108
|
+
const origin = `https://${seed.companyDomain.replace(/^www\./, "")}`;
|
|
109
|
+
const sitemaps = await sitemapUrls(origin, fetchOptions);
|
|
110
|
+
let sitemapReads = 0;
|
|
111
|
+
for (const sitemap of sitemaps) {
|
|
112
|
+
if (sitemapReads >= 3 || pages.length >= maxPages * 4) break;
|
|
113
|
+
try {
|
|
114
|
+
const page = await get(sitemap);
|
|
115
|
+
if (!page) continue;
|
|
116
|
+
sitemapReads += 1;
|
|
117
|
+
for (const loc of page.html.matchAll(/<loc>\s*([^<\s]+)\s*<\/loc>/gi)) {
|
|
118
|
+
const url = loc[1]!;
|
|
119
|
+
if (/sitemap/i.test(new URL(url).pathname) && JOB_PATH.test(url) && sitemapReads < 3 && !sitemaps.includes(url)) sitemaps.push(url);
|
|
120
|
+
else consider(url);
|
|
121
|
+
}
|
|
122
|
+
} catch (error) { result.issues.push(`${sitemap}: ${error instanceof Error ? error.message : String(error)}`); }
|
|
123
|
+
}
|
|
124
|
+
result.candidatePages = pages.length;
|
|
125
|
+
for (const url of pages.slice(0, maxPages)) {
|
|
126
|
+
try {
|
|
127
|
+
const page = await get(url);
|
|
128
|
+
if (!page) continue;
|
|
129
|
+
result.pagesFetched += 1;
|
|
130
|
+
result.postings.push(...extractJobPostings(page.html, page.finalUrl));
|
|
131
|
+
} catch (error) { result.issues.push(`${url}: ${error instanceof Error ? error.message : String(error)}`); }
|
|
132
|
+
}
|
|
133
|
+
const unique = new Map<string, SitePosting>();
|
|
134
|
+
for (const posting of result.postings) unique.set(posting.identifier ? `id:${posting.identifier}` : `url:${posting.url}:${posting.title}`, posting);
|
|
135
|
+
result.postings = [...unique.values()];
|
|
136
|
+
return result;
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
async function sitemapUrls(origin: string, fetchOptions: Parameters<typeof fetchSafePage>[1]): Promise<string[]> {
|
|
140
|
+
const urls = new Set<string>([`${origin}/sitemap.xml`]);
|
|
141
|
+
try {
|
|
142
|
+
const robots = await fetchSafePage(`${origin}/robots.txt`, { ...fetchOptions, maxBytes: 64_000 });
|
|
143
|
+
if (robots.status === 200) for (const match of robots.html.matchAll(/^\s*sitemap:\s*(\S+)/gim)) urls.add(match[1]!);
|
|
144
|
+
} catch { /* no robots: conventional sitemap only */ }
|
|
145
|
+
return [...urls];
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/** Jobs for the catalog from one company site, in the same shape every ATS adapter produces. */
|
|
149
|
+
export function sitePostingsToJobs(company: { slug: string; name: string }, postings: SitePosting[]): Job[] {
|
|
150
|
+
return postings.map((posting) => classifyJob({
|
|
151
|
+
id: `jobposting:${company.slug}:${posting.identifier ?? hash(posting.url + posting.title)}`,
|
|
152
|
+
company: company.name, title: posting.title, location: posting.location, remote: posting.remote, workMode: posting.remote ? "remote" : "unknown",
|
|
153
|
+
eligibleCountries: [], excludedCountries: [], eligibleRegions: [], eligibilityConfidence: "unknown",
|
|
154
|
+
url: posting.url, ...(posting.datePosted ? { updatedAt: posting.datePosted } : {}), description: posting.description,
|
|
155
|
+
}));
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
function hash(value: string): string { let h = 2166136261; for (const ch of value) { h ^= ch.charCodeAt(0); h = Math.imul(h, 16777619) >>> 0; } return h.toString(16); }
|
|
159
|
+
function text(value: unknown): string { return typeof value === "string" ? value.trim() : typeof value === "number" ? String(value) : ""; }
|
|
160
|
+
function isoDate(value: unknown): string | undefined { const time = Date.parse(text(value)); return Number.isFinite(time) ? new Date(time).toISOString() : undefined; }
|
|
161
|
+
function stripHtml(value: string): string { return value.replace(/<[^>]+>/g, " ").replace(/ /g, " ").replace(/&/g, "&").replace(/</g, "<").replace(/>/g, ">").replace(/\s+/g, " ").trim(); }
|
|
162
|
+
function isRecord(value: unknown): value is Record<string, unknown> { return typeof value === "object" && value !== null && !Array.isArray(value); }
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { readFile } from "node:fs/promises";
|
|
2
|
+
import { atomicJson } from "./atomic-file.ts";
|
|
3
|
+
import { providerSpec } from "./providers.ts";
|
|
4
|
+
import { mergeEnrichmentLeads, type EnrichmentLead } from "./enrichment-registry.ts";
|
|
5
|
+
import type { SourceCandidate } from "./types.ts";
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Keka tenants are discoverable by host, but a board token needs the org id that only the careers portal shell carries.
|
|
9
|
+
* One fetch per tenant turns a host list into verifiable candidates with the company name and, when Keka exposes it,
|
|
10
|
+
* the company's own website as the domain to verify against.
|
|
11
|
+
*/
|
|
12
|
+
export async function kekaTenantCandidates(hostsPath: string, outputPath: string, options: { fetcher?: typeof fetch; concurrency?: number; timeoutMs?: number; registryPath?: string } = {}): Promise<{ hosts: number; candidates: number; leads: number; failed: number }> {
|
|
13
|
+
const fetcher = options.fetcher ?? globalThis.fetch;
|
|
14
|
+
const hosts = [...new Set((await readFile(hostsPath, "utf8")).split(/\r?\n/).map((line) => line.trim().toLowerCase()).filter((line) => /^[a-z0-9-]+\.keka\.com$/.test(line)))];
|
|
15
|
+
const spec = providerSpec("keka")!;
|
|
16
|
+
const candidates: SourceCandidate[] = [];
|
|
17
|
+
const leads: EnrichmentLead[] = [];
|
|
18
|
+
let failed = 0;
|
|
19
|
+
let cursor = 0;
|
|
20
|
+
async function worker() {
|
|
21
|
+
while (cursor < hosts.length) {
|
|
22
|
+
const host = hosts[cursor++]!;
|
|
23
|
+
const tenant = host.split(".")[0]!;
|
|
24
|
+
try {
|
|
25
|
+
const get = async (url: string, format: "json" | "text" = "json") => { const reply = await fetcher(url, { signal: AbortSignal.timeout(options.timeoutMs ?? 15_000), headers: { accept: format === "text" ? "text/html" : "application/json" } }); if (!reply.ok) throw new Error(`HTTP ${reply.status}`); return format === "text" ? reply.text() : reply.json(); };
|
|
26
|
+
const shell = await get(`https://${host}/careers`, "text") as string;
|
|
27
|
+
const org = /\/ats\/documents\/([0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})\//i.exec(shell)?.[1]?.toLowerCase();
|
|
28
|
+
if (!org) { failed += 1; continue; }
|
|
29
|
+
const info = await spec.companyInfo!(`${tenant}/${org}`, get);
|
|
30
|
+
if (!info.name) { failed += 1; continue; }
|
|
31
|
+
const token = `${tenant}/${org}`;
|
|
32
|
+
const sourceUrl = spec.endpoint(token);
|
|
33
|
+
if (info.website) {
|
|
34
|
+
candidates.push({ companyName: info.name, companyDomain: info.website, sourceUrl, cohorts: ["IN"], discoveredFrom: { channel: "provider_directory", reference: `https://${host}/careers` } });
|
|
35
|
+
} else {
|
|
36
|
+
// No company website to verify against: the board tier can still admit it on the provider's own identity.
|
|
37
|
+
leads.push({ sourceKey: `keka:${token.toLowerCase()}`, sourceUrl, ats: "keka", token, discoveredFrom: [{ channel: "provider_directory", reference: `https://${host}/careers` }], companyMatches: [], identityEvidence: [], attempts: [] });
|
|
38
|
+
}
|
|
39
|
+
} catch { failed += 1; }
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
await Promise.all(Array.from({ length: Math.max(1, options.concurrency ?? 6) }, worker));
|
|
43
|
+
await atomicJson(outputPath, candidates);
|
|
44
|
+
if (options.registryPath && leads.length) await mergeEnrichmentLeads(options.registryPath, leads);
|
|
45
|
+
return { hosts: hosts.length, candidates: candidates.length, leads: leads.length, failed };
|
|
46
|
+
}
|
package/src/locations.ts
CHANGED
|
@@ -123,9 +123,13 @@ function detectCountries(location: string): string[] {
|
|
|
123
123
|
return [...new Set(["US", ...byCode.filter((code) => !US_STATE_CODES.has(code))])];
|
|
124
124
|
}
|
|
125
125
|
if (byName.includes("US")) return [...new Set([...byName, ...byCode.filter((code) => !US_STATE_CODES.has(code))])];
|
|
126
|
+
// "Hyderabad, TG, India": a bare code next to a named India is an Indian state (TG Telangana, KA Karnataka), not Togo or Kazakhstan.
|
|
127
|
+
if (byName.includes("IN") || INDIA_PATTERN.test(location)) return [...new Set([...byName, ...byCode.filter((code) => !INDIA_STATE_CODES.has(code))])];
|
|
126
128
|
return [...new Set([...byName, ...byCode])];
|
|
127
129
|
}
|
|
128
130
|
|
|
131
|
+
const INDIA_STATE_CODES = new Set(["AP", "AR", "AS", "BR", "CG", "CT", "GA", "GJ", "HR", "HP", "JH", "KA", "KL", "MP", "MH", "MN", "ML", "MZ", "NL", "OD", "OR", "PB", "RJ", "SK", "TN", "TG", "TS", "TR", "UP", "UK", "UT", "WB", "AN", "CH", "DN", "DD", "DL", "JK", "LA", "LD", "PY"]);
|
|
132
|
+
|
|
129
133
|
function detectEligibleCountries(description: string): string[] {
|
|
130
134
|
const matches: string[] = [];
|
|
131
135
|
for (const rule of countryRules()) {
|
package/src/providers.ts
CHANGED
|
@@ -8,8 +8,8 @@ import type { Ats, Company, Job } from "./types.ts";
|
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
10
|
type Rec = Record<string, unknown>;
|
|
11
|
-
/** Fetches a URL and returns its parsed JSON body; the caller supplies retry and pacing. */
|
|
12
|
-
export type JsonGet = (url: string) => Promise<unknown>;
|
|
11
|
+
/** Fetches a URL and returns its parsed JSON body (or the raw text when asked); the caller supplies retry and pacing. */
|
|
12
|
+
export type JsonGet = (url: string, format?: "json" | "text") => Promise<unknown>;
|
|
13
13
|
|
|
14
14
|
export interface ProviderSpec {
|
|
15
15
|
ats: Ats;
|
|
@@ -21,8 +21,12 @@ export interface ProviderSpec {
|
|
|
21
21
|
resolve(url: URL): string | null;
|
|
22
22
|
canonicalUrl(token: string): string;
|
|
23
23
|
endpoint(token: string): string;
|
|
24
|
+
/** "text" when the structured endpoint is not JSON (an RSS feed); `jobsFromBody` then receives the raw string. */
|
|
25
|
+
bodyFormat?: "json" | "text";
|
|
24
26
|
jobsFromBody(body: unknown): Rec[] | null;
|
|
25
27
|
providerName(jobs: Rec[], body: unknown): string;
|
|
28
|
+
/** Company identity from a separate tenant endpoint when the jobs payload carries no name; `website` is the company's own domain when the provider exposes it. */
|
|
29
|
+
companyInfo?(token: string, get: JsonGet): Promise<{ name: string; website?: string }>;
|
|
26
30
|
payloadVersion(body: unknown): string;
|
|
27
31
|
normalize(company: Company, job: Rec): Job;
|
|
28
32
|
/** Fetches every record when the endpoint paginates. Defaults to one request to `endpoint`. */
|
|
@@ -179,7 +183,84 @@ const freshteam: ProviderSpec = {
|
|
|
179
183
|
},
|
|
180
184
|
};
|
|
181
185
|
|
|
182
|
-
|
|
186
|
+
/** Keka (India HRMS): the careers portal calls an unauthenticated embed API. Token is `tenant/orgId`; the org id sits in the portal shell. */
|
|
187
|
+
const keka: ProviderSpec = {
|
|
188
|
+
ats: "keka",
|
|
189
|
+
label: "Keka",
|
|
190
|
+
hosts: ["keka.com"],
|
|
191
|
+
crawlPatterns: ["*.keka.com/careers*"],
|
|
192
|
+
resolve(url) {
|
|
193
|
+
const tenant = /^([a-z0-9-]+)\.keka\.com$/i.exec(url.hostname)?.[1]?.toLowerCase();
|
|
194
|
+
if (!tenant || ["www", "app", "hr", "academy", "developers", "help", "cdn", "api"].includes(tenant) && tenant !== "hr") return null;
|
|
195
|
+
const org = /\/(?:careers\/api\/embedjobs\/default\/active|ats\/documents)\/([0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})/i.exec(url.pathname)?.[1];
|
|
196
|
+
return org ? `${tenant}/${org.toLowerCase()}` : null;
|
|
197
|
+
},
|
|
198
|
+
canonicalUrl: (token) => `https://${token.split("/")[0]}.keka.com/careers`,
|
|
199
|
+
endpoint: (token) => `https://${token.split("/")[0]}.keka.com/careers/api/embedjobs/default/active/${token.split("/")[1] ?? ""}`,
|
|
200
|
+
jobsFromBody: (body) => asRecords(body),
|
|
201
|
+
providerName: () => "",
|
|
202
|
+
async companyInfo(token, get) {
|
|
203
|
+
const body = await get(`https://${token.split("/")[0]}.keka.com/careers/api/organization/default/careerportalinfo`);
|
|
204
|
+
const website = isRecord(body) ? str(body.companyWebsite).replace(/^https?:\/\//, "").replace(/^www\./, "").split("/")[0] ?? "" : "";
|
|
205
|
+
return { name: isRecord(body) ? str(body.name) || str(body.shortName) : "", ...(website && !/keka\.com$/i.test(website) ? { website } : {}) };
|
|
206
|
+
},
|
|
207
|
+
payloadVersion: () => "keka-embedjobs:v1",
|
|
208
|
+
normalize(company, job) {
|
|
209
|
+
const places = asRecords(job.jobLocations) ?? [];
|
|
210
|
+
const location = places.map((place) => [str(place.city) || str(place.name), str(place.state), countryLabel(str(place.countryCode)) || str(place.countryName)].filter(Boolean).join(", ")).filter(Boolean).join("; ") || "Unspecified";
|
|
211
|
+
const remote = /remote/i.test(location) || job.jobType === 3;
|
|
212
|
+
return classifyJob({
|
|
213
|
+
id: `keka:${company.slug}:${str(job.id)}`, company: company.name, title: str(job.title), location,
|
|
214
|
+
remote, workMode: remote ? "remote" : "unknown",
|
|
215
|
+
eligibleCountries: [], excludedCountries: [], eligibleRegions: [], eligibilityConfidence: "unknown",
|
|
216
|
+
url: `https://${company.token.split("/")[0]}.keka.com/careers/jobdetails/${encodeURIComponent(str(job.id))}`,
|
|
217
|
+
updatedAt: str(job.publishedOn) || undefined, description: plainText(str(job.description)),
|
|
218
|
+
});
|
|
219
|
+
},
|
|
220
|
+
};
|
|
221
|
+
|
|
222
|
+
/** Zoho Recruit careers portals publish an RSS feed of open positions. Token is the portal host (tenant.zohorecruit.com or .in). */
|
|
223
|
+
const zohorecruit: ProviderSpec = {
|
|
224
|
+
ats: "zohorecruit",
|
|
225
|
+
label: "Zoho Recruit",
|
|
226
|
+
hosts: ["zohorecruit.com", "zohorecruit.in", "zohorecruit.eu"],
|
|
227
|
+
crawlPatterns: ["*.zohorecruit.com/jobs/Careers*", "*.zohorecruit.in/jobs/Careers*"],
|
|
228
|
+
resolve(url) {
|
|
229
|
+
const match = /^([a-z0-9-]+)\.(zohorecruit\.(?:com|in|eu))$/i.exec(url.hostname);
|
|
230
|
+
if (!match || ["www", "help", "static", "img", "js", "css", "accounts"].includes(match[1]!.toLowerCase())) return null;
|
|
231
|
+
return /^\/jobs\/careers/i.test(url.pathname) ? url.hostname.toLowerCase() : null;
|
|
232
|
+
},
|
|
233
|
+
canonicalUrl: (token) => `https://${token}/jobs/Careers`,
|
|
234
|
+
endpoint: (token) => `https://${token}/jobs/Careers/rss`,
|
|
235
|
+
bodyFormat: "text",
|
|
236
|
+
jobsFromBody(body) {
|
|
237
|
+
if (typeof body !== "string" || !/<rss/i.test(body)) return null;
|
|
238
|
+
return [...body.matchAll(/<item>([\s\S]*?)<\/item>/gi)].map(([, item]) => ({
|
|
239
|
+
title: cdata(tag(item!, "title")), link: tag(item!, "link"), guid: cdata(tag(item!, "guid")), pubDate: tag(item!, "pubDate"), description: cdata(tag(item!, "description")),
|
|
240
|
+
}));
|
|
241
|
+
},
|
|
242
|
+
providerName: (_jobs, body) => (typeof body === "string" ? cdata(tag(body.split(/<item>/i)[0] ?? "", "title")).replace(/\s*[-–|]\s*careers?$/i, "").trim() : ""),
|
|
243
|
+
payloadVersion: () => "zohorecruit-rss:v1",
|
|
244
|
+
normalize(company, job) {
|
|
245
|
+
const html = str(job.description);
|
|
246
|
+
const locationText = plainText(/Location:\s*([\s\S]*?)(?:<br|<span|$)/i.exec(html)?.[1] ?? "").trim();
|
|
247
|
+
const description = plainText(html.replace(/^[\s\S]*?<span id="spandesc">/i, ""));
|
|
248
|
+
const remote = /remote/i.test(str(job.title)) || /remote/i.test(locationText);
|
|
249
|
+
const id = str(job.guid) || /\/jobs\/Careers\/(\d+)/i.exec(str(job.link))?.[1] || str(job.link);
|
|
250
|
+
const posted = Date.parse(str(job.pubDate));
|
|
251
|
+
return classifyJob({
|
|
252
|
+
id: `zohorecruit:${company.slug}:${id}`, company: company.name, title: str(job.title), location: locationText || "Unspecified",
|
|
253
|
+
remote, workMode: remote ? "remote" : "unknown",
|
|
254
|
+
eligibleCountries: [], excludedCountries: [], eligibleRegions: [], eligibilityConfidence: "unknown",
|
|
255
|
+
url: str(job.link).replace(/\?source=RSS$/i, ""), ...(Number.isFinite(posted) ? { updatedAt: new Date(posted).toISOString() } : {}), description,
|
|
256
|
+
});
|
|
257
|
+
},
|
|
258
|
+
};
|
|
259
|
+
|
|
260
|
+
function tag(xml: string, name: string): string { return new RegExp(`<${name}(?:\\s[^>]*)?>([\\s\\S]*?)</${name}>`, "i").exec(xml)?.[1]?.trim() ?? ""; }
|
|
261
|
+
function cdata(value: string): string { return value.replace(/^<!\[CDATA\[([\s\S]*?)\]\]>$/, "$1").trim(); }
|
|
262
|
+
|
|
263
|
+
export const PROVIDERS: ReadonlyArray<ProviderSpec> = [smartrecruiters, workable, breezy, freshteam, keka, zohorecruit];
|
|
183
264
|
|
|
184
265
|
export function providerSpec(ats: string): ProviderSpec | undefined {
|
|
185
266
|
return PROVIDERS.find((spec) => spec.ats === ats);
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import { readFile } from "node:fs/promises";
|
|
2
|
+
import { atomicJson } from "./atomic-file.ts";
|
|
3
|
+
import type { SiteProbeReport } from "./site-probe.ts";
|
|
4
|
+
import type { VerifiedCompany } from "./types.ts";
|
|
5
|
+
|
|
6
|
+
export interface SiteAdmissionResult { admitted: string[]; skippedKnown: string[]; catalogSize: number }
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Promote company sites that published JobPosting markup in a probe run into the verified catalog. Identity is
|
|
10
|
+
* structural: the postings were read from pages on the company's own domain, so the evidence kind is `company_site`.
|
|
11
|
+
* A company already covered by an ATS board is skipped so it is never counted twice.
|
|
12
|
+
*/
|
|
13
|
+
export async function admitSites(reportPath: string, catalogPath: string, options: { minPostings?: number; now?: Date } = {}): Promise<SiteAdmissionResult> {
|
|
14
|
+
const report = JSON.parse(await readFile(reportPath, "utf8")) as SiteProbeReport;
|
|
15
|
+
const catalog = JSON.parse(await readFile(catalogPath, "utf8")) as Record<string, Omit<VerifiedCompany, "slug">>;
|
|
16
|
+
const knownDomains = new Set(Object.values(catalog).map((entry) => normalize(entry.companyDomain ?? "")).filter(Boolean));
|
|
17
|
+
const checkedAt = (options.now ?? new Date()).toISOString();
|
|
18
|
+
const result: SiteAdmissionResult = { admitted: [], skippedKnown: [], catalogSize: 0 };
|
|
19
|
+
for (const site of report.sites) {
|
|
20
|
+
if (site.postings < (options.minPostings ?? 1) || !site.careerUrl) continue;
|
|
21
|
+
const domain = normalize(site.companyDomain);
|
|
22
|
+
if (knownDomains.has(domain)) { result.skippedKnown.push(domain); continue; }
|
|
23
|
+
let slug = domain.split(".")[0]!.replace(/[^a-z0-9-]/g, "-");
|
|
24
|
+
if (catalog[slug] && normalize(catalog[slug]!.companyDomain ?? "") !== domain) slug = `${slug}-site`;
|
|
25
|
+
catalog[slug] = {
|
|
26
|
+
name: site.companyName.trim(), ats: "jobposting", token: site.careerUrl, companyDomain: domain, sourceUrl: site.careerUrl,
|
|
27
|
+
...(site.indiaPostings > 0 ? { cohorts: ["IN"] } : {}),
|
|
28
|
+
discoveredFrom: { channel: "career_page", reference: site.careerUrl },
|
|
29
|
+
verification: { checkedAt, canonicalSourceUrl: site.careerUrl, observedCompanyName: site.companyName.trim(), identityEvidence: "company_site", contentType: "text/html", payloadVersion: "jobposting-jsonld:v1", jobCount: site.postings },
|
|
30
|
+
};
|
|
31
|
+
knownDomains.add(domain);
|
|
32
|
+
result.admitted.push(slug);
|
|
33
|
+
}
|
|
34
|
+
result.catalogSize = Object.keys(catalog).length;
|
|
35
|
+
await atomicJson(catalogPath, catalog);
|
|
36
|
+
return result;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
function normalize(domain: string): string { return domain.toLowerCase().replace(/^www\./, ""); }
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
import { readFile } from "node:fs/promises";
|
|
2
|
+
import { atomicJson } from "./atomic-file.ts";
|
|
3
|
+
import { crawlSite, sitePostingsToJobs, type SiteSeed } from "./jobposting-site.ts";
|
|
4
|
+
|
|
5
|
+
export interface SiteProbeOptions { limit?: number; sample?: number; concurrency?: number; maxPages?: number; delayMs?: number; timeoutMs?: number }
|
|
6
|
+
export interface SiteProbeReport {
|
|
7
|
+
generatedAt: string; seeds: number; checked: number; withCareersPage: number; withPostings: number; robotsBlocked: number;
|
|
8
|
+
postings: number; indiaPostings: number; withDate: number; withIdentifier: number; pagesFetched: number;
|
|
9
|
+
sites: Array<{ companyName: string; companyDomain: string; careerUrl?: string; candidatePages: number; pagesFetched: number; postings: number; indiaPostings: number; robotsBlocked: boolean; issues: number }>;
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
/** Bounded yield measurement over company seeds: how many company sites publish JobPosting markup, and how many roles that is. */
|
|
13
|
+
export async function probeSites(inputPath: string, reportPath: string, options: SiteProbeOptions = {}): Promise<SiteProbeReport> {
|
|
14
|
+
const all = JSON.parse(await readFile(inputPath, "utf8")) as SiteSeed[];
|
|
15
|
+
const byDomain = new Map<string, SiteSeed>();
|
|
16
|
+
for (const seed of all) if (seed.companyDomain && !byDomain.has(seed.companyDomain)) byDomain.set(seed.companyDomain, seed);
|
|
17
|
+
let seeds = [...byDomain.values()];
|
|
18
|
+
if (options.sample && options.sample < seeds.length) { let state = 12345; const random = () => (state = (state * 1103515245 + 12345) % 2147483648) / 2147483648; seeds = [...seeds].sort(() => random() - 0.5).slice(0, options.sample); }
|
|
19
|
+
if (options.limit) seeds = seeds.slice(0, options.limit);
|
|
20
|
+
const report: SiteProbeReport = { generatedAt: new Date().toISOString(), seeds: all.length, checked: 0, withCareersPage: 0, withPostings: 0, robotsBlocked: 0, postings: 0, indiaPostings: 0, withDate: 0, withIdentifier: 0, pagesFetched: 0, sites: [] };
|
|
21
|
+
let cursor = 0;
|
|
22
|
+
async function worker() {
|
|
23
|
+
while (cursor < seeds.length) {
|
|
24
|
+
const seed = seeds[cursor++]!;
|
|
25
|
+
const result = await crawlSite(seed, { maxPages: options.maxPages, delayMs: options.delayMs ?? 300, timeoutMs: options.timeoutMs }).catch((error) => ({ robotsBlocked: false, candidatePages: 0, pagesFetched: 0, postings: [], issues: [String(error)], careerUrl: undefined }));
|
|
26
|
+
const jobs = sitePostingsToJobs({ slug: seed.companyDomain, name: seed.companyName }, result.postings);
|
|
27
|
+
const india = jobs.filter((job) => job.eligibleCountries.includes("IN")).length;
|
|
28
|
+
report.checked += 1;
|
|
29
|
+
if (result.careerUrl) report.withCareersPage += 1;
|
|
30
|
+
if (result.postings.length) report.withPostings += 1;
|
|
31
|
+
if (result.robotsBlocked) report.robotsBlocked += 1;
|
|
32
|
+
report.postings += result.postings.length; report.indiaPostings += india; report.pagesFetched += result.pagesFetched;
|
|
33
|
+
report.withDate += result.postings.filter((posting) => posting.datePosted).length;
|
|
34
|
+
report.withIdentifier += result.postings.filter((posting) => posting.identifier).length;
|
|
35
|
+
report.sites.push({ companyName: seed.companyName, companyDomain: seed.companyDomain, careerUrl: result.careerUrl, candidatePages: result.candidatePages, pagesFetched: result.pagesFetched, postings: result.postings.length, indiaPostings: india, robotsBlocked: result.robotsBlocked, issues: result.issues.length });
|
|
36
|
+
if (report.checked % 25 === 0) await atomicJson(reportPath, report);
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
await Promise.all(Array.from({ length: Math.max(1, options.concurrency ?? 6) }, worker));
|
|
40
|
+
report.sites.sort((left, right) => right.postings - left.postings);
|
|
41
|
+
await atomicJson(reportPath, report);
|
|
42
|
+
return report;
|
|
43
|
+
}
|
package/src/source-enrichment.ts
CHANGED
|
@@ -84,7 +84,7 @@ function isVerifiedCatalogRecord(company: unknown): boolean {
|
|
|
84
84
|
&& verification.canonicalSourceUrl === source.canonicalSourceUrl
|
|
85
85
|
&& typeof verification.checkedAt === "string" && Number.isFinite(Date.parse(verification.checkedAt))
|
|
86
86
|
&& typeof verification.observedCompanyName === "string" && verification.observedCompanyName.length > 0
|
|
87
|
-
&& ["provider_company_name", "provider_tenant", "structured_domain_link", "company_redirect", "company_page_link"].includes(String(verification.identityEvidence))
|
|
87
|
+
&& ["provider_company_name", "provider_tenant", "structured_domain_link", "company_redirect", "company_page_link", "provider_board", "company_site"].includes(String(verification.identityEvidence))
|
|
88
88
|
&& typeof verification.contentType === "string" && typeof verification.payloadVersion === "string" && verification.payloadVersion.length > 0
|
|
89
89
|
&& Number.isInteger(verification.jobCount) && Number(verification.jobCount) > 0;
|
|
90
90
|
}
|