jd-intel 0.10.0 → 0.11.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -1
- package/package.json +1 -2
- package/src/adapters/ashby.js +6 -12
- package/src/adapters/greenhouse.js +79 -27
- package/src/adapters/index.js +24 -14
- package/src/adapters/lever.js +7 -12
- package/src/adapters/recruitee.js +6 -13
- package/src/adapters/smartrecruiters.js +27 -54
- package/src/adapters/teamtailor.js +3 -10
- package/src/adapters/workday.js +53 -88
- package/src/cli.js +30 -22
- package/src/filters.js +49 -5
- package/src/index.js +12 -1
- package/src/normalizer.js +26 -2
- package/src/registry.js +22 -29
package/README.md
CHANGED
|
@@ -240,7 +240,8 @@ No custom parsing per company.
|
|
|
240
240
|
| `locationType` | `remote`, `hybrid`, `onsite`, or `unknown` when neither the platform nor the location text says |
|
|
241
241
|
| `workplace` | `{ type, source }`. `type` repeats `locationType`; `source` is `ats` when the platform stated it, `text` when read from the location string, null when unknown |
|
|
242
242
|
| `salary` | Min-max range with `currency`, plus `period` (`year`, `month`, `hour`, or null) and `source` (`ats` when the platform supplied it, `text` when parsed from the posting). Null when nothing is stated |
|
|
243
|
-
| `description` | Full JD in clean markdown |
|
|
243
|
+
| `description` | Full JD in clean markdown. Empty when `content.status` is `missing` |
|
|
244
|
+
| `content` | `{ status, reason }`. `complete` when the posting was read. `missing` when Workday or SmartRecruiters listed the job but its detail request failed (`reason`: `http_503`, `http_429`, `network_error`), so `description` and `salary` are unknown |
|
|
244
245
|
| `url` | Direct link to the posting |
|
|
245
246
|
| `postedAt` | Publication date (when provided) |
|
|
246
247
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "jd-intel",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.11.1",
|
|
4
4
|
"description": "Fetch and normalize job descriptions across seven major ATS (Greenhouse, Lever, Ashby, Workday, and more), for your AI assistant. No copy-paste.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "src/index.js",
|
|
@@ -15,7 +15,6 @@
|
|
|
15
15
|
"scripts": {
|
|
16
16
|
"test": "node --test test/*.test.js",
|
|
17
17
|
"fetch": "node src/cli.js fetch",
|
|
18
|
-
"search": "node src/cli.js search",
|
|
19
18
|
"verify:registry": "node scripts/verify-registry.mjs",
|
|
20
19
|
"sync:registry-pages": "node scripts/sync-pages-registry.mjs",
|
|
21
20
|
"pack:mcpb": "node scripts/build-mcpb.mjs",
|
package/src/adapters/ashby.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize, extractSalaryFromText } from '../normalizer.js';
|
|
1
|
+
import { normalize, extractSalaryFromText, WORKPLACE_TYPES } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
import { atsFetch, probeResult } from '../http.js';
|
|
4
4
|
|
|
@@ -14,11 +14,9 @@ const BOARD_URL = 'https://api.ashbyhq.com/posting-api/job-board';
|
|
|
14
14
|
* into a silent empty result (issue #55).
|
|
15
15
|
*
|
|
16
16
|
* @param {string} slug - Company slug (e.g., 'notion', 'linear')
|
|
17
|
-
* @param {object} [ctx] - { report }; report is called once with
|
|
18
|
-
* { ats, org_name, org_url } when given
|
|
19
17
|
* @returns {Promise<Array>} Normalized job objects
|
|
20
18
|
*/
|
|
21
|
-
export async function fetchAshby(slug
|
|
19
|
+
export async function fetchAshby(slug) {
|
|
22
20
|
const url = `${BOARD_URL}/${slug}?includeCompensation=true`;
|
|
23
21
|
const resp = await atsFetch(url);
|
|
24
22
|
|
|
@@ -31,10 +29,8 @@ export async function fetchAshby(slug, ctx = {}) {
|
|
|
31
29
|
const jobs = data.jobs || [];
|
|
32
30
|
|
|
33
31
|
// The REST response is { jobs, apiVersion }: no organization name, and
|
|
34
|
-
// every link is on jobs.ashbyhq.com.
|
|
35
|
-
|
|
36
|
-
ctx.report({ ats: 'ashby', org_name: null, org_url: null });
|
|
37
|
-
}
|
|
32
|
+
// every link is on jobs.ashbyhq.com. Nothing to report, so the board's
|
|
33
|
+
// org_name and org_url stay null (issue #58).
|
|
38
34
|
|
|
39
35
|
return jobs.map(job => {
|
|
40
36
|
const comp = job.compensation || {};
|
|
@@ -69,16 +65,14 @@ export async function fetchAshby(slug, ctx = {}) {
|
|
|
69
65
|
});
|
|
70
66
|
}
|
|
71
67
|
|
|
72
|
-
const WORKPLACE_TYPES = { remote: 'remote', hybrid: 'hybrid', onsite: 'onsite' };
|
|
73
|
-
|
|
74
68
|
/**
|
|
75
69
|
* `workplaceType` is 'Remote', 'Hybrid' or 'OnSite'. `isRemote` is the
|
|
76
70
|
* older flag and can only say remote, so it is the fallback when the type
|
|
77
71
|
* is absent. false means nothing: the role may be hybrid or onsite.
|
|
78
72
|
*/
|
|
79
73
|
function parseAshbyWorkplace(job) {
|
|
80
|
-
const type =
|
|
81
|
-
if (type) return type;
|
|
74
|
+
const type = String(job.workplaceType || '').toLowerCase();
|
|
75
|
+
if (WORKPLACE_TYPES.has(type)) return type;
|
|
82
76
|
return job.isRemote === true ? 'remote' : null;
|
|
83
77
|
}
|
|
84
78
|
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize, decodeEntities } from '../normalizer.js';
|
|
1
|
+
import { normalize, decodeEntities, periodWord } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
import { atsFetch, probeResult } from '../http.js';
|
|
4
4
|
|
|
@@ -11,11 +11,12 @@ const BASE_URL = 'https://boards-api.greenhouse.io/v1/boards';
|
|
|
11
11
|
*
|
|
12
12
|
* @param {string} slug - Company slug (e.g., 'stripe', 'notion')
|
|
13
13
|
* @param {object} [ctx] - { report }; report is called once with
|
|
14
|
-
* {
|
|
14
|
+
* { org_name, org_url } when given
|
|
15
15
|
* @returns {Promise<Array>} Normalized job objects
|
|
16
16
|
*/
|
|
17
17
|
export async function fetchGreenhouse(slug, ctx = {}) {
|
|
18
|
-
|
|
18
|
+
// pay_transparency adds pay_input_ranges to each row (issue #86).
|
|
19
|
+
const url = `${BASE_URL}/${slug}/jobs?content=true&pay_transparency=true`;
|
|
19
20
|
const resp = await atsFetch(url);
|
|
20
21
|
|
|
21
22
|
if (!resp.ok) {
|
|
@@ -31,35 +32,86 @@ export async function fetchGreenhouse(slug, ctx = {}) {
|
|
|
31
32
|
// there is no company host to report (issue #58).
|
|
32
33
|
if (typeof ctx.report === 'function') {
|
|
33
34
|
ctx.report({
|
|
34
|
-
ats: 'greenhouse',
|
|
35
35
|
org_name: jobs.find(j => j.company_name)?.company_name || null,
|
|
36
36
|
org_url: null,
|
|
37
37
|
});
|
|
38
38
|
}
|
|
39
39
|
|
|
40
|
-
return jobs.map(job =>
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
40
|
+
return jobs.map(job => {
|
|
41
|
+
const payRanges = parsePayRanges(job.pay_input_ranges);
|
|
42
|
+
return normalize({
|
|
43
|
+
companySlug: slug,
|
|
44
|
+
company: data.name || slug,
|
|
45
|
+
title: job.title || '',
|
|
46
|
+
department: job.departments?.[0]?.name || '',
|
|
47
|
+
location: job.location?.name || '',
|
|
48
|
+
workplace: parseGreenhouseWorkplace(job.metadata),
|
|
49
|
+
// `content` arrives HTML-escaped (`<p>`). Decode that outer layer
|
|
50
|
+
// once so normalize() sees real tags; it strips and decodes the rest.
|
|
51
|
+
description: decodeEntities(job.content || ''),
|
|
52
|
+
url: job.absolute_url || '',
|
|
53
|
+
// updated_at is an edit time that many boards bulk-refresh, so it is not
|
|
54
|
+
// a posting date. first_published is. Fallback covers boards without it (#69).
|
|
55
|
+
postedAt: job.first_published || job.updated_at || null,
|
|
56
|
+
salary: salaryFromRanges(payRanges), // null without structured pay; normalize() then parses the text
|
|
57
|
+
metadata: {
|
|
58
|
+
greenhouseId: job.id,
|
|
59
|
+
internal_job_id: job.internal_job_id,
|
|
60
|
+
departments: job.departments?.map(d => d.name) || [],
|
|
61
|
+
offices: job.offices?.map(o => o.name) || [],
|
|
62
|
+
updatedAt: job.updated_at,
|
|
63
|
+
payRanges,
|
|
64
|
+
},
|
|
65
|
+
}, 'greenhouse');
|
|
66
|
+
});
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* `pay_input_ranges` is the board's pay transparency data:
|
|
71
|
+
* [{ min_cents, max_cents, currency_type, title, blurb }]. The blurb is
|
|
72
|
+
* boilerplate already rendered in the description, so it is dropped, and
|
|
73
|
+
* the platform can send the same entry twice, so entries are deduplicated.
|
|
74
|
+
* A board that does not publish pay sends an empty array or no key.
|
|
75
|
+
*/
|
|
76
|
+
function parsePayRanges(ranges) {
|
|
77
|
+
const seen = new Set();
|
|
78
|
+
const out = [];
|
|
79
|
+
for (const r of Array.isArray(ranges) ? ranges : []) {
|
|
80
|
+
const range = {
|
|
81
|
+
title: r.title || '',
|
|
82
|
+
min: Number.isFinite(r.min_cents) ? r.min_cents / 100 : null,
|
|
83
|
+
max: Number.isFinite(r.max_cents) ? r.max_cents / 100 : null,
|
|
84
|
+
currency: r.currency_type || 'USD',
|
|
85
|
+
};
|
|
86
|
+
const key = JSON.stringify(range);
|
|
87
|
+
if ((range.min === null && range.max === null) || seen.has(key)) continue;
|
|
88
|
+
seen.add(key);
|
|
89
|
+
out.push(range);
|
|
90
|
+
}
|
|
91
|
+
return out;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* One salary from the ranges. Ranges in one currency span (lowest min,
|
|
96
|
+
* highest max), as the Ashby adapter does for its tiers; with mixed
|
|
97
|
+
* currencies the first range stands alone. Every range stays in
|
|
98
|
+
* metadata.payRanges. The field has no period, so the period is the one
|
|
99
|
+
* the range titles state ("Annual", "Hourly") and null when they state
|
|
100
|
+
* none or disagree: an 'ats' value carries no guessed period.
|
|
101
|
+
*/
|
|
102
|
+
function salaryFromRanges(ranges) {
|
|
103
|
+
if (ranges.length === 0) return null;
|
|
104
|
+
const used = ranges.every(r => r.currency === ranges[0].currency) ? ranges : [ranges[0]];
|
|
105
|
+
const mins = used.map(r => r.min).filter(v => v !== null);
|
|
106
|
+
const maxes = used.map(r => r.max).filter(v => v !== null);
|
|
107
|
+
const periods = new Set(used.map(r => periodWord(r.title)));
|
|
108
|
+
return {
|
|
109
|
+
min: mins.length ? Math.min(...mins) : null,
|
|
110
|
+
max: maxes.length ? Math.max(...maxes) : null,
|
|
111
|
+
currency: used[0].currency,
|
|
112
|
+
period: periods.size === 1 ? [...periods][0] : null,
|
|
113
|
+
source: 'ats',
|
|
114
|
+
};
|
|
63
115
|
}
|
|
64
116
|
|
|
65
117
|
/**
|
package/src/adapters/index.js
CHANGED
|
@@ -1,19 +1,29 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
1
|
+
import { fetchGreenhouse, hasGreenhouse } from './greenhouse.js';
|
|
2
|
+
import { fetchLever, hasLever } from './lever.js';
|
|
3
|
+
import { fetchAshby, hasAshby } from './ashby.js';
|
|
4
|
+
import { fetchSmartrecruiters, hasSmartrecruiters } from './smartrecruiters.js';
|
|
5
|
+
import { fetchTeamtailor, hasTeamtailor } from './teamtailor.js';
|
|
6
|
+
import { fetchRecruitee, hasRecruitee } from './recruitee.js';
|
|
7
|
+
import { fetchWorkday, hasWorkday } from './workday.js';
|
|
8
|
+
|
|
9
|
+
export {
|
|
10
|
+
fetchGreenhouse, hasGreenhouse,
|
|
11
|
+
fetchLever, hasLever,
|
|
12
|
+
fetchAshby, hasAshby,
|
|
13
|
+
fetchSmartrecruiters, hasSmartrecruiters,
|
|
14
|
+
fetchTeamtailor, hasTeamtailor,
|
|
15
|
+
fetchRecruitee, hasRecruitee,
|
|
16
|
+
fetchWorkday, hasWorkday,
|
|
17
|
+
};
|
|
8
18
|
|
|
9
19
|
export const ADAPTERS = {
|
|
10
|
-
greenhouse: { fetch:
|
|
11
|
-
lever: { fetch:
|
|
12
|
-
ashby: { fetch:
|
|
13
|
-
smartrecruiters: { fetch:
|
|
14
|
-
teamtailor: { fetch:
|
|
15
|
-
recruitee: { fetch:
|
|
16
|
-
workday: { fetch:
|
|
20
|
+
greenhouse: { fetch: fetchGreenhouse, has: hasGreenhouse },
|
|
21
|
+
lever: { fetch: fetchLever, has: hasLever },
|
|
22
|
+
ashby: { fetch: fetchAshby, has: hasAshby },
|
|
23
|
+
smartrecruiters: { fetch: fetchSmartrecruiters, has: hasSmartrecruiters },
|
|
24
|
+
teamtailor: { fetch: fetchTeamtailor, has: hasTeamtailor },
|
|
25
|
+
recruitee: { fetch: fetchRecruitee, has: hasRecruitee },
|
|
26
|
+
workday: { fetch: fetchWorkday, has: hasWorkday },
|
|
17
27
|
};
|
|
18
28
|
|
|
19
29
|
export const ATS_NAMES = Object.keys(ADAPTERS);
|
package/src/adapters/lever.js
CHANGED
|
@@ -1,12 +1,10 @@
|
|
|
1
|
-
import { normalize, extractSalaryFromText } from '../normalizer.js';
|
|
1
|
+
import { normalize, extractSalaryFromText, toIso } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
import { atsFetch, probeResult } from '../http.js';
|
|
4
4
|
|
|
5
5
|
const BASE_URL = 'https://api.lever.co/v0/postings';
|
|
6
6
|
|
|
7
7
|
const PERIODS = { 'per-year-salary': 'year', 'per-month-salary': 'month', 'per-hour-wage': 'hour' };
|
|
8
|
-
// Lever's workplaceType is one of these or 'unspecified'.
|
|
9
|
-
const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
|
|
10
8
|
|
|
11
9
|
/**
|
|
12
10
|
* Fetch all jobs from a Lever job board.
|
|
@@ -14,11 +12,9 @@ const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
|
|
|
14
12
|
* Docs: https://github.com/lever/postings-api
|
|
15
13
|
*
|
|
16
14
|
* @param {string} slug - Company slug (e.g., 'stripe', 'figma')
|
|
17
|
-
* @param {object} [ctx] - { report }; report is called once with
|
|
18
|
-
* { ats, org_name, org_url } when given
|
|
19
15
|
* @returns {Promise<Array>} Normalized job objects
|
|
20
16
|
*/
|
|
21
|
-
export async function fetchLever(slug
|
|
17
|
+
export async function fetchLever(slug) {
|
|
22
18
|
const url = `${BASE_URL}/${slug}?mode=json`;
|
|
23
19
|
const resp = await atsFetch(url);
|
|
24
20
|
|
|
@@ -31,10 +27,8 @@ export async function fetchLever(slug, ctx = {}) {
|
|
|
31
27
|
if (!Array.isArray(jobs)) return [];
|
|
32
28
|
|
|
33
29
|
// The postings response is a bare array of jobs: no organization name
|
|
34
|
-
// anywhere, and every link is on jobs.lever.co.
|
|
35
|
-
|
|
36
|
-
ctx.report({ ats: 'lever', org_name: null, org_url: null });
|
|
37
|
-
}
|
|
30
|
+
// anywhere, and every link is on jobs.lever.co. Nothing to report, so
|
|
31
|
+
// the board's org_name and org_url stay null (issue #58).
|
|
38
32
|
|
|
39
33
|
return jobs.map(job => normalize({
|
|
40
34
|
companySlug: slug,
|
|
@@ -46,10 +40,11 @@ export async function fetchLever(slug, ctx = {}) {
|
|
|
46
40
|
department: job.categories?.department || job.categories?.team || '',
|
|
47
41
|
location: job.categories?.location || '',
|
|
48
42
|
locations: job.categories?.allLocations || [],
|
|
49
|
-
|
|
43
|
+
// 'remote', 'hybrid', 'onsite' or 'unspecified'; normalize() ignores the last.
|
|
44
|
+
workplace: job.workplaceType,
|
|
50
45
|
description: buildDescription(job),
|
|
51
46
|
url: job.hostedUrl || '',
|
|
52
|
-
postedAt:
|
|
47
|
+
postedAt: toIso(job.createdAt),
|
|
53
48
|
salary: parseLeverSalary(job.salaryRange, job.text),
|
|
54
49
|
metadata: {
|
|
55
50
|
leverId: job.id,
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize } from '../normalizer.js';
|
|
1
|
+
import { normalize, toIso } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
import { atsFetch, probeResult } from '../http.js';
|
|
4
4
|
import { orgHost } from '../boards.js';
|
|
@@ -21,7 +21,7 @@ import { orgHost } from '../boards.js';
|
|
|
21
21
|
*
|
|
22
22
|
* @param {string} slug - Recruitee company subdomain (e.g., 'vandebron')
|
|
23
23
|
* @param {object} [ctx] - { report }; report is called once with
|
|
24
|
-
* {
|
|
24
|
+
* { org_name, org_url } when given
|
|
25
25
|
* @returns {Promise<Array>} Normalized job objects
|
|
26
26
|
*/
|
|
27
27
|
export async function fetchRecruitee(slug, ctx = {}) {
|
|
@@ -41,7 +41,6 @@ export async function fetchRecruitee(slug, ctx = {}) {
|
|
|
41
41
|
// domain when the site has one and on {slug}.recruitee.com otherwise.
|
|
42
42
|
if (typeof ctx.report === 'function') {
|
|
43
43
|
ctx.report({
|
|
44
|
-
ats: 'recruitee',
|
|
45
44
|
org_name: offers.find(o => o.company_name)?.company_name || null,
|
|
46
45
|
org_url: orgHost(offers.find(o => o.careers_url)?.careers_url),
|
|
47
46
|
});
|
|
@@ -52,7 +51,7 @@ export async function fetchRecruitee(slug, ctx = {}) {
|
|
|
52
51
|
let location = place;
|
|
53
52
|
if (offer.remote) location = place ? `Remote - ${place}` : 'Remote';
|
|
54
53
|
|
|
55
|
-
const createdAt =
|
|
54
|
+
const createdAt = recruiteeDate(offer.created_at);
|
|
56
55
|
|
|
57
56
|
return normalize({
|
|
58
57
|
companySlug: slug,
|
|
@@ -66,7 +65,7 @@ export async function fetchRecruitee(slug, ctx = {}) {
|
|
|
66
65
|
url: offer.careers_url || offer.careers_apply_url || '',
|
|
67
66
|
// created_at can predate publication by years on long-lived offers,
|
|
68
67
|
// so it is not a posting date. published_at is.
|
|
69
|
-
postedAt:
|
|
68
|
+
postedAt: recruiteeDate(offer.published_at) || createdAt,
|
|
70
69
|
salary: parseRecruiteeSalary(offer.salary),
|
|
71
70
|
metadata: {
|
|
72
71
|
recruiteeId: offer.guid || offer.id,
|
|
@@ -78,14 +77,8 @@ export async function fetchRecruitee(slug, ctx = {}) {
|
|
|
78
77
|
});
|
|
79
78
|
}
|
|
80
79
|
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
*/
|
|
84
|
-
function toIso(ts) {
|
|
85
|
-
if (!ts) return null;
|
|
86
|
-
const d = new Date(ts.replace(' UTC', 'Z').replace(' ', 'T'));
|
|
87
|
-
return Number.isNaN(d.getTime()) ? null : d.toISOString();
|
|
88
|
-
}
|
|
80
|
+
// Recruitee returns "2026-05-13 07:38:11 UTC".
|
|
81
|
+
const recruiteeDate = (ts) => toIso(ts?.replace(' UTC', 'Z').replace(' ', 'T'));
|
|
89
82
|
|
|
90
83
|
/**
|
|
91
84
|
* Recruitee sends three booleans, not one enum. Hybrid wins when remote is
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { normalize } from '../normalizer.js';
|
|
1
|
+
import { normalize, missingContent } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
import { atsFetch, probeResult } from '../http.js';
|
|
4
|
-
import {
|
|
4
|
+
import { prefilterRows } from '../filters.js';
|
|
5
5
|
|
|
6
6
|
const BASE_URL = 'https://api.smartrecruiters.com/v1/companies';
|
|
7
7
|
const PAGE_SIZE = 100;
|
|
@@ -21,12 +21,12 @@ const MAX_DETAIL_FETCHES = 100;
|
|
|
21
21
|
* The list does carry name, location and releasedDate, so the same
|
|
22
22
|
* pre-filter and detail budget Workday applies run here: list-evaluable
|
|
23
23
|
* filters narrow the candidates, then at most MAX_DETAIL_FETCHES of them
|
|
24
|
-
* are hydrated (see
|
|
24
|
+
* are hydrated (see prefilterRows). Without a filterContext the
|
|
25
25
|
* cap still holds, so a direct call on a 400-posting tenant reads 100.
|
|
26
26
|
*
|
|
27
27
|
* @param {string} slug - SmartRecruiters company identifier (e.g., 'Visa')
|
|
28
28
|
* @param {object} [ctx] - { filterContext, report }; report is called once
|
|
29
|
-
* with {
|
|
29
|
+
* with { listed, prefiltered, hydrated, capped, org_name, org_url }
|
|
30
30
|
* when given
|
|
31
31
|
* @returns {Promise<Array>} Normalized job objects
|
|
32
32
|
*/
|
|
@@ -54,60 +54,28 @@ export async function fetchSmartrecruiters(slug, ctx = {}) {
|
|
|
54
54
|
if (content.length === 0 || offset >= (data.totalFound || 0)) break;
|
|
55
55
|
}
|
|
56
56
|
|
|
57
|
-
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost
|
|
58
|
-
// The list row carries name,
|
|
59
|
-
//
|
|
60
|
-
//
|
|
61
|
-
//
|
|
62
|
-
//
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
}
|
|
69
|
-
if (Array.isArray(fc.locationIncludes) && fc.locationIncludes.length > 0) {
|
|
70
|
-
const matchers = fc.locationIncludes.map(makeLocationMatcher);
|
|
71
|
-
candidates = candidates.filter(p => {
|
|
72
|
-
const loc = listLocation(p).location.toLowerCase();
|
|
73
|
-
return matchers.some(m => m(loc));
|
|
74
|
-
});
|
|
75
|
-
}
|
|
76
|
-
if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
|
|
77
|
-
const matchers = fc.locationExcludes.map(makeLocationMatcher);
|
|
78
|
-
candidates = candidates.filter(p => {
|
|
79
|
-
const loc = listLocation(p).location.toLowerCase();
|
|
80
|
-
return !loc || !matchers.some(m => m(loc));
|
|
81
|
-
});
|
|
82
|
-
}
|
|
83
|
-
if (typeof fc.postedWithinDays === 'number') {
|
|
84
|
-
// postedAt comes from releasedDate alone, so the library's rule can
|
|
85
|
-
// run here in full: a missing or unparseable date is out either way.
|
|
86
|
-
const cutoff = Date.now() - fc.postedWithinDays * 86400000;
|
|
87
|
-
candidates = candidates.filter(p => {
|
|
57
|
+
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost, then
|
|
58
|
+
// the detail budget (see prefilterRows). The list row carries name,
|
|
59
|
+
// location and releasedDate. The detail adds no location (unlike
|
|
60
|
+
// Workday's additionalLocations), so a row with none follows the
|
|
61
|
+
// library's rule now: out under includes, kept under excludes.
|
|
62
|
+
// postedAt comes from releasedDate alone, so a missing or unparseable
|
|
63
|
+
// date is out, as it is in the library.
|
|
64
|
+
const { candidates, hydrate } = prefilterRows(postings, fc, {
|
|
65
|
+
title: p => p.name || '',
|
|
66
|
+
location: p => listLocation(p).location.toLowerCase(),
|
|
67
|
+
postedWithin: (p, days) => {
|
|
88
68
|
const released = new Date(p.releasedDate || '').getTime();
|
|
89
|
-
return Number.isFinite(released) && released >=
|
|
90
|
-
}
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
// 3. Bound the detail-fetch set, Workday's reasoning verbatim: a
|
|
94
|
-
// description `filter` is applied by the library AFTER this returns,
|
|
95
|
-
// so that case keeps the full backstop instead of truncating to
|
|
96
|
-
// `limit` (which could hydrate jobs that all fail the regex while
|
|
97
|
-
// better matches go unscanned). The library pages with `offset`
|
|
98
|
-
// after this returns, so the budget covers the page plus what
|
|
99
|
-
// precedes it. Candidates keep list order.
|
|
100
|
-
const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
|
|
101
|
-
const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
|
|
102
|
-
const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
|
|
103
|
-
const hydrate = candidates.slice(0, cap);
|
|
69
|
+
return Number.isFinite(released) && released >= Date.now() - days * 86400000;
|
|
70
|
+
},
|
|
71
|
+
max: MAX_DETAIL_FETCHES,
|
|
72
|
+
});
|
|
104
73
|
|
|
105
74
|
// Every list row carries company { identifier, name }. Neither the list
|
|
106
75
|
// nor the detail has a company website, and postingUrl is always on
|
|
107
76
|
// jobs.smartrecruiters.com, so org_url stays null (issue #58).
|
|
108
77
|
if (typeof ctx.report === 'function') {
|
|
109
78
|
ctx.report({
|
|
110
|
-
ats: 'smartrecruiters',
|
|
111
79
|
listed: postings.length,
|
|
112
80
|
prefiltered: candidates.length,
|
|
113
81
|
hydrated: hydrate.length,
|
|
@@ -125,6 +93,7 @@ export async function fetchSmartrecruiters(slug, ctx = {}) {
|
|
|
125
93
|
let sections = {};
|
|
126
94
|
let postingUrl = '';
|
|
127
95
|
let salary = null;
|
|
96
|
+
let content; // set only when the detail could not be read (issue #85)
|
|
128
97
|
|
|
129
98
|
try {
|
|
130
99
|
const detailResp = await atsFetch(`${BASE_URL}/${slug}/postings/${p.id}`);
|
|
@@ -133,10 +102,13 @@ export async function fetchSmartrecruiters(slug, ctx = {}) {
|
|
|
133
102
|
sections = detail.jobAd?.sections || {};
|
|
134
103
|
postingUrl = detail.postingUrl || detail.applyUrl || '';
|
|
135
104
|
salary = parseCompensation(detail.compensation);
|
|
105
|
+
} else {
|
|
106
|
+
content = missingContent(detailResp);
|
|
136
107
|
}
|
|
137
|
-
} catch {
|
|
138
|
-
// Detail fetch failed, retries included:
|
|
139
|
-
//
|
|
108
|
+
} catch (err) {
|
|
109
|
+
// Detail fetch failed, retries included: list-only fields (no
|
|
110
|
+
// description, no url), marked missing.
|
|
111
|
+
content = missingContent(err);
|
|
140
112
|
}
|
|
141
113
|
|
|
142
114
|
const description = [
|
|
@@ -158,6 +130,7 @@ export async function fetchSmartrecruiters(slug, ctx = {}) {
|
|
|
158
130
|
url: postingUrl,
|
|
159
131
|
postedAt: p.releasedDate || null,
|
|
160
132
|
salary, // null when the detail has no compensation; normalize() then parses text
|
|
133
|
+
content,
|
|
161
134
|
metadata: {
|
|
162
135
|
smartRecruitersId: p.id,
|
|
163
136
|
refNumber: p.refNumber || '',
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize, decodeEntities } from '../normalizer.js';
|
|
1
|
+
import { normalize, decodeEntities, toIso } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
import { atsFetch } from '../http.js';
|
|
4
4
|
import { orgHost } from '../boards.js';
|
|
@@ -26,7 +26,7 @@ import { orgHost } from '../boards.js';
|
|
|
26
26
|
*
|
|
27
27
|
* @param {string} slug - TeamTailor career-site slug (e.g., 'tibber')
|
|
28
28
|
* @param {object} [ctx] - { report }; report is called once with
|
|
29
|
-
* {
|
|
29
|
+
* { org_name, org_url } when given
|
|
30
30
|
* @returns {Promise<Array>} Normalized job objects
|
|
31
31
|
*/
|
|
32
32
|
// Most sites are {slug}.teamtailor.com, but some sit on a regional
|
|
@@ -85,7 +85,6 @@ export async function fetchTeamtailor(slug, ctx = {}) {
|
|
|
85
85
|
|| xml.match(/<channel>[\s\S]*?<link>([\s\S]*?)<\/link>/)?.[1]
|
|
86
86
|
|| '';
|
|
87
87
|
ctx.report({
|
|
88
|
-
ats: 'teamtailor',
|
|
89
88
|
org_name: decodeEntities(channelTitle) || null,
|
|
90
89
|
org_url: orgHost(link.trim()),
|
|
91
90
|
});
|
|
@@ -117,12 +116,6 @@ export async function fetchTeamtailor(slug, ctx = {}) {
|
|
|
117
116
|
location = location ? `Remote - ${location}` : 'Remote';
|
|
118
117
|
}
|
|
119
118
|
|
|
120
|
-
let postedAt = null;
|
|
121
|
-
if (pubDateRaw) {
|
|
122
|
-
const d = new Date(pubDateRaw);
|
|
123
|
-
if (!Number.isNaN(d.getTime())) postedAt = d.toISOString();
|
|
124
|
-
}
|
|
125
|
-
|
|
126
119
|
return normalize({
|
|
127
120
|
companySlug: slug,
|
|
128
121
|
company,
|
|
@@ -133,7 +126,7 @@ export async function fetchTeamtailor(slug, ctx = {}) {
|
|
|
133
126
|
workplace: REMOTE_STATUS[remoteStatus.toLowerCase()] || null,
|
|
134
127
|
description: decodeEntities(pick('description')),
|
|
135
128
|
url: link,
|
|
136
|
-
postedAt,
|
|
129
|
+
postedAt: toIso(pubDateRaw),
|
|
137
130
|
salary: null, // No structured salary; normalizer parses from text
|
|
138
131
|
metadata: {
|
|
139
132
|
teamtailorId: guid,
|
package/src/adapters/workday.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { normalize } from '../normalizer.js';
|
|
1
|
+
import { normalize, missingContent, toIso } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
-
import {
|
|
3
|
+
import { prefilterRows } from '../filters.js';
|
|
4
4
|
import { atsFetch } from '../http.js';
|
|
5
5
|
import { orgHost } from '../boards.js';
|
|
6
6
|
|
|
@@ -35,7 +35,7 @@ const MULTI_LOCATION = /^\s*\d+\s+locations?\s*$/;
|
|
|
35
35
|
* @param {string} slug - normalized company slug (registry routing key)
|
|
36
36
|
* @param {object} [ctx] - { config:{tenant,env,site}, companyName, filterContext, report };
|
|
37
37
|
* report is called once, after hydration, with
|
|
38
|
-
* {
|
|
38
|
+
* { listed, prefiltered, hydrated, capped, org_name, org_url } when given
|
|
39
39
|
* @returns {Promise<Array>} Normalized job objects
|
|
40
40
|
*/
|
|
41
41
|
export async function fetchWorkday(slug, ctx = {}) {
|
|
@@ -92,55 +92,27 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
92
92
|
if (firstTotal > 0 && offset >= firstTotal) break;
|
|
93
93
|
}
|
|
94
94
|
|
|
95
|
-
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost
|
|
96
|
-
//
|
|
97
|
-
//
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
//
|
|
105
|
-
//
|
|
106
|
-
//
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
const inc = fc.locationIncludes.map(makeLocationMatcher);
|
|
111
|
-
candidates = candidates.filter(p => {
|
|
112
|
-
const loc = (p.locationsText || '').toLowerCase();
|
|
113
|
-
return MULTI_LOCATION.test(loc) || inc.some(m => m(loc));
|
|
114
|
-
});
|
|
115
|
-
}
|
|
116
|
-
if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
|
|
117
|
-
const exc = fc.locationExcludes.map(makeLocationMatcher);
|
|
118
|
-
candidates = candidates.filter(p => {
|
|
95
|
+
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost, then
|
|
96
|
+
// the detail budget (see prefilterRows). The list carries
|
|
97
|
+
// title/locationsText/postedOn, enough to apply titleFilter, location
|
|
98
|
+
// and recency without descriptions. "2 Locations" says nothing about
|
|
99
|
+
// where: the row stays a candidate through both location filters and
|
|
100
|
+
// the library's pass after hydration decides on the detail's location
|
|
101
|
+
// list (issue #61).
|
|
102
|
+
// NOTE: huge-tenant coverage is intentionally capped for v1. Two caps
|
|
103
|
+
// apply: the list scan above stops at LIST_PAGE_HARD_CAP pages (2000
|
|
104
|
+
// postings, enough for Salesforce's ~1398), and the detail set is cut
|
|
105
|
+
// to MAX_DETAIL_FETCHES here. Proper fix (smart pagination, surfaced
|
|
106
|
+
// truncation) is tracked in #26.
|
|
107
|
+
const { candidates, hydrate } = prefilterRows(postings, fc, {
|
|
108
|
+
title: p => p.title || '',
|
|
109
|
+
location: p => {
|
|
119
110
|
const loc = (p.locationsText || '').toLowerCase();
|
|
120
|
-
return MULTI_LOCATION.test(loc)
|
|
121
|
-
}
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
}
|
|
126
|
-
|
|
127
|
-
// 3. Bound the detail-fetch set.
|
|
128
|
-
// NOTE: huge-tenant coverage is intentionally capped for v1. Two
|
|
129
|
-
// caps apply: the list scan above stops at LIST_PAGE_HARD_CAP pages
|
|
130
|
-
// (2000 postings, enough for Salesforce's ~1398), and the detail set
|
|
131
|
-
// is cut to MAX_DETAIL_FETCHES here. A description `filter` is
|
|
132
|
-
// applied by the library AFTER this returns, so for that case we
|
|
133
|
-
// keep the full backstop instead of truncating tightly to `limit`
|
|
134
|
-
// (which could hydrate jobs that all fail the regex while better
|
|
135
|
-
// matches go unscanned). The library pages with `offset` after this
|
|
136
|
-
// returns, so the budget covers the page plus what precedes it.
|
|
137
|
-
// Proper fix (smart pagination / rate-limited concurrency / surfaced
|
|
138
|
-
// truncation) is tracked in #26, to be designed alongside
|
|
139
|
-
// retry/rate-limit work (#7).
|
|
140
|
-
const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
|
|
141
|
-
const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
|
|
142
|
-
const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
|
|
143
|
-
const hydrate = candidates.slice(0, cap);
|
|
111
|
+
return MULTI_LOCATION.test(loc) ? null : loc;
|
|
112
|
+
},
|
|
113
|
+
postedWithin: (p, days) => withinDays(p.postedOn, days),
|
|
114
|
+
max: MAX_DETAIL_FETCHES,
|
|
115
|
+
});
|
|
144
116
|
|
|
145
117
|
// 4. Hydrate descriptions via the per-posting detail endpoint. The detail
|
|
146
118
|
// also carries `hiringOrganization: { name, url }` next to
|
|
@@ -151,6 +123,7 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
151
123
|
const jobs = await Promise.all(hydrate.map(async (p, i) => {
|
|
152
124
|
const externalPath = p.externalPath || ''; // already begins with '/job/...'
|
|
153
125
|
let info = {};
|
|
126
|
+
let content; // set only when the detail could not be read (issue #85)
|
|
154
127
|
try {
|
|
155
128
|
// externalPath already carries the '/job/...' segment, so it is
|
|
156
129
|
// concatenated directly onto the CXS base. Inserting another
|
|
@@ -160,9 +133,12 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
160
133
|
const detail = await dResp.json();
|
|
161
134
|
info = detail.jobPostingInfo || {};
|
|
162
135
|
orgs[i] = detail.hiringOrganization || null;
|
|
136
|
+
} else {
|
|
137
|
+
content = missingContent(dResp);
|
|
163
138
|
}
|
|
164
|
-
} catch {
|
|
165
|
-
// detail failed, retries included:
|
|
139
|
+
} catch (err) {
|
|
140
|
+
// detail failed, retries included: list fields only, marked missing
|
|
141
|
+
content = missingContent(err);
|
|
166
142
|
}
|
|
167
143
|
|
|
168
144
|
return normalize({
|
|
@@ -175,8 +151,10 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
175
151
|
workplace: parseWorkdayRemoteType(info.remoteType),
|
|
176
152
|
description: info.jobDescription || '',
|
|
177
153
|
url: `https://${tenant}.${env}.myworkdayjobs.com/${site}${externalPath}`,
|
|
178
|
-
|
|
154
|
+
// startDate is "2026-05-01" or "May 1, 2026".
|
|
155
|
+
postedAt: toIso(info.startDate) || normalizePostedOn(p.postedOn),
|
|
179
156
|
salary: null, // normalizer extracts from description text
|
|
157
|
+
content,
|
|
180
158
|
metadata: {
|
|
181
159
|
workdayTenant: tenant,
|
|
182
160
|
workdayEnv: env,
|
|
@@ -191,7 +169,6 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
191
169
|
if (typeof ctx.report === 'function') {
|
|
192
170
|
const org = orgs.find(Boolean) || {};
|
|
193
171
|
ctx.report({
|
|
194
|
-
ats: 'workday',
|
|
195
172
|
listed: postings.length,
|
|
196
173
|
prefiltered: candidates.length,
|
|
197
174
|
hydrated: hydrate.length,
|
|
@@ -220,49 +197,37 @@ function parseWorkdayRemoteType(remoteType) {
|
|
|
220
197
|
|
|
221
198
|
/**
|
|
222
199
|
* Workday list `postedOn` is a relative string ("Posted Today",
|
|
223
|
-
* "Posted 5 Days Ago", "Posted 30+ Days Ago").
|
|
224
|
-
*
|
|
225
|
-
* the library re-filters authoritatively on the real postedAt after
|
|
226
|
-
* hydration, so a false-keep here is corrected downstream.
|
|
200
|
+
* "Posted 5 Days Ago", "Posted 30+ Days Ago"). The days it names, or null
|
|
201
|
+
* when it names none.
|
|
227
202
|
*/
|
|
228
|
-
function
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
if (/
|
|
232
|
-
if (/yesterday/.test(s)) return days >= 1;
|
|
203
|
+
function daysAgo(postedOn) {
|
|
204
|
+
const s = String(postedOn || '').toLowerCase();
|
|
205
|
+
if (/today/.test(s)) return 0;
|
|
206
|
+
if (/yesterday/.test(s)) return 1;
|
|
233
207
|
const m = s.match(/(\d+)\+?\s*days?\s*ago/);
|
|
234
|
-
|
|
235
|
-
return true;
|
|
208
|
+
return m ? parseInt(m[1], 10) : null;
|
|
236
209
|
}
|
|
237
210
|
|
|
238
211
|
/**
|
|
239
|
-
*
|
|
240
|
-
*
|
|
212
|
+
* Decide membership in the last N days WITHOUT a network call.
|
|
213
|
+
* Unparseable -> keep (true); the library re-filters authoritatively on
|
|
214
|
+
* the real postedAt after hydration, so a false-keep here is corrected
|
|
215
|
+
* downstream.
|
|
241
216
|
*/
|
|
242
|
-
function
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
if (Number.isFinite(direct.getTime())) return direct.toISOString();
|
|
246
|
-
const s = String(v).toLowerCase();
|
|
247
|
-
let daysAgo = null;
|
|
248
|
-
if (/today/.test(s)) daysAgo = 0;
|
|
249
|
-
else if (/yesterday/.test(s)) daysAgo = 1;
|
|
250
|
-
else {
|
|
251
|
-
const m = s.match(/(\d+)\+?\s*days?\s*ago/);
|
|
252
|
-
if (m) daysAgo = parseInt(m[1], 10);
|
|
253
|
-
}
|
|
254
|
-
if (daysAgo === null) return null;
|
|
255
|
-
return new Date(Date.now() - daysAgo * 86400000).toISOString();
|
|
217
|
+
function withinDays(postedOn, days) {
|
|
218
|
+
const n = daysAgo(postedOn);
|
|
219
|
+
return n === null || n <= days;
|
|
256
220
|
}
|
|
257
221
|
|
|
258
222
|
/**
|
|
259
|
-
* Workday
|
|
260
|
-
*
|
|
223
|
+
* Coerce a Workday list `postedOn` (relative) into an approx ISO date
|
|
224
|
+
* so the library's postedWithinDays re-filter has a value to compare.
|
|
261
225
|
*/
|
|
262
|
-
function
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
226
|
+
function normalizePostedOn(v) {
|
|
227
|
+
const direct = toIso(v);
|
|
228
|
+
if (direct) return direct;
|
|
229
|
+
const n = daysAgo(v);
|
|
230
|
+
return n === null ? null : new Date(Date.now() - n * 86400000).toISOString();
|
|
266
231
|
}
|
|
267
232
|
|
|
268
233
|
/**
|
package/src/cli.js
CHANGED
|
@@ -11,6 +11,7 @@
|
|
|
11
11
|
|
|
12
12
|
import { realpathSync } from 'node:fs';
|
|
13
13
|
import { fileURLToPath } from 'node:url';
|
|
14
|
+
import { parseArgs } from 'node:util';
|
|
14
15
|
import { fetchJobs } from './index.js';
|
|
15
16
|
import { detectAtsDetailed, searchRegistry } from './registry.js';
|
|
16
17
|
|
|
@@ -19,30 +20,35 @@ const [,, command, ...args] = process.argv;
|
|
|
19
20
|
async function main() {
|
|
20
21
|
switch (command) {
|
|
21
22
|
case 'fetch': {
|
|
22
|
-
const
|
|
23
|
+
const string = { type: 'string' };
|
|
24
|
+
const { values: flags, positionals } = parseArgs({
|
|
25
|
+
args,
|
|
26
|
+
allowPositionals: true,
|
|
27
|
+
options: {
|
|
28
|
+
ats: string, 'title-filter': string, filter: string, 'posted-within-days': string,
|
|
29
|
+
'location-include': string, 'location-exclude': string, limit: string,
|
|
30
|
+
'workday-tenant': string, 'workday-env': string, 'workday-site': string,
|
|
31
|
+
json: { type: 'boolean' },
|
|
32
|
+
},
|
|
33
|
+
});
|
|
34
|
+
const company = positionals[0];
|
|
23
35
|
if (!company) { console.error('Usage: jd-intel fetch <company> [--ats <platform>] (omit --ats to auto-detect; run "jd-intel" for the platform list)'); process.exit(1); }
|
|
24
|
-
const
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
const
|
|
30
|
-
const
|
|
31
|
-
const
|
|
32
|
-
const
|
|
33
|
-
const locIncludeRaw = getArg('--location-include');
|
|
34
|
-
const locationIncludes = locIncludeRaw ? locIncludeRaw.split(',').map(s => s.trim()).filter(Boolean) : undefined;
|
|
35
|
-
const locExcludeRaw = getArg('--location-exclude');
|
|
36
|
-
const locationExcludes = locExcludeRaw ? locExcludeRaw.split(',').map(s => s.trim()).filter(Boolean) : undefined;
|
|
37
|
-
const limitRaw = getArg('--limit');
|
|
38
|
-
const limit = limitRaw !== undefined ? Number(limitRaw) : undefined;
|
|
36
|
+
const number = (v) => (v !== undefined ? Number(v) : undefined);
|
|
37
|
+
const list = (v) => (v ? v.split(',').map(s => s.trim()).filter(Boolean) : undefined);
|
|
38
|
+
let ats = flags.ats;
|
|
39
|
+
const titleFilter = flags['title-filter'];
|
|
40
|
+
const filter = flags.filter;
|
|
41
|
+
const postedWithinDays = number(flags['posted-within-days']);
|
|
42
|
+
const locationIncludes = list(flags['location-include']);
|
|
43
|
+
const locationExcludes = list(flags['location-exclude']);
|
|
44
|
+
const limit = number(flags.limit);
|
|
39
45
|
|
|
40
46
|
// Workday is keyed by a {tenant, env, site} triple, not a slug.
|
|
41
47
|
// Supplying it here makes a Workday board reachable without a
|
|
42
48
|
// registry entry; presence of the flags infers --ats workday.
|
|
43
|
-
const wdTenant =
|
|
44
|
-
const wdEnv =
|
|
45
|
-
const wdSite =
|
|
49
|
+
const wdTenant = flags['workday-tenant'];
|
|
50
|
+
const wdEnv = flags['workday-env'];
|
|
51
|
+
const wdSite = flags['workday-site'];
|
|
46
52
|
let config;
|
|
47
53
|
if (wdTenant || wdEnv || wdSite) {
|
|
48
54
|
if (!wdTenant || !wdEnv || !wdSite) {
|
|
@@ -91,8 +97,10 @@ async function main() {
|
|
|
91
97
|
const loc = job.location ? ` | ${job.location}` : '';
|
|
92
98
|
const dept = job.department ? ` [${job.department}]` : '';
|
|
93
99
|
console.log(` ${job.title}${dept}${loc}${salary}`);
|
|
94
|
-
console.log(` ${job.url}`);
|
|
95
|
-
if (job.
|
|
100
|
+
console.log(` ${job.url || '(no URL: posting not read)'}`);
|
|
101
|
+
if (job.content?.status === 'missing') {
|
|
102
|
+
console.log(` posting not read: ${job.content.reason}`);
|
|
103
|
+
} else if (job.description) {
|
|
96
104
|
const preview = job.description.substring(0, 120).replace(/\n/g, ' ');
|
|
97
105
|
console.log(` ${preview}...`);
|
|
98
106
|
}
|
|
@@ -103,7 +111,7 @@ async function main() {
|
|
|
103
111
|
console.log(` ... and ${jobs.length - 20} more. Use --json for full output.`);
|
|
104
112
|
}
|
|
105
113
|
|
|
106
|
-
if (
|
|
114
|
+
if (flags.json) {
|
|
107
115
|
console.log(JSON.stringify(jobs, null, 2));
|
|
108
116
|
}
|
|
109
117
|
break;
|
package/src/filters.js
CHANGED
|
@@ -88,11 +88,7 @@ export function pageJobs(jobs, { order = 'newest', offset = 0, limit = 100 } = {
|
|
|
88
88
|
|
|
89
89
|
const start = typeof offset === 'number' && offset > 0 ? offset : 0;
|
|
90
90
|
const end = typeof limit === 'number' ? start + limit : undefined;
|
|
91
|
-
|
|
92
|
-
result = result.slice(start, end);
|
|
93
|
-
}
|
|
94
|
-
|
|
95
|
-
return result;
|
|
91
|
+
return result.slice(start, end);
|
|
96
92
|
}
|
|
97
93
|
|
|
98
94
|
/**
|
|
@@ -174,3 +170,51 @@ export function makeLocationMatcher(needle) {
|
|
|
174
170
|
}
|
|
175
171
|
return (loc) => loc.includes(lower);
|
|
176
172
|
}
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* The list pre-filter and detail budget the two-step adapters (Workday,
|
|
176
|
+
* SmartRecruiters) share: narrow the cheap list rows with the filters a row
|
|
177
|
+
* can answer, then bound how many get a detail request.
|
|
178
|
+
*
|
|
179
|
+
* The library re-applies every filter after hydration, so a keep here is
|
|
180
|
+
* never final. `location(row)` returns the lowercased location, or null
|
|
181
|
+
* when the row cannot say where it is (it then stays a candidate through
|
|
182
|
+
* both location filters). `postedWithin(row, days)` decides recency.
|
|
183
|
+
*
|
|
184
|
+
* A description `filter` runs only after hydration, so that case keeps the
|
|
185
|
+
* full `max` budget instead of truncating to the page (which could hydrate
|
|
186
|
+
* rows that all fail the regex while better matches go unscanned). Without
|
|
187
|
+
* one, the budget is the page plus the offset before it. List order is kept.
|
|
188
|
+
*
|
|
189
|
+
* @returns {{ candidates: Array, hydrate: Array }}
|
|
190
|
+
*/
|
|
191
|
+
export function prefilterRows(rows, fc, { title, location, postedWithin, max }) {
|
|
192
|
+
let candidates = rows;
|
|
193
|
+
|
|
194
|
+
if (fc.titleFilter) {
|
|
195
|
+
const re = new RegExp(fc.titleFilter, 'i');
|
|
196
|
+
candidates = candidates.filter(p => re.test(title(p)));
|
|
197
|
+
}
|
|
198
|
+
if (Array.isArray(fc.locationIncludes) && fc.locationIncludes.length > 0) {
|
|
199
|
+
const inc = fc.locationIncludes.map(makeLocationMatcher);
|
|
200
|
+
candidates = candidates.filter(p => {
|
|
201
|
+
const loc = location(p);
|
|
202
|
+
return loc === null || inc.some(m => m(loc));
|
|
203
|
+
});
|
|
204
|
+
}
|
|
205
|
+
if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
|
|
206
|
+
const exc = fc.locationExcludes.map(makeLocationMatcher);
|
|
207
|
+
candidates = candidates.filter(p => {
|
|
208
|
+
const loc = location(p);
|
|
209
|
+
return loc === null || !exc.some(m => m(loc));
|
|
210
|
+
});
|
|
211
|
+
}
|
|
212
|
+
if (typeof fc.postedWithinDays === 'number') {
|
|
213
|
+
candidates = candidates.filter(p => postedWithin(p, fc.postedWithinDays));
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
|
|
217
|
+
const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
|
|
218
|
+
const cap = fc.filter ? max : Math.min(skip + limit, max);
|
|
219
|
+
return { candidates, hydrate: candidates.slice(0, cap) };
|
|
220
|
+
}
|
package/src/index.js
CHANGED
|
@@ -57,6 +57,7 @@ export async function fetchJobs(options = {}) {
|
|
|
57
57
|
* jobs: Array,
|
|
58
58
|
* total_matched: number,
|
|
59
59
|
* total_before_filters: number,
|
|
60
|
+
* content_missing: number,
|
|
60
61
|
* match: 'registry'|'probe'|'workday_override',
|
|
61
62
|
* company: { key: string, name: string }|null,
|
|
62
63
|
* boards: Array<object>,
|
|
@@ -65,6 +66,9 @@ export async function fetchJobs(options = {}) {
|
|
|
65
66
|
* jobs: the page. total_matched: matches before offset and limit.
|
|
66
67
|
* total_before_filters: rows every board listed before any filter (on
|
|
67
68
|
* Workday and SmartRecruiters the list count, not the rows they hydrated).
|
|
69
|
+
* content_missing: jobs whose posting was not read because the detail
|
|
70
|
+
* request failed (Workday, SmartRecruiters), counted before the filters:
|
|
71
|
+
* a description filter cannot match text that never arrived.
|
|
68
72
|
* match: how the company was resolved. company: the registry row's name
|
|
69
73
|
* and its key (normalized name); null unless match is 'registry'.
|
|
70
74
|
* boards: one entry per board that answered (see src/boards.js), with
|
|
@@ -194,6 +198,7 @@ export async function fetchJobsDetailed({
|
|
|
194
198
|
jobs: pageJobs(matched, { order, offset, limit }),
|
|
195
199
|
total_matched: matched.length,
|
|
196
200
|
total_before_filters: boards.reduce((n, b) => n + b.jobs_found, 0),
|
|
201
|
+
content_missing: rows.filter(j => j.content?.status === 'missing').length,
|
|
197
202
|
match,
|
|
198
203
|
company: match === 'registry' ? { key: normSlug(targets[0].name), name: targets[0].name } : null,
|
|
199
204
|
boards,
|
|
@@ -205,7 +210,13 @@ function registryTarget(hit, config) {
|
|
|
205
210
|
return { ats: hit.ats, slug: hit.entry.slug, name: hit.entry.name, config: config || hit.entry.config };
|
|
206
211
|
}
|
|
207
212
|
|
|
208
|
-
|
|
213
|
+
/**
|
|
214
|
+
* The AtsError for a lookup where no board answered and at least one check
|
|
215
|
+
* failed: rate_limited when any failure was a 429, else ats_unreachable,
|
|
216
|
+
* with a message naming each failed check. `failed` is the list
|
|
217
|
+
* fetchJobsDetailed and detectAtsDetailed return.
|
|
218
|
+
*/
|
|
219
|
+
export function discoveryFailure(company, failed) {
|
|
209
220
|
const limited = failed.some(f => f.code === ERROR_CODES.RATE_LIMITED);
|
|
210
221
|
const checks = failed.map(f => `${f.ats} (${f.message})`).join('; ');
|
|
211
222
|
return new AtsError(
|
package/src/normalizer.js
CHANGED
|
@@ -1,5 +1,15 @@
|
|
|
1
1
|
import { createHash } from 'node:crypto';
|
|
2
2
|
|
|
3
|
+
/**
|
|
4
|
+
* A date value (string or epoch ms) as an ISO string, or null when it is
|
|
5
|
+
* missing or does not parse.
|
|
6
|
+
*/
|
|
7
|
+
export function toIso(value) {
|
|
8
|
+
if (!value) return null;
|
|
9
|
+
const d = new Date(value);
|
|
10
|
+
return Number.isNaN(d.getTime()) ? null : d.toISOString();
|
|
11
|
+
}
|
|
12
|
+
|
|
3
13
|
/**
|
|
4
14
|
* Generate a stable ID for a job posting.
|
|
5
15
|
*/
|
|
@@ -23,6 +33,9 @@ export function jobId(company, title, ats, location = '') {
|
|
|
23
33
|
* adapter to 'remote' | 'hybrid' | 'onsite', or null when the platform
|
|
24
34
|
* gives no signal. `raw.locations` lists every place the posting is open
|
|
25
35
|
* in; `location` stays the primary because it feeds the id (issue #68).
|
|
36
|
+
*
|
|
37
|
+
* `raw.content` is set only by a two-step adapter whose detail request
|
|
38
|
+
* failed (see missingContent). Every other job was read in full.
|
|
26
39
|
*/
|
|
27
40
|
export function normalize(raw, ats) {
|
|
28
41
|
const now = new Date().toISOString();
|
|
@@ -47,10 +60,21 @@ export function normalize(raw, ats) {
|
|
|
47
60
|
firstSeen: now,
|
|
48
61
|
lastSeen: now,
|
|
49
62
|
status: 'open',
|
|
63
|
+
content: raw.content || { status: 'complete', reason: null },
|
|
50
64
|
metadata: raw.metadata || {},
|
|
51
65
|
};
|
|
52
66
|
}
|
|
53
67
|
|
|
68
|
+
/**
|
|
69
|
+
* The `content` of a job whose detail request failed, so its description
|
|
70
|
+
* and pay were never read (issue #85). `failure` is the non-OK Response or
|
|
71
|
+
* the error atsFetch threw: an HTTP status gives "http_503", anything else
|
|
72
|
+
* "network_error".
|
|
73
|
+
*/
|
|
74
|
+
export function missingContent(failure) {
|
|
75
|
+
return { status: 'missing', reason: failure?.status ? `http_${failure.status}` : 'network_error' };
|
|
76
|
+
}
|
|
77
|
+
|
|
54
78
|
const CURRENCY_CODES = 'USD|EUR|GBP|CAD|AUD|NZD|CHF|SEK|NOK|DKK|PLN|CZK|HUF|INR|SGD|HKD|JPY|CNY|BRL|MXN|ZAR|AED|ILS';
|
|
55
79
|
const SYMBOL_CURRENCY = { $: 'USD', '€': 'EUR', '£': 'GBP' };
|
|
56
80
|
|
|
@@ -140,14 +164,14 @@ function detectPeriod(before, after, min) {
|
|
|
140
164
|
return min >= 10000 ? 'year' : null;
|
|
141
165
|
}
|
|
142
166
|
|
|
143
|
-
function periodWord(text) {
|
|
167
|
+
export function periodWord(text) {
|
|
144
168
|
if (HOUR_RE.test(text)) return 'hour';
|
|
145
169
|
if (MONTH_RE.test(text)) return 'month';
|
|
146
170
|
if (YEAR_RE.test(text)) return 'year';
|
|
147
171
|
return null;
|
|
148
172
|
}
|
|
149
173
|
|
|
150
|
-
const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
|
|
174
|
+
export const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
|
|
151
175
|
|
|
152
176
|
/**
|
|
153
177
|
* The platform's own value wins. Without one, a keyword in the location
|
package/src/registry.js
CHANGED
|
@@ -2,15 +2,11 @@ import { readFile } from 'node:fs/promises';
|
|
|
2
2
|
import { join, dirname } from 'node:path';
|
|
3
3
|
import { fileURLToPath } from 'node:url';
|
|
4
4
|
import { AtsError } from './errors.js';
|
|
5
|
+
import { ADAPTERS, ATS_NAMES } from './adapters/index.js';
|
|
5
6
|
|
|
6
7
|
const __dirname = dirname(fileURLToPath(import.meta.url));
|
|
7
8
|
const REGISTRY_DIR = join(__dirname, '..', 'registry');
|
|
8
9
|
|
|
9
|
-
// The one order the registry is ever walked in. Lookups, detectAts and the
|
|
10
|
-
// loaded object all follow it, so which file answers for a slug does not
|
|
11
|
-
// depend on which file's load finished first (issue #87).
|
|
12
|
-
const PLATFORMS = ['greenhouse', 'lever', 'ashby', 'smartrecruiters', 'teamtailor', 'recruitee', 'workday'];
|
|
13
|
-
|
|
14
10
|
// Network-first registry. A hosted copy lets installed bundles AND npx users
|
|
15
11
|
// pick up newly-added companies without reinstalling; the on-disk copy that
|
|
16
12
|
// ships with the package is the guaranteed offline fallback. The base URL is
|
|
@@ -72,8 +68,11 @@ async function loadPlatform(platform) {
|
|
|
72
68
|
*/
|
|
73
69
|
export async function loadRegistry(ats) {
|
|
74
70
|
if (ats) return loadPlatform(ats);
|
|
75
|
-
|
|
76
|
-
|
|
71
|
+
// ATS_NAMES is the one order the registry is ever walked in. Lookups,
|
|
72
|
+
// detectAts and this object all follow it, so which file answers for a
|
|
73
|
+
// slug does not depend on which file's load finished first (issue #87).
|
|
74
|
+
const lists = await Promise.all(ATS_NAMES.map(loadPlatform));
|
|
75
|
+
return Object.fromEntries(ATS_NAMES.map((platform, i) => [platform, lists[i]]));
|
|
77
76
|
}
|
|
78
77
|
|
|
79
78
|
/**
|
|
@@ -94,7 +93,10 @@ export function getRegistrySource() {
|
|
|
94
93
|
}
|
|
95
94
|
|
|
96
95
|
/**
|
|
97
|
-
* Search registry for companies matching a query
|
|
96
|
+
* Search registry for companies matching a query, best match first: an
|
|
97
|
+
* exact name or slug, then a name that starts with the query, then a name
|
|
98
|
+
* that contains it, then a sector-only match. Ties keep platform order, so
|
|
99
|
+
* a caller that cuts the list drops the weakest matches (issue #62).
|
|
98
100
|
*/
|
|
99
101
|
export async function searchRegistry(query) {
|
|
100
102
|
const all = await loadRegistry();
|
|
@@ -105,13 +107,16 @@ export async function searchRegistry(query) {
|
|
|
105
107
|
for (const company of companies) {
|
|
106
108
|
const name = (company.name || company.slug || '').toLowerCase();
|
|
107
109
|
const sector = (company.sector || '').toLowerCase();
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
110
|
+
const rank = name === lower || String(company.slug).toLowerCase() === lower ? 0
|
|
111
|
+
: name.startsWith(lower) ? 1
|
|
112
|
+
: name.includes(lower) ? 2
|
|
113
|
+
: sector.includes(lower) ? 3
|
|
114
|
+
: -1;
|
|
115
|
+
if (rank >= 0) results.push({ rank, row: { ...company, ats } });
|
|
111
116
|
}
|
|
112
117
|
}
|
|
113
118
|
|
|
114
|
-
return results;
|
|
119
|
+
return results.sort((a, b) => a.rank - b.rank).map(r => r.row);
|
|
115
120
|
}
|
|
116
121
|
|
|
117
122
|
// Slug match is case/punctuation-insensitive: registry slugs are stored
|
|
@@ -120,18 +125,7 @@ export async function searchRegistry(query) {
|
|
|
120
125
|
// normalized forms keeps registry-first routing working for those.
|
|
121
126
|
export const normSlug = (s) => String(s).toLowerCase().replace(/[^a-z0-9]/g, '');
|
|
122
127
|
|
|
123
|
-
|
|
124
|
-
// order the object's keys are in.
|
|
125
|
-
function platformEntries(all) {
|
|
126
|
-
return PLATFORMS.map(ats => [ats, all[ats] || []]);
|
|
127
|
-
}
|
|
128
|
-
|
|
129
|
-
function platformIndex(ats) {
|
|
130
|
-
const i = PLATFORMS.indexOf(ats);
|
|
131
|
-
return i === -1 ? PLATFORMS.length : i;
|
|
132
|
-
}
|
|
133
|
-
|
|
134
|
-
const byPlatform = (a, b) => platformIndex(a.ats) - platformIndex(b.ats);
|
|
128
|
+
const byPlatform = (a, b) => ATS_NAMES.indexOf(a.ats) - ATS_NAMES.indexOf(b.ats);
|
|
135
129
|
|
|
136
130
|
/**
|
|
137
131
|
* Look up which ATS a slug belongs to in the registry.
|
|
@@ -147,14 +141,14 @@ export async function findAtsBySlug(slug) {
|
|
|
147
141
|
* Unlike findAtsBySlug (returns just the ats name), this returns the
|
|
148
142
|
* whole entry so callers can read adapter-specific config (e.g. the
|
|
149
143
|
* Workday {tenant, env, site} triple). The files are searched in
|
|
150
|
-
*
|
|
144
|
+
* ATS_NAMES order, so the first match is the same on every call.
|
|
151
145
|
*
|
|
152
146
|
* @returns {Promise<{ats: string, entry: object}|null>}
|
|
153
147
|
*/
|
|
154
148
|
export async function findEntryBySlug(slug) {
|
|
155
149
|
const all = await loadRegistry();
|
|
156
150
|
const key = normSlug(slug);
|
|
157
|
-
for (const [ats, companies] of
|
|
151
|
+
for (const [ats, companies] of Object.entries(all)) {
|
|
158
152
|
const entry = companies.find(c => normSlug(c.slug) === key);
|
|
159
153
|
if (entry) return { ats, entry };
|
|
160
154
|
}
|
|
@@ -171,7 +165,7 @@ export async function findEntryBySlug(slug) {
|
|
|
171
165
|
* adds nothing, and an AtsError (429, 5xx, 401, network) goes to `failed`
|
|
172
166
|
* with its code, so a board the probe could not check never reads as
|
|
173
167
|
* absent (issue #55). Any other error is a bug and is rethrown. Both lists
|
|
174
|
-
* come back in
|
|
168
|
+
* come back in ATS_NAMES order, never in completion order.
|
|
175
169
|
*
|
|
176
170
|
* @returns {Promise<{
|
|
177
171
|
* boards: Array<{ ats: string, slug: string, source: 'registry'|'probe' }>,
|
|
@@ -179,14 +173,13 @@ export async function findEntryBySlug(slug) {
|
|
|
179
173
|
* }>}
|
|
180
174
|
*/
|
|
181
175
|
export async function detectAtsDetailed(companyName) {
|
|
182
|
-
const { ADAPTERS } = await import('./adapters/index.js');
|
|
183
176
|
const slug = normSlug(companyName);
|
|
184
177
|
const all = await loadRegistry();
|
|
185
178
|
|
|
186
179
|
const boards = [];
|
|
187
180
|
const failed = [];
|
|
188
181
|
const known = new Set();
|
|
189
|
-
for (const [ats, companies] of
|
|
182
|
+
for (const [ats, companies] of Object.entries(all)) {
|
|
190
183
|
const entry = companies.find(c => normSlug(c.slug) === slug);
|
|
191
184
|
if (entry) {
|
|
192
185
|
boards.push({ ats, slug: entry.slug, source: 'registry' });
|