jd-intel 0.10.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -1
- package/package.json +1 -1
- package/src/adapters/ashby.js +6 -12
- package/src/adapters/greenhouse.js +79 -27
- package/src/adapters/index.js +24 -14
- package/src/adapters/lever.js +5 -10
- package/src/adapters/recruitee.js +1 -2
- package/src/adapters/smartrecruiters.js +27 -54
- package/src/adapters/teamtailor.js +1 -2
- package/src/adapters/workday.js +50 -75
- package/src/cli.js +30 -22
- package/src/filters.js +49 -5
- package/src/index.js +12 -1
- package/src/normalizer.js +16 -2
- package/src/registry.js +16 -21
package/README.md
CHANGED
|
@@ -240,7 +240,8 @@ No custom parsing per company.
|
|
|
240
240
|
| `locationType` | `remote`, `hybrid`, `onsite`, or `unknown` when neither the platform nor the location text says |
|
|
241
241
|
| `workplace` | `{ type, source }`. `type` repeats `locationType`; `source` is `ats` when the platform stated it, `text` when read from the location string, null when unknown |
|
|
242
242
|
| `salary` | Min-max range with `currency`, plus `period` (`year`, `month`, `hour`, or null) and `source` (`ats` when the platform supplied it, `text` when parsed from the posting). Null when nothing is stated |
|
|
243
|
-
| `description` | Full JD in clean markdown |
|
|
243
|
+
| `description` | Full JD in clean markdown. Empty when `content.status` is `missing` |
|
|
244
|
+
| `content` | `{ status, reason }`. `complete` when the posting was read. `missing` when Workday or SmartRecruiters listed the job but its detail request failed (`reason`: `http_503`, `http_429`, `network_error`), so `description` and `salary` are unknown |
|
|
244
245
|
| `url` | Direct link to the posting |
|
|
245
246
|
| `postedAt` | Publication date (when provided) |
|
|
246
247
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "jd-intel",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.11.0",
|
|
4
4
|
"description": "Fetch and normalize job descriptions across seven major ATS (Greenhouse, Lever, Ashby, Workday, and more), for your AI assistant. No copy-paste.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "src/index.js",
|
package/src/adapters/ashby.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize, extractSalaryFromText } from '../normalizer.js';
|
|
1
|
+
import { normalize, extractSalaryFromText, WORKPLACE_TYPES } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
import { atsFetch, probeResult } from '../http.js';
|
|
4
4
|
|
|
@@ -14,11 +14,9 @@ const BOARD_URL = 'https://api.ashbyhq.com/posting-api/job-board';
|
|
|
14
14
|
* into a silent empty result (issue #55).
|
|
15
15
|
*
|
|
16
16
|
* @param {string} slug - Company slug (e.g., 'notion', 'linear')
|
|
17
|
-
* @param {object} [ctx] - { report }; report is called once with
|
|
18
|
-
* { ats, org_name, org_url } when given
|
|
19
17
|
* @returns {Promise<Array>} Normalized job objects
|
|
20
18
|
*/
|
|
21
|
-
export async function fetchAshby(slug
|
|
19
|
+
export async function fetchAshby(slug) {
|
|
22
20
|
const url = `${BOARD_URL}/${slug}?includeCompensation=true`;
|
|
23
21
|
const resp = await atsFetch(url);
|
|
24
22
|
|
|
@@ -31,10 +29,8 @@ export async function fetchAshby(slug, ctx = {}) {
|
|
|
31
29
|
const jobs = data.jobs || [];
|
|
32
30
|
|
|
33
31
|
// The REST response is { jobs, apiVersion }: no organization name, and
|
|
34
|
-
// every link is on jobs.ashbyhq.com.
|
|
35
|
-
|
|
36
|
-
ctx.report({ ats: 'ashby', org_name: null, org_url: null });
|
|
37
|
-
}
|
|
32
|
+
// every link is on jobs.ashbyhq.com. Nothing to report, so the board's
|
|
33
|
+
// org_name and org_url stay null (issue #58).
|
|
38
34
|
|
|
39
35
|
return jobs.map(job => {
|
|
40
36
|
const comp = job.compensation || {};
|
|
@@ -69,16 +65,14 @@ export async function fetchAshby(slug, ctx = {}) {
|
|
|
69
65
|
});
|
|
70
66
|
}
|
|
71
67
|
|
|
72
|
-
const WORKPLACE_TYPES = { remote: 'remote', hybrid: 'hybrid', onsite: 'onsite' };
|
|
73
|
-
|
|
74
68
|
/**
|
|
75
69
|
* `workplaceType` is 'Remote', 'Hybrid' or 'OnSite'. `isRemote` is the
|
|
76
70
|
* older flag and can only say remote, so it is the fallback when the type
|
|
77
71
|
* is absent. false means nothing: the role may be hybrid or onsite.
|
|
78
72
|
*/
|
|
79
73
|
function parseAshbyWorkplace(job) {
|
|
80
|
-
const type =
|
|
81
|
-
if (type) return type;
|
|
74
|
+
const type = String(job.workplaceType || '').toLowerCase();
|
|
75
|
+
if (WORKPLACE_TYPES.has(type)) return type;
|
|
82
76
|
return job.isRemote === true ? 'remote' : null;
|
|
83
77
|
}
|
|
84
78
|
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize, decodeEntities } from '../normalizer.js';
|
|
1
|
+
import { normalize, decodeEntities, periodWord } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
import { atsFetch, probeResult } from '../http.js';
|
|
4
4
|
|
|
@@ -11,11 +11,12 @@ const BASE_URL = 'https://boards-api.greenhouse.io/v1/boards';
|
|
|
11
11
|
*
|
|
12
12
|
* @param {string} slug - Company slug (e.g., 'stripe', 'notion')
|
|
13
13
|
* @param {object} [ctx] - { report }; report is called once with
|
|
14
|
-
* {
|
|
14
|
+
* { org_name, org_url } when given
|
|
15
15
|
* @returns {Promise<Array>} Normalized job objects
|
|
16
16
|
*/
|
|
17
17
|
export async function fetchGreenhouse(slug, ctx = {}) {
|
|
18
|
-
|
|
18
|
+
// pay_transparency adds pay_input_ranges to each row (issue #86).
|
|
19
|
+
const url = `${BASE_URL}/${slug}/jobs?content=true&pay_transparency=true`;
|
|
19
20
|
const resp = await atsFetch(url);
|
|
20
21
|
|
|
21
22
|
if (!resp.ok) {
|
|
@@ -31,35 +32,86 @@ export async function fetchGreenhouse(slug, ctx = {}) {
|
|
|
31
32
|
// there is no company host to report (issue #58).
|
|
32
33
|
if (typeof ctx.report === 'function') {
|
|
33
34
|
ctx.report({
|
|
34
|
-
ats: 'greenhouse',
|
|
35
35
|
org_name: jobs.find(j => j.company_name)?.company_name || null,
|
|
36
36
|
org_url: null,
|
|
37
37
|
});
|
|
38
38
|
}
|
|
39
39
|
|
|
40
|
-
return jobs.map(job =>
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
40
|
+
return jobs.map(job => {
|
|
41
|
+
const payRanges = parsePayRanges(job.pay_input_ranges);
|
|
42
|
+
return normalize({
|
|
43
|
+
companySlug: slug,
|
|
44
|
+
company: data.name || slug,
|
|
45
|
+
title: job.title || '',
|
|
46
|
+
department: job.departments?.[0]?.name || '',
|
|
47
|
+
location: job.location?.name || '',
|
|
48
|
+
workplace: parseGreenhouseWorkplace(job.metadata),
|
|
49
|
+
// `content` arrives HTML-escaped (`<p>`). Decode that outer layer
|
|
50
|
+
// once so normalize() sees real tags; it strips and decodes the rest.
|
|
51
|
+
description: decodeEntities(job.content || ''),
|
|
52
|
+
url: job.absolute_url || '',
|
|
53
|
+
// updated_at is an edit time that many boards bulk-refresh, so it is not
|
|
54
|
+
// a posting date. first_published is. Fallback covers boards without it (#69).
|
|
55
|
+
postedAt: job.first_published || job.updated_at || null,
|
|
56
|
+
salary: salaryFromRanges(payRanges), // null without structured pay; normalize() then parses the text
|
|
57
|
+
metadata: {
|
|
58
|
+
greenhouseId: job.id,
|
|
59
|
+
internal_job_id: job.internal_job_id,
|
|
60
|
+
departments: job.departments?.map(d => d.name) || [],
|
|
61
|
+
offices: job.offices?.map(o => o.name) || [],
|
|
62
|
+
updatedAt: job.updated_at,
|
|
63
|
+
payRanges,
|
|
64
|
+
},
|
|
65
|
+
}, 'greenhouse');
|
|
66
|
+
});
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* `pay_input_ranges` is the board's pay transparency data:
|
|
71
|
+
* [{ min_cents, max_cents, currency_type, title, blurb }]. The blurb is
|
|
72
|
+
* boilerplate already rendered in the description, so it is dropped, and
|
|
73
|
+
* the platform can send the same entry twice, so entries are deduplicated.
|
|
74
|
+
* A board that does not publish pay sends an empty array or no key.
|
|
75
|
+
*/
|
|
76
|
+
function parsePayRanges(ranges) {
|
|
77
|
+
const seen = new Set();
|
|
78
|
+
const out = [];
|
|
79
|
+
for (const r of Array.isArray(ranges) ? ranges : []) {
|
|
80
|
+
const range = {
|
|
81
|
+
title: r.title || '',
|
|
82
|
+
min: Number.isFinite(r.min_cents) ? r.min_cents / 100 : null,
|
|
83
|
+
max: Number.isFinite(r.max_cents) ? r.max_cents / 100 : null,
|
|
84
|
+
currency: r.currency_type || 'USD',
|
|
85
|
+
};
|
|
86
|
+
const key = JSON.stringify(range);
|
|
87
|
+
if ((range.min === null && range.max === null) || seen.has(key)) continue;
|
|
88
|
+
seen.add(key);
|
|
89
|
+
out.push(range);
|
|
90
|
+
}
|
|
91
|
+
return out;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* One salary from the ranges. Ranges in one currency span (lowest min,
|
|
96
|
+
* highest max), as the Ashby adapter does for its tiers; with mixed
|
|
97
|
+
* currencies the first range stands alone. Every range stays in
|
|
98
|
+
* metadata.payRanges. The field has no period, so the period is the one
|
|
99
|
+
* the range titles state ("Annual", "Hourly") and null when they state
|
|
100
|
+
* none or disagree: an 'ats' value carries no guessed period.
|
|
101
|
+
*/
|
|
102
|
+
function salaryFromRanges(ranges) {
|
|
103
|
+
if (ranges.length === 0) return null;
|
|
104
|
+
const used = ranges.every(r => r.currency === ranges[0].currency) ? ranges : [ranges[0]];
|
|
105
|
+
const mins = used.map(r => r.min).filter(v => v !== null);
|
|
106
|
+
const maxes = used.map(r => r.max).filter(v => v !== null);
|
|
107
|
+
const periods = new Set(used.map(r => periodWord(r.title)));
|
|
108
|
+
return {
|
|
109
|
+
min: mins.length ? Math.min(...mins) : null,
|
|
110
|
+
max: maxes.length ? Math.max(...maxes) : null,
|
|
111
|
+
currency: used[0].currency,
|
|
112
|
+
period: periods.size === 1 ? [...periods][0] : null,
|
|
113
|
+
source: 'ats',
|
|
114
|
+
};
|
|
63
115
|
}
|
|
64
116
|
|
|
65
117
|
/**
|
package/src/adapters/index.js
CHANGED
|
@@ -1,19 +1,29 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
1
|
+
import { fetchGreenhouse, hasGreenhouse } from './greenhouse.js';
|
|
2
|
+
import { fetchLever, hasLever } from './lever.js';
|
|
3
|
+
import { fetchAshby, hasAshby } from './ashby.js';
|
|
4
|
+
import { fetchSmartrecruiters, hasSmartrecruiters } from './smartrecruiters.js';
|
|
5
|
+
import { fetchTeamtailor, hasTeamtailor } from './teamtailor.js';
|
|
6
|
+
import { fetchRecruitee, hasRecruitee } from './recruitee.js';
|
|
7
|
+
import { fetchWorkday, hasWorkday } from './workday.js';
|
|
8
|
+
|
|
9
|
+
export {
|
|
10
|
+
fetchGreenhouse, hasGreenhouse,
|
|
11
|
+
fetchLever, hasLever,
|
|
12
|
+
fetchAshby, hasAshby,
|
|
13
|
+
fetchSmartrecruiters, hasSmartrecruiters,
|
|
14
|
+
fetchTeamtailor, hasTeamtailor,
|
|
15
|
+
fetchRecruitee, hasRecruitee,
|
|
16
|
+
fetchWorkday, hasWorkday,
|
|
17
|
+
};
|
|
8
18
|
|
|
9
19
|
export const ADAPTERS = {
|
|
10
|
-
greenhouse: { fetch:
|
|
11
|
-
lever: { fetch:
|
|
12
|
-
ashby: { fetch:
|
|
13
|
-
smartrecruiters: { fetch:
|
|
14
|
-
teamtailor: { fetch:
|
|
15
|
-
recruitee: { fetch:
|
|
16
|
-
workday: { fetch:
|
|
20
|
+
greenhouse: { fetch: fetchGreenhouse, has: hasGreenhouse },
|
|
21
|
+
lever: { fetch: fetchLever, has: hasLever },
|
|
22
|
+
ashby: { fetch: fetchAshby, has: hasAshby },
|
|
23
|
+
smartrecruiters: { fetch: fetchSmartrecruiters, has: hasSmartrecruiters },
|
|
24
|
+
teamtailor: { fetch: fetchTeamtailor, has: hasTeamtailor },
|
|
25
|
+
recruitee: { fetch: fetchRecruitee, has: hasRecruitee },
|
|
26
|
+
workday: { fetch: fetchWorkday, has: hasWorkday },
|
|
17
27
|
};
|
|
18
28
|
|
|
19
29
|
export const ATS_NAMES = Object.keys(ADAPTERS);
|
package/src/adapters/lever.js
CHANGED
|
@@ -5,8 +5,6 @@ import { atsFetch, probeResult } from '../http.js';
|
|
|
5
5
|
const BASE_URL = 'https://api.lever.co/v0/postings';
|
|
6
6
|
|
|
7
7
|
const PERIODS = { 'per-year-salary': 'year', 'per-month-salary': 'month', 'per-hour-wage': 'hour' };
|
|
8
|
-
// Lever's workplaceType is one of these or 'unspecified'.
|
|
9
|
-
const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
|
|
10
8
|
|
|
11
9
|
/**
|
|
12
10
|
* Fetch all jobs from a Lever job board.
|
|
@@ -14,11 +12,9 @@ const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
|
|
|
14
12
|
* Docs: https://github.com/lever/postings-api
|
|
15
13
|
*
|
|
16
14
|
* @param {string} slug - Company slug (e.g., 'stripe', 'figma')
|
|
17
|
-
* @param {object} [ctx] - { report }; report is called once with
|
|
18
|
-
* { ats, org_name, org_url } when given
|
|
19
15
|
* @returns {Promise<Array>} Normalized job objects
|
|
20
16
|
*/
|
|
21
|
-
export async function fetchLever(slug
|
|
17
|
+
export async function fetchLever(slug) {
|
|
22
18
|
const url = `${BASE_URL}/${slug}?mode=json`;
|
|
23
19
|
const resp = await atsFetch(url);
|
|
24
20
|
|
|
@@ -31,10 +27,8 @@ export async function fetchLever(slug, ctx = {}) {
|
|
|
31
27
|
if (!Array.isArray(jobs)) return [];
|
|
32
28
|
|
|
33
29
|
// The postings response is a bare array of jobs: no organization name
|
|
34
|
-
// anywhere, and every link is on jobs.lever.co.
|
|
35
|
-
|
|
36
|
-
ctx.report({ ats: 'lever', org_name: null, org_url: null });
|
|
37
|
-
}
|
|
30
|
+
// anywhere, and every link is on jobs.lever.co. Nothing to report, so
|
|
31
|
+
// the board's org_name and org_url stay null (issue #58).
|
|
38
32
|
|
|
39
33
|
return jobs.map(job => normalize({
|
|
40
34
|
companySlug: slug,
|
|
@@ -46,7 +40,8 @@ export async function fetchLever(slug, ctx = {}) {
|
|
|
46
40
|
department: job.categories?.department || job.categories?.team || '',
|
|
47
41
|
location: job.categories?.location || '',
|
|
48
42
|
locations: job.categories?.allLocations || [],
|
|
49
|
-
|
|
43
|
+
// 'remote', 'hybrid', 'onsite' or 'unspecified'; normalize() ignores the last.
|
|
44
|
+
workplace: job.workplaceType,
|
|
50
45
|
description: buildDescription(job),
|
|
51
46
|
url: job.hostedUrl || '',
|
|
52
47
|
postedAt: job.createdAt ? new Date(job.createdAt).toISOString() : null,
|
|
@@ -21,7 +21,7 @@ import { orgHost } from '../boards.js';
|
|
|
21
21
|
*
|
|
22
22
|
* @param {string} slug - Recruitee company subdomain (e.g., 'vandebron')
|
|
23
23
|
* @param {object} [ctx] - { report }; report is called once with
|
|
24
|
-
* {
|
|
24
|
+
* { org_name, org_url } when given
|
|
25
25
|
* @returns {Promise<Array>} Normalized job objects
|
|
26
26
|
*/
|
|
27
27
|
export async function fetchRecruitee(slug, ctx = {}) {
|
|
@@ -41,7 +41,6 @@ export async function fetchRecruitee(slug, ctx = {}) {
|
|
|
41
41
|
// domain when the site has one and on {slug}.recruitee.com otherwise.
|
|
42
42
|
if (typeof ctx.report === 'function') {
|
|
43
43
|
ctx.report({
|
|
44
|
-
ats: 'recruitee',
|
|
45
44
|
org_name: offers.find(o => o.company_name)?.company_name || null,
|
|
46
45
|
org_url: orgHost(offers.find(o => o.careers_url)?.careers_url),
|
|
47
46
|
});
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { normalize } from '../normalizer.js';
|
|
1
|
+
import { normalize, missingContent } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
import { atsFetch, probeResult } from '../http.js';
|
|
4
|
-
import {
|
|
4
|
+
import { prefilterRows } from '../filters.js';
|
|
5
5
|
|
|
6
6
|
const BASE_URL = 'https://api.smartrecruiters.com/v1/companies';
|
|
7
7
|
const PAGE_SIZE = 100;
|
|
@@ -21,12 +21,12 @@ const MAX_DETAIL_FETCHES = 100;
|
|
|
21
21
|
* The list does carry name, location and releasedDate, so the same
|
|
22
22
|
* pre-filter and detail budget Workday applies run here: list-evaluable
|
|
23
23
|
* filters narrow the candidates, then at most MAX_DETAIL_FETCHES of them
|
|
24
|
-
* are hydrated (see
|
|
24
|
+
* are hydrated (see prefilterRows). Without a filterContext the
|
|
25
25
|
* cap still holds, so a direct call on a 400-posting tenant reads 100.
|
|
26
26
|
*
|
|
27
27
|
* @param {string} slug - SmartRecruiters company identifier (e.g., 'Visa')
|
|
28
28
|
* @param {object} [ctx] - { filterContext, report }; report is called once
|
|
29
|
-
* with {
|
|
29
|
+
* with { listed, prefiltered, hydrated, capped, org_name, org_url }
|
|
30
30
|
* when given
|
|
31
31
|
* @returns {Promise<Array>} Normalized job objects
|
|
32
32
|
*/
|
|
@@ -54,60 +54,28 @@ export async function fetchSmartrecruiters(slug, ctx = {}) {
|
|
|
54
54
|
if (content.length === 0 || offset >= (data.totalFound || 0)) break;
|
|
55
55
|
}
|
|
56
56
|
|
|
57
|
-
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost
|
|
58
|
-
// The list row carries name,
|
|
59
|
-
//
|
|
60
|
-
//
|
|
61
|
-
//
|
|
62
|
-
//
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
}
|
|
69
|
-
if (Array.isArray(fc.locationIncludes) && fc.locationIncludes.length > 0) {
|
|
70
|
-
const matchers = fc.locationIncludes.map(makeLocationMatcher);
|
|
71
|
-
candidates = candidates.filter(p => {
|
|
72
|
-
const loc = listLocation(p).location.toLowerCase();
|
|
73
|
-
return matchers.some(m => m(loc));
|
|
74
|
-
});
|
|
75
|
-
}
|
|
76
|
-
if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
|
|
77
|
-
const matchers = fc.locationExcludes.map(makeLocationMatcher);
|
|
78
|
-
candidates = candidates.filter(p => {
|
|
79
|
-
const loc = listLocation(p).location.toLowerCase();
|
|
80
|
-
return !loc || !matchers.some(m => m(loc));
|
|
81
|
-
});
|
|
82
|
-
}
|
|
83
|
-
if (typeof fc.postedWithinDays === 'number') {
|
|
84
|
-
// postedAt comes from releasedDate alone, so the library's rule can
|
|
85
|
-
// run here in full: a missing or unparseable date is out either way.
|
|
86
|
-
const cutoff = Date.now() - fc.postedWithinDays * 86400000;
|
|
87
|
-
candidates = candidates.filter(p => {
|
|
57
|
+
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost, then
|
|
58
|
+
// the detail budget (see prefilterRows). The list row carries name,
|
|
59
|
+
// location and releasedDate. The detail adds no location (unlike
|
|
60
|
+
// Workday's additionalLocations), so a row with none follows the
|
|
61
|
+
// library's rule now: out under includes, kept under excludes.
|
|
62
|
+
// postedAt comes from releasedDate alone, so a missing or unparseable
|
|
63
|
+
// date is out, as it is in the library.
|
|
64
|
+
const { candidates, hydrate } = prefilterRows(postings, fc, {
|
|
65
|
+
title: p => p.name || '',
|
|
66
|
+
location: p => listLocation(p).location.toLowerCase(),
|
|
67
|
+
postedWithin: (p, days) => {
|
|
88
68
|
const released = new Date(p.releasedDate || '').getTime();
|
|
89
|
-
return Number.isFinite(released) && released >=
|
|
90
|
-
}
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
// 3. Bound the detail-fetch set, Workday's reasoning verbatim: a
|
|
94
|
-
// description `filter` is applied by the library AFTER this returns,
|
|
95
|
-
// so that case keeps the full backstop instead of truncating to
|
|
96
|
-
// `limit` (which could hydrate jobs that all fail the regex while
|
|
97
|
-
// better matches go unscanned). The library pages with `offset`
|
|
98
|
-
// after this returns, so the budget covers the page plus what
|
|
99
|
-
// precedes it. Candidates keep list order.
|
|
100
|
-
const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
|
|
101
|
-
const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
|
|
102
|
-
const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
|
|
103
|
-
const hydrate = candidates.slice(0, cap);
|
|
69
|
+
return Number.isFinite(released) && released >= Date.now() - days * 86400000;
|
|
70
|
+
},
|
|
71
|
+
max: MAX_DETAIL_FETCHES,
|
|
72
|
+
});
|
|
104
73
|
|
|
105
74
|
// Every list row carries company { identifier, name }. Neither the list
|
|
106
75
|
// nor the detail has a company website, and postingUrl is always on
|
|
107
76
|
// jobs.smartrecruiters.com, so org_url stays null (issue #58).
|
|
108
77
|
if (typeof ctx.report === 'function') {
|
|
109
78
|
ctx.report({
|
|
110
|
-
ats: 'smartrecruiters',
|
|
111
79
|
listed: postings.length,
|
|
112
80
|
prefiltered: candidates.length,
|
|
113
81
|
hydrated: hydrate.length,
|
|
@@ -125,6 +93,7 @@ export async function fetchSmartrecruiters(slug, ctx = {}) {
|
|
|
125
93
|
let sections = {};
|
|
126
94
|
let postingUrl = '';
|
|
127
95
|
let salary = null;
|
|
96
|
+
let content; // set only when the detail could not be read (issue #85)
|
|
128
97
|
|
|
129
98
|
try {
|
|
130
99
|
const detailResp = await atsFetch(`${BASE_URL}/${slug}/postings/${p.id}`);
|
|
@@ -133,10 +102,13 @@ export async function fetchSmartrecruiters(slug, ctx = {}) {
|
|
|
133
102
|
sections = detail.jobAd?.sections || {};
|
|
134
103
|
postingUrl = detail.postingUrl || detail.applyUrl || '';
|
|
135
104
|
salary = parseCompensation(detail.compensation);
|
|
105
|
+
} else {
|
|
106
|
+
content = missingContent(detailResp);
|
|
136
107
|
}
|
|
137
|
-
} catch {
|
|
138
|
-
// Detail fetch failed, retries included:
|
|
139
|
-
//
|
|
108
|
+
} catch (err) {
|
|
109
|
+
// Detail fetch failed, retries included: list-only fields (no
|
|
110
|
+
// description, no url), marked missing.
|
|
111
|
+
content = missingContent(err);
|
|
140
112
|
}
|
|
141
113
|
|
|
142
114
|
const description = [
|
|
@@ -158,6 +130,7 @@ export async function fetchSmartrecruiters(slug, ctx = {}) {
|
|
|
158
130
|
url: postingUrl,
|
|
159
131
|
postedAt: p.releasedDate || null,
|
|
160
132
|
salary, // null when the detail has no compensation; normalize() then parses text
|
|
133
|
+
content,
|
|
161
134
|
metadata: {
|
|
162
135
|
smartRecruitersId: p.id,
|
|
163
136
|
refNumber: p.refNumber || '',
|
|
@@ -26,7 +26,7 @@ import { orgHost } from '../boards.js';
|
|
|
26
26
|
*
|
|
27
27
|
* @param {string} slug - TeamTailor career-site slug (e.g., 'tibber')
|
|
28
28
|
* @param {object} [ctx] - { report }; report is called once with
|
|
29
|
-
* {
|
|
29
|
+
* { org_name, org_url } when given
|
|
30
30
|
* @returns {Promise<Array>} Normalized job objects
|
|
31
31
|
*/
|
|
32
32
|
// Most sites are {slug}.teamtailor.com, but some sit on a regional
|
|
@@ -85,7 +85,6 @@ export async function fetchTeamtailor(slug, ctx = {}) {
|
|
|
85
85
|
|| xml.match(/<channel>[\s\S]*?<link>([\s\S]*?)<\/link>/)?.[1]
|
|
86
86
|
|| '';
|
|
87
87
|
ctx.report({
|
|
88
|
-
ats: 'teamtailor',
|
|
89
88
|
org_name: decodeEntities(channelTitle) || null,
|
|
90
89
|
org_url: orgHost(link.trim()),
|
|
91
90
|
});
|
package/src/adapters/workday.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { normalize } from '../normalizer.js';
|
|
1
|
+
import { normalize, missingContent } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
-
import {
|
|
3
|
+
import { prefilterRows } from '../filters.js';
|
|
4
4
|
import { atsFetch } from '../http.js';
|
|
5
5
|
import { orgHost } from '../boards.js';
|
|
6
6
|
|
|
@@ -35,7 +35,7 @@ const MULTI_LOCATION = /^\s*\d+\s+locations?\s*$/;
|
|
|
35
35
|
* @param {string} slug - normalized company slug (registry routing key)
|
|
36
36
|
* @param {object} [ctx] - { config:{tenant,env,site}, companyName, filterContext, report };
|
|
37
37
|
* report is called once, after hydration, with
|
|
38
|
-
* {
|
|
38
|
+
* { listed, prefiltered, hydrated, capped, org_name, org_url } when given
|
|
39
39
|
* @returns {Promise<Array>} Normalized job objects
|
|
40
40
|
*/
|
|
41
41
|
export async function fetchWorkday(slug, ctx = {}) {
|
|
@@ -92,55 +92,27 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
92
92
|
if (firstTotal > 0 && offset >= firstTotal) break;
|
|
93
93
|
}
|
|
94
94
|
|
|
95
|
-
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost
|
|
96
|
-
//
|
|
97
|
-
//
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
//
|
|
105
|
-
//
|
|
106
|
-
//
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
const inc = fc.locationIncludes.map(makeLocationMatcher);
|
|
111
|
-
candidates = candidates.filter(p => {
|
|
112
|
-
const loc = (p.locationsText || '').toLowerCase();
|
|
113
|
-
return MULTI_LOCATION.test(loc) || inc.some(m => m(loc));
|
|
114
|
-
});
|
|
115
|
-
}
|
|
116
|
-
if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
|
|
117
|
-
const exc = fc.locationExcludes.map(makeLocationMatcher);
|
|
118
|
-
candidates = candidates.filter(p => {
|
|
95
|
+
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost, then
|
|
96
|
+
// the detail budget (see prefilterRows). The list carries
|
|
97
|
+
// title/locationsText/postedOn, enough to apply titleFilter, location
|
|
98
|
+
// and recency without descriptions. "2 Locations" says nothing about
|
|
99
|
+
// where: the row stays a candidate through both location filters and
|
|
100
|
+
// the library's pass after hydration decides on the detail's location
|
|
101
|
+
// list (issue #61).
|
|
102
|
+
// NOTE: huge-tenant coverage is intentionally capped for v1. Two caps
|
|
103
|
+
// apply: the list scan above stops at LIST_PAGE_HARD_CAP pages (2000
|
|
104
|
+
// postings, enough for Salesforce's ~1398), and the detail set is cut
|
|
105
|
+
// to MAX_DETAIL_FETCHES here. Proper fix (smart pagination, surfaced
|
|
106
|
+
// truncation) is tracked in #26.
|
|
107
|
+
const { candidates, hydrate } = prefilterRows(postings, fc, {
|
|
108
|
+
title: p => p.title || '',
|
|
109
|
+
location: p => {
|
|
119
110
|
const loc = (p.locationsText || '').toLowerCase();
|
|
120
|
-
return MULTI_LOCATION.test(loc)
|
|
121
|
-
}
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
}
|
|
126
|
-
|
|
127
|
-
// 3. Bound the detail-fetch set.
|
|
128
|
-
// NOTE: huge-tenant coverage is intentionally capped for v1. Two
|
|
129
|
-
// caps apply: the list scan above stops at LIST_PAGE_HARD_CAP pages
|
|
130
|
-
// (2000 postings, enough for Salesforce's ~1398), and the detail set
|
|
131
|
-
// is cut to MAX_DETAIL_FETCHES here. A description `filter` is
|
|
132
|
-
// applied by the library AFTER this returns, so for that case we
|
|
133
|
-
// keep the full backstop instead of truncating tightly to `limit`
|
|
134
|
-
// (which could hydrate jobs that all fail the regex while better
|
|
135
|
-
// matches go unscanned). The library pages with `offset` after this
|
|
136
|
-
// returns, so the budget covers the page plus what precedes it.
|
|
137
|
-
// Proper fix (smart pagination / rate-limited concurrency / surfaced
|
|
138
|
-
// truncation) is tracked in #26, to be designed alongside
|
|
139
|
-
// retry/rate-limit work (#7).
|
|
140
|
-
const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
|
|
141
|
-
const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
|
|
142
|
-
const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
|
|
143
|
-
const hydrate = candidates.slice(0, cap);
|
|
111
|
+
return MULTI_LOCATION.test(loc) ? null : loc;
|
|
112
|
+
},
|
|
113
|
+
postedWithin: (p, days) => withinDays(p.postedOn, days),
|
|
114
|
+
max: MAX_DETAIL_FETCHES,
|
|
115
|
+
});
|
|
144
116
|
|
|
145
117
|
// 4. Hydrate descriptions via the per-posting detail endpoint. The detail
|
|
146
118
|
// also carries `hiringOrganization: { name, url }` next to
|
|
@@ -151,6 +123,7 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
151
123
|
const jobs = await Promise.all(hydrate.map(async (p, i) => {
|
|
152
124
|
const externalPath = p.externalPath || ''; // already begins with '/job/...'
|
|
153
125
|
let info = {};
|
|
126
|
+
let content; // set only when the detail could not be read (issue #85)
|
|
154
127
|
try {
|
|
155
128
|
// externalPath already carries the '/job/...' segment, so it is
|
|
156
129
|
// concatenated directly onto the CXS base. Inserting another
|
|
@@ -160,9 +133,12 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
160
133
|
const detail = await dResp.json();
|
|
161
134
|
info = detail.jobPostingInfo || {};
|
|
162
135
|
orgs[i] = detail.hiringOrganization || null;
|
|
136
|
+
} else {
|
|
137
|
+
content = missingContent(dResp);
|
|
163
138
|
}
|
|
164
|
-
} catch {
|
|
165
|
-
// detail failed, retries included:
|
|
139
|
+
} catch (err) {
|
|
140
|
+
// detail failed, retries included: list fields only, marked missing
|
|
141
|
+
content = missingContent(err);
|
|
166
142
|
}
|
|
167
143
|
|
|
168
144
|
return normalize({
|
|
@@ -177,6 +153,7 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
177
153
|
url: `https://${tenant}.${env}.myworkdayjobs.com/${site}${externalPath}`,
|
|
178
154
|
postedAt: parseWorkdayDate(info.startDate) || normalizePostedOn(p.postedOn),
|
|
179
155
|
salary: null, // normalizer extracts from description text
|
|
156
|
+
content,
|
|
180
157
|
metadata: {
|
|
181
158
|
workdayTenant: tenant,
|
|
182
159
|
workdayEnv: env,
|
|
@@ -191,7 +168,6 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
191
168
|
if (typeof ctx.report === 'function') {
|
|
192
169
|
const org = orgs.find(Boolean) || {};
|
|
193
170
|
ctx.report({
|
|
194
|
-
ats: 'workday',
|
|
195
171
|
listed: postings.length,
|
|
196
172
|
prefiltered: candidates.length,
|
|
197
173
|
hydrated: hydrate.length,
|
|
@@ -220,19 +196,26 @@ function parseWorkdayRemoteType(remoteType) {
|
|
|
220
196
|
|
|
221
197
|
/**
|
|
222
198
|
* Workday list `postedOn` is a relative string ("Posted Today",
|
|
223
|
-
* "Posted 5 Days Ago", "Posted 30+ Days Ago").
|
|
224
|
-
*
|
|
225
|
-
* the library re-filters authoritatively on the real postedAt after
|
|
226
|
-
* hydration, so a false-keep here is corrected downstream.
|
|
199
|
+
* "Posted 5 Days Ago", "Posted 30+ Days Ago"). The days it names, or null
|
|
200
|
+
* when it names none.
|
|
227
201
|
*/
|
|
228
|
-
function
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
if (/
|
|
232
|
-
if (/yesterday/.test(s)) return days >= 1;
|
|
202
|
+
function daysAgo(postedOn) {
|
|
203
|
+
const s = String(postedOn || '').toLowerCase();
|
|
204
|
+
if (/today/.test(s)) return 0;
|
|
205
|
+
if (/yesterday/.test(s)) return 1;
|
|
233
206
|
const m = s.match(/(\d+)\+?\s*days?\s*ago/);
|
|
234
|
-
|
|
235
|
-
|
|
207
|
+
return m ? parseInt(m[1], 10) : null;
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
/**
|
|
211
|
+
* Decide membership in the last N days WITHOUT a network call.
|
|
212
|
+
* Unparseable -> keep (true); the library re-filters authoritatively on
|
|
213
|
+
* the real postedAt after hydration, so a false-keep here is corrected
|
|
214
|
+
* downstream.
|
|
215
|
+
*/
|
|
216
|
+
function withinDays(postedOn, days) {
|
|
217
|
+
const n = daysAgo(postedOn);
|
|
218
|
+
return n === null || n <= days;
|
|
236
219
|
}
|
|
237
220
|
|
|
238
221
|
/**
|
|
@@ -243,16 +226,8 @@ function normalizePostedOn(v) {
|
|
|
243
226
|
if (!v) return null;
|
|
244
227
|
const direct = new Date(v);
|
|
245
228
|
if (Number.isFinite(direct.getTime())) return direct.toISOString();
|
|
246
|
-
const
|
|
247
|
-
|
|
248
|
-
if (/today/.test(s)) daysAgo = 0;
|
|
249
|
-
else if (/yesterday/.test(s)) daysAgo = 1;
|
|
250
|
-
else {
|
|
251
|
-
const m = s.match(/(\d+)\+?\s*days?\s*ago/);
|
|
252
|
-
if (m) daysAgo = parseInt(m[1], 10);
|
|
253
|
-
}
|
|
254
|
-
if (daysAgo === null) return null;
|
|
255
|
-
return new Date(Date.now() - daysAgo * 86400000).toISOString();
|
|
229
|
+
const n = daysAgo(v);
|
|
230
|
+
return n === null ? null : new Date(Date.now() - n * 86400000).toISOString();
|
|
256
231
|
}
|
|
257
232
|
|
|
258
233
|
/**
|
package/src/cli.js
CHANGED
|
@@ -11,6 +11,7 @@
|
|
|
11
11
|
|
|
12
12
|
import { realpathSync } from 'node:fs';
|
|
13
13
|
import { fileURLToPath } from 'node:url';
|
|
14
|
+
import { parseArgs } from 'node:util';
|
|
14
15
|
import { fetchJobs } from './index.js';
|
|
15
16
|
import { detectAtsDetailed, searchRegistry } from './registry.js';
|
|
16
17
|
|
|
@@ -19,30 +20,35 @@ const [,, command, ...args] = process.argv;
|
|
|
19
20
|
async function main() {
|
|
20
21
|
switch (command) {
|
|
21
22
|
case 'fetch': {
|
|
22
|
-
const
|
|
23
|
+
const string = { type: 'string' };
|
|
24
|
+
const { values: flags, positionals } = parseArgs({
|
|
25
|
+
args,
|
|
26
|
+
allowPositionals: true,
|
|
27
|
+
options: {
|
|
28
|
+
ats: string, 'title-filter': string, filter: string, 'posted-within-days': string,
|
|
29
|
+
'location-include': string, 'location-exclude': string, limit: string,
|
|
30
|
+
'workday-tenant': string, 'workday-env': string, 'workday-site': string,
|
|
31
|
+
json: { type: 'boolean' },
|
|
32
|
+
},
|
|
33
|
+
});
|
|
34
|
+
const company = positionals[0];
|
|
23
35
|
if (!company) { console.error('Usage: jd-intel fetch <company> [--ats <platform>] (omit --ats to auto-detect; run "jd-intel" for the platform list)'); process.exit(1); }
|
|
24
|
-
const
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
const
|
|
30
|
-
const
|
|
31
|
-
const
|
|
32
|
-
const
|
|
33
|
-
const locIncludeRaw = getArg('--location-include');
|
|
34
|
-
const locationIncludes = locIncludeRaw ? locIncludeRaw.split(',').map(s => s.trim()).filter(Boolean) : undefined;
|
|
35
|
-
const locExcludeRaw = getArg('--location-exclude');
|
|
36
|
-
const locationExcludes = locExcludeRaw ? locExcludeRaw.split(',').map(s => s.trim()).filter(Boolean) : undefined;
|
|
37
|
-
const limitRaw = getArg('--limit');
|
|
38
|
-
const limit = limitRaw !== undefined ? Number(limitRaw) : undefined;
|
|
36
|
+
const number = (v) => (v !== undefined ? Number(v) : undefined);
|
|
37
|
+
const list = (v) => (v ? v.split(',').map(s => s.trim()).filter(Boolean) : undefined);
|
|
38
|
+
let ats = flags.ats;
|
|
39
|
+
const titleFilter = flags['title-filter'];
|
|
40
|
+
const filter = flags.filter;
|
|
41
|
+
const postedWithinDays = number(flags['posted-within-days']);
|
|
42
|
+
const locationIncludes = list(flags['location-include']);
|
|
43
|
+
const locationExcludes = list(flags['location-exclude']);
|
|
44
|
+
const limit = number(flags.limit);
|
|
39
45
|
|
|
40
46
|
// Workday is keyed by a {tenant, env, site} triple, not a slug.
|
|
41
47
|
// Supplying it here makes a Workday board reachable without a
|
|
42
48
|
// registry entry; presence of the flags infers --ats workday.
|
|
43
|
-
const wdTenant =
|
|
44
|
-
const wdEnv =
|
|
45
|
-
const wdSite =
|
|
49
|
+
const wdTenant = flags['workday-tenant'];
|
|
50
|
+
const wdEnv = flags['workday-env'];
|
|
51
|
+
const wdSite = flags['workday-site'];
|
|
46
52
|
let config;
|
|
47
53
|
if (wdTenant || wdEnv || wdSite) {
|
|
48
54
|
if (!wdTenant || !wdEnv || !wdSite) {
|
|
@@ -91,8 +97,10 @@ async function main() {
|
|
|
91
97
|
const loc = job.location ? ` | ${job.location}` : '';
|
|
92
98
|
const dept = job.department ? ` [${job.department}]` : '';
|
|
93
99
|
console.log(` ${job.title}${dept}${loc}${salary}`);
|
|
94
|
-
console.log(` ${job.url}`);
|
|
95
|
-
if (job.
|
|
100
|
+
console.log(` ${job.url || '(no URL: posting not read)'}`);
|
|
101
|
+
if (job.content?.status === 'missing') {
|
|
102
|
+
console.log(` posting not read: ${job.content.reason}`);
|
|
103
|
+
} else if (job.description) {
|
|
96
104
|
const preview = job.description.substring(0, 120).replace(/\n/g, ' ');
|
|
97
105
|
console.log(` ${preview}...`);
|
|
98
106
|
}
|
|
@@ -103,7 +111,7 @@ async function main() {
|
|
|
103
111
|
console.log(` ... and ${jobs.length - 20} more. Use --json for full output.`);
|
|
104
112
|
}
|
|
105
113
|
|
|
106
|
-
if (
|
|
114
|
+
if (flags.json) {
|
|
107
115
|
console.log(JSON.stringify(jobs, null, 2));
|
|
108
116
|
}
|
|
109
117
|
break;
|
package/src/filters.js
CHANGED
|
@@ -88,11 +88,7 @@ export function pageJobs(jobs, { order = 'newest', offset = 0, limit = 100 } = {
|
|
|
88
88
|
|
|
89
89
|
const start = typeof offset === 'number' && offset > 0 ? offset : 0;
|
|
90
90
|
const end = typeof limit === 'number' ? start + limit : undefined;
|
|
91
|
-
|
|
92
|
-
result = result.slice(start, end);
|
|
93
|
-
}
|
|
94
|
-
|
|
95
|
-
return result;
|
|
91
|
+
return result.slice(start, end);
|
|
96
92
|
}
|
|
97
93
|
|
|
98
94
|
/**
|
|
@@ -174,3 +170,51 @@ export function makeLocationMatcher(needle) {
|
|
|
174
170
|
}
|
|
175
171
|
return (loc) => loc.includes(lower);
|
|
176
172
|
}
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* The list pre-filter and detail budget the two-step adapters (Workday,
|
|
176
|
+
* SmartRecruiters) share: narrow the cheap list rows with the filters a row
|
|
177
|
+
* can answer, then bound how many get a detail request.
|
|
178
|
+
*
|
|
179
|
+
* The library re-applies every filter after hydration, so a keep here is
|
|
180
|
+
* never final. `location(row)` returns the lowercased location, or null
|
|
181
|
+
* when the row cannot say where it is (it then stays a candidate through
|
|
182
|
+
* both location filters). `postedWithin(row, days)` decides recency.
|
|
183
|
+
*
|
|
184
|
+
* A description `filter` runs only after hydration, so that case keeps the
|
|
185
|
+
* full `max` budget instead of truncating to the page (which could hydrate
|
|
186
|
+
* rows that all fail the regex while better matches go unscanned). Without
|
|
187
|
+
* one, the budget is the page plus the offset before it. List order is kept.
|
|
188
|
+
*
|
|
189
|
+
* @returns {{ candidates: Array, hydrate: Array }}
|
|
190
|
+
*/
|
|
191
|
+
export function prefilterRows(rows, fc, { title, location, postedWithin, max }) {
|
|
192
|
+
let candidates = rows;
|
|
193
|
+
|
|
194
|
+
if (fc.titleFilter) {
|
|
195
|
+
const re = new RegExp(fc.titleFilter, 'i');
|
|
196
|
+
candidates = candidates.filter(p => re.test(title(p)));
|
|
197
|
+
}
|
|
198
|
+
if (Array.isArray(fc.locationIncludes) && fc.locationIncludes.length > 0) {
|
|
199
|
+
const inc = fc.locationIncludes.map(makeLocationMatcher);
|
|
200
|
+
candidates = candidates.filter(p => {
|
|
201
|
+
const loc = location(p);
|
|
202
|
+
return loc === null || inc.some(m => m(loc));
|
|
203
|
+
});
|
|
204
|
+
}
|
|
205
|
+
if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
|
|
206
|
+
const exc = fc.locationExcludes.map(makeLocationMatcher);
|
|
207
|
+
candidates = candidates.filter(p => {
|
|
208
|
+
const loc = location(p);
|
|
209
|
+
return loc === null || !exc.some(m => m(loc));
|
|
210
|
+
});
|
|
211
|
+
}
|
|
212
|
+
if (typeof fc.postedWithinDays === 'number') {
|
|
213
|
+
candidates = candidates.filter(p => postedWithin(p, fc.postedWithinDays));
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
|
|
217
|
+
const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
|
|
218
|
+
const cap = fc.filter ? max : Math.min(skip + limit, max);
|
|
219
|
+
return { candidates, hydrate: candidates.slice(0, cap) };
|
|
220
|
+
}
|
package/src/index.js
CHANGED
|
@@ -57,6 +57,7 @@ export async function fetchJobs(options = {}) {
|
|
|
57
57
|
* jobs: Array,
|
|
58
58
|
* total_matched: number,
|
|
59
59
|
* total_before_filters: number,
|
|
60
|
+
* content_missing: number,
|
|
60
61
|
* match: 'registry'|'probe'|'workday_override',
|
|
61
62
|
* company: { key: string, name: string }|null,
|
|
62
63
|
* boards: Array<object>,
|
|
@@ -65,6 +66,9 @@ export async function fetchJobs(options = {}) {
|
|
|
65
66
|
* jobs: the page. total_matched: matches before offset and limit.
|
|
66
67
|
* total_before_filters: rows every board listed before any filter (on
|
|
67
68
|
* Workday and SmartRecruiters the list count, not the rows they hydrated).
|
|
69
|
+
* content_missing: jobs whose posting was not read because the detail
|
|
70
|
+
* request failed (Workday, SmartRecruiters), counted before the filters:
|
|
71
|
+
* a description filter cannot match text that never arrived.
|
|
68
72
|
* match: how the company was resolved. company: the registry row's name
|
|
69
73
|
* and its key (normalized name); null unless match is 'registry'.
|
|
70
74
|
* boards: one entry per board that answered (see src/boards.js), with
|
|
@@ -194,6 +198,7 @@ export async function fetchJobsDetailed({
|
|
|
194
198
|
jobs: pageJobs(matched, { order, offset, limit }),
|
|
195
199
|
total_matched: matched.length,
|
|
196
200
|
total_before_filters: boards.reduce((n, b) => n + b.jobs_found, 0),
|
|
201
|
+
content_missing: rows.filter(j => j.content?.status === 'missing').length,
|
|
197
202
|
match,
|
|
198
203
|
company: match === 'registry' ? { key: normSlug(targets[0].name), name: targets[0].name } : null,
|
|
199
204
|
boards,
|
|
@@ -205,7 +210,13 @@ function registryTarget(hit, config) {
|
|
|
205
210
|
return { ats: hit.ats, slug: hit.entry.slug, name: hit.entry.name, config: config || hit.entry.config };
|
|
206
211
|
}
|
|
207
212
|
|
|
208
|
-
|
|
213
|
+
/**
|
|
214
|
+
* The AtsError for a lookup where no board answered and at least one check
|
|
215
|
+
* failed: rate_limited when any failure was a 429, else ats_unreachable,
|
|
216
|
+
* with a message naming each failed check. `failed` is the list
|
|
217
|
+
* fetchJobsDetailed and detectAtsDetailed return.
|
|
218
|
+
*/
|
|
219
|
+
export function discoveryFailure(company, failed) {
|
|
209
220
|
const limited = failed.some(f => f.code === ERROR_CODES.RATE_LIMITED);
|
|
210
221
|
const checks = failed.map(f => `${f.ats} (${f.message})`).join('; ');
|
|
211
222
|
return new AtsError(
|
package/src/normalizer.js
CHANGED
|
@@ -23,6 +23,9 @@ export function jobId(company, title, ats, location = '') {
|
|
|
23
23
|
* adapter to 'remote' | 'hybrid' | 'onsite', or null when the platform
|
|
24
24
|
* gives no signal. `raw.locations` lists every place the posting is open
|
|
25
25
|
* in; `location` stays the primary because it feeds the id (issue #68).
|
|
26
|
+
*
|
|
27
|
+
* `raw.content` is set only by a two-step adapter whose detail request
|
|
28
|
+
* failed (see missingContent). Every other job was read in full.
|
|
26
29
|
*/
|
|
27
30
|
export function normalize(raw, ats) {
|
|
28
31
|
const now = new Date().toISOString();
|
|
@@ -47,10 +50,21 @@ export function normalize(raw, ats) {
|
|
|
47
50
|
firstSeen: now,
|
|
48
51
|
lastSeen: now,
|
|
49
52
|
status: 'open',
|
|
53
|
+
content: raw.content || { status: 'complete', reason: null },
|
|
50
54
|
metadata: raw.metadata || {},
|
|
51
55
|
};
|
|
52
56
|
}
|
|
53
57
|
|
|
58
|
+
/**
|
|
59
|
+
* The `content` of a job whose detail request failed, so its description
|
|
60
|
+
* and pay were never read (issue #85). `failure` is the non-OK Response or
|
|
61
|
+
* the error atsFetch threw: an HTTP status gives "http_503", anything else
|
|
62
|
+
* "network_error".
|
|
63
|
+
*/
|
|
64
|
+
export function missingContent(failure) {
|
|
65
|
+
return { status: 'missing', reason: failure?.status ? `http_${failure.status}` : 'network_error' };
|
|
66
|
+
}
|
|
67
|
+
|
|
54
68
|
const CURRENCY_CODES = 'USD|EUR|GBP|CAD|AUD|NZD|CHF|SEK|NOK|DKK|PLN|CZK|HUF|INR|SGD|HKD|JPY|CNY|BRL|MXN|ZAR|AED|ILS';
|
|
55
69
|
const SYMBOL_CURRENCY = { $: 'USD', '€': 'EUR', '£': 'GBP' };
|
|
56
70
|
|
|
@@ -140,14 +154,14 @@ function detectPeriod(before, after, min) {
|
|
|
140
154
|
return min >= 10000 ? 'year' : null;
|
|
141
155
|
}
|
|
142
156
|
|
|
143
|
-
function periodWord(text) {
|
|
157
|
+
export function periodWord(text) {
|
|
144
158
|
if (HOUR_RE.test(text)) return 'hour';
|
|
145
159
|
if (MONTH_RE.test(text)) return 'month';
|
|
146
160
|
if (YEAR_RE.test(text)) return 'year';
|
|
147
161
|
return null;
|
|
148
162
|
}
|
|
149
163
|
|
|
150
|
-
const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
|
|
164
|
+
export const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
|
|
151
165
|
|
|
152
166
|
/**
|
|
153
167
|
* The platform's own value wins. Without one, a keyword in the location
|
package/src/registry.js
CHANGED
|
@@ -2,6 +2,7 @@ import { readFile } from 'node:fs/promises';
|
|
|
2
2
|
import { join, dirname } from 'node:path';
|
|
3
3
|
import { fileURLToPath } from 'node:url';
|
|
4
4
|
import { AtsError } from './errors.js';
|
|
5
|
+
import { ADAPTERS, ATS_NAMES } from './adapters/index.js';
|
|
5
6
|
|
|
6
7
|
const __dirname = dirname(fileURLToPath(import.meta.url));
|
|
7
8
|
const REGISTRY_DIR = join(__dirname, '..', 'registry');
|
|
@@ -9,7 +10,7 @@ const REGISTRY_DIR = join(__dirname, '..', 'registry');
|
|
|
9
10
|
// The one order the registry is ever walked in. Lookups, detectAts and the
|
|
10
11
|
// loaded object all follow it, so which file answers for a slug does not
|
|
11
12
|
// depend on which file's load finished first (issue #87).
|
|
12
|
-
const PLATFORMS =
|
|
13
|
+
const PLATFORMS = ATS_NAMES;
|
|
13
14
|
|
|
14
15
|
// Network-first registry. A hosted copy lets installed bundles AND npx users
|
|
15
16
|
// pick up newly-added companies without reinstalling; the on-disk copy that
|
|
@@ -94,7 +95,10 @@ export function getRegistrySource() {
|
|
|
94
95
|
}
|
|
95
96
|
|
|
96
97
|
/**
|
|
97
|
-
* Search registry for companies matching a query
|
|
98
|
+
* Search registry for companies matching a query, best match first: an
|
|
99
|
+
* exact name or slug, then a name that starts with the query, then a name
|
|
100
|
+
* that contains it, then a sector-only match. Ties keep platform order, so
|
|
101
|
+
* a caller that cuts the list drops the weakest matches (issue #62).
|
|
98
102
|
*/
|
|
99
103
|
export async function searchRegistry(query) {
|
|
100
104
|
const all = await loadRegistry();
|
|
@@ -105,13 +109,16 @@ export async function searchRegistry(query) {
|
|
|
105
109
|
for (const company of companies) {
|
|
106
110
|
const name = (company.name || company.slug || '').toLowerCase();
|
|
107
111
|
const sector = (company.sector || '').toLowerCase();
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
112
|
+
const rank = name === lower || String(company.slug).toLowerCase() === lower ? 0
|
|
113
|
+
: name.startsWith(lower) ? 1
|
|
114
|
+
: name.includes(lower) ? 2
|
|
115
|
+
: sector.includes(lower) ? 3
|
|
116
|
+
: -1;
|
|
117
|
+
if (rank >= 0) results.push({ rank, row: { ...company, ats } });
|
|
111
118
|
}
|
|
112
119
|
}
|
|
113
120
|
|
|
114
|
-
return results;
|
|
121
|
+
return results.sort((a, b) => a.rank - b.rank).map(r => r.row);
|
|
115
122
|
}
|
|
116
123
|
|
|
117
124
|
// Slug match is case/punctuation-insensitive: registry slugs are stored
|
|
@@ -120,18 +127,7 @@ export async function searchRegistry(query) {
|
|
|
120
127
|
// normalized forms keeps registry-first routing working for those.
|
|
121
128
|
export const normSlug = (s) => String(s).toLowerCase().replace(/[^a-z0-9]/g, '');
|
|
122
129
|
|
|
123
|
-
|
|
124
|
-
// order the object's keys are in.
|
|
125
|
-
function platformEntries(all) {
|
|
126
|
-
return PLATFORMS.map(ats => [ats, all[ats] || []]);
|
|
127
|
-
}
|
|
128
|
-
|
|
129
|
-
function platformIndex(ats) {
|
|
130
|
-
const i = PLATFORMS.indexOf(ats);
|
|
131
|
-
return i === -1 ? PLATFORMS.length : i;
|
|
132
|
-
}
|
|
133
|
-
|
|
134
|
-
const byPlatform = (a, b) => platformIndex(a.ats) - platformIndex(b.ats);
|
|
130
|
+
const byPlatform = (a, b) => PLATFORMS.indexOf(a.ats) - PLATFORMS.indexOf(b.ats);
|
|
135
131
|
|
|
136
132
|
/**
|
|
137
133
|
* Look up which ATS a slug belongs to in the registry.
|
|
@@ -154,7 +150,7 @@ export async function findAtsBySlug(slug) {
|
|
|
154
150
|
export async function findEntryBySlug(slug) {
|
|
155
151
|
const all = await loadRegistry();
|
|
156
152
|
const key = normSlug(slug);
|
|
157
|
-
for (const [ats, companies] of
|
|
153
|
+
for (const [ats, companies] of Object.entries(all)) {
|
|
158
154
|
const entry = companies.find(c => normSlug(c.slug) === key);
|
|
159
155
|
if (entry) return { ats, entry };
|
|
160
156
|
}
|
|
@@ -179,14 +175,13 @@ export async function findEntryBySlug(slug) {
|
|
|
179
175
|
* }>}
|
|
180
176
|
*/
|
|
181
177
|
export async function detectAtsDetailed(companyName) {
|
|
182
|
-
const { ADAPTERS } = await import('./adapters/index.js');
|
|
183
178
|
const slug = normSlug(companyName);
|
|
184
179
|
const all = await loadRegistry();
|
|
185
180
|
|
|
186
181
|
const boards = [];
|
|
187
182
|
const failed = [];
|
|
188
183
|
const known = new Set();
|
|
189
|
-
for (const [ats, companies] of
|
|
184
|
+
for (const [ats, companies] of Object.entries(all)) {
|
|
190
185
|
const entry = companies.find(c => normSlug(c.slug) === slug);
|
|
191
186
|
if (entry) {
|
|
192
187
|
boards.push({ ats, slug: entry.slug, source: 'registry' });
|