jd-intel 0.8.3 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +35 -4
- package/package.json +1 -1
- package/src/adapters/ashby.js +69 -86
- package/src/adapters/greenhouse.js +46 -13
- package/src/adapters/lever.js +87 -45
- package/src/adapters/recruitee.js +87 -23
- package/src/adapters/smartrecruiters.js +146 -34
- package/src/adapters/teamtailor.js +57 -39
- package/src/adapters/workday.js +107 -31
- package/src/boards.js +80 -0
- package/src/cli.js +46 -15
- package/src/errors.js +14 -0
- package/src/filters.js +122 -29
- package/src/http.js +184 -0
- package/src/index.js +186 -59
- package/src/normalizer.js +200 -44
- package/src/registry.js +79 -25
package/src/adapters/workday.js
CHANGED
|
@@ -1,9 +1,17 @@
|
|
|
1
|
-
import { normalize
|
|
1
|
+
import { normalize } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { makeLocationMatcher } from '../filters.js';
|
|
4
|
+
import { atsFetch } from '../http.js';
|
|
5
|
+
import { orgHost } from '../boards.js';
|
|
3
6
|
|
|
4
7
|
const MAX_DETAIL_FETCHES = 100;
|
|
5
8
|
const LIST_PAGE_SIZE = 20;
|
|
6
|
-
|
|
9
|
+
// Upper bound on list pages per call: 100 pages of 20 = at most 2000
|
|
10
|
+
// postings scanned. Paging usually stops sooner, on a short page or when
|
|
11
|
+
// offset reaches the first page's total (see the loop below).
|
|
12
|
+
const LIST_PAGE_HARD_CAP = 100;
|
|
13
|
+
// A multi-location posting's list row reads "2 Locations", "14 Locations".
|
|
14
|
+
const MULTI_LOCATION = /^\s*\d+\s+locations?\s*$/;
|
|
7
15
|
|
|
8
16
|
/**
|
|
9
17
|
* Fetch jobs from a Workday tenant via the public "CXS" JSON API.
|
|
@@ -25,7 +33,9 @@ const LIST_PAGE_HARD_CAP = 100; // <= 2000 list items scanned per request
|
|
|
25
33
|
* detail set.
|
|
26
34
|
*
|
|
27
35
|
* @param {string} slug - normalized company slug (registry routing key)
|
|
28
|
-
* @param {object} [ctx] - { config:{tenant,env,site}, companyName, filterContext }
|
|
36
|
+
* @param {object} [ctx] - { config:{tenant,env,site}, companyName, filterContext, report };
|
|
37
|
+
* report is called once, after hydration, with
|
|
38
|
+
* { ats, listed, prefiltered, hydrated, capped, org_name, org_url } when given
|
|
29
39
|
* @returns {Promise<Array>} Normalized job objects
|
|
30
40
|
*/
|
|
31
41
|
export async function fetchWorkday(slug, ctx = {}) {
|
|
@@ -40,27 +50,46 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
40
50
|
const postings = [];
|
|
41
51
|
let offset = 0;
|
|
42
52
|
let pages = 0;
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
53
|
+
let firstTotal = 0;
|
|
54
|
+
let listCapped = false;
|
|
55
|
+
while (true) {
|
|
56
|
+
if (pages >= LIST_PAGE_HARD_CAP) {
|
|
57
|
+
listCapped = true;
|
|
58
|
+
break;
|
|
59
|
+
}
|
|
60
|
+
let resp;
|
|
61
|
+
try {
|
|
62
|
+
resp = await atsFetch(`${base}/jobs`, {
|
|
63
|
+
method: 'POST',
|
|
64
|
+
headers: { 'Content-Type': 'application/json' },
|
|
65
|
+
body: JSON.stringify({ appliedFacets: {}, limit: LIST_PAGE_SIZE, offset, searchText: '' }),
|
|
66
|
+
});
|
|
67
|
+
} catch (err) {
|
|
68
|
+
// A 429 or 5xx that outlasted the retries, or a network failure.
|
|
69
|
+
// After the first page, keep the postings already read.
|
|
70
|
+
if (offset === 0) throw err;
|
|
71
|
+
break;
|
|
72
|
+
}
|
|
49
73
|
|
|
50
74
|
if (!resp.ok) {
|
|
51
|
-
if (resp.status === 404) return []; // wrong site / no such board
|
|
52
75
|
if (offset === 0) {
|
|
76
|
+
if (resp.status === 404) return []; // wrong site / no such board
|
|
53
77
|
throw atsErrorFromStatus(resp.status, `Workday API error for ${slug} (${tenant}/${env}/${site}): ${resp.status}`);
|
|
54
78
|
}
|
|
55
|
-
break; // mid-paging failure: keep what we have
|
|
79
|
+
break; // mid-paging failure (any status): keep what we have
|
|
56
80
|
}
|
|
57
81
|
|
|
58
82
|
const data = await resp.json();
|
|
59
83
|
const page = data.jobPostings || [];
|
|
84
|
+
// Some tenants report the real `total` only at offset 0 and send
|
|
85
|
+
// `total: 0` on every later page, so only the first page's figure
|
|
86
|
+
// is trusted. A short page is the other stop signal.
|
|
87
|
+
if (pages === 0) firstTotal = data.total || 0;
|
|
60
88
|
postings.push(...page);
|
|
61
89
|
pages += 1;
|
|
62
90
|
offset += LIST_PAGE_SIZE;
|
|
63
|
-
if (page.length
|
|
91
|
+
if (page.length < LIST_PAGE_SIZE) break;
|
|
92
|
+
if (firstTotal > 0 && offset >= firstTotal) break;
|
|
64
93
|
}
|
|
65
94
|
|
|
66
95
|
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost.
|
|
@@ -72,18 +101,23 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
72
101
|
const re = new RegExp(fc.titleFilter, 'i');
|
|
73
102
|
candidates = candidates.filter(p => re.test(p.title || ''));
|
|
74
103
|
}
|
|
104
|
+
// Location rows go through the applyFilters matcher, so the pre-filter
|
|
105
|
+
// keeps exactly the rows the pass after hydration would (issue #61).
|
|
106
|
+
// "2 Locations" says nothing about where: the row stays a candidate
|
|
107
|
+
// through both filters and that later pass decides on the detail's
|
|
108
|
+
// location list.
|
|
75
109
|
if (Array.isArray(fc.locationIncludes) && fc.locationIncludes.length > 0) {
|
|
76
|
-
const inc = fc.locationIncludes.map(
|
|
110
|
+
const inc = fc.locationIncludes.map(makeLocationMatcher);
|
|
77
111
|
candidates = candidates.filter(p => {
|
|
78
112
|
const loc = (p.locationsText || '').toLowerCase();
|
|
79
|
-
return inc.some(
|
|
113
|
+
return MULTI_LOCATION.test(loc) || inc.some(m => m(loc));
|
|
80
114
|
});
|
|
81
115
|
}
|
|
82
116
|
if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
|
|
83
|
-
const exc = fc.locationExcludes.map(
|
|
117
|
+
const exc = fc.locationExcludes.map(makeLocationMatcher);
|
|
84
118
|
candidates = candidates.filter(p => {
|
|
85
119
|
const loc = (p.locationsText || '').toLowerCase();
|
|
86
|
-
return !exc.some(
|
|
120
|
+
return MULTI_LOCATION.test(loc) || !exc.some(m => m(loc));
|
|
87
121
|
});
|
|
88
122
|
}
|
|
89
123
|
if (typeof fc.postedWithinDays === 'number') {
|
|
@@ -91,33 +125,44 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
91
125
|
}
|
|
92
126
|
|
|
93
127
|
// 3. Bound the detail-fetch set.
|
|
94
|
-
// NOTE: huge-tenant coverage is intentionally capped for v1
|
|
95
|
-
//
|
|
96
|
-
//
|
|
97
|
-
//
|
|
98
|
-
//
|
|
99
|
-
//
|
|
100
|
-
//
|
|
101
|
-
//
|
|
128
|
+
// NOTE: huge-tenant coverage is intentionally capped for v1. Two
|
|
129
|
+
// caps apply: the list scan above stops at LIST_PAGE_HARD_CAP pages
|
|
130
|
+
// (2000 postings, enough for Salesforce's ~1398), and the detail set
|
|
131
|
+
// is cut to MAX_DETAIL_FETCHES here. A description `filter` is
|
|
132
|
+
// applied by the library AFTER this returns, so for that case we
|
|
133
|
+
// keep the full backstop instead of truncating tightly to `limit`
|
|
134
|
+
// (which could hydrate jobs that all fail the regex while better
|
|
135
|
+
// matches go unscanned). The library pages with `offset` after this
|
|
136
|
+
// returns, so the budget covers the page plus what precedes it.
|
|
137
|
+
// Proper fix (smart pagination / rate-limited concurrency / surfaced
|
|
138
|
+
// truncation) is tracked in #26, to be designed alongside
|
|
139
|
+
// retry/rate-limit work (#7).
|
|
102
140
|
const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
|
|
103
|
-
const
|
|
104
|
-
|
|
141
|
+
const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
|
|
142
|
+
const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
|
|
143
|
+
const hydrate = candidates.slice(0, cap);
|
|
105
144
|
|
|
106
|
-
// 4. Hydrate descriptions via the per-posting detail endpoint.
|
|
107
|
-
|
|
145
|
+
// 4. Hydrate descriptions via the per-posting detail endpoint. The detail
|
|
146
|
+
// also carries `hiringOrganization: { name, url }` next to
|
|
147
|
+
// jobPostingInfo; the list does not. Kept per posting in list order so
|
|
148
|
+
// the one reported is the first hydrated posting's, not whichever
|
|
149
|
+
// detail answered first (a tenant can post under several entities).
|
|
150
|
+
const orgs = [];
|
|
151
|
+
const jobs = await Promise.all(hydrate.map(async (p, i) => {
|
|
108
152
|
const externalPath = p.externalPath || ''; // already begins with '/job/...'
|
|
109
153
|
let info = {};
|
|
110
154
|
try {
|
|
111
155
|
// externalPath already carries the '/job/...' segment, so it is
|
|
112
156
|
// concatenated directly onto the CXS base. Inserting another
|
|
113
157
|
// '/job' here yields '/job/job/...' which Workday rejects (422).
|
|
114
|
-
const dResp = await
|
|
158
|
+
const dResp = await atsFetch(`${base}${externalPath}`);
|
|
115
159
|
if (dResp.ok) {
|
|
116
160
|
const detail = await dResp.json();
|
|
117
161
|
info = detail.jobPostingInfo || {};
|
|
162
|
+
orgs[i] = detail.hiringOrganization || null;
|
|
118
163
|
}
|
|
119
164
|
} catch {
|
|
120
|
-
// detail failed: fall back to list fields, empty description
|
|
165
|
+
// detail failed, retries included: fall back to list fields, empty description
|
|
121
166
|
}
|
|
122
167
|
|
|
123
168
|
return normalize({
|
|
@@ -126,7 +171,9 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
126
171
|
title: p.title || info.title || '',
|
|
127
172
|
department: '',
|
|
128
173
|
location: info.location || p.locationsText || '',
|
|
129
|
-
|
|
174
|
+
locations: info.additionalLocations || [],
|
|
175
|
+
workplace: parseWorkdayRemoteType(info.remoteType),
|
|
176
|
+
description: info.jobDescription || '',
|
|
130
177
|
url: `https://${tenant}.${env}.myworkdayjobs.com/${site}${externalPath}`,
|
|
131
178
|
postedAt: parseWorkdayDate(info.startDate) || normalizePostedOn(p.postedOn),
|
|
132
179
|
salary: null, // normalizer extracts from description text
|
|
@@ -139,9 +186,38 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
139
186
|
}, 'workday');
|
|
140
187
|
}));
|
|
141
188
|
|
|
189
|
+
// Nothing hydrated (a filter miss, an empty site) means no detail was
|
|
190
|
+
// read, so the org is unknown rather than absent: null, null.
|
|
191
|
+
if (typeof ctx.report === 'function') {
|
|
192
|
+
const org = orgs.find(Boolean) || {};
|
|
193
|
+
ctx.report({
|
|
194
|
+
ats: 'workday',
|
|
195
|
+
listed: postings.length,
|
|
196
|
+
prefiltered: candidates.length,
|
|
197
|
+
hydrated: hydrate.length,
|
|
198
|
+
capped: listCapped || hydrate.length < candidates.length,
|
|
199
|
+
org_name: org.name || null,
|
|
200
|
+
org_url: orgHost(org.url),
|
|
201
|
+
});
|
|
202
|
+
}
|
|
203
|
+
|
|
142
204
|
return jobs;
|
|
143
205
|
}
|
|
144
206
|
|
|
207
|
+
/**
|
|
208
|
+
* Detail `remoteType` is free text set per tenant: "Remote", "Hybrid",
|
|
209
|
+
* "Office - Flexible", "On-site". A flexible office arrangement counts as
|
|
210
|
+
* hybrid, so that check runs before the office one. Some tenants send no
|
|
211
|
+
* value at all; the location string decides then.
|
|
212
|
+
*/
|
|
213
|
+
function parseWorkdayRemoteType(remoteType) {
|
|
214
|
+
const s = String(remoteType || '').toLowerCase();
|
|
215
|
+
if (/remote/.test(s)) return 'remote';
|
|
216
|
+
if (/hybrid|flexible/.test(s)) return 'hybrid';
|
|
217
|
+
if (/office|on-?site/.test(s)) return 'onsite';
|
|
218
|
+
return null;
|
|
219
|
+
}
|
|
220
|
+
|
|
145
221
|
/**
|
|
146
222
|
* Workday list `postedOn` is a relative string ("Posted Today",
|
|
147
223
|
* "Posted 5 Days Ago", "Posted 30+ Days Ago"). Decide membership in
|
package/src/boards.js
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The boards[] entry of a fetchJobsDetailed result (issues #58, #60, #87).
|
|
3
|
+
*
|
|
4
|
+
* A board is one (ats, slug) the library fetched, with what came back. The
|
|
5
|
+
* fields fall in three groups: what the registry or the caller said about
|
|
6
|
+
* it (`name`, `site`), what the fetch found (`jobs_found`, `matched`,
|
|
7
|
+
* `scan`), and what the board says about itself (`org_name`, `org_url`).
|
|
8
|
+
* Only the adapters can read the third group, and they hand it over through
|
|
9
|
+
* ctx.report. Both fields are null where the platform exposes nothing, and
|
|
10
|
+
* they are never filled from the slug or the registry name: a slug is an
|
|
11
|
+
* address, not a confirmed identity.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
const BOARD_URLS = {
|
|
15
|
+
greenhouse: (slug) => `https://boards.greenhouse.io/${slug}`,
|
|
16
|
+
lever: (slug) => `https://jobs.lever.co/${slug}`,
|
|
17
|
+
ashby: (slug) => `https://jobs.ashbyhq.com/${slug}`,
|
|
18
|
+
smartrecruiters: (slug) => `https://careers.smartrecruiters.com/${slug}`,
|
|
19
|
+
teamtailor: (slug) => `https://${slug}.teamtailor.com`,
|
|
20
|
+
recruitee: (slug) => `https://${slug}.recruitee.com`,
|
|
21
|
+
workday: (slug, config) => (config ? `https://${config.tenant}.${config.env}.myworkdayjobs.com/${config.site}` : null),
|
|
22
|
+
};
|
|
23
|
+
|
|
24
|
+
// Domains the platforms own. A link there (boards.greenhouse.io,
|
|
25
|
+
// jobs.lever.co, testco.recruitee.com, cisco.wd5.myworkdayjobs.com) says
|
|
26
|
+
// which ATS hosts the board, nothing about whose board it is.
|
|
27
|
+
const ATS_DOMAINS = ['greenhouse.io', 'lever.co', 'ashbyhq.com', 'smartrecruiters.com', 'teamtailor.com', 'recruitee.com', 'myworkdayjobs.com'];
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* The page a person opens to see the board, or null when the ATS is unknown
|
|
31
|
+
* or, for Workday, no {tenant, env, site} is at hand.
|
|
32
|
+
*/
|
|
33
|
+
export function boardUrl(ats, slug, config) {
|
|
34
|
+
const build = BOARD_URLS[ats];
|
|
35
|
+
return build ? build(slug, config) : null;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* The bare hostname a board's link points at ("jobs.example.com"), for
|
|
40
|
+
* org_url. Null when the link is missing or malformed, and null when the
|
|
41
|
+
* host belongs to an ATS, since that carries no signal about the company.
|
|
42
|
+
*/
|
|
43
|
+
export function orgHost(link) {
|
|
44
|
+
let host;
|
|
45
|
+
try {
|
|
46
|
+
host = new URL(link).hostname.toLowerCase();
|
|
47
|
+
} catch {
|
|
48
|
+
return null;
|
|
49
|
+
}
|
|
50
|
+
if (!host || ATS_DOMAINS.some(d => host === d || host.endsWith(`.${d}`))) return null;
|
|
51
|
+
return host;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* @param {object} board
|
|
56
|
+
* @param {string} board.ats
|
|
57
|
+
* @param {string} board.slug - The slug the adapter was called with (canonical casing on a registry hit)
|
|
58
|
+
* @param {string|null} board.name - Registry row name; null for a probe or an override
|
|
59
|
+
* @param {object} [board.config] - Workday {tenant, env, site}, when one was used
|
|
60
|
+
* @param {string|null} board.org_name - The organization name the ATS response states, else null
|
|
61
|
+
* @param {string|null} board.org_url - The careers or company host the board links to (see orgHost), else null
|
|
62
|
+
* @param {number} board.jobs_found - Rows the board listed before any filter: the list count an adapter reported through ctx.report when it filters before hydrating, else the rows it returned
|
|
63
|
+
* @param {number} board.matched - Rows left after filters, before offset and limit
|
|
64
|
+
* @param {object|null} board.scan - The { listed, prefiltered, hydrated, capped } counts the adapter reported through ctx.report, else null
|
|
65
|
+
*/
|
|
66
|
+
export function describeBoard({ ats, slug, name = null, config, org_name = null, org_url = null, jobs_found, matched = 0, scan = null }) {
|
|
67
|
+
return {
|
|
68
|
+
ats,
|
|
69
|
+
slug,
|
|
70
|
+
name,
|
|
71
|
+
site: ats === 'workday' && config ? config.site : null,
|
|
72
|
+
board_url: boardUrl(ats, slug, config),
|
|
73
|
+
org_name,
|
|
74
|
+
org_url,
|
|
75
|
+
jobs_found,
|
|
76
|
+
matched,
|
|
77
|
+
selected: true,
|
|
78
|
+
scan,
|
|
79
|
+
};
|
|
80
|
+
}
|
package/src/cli.js
CHANGED
|
@@ -9,8 +9,10 @@
|
|
|
9
9
|
* jd-intel registry search <query>
|
|
10
10
|
*/
|
|
11
11
|
|
|
12
|
+
import { realpathSync } from 'node:fs';
|
|
13
|
+
import { fileURLToPath } from 'node:url';
|
|
12
14
|
import { fetchJobs } from './index.js';
|
|
13
|
-
import {
|
|
15
|
+
import { detectAtsDetailed, searchRegistry } from './registry.js';
|
|
14
16
|
|
|
15
17
|
const [,, command, ...args] = process.argv;
|
|
16
18
|
|
|
@@ -85,7 +87,7 @@ async function main() {
|
|
|
85
87
|
console.log(`Found ${jobs.length} jobs\n`);
|
|
86
88
|
|
|
87
89
|
for (const job of jobs.slice(0, 20)) {
|
|
88
|
-
const salary = job.salary ? ` |
|
|
90
|
+
const salary = job.salary ? ` | ${formatSalary(job.salary)}` : '';
|
|
89
91
|
const loc = job.location ? ` | ${job.location}` : '';
|
|
90
92
|
const dept = job.department ? ` [${job.department}]` : '';
|
|
91
93
|
console.log(` ${job.title}${dept}${loc}${salary}`);
|
|
@@ -111,13 +113,17 @@ async function main() {
|
|
|
111
113
|
const company = args[0];
|
|
112
114
|
if (!company) { console.error('Usage: jd-intel detect <company>'); process.exit(1); }
|
|
113
115
|
console.log(`Detecting ATS for ${company}...`);
|
|
114
|
-
const
|
|
115
|
-
|
|
116
|
-
console.log('
|
|
117
|
-
}
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
116
|
+
const { boards, failed } = await detectAtsDetailed(company);
|
|
117
|
+
for (const b of boards) {
|
|
118
|
+
console.log(` Found: ${b.ats} (slug: ${b.slug}, ${b.source === 'registry' ? 'in the registry' : 'live probe'})`);
|
|
119
|
+
}
|
|
120
|
+
for (const f of failed) {
|
|
121
|
+
console.log(` Could not check ${f.ats}: ${f.message}`);
|
|
122
|
+
}
|
|
123
|
+
if (boards.length === 0) {
|
|
124
|
+
console.log(failed.length > 0
|
|
125
|
+
? 'No ATS board confirmed. At least one check failed, so this is not a definite miss. Retry in a moment.'
|
|
126
|
+
: 'No ATS board found for this company.');
|
|
121
127
|
}
|
|
122
128
|
break;
|
|
123
129
|
}
|
|
@@ -162,8 +168,8 @@ Fetch options:
|
|
|
162
168
|
--title-filter pattern Regex matched against TITLE only (role identity)
|
|
163
169
|
--filter pattern Regex matched across title, department, description (topic/scope)
|
|
164
170
|
--posted-within-days N Only jobs posted in the last N days
|
|
165
|
-
--location-include "A,B,C" Keep jobs
|
|
166
|
-
--location-exclude "A,B,C" Drop jobs
|
|
171
|
+
--location-include "A,B,C" Keep jobs where any listed location contains one of these
|
|
172
|
+
--location-exclude "A,B,C" Drop jobs only when every listed location contains one of these
|
|
167
173
|
--limit N Cap results (default 100)
|
|
168
174
|
--json Output full JSON
|
|
169
175
|
|
|
@@ -184,7 +190,32 @@ Examples:
|
|
|
184
190
|
}
|
|
185
191
|
}
|
|
186
192
|
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
193
|
+
export function formatSalary({ min, max, currency, period }) {
|
|
194
|
+
const hasMin = min != null;
|
|
195
|
+
const hasMax = max != null;
|
|
196
|
+
let range;
|
|
197
|
+
if (hasMin && hasMax) range = `${min.toLocaleString()}-${max.toLocaleString()}`;
|
|
198
|
+
else if (hasMin) range = `from ${min.toLocaleString()}`;
|
|
199
|
+
else range = `up to ${max.toLocaleString()}`;
|
|
200
|
+
const unit = period === 'hour' ? '/hr' : period === 'month' ? '/mo' : '';
|
|
201
|
+
return `${range} ${currency}${unit}`;
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
// Boot only when this file is the script Node was started with, so a test
|
|
205
|
+
// can import formatSalary without running a command. argv[1] is resolved
|
|
206
|
+
// through realpath because npm installs the bin as a symlink into .bin/,
|
|
207
|
+
// while import.meta.url already points at the real file.
|
|
208
|
+
function isEntrypoint() {
|
|
209
|
+
try {
|
|
210
|
+
return realpathSync(process.argv[1]) === fileURLToPath(import.meta.url);
|
|
211
|
+
} catch {
|
|
212
|
+
return false;
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
if (isEntrypoint()) {
|
|
217
|
+
main().catch(err => {
|
|
218
|
+
console.error('Error:', err.message);
|
|
219
|
+
process.exit(1);
|
|
220
|
+
});
|
|
221
|
+
}
|
package/src/errors.js
CHANGED
|
@@ -34,6 +34,20 @@ export class AtsError extends Error {
|
|
|
34
34
|
}
|
|
35
35
|
}
|
|
36
36
|
|
|
37
|
+
/**
|
|
38
|
+
* Thrown when a call cannot proceed because of its arguments: a missing
|
|
39
|
+
* company, an unknown ATS name, a regex that does not compile. Carries
|
|
40
|
+
* `code: 'invalid_args'` so callers tell a bad request from a failed fetch
|
|
41
|
+
* (AtsError) without reading the message.
|
|
42
|
+
*/
|
|
43
|
+
export class ArgumentError extends Error {
|
|
44
|
+
constructor(message) {
|
|
45
|
+
super(message);
|
|
46
|
+
this.name = 'ArgumentError';
|
|
47
|
+
this.code = ERROR_CODES.INVALID_ARGS;
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
|
|
37
51
|
/**
|
|
38
52
|
* Helper for adapters: build an AtsError from an HTTP status (429 => rate
|
|
39
53
|
* limited, anything else => unreachable) with the given message.
|
package/src/filters.js
CHANGED
|
@@ -1,33 +1,56 @@
|
|
|
1
|
+
import { ArgumentError } from './errors.js';
|
|
2
|
+
|
|
1
3
|
/**
|
|
2
4
|
* Apply filters to a list of normalized jobs.
|
|
3
5
|
*
|
|
4
6
|
* Facts go here (deterministic field matches). Interpretations stay with the
|
|
5
7
|
* caller — this module does substring matching on structured fields, nothing
|
|
6
8
|
* semantic.
|
|
9
|
+
*
|
|
10
|
+
* Returns the page as an array. applyFiltersDetailed returns the same page
|
|
11
|
+
* plus total_matched, the match count before offset and limit.
|
|
7
12
|
*/
|
|
8
13
|
export function applyFilters(jobs, options = {}) {
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
14
|
+
return applyFiltersDetailed(jobs, options).jobs;
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Filter, sort, then page.
|
|
19
|
+
*
|
|
20
|
+
* Order is applied after the filters and before offset and limit, so a cut
|
|
21
|
+
* drops the oldest matches first. 'newest' sorts by postedAt descending with
|
|
22
|
+
* undated jobs last and ties broken by id, which keeps pages deterministic.
|
|
23
|
+
* 'board' keeps the order the adapter returned.
|
|
24
|
+
*
|
|
25
|
+
* @returns {{ jobs: Array, total_matched: number }}
|
|
26
|
+
*/
|
|
27
|
+
export function applyFiltersDetailed(jobs, options = {}) {
|
|
28
|
+
const { order = 'newest', offset = 0, limit = 100 } = options;
|
|
29
|
+
const matched = filterJobs(jobs, options);
|
|
30
|
+
return { jobs: pageJobs(matched, { order, offset, limit }), total_matched: matched.length };
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* The filter step on its own: every job that passes titleFilter, filter,
|
|
35
|
+
* postedWithinDays and the location filters, in the order given. No sort
|
|
36
|
+
* and no paging, so fetchJobsDetailed can count the matches per board
|
|
37
|
+
* before the page cut removes them.
|
|
38
|
+
*/
|
|
39
|
+
export function filterJobs(jobs, options = {}) {
|
|
40
|
+
const { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes } = options;
|
|
41
|
+
const { title, topic } = compileFilterPatterns({ titleFilter, filter });
|
|
17
42
|
|
|
18
43
|
let result = jobs;
|
|
19
44
|
|
|
20
|
-
if (
|
|
21
|
-
|
|
22
|
-
result = result.filter(j => pattern.test(j.title || ''));
|
|
45
|
+
if (title) {
|
|
46
|
+
result = result.filter(j => title.test(j.title || ''));
|
|
23
47
|
}
|
|
24
48
|
|
|
25
|
-
if (
|
|
26
|
-
const pattern = new RegExp(filter, 'i');
|
|
49
|
+
if (topic) {
|
|
27
50
|
result = result.filter(j =>
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
51
|
+
topic.test(j.title || '') ||
|
|
52
|
+
topic.test(j.department || '') ||
|
|
53
|
+
topic.test(j.description || '')
|
|
31
54
|
);
|
|
32
55
|
}
|
|
33
56
|
|
|
@@ -42,36 +65,106 @@ export function applyFilters(jobs, options = {}) {
|
|
|
42
65
|
|
|
43
66
|
if (Array.isArray(locationIncludes) && locationIncludes.length > 0) {
|
|
44
67
|
const matchers = locationIncludes.map(makeLocationMatcher);
|
|
45
|
-
result = result.filter(j =>
|
|
46
|
-
const loc = (j.location || '').toLowerCase();
|
|
47
|
-
return matchers.some(m => m(loc));
|
|
48
|
-
});
|
|
68
|
+
result = result.filter(j => jobLocations(j).some(loc => matchers.some(m => m(loc))));
|
|
49
69
|
}
|
|
50
70
|
|
|
51
71
|
if (Array.isArray(locationExcludes) && locationExcludes.length > 0) {
|
|
52
72
|
const matchers = locationExcludes.map(makeLocationMatcher);
|
|
53
|
-
result = result.filter(j =>
|
|
54
|
-
const loc = (j.location || '').toLowerCase();
|
|
55
|
-
return !matchers.some(m => m(loc));
|
|
56
|
-
});
|
|
73
|
+
result = result.filter(j => !jobLocations(j).every(loc => matchers.some(m => m(loc))));
|
|
57
74
|
}
|
|
58
75
|
|
|
59
|
-
|
|
60
|
-
|
|
76
|
+
return result;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* Sort, then cut the page (see applyFiltersDetailed for the order rules).
|
|
81
|
+
*/
|
|
82
|
+
export function pageJobs(jobs, { order = 'newest', offset = 0, limit = 100 } = {}) {
|
|
83
|
+
let result = jobs;
|
|
84
|
+
|
|
85
|
+
if (order !== 'board') {
|
|
86
|
+
result = [...result].sort(byNewest);
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
const start = typeof offset === 'number' && offset > 0 ? offset : 0;
|
|
90
|
+
const end = typeof limit === 'number' ? start + limit : undefined;
|
|
91
|
+
if (start > 0 || (end !== undefined && result.length > end)) {
|
|
92
|
+
result = result.slice(start, end);
|
|
61
93
|
}
|
|
62
94
|
|
|
63
95
|
return result;
|
|
64
96
|
}
|
|
65
97
|
|
|
66
98
|
/**
|
|
67
|
-
*
|
|
99
|
+
* Compile the two regex arguments, or throw ArgumentError naming the one
|
|
100
|
+
* that does not compile. Both are case-insensitive. fetchJobsDetailed calls
|
|
101
|
+
* this before its first request, so a bad pattern is reported as a bad
|
|
102
|
+
* argument and costs no upstream traffic.
|
|
103
|
+
*
|
|
104
|
+
* @returns {{ title: RegExp|null, topic: RegExp|null }}
|
|
105
|
+
*/
|
|
106
|
+
export function compileFilterPatterns({ titleFilter, filter } = {}) {
|
|
107
|
+
return {
|
|
108
|
+
title: titleFilter ? compilePattern(titleFilter, 'titleFilter') : null,
|
|
109
|
+
topic: filter ? compilePattern(filter, 'filter') : null,
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
function compilePattern(source, name) {
|
|
114
|
+
try {
|
|
115
|
+
return new RegExp(source, 'i');
|
|
116
|
+
} catch (err) {
|
|
117
|
+
throw new ArgumentError(`${name}: ${err.message}`);
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* Every location a job is open in, lowercased. A job passes an include when
|
|
123
|
+
* any of them matches and is dropped by an exclude only when all of them
|
|
124
|
+
* match: a role open in Berlin and New York is still open in Berlin for
|
|
125
|
+
* someone excluding the US (issue #68). Jobs from before `locations`
|
|
126
|
+
* existed fall back to the single `location` string.
|
|
127
|
+
*/
|
|
128
|
+
function jobLocations(job) {
|
|
129
|
+
const list = Array.isArray(job.locations) && job.locations.length > 0
|
|
130
|
+
? job.locations
|
|
131
|
+
: [job.location || ''];
|
|
132
|
+
return list.map(loc => String(loc).toLowerCase());
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
function postedTime(job) {
|
|
136
|
+
if (!job.postedAt) return null;
|
|
137
|
+
const t = new Date(job.postedAt).getTime();
|
|
138
|
+
return Number.isFinite(t) ? t : null;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
function byNewest(a, b) {
|
|
142
|
+
const ta = postedTime(a);
|
|
143
|
+
const tb = postedTime(b);
|
|
144
|
+
if (ta !== tb) {
|
|
145
|
+
if (ta === null) return 1;
|
|
146
|
+
if (tb === null) return -1;
|
|
147
|
+
return tb - ta;
|
|
148
|
+
}
|
|
149
|
+
const ia = a.id || '';
|
|
150
|
+
const ib = b.id || '';
|
|
151
|
+
return ia < ib ? -1 : ia > ib ? 1 : 0;
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* Build a matcher for a single location keyword. The matcher takes a
|
|
156
|
+
* lowercased location string.
|
|
68
157
|
*
|
|
69
158
|
* Short tokens (≤4 chars) use word-boundary matching to prevent substring
|
|
70
159
|
* collisions like "US" matching "Australia", "Brussels", "Belarus", or "UK"
|
|
71
|
-
* matching "
|
|
160
|
+
* matching "Ukraine". Longer tokens use substring matching so phrases like
|
|
72
161
|
* "United States" can match "United States of America".
|
|
162
|
+
*
|
|
163
|
+
* Exported for the Workday list pre-filter, so one rule (trim, empty
|
|
164
|
+
* keywords never match, word boundaries for short tokens) applies before
|
|
165
|
+
* and after detail hydration (issue #61).
|
|
73
166
|
*/
|
|
74
|
-
function makeLocationMatcher(needle) {
|
|
167
|
+
export function makeLocationMatcher(needle) {
|
|
75
168
|
const lower = (needle || '').toLowerCase().trim();
|
|
76
169
|
if (!lower) return () => false;
|
|
77
170
|
if (lower.length <= 4) {
|