jd-intel 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -0
- package/package.json +1 -1
- package/src/adapters/ashby.js +20 -72
- package/src/adapters/greenhouse.js +19 -9
- package/src/adapters/lever.js +16 -8
- package/src/adapters/recruitee.js +20 -9
- package/src/adapters/smartrecruiters.js +113 -36
- package/src/adapters/teamtailor.js +37 -15
- package/src/adapters/workday.js +60 -19
- package/src/boards.js +80 -0
- package/src/cli.js +12 -8
- package/src/errors.js +14 -0
- package/src/filters.js +62 -23
- package/src/http.js +184 -0
- package/src/index.js +168 -57
- package/src/registry.js +79 -25
package/src/adapters/workday.js
CHANGED
|
@@ -1,5 +1,8 @@
|
|
|
1
1
|
import { normalize } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { makeLocationMatcher } from '../filters.js';
|
|
4
|
+
import { atsFetch } from '../http.js';
|
|
5
|
+
import { orgHost } from '../boards.js';
|
|
3
6
|
|
|
4
7
|
const MAX_DETAIL_FETCHES = 100;
|
|
5
8
|
const LIST_PAGE_SIZE = 20;
|
|
@@ -30,7 +33,9 @@ const MULTI_LOCATION = /^\s*\d+\s+locations?\s*$/;
|
|
|
30
33
|
* detail set.
|
|
31
34
|
*
|
|
32
35
|
* @param {string} slug - normalized company slug (registry routing key)
|
|
33
|
-
* @param {object} [ctx] - { config:{tenant,env,site}, companyName, filterContext }
|
|
36
|
+
* @param {object} [ctx] - { config:{tenant,env,site}, companyName, filterContext, report };
|
|
37
|
+
* report is called once, after hydration, with
|
|
38
|
+
* { ats, listed, prefiltered, hydrated, capped, org_name, org_url } when given
|
|
34
39
|
* @returns {Promise<Array>} Normalized job objects
|
|
35
40
|
*/
|
|
36
41
|
export async function fetchWorkday(slug, ctx = {}) {
|
|
@@ -46,12 +51,25 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
46
51
|
let offset = 0;
|
|
47
52
|
let pages = 0;
|
|
48
53
|
let firstTotal = 0;
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
}
|
|
54
|
+
let listCapped = false;
|
|
55
|
+
while (true) {
|
|
56
|
+
if (pages >= LIST_PAGE_HARD_CAP) {
|
|
57
|
+
listCapped = true;
|
|
58
|
+
break;
|
|
59
|
+
}
|
|
60
|
+
let resp;
|
|
61
|
+
try {
|
|
62
|
+
resp = await atsFetch(`${base}/jobs`, {
|
|
63
|
+
method: 'POST',
|
|
64
|
+
headers: { 'Content-Type': 'application/json' },
|
|
65
|
+
body: JSON.stringify({ appliedFacets: {}, limit: LIST_PAGE_SIZE, offset, searchText: '' }),
|
|
66
|
+
});
|
|
67
|
+
} catch (err) {
|
|
68
|
+
// A 429 or 5xx that outlasted the retries, or a network failure.
|
|
69
|
+
// After the first page, keep the postings already read.
|
|
70
|
+
if (offset === 0) throw err;
|
|
71
|
+
break;
|
|
72
|
+
}
|
|
55
73
|
|
|
56
74
|
if (!resp.ok) {
|
|
57
75
|
if (offset === 0) {
|
|
@@ -83,21 +101,23 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
83
101
|
const re = new RegExp(fc.titleFilter, 'i');
|
|
84
102
|
candidates = candidates.filter(p => re.test(p.title || ''));
|
|
85
103
|
}
|
|
104
|
+
// Location rows go through the applyFilters matcher, so the pre-filter
|
|
105
|
+
// keeps exactly the rows the pass after hydration would (issue #61).
|
|
106
|
+
// "2 Locations" says nothing about where: the row stays a candidate
|
|
107
|
+
// through both filters and that later pass decides on the detail's
|
|
108
|
+
// location list.
|
|
86
109
|
if (Array.isArray(fc.locationIncludes) && fc.locationIncludes.length > 0) {
|
|
87
|
-
const inc = fc.locationIncludes.map(
|
|
110
|
+
const inc = fc.locationIncludes.map(makeLocationMatcher);
|
|
88
111
|
candidates = candidates.filter(p => {
|
|
89
112
|
const loc = (p.locationsText || '').toLowerCase();
|
|
90
|
-
|
|
91
|
-
// and the pass after hydration decides on the detail's location list.
|
|
92
|
-
if (MULTI_LOCATION.test(loc)) return true;
|
|
93
|
-
return inc.some(s => loc.includes(s));
|
|
113
|
+
return MULTI_LOCATION.test(loc) || inc.some(m => m(loc));
|
|
94
114
|
});
|
|
95
115
|
}
|
|
96
116
|
if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
|
|
97
|
-
const exc = fc.locationExcludes.map(
|
|
117
|
+
const exc = fc.locationExcludes.map(makeLocationMatcher);
|
|
98
118
|
candidates = candidates.filter(p => {
|
|
99
119
|
const loc = (p.locationsText || '').toLowerCase();
|
|
100
|
-
return !exc.some(
|
|
120
|
+
return MULTI_LOCATION.test(loc) || !exc.some(m => m(loc));
|
|
101
121
|
});
|
|
102
122
|
}
|
|
103
123
|
if (typeof fc.postedWithinDays === 'number') {
|
|
@@ -120,23 +140,29 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
120
140
|
const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
|
|
121
141
|
const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
|
|
122
142
|
const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
|
|
123
|
-
|
|
143
|
+
const hydrate = candidates.slice(0, cap);
|
|
124
144
|
|
|
125
|
-
// 4. Hydrate descriptions via the per-posting detail endpoint.
|
|
126
|
-
|
|
145
|
+
// 4. Hydrate descriptions via the per-posting detail endpoint. The detail
|
|
146
|
+
// also carries `hiringOrganization: { name, url }` next to
|
|
147
|
+
// jobPostingInfo; the list does not. Kept per posting in list order so
|
|
148
|
+
// the one reported is the first hydrated posting's, not whichever
|
|
149
|
+
// detail answered first (a tenant can post under several entities).
|
|
150
|
+
const orgs = [];
|
|
151
|
+
const jobs = await Promise.all(hydrate.map(async (p, i) => {
|
|
127
152
|
const externalPath = p.externalPath || ''; // already begins with '/job/...'
|
|
128
153
|
let info = {};
|
|
129
154
|
try {
|
|
130
155
|
// externalPath already carries the '/job/...' segment, so it is
|
|
131
156
|
// concatenated directly onto the CXS base. Inserting another
|
|
132
157
|
// '/job' here yields '/job/job/...' which Workday rejects (422).
|
|
133
|
-
const dResp = await
|
|
158
|
+
const dResp = await atsFetch(`${base}${externalPath}`);
|
|
134
159
|
if (dResp.ok) {
|
|
135
160
|
const detail = await dResp.json();
|
|
136
161
|
info = detail.jobPostingInfo || {};
|
|
162
|
+
orgs[i] = detail.hiringOrganization || null;
|
|
137
163
|
}
|
|
138
164
|
} catch {
|
|
139
|
-
// detail failed: fall back to list fields, empty description
|
|
165
|
+
// detail failed, retries included: fall back to list fields, empty description
|
|
140
166
|
}
|
|
141
167
|
|
|
142
168
|
return normalize({
|
|
@@ -160,6 +186,21 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
160
186
|
}, 'workday');
|
|
161
187
|
}));
|
|
162
188
|
|
|
189
|
+
// Nothing hydrated (a filter miss, an empty site) means no detail was
|
|
190
|
+
// read, so the org is unknown rather than absent: null, null.
|
|
191
|
+
if (typeof ctx.report === 'function') {
|
|
192
|
+
const org = orgs.find(Boolean) || {};
|
|
193
|
+
ctx.report({
|
|
194
|
+
ats: 'workday',
|
|
195
|
+
listed: postings.length,
|
|
196
|
+
prefiltered: candidates.length,
|
|
197
|
+
hydrated: hydrate.length,
|
|
198
|
+
capped: listCapped || hydrate.length < candidates.length,
|
|
199
|
+
org_name: org.name || null,
|
|
200
|
+
org_url: orgHost(org.url),
|
|
201
|
+
});
|
|
202
|
+
}
|
|
203
|
+
|
|
163
204
|
return jobs;
|
|
164
205
|
}
|
|
165
206
|
|
package/src/boards.js
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The boards[] entry of a fetchJobsDetailed result (issues #58, #60, #87).
|
|
3
|
+
*
|
|
4
|
+
* A board is one (ats, slug) the library fetched, with what came back. The
|
|
5
|
+
* fields fall in three groups: what the registry or the caller said about
|
|
6
|
+
* it (`name`, `site`), what the fetch found (`jobs_found`, `matched`,
|
|
7
|
+
* `scan`), and what the board says about itself (`org_name`, `org_url`).
|
|
8
|
+
* Only the adapters can read the third group, and they hand it over through
|
|
9
|
+
* ctx.report. Both fields are null where the platform exposes nothing, and
|
|
10
|
+
* they are never filled from the slug or the registry name: a slug is an
|
|
11
|
+
* address, not a confirmed identity.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
const BOARD_URLS = {
|
|
15
|
+
greenhouse: (slug) => `https://boards.greenhouse.io/${slug}`,
|
|
16
|
+
lever: (slug) => `https://jobs.lever.co/${slug}`,
|
|
17
|
+
ashby: (slug) => `https://jobs.ashbyhq.com/${slug}`,
|
|
18
|
+
smartrecruiters: (slug) => `https://careers.smartrecruiters.com/${slug}`,
|
|
19
|
+
teamtailor: (slug) => `https://${slug}.teamtailor.com`,
|
|
20
|
+
recruitee: (slug) => `https://${slug}.recruitee.com`,
|
|
21
|
+
workday: (slug, config) => (config ? `https://${config.tenant}.${config.env}.myworkdayjobs.com/${config.site}` : null),
|
|
22
|
+
};
|
|
23
|
+
|
|
24
|
+
// Domains the platforms own. A link there (boards.greenhouse.io,
|
|
25
|
+
// jobs.lever.co, testco.recruitee.com, cisco.wd5.myworkdayjobs.com) says
|
|
26
|
+
// which ATS hosts the board, nothing about whose board it is.
|
|
27
|
+
const ATS_DOMAINS = ['greenhouse.io', 'lever.co', 'ashbyhq.com', 'smartrecruiters.com', 'teamtailor.com', 'recruitee.com', 'myworkdayjobs.com'];
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* The page a person opens to see the board, or null when the ATS is unknown
|
|
31
|
+
* or, for Workday, no {tenant, env, site} is at hand.
|
|
32
|
+
*/
|
|
33
|
+
export function boardUrl(ats, slug, config) {
|
|
34
|
+
const build = BOARD_URLS[ats];
|
|
35
|
+
return build ? build(slug, config) : null;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* The bare hostname a board's link points at ("jobs.example.com"), for
|
|
40
|
+
* org_url. Null when the link is missing or malformed, and null when the
|
|
41
|
+
* host belongs to an ATS, since that carries no signal about the company.
|
|
42
|
+
*/
|
|
43
|
+
export function orgHost(link) {
|
|
44
|
+
let host;
|
|
45
|
+
try {
|
|
46
|
+
host = new URL(link).hostname.toLowerCase();
|
|
47
|
+
} catch {
|
|
48
|
+
return null;
|
|
49
|
+
}
|
|
50
|
+
if (!host || ATS_DOMAINS.some(d => host === d || host.endsWith(`.${d}`))) return null;
|
|
51
|
+
return host;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* @param {object} board
|
|
56
|
+
* @param {string} board.ats
|
|
57
|
+
* @param {string} board.slug - The slug the adapter was called with (canonical casing on a registry hit)
|
|
58
|
+
* @param {string|null} board.name - Registry row name; null for a probe or an override
|
|
59
|
+
* @param {object} [board.config] - Workday {tenant, env, site}, when one was used
|
|
60
|
+
* @param {string|null} board.org_name - The organization name the ATS response states, else null
|
|
61
|
+
* @param {string|null} board.org_url - The careers or company host the board links to (see orgHost), else null
|
|
62
|
+
* @param {number} board.jobs_found - Rows the board listed before any filter: the list count an adapter reported through ctx.report when it filters before hydrating, else the rows it returned
|
|
63
|
+
* @param {number} board.matched - Rows left after filters, before offset and limit
|
|
64
|
+
* @param {object|null} board.scan - The { listed, prefiltered, hydrated, capped } counts the adapter reported through ctx.report, else null
|
|
65
|
+
*/
|
|
66
|
+
export function describeBoard({ ats, slug, name = null, config, org_name = null, org_url = null, jobs_found, matched = 0, scan = null }) {
|
|
67
|
+
return {
|
|
68
|
+
ats,
|
|
69
|
+
slug,
|
|
70
|
+
name,
|
|
71
|
+
site: ats === 'workday' && config ? config.site : null,
|
|
72
|
+
board_url: boardUrl(ats, slug, config),
|
|
73
|
+
org_name,
|
|
74
|
+
org_url,
|
|
75
|
+
jobs_found,
|
|
76
|
+
matched,
|
|
77
|
+
selected: true,
|
|
78
|
+
scan,
|
|
79
|
+
};
|
|
80
|
+
}
|
package/src/cli.js
CHANGED
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
import { realpathSync } from 'node:fs';
|
|
13
13
|
import { fileURLToPath } from 'node:url';
|
|
14
14
|
import { fetchJobs } from './index.js';
|
|
15
|
-
import {
|
|
15
|
+
import { detectAtsDetailed, searchRegistry } from './registry.js';
|
|
16
16
|
|
|
17
17
|
const [,, command, ...args] = process.argv;
|
|
18
18
|
|
|
@@ -113,13 +113,17 @@ async function main() {
|
|
|
113
113
|
const company = args[0];
|
|
114
114
|
if (!company) { console.error('Usage: jd-intel detect <company>'); process.exit(1); }
|
|
115
115
|
console.log(`Detecting ATS for ${company}...`);
|
|
116
|
-
const
|
|
117
|
-
|
|
118
|
-
console.log('
|
|
119
|
-
}
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
116
|
+
const { boards, failed } = await detectAtsDetailed(company);
|
|
117
|
+
for (const b of boards) {
|
|
118
|
+
console.log(` Found: ${b.ats} (slug: ${b.slug}, ${b.source === 'registry' ? 'in the registry' : 'live probe'})`);
|
|
119
|
+
}
|
|
120
|
+
for (const f of failed) {
|
|
121
|
+
console.log(` Could not check ${f.ats}: ${f.message}`);
|
|
122
|
+
}
|
|
123
|
+
if (boards.length === 0) {
|
|
124
|
+
console.log(failed.length > 0
|
|
125
|
+
? 'No ATS board confirmed. At least one check failed, so this is not a definite miss. Retry in a moment.'
|
|
126
|
+
: 'No ATS board found for this company.');
|
|
123
127
|
}
|
|
124
128
|
break;
|
|
125
129
|
}
|
package/src/errors.js
CHANGED
|
@@ -34,6 +34,20 @@ export class AtsError extends Error {
|
|
|
34
34
|
}
|
|
35
35
|
}
|
|
36
36
|
|
|
37
|
+
/**
|
|
38
|
+
* Thrown when a call cannot proceed because of its arguments: a missing
|
|
39
|
+
* company, an unknown ATS name, a regex that does not compile. Carries
|
|
40
|
+
* `code: 'invalid_args'` so callers tell a bad request from a failed fetch
|
|
41
|
+
* (AtsError) without reading the message.
|
|
42
|
+
*/
|
|
43
|
+
export class ArgumentError extends Error {
|
|
44
|
+
constructor(message) {
|
|
45
|
+
super(message);
|
|
46
|
+
this.name = 'ArgumentError';
|
|
47
|
+
this.code = ERROR_CODES.INVALID_ARGS;
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
|
|
37
51
|
/**
|
|
38
52
|
* Helper for adapters: build an AtsError from an HTTP status (429 => rate
|
|
39
53
|
* limited, anything else => unreachable) with the given message.
|
package/src/filters.js
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { ArgumentError } from './errors.js';
|
|
2
|
+
|
|
1
3
|
/**
|
|
2
4
|
* Apply filters to a list of normalized jobs.
|
|
3
5
|
*
|
|
@@ -23,30 +25,32 @@ export function applyFilters(jobs, options = {}) {
|
|
|
23
25
|
* @returns {{ jobs: Array, total_matched: number }}
|
|
24
26
|
*/
|
|
25
27
|
export function applyFiltersDetailed(jobs, options = {}) {
|
|
26
|
-
const {
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
28
|
+
const { order = 'newest', offset = 0, limit = 100 } = options;
|
|
29
|
+
const matched = filterJobs(jobs, options);
|
|
30
|
+
return { jobs: pageJobs(matched, { order, offset, limit }), total_matched: matched.length };
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* The filter step on its own: every job that passes titleFilter, filter,
|
|
35
|
+
* postedWithinDays and the location filters, in the order given. No sort
|
|
36
|
+
* and no paging, so fetchJobsDetailed can count the matches per board
|
|
37
|
+
* before the page cut removes them.
|
|
38
|
+
*/
|
|
39
|
+
export function filterJobs(jobs, options = {}) {
|
|
40
|
+
const { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes } = options;
|
|
41
|
+
const { title, topic } = compileFilterPatterns({ titleFilter, filter });
|
|
36
42
|
|
|
37
43
|
let result = jobs;
|
|
38
44
|
|
|
39
|
-
if (
|
|
40
|
-
|
|
41
|
-
result = result.filter(j => pattern.test(j.title || ''));
|
|
45
|
+
if (title) {
|
|
46
|
+
result = result.filter(j => title.test(j.title || ''));
|
|
42
47
|
}
|
|
43
48
|
|
|
44
|
-
if (
|
|
45
|
-
const pattern = new RegExp(filter, 'i');
|
|
49
|
+
if (topic) {
|
|
46
50
|
result = result.filter(j =>
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
51
|
+
topic.test(j.title || '') ||
|
|
52
|
+
topic.test(j.department || '') ||
|
|
53
|
+
topic.test(j.description || '')
|
|
50
54
|
);
|
|
51
55
|
}
|
|
52
56
|
|
|
@@ -69,7 +73,14 @@ export function applyFiltersDetailed(jobs, options = {}) {
|
|
|
69
73
|
result = result.filter(j => !jobLocations(j).every(loc => matchers.some(m => m(loc))));
|
|
70
74
|
}
|
|
71
75
|
|
|
72
|
-
|
|
76
|
+
return result;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* Sort, then cut the page (see applyFiltersDetailed for the order rules).
|
|
81
|
+
*/
|
|
82
|
+
export function pageJobs(jobs, { order = 'newest', offset = 0, limit = 100 } = {}) {
|
|
83
|
+
let result = jobs;
|
|
73
84
|
|
|
74
85
|
if (order !== 'board') {
|
|
75
86
|
result = [...result].sort(byNewest);
|
|
@@ -81,7 +92,30 @@ export function applyFiltersDetailed(jobs, options = {}) {
|
|
|
81
92
|
result = result.slice(start, end);
|
|
82
93
|
}
|
|
83
94
|
|
|
84
|
-
return
|
|
95
|
+
return result;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Compile the two regex arguments, or throw ArgumentError naming the one
|
|
100
|
+
* that does not compile. Both are case-insensitive. fetchJobsDetailed calls
|
|
101
|
+
* this before its first request, so a bad pattern is reported as a bad
|
|
102
|
+
* argument and costs no upstream traffic.
|
|
103
|
+
*
|
|
104
|
+
* @returns {{ title: RegExp|null, topic: RegExp|null }}
|
|
105
|
+
*/
|
|
106
|
+
export function compileFilterPatterns({ titleFilter, filter } = {}) {
|
|
107
|
+
return {
|
|
108
|
+
title: titleFilter ? compilePattern(titleFilter, 'titleFilter') : null,
|
|
109
|
+
topic: filter ? compilePattern(filter, 'filter') : null,
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
function compilePattern(source, name) {
|
|
114
|
+
try {
|
|
115
|
+
return new RegExp(source, 'i');
|
|
116
|
+
} catch (err) {
|
|
117
|
+
throw new ArgumentError(`${name}: ${err.message}`);
|
|
118
|
+
}
|
|
85
119
|
}
|
|
86
120
|
|
|
87
121
|
/**
|
|
@@ -118,14 +152,19 @@ function byNewest(a, b) {
|
|
|
118
152
|
}
|
|
119
153
|
|
|
120
154
|
/**
|
|
121
|
-
* Build a matcher for a single location keyword.
|
|
155
|
+
* Build a matcher for a single location keyword. The matcher takes a
|
|
156
|
+
* lowercased location string.
|
|
122
157
|
*
|
|
123
158
|
* Short tokens (≤4 chars) use word-boundary matching to prevent substring
|
|
124
159
|
* collisions like "US" matching "Australia", "Brussels", "Belarus", or "UK"
|
|
125
|
-
* matching "
|
|
160
|
+
* matching "Ukraine". Longer tokens use substring matching so phrases like
|
|
126
161
|
* "United States" can match "United States of America".
|
|
162
|
+
*
|
|
163
|
+
* Exported for the Workday list pre-filter, so one rule (trim, empty
|
|
164
|
+
* keywords never match, word boundaries for short tokens) applies before
|
|
165
|
+
* and after detail hydration (issue #61).
|
|
127
166
|
*/
|
|
128
|
-
function makeLocationMatcher(needle) {
|
|
167
|
+
export function makeLocationMatcher(needle) {
|
|
129
168
|
const lower = (needle || '').toLowerCase().trim();
|
|
130
169
|
if (!lower) return () => false;
|
|
131
170
|
if (lower.length <= 4) {
|
package/src/http.js
ADDED
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
import { AtsError, ERROR_CODES, atsErrorFromStatus } from './errors.js';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* The one HTTP door for every adapter request (issue #7).
|
|
5
|
+
*
|
|
6
|
+
* atsFetch() wraps the global fetch with the politeness every ATS expects
|
|
7
|
+
* and the failure handling the adapters used to leave out:
|
|
8
|
+
* - a timeout per attempt on the wait for the response to start
|
|
9
|
+
* - retries with exponential backoff and jitter on 429, any 5xx, and
|
|
10
|
+
* network errors (DNS, reset, timeout), honoring Retry-After; a
|
|
11
|
+
* certificate failure is thrown at once, since it cannot pass later
|
|
12
|
+
* - a cap on requests in flight per host
|
|
13
|
+
*
|
|
14
|
+
* Every other status resolves normally, so an adapter keeps its own 404
|
|
15
|
+
* handling. Once the retries are used up the caller gets an AtsError:
|
|
16
|
+
* rate_limited for a 429, ats_unreachable for a 5xx or a network failure.
|
|
17
|
+
* The global fetch is read on every attempt so a test's mock of it applies.
|
|
18
|
+
*
|
|
19
|
+
* The timer stops once the headers are in. Reading the body is the caller's
|
|
20
|
+
* step (resp.json() in the adapter), and it is not timed: a signal left on
|
|
21
|
+
* the request would abort that read too, so a large board on a slow link
|
|
22
|
+
* would fail at the timeout with a raw TimeoutError thrown from resp.json(),
|
|
23
|
+
* outside this retry loop, where master downloaded it fine. undici's own
|
|
24
|
+
* body timeout (300s idle) still ends a stream that stalls.
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
const DEFAULTS = {
|
|
28
|
+
timeoutMs: 10_000,
|
|
29
|
+
retries: 3, // attempts per request in total; 1 turns retrying off
|
|
30
|
+
perHost: 4, // requests in flight per hostname
|
|
31
|
+
sleep: (ms) => new Promise((resolve) => setTimeout(resolve, ms)),
|
|
32
|
+
};
|
|
33
|
+
|
|
34
|
+
const BASE_BACKOFF_MS = 1000; // 1s, 2s, 4s, ...
|
|
35
|
+
const JITTER_MS = 250;
|
|
36
|
+
const MAX_RETRY_AFTER_MS = 30_000;
|
|
37
|
+
|
|
38
|
+
let settings = { ...DEFAULTS };
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* Replace the HTTP settings with the defaults plus `overrides`. For tests
|
|
42
|
+
* and scripts. `configureHttp()` restores the defaults. Returns a copy of
|
|
43
|
+
* the settings now in force.
|
|
44
|
+
*/
|
|
45
|
+
export function configureHttp(overrides = {}) {
|
|
46
|
+
settings = { ...DEFAULTS, ...overrides };
|
|
47
|
+
return { ...settings };
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* fetch(url, init) with a timeout, retries and a per-host queue.
|
|
52
|
+
*
|
|
53
|
+
* Resolves with the Response for any status that is not retried (2xx, 3xx,
|
|
54
|
+
* and 4xx other than 429). Throws AtsError once the retries are used up on
|
|
55
|
+
* a 429 or 5xx (with `.status`), or on a network error or a timeout waiting
|
|
56
|
+
* for the response to start.
|
|
57
|
+
*/
|
|
58
|
+
export async function atsFetch(url, init = {}) {
|
|
59
|
+
const host = new URL(url).hostname;
|
|
60
|
+
const { retries, sleep, timeoutMs } = settings;
|
|
61
|
+
|
|
62
|
+
for (let attempt = 1; ; attempt++) {
|
|
63
|
+
let resp;
|
|
64
|
+
try {
|
|
65
|
+
resp = await withHostSlot(host, () => fetchWithTimeout(url, init, timeoutMs));
|
|
66
|
+
} catch (err) {
|
|
67
|
+
if (attempt >= retries || isCertError(err)) {
|
|
68
|
+
const error = new AtsError(
|
|
69
|
+
ERROR_CODES.ATS_UNREACHABLE,
|
|
70
|
+
`${host}: ${describeCause(err, timeoutMs)} after ${attempts(attempt)}`
|
|
71
|
+
);
|
|
72
|
+
error.cause = err;
|
|
73
|
+
throw error;
|
|
74
|
+
}
|
|
75
|
+
await sleep(backoffMs(attempt));
|
|
76
|
+
continue;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
if (!isRetried(resp.status)) return resp;
|
|
80
|
+
if (attempt >= retries) {
|
|
81
|
+
throw atsErrorFromStatus(resp.status, `${host}: HTTP ${resp.status} after ${attempts(attempt)}`);
|
|
82
|
+
}
|
|
83
|
+
await sleep(retryAfterMs(resp) ?? backoffMs(attempt));
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* Three-state probe outcome for an adapter's has(): true on 2xx, false on
|
|
89
|
+
* a 404, and an AtsError for anything else (401, 403, ...), so a board the
|
|
90
|
+
* probe could not check never reads as "not here" (issue #55). A 429 or
|
|
91
|
+
* 5xx never reaches this point: atsFetch throws on those itself.
|
|
92
|
+
*/
|
|
93
|
+
export function probeResult(resp, label) {
|
|
94
|
+
if (resp.ok) return true;
|
|
95
|
+
if (resp.status === 404) return false;
|
|
96
|
+
throw atsErrorFromStatus(resp.status, `${label}: ${resp.status}`);
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
// The abort covers connecting and waiting for the headers, and is cleared as
|
|
100
|
+
// soon as fetch resolves so the body read that follows is never aborted
|
|
101
|
+
// (see the header comment). The reason is a TimeoutError like the one
|
|
102
|
+
// AbortSignal.timeout would raise, so describeCause reads both the same way.
|
|
103
|
+
async function fetchWithTimeout(url, init, timeoutMs) {
|
|
104
|
+
const controller = new AbortController();
|
|
105
|
+
const timer = setTimeout(
|
|
106
|
+
() => controller.abort(new DOMException(`Timed out after ${timeoutMs}ms`, 'TimeoutError')),
|
|
107
|
+
timeoutMs
|
|
108
|
+
);
|
|
109
|
+
try {
|
|
110
|
+
return await globalThis.fetch(url, { ...init, signal: controller.signal });
|
|
111
|
+
} finally {
|
|
112
|
+
clearTimeout(timer);
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
function isRetried(status) {
|
|
117
|
+
return status === 429 || status >= 500;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
function backoffMs(attempt) {
|
|
121
|
+
return BASE_BACKOFF_MS * 2 ** (attempt - 1) + Math.floor(Math.random() * JITTER_MS);
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
// Retry-After is either delay-seconds or an HTTP-date. Mocked responses may
|
|
125
|
+
// carry no headers at all.
|
|
126
|
+
function retryAfterMs(resp) {
|
|
127
|
+
const raw = resp.headers?.get?.('retry-after');
|
|
128
|
+
if (!raw) return null;
|
|
129
|
+
const seconds = Number(raw);
|
|
130
|
+
const ms = Number.isFinite(seconds) ? seconds * 1000 : Date.parse(raw) - Date.now();
|
|
131
|
+
if (!Number.isFinite(ms)) return null;
|
|
132
|
+
return Math.min(Math.max(ms, 0), MAX_RETRY_AFTER_MS);
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
// A certificate that fails validation fails the same way on the next
|
|
136
|
+
// attempt, so retrying it only costs time. Seen live: every
|
|
137
|
+
// {slug}.eu.teamtailor.com answers ERR_TLS_CERT_ALTNAME_INVALID.
|
|
138
|
+
const CERT_ERROR = /^(?:ERR_TLS_CERT_ALTNAME_INVALID|CERT_HAS_EXPIRED|CERT_NOT_YET_VALID|DEPTH_ZERO_SELF_SIGNED_CERT|SELF_SIGNED_CERT_IN_CHAIN|UNABLE_TO_VERIFY_LEAF_SIGNATURE|UNABLE_TO_GET_ISSUER_CERT(?:_LOCALLY)?)$/;
|
|
139
|
+
|
|
140
|
+
function isCertError(err) {
|
|
141
|
+
return CERT_ERROR.test(err?.cause?.code || '');
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
// undici reports socket failures as TypeError('fetch failed') with the OS
|
|
145
|
+
// code on `cause`, and rejects an aborted request with the signal's reason,
|
|
146
|
+
// here the TimeoutError from fetchWithTimeout.
|
|
147
|
+
function describeCause(err, timeoutMs) {
|
|
148
|
+
if (err?.name === 'TimeoutError') return `timed out after ${timeoutMs}ms`;
|
|
149
|
+
const detail = err?.cause?.code || err?.cause?.message;
|
|
150
|
+
const message = err?.message || String(err);
|
|
151
|
+
return detail ? `${message} (${detail})` : message;
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
function attempts(n) {
|
|
155
|
+
return `${n} attempt${n === 1 ? '' : 's'}`;
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
// Per-host queue. A finished request hands its slot straight to the next
|
|
159
|
+
// waiter (the count never dips in between), so the cap holds even when a new
|
|
160
|
+
// caller arrives while a waiter is being woken.
|
|
161
|
+
const hosts = new Map(); // hostname -> { active, waiting: [resolve] }
|
|
162
|
+
|
|
163
|
+
async function withHostSlot(host, run) {
|
|
164
|
+
let slot = hosts.get(host);
|
|
165
|
+
if (!slot) {
|
|
166
|
+
slot = { active: 0, waiting: [] };
|
|
167
|
+
hosts.set(host, slot);
|
|
168
|
+
}
|
|
169
|
+
if (slot.active >= settings.perHost) {
|
|
170
|
+
await new Promise((resolve) => slot.waiting.push(resolve));
|
|
171
|
+
} else {
|
|
172
|
+
slot.active += 1;
|
|
173
|
+
}
|
|
174
|
+
try {
|
|
175
|
+
return await run();
|
|
176
|
+
} finally {
|
|
177
|
+
const next = slot.waiting.shift();
|
|
178
|
+
if (next) next();
|
|
179
|
+
else {
|
|
180
|
+
slot.active -= 1;
|
|
181
|
+
if (slot.active === 0) hosts.delete(host);
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
}
|