jd-intel 0.9.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -1
- package/package.json +1 -1
- package/src/adapters/ashby.js +18 -76
- package/src/adapters/greenhouse.js +96 -34
- package/src/adapters/index.js +24 -14
- package/src/adapters/lever.js +13 -10
- package/src/adapters/recruitee.js +19 -9
- package/src/adapters/smartrecruiters.js +88 -38
- package/src/adapters/teamtailor.js +36 -15
- package/src/adapters/workday.js +97 -81
- package/src/boards.js +80 -0
- package/src/cli.js +42 -30
- package/src/errors.js +14 -0
- package/src/filters.js +109 -26
- package/src/http.js +184 -0
- package/src/index.js +179 -57
- package/src/normalizer.js +16 -2
- package/src/registry.js +80 -31
|
@@ -1,11 +1,14 @@
|
|
|
1
|
-
import { normalize } from '../normalizer.js';
|
|
1
|
+
import { normalize, missingContent } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { atsFetch, probeResult } from '../http.js';
|
|
4
|
+
import { prefilterRows } from '../filters.js';
|
|
3
5
|
|
|
4
6
|
const BASE_URL = 'https://api.smartrecruiters.com/v1/companies';
|
|
5
7
|
const PAGE_SIZE = 100;
|
|
8
|
+
const MAX_DETAIL_FETCHES = 100;
|
|
6
9
|
|
|
7
10
|
/**
|
|
8
|
-
* Fetch
|
|
11
|
+
* Fetch postings from a SmartRecruiters company.
|
|
9
12
|
* Public API, no auth required.
|
|
10
13
|
* Docs: https://developers.smartrecruiters.com/reference/postingsget-1
|
|
11
14
|
*
|
|
@@ -14,21 +17,29 @@ const PAGE_SIZE = 100;
|
|
|
14
17
|
* and the structured `compensation` block with it.
|
|
15
18
|
* - jd-intel's contract is "full JD text", so we must fetch each
|
|
16
19
|
* posting's DETAIL endpoint to get jobAd.sections.
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
+
*
|
|
21
|
+
* The list does carry name, location and releasedDate, so the same
|
|
22
|
+
* pre-filter and detail budget Workday applies run here: list-evaluable
|
|
23
|
+
* filters narrow the candidates, then at most MAX_DETAIL_FETCHES of them
|
|
24
|
+
* are hydrated (see prefilterRows). Without a filterContext the
|
|
25
|
+
* cap still holds, so a direct call on a 400-posting tenant reads 100.
|
|
20
26
|
*
|
|
21
27
|
* @param {string} slug - SmartRecruiters company identifier (e.g., 'Visa')
|
|
28
|
+
* @param {object} [ctx] - { filterContext, report }; report is called once
|
|
29
|
+
* with { listed, prefiltered, hydrated, capped, org_name, org_url }
|
|
30
|
+
* when given
|
|
22
31
|
* @returns {Promise<Array>} Normalized job objects
|
|
23
32
|
*/
|
|
24
|
-
export async function fetchSmartrecruiters(slug) {
|
|
33
|
+
export async function fetchSmartrecruiters(slug, ctx = {}) {
|
|
34
|
+
const fc = ctx.filterContext || {};
|
|
35
|
+
|
|
25
36
|
// 1. Page through the postings list.
|
|
26
37
|
const postings = [];
|
|
27
38
|
let offset = 0;
|
|
28
39
|
|
|
29
40
|
while (true) {
|
|
30
41
|
const listUrl = `${BASE_URL}/${slug}/postings?limit=${PAGE_SIZE}&offset=${offset}`;
|
|
31
|
-
const resp = await
|
|
42
|
+
const resp = await atsFetch(listUrl);
|
|
32
43
|
|
|
33
44
|
if (!resp.ok) {
|
|
34
45
|
if (resp.status === 404) return []; // Company not found
|
|
@@ -43,22 +54,61 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
43
54
|
if (content.length === 0 || offset >= (data.totalFound || 0)) break;
|
|
44
55
|
}
|
|
45
56
|
|
|
46
|
-
// 2.
|
|
47
|
-
|
|
57
|
+
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost, then
|
|
58
|
+
// the detail budget (see prefilterRows). The list row carries name,
|
|
59
|
+
// location and releasedDate. The detail adds no location (unlike
|
|
60
|
+
// Workday's additionalLocations), so a row with none follows the
|
|
61
|
+
// library's rule now: out under includes, kept under excludes.
|
|
62
|
+
// postedAt comes from releasedDate alone, so a missing or unparseable
|
|
63
|
+
// date is out, as it is in the library.
|
|
64
|
+
const { candidates, hydrate } = prefilterRows(postings, fc, {
|
|
65
|
+
title: p => p.name || '',
|
|
66
|
+
location: p => listLocation(p).location.toLowerCase(),
|
|
67
|
+
postedWithin: (p, days) => {
|
|
68
|
+
const released = new Date(p.releasedDate || '').getTime();
|
|
69
|
+
return Number.isFinite(released) && released >= Date.now() - days * 86400000;
|
|
70
|
+
},
|
|
71
|
+
max: MAX_DETAIL_FETCHES,
|
|
72
|
+
});
|
|
73
|
+
|
|
74
|
+
// Every list row carries company { identifier, name }. Neither the list
|
|
75
|
+
// nor the detail has a company website, and postingUrl is always on
|
|
76
|
+
// jobs.smartrecruiters.com, so org_url stays null (issue #58).
|
|
77
|
+
if (typeof ctx.report === 'function') {
|
|
78
|
+
ctx.report({
|
|
79
|
+
listed: postings.length,
|
|
80
|
+
prefiltered: candidates.length,
|
|
81
|
+
hydrated: hydrate.length,
|
|
82
|
+
capped: hydrate.length < candidates.length,
|
|
83
|
+
org_name: postings.find(p => p.company?.name)?.company.name || null,
|
|
84
|
+
org_url: null,
|
|
85
|
+
});
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
// 4. Fetch detail per candidate for the description. atsFetch's per-host
|
|
89
|
+
// queue keeps this fan-out to 4 requests at a time (a 412-posting
|
|
90
|
+
// tenant measured 53s unbounded), so the cap also holds the detail
|
|
91
|
+
// step to roughly 13s.
|
|
92
|
+
const jobs = await Promise.all(hydrate.map(async (p) => {
|
|
48
93
|
let sections = {};
|
|
49
94
|
let postingUrl = '';
|
|
50
95
|
let salary = null;
|
|
96
|
+
let content; // set only when the detail could not be read (issue #85)
|
|
51
97
|
|
|
52
98
|
try {
|
|
53
|
-
const detailResp = await
|
|
99
|
+
const detailResp = await atsFetch(`${BASE_URL}/${slug}/postings/${p.id}`);
|
|
54
100
|
if (detailResp.ok) {
|
|
55
101
|
const detail = await detailResp.json();
|
|
56
102
|
sections = detail.jobAd?.sections || {};
|
|
57
103
|
postingUrl = detail.postingUrl || detail.applyUrl || '';
|
|
58
104
|
salary = parseCompensation(detail.compensation);
|
|
105
|
+
} else {
|
|
106
|
+
content = missingContent(detailResp);
|
|
59
107
|
}
|
|
60
|
-
} catch {
|
|
61
|
-
// Detail fetch failed
|
|
108
|
+
} catch (err) {
|
|
109
|
+
// Detail fetch failed, retries included: list-only fields (no
|
|
110
|
+
// description, no url), marked missing.
|
|
111
|
+
content = missingContent(err);
|
|
62
112
|
}
|
|
63
113
|
|
|
64
114
|
const description = [
|
|
@@ -67,18 +117,7 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
67
117
|
sections.additionalInformation?.text,
|
|
68
118
|
].filter(Boolean).join('\n\n');
|
|
69
119
|
|
|
70
|
-
const
|
|
71
|
-
const place = loc.fullLocation
|
|
72
|
-
|| [loc.city, loc.region, loc.country].filter(Boolean).join(', ');
|
|
73
|
-
let location = place;
|
|
74
|
-
let workplace = null;
|
|
75
|
-
if (loc.remote) {
|
|
76
|
-
location = `Remote - ${place}`.replace(/ - $/, ' ');
|
|
77
|
-
workplace = 'remote';
|
|
78
|
-
} else if (loc.hybrid) {
|
|
79
|
-
location = `Hybrid - ${place}`.replace(/ - $/, ' ');
|
|
80
|
-
workplace = 'hybrid';
|
|
81
|
-
}
|
|
120
|
+
const { location, workplace } = listLocation(p);
|
|
82
121
|
|
|
83
122
|
return normalize({
|
|
84
123
|
companySlug: slug,
|
|
@@ -91,6 +130,7 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
91
130
|
url: postingUrl,
|
|
92
131
|
postedAt: p.releasedDate || null,
|
|
93
132
|
salary, // null when the detail has no compensation; normalize() then parses text
|
|
133
|
+
content,
|
|
94
134
|
metadata: {
|
|
95
135
|
smartRecruitersId: p.id,
|
|
96
136
|
refNumber: p.refNumber || '',
|
|
@@ -104,6 +144,20 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
104
144
|
return jobs;
|
|
105
145
|
}
|
|
106
146
|
|
|
147
|
+
/**
|
|
148
|
+
* The location string and workplace type a list row yields. Built once
|
|
149
|
+
* here so the pre-filter matches exactly what the normalized job carries,
|
|
150
|
+
* "Remote - " and "Hybrid - " prefixes included.
|
|
151
|
+
*/
|
|
152
|
+
function listLocation(p) {
|
|
153
|
+
const loc = p.location || {};
|
|
154
|
+
const place = loc.fullLocation
|
|
155
|
+
|| [loc.city, loc.region, loc.country].filter(Boolean).join(', ');
|
|
156
|
+
if (loc.remote) return { location: `Remote - ${place}`.replace(/ - $/, ' '), workplace: 'remote' };
|
|
157
|
+
if (loc.hybrid) return { location: `Hybrid - ${place}`.replace(/ - $/, ' '), workplace: 'hybrid' };
|
|
158
|
+
return { location: place, workplace: null };
|
|
159
|
+
}
|
|
160
|
+
|
|
107
161
|
const PERIODS = { YEARLY: 'year', MONTHLY: 'month', HOURLY: 'hour' };
|
|
108
162
|
|
|
109
163
|
/**
|
|
@@ -130,20 +184,16 @@ function parseCompensation(comp) {
|
|
|
130
184
|
}
|
|
131
185
|
|
|
132
186
|
/**
|
|
133
|
-
* Check if a company exists on SmartRecruiters.
|
|
134
|
-
* (HEAD isn't reliably supported on the postings endpoint, so
|
|
135
|
-
* minimal GET.)
|
|
187
|
+
* Check if a company exists on SmartRecruiters. See probeResult for the
|
|
188
|
+
* outcomes. (HEAD isn't reliably supported on the postings endpoint, so
|
|
189
|
+
* use a minimal GET.)
|
|
136
190
|
*/
|
|
137
191
|
export async function hasSmartrecruiters(slug) {
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
return (data.totalFound || 0) > 0 || (data.content || []).length > 0;
|
|
146
|
-
} catch {
|
|
147
|
-
return false;
|
|
148
|
-
}
|
|
192
|
+
const resp = await atsFetch(`${BASE_URL}/${slug}/postings?limit=1`);
|
|
193
|
+
if (!probeResult(resp, `SmartRecruiters probe for ${slug}`)) return false;
|
|
194
|
+
// SmartRecruiters returns 200 with an empty page (not 404) for unknown
|
|
195
|
+
// companies, so resp.ok alone false-positives on any slug. Confirm at
|
|
196
|
+
// least one real posting exists before claiming a match.
|
|
197
|
+
const data = await resp.json();
|
|
198
|
+
return (data.totalFound || 0) > 0 || (data.content || []).length > 0;
|
|
149
199
|
}
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import { normalize, decodeEntities } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { atsFetch } from '../http.js';
|
|
4
|
+
import { orgHost } from '../boards.js';
|
|
3
5
|
|
|
4
6
|
/**
|
|
5
7
|
* Fetch jobs from a TeamTailor career site via its public RSS feed.
|
|
@@ -23,26 +25,33 @@ import { atsErrorFromStatus } from '../errors.js';
|
|
|
23
25
|
* collapse by one layer per pass.
|
|
24
26
|
*
|
|
25
27
|
* @param {string} slug - TeamTailor career-site slug (e.g., 'tibber')
|
|
28
|
+
* @param {object} [ctx] - { report }; report is called once with
|
|
29
|
+
* { org_name, org_url } when given
|
|
26
30
|
* @returns {Promise<Array>} Normalized job objects
|
|
27
31
|
*/
|
|
28
32
|
// Most sites are {slug}.teamtailor.com, but some sit on a regional
|
|
29
|
-
// segment, e.g. crunchbase.na.teamtailor.com. '' is the base host.
|
|
30
|
-
|
|
33
|
+
// segment, e.g. crunchbase.na.teamtailor.com. '' is the base host. There
|
|
34
|
+
// is no reachable eu segment: {slug}.eu.teamtailor.com fails TLS for every
|
|
35
|
+
// slug, known or not, because the wildcard certificate covers one label
|
|
36
|
+
// only (live check 2026-09-27). Probing it was a guaranteed failure that
|
|
37
|
+
// the has() contract would now report as an outage.
|
|
38
|
+
const TT_REGIONS = ['', 'na'];
|
|
31
39
|
|
|
32
40
|
// Feeds send `none`, `hybrid`, `fully` or `onsite`. `none` is no signal.
|
|
33
41
|
const REMOTE_STATUS = { hybrid: 'hybrid', fully: 'remote', onsite: 'onsite' };
|
|
34
42
|
|
|
35
43
|
/**
|
|
36
44
|
* Resolve which TeamTailor host actually serves this slug's feed.
|
|
37
|
-
* Returns the first 200 Response, throws on a non-404 error
|
|
38
|
-
* returns null if no
|
|
45
|
+
* Returns the first 200 Response, throws on a non-404 error (atsFetch
|
|
46
|
+
* throws the 429, 5xx and network cases itself), or returns null if no
|
|
47
|
+
* region has a feed.
|
|
39
48
|
*/
|
|
40
49
|
async function resolveFeed(slug, method = 'GET') {
|
|
41
50
|
for (const region of TT_REGIONS) {
|
|
42
51
|
const host = region
|
|
43
52
|
? `${slug}.${region}.teamtailor.com`
|
|
44
53
|
: `${slug}.teamtailor.com`;
|
|
45
|
-
const resp = await
|
|
54
|
+
const resp = await atsFetch(`https://${host}/jobs.rss`, {
|
|
46
55
|
method,
|
|
47
56
|
redirect: 'follow',
|
|
48
57
|
});
|
|
@@ -55,18 +64,32 @@ async function resolveFeed(slug, method = 'GET') {
|
|
|
55
64
|
return null;
|
|
56
65
|
}
|
|
57
66
|
|
|
58
|
-
export async function fetchTeamtailor(slug) {
|
|
67
|
+
export async function fetchTeamtailor(slug, ctx = {}) {
|
|
59
68
|
const resp = await resolveFeed(slug, 'GET');
|
|
60
69
|
if (!resp) return []; // No TeamTailor site in any known region
|
|
61
70
|
|
|
62
71
|
const xml = await resp.text();
|
|
63
72
|
|
|
64
|
-
const
|
|
65
|
-
|
|
66
|
-
).trim();
|
|
73
|
+
const channelTitle = (xml.match(/<channel>[\s\S]*?<title>([\s\S]*?)<\/title>/)?.[1] || '').trim();
|
|
74
|
+
const company = channelTitle || slug;
|
|
67
75
|
|
|
68
76
|
const items = [...xml.matchAll(/<item>([\s\S]*?)<\/item>/g)].map(m => m[1]);
|
|
69
77
|
|
|
78
|
+
// The channel title is the company as the site names itself. The channel
|
|
79
|
+
// <link> always sits on {slug}.teamtailor.com, but item links follow the
|
|
80
|
+
// site's custom domain when it has one (jobs.tibber.com on a feed served
|
|
81
|
+
// from tibber.teamtailor.com), so the first item's link is the host that
|
|
82
|
+
// can say something; the channel link is the fallback for an empty feed.
|
|
83
|
+
if (typeof ctx.report === 'function') {
|
|
84
|
+
const link = items[0]?.match(/<link>([\s\S]*?)<\/link>/)?.[1]
|
|
85
|
+
|| xml.match(/<channel>[\s\S]*?<link>([\s\S]*?)<\/link>/)?.[1]
|
|
86
|
+
|| '';
|
|
87
|
+
ctx.report({
|
|
88
|
+
org_name: decodeEntities(channelTitle) || null,
|
|
89
|
+
org_url: orgHost(link.trim()),
|
|
90
|
+
});
|
|
91
|
+
}
|
|
92
|
+
|
|
70
93
|
return items.map(item => {
|
|
71
94
|
const pick = (tag, src = item) => {
|
|
72
95
|
const m = src.match(new RegExp(`<${tag}[^>]*>([\\s\\S]*?)</${tag}>`));
|
|
@@ -120,12 +143,10 @@ export async function fetchTeamtailor(slug) {
|
|
|
120
143
|
}
|
|
121
144
|
|
|
122
145
|
/**
|
|
123
|
-
* Check if a company has a TeamTailor career site
|
|
146
|
+
* Check if a company has a TeamTailor career site: true when a regional
|
|
147
|
+
* host serves the feed, false when every host answers 404, and the
|
|
148
|
+
* AtsError from resolveFeed for anything else.
|
|
124
149
|
*/
|
|
125
150
|
export async function hasTeamtailor(slug) {
|
|
126
|
-
|
|
127
|
-
return (await resolveFeed(slug, 'HEAD')) !== null;
|
|
128
|
-
} catch {
|
|
129
|
-
return false;
|
|
130
|
-
}
|
|
151
|
+
return (await resolveFeed(slug, 'HEAD')) !== null;
|
|
131
152
|
}
|
package/src/adapters/workday.js
CHANGED
|
@@ -1,5 +1,8 @@
|
|
|
1
|
-
import { normalize } from '../normalizer.js';
|
|
1
|
+
import { normalize, missingContent } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { prefilterRows } from '../filters.js';
|
|
4
|
+
import { atsFetch } from '../http.js';
|
|
5
|
+
import { orgHost } from '../boards.js';
|
|
3
6
|
|
|
4
7
|
const MAX_DETAIL_FETCHES = 100;
|
|
5
8
|
const LIST_PAGE_SIZE = 20;
|
|
@@ -30,7 +33,9 @@ const MULTI_LOCATION = /^\s*\d+\s+locations?\s*$/;
|
|
|
30
33
|
* detail set.
|
|
31
34
|
*
|
|
32
35
|
* @param {string} slug - normalized company slug (registry routing key)
|
|
33
|
-
* @param {object} [ctx] - { config:{tenant,env,site}, companyName, filterContext }
|
|
36
|
+
* @param {object} [ctx] - { config:{tenant,env,site}, companyName, filterContext, report };
|
|
37
|
+
* report is called once, after hydration, with
|
|
38
|
+
* { listed, prefiltered, hydrated, capped, org_name, org_url } when given
|
|
34
39
|
* @returns {Promise<Array>} Normalized job objects
|
|
35
40
|
*/
|
|
36
41
|
export async function fetchWorkday(slug, ctx = {}) {
|
|
@@ -46,12 +51,25 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
46
51
|
let offset = 0;
|
|
47
52
|
let pages = 0;
|
|
48
53
|
let firstTotal = 0;
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
}
|
|
54
|
+
let listCapped = false;
|
|
55
|
+
while (true) {
|
|
56
|
+
if (pages >= LIST_PAGE_HARD_CAP) {
|
|
57
|
+
listCapped = true;
|
|
58
|
+
break;
|
|
59
|
+
}
|
|
60
|
+
let resp;
|
|
61
|
+
try {
|
|
62
|
+
resp = await atsFetch(`${base}/jobs`, {
|
|
63
|
+
method: 'POST',
|
|
64
|
+
headers: { 'Content-Type': 'application/json' },
|
|
65
|
+
body: JSON.stringify({ appliedFacets: {}, limit: LIST_PAGE_SIZE, offset, searchText: '' }),
|
|
66
|
+
});
|
|
67
|
+
} catch (err) {
|
|
68
|
+
// A 429 or 5xx that outlasted the retries, or a network failure.
|
|
69
|
+
// After the first page, keep the postings already read.
|
|
70
|
+
if (offset === 0) throw err;
|
|
71
|
+
break;
|
|
72
|
+
}
|
|
55
73
|
|
|
56
74
|
if (!resp.ok) {
|
|
57
75
|
if (offset === 0) {
|
|
@@ -74,69 +92,53 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
74
92
|
if (firstTotal > 0 && offset >= firstTotal) break;
|
|
75
93
|
}
|
|
76
94
|
|
|
77
|
-
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost
|
|
78
|
-
//
|
|
79
|
-
//
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
95
|
+
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost, then
|
|
96
|
+
// the detail budget (see prefilterRows). The list carries
|
|
97
|
+
// title/locationsText/postedOn, enough to apply titleFilter, location
|
|
98
|
+
// and recency without descriptions. "2 Locations" says nothing about
|
|
99
|
+
// where: the row stays a candidate through both location filters and
|
|
100
|
+
// the library's pass after hydration decides on the detail's location
|
|
101
|
+
// list (issue #61).
|
|
102
|
+
// NOTE: huge-tenant coverage is intentionally capped for v1. Two caps
|
|
103
|
+
// apply: the list scan above stops at LIST_PAGE_HARD_CAP pages (2000
|
|
104
|
+
// postings, enough for Salesforce's ~1398), and the detail set is cut
|
|
105
|
+
// to MAX_DETAIL_FETCHES here. Proper fix (smart pagination, surfaced
|
|
106
|
+
// truncation) is tracked in #26.
|
|
107
|
+
const { candidates, hydrate } = prefilterRows(postings, fc, {
|
|
108
|
+
title: p => p.title || '',
|
|
109
|
+
location: p => {
|
|
89
110
|
const loc = (p.locationsText || '').toLowerCase();
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
if (typeof fc.postedWithinDays === 'number') {
|
|
104
|
-
candidates = candidates.filter(p => withinDays(p.postedOn, fc.postedWithinDays));
|
|
105
|
-
}
|
|
106
|
-
|
|
107
|
-
// 3. Bound the detail-fetch set.
|
|
108
|
-
// NOTE: huge-tenant coverage is intentionally capped for v1. Two
|
|
109
|
-
// caps apply: the list scan above stops at LIST_PAGE_HARD_CAP pages
|
|
110
|
-
// (2000 postings, enough for Salesforce's ~1398), and the detail set
|
|
111
|
-
// is cut to MAX_DETAIL_FETCHES here. A description `filter` is
|
|
112
|
-
// applied by the library AFTER this returns, so for that case we
|
|
113
|
-
// keep the full backstop instead of truncating tightly to `limit`
|
|
114
|
-
// (which could hydrate jobs that all fail the regex while better
|
|
115
|
-
// matches go unscanned). The library pages with `offset` after this
|
|
116
|
-
// returns, so the budget covers the page plus what precedes it.
|
|
117
|
-
// Proper fix (smart pagination / rate-limited concurrency / surfaced
|
|
118
|
-
// truncation) is tracked in #26, to be designed alongside
|
|
119
|
-
// retry/rate-limit work (#7).
|
|
120
|
-
const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
|
|
121
|
-
const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
|
|
122
|
-
const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
|
|
123
|
-
candidates = candidates.slice(0, cap);
|
|
124
|
-
|
|
125
|
-
// 4. Hydrate descriptions via the per-posting detail endpoint.
|
|
126
|
-
const jobs = await Promise.all(candidates.map(async (p) => {
|
|
111
|
+
return MULTI_LOCATION.test(loc) ? null : loc;
|
|
112
|
+
},
|
|
113
|
+
postedWithin: (p, days) => withinDays(p.postedOn, days),
|
|
114
|
+
max: MAX_DETAIL_FETCHES,
|
|
115
|
+
});
|
|
116
|
+
|
|
117
|
+
// 4. Hydrate descriptions via the per-posting detail endpoint. The detail
|
|
118
|
+
// also carries `hiringOrganization: { name, url }` next to
|
|
119
|
+
// jobPostingInfo; the list does not. Kept per posting in list order so
|
|
120
|
+
// the one reported is the first hydrated posting's, not whichever
|
|
121
|
+
// detail answered first (a tenant can post under several entities).
|
|
122
|
+
const orgs = [];
|
|
123
|
+
const jobs = await Promise.all(hydrate.map(async (p, i) => {
|
|
127
124
|
const externalPath = p.externalPath || ''; // already begins with '/job/...'
|
|
128
125
|
let info = {};
|
|
126
|
+
let content; // set only when the detail could not be read (issue #85)
|
|
129
127
|
try {
|
|
130
128
|
// externalPath already carries the '/job/...' segment, so it is
|
|
131
129
|
// concatenated directly onto the CXS base. Inserting another
|
|
132
130
|
// '/job' here yields '/job/job/...' which Workday rejects (422).
|
|
133
|
-
const dResp = await
|
|
131
|
+
const dResp = await atsFetch(`${base}${externalPath}`);
|
|
134
132
|
if (dResp.ok) {
|
|
135
133
|
const detail = await dResp.json();
|
|
136
134
|
info = detail.jobPostingInfo || {};
|
|
135
|
+
orgs[i] = detail.hiringOrganization || null;
|
|
136
|
+
} else {
|
|
137
|
+
content = missingContent(dResp);
|
|
137
138
|
}
|
|
138
|
-
} catch {
|
|
139
|
-
// detail failed
|
|
139
|
+
} catch (err) {
|
|
140
|
+
// detail failed, retries included: list fields only, marked missing
|
|
141
|
+
content = missingContent(err);
|
|
140
142
|
}
|
|
141
143
|
|
|
142
144
|
return normalize({
|
|
@@ -151,6 +153,7 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
151
153
|
url: `https://${tenant}.${env}.myworkdayjobs.com/${site}${externalPath}`,
|
|
152
154
|
postedAt: parseWorkdayDate(info.startDate) || normalizePostedOn(p.postedOn),
|
|
153
155
|
salary: null, // normalizer extracts from description text
|
|
156
|
+
content,
|
|
154
157
|
metadata: {
|
|
155
158
|
workdayTenant: tenant,
|
|
156
159
|
workdayEnv: env,
|
|
@@ -160,6 +163,20 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
160
163
|
}, 'workday');
|
|
161
164
|
}));
|
|
162
165
|
|
|
166
|
+
// Nothing hydrated (a filter miss, an empty site) means no detail was
|
|
167
|
+
// read, so the org is unknown rather than absent: null, null.
|
|
168
|
+
if (typeof ctx.report === 'function') {
|
|
169
|
+
const org = orgs.find(Boolean) || {};
|
|
170
|
+
ctx.report({
|
|
171
|
+
listed: postings.length,
|
|
172
|
+
prefiltered: candidates.length,
|
|
173
|
+
hydrated: hydrate.length,
|
|
174
|
+
capped: listCapped || hydrate.length < candidates.length,
|
|
175
|
+
org_name: org.name || null,
|
|
176
|
+
org_url: orgHost(org.url),
|
|
177
|
+
});
|
|
178
|
+
}
|
|
179
|
+
|
|
163
180
|
return jobs;
|
|
164
181
|
}
|
|
165
182
|
|
|
@@ -179,19 +196,26 @@ function parseWorkdayRemoteType(remoteType) {
|
|
|
179
196
|
|
|
180
197
|
/**
|
|
181
198
|
* Workday list `postedOn` is a relative string ("Posted Today",
|
|
182
|
-
* "Posted 5 Days Ago", "Posted 30+ Days Ago").
|
|
183
|
-
*
|
|
184
|
-
* the library re-filters authoritatively on the real postedAt after
|
|
185
|
-
* hydration, so a false-keep here is corrected downstream.
|
|
199
|
+
* "Posted 5 Days Ago", "Posted 30+ Days Ago"). The days it names, or null
|
|
200
|
+
* when it names none.
|
|
186
201
|
*/
|
|
187
|
-
function
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
if (/
|
|
191
|
-
if (/yesterday/.test(s)) return days >= 1;
|
|
202
|
+
function daysAgo(postedOn) {
|
|
203
|
+
const s = String(postedOn || '').toLowerCase();
|
|
204
|
+
if (/today/.test(s)) return 0;
|
|
205
|
+
if (/yesterday/.test(s)) return 1;
|
|
192
206
|
const m = s.match(/(\d+)\+?\s*days?\s*ago/);
|
|
193
|
-
|
|
194
|
-
|
|
207
|
+
return m ? parseInt(m[1], 10) : null;
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
/**
|
|
211
|
+
* Decide membership in the last N days WITHOUT a network call.
|
|
212
|
+
* Unparseable -> keep (true); the library re-filters authoritatively on
|
|
213
|
+
* the real postedAt after hydration, so a false-keep here is corrected
|
|
214
|
+
* downstream.
|
|
215
|
+
*/
|
|
216
|
+
function withinDays(postedOn, days) {
|
|
217
|
+
const n = daysAgo(postedOn);
|
|
218
|
+
return n === null || n <= days;
|
|
195
219
|
}
|
|
196
220
|
|
|
197
221
|
/**
|
|
@@ -202,16 +226,8 @@ function normalizePostedOn(v) {
|
|
|
202
226
|
if (!v) return null;
|
|
203
227
|
const direct = new Date(v);
|
|
204
228
|
if (Number.isFinite(direct.getTime())) return direct.toISOString();
|
|
205
|
-
const
|
|
206
|
-
|
|
207
|
-
if (/today/.test(s)) daysAgo = 0;
|
|
208
|
-
else if (/yesterday/.test(s)) daysAgo = 1;
|
|
209
|
-
else {
|
|
210
|
-
const m = s.match(/(\d+)\+?\s*days?\s*ago/);
|
|
211
|
-
if (m) daysAgo = parseInt(m[1], 10);
|
|
212
|
-
}
|
|
213
|
-
if (daysAgo === null) return null;
|
|
214
|
-
return new Date(Date.now() - daysAgo * 86400000).toISOString();
|
|
229
|
+
const n = daysAgo(v);
|
|
230
|
+
return n === null ? null : new Date(Date.now() - n * 86400000).toISOString();
|
|
215
231
|
}
|
|
216
232
|
|
|
217
233
|
/**
|
package/src/boards.js
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The boards[] entry of a fetchJobsDetailed result (issues #58, #60, #87).
|
|
3
|
+
*
|
|
4
|
+
* A board is one (ats, slug) the library fetched, with what came back. The
|
|
5
|
+
* fields fall in three groups: what the registry or the caller said about
|
|
6
|
+
* it (`name`, `site`), what the fetch found (`jobs_found`, `matched`,
|
|
7
|
+
* `scan`), and what the board says about itself (`org_name`, `org_url`).
|
|
8
|
+
* Only the adapters can read the third group, and they hand it over through
|
|
9
|
+
* ctx.report. Both fields are null where the platform exposes nothing, and
|
|
10
|
+
* they are never filled from the slug or the registry name: a slug is an
|
|
11
|
+
* address, not a confirmed identity.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
const BOARD_URLS = {
|
|
15
|
+
greenhouse: (slug) => `https://boards.greenhouse.io/${slug}`,
|
|
16
|
+
lever: (slug) => `https://jobs.lever.co/${slug}`,
|
|
17
|
+
ashby: (slug) => `https://jobs.ashbyhq.com/${slug}`,
|
|
18
|
+
smartrecruiters: (slug) => `https://careers.smartrecruiters.com/${slug}`,
|
|
19
|
+
teamtailor: (slug) => `https://${slug}.teamtailor.com`,
|
|
20
|
+
recruitee: (slug) => `https://${slug}.recruitee.com`,
|
|
21
|
+
workday: (slug, config) => (config ? `https://${config.tenant}.${config.env}.myworkdayjobs.com/${config.site}` : null),
|
|
22
|
+
};
|
|
23
|
+
|
|
24
|
+
// Domains the platforms own. A link there (boards.greenhouse.io,
|
|
25
|
+
// jobs.lever.co, testco.recruitee.com, cisco.wd5.myworkdayjobs.com) says
|
|
26
|
+
// which ATS hosts the board, nothing about whose board it is.
|
|
27
|
+
const ATS_DOMAINS = ['greenhouse.io', 'lever.co', 'ashbyhq.com', 'smartrecruiters.com', 'teamtailor.com', 'recruitee.com', 'myworkdayjobs.com'];
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* The page a person opens to see the board, or null when the ATS is unknown
|
|
31
|
+
* or, for Workday, no {tenant, env, site} is at hand.
|
|
32
|
+
*/
|
|
33
|
+
export function boardUrl(ats, slug, config) {
|
|
34
|
+
const build = BOARD_URLS[ats];
|
|
35
|
+
return build ? build(slug, config) : null;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* The bare hostname a board's link points at ("jobs.example.com"), for
|
|
40
|
+
* org_url. Null when the link is missing or malformed, and null when the
|
|
41
|
+
* host belongs to an ATS, since that carries no signal about the company.
|
|
42
|
+
*/
|
|
43
|
+
export function orgHost(link) {
|
|
44
|
+
let host;
|
|
45
|
+
try {
|
|
46
|
+
host = new URL(link).hostname.toLowerCase();
|
|
47
|
+
} catch {
|
|
48
|
+
return null;
|
|
49
|
+
}
|
|
50
|
+
if (!host || ATS_DOMAINS.some(d => host === d || host.endsWith(`.${d}`))) return null;
|
|
51
|
+
return host;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* @param {object} board
|
|
56
|
+
* @param {string} board.ats
|
|
57
|
+
* @param {string} board.slug - The slug the adapter was called with (canonical casing on a registry hit)
|
|
58
|
+
* @param {string|null} board.name - Registry row name; null for a probe or an override
|
|
59
|
+
* @param {object} [board.config] - Workday {tenant, env, site}, when one was used
|
|
60
|
+
* @param {string|null} board.org_name - The organization name the ATS response states, else null
|
|
61
|
+
* @param {string|null} board.org_url - The careers or company host the board links to (see orgHost), else null
|
|
62
|
+
* @param {number} board.jobs_found - Rows the board listed before any filter: the list count an adapter reported through ctx.report when it filters before hydrating, else the rows it returned
|
|
63
|
+
* @param {number} board.matched - Rows left after filters, before offset and limit
|
|
64
|
+
* @param {object|null} board.scan - The { listed, prefiltered, hydrated, capped } counts the adapter reported through ctx.report, else null
|
|
65
|
+
*/
|
|
66
|
+
export function describeBoard({ ats, slug, name = null, config, org_name = null, org_url = null, jobs_found, matched = 0, scan = null }) {
|
|
67
|
+
return {
|
|
68
|
+
ats,
|
|
69
|
+
slug,
|
|
70
|
+
name,
|
|
71
|
+
site: ats === 'workday' && config ? config.site : null,
|
|
72
|
+
board_url: boardUrl(ats, slug, config),
|
|
73
|
+
org_name,
|
|
74
|
+
org_url,
|
|
75
|
+
jobs_found,
|
|
76
|
+
matched,
|
|
77
|
+
selected: true,
|
|
78
|
+
scan,
|
|
79
|
+
};
|
|
80
|
+
}
|