jd-intel 0.8.3 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +35 -4
- package/package.json +1 -1
- package/src/adapters/ashby.js +69 -86
- package/src/adapters/greenhouse.js +46 -13
- package/src/adapters/lever.js +87 -45
- package/src/adapters/recruitee.js +87 -23
- package/src/adapters/smartrecruiters.js +146 -34
- package/src/adapters/teamtailor.js +57 -39
- package/src/adapters/workday.js +107 -31
- package/src/boards.js +80 -0
- package/src/cli.js +46 -15
- package/src/errors.js +14 -0
- package/src/filters.js +122 -29
- package/src/http.js +184 -0
- package/src/index.js +186 -59
- package/src/normalizer.js +200 -44
- package/src/registry.js +79 -25
|
@@ -1,21 +1,32 @@
|
|
|
1
|
-
import { normalize
|
|
1
|
+
import { normalize } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { atsFetch, probeResult } from '../http.js';
|
|
4
|
+
import { orgHost } from '../boards.js';
|
|
3
5
|
|
|
4
6
|
/**
|
|
5
7
|
* Fetch jobs from a Recruitee career site.
|
|
6
8
|
* Public API, no auth required.
|
|
7
9
|
* Docs: https://docs.recruitee.com/reference/offers
|
|
8
10
|
*
|
|
9
|
-
* Single GET returns every offer
|
|
10
|
-
*
|
|
11
|
-
*
|
|
11
|
+
* Single GET returns every offer inline — no N+1 (unlike SmartRecruiters),
|
|
12
|
+
* no XML (unlike TeamTailor/Personio). The simplest adapter shape in the
|
|
13
|
+
* toolkit.
|
|
14
|
+
*
|
|
15
|
+
* Each offer carries two HTML fields, `description` and `requirements`.
|
|
16
|
+
* Which one holds the role depends on the tenant's template (and sometimes
|
|
17
|
+
* the posting): some keep the duties in `description` and the candidate
|
|
18
|
+
* profile in `requirements`, others put a company intro in `description`
|
|
19
|
+
* and everything else in `requirements`. Neither alone is the posting, so
|
|
20
|
+
* both are joined before normalize() strips them (issue #65).
|
|
12
21
|
*
|
|
13
22
|
* @param {string} slug - Recruitee company subdomain (e.g., 'vandebron')
|
|
23
|
+
* @param {object} [ctx] - { report }; report is called once with
|
|
24
|
+
* { ats, org_name, org_url } when given
|
|
14
25
|
* @returns {Promise<Array>} Normalized job objects
|
|
15
26
|
*/
|
|
16
|
-
export async function fetchRecruitee(slug) {
|
|
27
|
+
export async function fetchRecruitee(slug, ctx = {}) {
|
|
17
28
|
const url = `https://${slug}.recruitee.com/api/offers/`;
|
|
18
|
-
const resp = await
|
|
29
|
+
const resp = await atsFetch(url);
|
|
19
30
|
|
|
20
31
|
if (!resp.ok) {
|
|
21
32
|
if (resp.status === 404) return []; // No Recruitee site for this slug
|
|
@@ -25,18 +36,23 @@ export async function fetchRecruitee(slug) {
|
|
|
25
36
|
const data = await resp.json();
|
|
26
37
|
const offers = data.offers || [];
|
|
27
38
|
|
|
39
|
+
// The response is { offers } only, so the identity lives on the rows:
|
|
40
|
+
// company_name, and careers_url, which sits on the company's own careers
|
|
41
|
+
// domain when the site has one and on {slug}.recruitee.com otherwise.
|
|
42
|
+
if (typeof ctx.report === 'function') {
|
|
43
|
+
ctx.report({
|
|
44
|
+
ats: 'recruitee',
|
|
45
|
+
org_name: offers.find(o => o.company_name)?.company_name || null,
|
|
46
|
+
org_url: orgHost(offers.find(o => o.careers_url)?.careers_url),
|
|
47
|
+
});
|
|
48
|
+
}
|
|
49
|
+
|
|
28
50
|
return offers.map(offer => {
|
|
29
51
|
const place = [offer.city, offer.country].filter(Boolean).join(', ');
|
|
30
52
|
let location = place;
|
|
31
53
|
if (offer.remote) location = place ? `Remote - ${place}` : 'Remote';
|
|
32
54
|
|
|
33
|
-
|
|
34
|
-
if (offer.created_at) {
|
|
35
|
-
// Recruitee returns "2026-05-13 07:38:11 UTC"; coerce to ISO.
|
|
36
|
-
const iso = offer.created_at.replace(' UTC', 'Z').replace(' ', 'T');
|
|
37
|
-
const d = new Date(iso);
|
|
38
|
-
if (!Number.isNaN(d.getTime())) postedAt = d.toISOString();
|
|
39
|
-
}
|
|
55
|
+
const createdAt = toIso(offer.created_at);
|
|
40
56
|
|
|
41
57
|
return normalize({
|
|
42
58
|
companySlug: slug,
|
|
@@ -44,27 +60,75 @@ export async function fetchRecruitee(slug) {
|
|
|
44
60
|
title: offer.title || '',
|
|
45
61
|
department: offer.department || '',
|
|
46
62
|
location,
|
|
47
|
-
|
|
63
|
+
locations: (offer.locations || []).map(l => [l.city, l.country].filter(Boolean).join(', ')),
|
|
64
|
+
workplace: parseRecruiteeWorkplace(offer),
|
|
65
|
+
description: [offer.description, offer.requirements].filter(Boolean).join('\n'),
|
|
48
66
|
url: offer.careers_url || offer.careers_apply_url || '',
|
|
49
|
-
|
|
50
|
-
|
|
67
|
+
// created_at can predate publication by years on long-lived offers,
|
|
68
|
+
// so it is not a posting date. published_at is.
|
|
69
|
+
postedAt: toIso(offer.published_at) || createdAt,
|
|
70
|
+
salary: parseRecruiteeSalary(offer.salary),
|
|
51
71
|
metadata: {
|
|
52
72
|
recruiteeId: offer.guid || offer.id,
|
|
53
73
|
employmentType: offer.employment_type_code || '',
|
|
54
74
|
category: offer.category_code || '',
|
|
75
|
+
createdAt,
|
|
55
76
|
},
|
|
56
77
|
}, 'recruitee');
|
|
57
78
|
});
|
|
58
79
|
}
|
|
59
80
|
|
|
60
81
|
/**
|
|
61
|
-
*
|
|
82
|
+
* Recruitee returns "2026-05-13 07:38:11 UTC"; coerce to ISO.
|
|
83
|
+
*/
|
|
84
|
+
function toIso(ts) {
|
|
85
|
+
if (!ts) return null;
|
|
86
|
+
const d = new Date(ts.replace(' UTC', 'Z').replace(' ', 'T'));
|
|
87
|
+
return Number.isNaN(d.getTime()) ? null : d.toISOString();
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* Recruitee sends three booleans, not one enum. Hybrid wins when remote is
|
|
92
|
+
* also set, and on_site alone is onsite. All false is no signal.
|
|
93
|
+
*/
|
|
94
|
+
function parseRecruiteeWorkplace(offer) {
|
|
95
|
+
if (offer.hybrid) return 'hybrid';
|
|
96
|
+
if (offer.remote) return 'remote';
|
|
97
|
+
if (offer.on_site) return 'onsite';
|
|
98
|
+
return null;
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
const PERIODS = new Set(['year', 'month', 'hour']);
|
|
102
|
+
|
|
103
|
+
/**
|
|
104
|
+
* Recruitee sends `salary` as `{min, max, period, currency}` with string
|
|
105
|
+
* amounts. Offers without pay still carry the object, either all-null or
|
|
106
|
+
* as a "0"/"0" placeholder, so anything without a positive side returns
|
|
107
|
+
* null and normalize() falls back to the posting text.
|
|
108
|
+
*/
|
|
109
|
+
function parseRecruiteeSalary(salary) {
|
|
110
|
+
if (!salary) return null;
|
|
111
|
+
const min = toAmount(salary.min);
|
|
112
|
+
const max = toAmount(salary.max);
|
|
113
|
+
if (min === null && max === null) return null;
|
|
114
|
+
return {
|
|
115
|
+
min,
|
|
116
|
+
max,
|
|
117
|
+
currency: (salary.currency || '').toUpperCase(),
|
|
118
|
+
period: PERIODS.has(salary.period) ? salary.period : null,
|
|
119
|
+
source: 'ats',
|
|
120
|
+
};
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
function toAmount(value) {
|
|
124
|
+
const n = parseFloat(value);
|
|
125
|
+
return Number.isFinite(n) && n > 0 ? n : null;
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
/**
|
|
129
|
+
* Check if a company has a Recruitee career site. See probeResult for the outcomes.
|
|
62
130
|
*/
|
|
63
131
|
export async function hasRecruitee(slug) {
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
return resp.ok;
|
|
67
|
-
} catch {
|
|
68
|
-
return false;
|
|
69
|
-
}
|
|
132
|
+
const resp = await atsFetch(`https://${slug}.recruitee.com/api/offers/`);
|
|
133
|
+
return probeResult(resp, `Recruitee probe for ${slug}`);
|
|
70
134
|
}
|
|
@@ -1,33 +1,45 @@
|
|
|
1
|
-
import { normalize
|
|
1
|
+
import { normalize } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { atsFetch, probeResult } from '../http.js';
|
|
4
|
+
import { makeLocationMatcher } from '../filters.js';
|
|
3
5
|
|
|
4
6
|
const BASE_URL = 'https://api.smartrecruiters.com/v1/companies';
|
|
5
7
|
const PAGE_SIZE = 100;
|
|
8
|
+
const MAX_DETAIL_FETCHES = 100;
|
|
6
9
|
|
|
7
10
|
/**
|
|
8
|
-
* Fetch
|
|
11
|
+
* Fetch postings from a SmartRecruiters company.
|
|
9
12
|
* Public API, no auth required.
|
|
10
13
|
* Docs: https://developers.smartrecruiters.com/reference/postingsget-1
|
|
11
14
|
*
|
|
12
15
|
* Two-step flow (unavoidable N+1):
|
|
13
|
-
* - The postings LIST endpoint omits the job description entirely
|
|
16
|
+
* - The postings LIST endpoint omits the job description entirely,
|
|
17
|
+
* and the structured `compensation` block with it.
|
|
14
18
|
* - jd-intel's contract is "full JD text", so we must fetch each
|
|
15
19
|
* posting's DETAIL endpoint to get jobAd.sections.
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
*
|
|
20
|
+
*
|
|
21
|
+
* The list does carry name, location and releasedDate, so the same
|
|
22
|
+
* pre-filter and detail budget Workday applies run here: list-evaluable
|
|
23
|
+
* filters narrow the candidates, then at most MAX_DETAIL_FETCHES of them
|
|
24
|
+
* are hydrated (see the budget note below). Without a filterContext the
|
|
25
|
+
* cap still holds, so a direct call on a 400-posting tenant reads 100.
|
|
19
26
|
*
|
|
20
27
|
* @param {string} slug - SmartRecruiters company identifier (e.g., 'Visa')
|
|
28
|
+
* @param {object} [ctx] - { filterContext, report }; report is called once
|
|
29
|
+
* with { ats, listed, prefiltered, hydrated, capped, org_name, org_url }
|
|
30
|
+
* when given
|
|
21
31
|
* @returns {Promise<Array>} Normalized job objects
|
|
22
32
|
*/
|
|
23
|
-
export async function fetchSmartrecruiters(slug) {
|
|
33
|
+
export async function fetchSmartrecruiters(slug, ctx = {}) {
|
|
34
|
+
const fc = ctx.filterContext || {};
|
|
35
|
+
|
|
24
36
|
// 1. Page through the postings list.
|
|
25
37
|
const postings = [];
|
|
26
38
|
let offset = 0;
|
|
27
39
|
|
|
28
40
|
while (true) {
|
|
29
41
|
const listUrl = `${BASE_URL}/${slug}/postings?limit=${PAGE_SIZE}&offset=${offset}`;
|
|
30
|
-
const resp = await
|
|
42
|
+
const resp = await atsFetch(listUrl);
|
|
31
43
|
|
|
32
44
|
if (!resp.ok) {
|
|
33
45
|
if (resp.status === 404) return []; // Company not found
|
|
@@ -42,20 +54,89 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
42
54
|
if (content.length === 0 || offset >= (data.totalFound || 0)) break;
|
|
43
55
|
}
|
|
44
56
|
|
|
45
|
-
// 2.
|
|
46
|
-
|
|
57
|
+
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost.
|
|
58
|
+
// The list row carries name, location and releasedDate, and the
|
|
59
|
+
// library re-applies every filter after this returns, so a keep here
|
|
60
|
+
// is never final. The detail adds no location (unlike Workday's
|
|
61
|
+
// additionalLocations), so a row with none follows the library's
|
|
62
|
+
// rule now: out under includes, kept under excludes.
|
|
63
|
+
let candidates = postings;
|
|
64
|
+
|
|
65
|
+
if (fc.titleFilter) {
|
|
66
|
+
const re = new RegExp(fc.titleFilter, 'i');
|
|
67
|
+
candidates = candidates.filter(p => re.test(p.name || ''));
|
|
68
|
+
}
|
|
69
|
+
if (Array.isArray(fc.locationIncludes) && fc.locationIncludes.length > 0) {
|
|
70
|
+
const matchers = fc.locationIncludes.map(makeLocationMatcher);
|
|
71
|
+
candidates = candidates.filter(p => {
|
|
72
|
+
const loc = listLocation(p).location.toLowerCase();
|
|
73
|
+
return matchers.some(m => m(loc));
|
|
74
|
+
});
|
|
75
|
+
}
|
|
76
|
+
if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
|
|
77
|
+
const matchers = fc.locationExcludes.map(makeLocationMatcher);
|
|
78
|
+
candidates = candidates.filter(p => {
|
|
79
|
+
const loc = listLocation(p).location.toLowerCase();
|
|
80
|
+
return !loc || !matchers.some(m => m(loc));
|
|
81
|
+
});
|
|
82
|
+
}
|
|
83
|
+
if (typeof fc.postedWithinDays === 'number') {
|
|
84
|
+
// postedAt comes from releasedDate alone, so the library's rule can
|
|
85
|
+
// run here in full: a missing or unparseable date is out either way.
|
|
86
|
+
const cutoff = Date.now() - fc.postedWithinDays * 86400000;
|
|
87
|
+
candidates = candidates.filter(p => {
|
|
88
|
+
const released = new Date(p.releasedDate || '').getTime();
|
|
89
|
+
return Number.isFinite(released) && released >= cutoff;
|
|
90
|
+
});
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// 3. Bound the detail-fetch set, Workday's reasoning verbatim: a
|
|
94
|
+
// description `filter` is applied by the library AFTER this returns,
|
|
95
|
+
// so that case keeps the full backstop instead of truncating to
|
|
96
|
+
// `limit` (which could hydrate jobs that all fail the regex while
|
|
97
|
+
// better matches go unscanned). The library pages with `offset`
|
|
98
|
+
// after this returns, so the budget covers the page plus what
|
|
99
|
+
// precedes it. Candidates keep list order.
|
|
100
|
+
const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
|
|
101
|
+
const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
|
|
102
|
+
const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
|
|
103
|
+
const hydrate = candidates.slice(0, cap);
|
|
104
|
+
|
|
105
|
+
// Every list row carries company { identifier, name }. Neither the list
|
|
106
|
+
// nor the detail has a company website, and postingUrl is always on
|
|
107
|
+
// jobs.smartrecruiters.com, so org_url stays null (issue #58).
|
|
108
|
+
if (typeof ctx.report === 'function') {
|
|
109
|
+
ctx.report({
|
|
110
|
+
ats: 'smartrecruiters',
|
|
111
|
+
listed: postings.length,
|
|
112
|
+
prefiltered: candidates.length,
|
|
113
|
+
hydrated: hydrate.length,
|
|
114
|
+
capped: hydrate.length < candidates.length,
|
|
115
|
+
org_name: postings.find(p => p.company?.name)?.company.name || null,
|
|
116
|
+
org_url: null,
|
|
117
|
+
});
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
// 4. Fetch detail per candidate for the description. atsFetch's per-host
|
|
121
|
+
// queue keeps this fan-out to 4 requests at a time (a 412-posting
|
|
122
|
+
// tenant measured 53s unbounded), so the cap also holds the detail
|
|
123
|
+
// step to roughly 13s.
|
|
124
|
+
const jobs = await Promise.all(hydrate.map(async (p) => {
|
|
47
125
|
let sections = {};
|
|
48
126
|
let postingUrl = '';
|
|
127
|
+
let salary = null;
|
|
49
128
|
|
|
50
129
|
try {
|
|
51
|
-
const detailResp = await
|
|
130
|
+
const detailResp = await atsFetch(`${BASE_URL}/${slug}/postings/${p.id}`);
|
|
52
131
|
if (detailResp.ok) {
|
|
53
132
|
const detail = await detailResp.json();
|
|
54
133
|
sections = detail.jobAd?.sections || {};
|
|
55
134
|
postingUrl = detail.postingUrl || detail.applyUrl || '';
|
|
135
|
+
salary = parseCompensation(detail.compensation);
|
|
56
136
|
}
|
|
57
137
|
} catch {
|
|
58
|
-
// Detail fetch failed: fall back to list-only
|
|
138
|
+
// Detail fetch failed, retries included: fall back to list-only
|
|
139
|
+
// fields (no description). Reporting this is #85.
|
|
59
140
|
}
|
|
60
141
|
|
|
61
142
|
const description = [
|
|
@@ -64,12 +145,7 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
64
145
|
sections.additionalInformation?.text,
|
|
65
146
|
].filter(Boolean).join('\n\n');
|
|
66
147
|
|
|
67
|
-
const
|
|
68
|
-
const place = loc.fullLocation
|
|
69
|
-
|| [loc.city, loc.region, loc.country].filter(Boolean).join(', ');
|
|
70
|
-
let location = place;
|
|
71
|
-
if (loc.remote) location = `Remote - ${place}`.replace(/ - $/, ' ');
|
|
72
|
-
else if (loc.hybrid) location = `Hybrid - ${place}`.replace(/ - $/, ' ');
|
|
148
|
+
const { location, workplace } = listLocation(p);
|
|
73
149
|
|
|
74
150
|
return normalize({
|
|
75
151
|
companySlug: slug,
|
|
@@ -77,10 +153,11 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
77
153
|
title: p.name || '',
|
|
78
154
|
department: p.department?.label || p.function?.label || '',
|
|
79
155
|
location,
|
|
80
|
-
|
|
156
|
+
workplace,
|
|
157
|
+
description,
|
|
81
158
|
url: postingUrl,
|
|
82
159
|
postedAt: p.releasedDate || null,
|
|
83
|
-
salary
|
|
160
|
+
salary, // null when the detail has no compensation; normalize() then parses text
|
|
84
161
|
metadata: {
|
|
85
162
|
smartRecruitersId: p.id,
|
|
86
163
|
refNumber: p.refNumber || '',
|
|
@@ -95,20 +172,55 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
95
172
|
}
|
|
96
173
|
|
|
97
174
|
/**
|
|
98
|
-
*
|
|
99
|
-
*
|
|
100
|
-
*
|
|
175
|
+
* The location string and workplace type a list row yields. Built once
|
|
176
|
+
* here so the pre-filter matches exactly what the normalized job carries,
|
|
177
|
+
* "Remote - " and "Hybrid - " prefixes included.
|
|
178
|
+
*/
|
|
179
|
+
function listLocation(p) {
|
|
180
|
+
const loc = p.location || {};
|
|
181
|
+
const place = loc.fullLocation
|
|
182
|
+
|| [loc.city, loc.region, loc.country].filter(Boolean).join(', ');
|
|
183
|
+
if (loc.remote) return { location: `Remote - ${place}`.replace(/ - $/, ' '), workplace: 'remote' };
|
|
184
|
+
if (loc.hybrid) return { location: `Hybrid - ${place}`.replace(/ - $/, ' '), workplace: 'hybrid' };
|
|
185
|
+
return { location: place, workplace: null };
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
const PERIODS = { YEARLY: 'year', MONTHLY: 'month', HOURLY: 'hour' };
|
|
189
|
+
|
|
190
|
+
/**
|
|
191
|
+
* Map the detail response's `compensation` to the shared salary shape.
|
|
192
|
+
*
|
|
193
|
+
* SmartRecruiters publishes `{min?, max?, currency, period}`, and both
|
|
194
|
+
* one-sided cases occur (a "max only" cap, a "from" floor), so each bound
|
|
195
|
+
* is passed through as null when absent rather than dropping the whole
|
|
196
|
+
* range. The period is kept as published: a MONTHLY figure is not
|
|
197
|
+
* annualized because tenants occasionally mislabel it (issue #70).
|
|
198
|
+
*/
|
|
199
|
+
function parseCompensation(comp) {
|
|
200
|
+
if (!comp || !comp.currency) return null;
|
|
201
|
+
const min = Number.isFinite(comp.min) ? comp.min : null;
|
|
202
|
+
const max = Number.isFinite(comp.max) ? comp.max : null;
|
|
203
|
+
if (min === null && max === null) return null;
|
|
204
|
+
return {
|
|
205
|
+
min,
|
|
206
|
+
max,
|
|
207
|
+
currency: comp.currency,
|
|
208
|
+
period: PERIODS[comp.period] ?? null,
|
|
209
|
+
source: 'ats',
|
|
210
|
+
};
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
/**
|
|
214
|
+
* Check if a company exists on SmartRecruiters. See probeResult for the
|
|
215
|
+
* outcomes. (HEAD isn't reliably supported on the postings endpoint, so
|
|
216
|
+
* use a minimal GET.)
|
|
101
217
|
*/
|
|
102
218
|
export async function hasSmartrecruiters(slug) {
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
return (data.totalFound || 0) > 0 || (data.content || []).length > 0;
|
|
111
|
-
} catch {
|
|
112
|
-
return false;
|
|
113
|
-
}
|
|
219
|
+
const resp = await atsFetch(`${BASE_URL}/${slug}/postings?limit=1`);
|
|
220
|
+
if (!probeResult(resp, `SmartRecruiters probe for ${slug}`)) return false;
|
|
221
|
+
// SmartRecruiters returns 200 with an empty page (not 404) for unknown
|
|
222
|
+
// companies, so resp.ok alone false-positives on any slug. Confirm at
|
|
223
|
+
// least one real posting exists before claiming a match.
|
|
224
|
+
const data = await resp.json();
|
|
225
|
+
return (data.totalFound || 0) > 0 || (data.content || []).length > 0;
|
|
114
226
|
}
|
|
@@ -1,5 +1,7 @@
|
|
|
1
|
-
import { normalize,
|
|
1
|
+
import { normalize, decodeEntities } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { atsFetch } from '../http.js';
|
|
4
|
+
import { orgHost } from '../boards.js';
|
|
3
5
|
|
|
4
6
|
/**
|
|
5
7
|
* Fetch jobs from a TeamTailor career site via its public RSS feed.
|
|
@@ -16,29 +18,40 @@ import { atsErrorFromStatus } from '../errors.js';
|
|
|
16
18
|
* to a custom domain (e.g. jobs.tibber.com).
|
|
17
19
|
*
|
|
18
20
|
* RSS quirk: descriptions are HTML-entity-encoded inside the XML
|
|
19
|
-
* (`<p>...`). We decode that outer layer to real HTML
|
|
20
|
-
*
|
|
21
|
-
* entities. Decode order matters —
|
|
22
|
-
* double-encoded sequences (`&amp;`)
|
|
21
|
+
* (`<p>...`). We decode that outer layer to real HTML with the
|
|
22
|
+
* shared decodeEntities() and hand the HTML to normalize(), which
|
|
23
|
+
* strips tags and resolves the inner entities. Decode order matters —
|
|
24
|
+
* `&` resolves LAST so double-encoded sequences (`&amp;`)
|
|
25
|
+
* collapse by one layer per pass.
|
|
23
26
|
*
|
|
24
27
|
* @param {string} slug - TeamTailor career-site slug (e.g., 'tibber')
|
|
28
|
+
* @param {object} [ctx] - { report }; report is called once with
|
|
29
|
+
* { ats, org_name, org_url } when given
|
|
25
30
|
* @returns {Promise<Array>} Normalized job objects
|
|
26
31
|
*/
|
|
27
32
|
// Most sites are {slug}.teamtailor.com, but some sit on a regional
|
|
28
|
-
// segment, e.g. crunchbase.na.teamtailor.com. '' is the base host.
|
|
29
|
-
|
|
33
|
+
// segment, e.g. crunchbase.na.teamtailor.com. '' is the base host. There
|
|
34
|
+
// is no reachable eu segment: {slug}.eu.teamtailor.com fails TLS for every
|
|
35
|
+
// slug, known or not, because the wildcard certificate covers one label
|
|
36
|
+
// only (live check 2026-09-27). Probing it was a guaranteed failure that
|
|
37
|
+
// the has() contract would now report as an outage.
|
|
38
|
+
const TT_REGIONS = ['', 'na'];
|
|
39
|
+
|
|
40
|
+
// Feeds send `none`, `hybrid`, `fully` or `onsite`. `none` is no signal.
|
|
41
|
+
const REMOTE_STATUS = { hybrid: 'hybrid', fully: 'remote', onsite: 'onsite' };
|
|
30
42
|
|
|
31
43
|
/**
|
|
32
44
|
* Resolve which TeamTailor host actually serves this slug's feed.
|
|
33
|
-
* Returns the first 200 Response, throws on a non-404 error
|
|
34
|
-
* returns null if no
|
|
45
|
+
* Returns the first 200 Response, throws on a non-404 error (atsFetch
|
|
46
|
+
* throws the 429, 5xx and network cases itself), or returns null if no
|
|
47
|
+
* region has a feed.
|
|
35
48
|
*/
|
|
36
49
|
async function resolveFeed(slug, method = 'GET') {
|
|
37
50
|
for (const region of TT_REGIONS) {
|
|
38
51
|
const host = region
|
|
39
52
|
? `${slug}.${region}.teamtailor.com`
|
|
40
53
|
: `${slug}.teamtailor.com`;
|
|
41
|
-
const resp = await
|
|
54
|
+
const resp = await atsFetch(`https://${host}/jobs.rss`, {
|
|
42
55
|
method,
|
|
43
56
|
redirect: 'follow',
|
|
44
57
|
});
|
|
@@ -51,21 +64,36 @@ async function resolveFeed(slug, method = 'GET') {
|
|
|
51
64
|
return null;
|
|
52
65
|
}
|
|
53
66
|
|
|
54
|
-
export async function fetchTeamtailor(slug) {
|
|
67
|
+
export async function fetchTeamtailor(slug, ctx = {}) {
|
|
55
68
|
const resp = await resolveFeed(slug, 'GET');
|
|
56
69
|
if (!resp) return []; // No TeamTailor site in any known region
|
|
57
70
|
|
|
58
71
|
const xml = await resp.text();
|
|
59
72
|
|
|
60
|
-
const
|
|
61
|
-
|
|
62
|
-
).trim();
|
|
73
|
+
const channelTitle = (xml.match(/<channel>[\s\S]*?<title>([\s\S]*?)<\/title>/)?.[1] || '').trim();
|
|
74
|
+
const company = channelTitle || slug;
|
|
63
75
|
|
|
64
76
|
const items = [...xml.matchAll(/<item>([\s\S]*?)<\/item>/g)].map(m => m[1]);
|
|
65
77
|
|
|
78
|
+
// The channel title is the company as the site names itself. The channel
|
|
79
|
+
// <link> always sits on {slug}.teamtailor.com, but item links follow the
|
|
80
|
+
// site's custom domain when it has one (jobs.tibber.com on a feed served
|
|
81
|
+
// from tibber.teamtailor.com), so the first item's link is the host that
|
|
82
|
+
// can say something; the channel link is the fallback for an empty feed.
|
|
83
|
+
if (typeof ctx.report === 'function') {
|
|
84
|
+
const link = items[0]?.match(/<link>([\s\S]*?)<\/link>/)?.[1]
|
|
85
|
+
|| xml.match(/<channel>[\s\S]*?<link>([\s\S]*?)<\/link>/)?.[1]
|
|
86
|
+
|| '';
|
|
87
|
+
ctx.report({
|
|
88
|
+
ats: 'teamtailor',
|
|
89
|
+
org_name: decodeEntities(channelTitle) || null,
|
|
90
|
+
org_url: orgHost(link.trim()),
|
|
91
|
+
});
|
|
92
|
+
}
|
|
93
|
+
|
|
66
94
|
return items.map(item => {
|
|
67
|
-
const pick = (tag) => {
|
|
68
|
-
const m =
|
|
95
|
+
const pick = (tag, src = item) => {
|
|
96
|
+
const m = src.match(new RegExp(`<${tag}[^>]*>([\\s\\S]*?)</${tag}>`));
|
|
69
97
|
return m ? m[1].trim() : '';
|
|
70
98
|
};
|
|
71
99
|
|
|
@@ -78,6 +106,12 @@ export async function fetchTeamtailor(slug) {
|
|
|
78
106
|
const country = decodeEntities(pick('tt:country'));
|
|
79
107
|
const remoteStatus = decodeEntities(pick('remoteStatus'));
|
|
80
108
|
|
|
109
|
+
// One <tt:location> per office the posting is open in, read the same
|
|
110
|
+
// way as the primary above so the entries line up.
|
|
111
|
+
const locations = [...item.matchAll(/<tt:location>([\s\S]*?)<\/tt:location>/g)].map(m =>
|
|
112
|
+
[decodeEntities(pick('tt:city', m[1])), decodeEntities(pick('tt:country', m[1]))].filter(Boolean).join(', ')
|
|
113
|
+
);
|
|
114
|
+
|
|
81
115
|
let location = [city, country].filter(Boolean).join(', ');
|
|
82
116
|
if (/remote/i.test(remoteStatus)) {
|
|
83
117
|
location = location ? `Remote - ${location}` : 'Remote';
|
|
@@ -95,7 +129,9 @@ export async function fetchTeamtailor(slug) {
|
|
|
95
129
|
title,
|
|
96
130
|
department,
|
|
97
131
|
location,
|
|
98
|
-
|
|
132
|
+
locations,
|
|
133
|
+
workplace: REMOTE_STATUS[remoteStatus.toLowerCase()] || null,
|
|
134
|
+
description: decodeEntities(pick('description')),
|
|
99
135
|
url: link,
|
|
100
136
|
postedAt,
|
|
101
137
|
salary: null, // No structured salary; normalizer parses from text
|
|
@@ -108,28 +144,10 @@ export async function fetchTeamtailor(slug) {
|
|
|
108
144
|
}
|
|
109
145
|
|
|
110
146
|
/**
|
|
111
|
-
*
|
|
112
|
-
*
|
|
113
|
-
|
|
114
|
-
function decodeEntities(s) {
|
|
115
|
-
if (!s) return '';
|
|
116
|
-
return s
|
|
117
|
-
.replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1')
|
|
118
|
-
.replace(/</g, '<')
|
|
119
|
-
.replace(/>/g, '>')
|
|
120
|
-
.replace(/"/g, '"')
|
|
121
|
-
.replace(/'/g, "'")
|
|
122
|
-
.replace(/'/g, "'")
|
|
123
|
-
.replace(/&/g, '&');
|
|
124
|
-
}
|
|
125
|
-
|
|
126
|
-
/**
|
|
127
|
-
* Check if a company has a TeamTailor career site.
|
|
147
|
+
* Check if a company has a TeamTailor career site: true when a regional
|
|
148
|
+
* host serves the feed, false when every host answers 404, and the
|
|
149
|
+
* AtsError from resolveFeed for anything else.
|
|
128
150
|
*/
|
|
129
151
|
export async function hasTeamtailor(slug) {
|
|
130
|
-
|
|
131
|
-
return (await resolveFeed(slug, 'HEAD')) !== null;
|
|
132
|
-
} catch {
|
|
133
|
-
return false;
|
|
134
|
-
}
|
|
152
|
+
return (await resolveFeed(slug, 'HEAD')) !== null;
|
|
135
153
|
}
|