jd-intel 0.8.2 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +22 -5
- package/package.json +1 -1
- package/registry/ashby.json +154 -118
- package/registry/greenhouse.json +60 -3
- package/registry/lever.json +78 -68
- package/registry/recruitee.json +31 -1
- package/registry/smartrecruiters.json +6 -1
- package/registry/teamtailor.json +36 -1
- package/registry/workday.json +242 -197
- package/src/adapters/ashby.js +52 -17
- package/src/adapters/greenhouse.js +27 -4
- package/src/adapters/lever.js +71 -37
- package/src/adapters/recruitee.js +67 -14
- package/src/adapters/smartrecruiters.js +41 -6
- package/src/adapters/teamtailor.js +20 -24
- package/src/adapters/workday.js +50 -15
- package/src/cli.js +34 -7
- package/src/filters.js +65 -11
- package/src/index.js +25 -9
- package/src/normalizer.js +200 -44
package/src/adapters/ashby.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize } from '../normalizer.js';
|
|
1
|
+
import { normalize, extractSalaryFromText } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
|
|
4
4
|
const API_URL = 'https://jobs.ashbyhq.com/api/non-user-graphql';
|
|
@@ -36,23 +36,33 @@ async function fetchAshbyRest(slug) {
|
|
|
36
36
|
const jobs = data.jobs || [];
|
|
37
37
|
|
|
38
38
|
return jobs.map(job => {
|
|
39
|
-
const
|
|
39
|
+
const comp = job.compensation || {};
|
|
40
40
|
|
|
41
41
|
return normalize({
|
|
42
42
|
companySlug: slug,
|
|
43
43
|
company: data.organizationName || slug,
|
|
44
44
|
title: job.title || '',
|
|
45
|
-
department: job.
|
|
45
|
+
department: job.department || '',
|
|
46
46
|
location: job.location || '',
|
|
47
|
+
locations: (job.secondaryLocations || []).map(l => l?.location || ''),
|
|
48
|
+
workplace: parseAshbyWorkplace(job),
|
|
47
49
|
description: job.descriptionHtml || job.descriptionPlain || '',
|
|
48
50
|
url: `https://jobs.ashbyhq.com/${slug}/${job.id}`,
|
|
49
51
|
postedAt: job.publishedAt || null,
|
|
50
|
-
salary,
|
|
52
|
+
salary: parseAshbyCompensation(comp),
|
|
51
53
|
metadata: {
|
|
52
54
|
ashbyId: job.id,
|
|
53
55
|
employmentType: job.employmentType || '',
|
|
54
56
|
isRemote: job.isRemote || false,
|
|
55
|
-
team: job.
|
|
57
|
+
team: job.team || '',
|
|
58
|
+
// The rendered summaries keep what min/max drop: "Offers Equity",
|
|
59
|
+
// "Multiple Ranges", and per-location tiers labelled OTE.
|
|
60
|
+
compensationSummary: comp.compensationTierSummary || '',
|
|
61
|
+
compensationTiers: (comp.compensationTiers || []).map(tier => ({
|
|
62
|
+
title: tier.title || '',
|
|
63
|
+
summary: tier.tierSummary || '',
|
|
64
|
+
additionalInformation: tier.additionalInformation || '',
|
|
65
|
+
})),
|
|
56
66
|
},
|
|
57
67
|
}, 'ashby');
|
|
58
68
|
});
|
|
@@ -110,22 +120,47 @@ async function fetchAshbyGraphQL(slug) {
|
|
|
110
120
|
}, 'ashby'));
|
|
111
121
|
}
|
|
112
122
|
|
|
123
|
+
const WORKPLACE_TYPES = { remote: 'remote', hybrid: 'hybrid', onsite: 'onsite' };
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* `workplaceType` is 'Remote', 'Hybrid' or 'OnSite'. `isRemote` is the
|
|
127
|
+
* older flag and can only say remote, so it is the fallback when the type
|
|
128
|
+
* is absent. false means nothing: the role may be hybrid or onsite.
|
|
129
|
+
*/
|
|
130
|
+
function parseAshbyWorkplace(job) {
|
|
131
|
+
const type = WORKPLACE_TYPES[String(job.workplaceType || '').toLowerCase()];
|
|
132
|
+
if (type) return type;
|
|
133
|
+
return job.isRemote === true ? 'remote' : null;
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
const INTERVAL_PERIOD = { '1 YEAR': 'year', '1 MONTH': 'month', '1 HOUR': 'hour' };
|
|
137
|
+
|
|
138
|
+
/**
|
|
139
|
+
* Read pay from Ashby's `compensation` object (issue #67).
|
|
140
|
+
*
|
|
141
|
+
* `summaryComponents` carries one structured entry per component type
|
|
142
|
+
* (Salary, Bonus, Commission, Equity); the Salary entry spans every tier.
|
|
143
|
+
* `scrapeableCompensationSalarySummary` and `compensationTierSummary` are
|
|
144
|
+
* the rendered strings. A board that publishes no pay still sends the
|
|
145
|
+
* object, with null summaries and empty arrays, so a miss here has to
|
|
146
|
+
* return null for the normalizer's text fallback to run.
|
|
147
|
+
*/
|
|
113
148
|
function parseAshbyCompensation(comp) {
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
if (typeof comp === 'string') {
|
|
117
|
-
const match = comp.match(/\$?([\d,]+)\s*[-–]\s*\$?([\d,]+)/);
|
|
118
|
-
if (!match) return null;
|
|
149
|
+
const salary = (comp.summaryComponents || []).find(c => c.compensationType === 'Salary');
|
|
150
|
+
if (salary && (salary.minValue != null || salary.maxValue != null)) {
|
|
119
151
|
return {
|
|
120
|
-
min:
|
|
121
|
-
max:
|
|
122
|
-
currency: 'USD',
|
|
152
|
+
min: salary.minValue ?? null,
|
|
153
|
+
max: salary.maxValue ?? null,
|
|
154
|
+
currency: salary.currencyCode || 'USD',
|
|
155
|
+
period: INTERVAL_PERIOD[salary.interval] || null,
|
|
156
|
+
source: 'ats',
|
|
123
157
|
};
|
|
124
158
|
}
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
159
|
+
// The summaries are still the ATS's own compensation field, so a range
|
|
160
|
+
// read out of one counts as source 'ats'.
|
|
161
|
+
const parsed = extractSalaryFromText(comp.scrapeableCompensationSalarySummary)
|
|
162
|
+
|| extractSalaryFromText(comp.compensationTierSummary);
|
|
163
|
+
return parsed ? { ...parsed, source: 'ats' } : null;
|
|
129
164
|
}
|
|
130
165
|
|
|
131
166
|
export async function hasAshby(slug) {
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize,
|
|
1
|
+
import { normalize, decodeEntities } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
|
|
4
4
|
const BASE_URL = 'https://boards-api.greenhouse.io/v1/boards';
|
|
@@ -29,19 +29,42 @@ export async function fetchGreenhouse(slug) {
|
|
|
29
29
|
title: job.title || '',
|
|
30
30
|
department: job.departments?.[0]?.name || '',
|
|
31
31
|
location: job.location?.name || '',
|
|
32
|
-
|
|
32
|
+
workplace: parseGreenhouseWorkplace(job.metadata),
|
|
33
|
+
// `content` arrives HTML-escaped (`<p>`). Decode that outer layer
|
|
34
|
+
// once so normalize() sees real tags; it strips and decodes the rest.
|
|
35
|
+
description: decodeEntities(job.content || ''),
|
|
33
36
|
url: job.absolute_url || '',
|
|
34
|
-
|
|
35
|
-
|
|
37
|
+
// updated_at is an edit time that many boards bulk-refresh, so it is not
|
|
38
|
+
// a posting date. first_published is. Fallback covers boards without it (#69).
|
|
39
|
+
postedAt: job.first_published || job.updated_at || null,
|
|
40
|
+
salary: null, // list endpoint has no structured pay; normalizer parses the pay transparency text
|
|
36
41
|
metadata: {
|
|
37
42
|
greenhouseId: job.id,
|
|
38
43
|
internal_job_id: job.internal_job_id,
|
|
39
44
|
departments: job.departments?.map(d => d.name) || [],
|
|
40
45
|
offices: job.offices?.map(o => o.name) || [],
|
|
46
|
+
updatedAt: job.updated_at,
|
|
41
47
|
},
|
|
42
48
|
}, 'greenhouse'));
|
|
43
49
|
}
|
|
44
50
|
|
|
51
|
+
/**
|
|
52
|
+
* Greenhouse has no native workplace field. Boards that track it define a
|
|
53
|
+
* custom field ("Location Type", "Workplace Type") that arrives in the
|
|
54
|
+
* job's `metadata[]`, with `value` a string for single-select fields and
|
|
55
|
+
* an array for multi-select. Values seen: On-Site, Hybrid (Travel-Required),
|
|
56
|
+
* Remote. Anything else is no signal and the location string decides.
|
|
57
|
+
*/
|
|
58
|
+
function parseGreenhouseWorkplace(metadata) {
|
|
59
|
+
const field = (metadata || []).find(m => /location type|workplace type/i.test(m?.name || ''));
|
|
60
|
+
if (!field) return null;
|
|
61
|
+
const value = [].concat(field.value ?? []).join(' ').toLowerCase();
|
|
62
|
+
if (/remote/.test(value)) return 'remote';
|
|
63
|
+
if (/hybrid/.test(value)) return 'hybrid';
|
|
64
|
+
if (/on-?site/.test(value)) return 'onsite';
|
|
65
|
+
return null;
|
|
66
|
+
}
|
|
67
|
+
|
|
45
68
|
/**
|
|
46
69
|
* Check if a company has a Greenhouse board.
|
|
47
70
|
*/
|
package/src/adapters/lever.js
CHANGED
|
@@ -1,11 +1,16 @@
|
|
|
1
|
-
import { normalize,
|
|
1
|
+
import { normalize, extractSalaryFromText } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
|
|
4
4
|
const BASE_URL = 'https://api.lever.co/v0/postings';
|
|
5
5
|
|
|
6
|
+
const PERIODS = { 'per-year-salary': 'year', 'per-month-salary': 'month', 'per-hour-wage': 'hour' };
|
|
7
|
+
// Lever's workplaceType is one of these or 'unspecified'.
|
|
8
|
+
const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
|
|
9
|
+
|
|
6
10
|
/**
|
|
7
11
|
* Fetch all jobs from a Lever job board.
|
|
8
12
|
* Public API, no auth required.
|
|
13
|
+
* Docs: https://github.com/lever/postings-api
|
|
9
14
|
*
|
|
10
15
|
* @param {string} slug - Company slug (e.g., 'stripe', 'figma')
|
|
11
16
|
* @returns {Promise<Array>} Normalized job objects
|
|
@@ -22,30 +27,72 @@ export async function fetchLever(slug) {
|
|
|
22
27
|
const jobs = await resp.json();
|
|
23
28
|
if (!Array.isArray(jobs)) return [];
|
|
24
29
|
|
|
25
|
-
return jobs.map(job => {
|
|
26
|
-
|
|
30
|
+
return jobs.map(job => normalize({
|
|
31
|
+
companySlug: slug,
|
|
32
|
+
// Lever's API doesn't return the company name at the board or job level,
|
|
33
|
+
// so the slug is the honest fallback. `categories.team` is the team within
|
|
34
|
+
// the company ("Payments Platform"), not the company itself.
|
|
35
|
+
company: titleCaseSlug(slug),
|
|
36
|
+
title: job.text || '',
|
|
37
|
+
department: job.categories?.department || job.categories?.team || '',
|
|
38
|
+
location: job.categories?.location || '',
|
|
39
|
+
locations: job.categories?.allLocations || [],
|
|
40
|
+
workplace: WORKPLACE_TYPES.has(job.workplaceType) ? job.workplaceType : null,
|
|
41
|
+
description: buildDescription(job),
|
|
42
|
+
url: job.hostedUrl || '',
|
|
43
|
+
postedAt: job.createdAt ? new Date(job.createdAt).toISOString() : null,
|
|
44
|
+
salary: parseLeverSalary(job.salaryRange, job.text),
|
|
45
|
+
metadata: {
|
|
46
|
+
leverId: job.id,
|
|
47
|
+
team: job.categories?.team || '',
|
|
48
|
+
commitment: job.categories?.commitment || '', // Full-time, Part-time, etc.
|
|
49
|
+
workplaceType: job.workplaceType || '',
|
|
50
|
+
salaryDescription: job.salaryDescriptionPlain || '',
|
|
51
|
+
},
|
|
52
|
+
}, 'lever'));
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Lever splits a posting across `description` (company intro plus overview),
|
|
57
|
+
* `lists` (one `{text, content}` per section: responsibilities, requirements,
|
|
58
|
+
* location details) and `additional` (benefits, EEO). Only the first used to
|
|
59
|
+
* reach the description, so requirements were invisible to filters and to
|
|
60
|
+
* the assistant (issue #64). Reassemble the whole posting as HTML and let
|
|
61
|
+
* normalize() render the headings and bullets.
|
|
62
|
+
*/
|
|
63
|
+
function buildDescription(job) {
|
|
64
|
+
const parts = [job.description || job.descriptionPlain || ''];
|
|
65
|
+
for (const list of job.lists || []) {
|
|
66
|
+
const heading = (list.text || '').trim();
|
|
67
|
+
parts.push((heading ? `<h3>${escapeHtml(heading)}</h3>` : '') + (list.content || ''));
|
|
68
|
+
}
|
|
69
|
+
parts.push(job.additional || job.additionalPlain || '');
|
|
70
|
+
return parts.filter(Boolean).join('\n');
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
// `lists[].text` is plain text ("Skills & Experience"). Escaped, normalize()
|
|
74
|
+
// decodes it back; raw, a stray `<` would be stripped as a tag.
|
|
75
|
+
function escapeHtml(s) {
|
|
76
|
+
return s.replace(/&/g, '&').replace(/</g, '<').replace(/>/g, '>');
|
|
77
|
+
}
|
|
27
78
|
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
workplaceType: job.workplaceType || '',
|
|
46
|
-
},
|
|
47
|
-
}, 'lever');
|
|
48
|
-
});
|
|
79
|
+
/**
|
|
80
|
+
* Lever publishes `salaryRange: {min, max, currency, interval}` on boards
|
|
81
|
+
* that state pay. Boards that don't sometimes put the range in the title.
|
|
82
|
+
*/
|
|
83
|
+
function parseLeverSalary(range, title) {
|
|
84
|
+
const min = range?.min || null;
|
|
85
|
+
const max = range?.max || null;
|
|
86
|
+
if (min || max) {
|
|
87
|
+
return {
|
|
88
|
+
min,
|
|
89
|
+
max,
|
|
90
|
+
currency: range.currency || 'USD',
|
|
91
|
+
period: PERIODS[range.interval] || null,
|
|
92
|
+
source: 'ats',
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
return extractSalaryFromText(title || '');
|
|
49
96
|
}
|
|
50
97
|
|
|
51
98
|
function titleCaseSlug(slug) {
|
|
@@ -55,19 +102,6 @@ function titleCaseSlug(slug) {
|
|
|
55
102
|
return slug.charAt(0).toUpperCase() + slug.slice(1);
|
|
56
103
|
}
|
|
57
104
|
|
|
58
|
-
function parseLeverSalary(commitment, title) {
|
|
59
|
-
// Lever doesn't have a salary field, but sometimes it's in the title
|
|
60
|
-
const match = (title || '').match(/\$[\d,]+\s*[-–]\s*\$[\d,]+/);
|
|
61
|
-
if (!match) return null;
|
|
62
|
-
const nums = match[0].match(/[\d,]+/g);
|
|
63
|
-
if (!nums || nums.length < 2) return null;
|
|
64
|
-
return {
|
|
65
|
-
min: parseInt(nums[0].replace(/,/g, '')),
|
|
66
|
-
max: parseInt(nums[1].replace(/,/g, '')),
|
|
67
|
-
currency: 'USD',
|
|
68
|
-
};
|
|
69
|
-
}
|
|
70
|
-
|
|
71
105
|
export async function hasLever(slug) {
|
|
72
106
|
try {
|
|
73
107
|
const resp = await fetch(`${BASE_URL}/${slug}?mode=json`, { method: 'HEAD' });
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize
|
|
1
|
+
import { normalize } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
@@ -6,9 +6,16 @@ import { atsErrorFromStatus } from '../errors.js';
|
|
|
6
6
|
* Public API, no auth required.
|
|
7
7
|
* Docs: https://docs.recruitee.com/reference/offers
|
|
8
8
|
*
|
|
9
|
-
* Single GET returns every offer
|
|
10
|
-
*
|
|
11
|
-
*
|
|
9
|
+
* Single GET returns every offer inline — no N+1 (unlike SmartRecruiters),
|
|
10
|
+
* no XML (unlike TeamTailor/Personio). The simplest adapter shape in the
|
|
11
|
+
* toolkit.
|
|
12
|
+
*
|
|
13
|
+
* Each offer carries two HTML fields, `description` and `requirements`.
|
|
14
|
+
* Which one holds the role depends on the tenant's template (and sometimes
|
|
15
|
+
* the posting): some keep the duties in `description` and the candidate
|
|
16
|
+
* profile in `requirements`, others put a company intro in `description`
|
|
17
|
+
* and everything else in `requirements`. Neither alone is the posting, so
|
|
18
|
+
* both are joined before normalize() strips them (issue #65).
|
|
12
19
|
*
|
|
13
20
|
* @param {string} slug - Recruitee company subdomain (e.g., 'vandebron')
|
|
14
21
|
* @returns {Promise<Array>} Normalized job objects
|
|
@@ -30,13 +37,7 @@ export async function fetchRecruitee(slug) {
|
|
|
30
37
|
let location = place;
|
|
31
38
|
if (offer.remote) location = place ? `Remote - ${place}` : 'Remote';
|
|
32
39
|
|
|
33
|
-
|
|
34
|
-
if (offer.created_at) {
|
|
35
|
-
// Recruitee returns "2026-05-13 07:38:11 UTC"; coerce to ISO.
|
|
36
|
-
const iso = offer.created_at.replace(' UTC', 'Z').replace(' ', 'T');
|
|
37
|
-
const d = new Date(iso);
|
|
38
|
-
if (!Number.isNaN(d.getTime())) postedAt = d.toISOString();
|
|
39
|
-
}
|
|
40
|
+
const createdAt = toIso(offer.created_at);
|
|
40
41
|
|
|
41
42
|
return normalize({
|
|
42
43
|
companySlug: slug,
|
|
@@ -44,19 +45,71 @@ export async function fetchRecruitee(slug) {
|
|
|
44
45
|
title: offer.title || '',
|
|
45
46
|
department: offer.department || '',
|
|
46
47
|
location,
|
|
47
|
-
|
|
48
|
+
locations: (offer.locations || []).map(l => [l.city, l.country].filter(Boolean).join(', ')),
|
|
49
|
+
workplace: parseRecruiteeWorkplace(offer),
|
|
50
|
+
description: [offer.description, offer.requirements].filter(Boolean).join('\n'),
|
|
48
51
|
url: offer.careers_url || offer.careers_apply_url || '',
|
|
49
|
-
|
|
50
|
-
|
|
52
|
+
// created_at can predate publication by years on long-lived offers,
|
|
53
|
+
// so it is not a posting date. published_at is.
|
|
54
|
+
postedAt: toIso(offer.published_at) || createdAt,
|
|
55
|
+
salary: parseRecruiteeSalary(offer.salary),
|
|
51
56
|
metadata: {
|
|
52
57
|
recruiteeId: offer.guid || offer.id,
|
|
53
58
|
employmentType: offer.employment_type_code || '',
|
|
54
59
|
category: offer.category_code || '',
|
|
60
|
+
createdAt,
|
|
55
61
|
},
|
|
56
62
|
}, 'recruitee');
|
|
57
63
|
});
|
|
58
64
|
}
|
|
59
65
|
|
|
66
|
+
/**
|
|
67
|
+
* Recruitee returns "2026-05-13 07:38:11 UTC"; coerce to ISO.
|
|
68
|
+
*/
|
|
69
|
+
function toIso(ts) {
|
|
70
|
+
if (!ts) return null;
|
|
71
|
+
const d = new Date(ts.replace(' UTC', 'Z').replace(' ', 'T'));
|
|
72
|
+
return Number.isNaN(d.getTime()) ? null : d.toISOString();
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Recruitee sends three booleans, not one enum. Hybrid wins when remote is
|
|
77
|
+
* also set, and on_site alone is onsite. All false is no signal.
|
|
78
|
+
*/
|
|
79
|
+
function parseRecruiteeWorkplace(offer) {
|
|
80
|
+
if (offer.hybrid) return 'hybrid';
|
|
81
|
+
if (offer.remote) return 'remote';
|
|
82
|
+
if (offer.on_site) return 'onsite';
|
|
83
|
+
return null;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
const PERIODS = new Set(['year', 'month', 'hour']);
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* Recruitee sends `salary` as `{min, max, period, currency}` with string
|
|
90
|
+
* amounts. Offers without pay still carry the object, either all-null or
|
|
91
|
+
* as a "0"/"0" placeholder, so anything without a positive side returns
|
|
92
|
+
* null and normalize() falls back to the posting text.
|
|
93
|
+
*/
|
|
94
|
+
function parseRecruiteeSalary(salary) {
|
|
95
|
+
if (!salary) return null;
|
|
96
|
+
const min = toAmount(salary.min);
|
|
97
|
+
const max = toAmount(salary.max);
|
|
98
|
+
if (min === null && max === null) return null;
|
|
99
|
+
return {
|
|
100
|
+
min,
|
|
101
|
+
max,
|
|
102
|
+
currency: (salary.currency || '').toUpperCase(),
|
|
103
|
+
period: PERIODS.has(salary.period) ? salary.period : null,
|
|
104
|
+
source: 'ats',
|
|
105
|
+
};
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
function toAmount(value) {
|
|
109
|
+
const n = parseFloat(value);
|
|
110
|
+
return Number.isFinite(n) && n > 0 ? n : null;
|
|
111
|
+
}
|
|
112
|
+
|
|
60
113
|
/**
|
|
61
114
|
* Check if a company has a Recruitee career site.
|
|
62
115
|
*/
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize
|
|
1
|
+
import { normalize } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
|
|
4
4
|
const BASE_URL = 'https://api.smartrecruiters.com/v1/companies';
|
|
@@ -10,7 +10,8 @@ const PAGE_SIZE = 100;
|
|
|
10
10
|
* Docs: https://developers.smartrecruiters.com/reference/postingsget-1
|
|
11
11
|
*
|
|
12
12
|
* Two-step flow (unavoidable N+1):
|
|
13
|
-
* - The postings LIST endpoint omits the job description entirely
|
|
13
|
+
* - The postings LIST endpoint omits the job description entirely,
|
|
14
|
+
* and the structured `compensation` block with it.
|
|
14
15
|
* - jd-intel's contract is "full JD text", so we must fetch each
|
|
15
16
|
* posting's DETAIL endpoint to get jobAd.sections.
|
|
16
17
|
* Large enterprise tenants with hundreds of openings will therefore be
|
|
@@ -46,6 +47,7 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
46
47
|
const jobs = await Promise.all(postings.map(async (p) => {
|
|
47
48
|
let sections = {};
|
|
48
49
|
let postingUrl = '';
|
|
50
|
+
let salary = null;
|
|
49
51
|
|
|
50
52
|
try {
|
|
51
53
|
const detailResp = await fetch(`${BASE_URL}/${slug}/postings/${p.id}`);
|
|
@@ -53,6 +55,7 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
53
55
|
const detail = await detailResp.json();
|
|
54
56
|
sections = detail.jobAd?.sections || {};
|
|
55
57
|
postingUrl = detail.postingUrl || detail.applyUrl || '';
|
|
58
|
+
salary = parseCompensation(detail.compensation);
|
|
56
59
|
}
|
|
57
60
|
} catch {
|
|
58
61
|
// Detail fetch failed: fall back to list-only fields (no description).
|
|
@@ -68,8 +71,14 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
68
71
|
const place = loc.fullLocation
|
|
69
72
|
|| [loc.city, loc.region, loc.country].filter(Boolean).join(', ');
|
|
70
73
|
let location = place;
|
|
71
|
-
|
|
72
|
-
|
|
74
|
+
let workplace = null;
|
|
75
|
+
if (loc.remote) {
|
|
76
|
+
location = `Remote - ${place}`.replace(/ - $/, ' ');
|
|
77
|
+
workplace = 'remote';
|
|
78
|
+
} else if (loc.hybrid) {
|
|
79
|
+
location = `Hybrid - ${place}`.replace(/ - $/, ' ');
|
|
80
|
+
workplace = 'hybrid';
|
|
81
|
+
}
|
|
73
82
|
|
|
74
83
|
return normalize({
|
|
75
84
|
companySlug: slug,
|
|
@@ -77,10 +86,11 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
77
86
|
title: p.name || '',
|
|
78
87
|
department: p.department?.label || p.function?.label || '',
|
|
79
88
|
location,
|
|
80
|
-
|
|
89
|
+
workplace,
|
|
90
|
+
description,
|
|
81
91
|
url: postingUrl,
|
|
82
92
|
postedAt: p.releasedDate || null,
|
|
83
|
-
salary
|
|
93
|
+
salary, // null when the detail has no compensation; normalize() then parses text
|
|
84
94
|
metadata: {
|
|
85
95
|
smartRecruitersId: p.id,
|
|
86
96
|
refNumber: p.refNumber || '',
|
|
@@ -94,6 +104,31 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
94
104
|
return jobs;
|
|
95
105
|
}
|
|
96
106
|
|
|
107
|
+
const PERIODS = { YEARLY: 'year', MONTHLY: 'month', HOURLY: 'hour' };
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* Map the detail response's `compensation` to the shared salary shape.
|
|
111
|
+
*
|
|
112
|
+
* SmartRecruiters publishes `{min?, max?, currency, period}`, and both
|
|
113
|
+
* one-sided cases occur (a "max only" cap, a "from" floor), so each bound
|
|
114
|
+
* is passed through as null when absent rather than dropping the whole
|
|
115
|
+
* range. The period is kept as published: a MONTHLY figure is not
|
|
116
|
+
* annualized because tenants occasionally mislabel it (issue #70).
|
|
117
|
+
*/
|
|
118
|
+
function parseCompensation(comp) {
|
|
119
|
+
if (!comp || !comp.currency) return null;
|
|
120
|
+
const min = Number.isFinite(comp.min) ? comp.min : null;
|
|
121
|
+
const max = Number.isFinite(comp.max) ? comp.max : null;
|
|
122
|
+
if (min === null && max === null) return null;
|
|
123
|
+
return {
|
|
124
|
+
min,
|
|
125
|
+
max,
|
|
126
|
+
currency: comp.currency,
|
|
127
|
+
period: PERIODS[comp.period] ?? null,
|
|
128
|
+
source: 'ats',
|
|
129
|
+
};
|
|
130
|
+
}
|
|
131
|
+
|
|
97
132
|
/**
|
|
98
133
|
* Check if a company exists on SmartRecruiters.
|
|
99
134
|
* (HEAD isn't reliably supported on the postings endpoint, so use a
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize,
|
|
1
|
+
import { normalize, decodeEntities } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
@@ -16,10 +16,11 @@ import { atsErrorFromStatus } from '../errors.js';
|
|
|
16
16
|
* to a custom domain (e.g. jobs.tibber.com).
|
|
17
17
|
*
|
|
18
18
|
* RSS quirk: descriptions are HTML-entity-encoded inside the XML
|
|
19
|
-
* (`<p>...`). We decode that outer layer to real HTML
|
|
20
|
-
*
|
|
21
|
-
* entities. Decode order matters —
|
|
22
|
-
* double-encoded sequences (`&amp;`)
|
|
19
|
+
* (`<p>...`). We decode that outer layer to real HTML with the
|
|
20
|
+
* shared decodeEntities() and hand the HTML to normalize(), which
|
|
21
|
+
* strips tags and resolves the inner entities. Decode order matters —
|
|
22
|
+
* `&` resolves LAST so double-encoded sequences (`&amp;`)
|
|
23
|
+
* collapse by one layer per pass.
|
|
23
24
|
*
|
|
24
25
|
* @param {string} slug - TeamTailor career-site slug (e.g., 'tibber')
|
|
25
26
|
* @returns {Promise<Array>} Normalized job objects
|
|
@@ -28,6 +29,9 @@ import { atsErrorFromStatus } from '../errors.js';
|
|
|
28
29
|
// segment, e.g. crunchbase.na.teamtailor.com. '' is the base host.
|
|
29
30
|
const TT_REGIONS = ['', 'na', 'eu'];
|
|
30
31
|
|
|
32
|
+
// Feeds send `none`, `hybrid`, `fully` or `onsite`. `none` is no signal.
|
|
33
|
+
const REMOTE_STATUS = { hybrid: 'hybrid', fully: 'remote', onsite: 'onsite' };
|
|
34
|
+
|
|
31
35
|
/**
|
|
32
36
|
* Resolve which TeamTailor host actually serves this slug's feed.
|
|
33
37
|
* Returns the first 200 Response, throws on a non-404 error, or
|
|
@@ -64,8 +68,8 @@ export async function fetchTeamtailor(slug) {
|
|
|
64
68
|
const items = [...xml.matchAll(/<item>([\s\S]*?)<\/item>/g)].map(m => m[1]);
|
|
65
69
|
|
|
66
70
|
return items.map(item => {
|
|
67
|
-
const pick = (tag) => {
|
|
68
|
-
const m =
|
|
71
|
+
const pick = (tag, src = item) => {
|
|
72
|
+
const m = src.match(new RegExp(`<${tag}[^>]*>([\\s\\S]*?)</${tag}>`));
|
|
69
73
|
return m ? m[1].trim() : '';
|
|
70
74
|
};
|
|
71
75
|
|
|
@@ -78,6 +82,12 @@ export async function fetchTeamtailor(slug) {
|
|
|
78
82
|
const country = decodeEntities(pick('tt:country'));
|
|
79
83
|
const remoteStatus = decodeEntities(pick('remoteStatus'));
|
|
80
84
|
|
|
85
|
+
// One <tt:location> per office the posting is open in, read the same
|
|
86
|
+
// way as the primary above so the entries line up.
|
|
87
|
+
const locations = [...item.matchAll(/<tt:location>([\s\S]*?)<\/tt:location>/g)].map(m =>
|
|
88
|
+
[decodeEntities(pick('tt:city', m[1])), decodeEntities(pick('tt:country', m[1]))].filter(Boolean).join(', ')
|
|
89
|
+
);
|
|
90
|
+
|
|
81
91
|
let location = [city, country].filter(Boolean).join(', ');
|
|
82
92
|
if (/remote/i.test(remoteStatus)) {
|
|
83
93
|
location = location ? `Remote - ${location}` : 'Remote';
|
|
@@ -95,7 +105,9 @@ export async function fetchTeamtailor(slug) {
|
|
|
95
105
|
title,
|
|
96
106
|
department,
|
|
97
107
|
location,
|
|
98
|
-
|
|
108
|
+
locations,
|
|
109
|
+
workplace: REMOTE_STATUS[remoteStatus.toLowerCase()] || null,
|
|
110
|
+
description: decodeEntities(pick('description')),
|
|
99
111
|
url: link,
|
|
100
112
|
postedAt,
|
|
101
113
|
salary: null, // No structured salary; normalizer parses from text
|
|
@@ -107,22 +119,6 @@ export async function fetchTeamtailor(slug) {
|
|
|
107
119
|
});
|
|
108
120
|
}
|
|
109
121
|
|
|
110
|
-
/**
|
|
111
|
-
* Decode the RSS entity/CDATA layer to real HTML.
|
|
112
|
-
* `&` is intentionally resolved LAST.
|
|
113
|
-
*/
|
|
114
|
-
function decodeEntities(s) {
|
|
115
|
-
if (!s) return '';
|
|
116
|
-
return s
|
|
117
|
-
.replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1')
|
|
118
|
-
.replace(/</g, '<')
|
|
119
|
-
.replace(/>/g, '>')
|
|
120
|
-
.replace(/"/g, '"')
|
|
121
|
-
.replace(/'/g, "'")
|
|
122
|
-
.replace(/'/g, "'")
|
|
123
|
-
.replace(/&/g, '&');
|
|
124
|
-
}
|
|
125
|
-
|
|
126
122
|
/**
|
|
127
123
|
* Check if a company has a TeamTailor career site.
|
|
128
124
|
*/
|