jd-intel 0.8.2 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +22 -5
- package/package.json +1 -1
- package/registry/ashby.json +154 -118
- package/registry/greenhouse.json +60 -3
- package/registry/lever.json +78 -68
- package/registry/recruitee.json +31 -1
- package/registry/smartrecruiters.json +6 -1
- package/registry/teamtailor.json +36 -1
- package/registry/workday.json +242 -197
- package/src/adapters/ashby.js +52 -17
- package/src/adapters/greenhouse.js +27 -4
- package/src/adapters/lever.js +71 -37
- package/src/adapters/recruitee.js +67 -14
- package/src/adapters/smartrecruiters.js +41 -6
- package/src/adapters/teamtailor.js +20 -24
- package/src/adapters/workday.js +50 -15
- package/src/cli.js +34 -7
- package/src/filters.js +65 -11
- package/src/index.js +25 -9
- package/src/normalizer.js +200 -44
package/src/adapters/workday.js
CHANGED
|
@@ -1,9 +1,14 @@
|
|
|
1
|
-
import { normalize
|
|
1
|
+
import { normalize } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
|
|
4
4
|
const MAX_DETAIL_FETCHES = 100;
|
|
5
5
|
const LIST_PAGE_SIZE = 20;
|
|
6
|
-
|
|
6
|
+
// Upper bound on list pages per call: 100 pages of 20 = at most 2000
|
|
7
|
+
// postings scanned. Paging usually stops sooner, on a short page or when
|
|
8
|
+
// offset reaches the first page's total (see the loop below).
|
|
9
|
+
const LIST_PAGE_HARD_CAP = 100;
|
|
10
|
+
// A multi-location posting's list row reads "2 Locations", "14 Locations".
|
|
11
|
+
const MULTI_LOCATION = /^\s*\d+\s+locations?\s*$/;
|
|
7
12
|
|
|
8
13
|
/**
|
|
9
14
|
* Fetch jobs from a Workday tenant via the public "CXS" JSON API.
|
|
@@ -40,6 +45,7 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
40
45
|
const postings = [];
|
|
41
46
|
let offset = 0;
|
|
42
47
|
let pages = 0;
|
|
48
|
+
let firstTotal = 0;
|
|
43
49
|
while (pages < LIST_PAGE_HARD_CAP) {
|
|
44
50
|
const resp = await fetch(`${base}/jobs`, {
|
|
45
51
|
method: 'POST',
|
|
@@ -48,19 +54,24 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
48
54
|
});
|
|
49
55
|
|
|
50
56
|
if (!resp.ok) {
|
|
51
|
-
if (resp.status === 404) return []; // wrong site / no such board
|
|
52
57
|
if (offset === 0) {
|
|
58
|
+
if (resp.status === 404) return []; // wrong site / no such board
|
|
53
59
|
throw atsErrorFromStatus(resp.status, `Workday API error for ${slug} (${tenant}/${env}/${site}): ${resp.status}`);
|
|
54
60
|
}
|
|
55
|
-
break; // mid-paging failure: keep what we have
|
|
61
|
+
break; // mid-paging failure (any status): keep what we have
|
|
56
62
|
}
|
|
57
63
|
|
|
58
64
|
const data = await resp.json();
|
|
59
65
|
const page = data.jobPostings || [];
|
|
66
|
+
// Some tenants report the real `total` only at offset 0 and send
|
|
67
|
+
// `total: 0` on every later page, so only the first page's figure
|
|
68
|
+
// is trusted. A short page is the other stop signal.
|
|
69
|
+
if (pages === 0) firstTotal = data.total || 0;
|
|
60
70
|
postings.push(...page);
|
|
61
71
|
pages += 1;
|
|
62
72
|
offset += LIST_PAGE_SIZE;
|
|
63
|
-
if (page.length
|
|
73
|
+
if (page.length < LIST_PAGE_SIZE) break;
|
|
74
|
+
if (firstTotal > 0 && offset >= firstTotal) break;
|
|
64
75
|
}
|
|
65
76
|
|
|
66
77
|
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost.
|
|
@@ -76,6 +87,9 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
76
87
|
const inc = fc.locationIncludes.map(s => String(s).toLowerCase());
|
|
77
88
|
candidates = candidates.filter(p => {
|
|
78
89
|
const loc = (p.locationsText || '').toLowerCase();
|
|
90
|
+
// "2 Locations" says nothing about where. The row stays a candidate
|
|
91
|
+
// and the pass after hydration decides on the detail's location list.
|
|
92
|
+
if (MULTI_LOCATION.test(loc)) return true;
|
|
79
93
|
return inc.some(s => loc.includes(s));
|
|
80
94
|
});
|
|
81
95
|
}
|
|
@@ -91,16 +105,21 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
91
105
|
}
|
|
92
106
|
|
|
93
107
|
// 3. Bound the detail-fetch set.
|
|
94
|
-
// NOTE: huge-tenant coverage is intentionally capped for v1
|
|
95
|
-
//
|
|
96
|
-
//
|
|
97
|
-
//
|
|
98
|
-
//
|
|
99
|
-
//
|
|
100
|
-
//
|
|
101
|
-
//
|
|
108
|
+
// NOTE: huge-tenant coverage is intentionally capped for v1. Two
|
|
109
|
+
// caps apply: the list scan above stops at LIST_PAGE_HARD_CAP pages
|
|
110
|
+
// (2000 postings, enough for Salesforce's ~1398), and the detail set
|
|
111
|
+
// is cut to MAX_DETAIL_FETCHES here. A description `filter` is
|
|
112
|
+
// applied by the library AFTER this returns, so for that case we
|
|
113
|
+
// keep the full backstop instead of truncating tightly to `limit`
|
|
114
|
+
// (which could hydrate jobs that all fail the regex while better
|
|
115
|
+
// matches go unscanned). The library pages with `offset` after this
|
|
116
|
+
// returns, so the budget covers the page plus what precedes it.
|
|
117
|
+
// Proper fix (smart pagination / rate-limited concurrency / surfaced
|
|
118
|
+
// truncation) is tracked in #26, to be designed alongside
|
|
119
|
+
// retry/rate-limit work (#7).
|
|
102
120
|
const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
|
|
103
|
-
const
|
|
121
|
+
const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
|
|
122
|
+
const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
|
|
104
123
|
candidates = candidates.slice(0, cap);
|
|
105
124
|
|
|
106
125
|
// 4. Hydrate descriptions via the per-posting detail endpoint.
|
|
@@ -126,7 +145,9 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
126
145
|
title: p.title || info.title || '',
|
|
127
146
|
department: '',
|
|
128
147
|
location: info.location || p.locationsText || '',
|
|
129
|
-
|
|
148
|
+
locations: info.additionalLocations || [],
|
|
149
|
+
workplace: parseWorkdayRemoteType(info.remoteType),
|
|
150
|
+
description: info.jobDescription || '',
|
|
130
151
|
url: `https://${tenant}.${env}.myworkdayjobs.com/${site}${externalPath}`,
|
|
131
152
|
postedAt: parseWorkdayDate(info.startDate) || normalizePostedOn(p.postedOn),
|
|
132
153
|
salary: null, // normalizer extracts from description text
|
|
@@ -142,6 +163,20 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
142
163
|
return jobs;
|
|
143
164
|
}
|
|
144
165
|
|
|
166
|
+
/**
|
|
167
|
+
* Detail `remoteType` is free text set per tenant: "Remote", "Hybrid",
|
|
168
|
+
* "Office - Flexible", "On-site". A flexible office arrangement counts as
|
|
169
|
+
* hybrid, so that check runs before the office one. Some tenants send no
|
|
170
|
+
* value at all; the location string decides then.
|
|
171
|
+
*/
|
|
172
|
+
function parseWorkdayRemoteType(remoteType) {
|
|
173
|
+
const s = String(remoteType || '').toLowerCase();
|
|
174
|
+
if (/remote/.test(s)) return 'remote';
|
|
175
|
+
if (/hybrid|flexible/.test(s)) return 'hybrid';
|
|
176
|
+
if (/office|on-?site/.test(s)) return 'onsite';
|
|
177
|
+
return null;
|
|
178
|
+
}
|
|
179
|
+
|
|
145
180
|
/**
|
|
146
181
|
* Workday list `postedOn` is a relative string ("Posted Today",
|
|
147
182
|
* "Posted 5 Days Ago", "Posted 30+ Days Ago"). Decide membership in
|
package/src/cli.js
CHANGED
|
@@ -9,6 +9,8 @@
|
|
|
9
9
|
* jd-intel registry search <query>
|
|
10
10
|
*/
|
|
11
11
|
|
|
12
|
+
import { realpathSync } from 'node:fs';
|
|
13
|
+
import { fileURLToPath } from 'node:url';
|
|
12
14
|
import { fetchJobs } from './index.js';
|
|
13
15
|
import { detectAts, searchRegistry } from './registry.js';
|
|
14
16
|
|
|
@@ -85,7 +87,7 @@ async function main() {
|
|
|
85
87
|
console.log(`Found ${jobs.length} jobs\n`);
|
|
86
88
|
|
|
87
89
|
for (const job of jobs.slice(0, 20)) {
|
|
88
|
-
const salary = job.salary ? ` |
|
|
90
|
+
const salary = job.salary ? ` | ${formatSalary(job.salary)}` : '';
|
|
89
91
|
const loc = job.location ? ` | ${job.location}` : '';
|
|
90
92
|
const dept = job.department ? ` [${job.department}]` : '';
|
|
91
93
|
console.log(` ${job.title}${dept}${loc}${salary}`);
|
|
@@ -162,8 +164,8 @@ Fetch options:
|
|
|
162
164
|
--title-filter pattern Regex matched against TITLE only (role identity)
|
|
163
165
|
--filter pattern Regex matched across title, department, description (topic/scope)
|
|
164
166
|
--posted-within-days N Only jobs posted in the last N days
|
|
165
|
-
--location-include "A,B,C" Keep jobs
|
|
166
|
-
--location-exclude "A,B,C" Drop jobs
|
|
167
|
+
--location-include "A,B,C" Keep jobs where any listed location contains one of these
|
|
168
|
+
--location-exclude "A,B,C" Drop jobs only when every listed location contains one of these
|
|
167
169
|
--limit N Cap results (default 100)
|
|
168
170
|
--json Output full JSON
|
|
169
171
|
|
|
@@ -184,7 +186,32 @@ Examples:
|
|
|
184
186
|
}
|
|
185
187
|
}
|
|
186
188
|
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
189
|
+
export function formatSalary({ min, max, currency, period }) {
|
|
190
|
+
const hasMin = min != null;
|
|
191
|
+
const hasMax = max != null;
|
|
192
|
+
let range;
|
|
193
|
+
if (hasMin && hasMax) range = `${min.toLocaleString()}-${max.toLocaleString()}`;
|
|
194
|
+
else if (hasMin) range = `from ${min.toLocaleString()}`;
|
|
195
|
+
else range = `up to ${max.toLocaleString()}`;
|
|
196
|
+
const unit = period === 'hour' ? '/hr' : period === 'month' ? '/mo' : '';
|
|
197
|
+
return `${range} ${currency}${unit}`;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
// Boot only when this file is the script Node was started with, so a test
|
|
201
|
+
// can import formatSalary without running a command. argv[1] is resolved
|
|
202
|
+
// through realpath because npm installs the bin as a symlink into .bin/,
|
|
203
|
+
// while import.meta.url already points at the real file.
|
|
204
|
+
function isEntrypoint() {
|
|
205
|
+
try {
|
|
206
|
+
return realpathSync(process.argv[1]) === fileURLToPath(import.meta.url);
|
|
207
|
+
} catch {
|
|
208
|
+
return false;
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
if (isEntrypoint()) {
|
|
213
|
+
main().catch(err => {
|
|
214
|
+
console.error('Error:', err.message);
|
|
215
|
+
process.exit(1);
|
|
216
|
+
});
|
|
217
|
+
}
|
package/src/filters.js
CHANGED
|
@@ -4,14 +4,33 @@
|
|
|
4
4
|
* Facts go here (deterministic field matches). Interpretations stay with the
|
|
5
5
|
* caller — this module does substring matching on structured fields, nothing
|
|
6
6
|
* semantic.
|
|
7
|
+
*
|
|
8
|
+
* Returns the page as an array. applyFiltersDetailed returns the same page
|
|
9
|
+
* plus total_matched, the match count before offset and limit.
|
|
7
10
|
*/
|
|
8
11
|
export function applyFilters(jobs, options = {}) {
|
|
12
|
+
return applyFiltersDetailed(jobs, options).jobs;
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* Filter, sort, then page.
|
|
17
|
+
*
|
|
18
|
+
* Order is applied after the filters and before offset and limit, so a cut
|
|
19
|
+
* drops the oldest matches first. 'newest' sorts by postedAt descending with
|
|
20
|
+
* undated jobs last and ties broken by id, which keeps pages deterministic.
|
|
21
|
+
* 'board' keeps the order the adapter returned.
|
|
22
|
+
*
|
|
23
|
+
* @returns {{ jobs: Array, total_matched: number }}
|
|
24
|
+
*/
|
|
25
|
+
export function applyFiltersDetailed(jobs, options = {}) {
|
|
9
26
|
const {
|
|
10
27
|
titleFilter,
|
|
11
28
|
filter,
|
|
12
29
|
postedWithinDays,
|
|
13
30
|
locationIncludes,
|
|
14
31
|
locationExcludes,
|
|
32
|
+
order = 'newest',
|
|
33
|
+
offset = 0,
|
|
15
34
|
limit = 100,
|
|
16
35
|
} = options;
|
|
17
36
|
|
|
@@ -42,25 +61,60 @@ export function applyFilters(jobs, options = {}) {
|
|
|
42
61
|
|
|
43
62
|
if (Array.isArray(locationIncludes) && locationIncludes.length > 0) {
|
|
44
63
|
const matchers = locationIncludes.map(makeLocationMatcher);
|
|
45
|
-
result = result.filter(j =>
|
|
46
|
-
const loc = (j.location || '').toLowerCase();
|
|
47
|
-
return matchers.some(m => m(loc));
|
|
48
|
-
});
|
|
64
|
+
result = result.filter(j => jobLocations(j).some(loc => matchers.some(m => m(loc))));
|
|
49
65
|
}
|
|
50
66
|
|
|
51
67
|
if (Array.isArray(locationExcludes) && locationExcludes.length > 0) {
|
|
52
68
|
const matchers = locationExcludes.map(makeLocationMatcher);
|
|
53
|
-
result = result.filter(j =>
|
|
54
|
-
const loc = (j.location || '').toLowerCase();
|
|
55
|
-
return !matchers.some(m => m(loc));
|
|
56
|
-
});
|
|
69
|
+
result = result.filter(j => !jobLocations(j).every(loc => matchers.some(m => m(loc))));
|
|
57
70
|
}
|
|
58
71
|
|
|
59
|
-
|
|
60
|
-
|
|
72
|
+
const total_matched = result.length;
|
|
73
|
+
|
|
74
|
+
if (order !== 'board') {
|
|
75
|
+
result = [...result].sort(byNewest);
|
|
61
76
|
}
|
|
62
77
|
|
|
63
|
-
|
|
78
|
+
const start = typeof offset === 'number' && offset > 0 ? offset : 0;
|
|
79
|
+
const end = typeof limit === 'number' ? start + limit : undefined;
|
|
80
|
+
if (start > 0 || (end !== undefined && result.length > end)) {
|
|
81
|
+
result = result.slice(start, end);
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
return { jobs: result, total_matched };
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* Every location a job is open in, lowercased. A job passes an include when
|
|
89
|
+
* any of them matches and is dropped by an exclude only when all of them
|
|
90
|
+
* match: a role open in Berlin and New York is still open in Berlin for
|
|
91
|
+
* someone excluding the US (issue #68). Jobs from before `locations`
|
|
92
|
+
* existed fall back to the single `location` string.
|
|
93
|
+
*/
|
|
94
|
+
function jobLocations(job) {
|
|
95
|
+
const list = Array.isArray(job.locations) && job.locations.length > 0
|
|
96
|
+
? job.locations
|
|
97
|
+
: [job.location || ''];
|
|
98
|
+
return list.map(loc => String(loc).toLowerCase());
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
function postedTime(job) {
|
|
102
|
+
if (!job.postedAt) return null;
|
|
103
|
+
const t = new Date(job.postedAt).getTime();
|
|
104
|
+
return Number.isFinite(t) ? t : null;
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
function byNewest(a, b) {
|
|
108
|
+
const ta = postedTime(a);
|
|
109
|
+
const tb = postedTime(b);
|
|
110
|
+
if (ta !== tb) {
|
|
111
|
+
if (ta === null) return 1;
|
|
112
|
+
if (tb === null) return -1;
|
|
113
|
+
return tb - ta;
|
|
114
|
+
}
|
|
115
|
+
const ia = a.id || '';
|
|
116
|
+
const ib = b.id || '';
|
|
117
|
+
return ia < ib ? -1 : ia > ib ? 1 : 0;
|
|
64
118
|
}
|
|
65
119
|
|
|
66
120
|
/**
|
package/src/index.js
CHANGED
|
@@ -8,11 +8,23 @@
|
|
|
8
8
|
|
|
9
9
|
import { ADAPTERS, ATS_NAMES } from './adapters/index.js';
|
|
10
10
|
import { loadRegistry, searchRegistry, detectAts, findAtsBySlug, findEntryBySlug, getRegistrySource } from './registry.js';
|
|
11
|
-
import {
|
|
11
|
+
import { applyFiltersDetailed } from './filters.js';
|
|
12
12
|
|
|
13
13
|
/**
|
|
14
14
|
* Fetch jobs from a company's ATS board.
|
|
15
15
|
*
|
|
16
|
+
* Same options as fetchJobsDetailed; returns the page as an array.
|
|
17
|
+
*
|
|
18
|
+
* @returns {Promise<Array>} Normalized, filtered job objects
|
|
19
|
+
*/
|
|
20
|
+
export async function fetchJobs(options = {}) {
|
|
21
|
+
const { jobs } = await fetchJobsDetailed(options);
|
|
22
|
+
return jobs;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Fetch jobs from a company's ATS board, with the match count.
|
|
27
|
+
*
|
|
16
28
|
* @param {Object} options
|
|
17
29
|
* @param {string} options.company - Company slug or name
|
|
18
30
|
* @param {string} [options.ats] - Specific ATS platform. If omitted, auto-detects.
|
|
@@ -20,12 +32,14 @@ import { applyFilters } from './filters.js';
|
|
|
20
32
|
* @param {string} [options.titleFilter] - Regex matched against title only. Use for role identity ("product manager", "staff engineer").
|
|
21
33
|
* @param {string} [options.filter] - Regex matched across title, department, description. Use for topic/scope.
|
|
22
34
|
* @param {number} [options.postedWithinDays] - Only return jobs posted within N days.
|
|
23
|
-
* @param {string[]} [options.locationIncludes] - Keep jobs
|
|
24
|
-
* @param {string[]} [options.locationExcludes] - Drop jobs
|
|
25
|
-
* @param {
|
|
26
|
-
* @
|
|
35
|
+
* @param {string[]} [options.locationIncludes] - Keep jobs where any listed location contains any of these (case-insensitive).
|
|
36
|
+
* @param {string[]} [options.locationExcludes] - Drop jobs only when every listed location contains one of these (case-insensitive).
|
|
37
|
+
* @param {'newest'|'board'} [options.order='newest'] - 'newest': by postedAt descending, undated last, ties by id. 'board': the adapter's own order.
|
|
38
|
+
* @param {number} [options.offset=0] - Matches to skip after sorting (paging).
|
|
39
|
+
* @param {number} [options.limit=100] - Maximum jobs to return after offset.
|
|
40
|
+
* @returns {Promise<{ jobs: Array, total_matched: number }>} The page, plus the match count before offset and limit
|
|
27
41
|
*/
|
|
28
|
-
export async function
|
|
42
|
+
export async function fetchJobsDetailed({
|
|
29
43
|
company,
|
|
30
44
|
ats,
|
|
31
45
|
config,
|
|
@@ -34,6 +48,8 @@ export async function fetchJobs({
|
|
|
34
48
|
postedWithinDays,
|
|
35
49
|
locationIncludes,
|
|
36
50
|
locationExcludes,
|
|
51
|
+
order = 'newest',
|
|
52
|
+
offset = 0,
|
|
37
53
|
limit = 100,
|
|
38
54
|
} = {}) {
|
|
39
55
|
if (!company) throw new Error('Company slug required');
|
|
@@ -45,7 +61,7 @@ export async function fetchJobs({
|
|
|
45
61
|
// adapters declare fetch{Name}(slug) and ignore extra positional args
|
|
46
62
|
// (JS no-op), so this is backward-compatible. Filter-aware adapters
|
|
47
63
|
// (e.g. Workday) use it to avoid mass detail-fetching on huge tenants.
|
|
48
|
-
const filterContext = { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes, limit };
|
|
64
|
+
const filterContext = { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes, offset, limit };
|
|
49
65
|
|
|
50
66
|
let jobs;
|
|
51
67
|
if (ats) {
|
|
@@ -93,7 +109,7 @@ export async function fetchJobs({
|
|
|
93
109
|
}
|
|
94
110
|
}
|
|
95
111
|
|
|
96
|
-
return
|
|
112
|
+
return applyFiltersDetailed(jobs, { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes, order, offset, limit });
|
|
97
113
|
}
|
|
98
114
|
|
|
99
115
|
/**
|
|
@@ -134,7 +150,7 @@ export { fetchLever } from './adapters/lever.js';
|
|
|
134
150
|
export { fetchAshby } from './adapters/ashby.js';
|
|
135
151
|
|
|
136
152
|
// Re-export filter logic for reuse (e.g., by the MCP server)
|
|
137
|
-
export { applyFilters } from './filters.js';
|
|
153
|
+
export { applyFilters, applyFiltersDetailed } from './filters.js';
|
|
138
154
|
|
|
139
155
|
// Re-export the list of supported ATS names (e.g. so the MCP layer can report
|
|
140
156
|
// the full set detectAts probes, instead of hardcoding a stale subset).
|
package/src/normalizer.js
CHANGED
|
@@ -13,20 +13,35 @@ export function jobId(company, title, ats, location = '') {
|
|
|
13
13
|
|
|
14
14
|
/**
|
|
15
15
|
* Normalize a raw ATS job object into the unified schema.
|
|
16
|
+
*
|
|
17
|
+
* Adapters pass `description` as HTML. This is the one place it is
|
|
18
|
+
* stripped and decoded (issue #66): a second pass would delete text the
|
|
19
|
+
* author escaped on purpose (`<5 years`) and leave entities the first
|
|
20
|
+
* pass exposed (`&mdash;` -> `—`) as literal noise.
|
|
21
|
+
*
|
|
22
|
+
* `raw.workplace` is the ATS's own arrangement, already mapped by the
|
|
23
|
+
* adapter to 'remote' | 'hybrid' | 'onsite', or null when the platform
|
|
24
|
+
* gives no signal. `raw.locations` lists every place the posting is open
|
|
25
|
+
* in; `location` stays the primary because it feeds the id (issue #68).
|
|
16
26
|
*/
|
|
17
27
|
export function normalize(raw, ats) {
|
|
18
28
|
const now = new Date().toISOString();
|
|
29
|
+
const description = stripHtml(raw.description || '');
|
|
30
|
+
const location = raw.location || '';
|
|
31
|
+
const workplace = resolveWorkplace(raw.workplace, location);
|
|
19
32
|
return {
|
|
20
|
-
id: jobId(raw.company || raw.companySlug, raw.title, ats,
|
|
33
|
+
id: jobId(raw.company || raw.companySlug, raw.title, ats, location),
|
|
21
34
|
company: raw.company || raw.companySlug || '',
|
|
22
35
|
companySlug: raw.companySlug || '',
|
|
23
36
|
ats,
|
|
24
37
|
title: raw.title || '',
|
|
25
38
|
department: raw.department || '',
|
|
26
|
-
location
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
39
|
+
location,
|
|
40
|
+
locations: uniqueLocations(location, raw.locations),
|
|
41
|
+
locationType: workplace.type,
|
|
42
|
+
workplace,
|
|
43
|
+
salary: raw.salary || extractSalaryFromText(description),
|
|
44
|
+
description,
|
|
30
45
|
url: raw.url || '',
|
|
31
46
|
postedAt: raw.postedAt || null,
|
|
32
47
|
firstSeen: now,
|
|
@@ -36,62 +51,203 @@ export function normalize(raw, ats) {
|
|
|
36
51
|
};
|
|
37
52
|
}
|
|
38
53
|
|
|
54
|
+
const CURRENCY_CODES = 'USD|EUR|GBP|CAD|AUD|NZD|CHF|SEK|NOK|DKK|PLN|CZK|HUF|INR|SGD|HKD|JPY|CNY|BRL|MXN|ZAR|AED|ILS';
|
|
55
|
+
const SYMBOL_CURRENCY = { $: 'USD', '€': 'EUR', '£': 'GBP' };
|
|
56
|
+
|
|
57
|
+
// A number as job posts write it: 1,234,567 / 1.234.567 / 1234, with an
|
|
58
|
+
// optional one- or two-digit decimal part (211.4, 40.50, 60.000,50).
|
|
59
|
+
// Exactly three digits after a dot are a thousands group, the way Dutch
|
|
60
|
+
// and German boards write it: "€60.000" is sixty thousand, not sixty.
|
|
61
|
+
const NUMBER =
|
|
62
|
+
'\\d{1,3}(?:,\\d{3})+(?:\\.\\d{1,2})?' +
|
|
63
|
+
'|\\d{1,3}(?:\\.\\d{3})+(?:,\\d{1,2})?' +
|
|
64
|
+
'|\\d+(?:[.,]\\d{1,2})?';
|
|
65
|
+
|
|
66
|
+
// One side of a range: optional code before, optional symbol, the number,
|
|
67
|
+
// optional K, optional code after. The lookarounds keep the number from
|
|
68
|
+
// starting or ending inside a longer one ("234.567" out of "1.234.567").
|
|
69
|
+
const amountPattern = (p) =>
|
|
70
|
+
`(?:\\b(?<${p}CodeBefore>${CURRENCY_CODES})\\s?)?` +
|
|
71
|
+
`(?<${p}Sym>[$€£])?\\s?` +
|
|
72
|
+
`(?<![\\d.,])(?<${p}Num>${NUMBER})(?!\\d|[.,]\\d)\\s?` +
|
|
73
|
+
`(?<${p}K>[kK]\\b)?` +
|
|
74
|
+
`(?:\\s?(?<${p}CodeAfter>${CURRENCY_CODES})\\b)?`;
|
|
75
|
+
|
|
76
|
+
const SALARY_RANGE = new RegExp(
|
|
77
|
+
`${amountPattern('lo')}\\s*(?:[-–—]|\\bto\\b)\\s*${amountPattern('hi')}`,
|
|
78
|
+
'g'
|
|
79
|
+
);
|
|
80
|
+
|
|
81
|
+
const HOUR_RE = /\b(?:per|an|each)\s+hour\b|\/\s*(?:hr|hour)\b|\bhourly\b/i;
|
|
82
|
+
const MONTH_RE = /\b(?:per|a|each)\s+month\b|\/\s*(?:mo|month)\b|\bmonthly\b/i;
|
|
83
|
+
const YEAR_RE = /\b(?:per|a|each)\s+(?:year|annum)\b|\/\s*(?:yr|year)\b|\b(?:annual(?:ly|ized)?|yearly)\b/i;
|
|
84
|
+
|
|
39
85
|
/**
|
|
40
|
-
*
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
*
|
|
44
|
-
*
|
|
86
|
+
* Extract a salary range from decoded job text.
|
|
87
|
+
*
|
|
88
|
+
* Accepts hyphen, en dash, em dash or "to" between the two amounts, an
|
|
89
|
+
* optional ISO currency code before, between or after them, `$` / EUR /
|
|
90
|
+
* GBP symbols, decimal K shorthand ($211.4K), and thousands grouped with
|
|
91
|
+
* either a comma or a dot (60,000 and 60.000 are both sixty thousand).
|
|
92
|
+
* A code wins over a symbol, so "$120,000 - $150,000 CAD" is CAD. Ranges
|
|
93
|
+
* with no currency marker at all (years, headcounts) are ignored.
|
|
94
|
+
*
|
|
95
|
+
* @returns {{min:number,max:number,currency:string,period:('year'|'month'|'hour'|null),source:'text'}|null}
|
|
45
96
|
*/
|
|
46
|
-
function extractSalaryFromText(text) {
|
|
97
|
+
export function extractSalaryFromText(text) {
|
|
47
98
|
if (!text) return null;
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
99
|
+
for (const m of text.matchAll(SALARY_RANGE)) {
|
|
100
|
+
const g = m.groups;
|
|
101
|
+
const end = m.index + m[0].length;
|
|
102
|
+
const after = text.slice(end, end + 40);
|
|
103
|
+
// "$20 - $30 million" is a revenue figure, not pay.
|
|
104
|
+
if (/^\s*(?:million|billion|m|bn?)\b/i.test(after)) continue;
|
|
105
|
+
|
|
106
|
+
const code = g.loCodeBefore || g.loCodeAfter || g.hiCodeBefore || g.hiCodeAfter;
|
|
107
|
+
const sym = g.loSym || g.hiSym;
|
|
108
|
+
if (!code && !sym) continue;
|
|
109
|
+
|
|
110
|
+
let min = parseAmount(g.loNum);
|
|
111
|
+
let max = parseAmount(g.hiNum);
|
|
112
|
+
if (g.loK || g.hiK) {
|
|
113
|
+
// "$150-200K" carries the K once for both sides.
|
|
114
|
+
if (min < 1000) min = Math.round(min * 1000);
|
|
115
|
+
if (max < 1000) max = Math.round(max * 1000);
|
|
116
|
+
}
|
|
117
|
+
if (!(min > 0) || !(max > 0)) continue;
|
|
118
|
+
|
|
119
|
+
const before = text.slice(Math.max(0, m.index - 40), m.index);
|
|
60
120
|
return {
|
|
61
|
-
min
|
|
62
|
-
max
|
|
63
|
-
currency:
|
|
121
|
+
min,
|
|
122
|
+
max,
|
|
123
|
+
currency: (code || SYMBOL_CURRENCY[sym]).toUpperCase(),
|
|
124
|
+
period: detectPeriod(before, after, min),
|
|
125
|
+
source: 'text',
|
|
64
126
|
};
|
|
65
127
|
}
|
|
66
128
|
return null;
|
|
67
129
|
}
|
|
68
130
|
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
if (
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
131
|
+
// Dots grouping thousands mean a comma is the decimal mark, and vice versa.
|
|
132
|
+
function parseAmount(s) {
|
|
133
|
+
if (/^\d{1,3}(?:\.\d{3})+/.test(s)) return Number(s.replace(/\./g, '').replace(',', '.'));
|
|
134
|
+
return Number(s.replace(/,(?=\d{3})/g, '').replace(',', '.'));
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
function detectPeriod(before, after, min) {
|
|
138
|
+
const explicit = periodWord(after) || periodWord(before);
|
|
139
|
+
if (explicit) return explicit;
|
|
140
|
+
return min >= 10000 ? 'year' : null;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
function periodWord(text) {
|
|
144
|
+
if (HOUR_RE.test(text)) return 'hour';
|
|
145
|
+
if (MONTH_RE.test(text)) return 'month';
|
|
146
|
+
if (YEAR_RE.test(text)) return 'year';
|
|
147
|
+
return null;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
|
|
151
|
+
|
|
152
|
+
/**
|
|
153
|
+
* The platform's own value wins. Without one, a keyword in the location
|
|
154
|
+
* string is the next best signal. Without either the type is 'unknown':
|
|
155
|
+
* a city name alone does not say the role is onsite, and a guessed
|
|
156
|
+
* 'onsite' reads as a fact to whoever consumes it.
|
|
157
|
+
*
|
|
158
|
+
* @returns {{type:('remote'|'hybrid'|'onsite'|'unknown'), source:('ats'|'text'|null)}}
|
|
159
|
+
*/
|
|
160
|
+
function resolveWorkplace(native, location) {
|
|
161
|
+
if (WORKPLACE_TYPES.has(native)) return { type: native, source: 'ats' };
|
|
162
|
+
const guessed = workplaceFromText(location);
|
|
163
|
+
if (guessed) return { type: guessed, source: 'text' };
|
|
164
|
+
return { type: 'unknown', source: null };
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
function workplaceFromText(location) {
|
|
168
|
+
const lower = (location || '').toLowerCase();
|
|
169
|
+
if (/remote/.test(lower)) return 'remote';
|
|
170
|
+
if (/hybrid/.test(lower)) return 'hybrid';
|
|
171
|
+
if (/on-?site/.test(lower)) return 'onsite';
|
|
172
|
+
return null;
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
function uniqueLocations(primary, extra) {
|
|
176
|
+
const out = [];
|
|
177
|
+
const seen = new Set();
|
|
178
|
+
for (const loc of [primary, ...(Array.isArray(extra) ? extra : [])]) {
|
|
179
|
+
const s = typeof loc === 'string' ? loc.trim() : '';
|
|
180
|
+
if (!s || seen.has(s.toLowerCase())) continue;
|
|
181
|
+
seen.add(s.toLowerCase());
|
|
182
|
+
out.push(s);
|
|
183
|
+
}
|
|
184
|
+
return out;
|
|
75
185
|
}
|
|
76
186
|
|
|
77
187
|
/**
|
|
78
188
|
* Strip HTML tags and convert to clean text.
|
|
189
|
+
*
|
|
190
|
+
* Block closers become line breaks and list items become bullets before
|
|
191
|
+
* the remaining tags are removed. Entities are decoded LAST, so a literal
|
|
192
|
+
* `<` in the source text never turns into a tag that gets stripped.
|
|
79
193
|
*/
|
|
80
194
|
export function stripHtml(html) {
|
|
81
195
|
if (!html) return '';
|
|
82
|
-
|
|
196
|
+
const text = html
|
|
83
197
|
.replace(/<br\s*\/?>/gi, '\n')
|
|
84
|
-
.replace(/<\/p
|
|
85
|
-
.replace(/<\/li
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
.replace(/<
|
|
89
|
-
.replace(/<[^>]
|
|
90
|
-
.replace(
|
|
91
|
-
|
|
92
|
-
.replace(
|
|
93
|
-
.replace(/ /g, ' ')
|
|
94
|
-
.replace(/&#\d+;/g, '')
|
|
198
|
+
.replace(/<\/(?:p|h[1-6])\s*>/gi, '\n\n')
|
|
199
|
+
.replace(/<\/(?:li|div|td|tr|ul|ol|table|section)\s*>/gi, '\n')
|
|
200
|
+
// Word-pasted markup (Lever lists, issue #64) opens a <p> inside each
|
|
201
|
+
// <li>; swallowing it keeps the item text on the bullet's line.
|
|
202
|
+
.replace(/<li\b[^>]*>\s*(?:<p\b[^>]*>\s*)?/gi, '- ')
|
|
203
|
+
.replace(/<h[1-6]\b[^>]*>/gi, '## ')
|
|
204
|
+
.replace(/<[^>]+>/g, '');
|
|
205
|
+
return decodeEntities(text)
|
|
206
|
+
.replace(/\u00a0/g, ' ')
|
|
95
207
|
.replace(/\n{3,}/g, '\n\n')
|
|
96
208
|
.trim();
|
|
97
209
|
}
|
|
210
|
+
|
|
211
|
+
const NAMED_ENTITIES = {
|
|
212
|
+
lt: '<', gt: '>', quot: '"', apos: "'", nbsp: '\u00a0',
|
|
213
|
+
mdash: '—', ndash: '–', hellip: '…',
|
|
214
|
+
lsquo: '‘', rsquo: '’', ldquo: '“', rdquo: '”',
|
|
215
|
+
sbquo: '‚', bdquo: '„', laquo: '«', raquo: '»',
|
|
216
|
+
bull: '•', middot: '·', copy: '©', reg: '®', trade: '™',
|
|
217
|
+
deg: '°', times: '×', euro: '€', pound: '£', yen: '¥', cent: '¢',
|
|
218
|
+
agrave: 'à', aacute: 'á', acirc: 'â', atilde: 'ã', auml: 'ä', aring: 'å', aelig: 'æ',
|
|
219
|
+
ccedil: 'ç', egrave: 'è', eacute: 'é', ecirc: 'ê', euml: 'ë',
|
|
220
|
+
igrave: 'ì', iacute: 'í', icirc: 'î', iuml: 'ï', ntilde: 'ñ',
|
|
221
|
+
ograve: 'ò', oacute: 'ó', ocirc: 'ô', otilde: 'õ', ouml: 'ö', oslash: 'ø',
|
|
222
|
+
ugrave: 'ù', uacute: 'ú', ucirc: 'û', uuml: 'ü', yacute: 'ý', yuml: 'ÿ', szlig: 'ß',
|
|
223
|
+
Agrave: 'À', Aacute: 'Á', Acirc: 'Â', Atilde: 'Ã', Auml: 'Ä', Aring: 'Å', AElig: 'Æ',
|
|
224
|
+
Ccedil: 'Ç', Egrave: 'È', Eacute: 'É', Ecirc: 'Ê', Euml: 'Ë',
|
|
225
|
+
Igrave: 'Ì', Iacute: 'Í', Icirc: 'Î', Iuml: 'Ï', Ntilde: 'Ñ',
|
|
226
|
+
Ograve: 'Ò', Oacute: 'Ó', Ocirc: 'Ô', Otilde: 'Õ', Ouml: 'Ö', Oslash: 'Ø',
|
|
227
|
+
Ugrave: 'Ù', Uacute: 'Ú', Ucirc: 'Û', Uuml: 'Ü', Yacute: 'Ý',
|
|
228
|
+
};
|
|
229
|
+
|
|
230
|
+
/**
|
|
231
|
+
* Decode one layer of entity encoding (and unwrap CDATA) to real text.
|
|
232
|
+
*
|
|
233
|
+
* Decimal and hex references go through String.fromCodePoint, named
|
|
234
|
+
* references through the table above. `&` is intentionally resolved
|
|
235
|
+
* LAST so double-encoded sequences (`&mdash;`, `&amp;`) collapse
|
|
236
|
+
* by exactly one layer per call. Used for the outer escaping Greenhouse
|
|
237
|
+
* and the Teamtailor RSS feed apply, and as stripHtml's final step.
|
|
238
|
+
*/
|
|
239
|
+
export function decodeEntities(s) {
|
|
240
|
+
if (!s) return '';
|
|
241
|
+
return s
|
|
242
|
+
.replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1')
|
|
243
|
+
.replace(/&#(\d+);/g, (m, dec) => codePointToString(parseInt(dec, 10), m))
|
|
244
|
+
.replace(/&#[xX]([0-9a-fA-F]+);/g, (m, hex) => codePointToString(parseInt(hex, 16), m))
|
|
245
|
+
.replace(/&([A-Za-z][A-Za-z0-9]*);/g, (m, name) =>
|
|
246
|
+
(name !== 'amp' && Object.hasOwn(NAMED_ENTITIES, name)) ? NAMED_ENTITIES[name] : m)
|
|
247
|
+
.replace(/&/g, '&');
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
function codePointToString(cp, fallback) {
|
|
251
|
+
if (!cp || cp > 0x10ffff || (cp >= 0xd800 && cp <= 0xdfff)) return fallback;
|
|
252
|
+
return String.fromCodePoint(cp);
|
|
253
|
+
}
|