jd-intel 0.8.3 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +21 -4
- package/package.json +1 -1
- package/src/adapters/ashby.js +52 -17
- package/src/adapters/greenhouse.js +27 -4
- package/src/adapters/lever.js +71 -37
- package/src/adapters/recruitee.js +67 -14
- package/src/adapters/smartrecruiters.js +41 -6
- package/src/adapters/teamtailor.js +20 -24
- package/src/adapters/workday.js +50 -15
- package/src/cli.js +34 -7
- package/src/filters.js +65 -11
- package/src/index.js +25 -9
- package/src/normalizer.js +200 -44
package/README.md
CHANGED
|
@@ -131,6 +131,19 @@ const jobs = await fetchJobs({
|
|
|
131
131
|
});
|
|
132
132
|
```
|
|
133
133
|
|
|
134
|
+
Results come back newest first by `postedAt`, undated last (`order: 'board'` keeps the ATS's own order). `fetchJobs` returns the page as an array. `fetchJobsDetailed` returns the same page plus `total_matched`, the number of matches before `offset` and `limit`, so you can tell a small board from a cut and page through the rest:
|
|
135
|
+
|
|
136
|
+
```js
|
|
137
|
+
import { fetchJobsDetailed } from 'jd-intel';
|
|
138
|
+
|
|
139
|
+
const { jobs, total_matched } = await fetchJobsDetailed({
|
|
140
|
+
company: '<your-target-company>',
|
|
141
|
+
titleFilter: 'engineer',
|
|
142
|
+
limit: 20,
|
|
143
|
+
offset: 20, // second page
|
|
144
|
+
});
|
|
145
|
+
```
|
|
146
|
+
|
|
134
147
|
CLI usage: `npx jd-intel fetch <company-slug> --title-filter "engineer" --posted-within-days 14`. Full filter reference [below](#filters-quick-reference).
|
|
135
148
|
|
|
136
149
|
Node.js 18+. No API keys. No configuration.
|
|
@@ -189,8 +202,10 @@ Every job normalizes to one schema, across every platform:
|
|
|
189
202
|
"title": "Senior Software Engineer, Platform",
|
|
190
203
|
"department": "Engineering",
|
|
191
204
|
"location": "Remote - US",
|
|
205
|
+
"locations": ["Remote - US", "Toronto, Canada"],
|
|
192
206
|
"locationType": "remote",
|
|
193
|
-
"
|
|
207
|
+
"workplace": { "type": "remote", "source": "ats" },
|
|
208
|
+
"salary": { "min": 180000, "max": 240000, "currency": "USD", "period": "year", "source": "text" },
|
|
194
209
|
"description": "Design and build the API surface our customers integrate against...",
|
|
195
210
|
"url": "https://boards.example.com/jobs/12345",
|
|
196
211
|
"postedAt": "2026-04-10T14:30:00Z"
|
|
@@ -206,9 +221,11 @@ No custom parsing per company.
|
|
|
206
221
|
| `title` | Full job title |
|
|
207
222
|
| `company` | Normalized company name |
|
|
208
223
|
| `department` | Team or department (when provided) |
|
|
209
|
-
| `location` |
|
|
210
|
-
| `
|
|
211
|
-
| `
|
|
224
|
+
| `location` | Primary location: city, state, country, or remote |
|
|
225
|
+
| `locations` | Every location the posting is open in, primary first. The location filters check each entry |
|
|
226
|
+
| `locationType` | `remote`, `hybrid`, `onsite`, or `unknown` when neither the platform nor the location text says |
|
|
227
|
+
| `workplace` | `{ type, source }`. `type` repeats `locationType`; `source` is `ats` when the platform stated it, `text` when read from the location string, null when unknown |
|
|
228
|
+
| `salary` | Min-max range with `currency`, plus `period` (`year`, `month`, `hour`, or null) and `source` (`ats` when the platform supplied it, `text` when parsed from the posting). Null when nothing is stated |
|
|
212
229
|
| `description` | Full JD in clean markdown |
|
|
213
230
|
| `url` | Direct link to the posting |
|
|
214
231
|
| `postedAt` | Publication date (when provided) |
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "jd-intel",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.9.0",
|
|
4
4
|
"description": "Fetch and normalize job descriptions across seven major ATS (Greenhouse, Lever, Ashby, Workday, and more), for your AI assistant. No copy-paste.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "src/index.js",
|
package/src/adapters/ashby.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize } from '../normalizer.js';
|
|
1
|
+
import { normalize, extractSalaryFromText } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
|
|
4
4
|
const API_URL = 'https://jobs.ashbyhq.com/api/non-user-graphql';
|
|
@@ -36,23 +36,33 @@ async function fetchAshbyRest(slug) {
|
|
|
36
36
|
const jobs = data.jobs || [];
|
|
37
37
|
|
|
38
38
|
return jobs.map(job => {
|
|
39
|
-
const
|
|
39
|
+
const comp = job.compensation || {};
|
|
40
40
|
|
|
41
41
|
return normalize({
|
|
42
42
|
companySlug: slug,
|
|
43
43
|
company: data.organizationName || slug,
|
|
44
44
|
title: job.title || '',
|
|
45
|
-
department: job.
|
|
45
|
+
department: job.department || '',
|
|
46
46
|
location: job.location || '',
|
|
47
|
+
locations: (job.secondaryLocations || []).map(l => l?.location || ''),
|
|
48
|
+
workplace: parseAshbyWorkplace(job),
|
|
47
49
|
description: job.descriptionHtml || job.descriptionPlain || '',
|
|
48
50
|
url: `https://jobs.ashbyhq.com/${slug}/${job.id}`,
|
|
49
51
|
postedAt: job.publishedAt || null,
|
|
50
|
-
salary,
|
|
52
|
+
salary: parseAshbyCompensation(comp),
|
|
51
53
|
metadata: {
|
|
52
54
|
ashbyId: job.id,
|
|
53
55
|
employmentType: job.employmentType || '',
|
|
54
56
|
isRemote: job.isRemote || false,
|
|
55
|
-
team: job.
|
|
57
|
+
team: job.team || '',
|
|
58
|
+
// The rendered summaries keep what min/max drop: "Offers Equity",
|
|
59
|
+
// "Multiple Ranges", and per-location tiers labelled OTE.
|
|
60
|
+
compensationSummary: comp.compensationTierSummary || '',
|
|
61
|
+
compensationTiers: (comp.compensationTiers || []).map(tier => ({
|
|
62
|
+
title: tier.title || '',
|
|
63
|
+
summary: tier.tierSummary || '',
|
|
64
|
+
additionalInformation: tier.additionalInformation || '',
|
|
65
|
+
})),
|
|
56
66
|
},
|
|
57
67
|
}, 'ashby');
|
|
58
68
|
});
|
|
@@ -110,22 +120,47 @@ async function fetchAshbyGraphQL(slug) {
|
|
|
110
120
|
}, 'ashby'));
|
|
111
121
|
}
|
|
112
122
|
|
|
123
|
+
const WORKPLACE_TYPES = { remote: 'remote', hybrid: 'hybrid', onsite: 'onsite' };
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* `workplaceType` is 'Remote', 'Hybrid' or 'OnSite'. `isRemote` is the
|
|
127
|
+
* older flag and can only say remote, so it is the fallback when the type
|
|
128
|
+
* is absent. false means nothing: the role may be hybrid or onsite.
|
|
129
|
+
*/
|
|
130
|
+
function parseAshbyWorkplace(job) {
|
|
131
|
+
const type = WORKPLACE_TYPES[String(job.workplaceType || '').toLowerCase()];
|
|
132
|
+
if (type) return type;
|
|
133
|
+
return job.isRemote === true ? 'remote' : null;
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
const INTERVAL_PERIOD = { '1 YEAR': 'year', '1 MONTH': 'month', '1 HOUR': 'hour' };
|
|
137
|
+
|
|
138
|
+
/**
|
|
139
|
+
* Read pay from Ashby's `compensation` object (issue #67).
|
|
140
|
+
*
|
|
141
|
+
* `summaryComponents` carries one structured entry per component type
|
|
142
|
+
* (Salary, Bonus, Commission, Equity); the Salary entry spans every tier.
|
|
143
|
+
* `scrapeableCompensationSalarySummary` and `compensationTierSummary` are
|
|
144
|
+
* the rendered strings. A board that publishes no pay still sends the
|
|
145
|
+
* object, with null summaries and empty arrays, so a miss here has to
|
|
146
|
+
* return null for the normalizer's text fallback to run.
|
|
147
|
+
*/
|
|
113
148
|
function parseAshbyCompensation(comp) {
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
if (typeof comp === 'string') {
|
|
117
|
-
const match = comp.match(/\$?([\d,]+)\s*[-–]\s*\$?([\d,]+)/);
|
|
118
|
-
if (!match) return null;
|
|
149
|
+
const salary = (comp.summaryComponents || []).find(c => c.compensationType === 'Salary');
|
|
150
|
+
if (salary && (salary.minValue != null || salary.maxValue != null)) {
|
|
119
151
|
return {
|
|
120
|
-
min:
|
|
121
|
-
max:
|
|
122
|
-
currency: 'USD',
|
|
152
|
+
min: salary.minValue ?? null,
|
|
153
|
+
max: salary.maxValue ?? null,
|
|
154
|
+
currency: salary.currencyCode || 'USD',
|
|
155
|
+
period: INTERVAL_PERIOD[salary.interval] || null,
|
|
156
|
+
source: 'ats',
|
|
123
157
|
};
|
|
124
158
|
}
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
159
|
+
// The summaries are still the ATS's own compensation field, so a range
|
|
160
|
+
// read out of one counts as source 'ats'.
|
|
161
|
+
const parsed = extractSalaryFromText(comp.scrapeableCompensationSalarySummary)
|
|
162
|
+
|| extractSalaryFromText(comp.compensationTierSummary);
|
|
163
|
+
return parsed ? { ...parsed, source: 'ats' } : null;
|
|
129
164
|
}
|
|
130
165
|
|
|
131
166
|
export async function hasAshby(slug) {
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize,
|
|
1
|
+
import { normalize, decodeEntities } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
|
|
4
4
|
const BASE_URL = 'https://boards-api.greenhouse.io/v1/boards';
|
|
@@ -29,19 +29,42 @@ export async function fetchGreenhouse(slug) {
|
|
|
29
29
|
title: job.title || '',
|
|
30
30
|
department: job.departments?.[0]?.name || '',
|
|
31
31
|
location: job.location?.name || '',
|
|
32
|
-
|
|
32
|
+
workplace: parseGreenhouseWorkplace(job.metadata),
|
|
33
|
+
// `content` arrives HTML-escaped (`<p>`). Decode that outer layer
|
|
34
|
+
// once so normalize() sees real tags; it strips and decodes the rest.
|
|
35
|
+
description: decodeEntities(job.content || ''),
|
|
33
36
|
url: job.absolute_url || '',
|
|
34
|
-
|
|
35
|
-
|
|
37
|
+
// updated_at is an edit time that many boards bulk-refresh, so it is not
|
|
38
|
+
// a posting date. first_published is. Fallback covers boards without it (#69).
|
|
39
|
+
postedAt: job.first_published || job.updated_at || null,
|
|
40
|
+
salary: null, // list endpoint has no structured pay; normalizer parses the pay transparency text
|
|
36
41
|
metadata: {
|
|
37
42
|
greenhouseId: job.id,
|
|
38
43
|
internal_job_id: job.internal_job_id,
|
|
39
44
|
departments: job.departments?.map(d => d.name) || [],
|
|
40
45
|
offices: job.offices?.map(o => o.name) || [],
|
|
46
|
+
updatedAt: job.updated_at,
|
|
41
47
|
},
|
|
42
48
|
}, 'greenhouse'));
|
|
43
49
|
}
|
|
44
50
|
|
|
51
|
+
/**
|
|
52
|
+
* Greenhouse has no native workplace field. Boards that track it define a
|
|
53
|
+
* custom field ("Location Type", "Workplace Type") that arrives in the
|
|
54
|
+
* job's `metadata[]`, with `value` a string for single-select fields and
|
|
55
|
+
* an array for multi-select. Values seen: On-Site, Hybrid (Travel-Required),
|
|
56
|
+
* Remote. Anything else is no signal and the location string decides.
|
|
57
|
+
*/
|
|
58
|
+
function parseGreenhouseWorkplace(metadata) {
|
|
59
|
+
const field = (metadata || []).find(m => /location type|workplace type/i.test(m?.name || ''));
|
|
60
|
+
if (!field) return null;
|
|
61
|
+
const value = [].concat(field.value ?? []).join(' ').toLowerCase();
|
|
62
|
+
if (/remote/.test(value)) return 'remote';
|
|
63
|
+
if (/hybrid/.test(value)) return 'hybrid';
|
|
64
|
+
if (/on-?site/.test(value)) return 'onsite';
|
|
65
|
+
return null;
|
|
66
|
+
}
|
|
67
|
+
|
|
45
68
|
/**
|
|
46
69
|
* Check if a company has a Greenhouse board.
|
|
47
70
|
*/
|
package/src/adapters/lever.js
CHANGED
|
@@ -1,11 +1,16 @@
|
|
|
1
|
-
import { normalize,
|
|
1
|
+
import { normalize, extractSalaryFromText } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
|
|
4
4
|
const BASE_URL = 'https://api.lever.co/v0/postings';
|
|
5
5
|
|
|
6
|
+
const PERIODS = { 'per-year-salary': 'year', 'per-month-salary': 'month', 'per-hour-wage': 'hour' };
|
|
7
|
+
// Lever's workplaceType is one of these or 'unspecified'.
|
|
8
|
+
const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
|
|
9
|
+
|
|
6
10
|
/**
|
|
7
11
|
* Fetch all jobs from a Lever job board.
|
|
8
12
|
* Public API, no auth required.
|
|
13
|
+
* Docs: https://github.com/lever/postings-api
|
|
9
14
|
*
|
|
10
15
|
* @param {string} slug - Company slug (e.g., 'stripe', 'figma')
|
|
11
16
|
* @returns {Promise<Array>} Normalized job objects
|
|
@@ -22,30 +27,72 @@ export async function fetchLever(slug) {
|
|
|
22
27
|
const jobs = await resp.json();
|
|
23
28
|
if (!Array.isArray(jobs)) return [];
|
|
24
29
|
|
|
25
|
-
return jobs.map(job => {
|
|
26
|
-
|
|
30
|
+
return jobs.map(job => normalize({
|
|
31
|
+
companySlug: slug,
|
|
32
|
+
// Lever's API doesn't return the company name at the board or job level,
|
|
33
|
+
// so the slug is the honest fallback. `categories.team` is the team within
|
|
34
|
+
// the company ("Payments Platform"), not the company itself.
|
|
35
|
+
company: titleCaseSlug(slug),
|
|
36
|
+
title: job.text || '',
|
|
37
|
+
department: job.categories?.department || job.categories?.team || '',
|
|
38
|
+
location: job.categories?.location || '',
|
|
39
|
+
locations: job.categories?.allLocations || [],
|
|
40
|
+
workplace: WORKPLACE_TYPES.has(job.workplaceType) ? job.workplaceType : null,
|
|
41
|
+
description: buildDescription(job),
|
|
42
|
+
url: job.hostedUrl || '',
|
|
43
|
+
postedAt: job.createdAt ? new Date(job.createdAt).toISOString() : null,
|
|
44
|
+
salary: parseLeverSalary(job.salaryRange, job.text),
|
|
45
|
+
metadata: {
|
|
46
|
+
leverId: job.id,
|
|
47
|
+
team: job.categories?.team || '',
|
|
48
|
+
commitment: job.categories?.commitment || '', // Full-time, Part-time, etc.
|
|
49
|
+
workplaceType: job.workplaceType || '',
|
|
50
|
+
salaryDescription: job.salaryDescriptionPlain || '',
|
|
51
|
+
},
|
|
52
|
+
}, 'lever'));
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Lever splits a posting across `description` (company intro plus overview),
|
|
57
|
+
* `lists` (one `{text, content}` per section: responsibilities, requirements,
|
|
58
|
+
* location details) and `additional` (benefits, EEO). Only the first used to
|
|
59
|
+
* reach the description, so requirements were invisible to filters and to
|
|
60
|
+
* the assistant (issue #64). Reassemble the whole posting as HTML and let
|
|
61
|
+
* normalize() render the headings and bullets.
|
|
62
|
+
*/
|
|
63
|
+
function buildDescription(job) {
|
|
64
|
+
const parts = [job.description || job.descriptionPlain || ''];
|
|
65
|
+
for (const list of job.lists || []) {
|
|
66
|
+
const heading = (list.text || '').trim();
|
|
67
|
+
parts.push((heading ? `<h3>${escapeHtml(heading)}</h3>` : '') + (list.content || ''));
|
|
68
|
+
}
|
|
69
|
+
parts.push(job.additional || job.additionalPlain || '');
|
|
70
|
+
return parts.filter(Boolean).join('\n');
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
// `lists[].text` is plain text ("Skills & Experience"). Escaped, normalize()
|
|
74
|
+
// decodes it back; raw, a stray `<` would be stripped as a tag.
|
|
75
|
+
function escapeHtml(s) {
|
|
76
|
+
return s.replace(/&/g, '&').replace(/</g, '<').replace(/>/g, '>');
|
|
77
|
+
}
|
|
27
78
|
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
workplaceType: job.workplaceType || '',
|
|
46
|
-
},
|
|
47
|
-
}, 'lever');
|
|
48
|
-
});
|
|
79
|
+
/**
|
|
80
|
+
* Lever publishes `salaryRange: {min, max, currency, interval}` on boards
|
|
81
|
+
* that state pay. Boards that don't sometimes put the range in the title.
|
|
82
|
+
*/
|
|
83
|
+
function parseLeverSalary(range, title) {
|
|
84
|
+
const min = range?.min || null;
|
|
85
|
+
const max = range?.max || null;
|
|
86
|
+
if (min || max) {
|
|
87
|
+
return {
|
|
88
|
+
min,
|
|
89
|
+
max,
|
|
90
|
+
currency: range.currency || 'USD',
|
|
91
|
+
period: PERIODS[range.interval] || null,
|
|
92
|
+
source: 'ats',
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
return extractSalaryFromText(title || '');
|
|
49
96
|
}
|
|
50
97
|
|
|
51
98
|
function titleCaseSlug(slug) {
|
|
@@ -55,19 +102,6 @@ function titleCaseSlug(slug) {
|
|
|
55
102
|
return slug.charAt(0).toUpperCase() + slug.slice(1);
|
|
56
103
|
}
|
|
57
104
|
|
|
58
|
-
function parseLeverSalary(commitment, title) {
|
|
59
|
-
// Lever doesn't have a salary field, but sometimes it's in the title
|
|
60
|
-
const match = (title || '').match(/\$[\d,]+\s*[-–]\s*\$[\d,]+/);
|
|
61
|
-
if (!match) return null;
|
|
62
|
-
const nums = match[0].match(/[\d,]+/g);
|
|
63
|
-
if (!nums || nums.length < 2) return null;
|
|
64
|
-
return {
|
|
65
|
-
min: parseInt(nums[0].replace(/,/g, '')),
|
|
66
|
-
max: parseInt(nums[1].replace(/,/g, '')),
|
|
67
|
-
currency: 'USD',
|
|
68
|
-
};
|
|
69
|
-
}
|
|
70
|
-
|
|
71
105
|
export async function hasLever(slug) {
|
|
72
106
|
try {
|
|
73
107
|
const resp = await fetch(`${BASE_URL}/${slug}?mode=json`, { method: 'HEAD' });
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize
|
|
1
|
+
import { normalize } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
@@ -6,9 +6,16 @@ import { atsErrorFromStatus } from '../errors.js';
|
|
|
6
6
|
* Public API, no auth required.
|
|
7
7
|
* Docs: https://docs.recruitee.com/reference/offers
|
|
8
8
|
*
|
|
9
|
-
* Single GET returns every offer
|
|
10
|
-
*
|
|
11
|
-
*
|
|
9
|
+
* Single GET returns every offer inline — no N+1 (unlike SmartRecruiters),
|
|
10
|
+
* no XML (unlike TeamTailor/Personio). The simplest adapter shape in the
|
|
11
|
+
* toolkit.
|
|
12
|
+
*
|
|
13
|
+
* Each offer carries two HTML fields, `description` and `requirements`.
|
|
14
|
+
* Which one holds the role depends on the tenant's template (and sometimes
|
|
15
|
+
* the posting): some keep the duties in `description` and the candidate
|
|
16
|
+
* profile in `requirements`, others put a company intro in `description`
|
|
17
|
+
* and everything else in `requirements`. Neither alone is the posting, so
|
|
18
|
+
* both are joined before normalize() strips them (issue #65).
|
|
12
19
|
*
|
|
13
20
|
* @param {string} slug - Recruitee company subdomain (e.g., 'vandebron')
|
|
14
21
|
* @returns {Promise<Array>} Normalized job objects
|
|
@@ -30,13 +37,7 @@ export async function fetchRecruitee(slug) {
|
|
|
30
37
|
let location = place;
|
|
31
38
|
if (offer.remote) location = place ? `Remote - ${place}` : 'Remote';
|
|
32
39
|
|
|
33
|
-
|
|
34
|
-
if (offer.created_at) {
|
|
35
|
-
// Recruitee returns "2026-05-13 07:38:11 UTC"; coerce to ISO.
|
|
36
|
-
const iso = offer.created_at.replace(' UTC', 'Z').replace(' ', 'T');
|
|
37
|
-
const d = new Date(iso);
|
|
38
|
-
if (!Number.isNaN(d.getTime())) postedAt = d.toISOString();
|
|
39
|
-
}
|
|
40
|
+
const createdAt = toIso(offer.created_at);
|
|
40
41
|
|
|
41
42
|
return normalize({
|
|
42
43
|
companySlug: slug,
|
|
@@ -44,19 +45,71 @@ export async function fetchRecruitee(slug) {
|
|
|
44
45
|
title: offer.title || '',
|
|
45
46
|
department: offer.department || '',
|
|
46
47
|
location,
|
|
47
|
-
|
|
48
|
+
locations: (offer.locations || []).map(l => [l.city, l.country].filter(Boolean).join(', ')),
|
|
49
|
+
workplace: parseRecruiteeWorkplace(offer),
|
|
50
|
+
description: [offer.description, offer.requirements].filter(Boolean).join('\n'),
|
|
48
51
|
url: offer.careers_url || offer.careers_apply_url || '',
|
|
49
|
-
|
|
50
|
-
|
|
52
|
+
// created_at can predate publication by years on long-lived offers,
|
|
53
|
+
// so it is not a posting date. published_at is.
|
|
54
|
+
postedAt: toIso(offer.published_at) || createdAt,
|
|
55
|
+
salary: parseRecruiteeSalary(offer.salary),
|
|
51
56
|
metadata: {
|
|
52
57
|
recruiteeId: offer.guid || offer.id,
|
|
53
58
|
employmentType: offer.employment_type_code || '',
|
|
54
59
|
category: offer.category_code || '',
|
|
60
|
+
createdAt,
|
|
55
61
|
},
|
|
56
62
|
}, 'recruitee');
|
|
57
63
|
});
|
|
58
64
|
}
|
|
59
65
|
|
|
66
|
+
/**
|
|
67
|
+
* Recruitee returns "2026-05-13 07:38:11 UTC"; coerce to ISO.
|
|
68
|
+
*/
|
|
69
|
+
function toIso(ts) {
|
|
70
|
+
if (!ts) return null;
|
|
71
|
+
const d = new Date(ts.replace(' UTC', 'Z').replace(' ', 'T'));
|
|
72
|
+
return Number.isNaN(d.getTime()) ? null : d.toISOString();
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Recruitee sends three booleans, not one enum. Hybrid wins when remote is
|
|
77
|
+
* also set, and on_site alone is onsite. All false is no signal.
|
|
78
|
+
*/
|
|
79
|
+
function parseRecruiteeWorkplace(offer) {
|
|
80
|
+
if (offer.hybrid) return 'hybrid';
|
|
81
|
+
if (offer.remote) return 'remote';
|
|
82
|
+
if (offer.on_site) return 'onsite';
|
|
83
|
+
return null;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
const PERIODS = new Set(['year', 'month', 'hour']);
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* Recruitee sends `salary` as `{min, max, period, currency}` with string
|
|
90
|
+
* amounts. Offers without pay still carry the object, either all-null or
|
|
91
|
+
* as a "0"/"0" placeholder, so anything without a positive side returns
|
|
92
|
+
* null and normalize() falls back to the posting text.
|
|
93
|
+
*/
|
|
94
|
+
function parseRecruiteeSalary(salary) {
|
|
95
|
+
if (!salary) return null;
|
|
96
|
+
const min = toAmount(salary.min);
|
|
97
|
+
const max = toAmount(salary.max);
|
|
98
|
+
if (min === null && max === null) return null;
|
|
99
|
+
return {
|
|
100
|
+
min,
|
|
101
|
+
max,
|
|
102
|
+
currency: (salary.currency || '').toUpperCase(),
|
|
103
|
+
period: PERIODS.has(salary.period) ? salary.period : null,
|
|
104
|
+
source: 'ats',
|
|
105
|
+
};
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
function toAmount(value) {
|
|
109
|
+
const n = parseFloat(value);
|
|
110
|
+
return Number.isFinite(n) && n > 0 ? n : null;
|
|
111
|
+
}
|
|
112
|
+
|
|
60
113
|
/**
|
|
61
114
|
* Check if a company has a Recruitee career site.
|
|
62
115
|
*/
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize
|
|
1
|
+
import { normalize } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
|
|
4
4
|
const BASE_URL = 'https://api.smartrecruiters.com/v1/companies';
|
|
@@ -10,7 +10,8 @@ const PAGE_SIZE = 100;
|
|
|
10
10
|
* Docs: https://developers.smartrecruiters.com/reference/postingsget-1
|
|
11
11
|
*
|
|
12
12
|
* Two-step flow (unavoidable N+1):
|
|
13
|
-
* - The postings LIST endpoint omits the job description entirely
|
|
13
|
+
* - The postings LIST endpoint omits the job description entirely,
|
|
14
|
+
* and the structured `compensation` block with it.
|
|
14
15
|
* - jd-intel's contract is "full JD text", so we must fetch each
|
|
15
16
|
* posting's DETAIL endpoint to get jobAd.sections.
|
|
16
17
|
* Large enterprise tenants with hundreds of openings will therefore be
|
|
@@ -46,6 +47,7 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
46
47
|
const jobs = await Promise.all(postings.map(async (p) => {
|
|
47
48
|
let sections = {};
|
|
48
49
|
let postingUrl = '';
|
|
50
|
+
let salary = null;
|
|
49
51
|
|
|
50
52
|
try {
|
|
51
53
|
const detailResp = await fetch(`${BASE_URL}/${slug}/postings/${p.id}`);
|
|
@@ -53,6 +55,7 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
53
55
|
const detail = await detailResp.json();
|
|
54
56
|
sections = detail.jobAd?.sections || {};
|
|
55
57
|
postingUrl = detail.postingUrl || detail.applyUrl || '';
|
|
58
|
+
salary = parseCompensation(detail.compensation);
|
|
56
59
|
}
|
|
57
60
|
} catch {
|
|
58
61
|
// Detail fetch failed: fall back to list-only fields (no description).
|
|
@@ -68,8 +71,14 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
68
71
|
const place = loc.fullLocation
|
|
69
72
|
|| [loc.city, loc.region, loc.country].filter(Boolean).join(', ');
|
|
70
73
|
let location = place;
|
|
71
|
-
|
|
72
|
-
|
|
74
|
+
let workplace = null;
|
|
75
|
+
if (loc.remote) {
|
|
76
|
+
location = `Remote - ${place}`.replace(/ - $/, ' ');
|
|
77
|
+
workplace = 'remote';
|
|
78
|
+
} else if (loc.hybrid) {
|
|
79
|
+
location = `Hybrid - ${place}`.replace(/ - $/, ' ');
|
|
80
|
+
workplace = 'hybrid';
|
|
81
|
+
}
|
|
73
82
|
|
|
74
83
|
return normalize({
|
|
75
84
|
companySlug: slug,
|
|
@@ -77,10 +86,11 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
77
86
|
title: p.name || '',
|
|
78
87
|
department: p.department?.label || p.function?.label || '',
|
|
79
88
|
location,
|
|
80
|
-
|
|
89
|
+
workplace,
|
|
90
|
+
description,
|
|
81
91
|
url: postingUrl,
|
|
82
92
|
postedAt: p.releasedDate || null,
|
|
83
|
-
salary
|
|
93
|
+
salary, // null when the detail has no compensation; normalize() then parses text
|
|
84
94
|
metadata: {
|
|
85
95
|
smartRecruitersId: p.id,
|
|
86
96
|
refNumber: p.refNumber || '',
|
|
@@ -94,6 +104,31 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
94
104
|
return jobs;
|
|
95
105
|
}
|
|
96
106
|
|
|
107
|
+
const PERIODS = { YEARLY: 'year', MONTHLY: 'month', HOURLY: 'hour' };
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* Map the detail response's `compensation` to the shared salary shape.
|
|
111
|
+
*
|
|
112
|
+
* SmartRecruiters publishes `{min?, max?, currency, period}`, and both
|
|
113
|
+
* one-sided cases occur (a "max only" cap, a "from" floor), so each bound
|
|
114
|
+
* is passed through as null when absent rather than dropping the whole
|
|
115
|
+
* range. The period is kept as published: a MONTHLY figure is not
|
|
116
|
+
* annualized because tenants occasionally mislabel it (issue #70).
|
|
117
|
+
*/
|
|
118
|
+
function parseCompensation(comp) {
|
|
119
|
+
if (!comp || !comp.currency) return null;
|
|
120
|
+
const min = Number.isFinite(comp.min) ? comp.min : null;
|
|
121
|
+
const max = Number.isFinite(comp.max) ? comp.max : null;
|
|
122
|
+
if (min === null && max === null) return null;
|
|
123
|
+
return {
|
|
124
|
+
min,
|
|
125
|
+
max,
|
|
126
|
+
currency: comp.currency,
|
|
127
|
+
period: PERIODS[comp.period] ?? null,
|
|
128
|
+
source: 'ats',
|
|
129
|
+
};
|
|
130
|
+
}
|
|
131
|
+
|
|
97
132
|
/**
|
|
98
133
|
* Check if a company exists on SmartRecruiters.
|
|
99
134
|
* (HEAD isn't reliably supported on the postings endpoint, so use a
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize,
|
|
1
|
+
import { normalize, decodeEntities } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
@@ -16,10 +16,11 @@ import { atsErrorFromStatus } from '../errors.js';
|
|
|
16
16
|
* to a custom domain (e.g. jobs.tibber.com).
|
|
17
17
|
*
|
|
18
18
|
* RSS quirk: descriptions are HTML-entity-encoded inside the XML
|
|
19
|
-
* (`<p>...`). We decode that outer layer to real HTML
|
|
20
|
-
*
|
|
21
|
-
* entities. Decode order matters —
|
|
22
|
-
* double-encoded sequences (`&amp;`)
|
|
19
|
+
* (`<p>...`). We decode that outer layer to real HTML with the
|
|
20
|
+
* shared decodeEntities() and hand the HTML to normalize(), which
|
|
21
|
+
* strips tags and resolves the inner entities. Decode order matters —
|
|
22
|
+
* `&` resolves LAST so double-encoded sequences (`&amp;`)
|
|
23
|
+
* collapse by one layer per pass.
|
|
23
24
|
*
|
|
24
25
|
* @param {string} slug - TeamTailor career-site slug (e.g., 'tibber')
|
|
25
26
|
* @returns {Promise<Array>} Normalized job objects
|
|
@@ -28,6 +29,9 @@ import { atsErrorFromStatus } from '../errors.js';
|
|
|
28
29
|
// segment, e.g. crunchbase.na.teamtailor.com. '' is the base host.
|
|
29
30
|
const TT_REGIONS = ['', 'na', 'eu'];
|
|
30
31
|
|
|
32
|
+
// Feeds send `none`, `hybrid`, `fully` or `onsite`. `none` is no signal.
|
|
33
|
+
const REMOTE_STATUS = { hybrid: 'hybrid', fully: 'remote', onsite: 'onsite' };
|
|
34
|
+
|
|
31
35
|
/**
|
|
32
36
|
* Resolve which TeamTailor host actually serves this slug's feed.
|
|
33
37
|
* Returns the first 200 Response, throws on a non-404 error, or
|
|
@@ -64,8 +68,8 @@ export async function fetchTeamtailor(slug) {
|
|
|
64
68
|
const items = [...xml.matchAll(/<item>([\s\S]*?)<\/item>/g)].map(m => m[1]);
|
|
65
69
|
|
|
66
70
|
return items.map(item => {
|
|
67
|
-
const pick = (tag) => {
|
|
68
|
-
const m =
|
|
71
|
+
const pick = (tag, src = item) => {
|
|
72
|
+
const m = src.match(new RegExp(`<${tag}[^>]*>([\\s\\S]*?)</${tag}>`));
|
|
69
73
|
return m ? m[1].trim() : '';
|
|
70
74
|
};
|
|
71
75
|
|
|
@@ -78,6 +82,12 @@ export async function fetchTeamtailor(slug) {
|
|
|
78
82
|
const country = decodeEntities(pick('tt:country'));
|
|
79
83
|
const remoteStatus = decodeEntities(pick('remoteStatus'));
|
|
80
84
|
|
|
85
|
+
// One <tt:location> per office the posting is open in, read the same
|
|
86
|
+
// way as the primary above so the entries line up.
|
|
87
|
+
const locations = [...item.matchAll(/<tt:location>([\s\S]*?)<\/tt:location>/g)].map(m =>
|
|
88
|
+
[decodeEntities(pick('tt:city', m[1])), decodeEntities(pick('tt:country', m[1]))].filter(Boolean).join(', ')
|
|
89
|
+
);
|
|
90
|
+
|
|
81
91
|
let location = [city, country].filter(Boolean).join(', ');
|
|
82
92
|
if (/remote/i.test(remoteStatus)) {
|
|
83
93
|
location = location ? `Remote - ${location}` : 'Remote';
|
|
@@ -95,7 +105,9 @@ export async function fetchTeamtailor(slug) {
|
|
|
95
105
|
title,
|
|
96
106
|
department,
|
|
97
107
|
location,
|
|
98
|
-
|
|
108
|
+
locations,
|
|
109
|
+
workplace: REMOTE_STATUS[remoteStatus.toLowerCase()] || null,
|
|
110
|
+
description: decodeEntities(pick('description')),
|
|
99
111
|
url: link,
|
|
100
112
|
postedAt,
|
|
101
113
|
salary: null, // No structured salary; normalizer parses from text
|
|
@@ -107,22 +119,6 @@ export async function fetchTeamtailor(slug) {
|
|
|
107
119
|
});
|
|
108
120
|
}
|
|
109
121
|
|
|
110
|
-
/**
|
|
111
|
-
* Decode the RSS entity/CDATA layer to real HTML.
|
|
112
|
-
* `&` is intentionally resolved LAST.
|
|
113
|
-
*/
|
|
114
|
-
function decodeEntities(s) {
|
|
115
|
-
if (!s) return '';
|
|
116
|
-
return s
|
|
117
|
-
.replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1')
|
|
118
|
-
.replace(/</g, '<')
|
|
119
|
-
.replace(/>/g, '>')
|
|
120
|
-
.replace(/"/g, '"')
|
|
121
|
-
.replace(/'/g, "'")
|
|
122
|
-
.replace(/'/g, "'")
|
|
123
|
-
.replace(/&/g, '&');
|
|
124
|
-
}
|
|
125
|
-
|
|
126
122
|
/**
|
|
127
123
|
* Check if a company has a TeamTailor career site.
|
|
128
124
|
*/
|
package/src/adapters/workday.js
CHANGED
|
@@ -1,9 +1,14 @@
|
|
|
1
|
-
import { normalize
|
|
1
|
+
import { normalize } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
|
|
4
4
|
const MAX_DETAIL_FETCHES = 100;
|
|
5
5
|
const LIST_PAGE_SIZE = 20;
|
|
6
|
-
|
|
6
|
+
// Upper bound on list pages per call: 100 pages of 20 = at most 2000
|
|
7
|
+
// postings scanned. Paging usually stops sooner, on a short page or when
|
|
8
|
+
// offset reaches the first page's total (see the loop below).
|
|
9
|
+
const LIST_PAGE_HARD_CAP = 100;
|
|
10
|
+
// A multi-location posting's list row reads "2 Locations", "14 Locations".
|
|
11
|
+
const MULTI_LOCATION = /^\s*\d+\s+locations?\s*$/;
|
|
7
12
|
|
|
8
13
|
/**
|
|
9
14
|
* Fetch jobs from a Workday tenant via the public "CXS" JSON API.
|
|
@@ -40,6 +45,7 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
40
45
|
const postings = [];
|
|
41
46
|
let offset = 0;
|
|
42
47
|
let pages = 0;
|
|
48
|
+
let firstTotal = 0;
|
|
43
49
|
while (pages < LIST_PAGE_HARD_CAP) {
|
|
44
50
|
const resp = await fetch(`${base}/jobs`, {
|
|
45
51
|
method: 'POST',
|
|
@@ -48,19 +54,24 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
48
54
|
});
|
|
49
55
|
|
|
50
56
|
if (!resp.ok) {
|
|
51
|
-
if (resp.status === 404) return []; // wrong site / no such board
|
|
52
57
|
if (offset === 0) {
|
|
58
|
+
if (resp.status === 404) return []; // wrong site / no such board
|
|
53
59
|
throw atsErrorFromStatus(resp.status, `Workday API error for ${slug} (${tenant}/${env}/${site}): ${resp.status}`);
|
|
54
60
|
}
|
|
55
|
-
break; // mid-paging failure: keep what we have
|
|
61
|
+
break; // mid-paging failure (any status): keep what we have
|
|
56
62
|
}
|
|
57
63
|
|
|
58
64
|
const data = await resp.json();
|
|
59
65
|
const page = data.jobPostings || [];
|
|
66
|
+
// Some tenants report the real `total` only at offset 0 and send
|
|
67
|
+
// `total: 0` on every later page, so only the first page's figure
|
|
68
|
+
// is trusted. A short page is the other stop signal.
|
|
69
|
+
if (pages === 0) firstTotal = data.total || 0;
|
|
60
70
|
postings.push(...page);
|
|
61
71
|
pages += 1;
|
|
62
72
|
offset += LIST_PAGE_SIZE;
|
|
63
|
-
if (page.length
|
|
73
|
+
if (page.length < LIST_PAGE_SIZE) break;
|
|
74
|
+
if (firstTotal > 0 && offset >= firstTotal) break;
|
|
64
75
|
}
|
|
65
76
|
|
|
66
77
|
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost.
|
|
@@ -76,6 +87,9 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
76
87
|
const inc = fc.locationIncludes.map(s => String(s).toLowerCase());
|
|
77
88
|
candidates = candidates.filter(p => {
|
|
78
89
|
const loc = (p.locationsText || '').toLowerCase();
|
|
90
|
+
// "2 Locations" says nothing about where. The row stays a candidate
|
|
91
|
+
// and the pass after hydration decides on the detail's location list.
|
|
92
|
+
if (MULTI_LOCATION.test(loc)) return true;
|
|
79
93
|
return inc.some(s => loc.includes(s));
|
|
80
94
|
});
|
|
81
95
|
}
|
|
@@ -91,16 +105,21 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
91
105
|
}
|
|
92
106
|
|
|
93
107
|
// 3. Bound the detail-fetch set.
|
|
94
|
-
// NOTE: huge-tenant coverage is intentionally capped for v1
|
|
95
|
-
//
|
|
96
|
-
//
|
|
97
|
-
//
|
|
98
|
-
//
|
|
99
|
-
//
|
|
100
|
-
//
|
|
101
|
-
//
|
|
108
|
+
// NOTE: huge-tenant coverage is intentionally capped for v1. Two
|
|
109
|
+
// caps apply: the list scan above stops at LIST_PAGE_HARD_CAP pages
|
|
110
|
+
// (2000 postings, enough for Salesforce's ~1398), and the detail set
|
|
111
|
+
// is cut to MAX_DETAIL_FETCHES here. A description `filter` is
|
|
112
|
+
// applied by the library AFTER this returns, so for that case we
|
|
113
|
+
// keep the full backstop instead of truncating tightly to `limit`
|
|
114
|
+
// (which could hydrate jobs that all fail the regex while better
|
|
115
|
+
// matches go unscanned). The library pages with `offset` after this
|
|
116
|
+
// returns, so the budget covers the page plus what precedes it.
|
|
117
|
+
// Proper fix (smart pagination / rate-limited concurrency / surfaced
|
|
118
|
+
// truncation) is tracked in #26, to be designed alongside
|
|
119
|
+
// retry/rate-limit work (#7).
|
|
102
120
|
const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
|
|
103
|
-
const
|
|
121
|
+
const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
|
|
122
|
+
const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
|
|
104
123
|
candidates = candidates.slice(0, cap);
|
|
105
124
|
|
|
106
125
|
// 4. Hydrate descriptions via the per-posting detail endpoint.
|
|
@@ -126,7 +145,9 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
126
145
|
title: p.title || info.title || '',
|
|
127
146
|
department: '',
|
|
128
147
|
location: info.location || p.locationsText || '',
|
|
129
|
-
|
|
148
|
+
locations: info.additionalLocations || [],
|
|
149
|
+
workplace: parseWorkdayRemoteType(info.remoteType),
|
|
150
|
+
description: info.jobDescription || '',
|
|
130
151
|
url: `https://${tenant}.${env}.myworkdayjobs.com/${site}${externalPath}`,
|
|
131
152
|
postedAt: parseWorkdayDate(info.startDate) || normalizePostedOn(p.postedOn),
|
|
132
153
|
salary: null, // normalizer extracts from description text
|
|
@@ -142,6 +163,20 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
142
163
|
return jobs;
|
|
143
164
|
}
|
|
144
165
|
|
|
166
|
+
/**
|
|
167
|
+
* Detail `remoteType` is free text set per tenant: "Remote", "Hybrid",
|
|
168
|
+
* "Office - Flexible", "On-site". A flexible office arrangement counts as
|
|
169
|
+
* hybrid, so that check runs before the office one. Some tenants send no
|
|
170
|
+
* value at all; the location string decides then.
|
|
171
|
+
*/
|
|
172
|
+
function parseWorkdayRemoteType(remoteType) {
|
|
173
|
+
const s = String(remoteType || '').toLowerCase();
|
|
174
|
+
if (/remote/.test(s)) return 'remote';
|
|
175
|
+
if (/hybrid|flexible/.test(s)) return 'hybrid';
|
|
176
|
+
if (/office|on-?site/.test(s)) return 'onsite';
|
|
177
|
+
return null;
|
|
178
|
+
}
|
|
179
|
+
|
|
145
180
|
/**
|
|
146
181
|
* Workday list `postedOn` is a relative string ("Posted Today",
|
|
147
182
|
* "Posted 5 Days Ago", "Posted 30+ Days Ago"). Decide membership in
|
package/src/cli.js
CHANGED
|
@@ -9,6 +9,8 @@
|
|
|
9
9
|
* jd-intel registry search <query>
|
|
10
10
|
*/
|
|
11
11
|
|
|
12
|
+
import { realpathSync } from 'node:fs';
|
|
13
|
+
import { fileURLToPath } from 'node:url';
|
|
12
14
|
import { fetchJobs } from './index.js';
|
|
13
15
|
import { detectAts, searchRegistry } from './registry.js';
|
|
14
16
|
|
|
@@ -85,7 +87,7 @@ async function main() {
|
|
|
85
87
|
console.log(`Found ${jobs.length} jobs\n`);
|
|
86
88
|
|
|
87
89
|
for (const job of jobs.slice(0, 20)) {
|
|
88
|
-
const salary = job.salary ? ` |
|
|
90
|
+
const salary = job.salary ? ` | ${formatSalary(job.salary)}` : '';
|
|
89
91
|
const loc = job.location ? ` | ${job.location}` : '';
|
|
90
92
|
const dept = job.department ? ` [${job.department}]` : '';
|
|
91
93
|
console.log(` ${job.title}${dept}${loc}${salary}`);
|
|
@@ -162,8 +164,8 @@ Fetch options:
|
|
|
162
164
|
--title-filter pattern Regex matched against TITLE only (role identity)
|
|
163
165
|
--filter pattern Regex matched across title, department, description (topic/scope)
|
|
164
166
|
--posted-within-days N Only jobs posted in the last N days
|
|
165
|
-
--location-include "A,B,C" Keep jobs
|
|
166
|
-
--location-exclude "A,B,C" Drop jobs
|
|
167
|
+
--location-include "A,B,C" Keep jobs where any listed location contains one of these
|
|
168
|
+
--location-exclude "A,B,C" Drop jobs only when every listed location contains one of these
|
|
167
169
|
--limit N Cap results (default 100)
|
|
168
170
|
--json Output full JSON
|
|
169
171
|
|
|
@@ -184,7 +186,32 @@ Examples:
|
|
|
184
186
|
}
|
|
185
187
|
}
|
|
186
188
|
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
189
|
+
export function formatSalary({ min, max, currency, period }) {
|
|
190
|
+
const hasMin = min != null;
|
|
191
|
+
const hasMax = max != null;
|
|
192
|
+
let range;
|
|
193
|
+
if (hasMin && hasMax) range = `${min.toLocaleString()}-${max.toLocaleString()}`;
|
|
194
|
+
else if (hasMin) range = `from ${min.toLocaleString()}`;
|
|
195
|
+
else range = `up to ${max.toLocaleString()}`;
|
|
196
|
+
const unit = period === 'hour' ? '/hr' : period === 'month' ? '/mo' : '';
|
|
197
|
+
return `${range} ${currency}${unit}`;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
// Boot only when this file is the script Node was started with, so a test
|
|
201
|
+
// can import formatSalary without running a command. argv[1] is resolved
|
|
202
|
+
// through realpath because npm installs the bin as a symlink into .bin/,
|
|
203
|
+
// while import.meta.url already points at the real file.
|
|
204
|
+
function isEntrypoint() {
|
|
205
|
+
try {
|
|
206
|
+
return realpathSync(process.argv[1]) === fileURLToPath(import.meta.url);
|
|
207
|
+
} catch {
|
|
208
|
+
return false;
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
if (isEntrypoint()) {
|
|
213
|
+
main().catch(err => {
|
|
214
|
+
console.error('Error:', err.message);
|
|
215
|
+
process.exit(1);
|
|
216
|
+
});
|
|
217
|
+
}
|
package/src/filters.js
CHANGED
|
@@ -4,14 +4,33 @@
|
|
|
4
4
|
* Facts go here (deterministic field matches). Interpretations stay with the
|
|
5
5
|
* caller — this module does substring matching on structured fields, nothing
|
|
6
6
|
* semantic.
|
|
7
|
+
*
|
|
8
|
+
* Returns the page as an array. applyFiltersDetailed returns the same page
|
|
9
|
+
* plus total_matched, the match count before offset and limit.
|
|
7
10
|
*/
|
|
8
11
|
export function applyFilters(jobs, options = {}) {
|
|
12
|
+
return applyFiltersDetailed(jobs, options).jobs;
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* Filter, sort, then page.
|
|
17
|
+
*
|
|
18
|
+
* Order is applied after the filters and before offset and limit, so a cut
|
|
19
|
+
* drops the oldest matches first. 'newest' sorts by postedAt descending with
|
|
20
|
+
* undated jobs last and ties broken by id, which keeps pages deterministic.
|
|
21
|
+
* 'board' keeps the order the adapter returned.
|
|
22
|
+
*
|
|
23
|
+
* @returns {{ jobs: Array, total_matched: number }}
|
|
24
|
+
*/
|
|
25
|
+
export function applyFiltersDetailed(jobs, options = {}) {
|
|
9
26
|
const {
|
|
10
27
|
titleFilter,
|
|
11
28
|
filter,
|
|
12
29
|
postedWithinDays,
|
|
13
30
|
locationIncludes,
|
|
14
31
|
locationExcludes,
|
|
32
|
+
order = 'newest',
|
|
33
|
+
offset = 0,
|
|
15
34
|
limit = 100,
|
|
16
35
|
} = options;
|
|
17
36
|
|
|
@@ -42,25 +61,60 @@ export function applyFilters(jobs, options = {}) {
|
|
|
42
61
|
|
|
43
62
|
if (Array.isArray(locationIncludes) && locationIncludes.length > 0) {
|
|
44
63
|
const matchers = locationIncludes.map(makeLocationMatcher);
|
|
45
|
-
result = result.filter(j =>
|
|
46
|
-
const loc = (j.location || '').toLowerCase();
|
|
47
|
-
return matchers.some(m => m(loc));
|
|
48
|
-
});
|
|
64
|
+
result = result.filter(j => jobLocations(j).some(loc => matchers.some(m => m(loc))));
|
|
49
65
|
}
|
|
50
66
|
|
|
51
67
|
if (Array.isArray(locationExcludes) && locationExcludes.length > 0) {
|
|
52
68
|
const matchers = locationExcludes.map(makeLocationMatcher);
|
|
53
|
-
result = result.filter(j =>
|
|
54
|
-
const loc = (j.location || '').toLowerCase();
|
|
55
|
-
return !matchers.some(m => m(loc));
|
|
56
|
-
});
|
|
69
|
+
result = result.filter(j => !jobLocations(j).every(loc => matchers.some(m => m(loc))));
|
|
57
70
|
}
|
|
58
71
|
|
|
59
|
-
|
|
60
|
-
|
|
72
|
+
const total_matched = result.length;
|
|
73
|
+
|
|
74
|
+
if (order !== 'board') {
|
|
75
|
+
result = [...result].sort(byNewest);
|
|
61
76
|
}
|
|
62
77
|
|
|
63
|
-
|
|
78
|
+
const start = typeof offset === 'number' && offset > 0 ? offset : 0;
|
|
79
|
+
const end = typeof limit === 'number' ? start + limit : undefined;
|
|
80
|
+
if (start > 0 || (end !== undefined && result.length > end)) {
|
|
81
|
+
result = result.slice(start, end);
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
return { jobs: result, total_matched };
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* Every location a job is open in, lowercased. A job passes an include when
|
|
89
|
+
* any of them matches and is dropped by an exclude only when all of them
|
|
90
|
+
* match: a role open in Berlin and New York is still open in Berlin for
|
|
91
|
+
* someone excluding the US (issue #68). Jobs from before `locations`
|
|
92
|
+
* existed fall back to the single `location` string.
|
|
93
|
+
*/
|
|
94
|
+
function jobLocations(job) {
|
|
95
|
+
const list = Array.isArray(job.locations) && job.locations.length > 0
|
|
96
|
+
? job.locations
|
|
97
|
+
: [job.location || ''];
|
|
98
|
+
return list.map(loc => String(loc).toLowerCase());
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
function postedTime(job) {
|
|
102
|
+
if (!job.postedAt) return null;
|
|
103
|
+
const t = new Date(job.postedAt).getTime();
|
|
104
|
+
return Number.isFinite(t) ? t : null;
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
function byNewest(a, b) {
|
|
108
|
+
const ta = postedTime(a);
|
|
109
|
+
const tb = postedTime(b);
|
|
110
|
+
if (ta !== tb) {
|
|
111
|
+
if (ta === null) return 1;
|
|
112
|
+
if (tb === null) return -1;
|
|
113
|
+
return tb - ta;
|
|
114
|
+
}
|
|
115
|
+
const ia = a.id || '';
|
|
116
|
+
const ib = b.id || '';
|
|
117
|
+
return ia < ib ? -1 : ia > ib ? 1 : 0;
|
|
64
118
|
}
|
|
65
119
|
|
|
66
120
|
/**
|
package/src/index.js
CHANGED
|
@@ -8,11 +8,23 @@
|
|
|
8
8
|
|
|
9
9
|
import { ADAPTERS, ATS_NAMES } from './adapters/index.js';
|
|
10
10
|
import { loadRegistry, searchRegistry, detectAts, findAtsBySlug, findEntryBySlug, getRegistrySource } from './registry.js';
|
|
11
|
-
import {
|
|
11
|
+
import { applyFiltersDetailed } from './filters.js';
|
|
12
12
|
|
|
13
13
|
/**
|
|
14
14
|
* Fetch jobs from a company's ATS board.
|
|
15
15
|
*
|
|
16
|
+
* Same options as fetchJobsDetailed; returns the page as an array.
|
|
17
|
+
*
|
|
18
|
+
* @returns {Promise<Array>} Normalized, filtered job objects
|
|
19
|
+
*/
|
|
20
|
+
export async function fetchJobs(options = {}) {
|
|
21
|
+
const { jobs } = await fetchJobsDetailed(options);
|
|
22
|
+
return jobs;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Fetch jobs from a company's ATS board, with the match count.
|
|
27
|
+
*
|
|
16
28
|
* @param {Object} options
|
|
17
29
|
* @param {string} options.company - Company slug or name
|
|
18
30
|
* @param {string} [options.ats] - Specific ATS platform. If omitted, auto-detects.
|
|
@@ -20,12 +32,14 @@ import { applyFilters } from './filters.js';
|
|
|
20
32
|
* @param {string} [options.titleFilter] - Regex matched against title only. Use for role identity ("product manager", "staff engineer").
|
|
21
33
|
* @param {string} [options.filter] - Regex matched across title, department, description. Use for topic/scope.
|
|
22
34
|
* @param {number} [options.postedWithinDays] - Only return jobs posted within N days.
|
|
23
|
-
* @param {string[]} [options.locationIncludes] - Keep jobs
|
|
24
|
-
* @param {string[]} [options.locationExcludes] - Drop jobs
|
|
25
|
-
* @param {
|
|
26
|
-
* @
|
|
35
|
+
* @param {string[]} [options.locationIncludes] - Keep jobs where any listed location contains any of these (case-insensitive).
|
|
36
|
+
* @param {string[]} [options.locationExcludes] - Drop jobs only when every listed location contains one of these (case-insensitive).
|
|
37
|
+
* @param {'newest'|'board'} [options.order='newest'] - 'newest': by postedAt descending, undated last, ties by id. 'board': the adapter's own order.
|
|
38
|
+
* @param {number} [options.offset=0] - Matches to skip after sorting (paging).
|
|
39
|
+
* @param {number} [options.limit=100] - Maximum jobs to return after offset.
|
|
40
|
+
* @returns {Promise<{ jobs: Array, total_matched: number }>} The page, plus the match count before offset and limit
|
|
27
41
|
*/
|
|
28
|
-
export async function
|
|
42
|
+
export async function fetchJobsDetailed({
|
|
29
43
|
company,
|
|
30
44
|
ats,
|
|
31
45
|
config,
|
|
@@ -34,6 +48,8 @@ export async function fetchJobs({
|
|
|
34
48
|
postedWithinDays,
|
|
35
49
|
locationIncludes,
|
|
36
50
|
locationExcludes,
|
|
51
|
+
order = 'newest',
|
|
52
|
+
offset = 0,
|
|
37
53
|
limit = 100,
|
|
38
54
|
} = {}) {
|
|
39
55
|
if (!company) throw new Error('Company slug required');
|
|
@@ -45,7 +61,7 @@ export async function fetchJobs({
|
|
|
45
61
|
// adapters declare fetch{Name}(slug) and ignore extra positional args
|
|
46
62
|
// (JS no-op), so this is backward-compatible. Filter-aware adapters
|
|
47
63
|
// (e.g. Workday) use it to avoid mass detail-fetching on huge tenants.
|
|
48
|
-
const filterContext = { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes, limit };
|
|
64
|
+
const filterContext = { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes, offset, limit };
|
|
49
65
|
|
|
50
66
|
let jobs;
|
|
51
67
|
if (ats) {
|
|
@@ -93,7 +109,7 @@ export async function fetchJobs({
|
|
|
93
109
|
}
|
|
94
110
|
}
|
|
95
111
|
|
|
96
|
-
return
|
|
112
|
+
return applyFiltersDetailed(jobs, { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes, order, offset, limit });
|
|
97
113
|
}
|
|
98
114
|
|
|
99
115
|
/**
|
|
@@ -134,7 +150,7 @@ export { fetchLever } from './adapters/lever.js';
|
|
|
134
150
|
export { fetchAshby } from './adapters/ashby.js';
|
|
135
151
|
|
|
136
152
|
// Re-export filter logic for reuse (e.g., by the MCP server)
|
|
137
|
-
export { applyFilters } from './filters.js';
|
|
153
|
+
export { applyFilters, applyFiltersDetailed } from './filters.js';
|
|
138
154
|
|
|
139
155
|
// Re-export the list of supported ATS names (e.g. so the MCP layer can report
|
|
140
156
|
// the full set detectAts probes, instead of hardcoding a stale subset).
|
package/src/normalizer.js
CHANGED
|
@@ -13,20 +13,35 @@ export function jobId(company, title, ats, location = '') {
|
|
|
13
13
|
|
|
14
14
|
/**
|
|
15
15
|
* Normalize a raw ATS job object into the unified schema.
|
|
16
|
+
*
|
|
17
|
+
* Adapters pass `description` as HTML. This is the one place it is
|
|
18
|
+
* stripped and decoded (issue #66): a second pass would delete text the
|
|
19
|
+
* author escaped on purpose (`<5 years`) and leave entities the first
|
|
20
|
+
* pass exposed (`&mdash;` -> `—`) as literal noise.
|
|
21
|
+
*
|
|
22
|
+
* `raw.workplace` is the ATS's own arrangement, already mapped by the
|
|
23
|
+
* adapter to 'remote' | 'hybrid' | 'onsite', or null when the platform
|
|
24
|
+
* gives no signal. `raw.locations` lists every place the posting is open
|
|
25
|
+
* in; `location` stays the primary because it feeds the id (issue #68).
|
|
16
26
|
*/
|
|
17
27
|
export function normalize(raw, ats) {
|
|
18
28
|
const now = new Date().toISOString();
|
|
29
|
+
const description = stripHtml(raw.description || '');
|
|
30
|
+
const location = raw.location || '';
|
|
31
|
+
const workplace = resolveWorkplace(raw.workplace, location);
|
|
19
32
|
return {
|
|
20
|
-
id: jobId(raw.company || raw.companySlug, raw.title, ats,
|
|
33
|
+
id: jobId(raw.company || raw.companySlug, raw.title, ats, location),
|
|
21
34
|
company: raw.company || raw.companySlug || '',
|
|
22
35
|
companySlug: raw.companySlug || '',
|
|
23
36
|
ats,
|
|
24
37
|
title: raw.title || '',
|
|
25
38
|
department: raw.department || '',
|
|
26
|
-
location
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
39
|
+
location,
|
|
40
|
+
locations: uniqueLocations(location, raw.locations),
|
|
41
|
+
locationType: workplace.type,
|
|
42
|
+
workplace,
|
|
43
|
+
salary: raw.salary || extractSalaryFromText(description),
|
|
44
|
+
description,
|
|
30
45
|
url: raw.url || '',
|
|
31
46
|
postedAt: raw.postedAt || null,
|
|
32
47
|
firstSeen: now,
|
|
@@ -36,62 +51,203 @@ export function normalize(raw, ats) {
|
|
|
36
51
|
};
|
|
37
52
|
}
|
|
38
53
|
|
|
54
|
+
const CURRENCY_CODES = 'USD|EUR|GBP|CAD|AUD|NZD|CHF|SEK|NOK|DKK|PLN|CZK|HUF|INR|SGD|HKD|JPY|CNY|BRL|MXN|ZAR|AED|ILS';
|
|
55
|
+
const SYMBOL_CURRENCY = { $: 'USD', '€': 'EUR', '£': 'GBP' };
|
|
56
|
+
|
|
57
|
+
// A number as job posts write it: 1,234,567 / 1.234.567 / 1234, with an
|
|
58
|
+
// optional one- or two-digit decimal part (211.4, 40.50, 60.000,50).
|
|
59
|
+
// Exactly three digits after a dot are a thousands group, the way Dutch
|
|
60
|
+
// and German boards write it: "€60.000" is sixty thousand, not sixty.
|
|
61
|
+
const NUMBER =
|
|
62
|
+
'\\d{1,3}(?:,\\d{3})+(?:\\.\\d{1,2})?' +
|
|
63
|
+
'|\\d{1,3}(?:\\.\\d{3})+(?:,\\d{1,2})?' +
|
|
64
|
+
'|\\d+(?:[.,]\\d{1,2})?';
|
|
65
|
+
|
|
66
|
+
// One side of a range: optional code before, optional symbol, the number,
|
|
67
|
+
// optional K, optional code after. The lookarounds keep the number from
|
|
68
|
+
// starting or ending inside a longer one ("234.567" out of "1.234.567").
|
|
69
|
+
const amountPattern = (p) =>
|
|
70
|
+
`(?:\\b(?<${p}CodeBefore>${CURRENCY_CODES})\\s?)?` +
|
|
71
|
+
`(?<${p}Sym>[$€£])?\\s?` +
|
|
72
|
+
`(?<![\\d.,])(?<${p}Num>${NUMBER})(?!\\d|[.,]\\d)\\s?` +
|
|
73
|
+
`(?<${p}K>[kK]\\b)?` +
|
|
74
|
+
`(?:\\s?(?<${p}CodeAfter>${CURRENCY_CODES})\\b)?`;
|
|
75
|
+
|
|
76
|
+
const SALARY_RANGE = new RegExp(
|
|
77
|
+
`${amountPattern('lo')}\\s*(?:[-–—]|\\bto\\b)\\s*${amountPattern('hi')}`,
|
|
78
|
+
'g'
|
|
79
|
+
);
|
|
80
|
+
|
|
81
|
+
const HOUR_RE = /\b(?:per|an|each)\s+hour\b|\/\s*(?:hr|hour)\b|\bhourly\b/i;
|
|
82
|
+
const MONTH_RE = /\b(?:per|a|each)\s+month\b|\/\s*(?:mo|month)\b|\bmonthly\b/i;
|
|
83
|
+
const YEAR_RE = /\b(?:per|a|each)\s+(?:year|annum)\b|\/\s*(?:yr|year)\b|\b(?:annual(?:ly|ized)?|yearly)\b/i;
|
|
84
|
+
|
|
39
85
|
/**
|
|
40
|
-
*
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
*
|
|
44
|
-
*
|
|
86
|
+
* Extract a salary range from decoded job text.
|
|
87
|
+
*
|
|
88
|
+
* Accepts hyphen, en dash, em dash or "to" between the two amounts, an
|
|
89
|
+
* optional ISO currency code before, between or after them, `$` / EUR /
|
|
90
|
+
* GBP symbols, decimal K shorthand ($211.4K), and thousands grouped with
|
|
91
|
+
* either a comma or a dot (60,000 and 60.000 are both sixty thousand).
|
|
92
|
+
* A code wins over a symbol, so "$120,000 - $150,000 CAD" is CAD. Ranges
|
|
93
|
+
* with no currency marker at all (years, headcounts) are ignored.
|
|
94
|
+
*
|
|
95
|
+
* @returns {{min:number,max:number,currency:string,period:('year'|'month'|'hour'|null),source:'text'}|null}
|
|
45
96
|
*/
|
|
46
|
-
function extractSalaryFromText(text) {
|
|
97
|
+
export function extractSalaryFromText(text) {
|
|
47
98
|
if (!text) return null;
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
99
|
+
for (const m of text.matchAll(SALARY_RANGE)) {
|
|
100
|
+
const g = m.groups;
|
|
101
|
+
const end = m.index + m[0].length;
|
|
102
|
+
const after = text.slice(end, end + 40);
|
|
103
|
+
// "$20 - $30 million" is a revenue figure, not pay.
|
|
104
|
+
if (/^\s*(?:million|billion|m|bn?)\b/i.test(after)) continue;
|
|
105
|
+
|
|
106
|
+
const code = g.loCodeBefore || g.loCodeAfter || g.hiCodeBefore || g.hiCodeAfter;
|
|
107
|
+
const sym = g.loSym || g.hiSym;
|
|
108
|
+
if (!code && !sym) continue;
|
|
109
|
+
|
|
110
|
+
let min = parseAmount(g.loNum);
|
|
111
|
+
let max = parseAmount(g.hiNum);
|
|
112
|
+
if (g.loK || g.hiK) {
|
|
113
|
+
// "$150-200K" carries the K once for both sides.
|
|
114
|
+
if (min < 1000) min = Math.round(min * 1000);
|
|
115
|
+
if (max < 1000) max = Math.round(max * 1000);
|
|
116
|
+
}
|
|
117
|
+
if (!(min > 0) || !(max > 0)) continue;
|
|
118
|
+
|
|
119
|
+
const before = text.slice(Math.max(0, m.index - 40), m.index);
|
|
60
120
|
return {
|
|
61
|
-
min
|
|
62
|
-
max
|
|
63
|
-
currency:
|
|
121
|
+
min,
|
|
122
|
+
max,
|
|
123
|
+
currency: (code || SYMBOL_CURRENCY[sym]).toUpperCase(),
|
|
124
|
+
period: detectPeriod(before, after, min),
|
|
125
|
+
source: 'text',
|
|
64
126
|
};
|
|
65
127
|
}
|
|
66
128
|
return null;
|
|
67
129
|
}
|
|
68
130
|
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
if (
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
131
|
+
// Dots grouping thousands mean a comma is the decimal mark, and vice versa.
|
|
132
|
+
function parseAmount(s) {
|
|
133
|
+
if (/^\d{1,3}(?:\.\d{3})+/.test(s)) return Number(s.replace(/\./g, '').replace(',', '.'));
|
|
134
|
+
return Number(s.replace(/,(?=\d{3})/g, '').replace(',', '.'));
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
function detectPeriod(before, after, min) {
|
|
138
|
+
const explicit = periodWord(after) || periodWord(before);
|
|
139
|
+
if (explicit) return explicit;
|
|
140
|
+
return min >= 10000 ? 'year' : null;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
function periodWord(text) {
|
|
144
|
+
if (HOUR_RE.test(text)) return 'hour';
|
|
145
|
+
if (MONTH_RE.test(text)) return 'month';
|
|
146
|
+
if (YEAR_RE.test(text)) return 'year';
|
|
147
|
+
return null;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
|
|
151
|
+
|
|
152
|
+
/**
|
|
153
|
+
* The platform's own value wins. Without one, a keyword in the location
|
|
154
|
+
* string is the next best signal. Without either the type is 'unknown':
|
|
155
|
+
* a city name alone does not say the role is onsite, and a guessed
|
|
156
|
+
* 'onsite' reads as a fact to whoever consumes it.
|
|
157
|
+
*
|
|
158
|
+
* @returns {{type:('remote'|'hybrid'|'onsite'|'unknown'), source:('ats'|'text'|null)}}
|
|
159
|
+
*/
|
|
160
|
+
function resolveWorkplace(native, location) {
|
|
161
|
+
if (WORKPLACE_TYPES.has(native)) return { type: native, source: 'ats' };
|
|
162
|
+
const guessed = workplaceFromText(location);
|
|
163
|
+
if (guessed) return { type: guessed, source: 'text' };
|
|
164
|
+
return { type: 'unknown', source: null };
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
function workplaceFromText(location) {
|
|
168
|
+
const lower = (location || '').toLowerCase();
|
|
169
|
+
if (/remote/.test(lower)) return 'remote';
|
|
170
|
+
if (/hybrid/.test(lower)) return 'hybrid';
|
|
171
|
+
if (/on-?site/.test(lower)) return 'onsite';
|
|
172
|
+
return null;
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
function uniqueLocations(primary, extra) {
|
|
176
|
+
const out = [];
|
|
177
|
+
const seen = new Set();
|
|
178
|
+
for (const loc of [primary, ...(Array.isArray(extra) ? extra : [])]) {
|
|
179
|
+
const s = typeof loc === 'string' ? loc.trim() : '';
|
|
180
|
+
if (!s || seen.has(s.toLowerCase())) continue;
|
|
181
|
+
seen.add(s.toLowerCase());
|
|
182
|
+
out.push(s);
|
|
183
|
+
}
|
|
184
|
+
return out;
|
|
75
185
|
}
|
|
76
186
|
|
|
77
187
|
/**
|
|
78
188
|
* Strip HTML tags and convert to clean text.
|
|
189
|
+
*
|
|
190
|
+
* Block closers become line breaks and list items become bullets before
|
|
191
|
+
* the remaining tags are removed. Entities are decoded LAST, so a literal
|
|
192
|
+
* `<` in the source text never turns into a tag that gets stripped.
|
|
79
193
|
*/
|
|
80
194
|
export function stripHtml(html) {
|
|
81
195
|
if (!html) return '';
|
|
82
|
-
|
|
196
|
+
const text = html
|
|
83
197
|
.replace(/<br\s*\/?>/gi, '\n')
|
|
84
|
-
.replace(/<\/p
|
|
85
|
-
.replace(/<\/li
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
.replace(/<
|
|
89
|
-
.replace(/<[^>]
|
|
90
|
-
.replace(
|
|
91
|
-
|
|
92
|
-
.replace(
|
|
93
|
-
.replace(/ /g, ' ')
|
|
94
|
-
.replace(/&#\d+;/g, '')
|
|
198
|
+
.replace(/<\/(?:p|h[1-6])\s*>/gi, '\n\n')
|
|
199
|
+
.replace(/<\/(?:li|div|td|tr|ul|ol|table|section)\s*>/gi, '\n')
|
|
200
|
+
// Word-pasted markup (Lever lists, issue #64) opens a <p> inside each
|
|
201
|
+
// <li>; swallowing it keeps the item text on the bullet's line.
|
|
202
|
+
.replace(/<li\b[^>]*>\s*(?:<p\b[^>]*>\s*)?/gi, '- ')
|
|
203
|
+
.replace(/<h[1-6]\b[^>]*>/gi, '## ')
|
|
204
|
+
.replace(/<[^>]+>/g, '');
|
|
205
|
+
return decodeEntities(text)
|
|
206
|
+
.replace(/\u00a0/g, ' ')
|
|
95
207
|
.replace(/\n{3,}/g, '\n\n')
|
|
96
208
|
.trim();
|
|
97
209
|
}
|
|
210
|
+
|
|
211
|
+
const NAMED_ENTITIES = {
|
|
212
|
+
lt: '<', gt: '>', quot: '"', apos: "'", nbsp: '\u00a0',
|
|
213
|
+
mdash: '—', ndash: '–', hellip: '…',
|
|
214
|
+
lsquo: '‘', rsquo: '’', ldquo: '“', rdquo: '”',
|
|
215
|
+
sbquo: '‚', bdquo: '„', laquo: '«', raquo: '»',
|
|
216
|
+
bull: '•', middot: '·', copy: '©', reg: '®', trade: '™',
|
|
217
|
+
deg: '°', times: '×', euro: '€', pound: '£', yen: '¥', cent: '¢',
|
|
218
|
+
agrave: 'à', aacute: 'á', acirc: 'â', atilde: 'ã', auml: 'ä', aring: 'å', aelig: 'æ',
|
|
219
|
+
ccedil: 'ç', egrave: 'è', eacute: 'é', ecirc: 'ê', euml: 'ë',
|
|
220
|
+
igrave: 'ì', iacute: 'í', icirc: 'î', iuml: 'ï', ntilde: 'ñ',
|
|
221
|
+
ograve: 'ò', oacute: 'ó', ocirc: 'ô', otilde: 'õ', ouml: 'ö', oslash: 'ø',
|
|
222
|
+
ugrave: 'ù', uacute: 'ú', ucirc: 'û', uuml: 'ü', yacute: 'ý', yuml: 'ÿ', szlig: 'ß',
|
|
223
|
+
Agrave: 'À', Aacute: 'Á', Acirc: 'Â', Atilde: 'Ã', Auml: 'Ä', Aring: 'Å', AElig: 'Æ',
|
|
224
|
+
Ccedil: 'Ç', Egrave: 'È', Eacute: 'É', Ecirc: 'Ê', Euml: 'Ë',
|
|
225
|
+
Igrave: 'Ì', Iacute: 'Í', Icirc: 'Î', Iuml: 'Ï', Ntilde: 'Ñ',
|
|
226
|
+
Ograve: 'Ò', Oacute: 'Ó', Ocirc: 'Ô', Otilde: 'Õ', Ouml: 'Ö', Oslash: 'Ø',
|
|
227
|
+
Ugrave: 'Ù', Uacute: 'Ú', Ucirc: 'Û', Uuml: 'Ü', Yacute: 'Ý',
|
|
228
|
+
};
|
|
229
|
+
|
|
230
|
+
/**
|
|
231
|
+
* Decode one layer of entity encoding (and unwrap CDATA) to real text.
|
|
232
|
+
*
|
|
233
|
+
* Decimal and hex references go through String.fromCodePoint, named
|
|
234
|
+
* references through the table above. `&` is intentionally resolved
|
|
235
|
+
* LAST so double-encoded sequences (`&mdash;`, `&amp;`) collapse
|
|
236
|
+
* by exactly one layer per call. Used for the outer escaping Greenhouse
|
|
237
|
+
* and the Teamtailor RSS feed apply, and as stripHtml's final step.
|
|
238
|
+
*/
|
|
239
|
+
export function decodeEntities(s) {
|
|
240
|
+
if (!s) return '';
|
|
241
|
+
return s
|
|
242
|
+
.replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1')
|
|
243
|
+
.replace(/&#(\d+);/g, (m, dec) => codePointToString(parseInt(dec, 10), m))
|
|
244
|
+
.replace(/&#[xX]([0-9a-fA-F]+);/g, (m, hex) => codePointToString(parseInt(hex, 16), m))
|
|
245
|
+
.replace(/&([A-Za-z][A-Za-z0-9]*);/g, (m, name) =>
|
|
246
|
+
(name !== 'amp' && Object.hasOwn(NAMED_ENTITIES, name)) ? NAMED_ENTITIES[name] : m)
|
|
247
|
+
.replace(/&/g, '&');
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
function codePointToString(cp, fallback) {
|
|
251
|
+
if (!cp || cp > 0x10ffff || (cp >= 0xd800 && cp <= 0xdfff)) return fallback;
|
|
252
|
+
return String.fromCodePoint(cp);
|
|
253
|
+
}
|