jd-intel 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -0
- package/package.json +1 -1
- package/src/adapters/ashby.js +20 -72
- package/src/adapters/greenhouse.js +19 -9
- package/src/adapters/lever.js +16 -8
- package/src/adapters/recruitee.js +20 -9
- package/src/adapters/smartrecruiters.js +113 -36
- package/src/adapters/teamtailor.js +37 -15
- package/src/adapters/workday.js +60 -19
- package/src/boards.js +80 -0
- package/src/cli.js +12 -8
- package/src/errors.js +14 -0
- package/src/filters.js +62 -23
- package/src/http.js +184 -0
- package/src/index.js +168 -57
- package/src/registry.js +79 -25
package/README.md
CHANGED
|
@@ -144,8 +144,22 @@ const { jobs, total_matched } = await fetchJobsDetailed({
|
|
|
144
144
|
});
|
|
145
145
|
```
|
|
146
146
|
|
|
147
|
+
The same result says how the company was resolved and which boards answered:
|
|
148
|
+
|
|
149
|
+
- `match`: `registry` (a known company, one adapter call), `probe` (not in the registry, every adapter asked) or `workday_override` (an explicit Workday config).
|
|
150
|
+
- `company`: `{ key, name }` from the registry row on a registry match, null otherwise.
|
|
151
|
+
- `boards`: one entry per board that answered, `{ ats, slug, name, site, board_url, org_name, org_url, jobs_found, matched, selected, scan }`. `jobs_found` counts the rows the board listed before any filter. Workday and SmartRecruiters filter their list before fetching details, so for them it is the list count, not the rows that came back (a lower bound when the list scan caps). `matched` is the rows left after filters and before `offset` and `limit`, so a board whose rows were all cut from the page still shows up. `board_url` is built from the slug. `org_name` and `org_url` are what the board states about itself, the organization name in the ATS response and the careers or company host its links point at, and they are null where the platform exposes nothing (Lever and Ashby expose neither); nothing is filled in from the slug or the registry name. A `probe` match is a slug match, not a confirmed identity: compare `org_name` and `org_url` with the company you meant, and open `board_url` when they are null, before treating the jobs as that company's.
|
|
152
|
+
- `failed`: adapters that threw during discovery, `{ ats, slug, name, code, message }`, with `code` either `rate_limited` or `ats_unreachable`. When no board answered and a check failed, `fetchJobs` throws that error instead of returning `[]`, so an outage never reads as "not found".
|
|
153
|
+
- `total_before_filters`: the sum of `jobs_found`. Zero matches with a positive `total_before_filters` means the company is hiring and nothing passed the filters, on every ATS.
|
|
154
|
+
|
|
155
|
+
`detectAtsDetailed(company)` returns `{ boards, failed }`: registry rows first (`source: 'registry'`, never probed), then every live probe that answered (`source: 'probe'`), in platform order; `failed` lists the probes that could not be checked, with the same codes. `detectAts` keeps returning `[{ ats, slug }]`.
|
|
156
|
+
|
|
157
|
+
A call the library cannot make throws `ArgumentError` (`code: 'invalid_args'`): no company, an unknown `ats`, or a `titleFilter` or `filter` that does not compile as a regex. It is thrown before any request goes out.
|
|
158
|
+
|
|
147
159
|
CLI usage: `npx jd-intel fetch <company-slug> --title-filter "engineer" --posted-within-days 14`. Full filter reference [below](#filters-quick-reference).
|
|
148
160
|
|
|
161
|
+
Each ATS request gives the server 10 seconds to start responding. A 429, a 5xx or a network error is retried up to three attempts with backoff (1s, 2s), honoring `Retry-After` when the ATS sends one, and at most 4 requests run at a time per host. A failure that outlasts the retries throws an `AtsError` whose `code` is `rate_limited` or `ats_unreachable`.
|
|
162
|
+
|
|
149
163
|
Node.js 18+. No API keys. No configuration.
|
|
150
164
|
|
|
151
165
|
### Manual install (fallback)
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "jd-intel",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.10.0",
|
|
4
4
|
"description": "Fetch and normalize job descriptions across seven major ATS (Greenhouse, Lever, Ashby, Workday, and more), for your AI assistant. No copy-paste.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "src/index.js",
|
package/src/adapters/ashby.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { normalize, extractSalaryFromText } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { atsFetch, probeResult } from '../http.js';
|
|
3
4
|
|
|
4
|
-
const API_URL = 'https://jobs.ashbyhq.com/api/non-user-graphql';
|
|
5
5
|
const BOARD_URL = 'https://api.ashbyhq.com/posting-api/job-board';
|
|
6
6
|
|
|
7
7
|
/**
|
|
@@ -9,23 +9,18 @@ const BOARD_URL = 'https://api.ashbyhq.com/posting-api/job-board';
|
|
|
9
9
|
* Public API, no auth required.
|
|
10
10
|
* Docs: https://developers.ashbyhq.com/docs/public-job-posting-api
|
|
11
11
|
*
|
|
12
|
+
* REST only. The GraphQL fallback this adapter once carried never named a
|
|
13
|
+
* board, so it never returned a job, and it turned every REST 429 or 5xx
|
|
14
|
+
* into a silent empty result (issue #55).
|
|
15
|
+
*
|
|
12
16
|
* @param {string} slug - Company slug (e.g., 'notion', 'linear')
|
|
17
|
+
* @param {object} [ctx] - { report }; report is called once with
|
|
18
|
+
* { ats, org_name, org_url } when given
|
|
13
19
|
* @returns {Promise<Array>} Normalized job objects
|
|
14
20
|
*/
|
|
15
|
-
export async function fetchAshby(slug) {
|
|
16
|
-
// Try the REST API first (simpler, includes compensation)
|
|
17
|
-
try {
|
|
18
|
-
const restJobs = await fetchAshbyRest(slug);
|
|
19
|
-
if (restJobs.length > 0) return restJobs;
|
|
20
|
-
} catch { /* fall through to GraphQL */ }
|
|
21
|
-
|
|
22
|
-
// Fallback: GraphQL API
|
|
23
|
-
return fetchAshbyGraphQL(slug);
|
|
24
|
-
}
|
|
25
|
-
|
|
26
|
-
async function fetchAshbyRest(slug) {
|
|
21
|
+
export async function fetchAshby(slug, ctx = {}) {
|
|
27
22
|
const url = `${BOARD_URL}/${slug}?includeCompensation=true`;
|
|
28
|
-
const resp = await
|
|
23
|
+
const resp = await atsFetch(url);
|
|
29
24
|
|
|
30
25
|
if (!resp.ok) {
|
|
31
26
|
if (resp.status === 404) return [];
|
|
@@ -35,6 +30,12 @@ async function fetchAshbyRest(slug) {
|
|
|
35
30
|
const data = await resp.json();
|
|
36
31
|
const jobs = data.jobs || [];
|
|
37
32
|
|
|
33
|
+
// The REST response is { jobs, apiVersion }: no organization name, and
|
|
34
|
+
// every link is on jobs.ashbyhq.com. Both null (issue #58).
|
|
35
|
+
if (typeof ctx.report === 'function') {
|
|
36
|
+
ctx.report({ ats: 'ashby', org_name: null, org_url: null });
|
|
37
|
+
}
|
|
38
|
+
|
|
38
39
|
return jobs.map(job => {
|
|
39
40
|
const comp = job.compensation || {};
|
|
40
41
|
|
|
@@ -68,58 +69,6 @@ async function fetchAshbyRest(slug) {
|
|
|
68
69
|
});
|
|
69
70
|
}
|
|
70
71
|
|
|
71
|
-
async function fetchAshbyGraphQL(slug) {
|
|
72
|
-
const query = `{
|
|
73
|
-
jobBoard {
|
|
74
|
-
title
|
|
75
|
-
jobPostings {
|
|
76
|
-
id
|
|
77
|
-
title
|
|
78
|
-
locationName
|
|
79
|
-
employmentType
|
|
80
|
-
descriptionHtml
|
|
81
|
-
publishedDate
|
|
82
|
-
compensationTierSummary
|
|
83
|
-
}
|
|
84
|
-
}
|
|
85
|
-
}`;
|
|
86
|
-
|
|
87
|
-
const resp = await fetch(API_URL, {
|
|
88
|
-
method: 'POST',
|
|
89
|
-
headers: { 'Content-Type': 'application/json' },
|
|
90
|
-
body: JSON.stringify({
|
|
91
|
-
operationName: 'ApiJobBoardWithTeams',
|
|
92
|
-
variables: { organizationHostedJobsPageName: slug },
|
|
93
|
-
query,
|
|
94
|
-
}),
|
|
95
|
-
});
|
|
96
|
-
|
|
97
|
-
if (!resp.ok) return [];
|
|
98
|
-
|
|
99
|
-
const data = await resp.json();
|
|
100
|
-
const board = data.data?.jobBoard;
|
|
101
|
-
if (!board) return [];
|
|
102
|
-
|
|
103
|
-
const postings = board.jobPostings || [];
|
|
104
|
-
|
|
105
|
-
return postings.map(job => normalize({
|
|
106
|
-
companySlug: slug,
|
|
107
|
-
company: board.title || slug,
|
|
108
|
-
title: job.title || '',
|
|
109
|
-
department: '',
|
|
110
|
-
location: job.locationName || '',
|
|
111
|
-
description: job.descriptionHtml || '',
|
|
112
|
-
url: `https://jobs.ashbyhq.com/${slug}/${job.id}`,
|
|
113
|
-
postedAt: job.publishedDate || null,
|
|
114
|
-
salary: null,
|
|
115
|
-
metadata: {
|
|
116
|
-
ashbyId: job.id,
|
|
117
|
-
employmentType: job.employmentType || '',
|
|
118
|
-
compensationSummary: job.compensationTierSummary || '',
|
|
119
|
-
},
|
|
120
|
-
}, 'ashby'));
|
|
121
|
-
}
|
|
122
|
-
|
|
123
72
|
const WORKPLACE_TYPES = { remote: 'remote', hybrid: 'hybrid', onsite: 'onsite' };
|
|
124
73
|
|
|
125
74
|
/**
|
|
@@ -163,11 +112,10 @@ function parseAshbyCompensation(comp) {
|
|
|
163
112
|
return parsed ? { ...parsed, source: 'ats' } : null;
|
|
164
113
|
}
|
|
165
114
|
|
|
115
|
+
/**
|
|
116
|
+
* Check if a company has an Ashby board. See probeResult for the outcomes.
|
|
117
|
+
*/
|
|
166
118
|
export async function hasAshby(slug) {
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
return resp.ok;
|
|
170
|
-
} catch {
|
|
171
|
-
return false;
|
|
172
|
-
}
|
|
119
|
+
const resp = await atsFetch(`${BOARD_URL}/${slug}`, { method: 'HEAD' });
|
|
120
|
+
return probeResult(resp, `Ashby probe for ${slug}`);
|
|
173
121
|
}
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { normalize, decodeEntities } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { atsFetch, probeResult } from '../http.js';
|
|
3
4
|
|
|
4
5
|
const BASE_URL = 'https://boards-api.greenhouse.io/v1/boards';
|
|
5
6
|
|
|
@@ -9,11 +10,13 @@ const BASE_URL = 'https://boards-api.greenhouse.io/v1/boards';
|
|
|
9
10
|
* Docs: https://developers.greenhouse.io/job-board.html
|
|
10
11
|
*
|
|
11
12
|
* @param {string} slug - Company slug (e.g., 'stripe', 'notion')
|
|
13
|
+
* @param {object} [ctx] - { report }; report is called once with
|
|
14
|
+
* { ats, org_name, org_url } when given
|
|
12
15
|
* @returns {Promise<Array>} Normalized job objects
|
|
13
16
|
*/
|
|
14
|
-
export async function fetchGreenhouse(slug) {
|
|
17
|
+
export async function fetchGreenhouse(slug, ctx = {}) {
|
|
15
18
|
const url = `${BASE_URL}/${slug}/jobs?content=true`;
|
|
16
|
-
const resp = await
|
|
19
|
+
const resp = await atsFetch(url);
|
|
17
20
|
|
|
18
21
|
if (!resp.ok) {
|
|
19
22
|
if (resp.status === 404) return []; // Company not found or no jobs
|
|
@@ -23,6 +26,17 @@ export async function fetchGreenhouse(slug) {
|
|
|
23
26
|
const data = await resp.json();
|
|
24
27
|
const jobs = data.jobs || [];
|
|
25
28
|
|
|
29
|
+
// The list response has no top-level name, but each row carries the
|
|
30
|
+
// board's company_name. Its only links are job-boards.greenhouse.io, so
|
|
31
|
+
// there is no company host to report (issue #58).
|
|
32
|
+
if (typeof ctx.report === 'function') {
|
|
33
|
+
ctx.report({
|
|
34
|
+
ats: 'greenhouse',
|
|
35
|
+
org_name: jobs.find(j => j.company_name)?.company_name || null,
|
|
36
|
+
org_url: null,
|
|
37
|
+
});
|
|
38
|
+
}
|
|
39
|
+
|
|
26
40
|
return jobs.map(job => normalize({
|
|
27
41
|
companySlug: slug,
|
|
28
42
|
company: data.name || slug,
|
|
@@ -66,13 +80,9 @@ function parseGreenhouseWorkplace(metadata) {
|
|
|
66
80
|
}
|
|
67
81
|
|
|
68
82
|
/**
|
|
69
|
-
* Check if a company has a Greenhouse board.
|
|
83
|
+
* Check if a company has a Greenhouse board. See probeResult for the outcomes.
|
|
70
84
|
*/
|
|
71
85
|
export async function hasGreenhouse(slug) {
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
return resp.ok;
|
|
75
|
-
} catch {
|
|
76
|
-
return false;
|
|
77
|
-
}
|
|
86
|
+
const resp = await atsFetch(`${BASE_URL}/${slug}`, { method: 'HEAD' });
|
|
87
|
+
return probeResult(resp, `Greenhouse probe for ${slug}`);
|
|
78
88
|
}
|
package/src/adapters/lever.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { normalize, extractSalaryFromText } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { atsFetch, probeResult } from '../http.js';
|
|
3
4
|
|
|
4
5
|
const BASE_URL = 'https://api.lever.co/v0/postings';
|
|
5
6
|
|
|
@@ -13,11 +14,13 @@ const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
|
|
|
13
14
|
* Docs: https://github.com/lever/postings-api
|
|
14
15
|
*
|
|
15
16
|
* @param {string} slug - Company slug (e.g., 'stripe', 'figma')
|
|
17
|
+
* @param {object} [ctx] - { report }; report is called once with
|
|
18
|
+
* { ats, org_name, org_url } when given
|
|
16
19
|
* @returns {Promise<Array>} Normalized job objects
|
|
17
20
|
*/
|
|
18
|
-
export async function fetchLever(slug) {
|
|
21
|
+
export async function fetchLever(slug, ctx = {}) {
|
|
19
22
|
const url = `${BASE_URL}/${slug}?mode=json`;
|
|
20
|
-
const resp = await
|
|
23
|
+
const resp = await atsFetch(url);
|
|
21
24
|
|
|
22
25
|
if (!resp.ok) {
|
|
23
26
|
if (resp.status === 404) return [];
|
|
@@ -27,6 +30,12 @@ export async function fetchLever(slug) {
|
|
|
27
30
|
const jobs = await resp.json();
|
|
28
31
|
if (!Array.isArray(jobs)) return [];
|
|
29
32
|
|
|
33
|
+
// The postings response is a bare array of jobs: no organization name
|
|
34
|
+
// anywhere, and every link is on jobs.lever.co. Both null (issue #58).
|
|
35
|
+
if (typeof ctx.report === 'function') {
|
|
36
|
+
ctx.report({ ats: 'lever', org_name: null, org_url: null });
|
|
37
|
+
}
|
|
38
|
+
|
|
30
39
|
return jobs.map(job => normalize({
|
|
31
40
|
companySlug: slug,
|
|
32
41
|
// Lever's API doesn't return the company name at the board or job level,
|
|
@@ -102,11 +111,10 @@ function titleCaseSlug(slug) {
|
|
|
102
111
|
return slug.charAt(0).toUpperCase() + slug.slice(1);
|
|
103
112
|
}
|
|
104
113
|
|
|
114
|
+
/**
|
|
115
|
+
* Check if a company has a Lever board. See probeResult for the outcomes.
|
|
116
|
+
*/
|
|
105
117
|
export async function hasLever(slug) {
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
return resp.ok;
|
|
109
|
-
} catch {
|
|
110
|
-
return false;
|
|
111
|
-
}
|
|
118
|
+
const resp = await atsFetch(`${BASE_URL}/${slug}?mode=json`, { method: 'HEAD' });
|
|
119
|
+
return probeResult(resp, `Lever probe for ${slug}`);
|
|
112
120
|
}
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import { normalize } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { atsFetch, probeResult } from '../http.js';
|
|
4
|
+
import { orgHost } from '../boards.js';
|
|
3
5
|
|
|
4
6
|
/**
|
|
5
7
|
* Fetch jobs from a Recruitee career site.
|
|
@@ -18,11 +20,13 @@ import { atsErrorFromStatus } from '../errors.js';
|
|
|
18
20
|
* both are joined before normalize() strips them (issue #65).
|
|
19
21
|
*
|
|
20
22
|
* @param {string} slug - Recruitee company subdomain (e.g., 'vandebron')
|
|
23
|
+
* @param {object} [ctx] - { report }; report is called once with
|
|
24
|
+
* { ats, org_name, org_url } when given
|
|
21
25
|
* @returns {Promise<Array>} Normalized job objects
|
|
22
26
|
*/
|
|
23
|
-
export async function fetchRecruitee(slug) {
|
|
27
|
+
export async function fetchRecruitee(slug, ctx = {}) {
|
|
24
28
|
const url = `https://${slug}.recruitee.com/api/offers/`;
|
|
25
|
-
const resp = await
|
|
29
|
+
const resp = await atsFetch(url);
|
|
26
30
|
|
|
27
31
|
if (!resp.ok) {
|
|
28
32
|
if (resp.status === 404) return []; // No Recruitee site for this slug
|
|
@@ -32,6 +36,17 @@ export async function fetchRecruitee(slug) {
|
|
|
32
36
|
const data = await resp.json();
|
|
33
37
|
const offers = data.offers || [];
|
|
34
38
|
|
|
39
|
+
// The response is { offers } only, so the identity lives on the rows:
|
|
40
|
+
// company_name, and careers_url, which sits on the company's own careers
|
|
41
|
+
// domain when the site has one and on {slug}.recruitee.com otherwise.
|
|
42
|
+
if (typeof ctx.report === 'function') {
|
|
43
|
+
ctx.report({
|
|
44
|
+
ats: 'recruitee',
|
|
45
|
+
org_name: offers.find(o => o.company_name)?.company_name || null,
|
|
46
|
+
org_url: orgHost(offers.find(o => o.careers_url)?.careers_url),
|
|
47
|
+
});
|
|
48
|
+
}
|
|
49
|
+
|
|
35
50
|
return offers.map(offer => {
|
|
36
51
|
const place = [offer.city, offer.country].filter(Boolean).join(', ');
|
|
37
52
|
let location = place;
|
|
@@ -111,13 +126,9 @@ function toAmount(value) {
|
|
|
111
126
|
}
|
|
112
127
|
|
|
113
128
|
/**
|
|
114
|
-
* Check if a company has a Recruitee career site.
|
|
129
|
+
* Check if a company has a Recruitee career site. See probeResult for the outcomes.
|
|
115
130
|
*/
|
|
116
131
|
export async function hasRecruitee(slug) {
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
return resp.ok;
|
|
120
|
-
} catch {
|
|
121
|
-
return false;
|
|
122
|
-
}
|
|
132
|
+
const resp = await atsFetch(`https://${slug}.recruitee.com/api/offers/`);
|
|
133
|
+
return probeResult(resp, `Recruitee probe for ${slug}`);
|
|
123
134
|
}
|
|
@@ -1,11 +1,14 @@
|
|
|
1
1
|
import { normalize } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { atsFetch, probeResult } from '../http.js';
|
|
4
|
+
import { makeLocationMatcher } from '../filters.js';
|
|
3
5
|
|
|
4
6
|
const BASE_URL = 'https://api.smartrecruiters.com/v1/companies';
|
|
5
7
|
const PAGE_SIZE = 100;
|
|
8
|
+
const MAX_DETAIL_FETCHES = 100;
|
|
6
9
|
|
|
7
10
|
/**
|
|
8
|
-
* Fetch
|
|
11
|
+
* Fetch postings from a SmartRecruiters company.
|
|
9
12
|
* Public API, no auth required.
|
|
10
13
|
* Docs: https://developers.smartrecruiters.com/reference/postingsget-1
|
|
11
14
|
*
|
|
@@ -14,21 +17,29 @@ const PAGE_SIZE = 100;
|
|
|
14
17
|
* and the structured `compensation` block with it.
|
|
15
18
|
* - jd-intel's contract is "full JD text", so we must fetch each
|
|
16
19
|
* posting's DETAIL endpoint to get jobAd.sections.
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
+
*
|
|
21
|
+
* The list does carry name, location and releasedDate, so the same
|
|
22
|
+
* pre-filter and detail budget Workday applies run here: list-evaluable
|
|
23
|
+
* filters narrow the candidates, then at most MAX_DETAIL_FETCHES of them
|
|
24
|
+
* are hydrated (see the budget note below). Without a filterContext the
|
|
25
|
+
* cap still holds, so a direct call on a 400-posting tenant reads 100.
|
|
20
26
|
*
|
|
21
27
|
* @param {string} slug - SmartRecruiters company identifier (e.g., 'Visa')
|
|
28
|
+
* @param {object} [ctx] - { filterContext, report }; report is called once
|
|
29
|
+
* with { ats, listed, prefiltered, hydrated, capped, org_name, org_url }
|
|
30
|
+
* when given
|
|
22
31
|
* @returns {Promise<Array>} Normalized job objects
|
|
23
32
|
*/
|
|
24
|
-
export async function fetchSmartrecruiters(slug) {
|
|
33
|
+
export async function fetchSmartrecruiters(slug, ctx = {}) {
|
|
34
|
+
const fc = ctx.filterContext || {};
|
|
35
|
+
|
|
25
36
|
// 1. Page through the postings list.
|
|
26
37
|
const postings = [];
|
|
27
38
|
let offset = 0;
|
|
28
39
|
|
|
29
40
|
while (true) {
|
|
30
41
|
const listUrl = `${BASE_URL}/${slug}/postings?limit=${PAGE_SIZE}&offset=${offset}`;
|
|
31
|
-
const resp = await
|
|
42
|
+
const resp = await atsFetch(listUrl);
|
|
32
43
|
|
|
33
44
|
if (!resp.ok) {
|
|
34
45
|
if (resp.status === 404) return []; // Company not found
|
|
@@ -43,14 +54,80 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
43
54
|
if (content.length === 0 || offset >= (data.totalFound || 0)) break;
|
|
44
55
|
}
|
|
45
56
|
|
|
46
|
-
// 2.
|
|
47
|
-
|
|
57
|
+
// 2. Filter-aware candidate selection BEFORE the N+1 detail cost.
|
|
58
|
+
// The list row carries name, location and releasedDate, and the
|
|
59
|
+
// library re-applies every filter after this returns, so a keep here
|
|
60
|
+
// is never final. The detail adds no location (unlike Workday's
|
|
61
|
+
// additionalLocations), so a row with none follows the library's
|
|
62
|
+
// rule now: out under includes, kept under excludes.
|
|
63
|
+
let candidates = postings;
|
|
64
|
+
|
|
65
|
+
if (fc.titleFilter) {
|
|
66
|
+
const re = new RegExp(fc.titleFilter, 'i');
|
|
67
|
+
candidates = candidates.filter(p => re.test(p.name || ''));
|
|
68
|
+
}
|
|
69
|
+
if (Array.isArray(fc.locationIncludes) && fc.locationIncludes.length > 0) {
|
|
70
|
+
const matchers = fc.locationIncludes.map(makeLocationMatcher);
|
|
71
|
+
candidates = candidates.filter(p => {
|
|
72
|
+
const loc = listLocation(p).location.toLowerCase();
|
|
73
|
+
return matchers.some(m => m(loc));
|
|
74
|
+
});
|
|
75
|
+
}
|
|
76
|
+
if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
|
|
77
|
+
const matchers = fc.locationExcludes.map(makeLocationMatcher);
|
|
78
|
+
candidates = candidates.filter(p => {
|
|
79
|
+
const loc = listLocation(p).location.toLowerCase();
|
|
80
|
+
return !loc || !matchers.some(m => m(loc));
|
|
81
|
+
});
|
|
82
|
+
}
|
|
83
|
+
if (typeof fc.postedWithinDays === 'number') {
|
|
84
|
+
// postedAt comes from releasedDate alone, so the library's rule can
|
|
85
|
+
// run here in full: a missing or unparseable date is out either way.
|
|
86
|
+
const cutoff = Date.now() - fc.postedWithinDays * 86400000;
|
|
87
|
+
candidates = candidates.filter(p => {
|
|
88
|
+
const released = new Date(p.releasedDate || '').getTime();
|
|
89
|
+
return Number.isFinite(released) && released >= cutoff;
|
|
90
|
+
});
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// 3. Bound the detail-fetch set, Workday's reasoning verbatim: a
|
|
94
|
+
// description `filter` is applied by the library AFTER this returns,
|
|
95
|
+
// so that case keeps the full backstop instead of truncating to
|
|
96
|
+
// `limit` (which could hydrate jobs that all fail the regex while
|
|
97
|
+
// better matches go unscanned). The library pages with `offset`
|
|
98
|
+
// after this returns, so the budget covers the page plus what
|
|
99
|
+
// precedes it. Candidates keep list order.
|
|
100
|
+
const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
|
|
101
|
+
const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
|
|
102
|
+
const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
|
|
103
|
+
const hydrate = candidates.slice(0, cap);
|
|
104
|
+
|
|
105
|
+
// Every list row carries company { identifier, name }. Neither the list
|
|
106
|
+
// nor the detail has a company website, and postingUrl is always on
|
|
107
|
+
// jobs.smartrecruiters.com, so org_url stays null (issue #58).
|
|
108
|
+
if (typeof ctx.report === 'function') {
|
|
109
|
+
ctx.report({
|
|
110
|
+
ats: 'smartrecruiters',
|
|
111
|
+
listed: postings.length,
|
|
112
|
+
prefiltered: candidates.length,
|
|
113
|
+
hydrated: hydrate.length,
|
|
114
|
+
capped: hydrate.length < candidates.length,
|
|
115
|
+
org_name: postings.find(p => p.company?.name)?.company.name || null,
|
|
116
|
+
org_url: null,
|
|
117
|
+
});
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
// 4. Fetch detail per candidate for the description. atsFetch's per-host
|
|
121
|
+
// queue keeps this fan-out to 4 requests at a time (a 412-posting
|
|
122
|
+
// tenant measured 53s unbounded), so the cap also holds the detail
|
|
123
|
+
// step to roughly 13s.
|
|
124
|
+
const jobs = await Promise.all(hydrate.map(async (p) => {
|
|
48
125
|
let sections = {};
|
|
49
126
|
let postingUrl = '';
|
|
50
127
|
let salary = null;
|
|
51
128
|
|
|
52
129
|
try {
|
|
53
|
-
const detailResp = await
|
|
130
|
+
const detailResp = await atsFetch(`${BASE_URL}/${slug}/postings/${p.id}`);
|
|
54
131
|
if (detailResp.ok) {
|
|
55
132
|
const detail = await detailResp.json();
|
|
56
133
|
sections = detail.jobAd?.sections || {};
|
|
@@ -58,7 +135,8 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
58
135
|
salary = parseCompensation(detail.compensation);
|
|
59
136
|
}
|
|
60
137
|
} catch {
|
|
61
|
-
// Detail fetch failed: fall back to list-only
|
|
138
|
+
// Detail fetch failed, retries included: fall back to list-only
|
|
139
|
+
// fields (no description). Reporting this is #85.
|
|
62
140
|
}
|
|
63
141
|
|
|
64
142
|
const description = [
|
|
@@ -67,18 +145,7 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
67
145
|
sections.additionalInformation?.text,
|
|
68
146
|
].filter(Boolean).join('\n\n');
|
|
69
147
|
|
|
70
|
-
const
|
|
71
|
-
const place = loc.fullLocation
|
|
72
|
-
|| [loc.city, loc.region, loc.country].filter(Boolean).join(', ');
|
|
73
|
-
let location = place;
|
|
74
|
-
let workplace = null;
|
|
75
|
-
if (loc.remote) {
|
|
76
|
-
location = `Remote - ${place}`.replace(/ - $/, ' ');
|
|
77
|
-
workplace = 'remote';
|
|
78
|
-
} else if (loc.hybrid) {
|
|
79
|
-
location = `Hybrid - ${place}`.replace(/ - $/, ' ');
|
|
80
|
-
workplace = 'hybrid';
|
|
81
|
-
}
|
|
148
|
+
const { location, workplace } = listLocation(p);
|
|
82
149
|
|
|
83
150
|
return normalize({
|
|
84
151
|
companySlug: slug,
|
|
@@ -104,6 +171,20 @@ export async function fetchSmartrecruiters(slug) {
|
|
|
104
171
|
return jobs;
|
|
105
172
|
}
|
|
106
173
|
|
|
174
|
+
/**
|
|
175
|
+
* The location string and workplace type a list row yields. Built once
|
|
176
|
+
* here so the pre-filter matches exactly what the normalized job carries,
|
|
177
|
+
* "Remote - " and "Hybrid - " prefixes included.
|
|
178
|
+
*/
|
|
179
|
+
function listLocation(p) {
|
|
180
|
+
const loc = p.location || {};
|
|
181
|
+
const place = loc.fullLocation
|
|
182
|
+
|| [loc.city, loc.region, loc.country].filter(Boolean).join(', ');
|
|
183
|
+
if (loc.remote) return { location: `Remote - ${place}`.replace(/ - $/, ' '), workplace: 'remote' };
|
|
184
|
+
if (loc.hybrid) return { location: `Hybrid - ${place}`.replace(/ - $/, ' '), workplace: 'hybrid' };
|
|
185
|
+
return { location: place, workplace: null };
|
|
186
|
+
}
|
|
187
|
+
|
|
107
188
|
const PERIODS = { YEARLY: 'year', MONTHLY: 'month', HOURLY: 'hour' };
|
|
108
189
|
|
|
109
190
|
/**
|
|
@@ -130,20 +211,16 @@ function parseCompensation(comp) {
|
|
|
130
211
|
}
|
|
131
212
|
|
|
132
213
|
/**
|
|
133
|
-
* Check if a company exists on SmartRecruiters.
|
|
134
|
-
* (HEAD isn't reliably supported on the postings endpoint, so
|
|
135
|
-
* minimal GET.)
|
|
214
|
+
* Check if a company exists on SmartRecruiters. See probeResult for the
|
|
215
|
+
* outcomes. (HEAD isn't reliably supported on the postings endpoint, so
|
|
216
|
+
* use a minimal GET.)
|
|
136
217
|
*/
|
|
137
218
|
export async function hasSmartrecruiters(slug) {
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
return (data.totalFound || 0) > 0 || (data.content || []).length > 0;
|
|
146
|
-
} catch {
|
|
147
|
-
return false;
|
|
148
|
-
}
|
|
219
|
+
const resp = await atsFetch(`${BASE_URL}/${slug}/postings?limit=1`);
|
|
220
|
+
if (!probeResult(resp, `SmartRecruiters probe for ${slug}`)) return false;
|
|
221
|
+
// SmartRecruiters returns 200 with an empty page (not 404) for unknown
|
|
222
|
+
// companies, so resp.ok alone false-positives on any slug. Confirm at
|
|
223
|
+
// least one real posting exists before claiming a match.
|
|
224
|
+
const data = await resp.json();
|
|
225
|
+
return (data.totalFound || 0) > 0 || (data.content || []).length > 0;
|
|
149
226
|
}
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import { normalize, decodeEntities } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { atsFetch } from '../http.js';
|
|
4
|
+
import { orgHost } from '../boards.js';
|
|
3
5
|
|
|
4
6
|
/**
|
|
5
7
|
* Fetch jobs from a TeamTailor career site via its public RSS feed.
|
|
@@ -23,26 +25,33 @@ import { atsErrorFromStatus } from '../errors.js';
|
|
|
23
25
|
* collapse by one layer per pass.
|
|
24
26
|
*
|
|
25
27
|
* @param {string} slug - TeamTailor career-site slug (e.g., 'tibber')
|
|
28
|
+
* @param {object} [ctx] - { report }; report is called once with
|
|
29
|
+
* { ats, org_name, org_url } when given
|
|
26
30
|
* @returns {Promise<Array>} Normalized job objects
|
|
27
31
|
*/
|
|
28
32
|
// Most sites are {slug}.teamtailor.com, but some sit on a regional
|
|
29
|
-
// segment, e.g. crunchbase.na.teamtailor.com. '' is the base host.
|
|
30
|
-
|
|
33
|
+
// segment, e.g. crunchbase.na.teamtailor.com. '' is the base host. There
|
|
34
|
+
// is no reachable eu segment: {slug}.eu.teamtailor.com fails TLS for every
|
|
35
|
+
// slug, known or not, because the wildcard certificate covers one label
|
|
36
|
+
// only (live check 2026-09-27). Probing it was a guaranteed failure that
|
|
37
|
+
// the has() contract would now report as an outage.
|
|
38
|
+
const TT_REGIONS = ['', 'na'];
|
|
31
39
|
|
|
32
40
|
// Feeds send `none`, `hybrid`, `fully` or `onsite`. `none` is no signal.
|
|
33
41
|
const REMOTE_STATUS = { hybrid: 'hybrid', fully: 'remote', onsite: 'onsite' };
|
|
34
42
|
|
|
35
43
|
/**
|
|
36
44
|
* Resolve which TeamTailor host actually serves this slug's feed.
|
|
37
|
-
* Returns the first 200 Response, throws on a non-404 error
|
|
38
|
-
* returns null if no
|
|
45
|
+
* Returns the first 200 Response, throws on a non-404 error (atsFetch
|
|
46
|
+
* throws the 429, 5xx and network cases itself), or returns null if no
|
|
47
|
+
* region has a feed.
|
|
39
48
|
*/
|
|
40
49
|
async function resolveFeed(slug, method = 'GET') {
|
|
41
50
|
for (const region of TT_REGIONS) {
|
|
42
51
|
const host = region
|
|
43
52
|
? `${slug}.${region}.teamtailor.com`
|
|
44
53
|
: `${slug}.teamtailor.com`;
|
|
45
|
-
const resp = await
|
|
54
|
+
const resp = await atsFetch(`https://${host}/jobs.rss`, {
|
|
46
55
|
method,
|
|
47
56
|
redirect: 'follow',
|
|
48
57
|
});
|
|
@@ -55,18 +64,33 @@ async function resolveFeed(slug, method = 'GET') {
|
|
|
55
64
|
return null;
|
|
56
65
|
}
|
|
57
66
|
|
|
58
|
-
export async function fetchTeamtailor(slug) {
|
|
67
|
+
export async function fetchTeamtailor(slug, ctx = {}) {
|
|
59
68
|
const resp = await resolveFeed(slug, 'GET');
|
|
60
69
|
if (!resp) return []; // No TeamTailor site in any known region
|
|
61
70
|
|
|
62
71
|
const xml = await resp.text();
|
|
63
72
|
|
|
64
|
-
const
|
|
65
|
-
|
|
66
|
-
).trim();
|
|
73
|
+
const channelTitle = (xml.match(/<channel>[\s\S]*?<title>([\s\S]*?)<\/title>/)?.[1] || '').trim();
|
|
74
|
+
const company = channelTitle || slug;
|
|
67
75
|
|
|
68
76
|
const items = [...xml.matchAll(/<item>([\s\S]*?)<\/item>/g)].map(m => m[1]);
|
|
69
77
|
|
|
78
|
+
// The channel title is the company as the site names itself. The channel
|
|
79
|
+
// <link> always sits on {slug}.teamtailor.com, but item links follow the
|
|
80
|
+
// site's custom domain when it has one (jobs.tibber.com on a feed served
|
|
81
|
+
// from tibber.teamtailor.com), so the first item's link is the host that
|
|
82
|
+
// can say something; the channel link is the fallback for an empty feed.
|
|
83
|
+
if (typeof ctx.report === 'function') {
|
|
84
|
+
const link = items[0]?.match(/<link>([\s\S]*?)<\/link>/)?.[1]
|
|
85
|
+
|| xml.match(/<channel>[\s\S]*?<link>([\s\S]*?)<\/link>/)?.[1]
|
|
86
|
+
|| '';
|
|
87
|
+
ctx.report({
|
|
88
|
+
ats: 'teamtailor',
|
|
89
|
+
org_name: decodeEntities(channelTitle) || null,
|
|
90
|
+
org_url: orgHost(link.trim()),
|
|
91
|
+
});
|
|
92
|
+
}
|
|
93
|
+
|
|
70
94
|
return items.map(item => {
|
|
71
95
|
const pick = (tag, src = item) => {
|
|
72
96
|
const m = src.match(new RegExp(`<${tag}[^>]*>([\\s\\S]*?)</${tag}>`));
|
|
@@ -120,12 +144,10 @@ export async function fetchTeamtailor(slug) {
|
|
|
120
144
|
}
|
|
121
145
|
|
|
122
146
|
/**
|
|
123
|
-
* Check if a company has a TeamTailor career site
|
|
147
|
+
* Check if a company has a TeamTailor career site: true when a regional
|
|
148
|
+
* host serves the feed, false when every host answers 404, and the
|
|
149
|
+
* AtsError from resolveFeed for anything else.
|
|
124
150
|
*/
|
|
125
151
|
export async function hasTeamtailor(slug) {
|
|
126
|
-
|
|
127
|
-
return (await resolveFeed(slug, 'HEAD')) !== null;
|
|
128
|
-
} catch {
|
|
129
|
-
return false;
|
|
130
|
-
}
|
|
152
|
+
return (await resolveFeed(slug, 'HEAD')) !== null;
|
|
131
153
|
}
|