jd-intel 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -144,8 +144,22 @@ const { jobs, total_matched } = await fetchJobsDetailed({
144
144
  });
145
145
  ```
146
146
 
147
+ The same result says how the company was resolved and which boards answered:
148
+
149
+ - `match`: `registry` (a known company, one adapter call), `probe` (not in the registry, every adapter asked) or `workday_override` (an explicit Workday config).
150
+ - `company`: `{ key, name }` from the registry row on a registry match, null otherwise.
151
+ - `boards`: one entry per board that answered, `{ ats, slug, name, site, board_url, org_name, org_url, jobs_found, matched, selected, scan }`. `jobs_found` counts the rows the board listed before any filter. Workday and SmartRecruiters filter their list before fetching details, so for them it is the list count, not the rows that came back (a lower bound when the list scan caps). `matched` is the rows left after filters and before `offset` and `limit`, so a board whose rows were all cut from the page still shows up. `board_url` is built from the slug. `org_name` and `org_url` are what the board states about itself, the organization name in the ATS response and the careers or company host its links point at, and they are null where the platform exposes nothing (Lever and Ashby expose neither); nothing is filled in from the slug or the registry name. A `probe` match is a slug match, not a confirmed identity: compare `org_name` and `org_url` with the company you meant, and open `board_url` when they are null, before treating the jobs as that company's.
152
+ - `failed`: adapters that threw during discovery, `{ ats, slug, name, code, message }`, with `code` either `rate_limited` or `ats_unreachable`. When no board answered and a check failed, `fetchJobs` throws that error instead of returning `[]`, so an outage never reads as "not found".
153
+ - `total_before_filters`: the sum of `jobs_found`. Zero matches with a positive `total_before_filters` means the company is hiring and nothing passed the filters, on every ATS.
154
+
155
+ `detectAtsDetailed(company)` returns `{ boards, failed }`: registry rows first (`source: 'registry'`, never probed), then every live probe that answered (`source: 'probe'`), in platform order; `failed` lists the probes that could not be checked, with the same codes. `detectAts` keeps returning `[{ ats, slug }]`.
156
+
157
+ A call the library cannot make throws `ArgumentError` (`code: 'invalid_args'`): no company, an unknown `ats`, or a `titleFilter` or `filter` that does not compile as a regex. It is thrown before any request goes out.
158
+
147
159
  CLI usage: `npx jd-intel fetch <company-slug> --title-filter "engineer" --posted-within-days 14`. Full filter reference [below](#filters-quick-reference).
148
160
 
161
+ Each ATS request gives the server 10 seconds to start responding. A 429, a 5xx or a network error is retried up to three attempts with backoff (1s, 2s), honoring `Retry-After` when the ATS sends one, and at most 4 requests run at a time per host. A failure that outlasts the retries throws an `AtsError` whose `code` is `rate_limited` or `ats_unreachable`.
162
+
149
163
  Node.js 18+. No API keys. No configuration.
150
164
 
151
165
  ### Manual install (fallback)
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "jd-intel",
3
- "version": "0.9.0",
3
+ "version": "0.10.0",
4
4
  "description": "Fetch and normalize job descriptions across seven major ATS (Greenhouse, Lever, Ashby, Workday, and more), for your AI assistant. No copy-paste.",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
@@ -1,7 +1,7 @@
1
1
  import { normalize, extractSalaryFromText } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
+ import { atsFetch, probeResult } from '../http.js';
3
4
 
4
- const API_URL = 'https://jobs.ashbyhq.com/api/non-user-graphql';
5
5
  const BOARD_URL = 'https://api.ashbyhq.com/posting-api/job-board';
6
6
 
7
7
  /**
@@ -9,23 +9,18 @@ const BOARD_URL = 'https://api.ashbyhq.com/posting-api/job-board';
9
9
  * Public API, no auth required.
10
10
  * Docs: https://developers.ashbyhq.com/docs/public-job-posting-api
11
11
  *
12
+ * REST only. The GraphQL fallback this adapter once carried never named a
13
+ * board, so it never returned a job, and it turned every REST 429 or 5xx
14
+ * into a silent empty result (issue #55).
15
+ *
12
16
  * @param {string} slug - Company slug (e.g., 'notion', 'linear')
17
+ * @param {object} [ctx] - { report }; report is called once with
18
+ * { ats, org_name, org_url } when given
13
19
  * @returns {Promise<Array>} Normalized job objects
14
20
  */
15
- export async function fetchAshby(slug) {
16
- // Try the REST API first (simpler, includes compensation)
17
- try {
18
- const restJobs = await fetchAshbyRest(slug);
19
- if (restJobs.length > 0) return restJobs;
20
- } catch { /* fall through to GraphQL */ }
21
-
22
- // Fallback: GraphQL API
23
- return fetchAshbyGraphQL(slug);
24
- }
25
-
26
- async function fetchAshbyRest(slug) {
21
+ export async function fetchAshby(slug, ctx = {}) {
27
22
  const url = `${BOARD_URL}/${slug}?includeCompensation=true`;
28
- const resp = await fetch(url);
23
+ const resp = await atsFetch(url);
29
24
 
30
25
  if (!resp.ok) {
31
26
  if (resp.status === 404) return [];
@@ -35,6 +30,12 @@ async function fetchAshbyRest(slug) {
35
30
  const data = await resp.json();
36
31
  const jobs = data.jobs || [];
37
32
 
33
+ // The REST response is { jobs, apiVersion }: no organization name, and
34
+ // every link is on jobs.ashbyhq.com. Both null (issue #58).
35
+ if (typeof ctx.report === 'function') {
36
+ ctx.report({ ats: 'ashby', org_name: null, org_url: null });
37
+ }
38
+
38
39
  return jobs.map(job => {
39
40
  const comp = job.compensation || {};
40
41
 
@@ -68,58 +69,6 @@ async function fetchAshbyRest(slug) {
68
69
  });
69
70
  }
70
71
 
71
- async function fetchAshbyGraphQL(slug) {
72
- const query = `{
73
- jobBoard {
74
- title
75
- jobPostings {
76
- id
77
- title
78
- locationName
79
- employmentType
80
- descriptionHtml
81
- publishedDate
82
- compensationTierSummary
83
- }
84
- }
85
- }`;
86
-
87
- const resp = await fetch(API_URL, {
88
- method: 'POST',
89
- headers: { 'Content-Type': 'application/json' },
90
- body: JSON.stringify({
91
- operationName: 'ApiJobBoardWithTeams',
92
- variables: { organizationHostedJobsPageName: slug },
93
- query,
94
- }),
95
- });
96
-
97
- if (!resp.ok) return [];
98
-
99
- const data = await resp.json();
100
- const board = data.data?.jobBoard;
101
- if (!board) return [];
102
-
103
- const postings = board.jobPostings || [];
104
-
105
- return postings.map(job => normalize({
106
- companySlug: slug,
107
- company: board.title || slug,
108
- title: job.title || '',
109
- department: '',
110
- location: job.locationName || '',
111
- description: job.descriptionHtml || '',
112
- url: `https://jobs.ashbyhq.com/${slug}/${job.id}`,
113
- postedAt: job.publishedDate || null,
114
- salary: null,
115
- metadata: {
116
- ashbyId: job.id,
117
- employmentType: job.employmentType || '',
118
- compensationSummary: job.compensationTierSummary || '',
119
- },
120
- }, 'ashby'));
121
- }
122
-
123
72
  const WORKPLACE_TYPES = { remote: 'remote', hybrid: 'hybrid', onsite: 'onsite' };
124
73
 
125
74
  /**
@@ -163,11 +112,10 @@ function parseAshbyCompensation(comp) {
163
112
  return parsed ? { ...parsed, source: 'ats' } : null;
164
113
  }
165
114
 
115
+ /**
116
+ * Check if a company has an Ashby board. See probeResult for the outcomes.
117
+ */
166
118
  export async function hasAshby(slug) {
167
- try {
168
- const resp = await fetch(`${BOARD_URL}/${slug}`, { method: 'HEAD' });
169
- return resp.ok;
170
- } catch {
171
- return false;
172
- }
119
+ const resp = await atsFetch(`${BOARD_URL}/${slug}`, { method: 'HEAD' });
120
+ return probeResult(resp, `Ashby probe for ${slug}`);
173
121
  }
@@ -1,5 +1,6 @@
1
1
  import { normalize, decodeEntities } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
+ import { atsFetch, probeResult } from '../http.js';
3
4
 
4
5
  const BASE_URL = 'https://boards-api.greenhouse.io/v1/boards';
5
6
 
@@ -9,11 +10,13 @@ const BASE_URL = 'https://boards-api.greenhouse.io/v1/boards';
9
10
  * Docs: https://developers.greenhouse.io/job-board.html
10
11
  *
11
12
  * @param {string} slug - Company slug (e.g., 'stripe', 'notion')
13
+ * @param {object} [ctx] - { report }; report is called once with
14
+ * { ats, org_name, org_url } when given
12
15
  * @returns {Promise<Array>} Normalized job objects
13
16
  */
14
- export async function fetchGreenhouse(slug) {
17
+ export async function fetchGreenhouse(slug, ctx = {}) {
15
18
  const url = `${BASE_URL}/${slug}/jobs?content=true`;
16
- const resp = await fetch(url);
19
+ const resp = await atsFetch(url);
17
20
 
18
21
  if (!resp.ok) {
19
22
  if (resp.status === 404) return []; // Company not found or no jobs
@@ -23,6 +26,17 @@ export async function fetchGreenhouse(slug) {
23
26
  const data = await resp.json();
24
27
  const jobs = data.jobs || [];
25
28
 
29
+ // The list response has no top-level name, but each row carries the
30
+ // board's company_name. Its only links are job-boards.greenhouse.io, so
31
+ // there is no company host to report (issue #58).
32
+ if (typeof ctx.report === 'function') {
33
+ ctx.report({
34
+ ats: 'greenhouse',
35
+ org_name: jobs.find(j => j.company_name)?.company_name || null,
36
+ org_url: null,
37
+ });
38
+ }
39
+
26
40
  return jobs.map(job => normalize({
27
41
  companySlug: slug,
28
42
  company: data.name || slug,
@@ -66,13 +80,9 @@ function parseGreenhouseWorkplace(metadata) {
66
80
  }
67
81
 
68
82
  /**
69
- * Check if a company has a Greenhouse board.
83
+ * Check if a company has a Greenhouse board. See probeResult for the outcomes.
70
84
  */
71
85
  export async function hasGreenhouse(slug) {
72
- try {
73
- const resp = await fetch(`${BASE_URL}/${slug}`, { method: 'HEAD' });
74
- return resp.ok;
75
- } catch {
76
- return false;
77
- }
86
+ const resp = await atsFetch(`${BASE_URL}/${slug}`, { method: 'HEAD' });
87
+ return probeResult(resp, `Greenhouse probe for ${slug}`);
78
88
  }
@@ -1,5 +1,6 @@
1
1
  import { normalize, extractSalaryFromText } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
+ import { atsFetch, probeResult } from '../http.js';
3
4
 
4
5
  const BASE_URL = 'https://api.lever.co/v0/postings';
5
6
 
@@ -13,11 +14,13 @@ const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
13
14
  * Docs: https://github.com/lever/postings-api
14
15
  *
15
16
  * @param {string} slug - Company slug (e.g., 'stripe', 'figma')
17
+ * @param {object} [ctx] - { report }; report is called once with
18
+ * { ats, org_name, org_url } when given
16
19
  * @returns {Promise<Array>} Normalized job objects
17
20
  */
18
- export async function fetchLever(slug) {
21
+ export async function fetchLever(slug, ctx = {}) {
19
22
  const url = `${BASE_URL}/${slug}?mode=json`;
20
- const resp = await fetch(url);
23
+ const resp = await atsFetch(url);
21
24
 
22
25
  if (!resp.ok) {
23
26
  if (resp.status === 404) return [];
@@ -27,6 +30,12 @@ export async function fetchLever(slug) {
27
30
  const jobs = await resp.json();
28
31
  if (!Array.isArray(jobs)) return [];
29
32
 
33
+ // The postings response is a bare array of jobs: no organization name
34
+ // anywhere, and every link is on jobs.lever.co. Both null (issue #58).
35
+ if (typeof ctx.report === 'function') {
36
+ ctx.report({ ats: 'lever', org_name: null, org_url: null });
37
+ }
38
+
30
39
  return jobs.map(job => normalize({
31
40
  companySlug: slug,
32
41
  // Lever's API doesn't return the company name at the board or job level,
@@ -102,11 +111,10 @@ function titleCaseSlug(slug) {
102
111
  return slug.charAt(0).toUpperCase() + slug.slice(1);
103
112
  }
104
113
 
114
+ /**
115
+ * Check if a company has a Lever board. See probeResult for the outcomes.
116
+ */
105
117
  export async function hasLever(slug) {
106
- try {
107
- const resp = await fetch(`${BASE_URL}/${slug}?mode=json`, { method: 'HEAD' });
108
- return resp.ok;
109
- } catch {
110
- return false;
111
- }
118
+ const resp = await atsFetch(`${BASE_URL}/${slug}?mode=json`, { method: 'HEAD' });
119
+ return probeResult(resp, `Lever probe for ${slug}`);
112
120
  }
@@ -1,5 +1,7 @@
1
1
  import { normalize } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
+ import { atsFetch, probeResult } from '../http.js';
4
+ import { orgHost } from '../boards.js';
3
5
 
4
6
  /**
5
7
  * Fetch jobs from a Recruitee career site.
@@ -18,11 +20,13 @@ import { atsErrorFromStatus } from '../errors.js';
18
20
  * both are joined before normalize() strips them (issue #65).
19
21
  *
20
22
  * @param {string} slug - Recruitee company subdomain (e.g., 'vandebron')
23
+ * @param {object} [ctx] - { report }; report is called once with
24
+ * { ats, org_name, org_url } when given
21
25
  * @returns {Promise<Array>} Normalized job objects
22
26
  */
23
- export async function fetchRecruitee(slug) {
27
+ export async function fetchRecruitee(slug, ctx = {}) {
24
28
  const url = `https://${slug}.recruitee.com/api/offers/`;
25
- const resp = await fetch(url);
29
+ const resp = await atsFetch(url);
26
30
 
27
31
  if (!resp.ok) {
28
32
  if (resp.status === 404) return []; // No Recruitee site for this slug
@@ -32,6 +36,17 @@ export async function fetchRecruitee(slug) {
32
36
  const data = await resp.json();
33
37
  const offers = data.offers || [];
34
38
 
39
+ // The response is { offers } only, so the identity lives on the rows:
40
+ // company_name, and careers_url, which sits on the company's own careers
41
+ // domain when the site has one and on {slug}.recruitee.com otherwise.
42
+ if (typeof ctx.report === 'function') {
43
+ ctx.report({
44
+ ats: 'recruitee',
45
+ org_name: offers.find(o => o.company_name)?.company_name || null,
46
+ org_url: orgHost(offers.find(o => o.careers_url)?.careers_url),
47
+ });
48
+ }
49
+
35
50
  return offers.map(offer => {
36
51
  const place = [offer.city, offer.country].filter(Boolean).join(', ');
37
52
  let location = place;
@@ -111,13 +126,9 @@ function toAmount(value) {
111
126
  }
112
127
 
113
128
  /**
114
- * Check if a company has a Recruitee career site.
129
+ * Check if a company has a Recruitee career site. See probeResult for the outcomes.
115
130
  */
116
131
  export async function hasRecruitee(slug) {
117
- try {
118
- const resp = await fetch(`https://${slug}.recruitee.com/api/offers/`);
119
- return resp.ok;
120
- } catch {
121
- return false;
122
- }
132
+ const resp = await atsFetch(`https://${slug}.recruitee.com/api/offers/`);
133
+ return probeResult(resp, `Recruitee probe for ${slug}`);
123
134
  }
@@ -1,11 +1,14 @@
1
1
  import { normalize } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
+ import { atsFetch, probeResult } from '../http.js';
4
+ import { makeLocationMatcher } from '../filters.js';
3
5
 
4
6
  const BASE_URL = 'https://api.smartrecruiters.com/v1/companies';
5
7
  const PAGE_SIZE = 100;
8
+ const MAX_DETAIL_FETCHES = 100;
6
9
 
7
10
  /**
8
- * Fetch all postings from a SmartRecruiters company.
11
+ * Fetch postings from a SmartRecruiters company.
9
12
  * Public API, no auth required.
10
13
  * Docs: https://developers.smartrecruiters.com/reference/postingsget-1
11
14
  *
@@ -14,21 +17,29 @@ const PAGE_SIZE = 100;
14
17
  * and the structured `compensation` block with it.
15
18
  * - jd-intel's contract is "full JD text", so we must fetch each
16
19
  * posting's DETAIL endpoint to get jobAd.sections.
17
- * Large enterprise tenants with hundreds of openings will therefore be
18
- * slow against SmartRecruiters specifically. This is the API's shape,
19
- * not a bug here.
20
+ *
21
+ * The list does carry name, location and releasedDate, so the same
22
+ * pre-filter and detail budget Workday applies run here: list-evaluable
23
+ * filters narrow the candidates, then at most MAX_DETAIL_FETCHES of them
24
+ * are hydrated (see the budget note below). Without a filterContext the
25
+ * cap still holds, so a direct call on a 400-posting tenant reads 100.
20
26
  *
21
27
  * @param {string} slug - SmartRecruiters company identifier (e.g., 'Visa')
28
+ * @param {object} [ctx] - { filterContext, report }; report is called once
29
+ * with { ats, listed, prefiltered, hydrated, capped, org_name, org_url }
30
+ * when given
22
31
  * @returns {Promise<Array>} Normalized job objects
23
32
  */
24
- export async function fetchSmartrecruiters(slug) {
33
+ export async function fetchSmartrecruiters(slug, ctx = {}) {
34
+ const fc = ctx.filterContext || {};
35
+
25
36
  // 1. Page through the postings list.
26
37
  const postings = [];
27
38
  let offset = 0;
28
39
 
29
40
  while (true) {
30
41
  const listUrl = `${BASE_URL}/${slug}/postings?limit=${PAGE_SIZE}&offset=${offset}`;
31
- const resp = await fetch(listUrl);
42
+ const resp = await atsFetch(listUrl);
32
43
 
33
44
  if (!resp.ok) {
34
45
  if (resp.status === 404) return []; // Company not found
@@ -43,14 +54,80 @@ export async function fetchSmartrecruiters(slug) {
43
54
  if (content.length === 0 || offset >= (data.totalFound || 0)) break;
44
55
  }
45
56
 
46
- // 2. Fetch detail per posting for the description.
47
- const jobs = await Promise.all(postings.map(async (p) => {
57
+ // 2. Filter-aware candidate selection BEFORE the N+1 detail cost.
58
+ // The list row carries name, location and releasedDate, and the
59
+ // library re-applies every filter after this returns, so a keep here
60
+ // is never final. The detail adds no location (unlike Workday's
61
+ // additionalLocations), so a row with none follows the library's
62
+ // rule now: out under includes, kept under excludes.
63
+ let candidates = postings;
64
+
65
+ if (fc.titleFilter) {
66
+ const re = new RegExp(fc.titleFilter, 'i');
67
+ candidates = candidates.filter(p => re.test(p.name || ''));
68
+ }
69
+ if (Array.isArray(fc.locationIncludes) && fc.locationIncludes.length > 0) {
70
+ const matchers = fc.locationIncludes.map(makeLocationMatcher);
71
+ candidates = candidates.filter(p => {
72
+ const loc = listLocation(p).location.toLowerCase();
73
+ return matchers.some(m => m(loc));
74
+ });
75
+ }
76
+ if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
77
+ const matchers = fc.locationExcludes.map(makeLocationMatcher);
78
+ candidates = candidates.filter(p => {
79
+ const loc = listLocation(p).location.toLowerCase();
80
+ return !loc || !matchers.some(m => m(loc));
81
+ });
82
+ }
83
+ if (typeof fc.postedWithinDays === 'number') {
84
+ // postedAt comes from releasedDate alone, so the library's rule can
85
+ // run here in full: a missing or unparseable date is out either way.
86
+ const cutoff = Date.now() - fc.postedWithinDays * 86400000;
87
+ candidates = candidates.filter(p => {
88
+ const released = new Date(p.releasedDate || '').getTime();
89
+ return Number.isFinite(released) && released >= cutoff;
90
+ });
91
+ }
92
+
93
+ // 3. Bound the detail-fetch set, Workday's reasoning verbatim: a
94
+ // description `filter` is applied by the library AFTER this returns,
95
+ // so that case keeps the full backstop instead of truncating to
96
+ // `limit` (which could hydrate jobs that all fail the regex while
97
+ // better matches go unscanned). The library pages with `offset`
98
+ // after this returns, so the budget covers the page plus what
99
+ // precedes it. Candidates keep list order.
100
+ const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
101
+ const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
102
+ const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
103
+ const hydrate = candidates.slice(0, cap);
104
+
105
+ // Every list row carries company { identifier, name }. Neither the list
106
+ // nor the detail has a company website, and postingUrl is always on
107
+ // jobs.smartrecruiters.com, so org_url stays null (issue #58).
108
+ if (typeof ctx.report === 'function') {
109
+ ctx.report({
110
+ ats: 'smartrecruiters',
111
+ listed: postings.length,
112
+ prefiltered: candidates.length,
113
+ hydrated: hydrate.length,
114
+ capped: hydrate.length < candidates.length,
115
+ org_name: postings.find(p => p.company?.name)?.company.name || null,
116
+ org_url: null,
117
+ });
118
+ }
119
+
120
+ // 4. Fetch detail per candidate for the description. atsFetch's per-host
121
+ // queue keeps this fan-out to 4 requests at a time (a 412-posting
122
+ // tenant measured 53s unbounded), so the cap also holds the detail
123
+ // step to roughly 13s.
124
+ const jobs = await Promise.all(hydrate.map(async (p) => {
48
125
  let sections = {};
49
126
  let postingUrl = '';
50
127
  let salary = null;
51
128
 
52
129
  try {
53
- const detailResp = await fetch(`${BASE_URL}/${slug}/postings/${p.id}`);
130
+ const detailResp = await atsFetch(`${BASE_URL}/${slug}/postings/${p.id}`);
54
131
  if (detailResp.ok) {
55
132
  const detail = await detailResp.json();
56
133
  sections = detail.jobAd?.sections || {};
@@ -58,7 +135,8 @@ export async function fetchSmartrecruiters(slug) {
58
135
  salary = parseCompensation(detail.compensation);
59
136
  }
60
137
  } catch {
61
- // Detail fetch failed: fall back to list-only fields (no description).
138
+ // Detail fetch failed, retries included: fall back to list-only
139
+ // fields (no description). Reporting this is #85.
62
140
  }
63
141
 
64
142
  const description = [
@@ -67,18 +145,7 @@ export async function fetchSmartrecruiters(slug) {
67
145
  sections.additionalInformation?.text,
68
146
  ].filter(Boolean).join('\n\n');
69
147
 
70
- const loc = p.location || {};
71
- const place = loc.fullLocation
72
- || [loc.city, loc.region, loc.country].filter(Boolean).join(', ');
73
- let location = place;
74
- let workplace = null;
75
- if (loc.remote) {
76
- location = `Remote - ${place}`.replace(/ - $/, ' ');
77
- workplace = 'remote';
78
- } else if (loc.hybrid) {
79
- location = `Hybrid - ${place}`.replace(/ - $/, ' ');
80
- workplace = 'hybrid';
81
- }
148
+ const { location, workplace } = listLocation(p);
82
149
 
83
150
  return normalize({
84
151
  companySlug: slug,
@@ -104,6 +171,20 @@ export async function fetchSmartrecruiters(slug) {
104
171
  return jobs;
105
172
  }
106
173
 
174
+ /**
175
+ * The location string and workplace type a list row yields. Built once
176
+ * here so the pre-filter matches exactly what the normalized job carries,
177
+ * "Remote - " and "Hybrid - " prefixes included.
178
+ */
179
+ function listLocation(p) {
180
+ const loc = p.location || {};
181
+ const place = loc.fullLocation
182
+ || [loc.city, loc.region, loc.country].filter(Boolean).join(', ');
183
+ if (loc.remote) return { location: `Remote - ${place}`.replace(/ - $/, ' '), workplace: 'remote' };
184
+ if (loc.hybrid) return { location: `Hybrid - ${place}`.replace(/ - $/, ' '), workplace: 'hybrid' };
185
+ return { location: place, workplace: null };
186
+ }
187
+
107
188
  const PERIODS = { YEARLY: 'year', MONTHLY: 'month', HOURLY: 'hour' };
108
189
 
109
190
  /**
@@ -130,20 +211,16 @@ function parseCompensation(comp) {
130
211
  }
131
212
 
132
213
  /**
133
- * Check if a company exists on SmartRecruiters.
134
- * (HEAD isn't reliably supported on the postings endpoint, so use a
135
- * minimal GET.)
214
+ * Check if a company exists on SmartRecruiters. See probeResult for the
215
+ * outcomes. (HEAD isn't reliably supported on the postings endpoint, so
216
+ * use a minimal GET.)
136
217
  */
137
218
  export async function hasSmartrecruiters(slug) {
138
- try {
139
- const resp = await fetch(`${BASE_URL}/${slug}/postings?limit=1`);
140
- if (!resp.ok) return false;
141
- // SmartRecruiters returns 200 with an empty page (not 404) for unknown
142
- // companies, so resp.ok alone false-positives on any slug. Confirm at
143
- // least one real posting exists before claiming a match.
144
- const data = await resp.json();
145
- return (data.totalFound || 0) > 0 || (data.content || []).length > 0;
146
- } catch {
147
- return false;
148
- }
219
+ const resp = await atsFetch(`${BASE_URL}/${slug}/postings?limit=1`);
220
+ if (!probeResult(resp, `SmartRecruiters probe for ${slug}`)) return false;
221
+ // SmartRecruiters returns 200 with an empty page (not 404) for unknown
222
+ // companies, so resp.ok alone false-positives on any slug. Confirm at
223
+ // least one real posting exists before claiming a match.
224
+ const data = await resp.json();
225
+ return (data.totalFound || 0) > 0 || (data.content || []).length > 0;
149
226
  }
@@ -1,5 +1,7 @@
1
1
  import { normalize, decodeEntities } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
+ import { atsFetch } from '../http.js';
4
+ import { orgHost } from '../boards.js';
3
5
 
4
6
  /**
5
7
  * Fetch jobs from a TeamTailor career site via its public RSS feed.
@@ -23,26 +25,33 @@ import { atsErrorFromStatus } from '../errors.js';
23
25
  * collapse by one layer per pass.
24
26
  *
25
27
  * @param {string} slug - TeamTailor career-site slug (e.g., 'tibber')
28
+ * @param {object} [ctx] - { report }; report is called once with
29
+ * { ats, org_name, org_url } when given
26
30
  * @returns {Promise<Array>} Normalized job objects
27
31
  */
28
32
  // Most sites are {slug}.teamtailor.com, but some sit on a regional
29
- // segment, e.g. crunchbase.na.teamtailor.com. '' is the base host.
30
- const TT_REGIONS = ['', 'na', 'eu'];
33
+ // segment, e.g. crunchbase.na.teamtailor.com. '' is the base host. There
34
+ // is no reachable eu segment: {slug}.eu.teamtailor.com fails TLS for every
35
+ // slug, known or not, because the wildcard certificate covers one label
36
+ // only (live check 2026-09-27). Probing it was a guaranteed failure that
37
+ // the has() contract would now report as an outage.
38
+ const TT_REGIONS = ['', 'na'];
31
39
 
32
40
  // Feeds send `none`, `hybrid`, `fully` or `onsite`. `none` is no signal.
33
41
  const REMOTE_STATUS = { hybrid: 'hybrid', fully: 'remote', onsite: 'onsite' };
34
42
 
35
43
  /**
36
44
  * Resolve which TeamTailor host actually serves this slug's feed.
37
- * Returns the first 200 Response, throws on a non-404 error, or
38
- * returns null if no region has a feed.
45
+ * Returns the first 200 Response, throws on a non-404 error (atsFetch
46
+ * throws the 429, 5xx and network cases itself), or returns null if no
47
+ * region has a feed.
39
48
  */
40
49
  async function resolveFeed(slug, method = 'GET') {
41
50
  for (const region of TT_REGIONS) {
42
51
  const host = region
43
52
  ? `${slug}.${region}.teamtailor.com`
44
53
  : `${slug}.teamtailor.com`;
45
- const resp = await fetch(`https://${host}/jobs.rss`, {
54
+ const resp = await atsFetch(`https://${host}/jobs.rss`, {
46
55
  method,
47
56
  redirect: 'follow',
48
57
  });
@@ -55,18 +64,33 @@ async function resolveFeed(slug, method = 'GET') {
55
64
  return null;
56
65
  }
57
66
 
58
- export async function fetchTeamtailor(slug) {
67
+ export async function fetchTeamtailor(slug, ctx = {}) {
59
68
  const resp = await resolveFeed(slug, 'GET');
60
69
  if (!resp) return []; // No TeamTailor site in any known region
61
70
 
62
71
  const xml = await resp.text();
63
72
 
64
- const company = (
65
- xml.match(/<channel>[\s\S]*?<title>([\s\S]*?)<\/title>/)?.[1] || slug
66
- ).trim();
73
+ const channelTitle = (xml.match(/<channel>[\s\S]*?<title>([\s\S]*?)<\/title>/)?.[1] || '').trim();
74
+ const company = channelTitle || slug;
67
75
 
68
76
  const items = [...xml.matchAll(/<item>([\s\S]*?)<\/item>/g)].map(m => m[1]);
69
77
 
78
+ // The channel title is the company as the site names itself. The channel
79
+ // <link> always sits on {slug}.teamtailor.com, but item links follow the
80
+ // site's custom domain when it has one (jobs.tibber.com on a feed served
81
+ // from tibber.teamtailor.com), so the first item's link is the host that
82
+ // can say something; the channel link is the fallback for an empty feed.
83
+ if (typeof ctx.report === 'function') {
84
+ const link = items[0]?.match(/<link>([\s\S]*?)<\/link>/)?.[1]
85
+ || xml.match(/<channel>[\s\S]*?<link>([\s\S]*?)<\/link>/)?.[1]
86
+ || '';
87
+ ctx.report({
88
+ ats: 'teamtailor',
89
+ org_name: decodeEntities(channelTitle) || null,
90
+ org_url: orgHost(link.trim()),
91
+ });
92
+ }
93
+
70
94
  return items.map(item => {
71
95
  const pick = (tag, src = item) => {
72
96
  const m = src.match(new RegExp(`<${tag}[^>]*>([\\s\\S]*?)</${tag}>`));
@@ -120,12 +144,10 @@ export async function fetchTeamtailor(slug) {
120
144
  }
121
145
 
122
146
  /**
123
- * Check if a company has a TeamTailor career site.
147
+ * Check if a company has a TeamTailor career site: true when a regional
148
+ * host serves the feed, false when every host answers 404, and the
149
+ * AtsError from resolveFeed for anything else.
124
150
  */
125
151
  export async function hasTeamtailor(slug) {
126
- try {
127
- return (await resolveFeed(slug, 'HEAD')) !== null;
128
- } catch {
129
- return false;
130
- }
152
+ return (await resolveFeed(slug, 'HEAD')) !== null;
131
153
  }