jd-intel 0.8.3 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,21 +1,32 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
+ import { atsFetch, probeResult } from '../http.js';
4
+ import { orgHost } from '../boards.js';
3
5
 
4
6
  /**
5
7
  * Fetch jobs from a Recruitee career site.
6
8
  * Public API, no auth required.
7
9
  * Docs: https://docs.recruitee.com/reference/offers
8
10
  *
9
- * Single GET returns every offer with the full HTML description
10
- * inline — no N+1 (unlike SmartRecruiters), no XML (unlike
11
- * TeamTailor/Personio). The simplest adapter shape in the toolkit.
11
+ * Single GET returns every offer inline — no N+1 (unlike SmartRecruiters),
12
+ * no XML (unlike TeamTailor/Personio). The simplest adapter shape in the
13
+ * toolkit.
14
+ *
15
+ * Each offer carries two HTML fields, `description` and `requirements`.
16
+ * Which one holds the role depends on the tenant's template (and sometimes
17
+ * the posting): some keep the duties in `description` and the candidate
18
+ * profile in `requirements`, others put a company intro in `description`
19
+ * and everything else in `requirements`. Neither alone is the posting, so
20
+ * both are joined before normalize() strips them (issue #65).
12
21
  *
13
22
  * @param {string} slug - Recruitee company subdomain (e.g., 'vandebron')
23
+ * @param {object} [ctx] - { report }; report is called once with
24
+ * { ats, org_name, org_url } when given
14
25
  * @returns {Promise<Array>} Normalized job objects
15
26
  */
16
- export async function fetchRecruitee(slug) {
27
+ export async function fetchRecruitee(slug, ctx = {}) {
17
28
  const url = `https://${slug}.recruitee.com/api/offers/`;
18
- const resp = await fetch(url);
29
+ const resp = await atsFetch(url);
19
30
 
20
31
  if (!resp.ok) {
21
32
  if (resp.status === 404) return []; // No Recruitee site for this slug
@@ -25,18 +36,23 @@ export async function fetchRecruitee(slug) {
25
36
  const data = await resp.json();
26
37
  const offers = data.offers || [];
27
38
 
39
+ // The response is { offers } only, so the identity lives on the rows:
40
+ // company_name, and careers_url, which sits on the company's own careers
41
+ // domain when the site has one and on {slug}.recruitee.com otherwise.
42
+ if (typeof ctx.report === 'function') {
43
+ ctx.report({
44
+ ats: 'recruitee',
45
+ org_name: offers.find(o => o.company_name)?.company_name || null,
46
+ org_url: orgHost(offers.find(o => o.careers_url)?.careers_url),
47
+ });
48
+ }
49
+
28
50
  return offers.map(offer => {
29
51
  const place = [offer.city, offer.country].filter(Boolean).join(', ');
30
52
  let location = place;
31
53
  if (offer.remote) location = place ? `Remote - ${place}` : 'Remote';
32
54
 
33
- let postedAt = null;
34
- if (offer.created_at) {
35
- // Recruitee returns "2026-05-13 07:38:11 UTC"; coerce to ISO.
36
- const iso = offer.created_at.replace(' UTC', 'Z').replace(' ', 'T');
37
- const d = new Date(iso);
38
- if (!Number.isNaN(d.getTime())) postedAt = d.toISOString();
39
- }
55
+ const createdAt = toIso(offer.created_at);
40
56
 
41
57
  return normalize({
42
58
  companySlug: slug,
@@ -44,27 +60,75 @@ export async function fetchRecruitee(slug) {
44
60
  title: offer.title || '',
45
61
  department: offer.department || '',
46
62
  location,
47
- description: stripHtml(offer.description || ''),
63
+ locations: (offer.locations || []).map(l => [l.city, l.country].filter(Boolean).join(', ')),
64
+ workplace: parseRecruiteeWorkplace(offer),
65
+ description: [offer.description, offer.requirements].filter(Boolean).join('\n'),
48
66
  url: offer.careers_url || offer.careers_apply_url || '',
49
- postedAt,
50
- salary: null, // No structured salary; normalizer parses from text
67
+ // created_at can predate publication by years on long-lived offers,
68
+ // so it is not a posting date. published_at is.
69
+ postedAt: toIso(offer.published_at) || createdAt,
70
+ salary: parseRecruiteeSalary(offer.salary),
51
71
  metadata: {
52
72
  recruiteeId: offer.guid || offer.id,
53
73
  employmentType: offer.employment_type_code || '',
54
74
  category: offer.category_code || '',
75
+ createdAt,
55
76
  },
56
77
  }, 'recruitee');
57
78
  });
58
79
  }
59
80
 
60
81
  /**
61
- * Check if a company has a Recruitee career site.
82
+ * Recruitee returns "2026-05-13 07:38:11 UTC"; coerce to ISO.
83
+ */
84
+ function toIso(ts) {
85
+ if (!ts) return null;
86
+ const d = new Date(ts.replace(' UTC', 'Z').replace(' ', 'T'));
87
+ return Number.isNaN(d.getTime()) ? null : d.toISOString();
88
+ }
89
+
90
+ /**
91
+ * Recruitee sends three booleans, not one enum. Hybrid wins when remote is
92
+ * also set, and on_site alone is onsite. All false is no signal.
93
+ */
94
+ function parseRecruiteeWorkplace(offer) {
95
+ if (offer.hybrid) return 'hybrid';
96
+ if (offer.remote) return 'remote';
97
+ if (offer.on_site) return 'onsite';
98
+ return null;
99
+ }
100
+
101
+ const PERIODS = new Set(['year', 'month', 'hour']);
102
+
103
+ /**
104
+ * Recruitee sends `salary` as `{min, max, period, currency}` with string
105
+ * amounts. Offers without pay still carry the object, either all-null or
106
+ * as a "0"/"0" placeholder, so anything without a positive side returns
107
+ * null and normalize() falls back to the posting text.
108
+ */
109
+ function parseRecruiteeSalary(salary) {
110
+ if (!salary) return null;
111
+ const min = toAmount(salary.min);
112
+ const max = toAmount(salary.max);
113
+ if (min === null && max === null) return null;
114
+ return {
115
+ min,
116
+ max,
117
+ currency: (salary.currency || '').toUpperCase(),
118
+ period: PERIODS.has(salary.period) ? salary.period : null,
119
+ source: 'ats',
120
+ };
121
+ }
122
+
123
+ function toAmount(value) {
124
+ const n = parseFloat(value);
125
+ return Number.isFinite(n) && n > 0 ? n : null;
126
+ }
127
+
128
+ /**
129
+ * Check if a company has a Recruitee career site. See probeResult for the outcomes.
62
130
  */
63
131
  export async function hasRecruitee(slug) {
64
- try {
65
- const resp = await fetch(`https://${slug}.recruitee.com/api/offers/`);
66
- return resp.ok;
67
- } catch {
68
- return false;
69
- }
132
+ const resp = await atsFetch(`https://${slug}.recruitee.com/api/offers/`);
133
+ return probeResult(resp, `Recruitee probe for ${slug}`);
70
134
  }
@@ -1,33 +1,45 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
+ import { atsFetch, probeResult } from '../http.js';
4
+ import { makeLocationMatcher } from '../filters.js';
3
5
 
4
6
  const BASE_URL = 'https://api.smartrecruiters.com/v1/companies';
5
7
  const PAGE_SIZE = 100;
8
+ const MAX_DETAIL_FETCHES = 100;
6
9
 
7
10
  /**
8
- * Fetch all postings from a SmartRecruiters company.
11
+ * Fetch postings from a SmartRecruiters company.
9
12
  * Public API, no auth required.
10
13
  * Docs: https://developers.smartrecruiters.com/reference/postingsget-1
11
14
  *
12
15
  * Two-step flow (unavoidable N+1):
13
- * - The postings LIST endpoint omits the job description entirely.
16
+ * - The postings LIST endpoint omits the job description entirely,
17
+ * and the structured `compensation` block with it.
14
18
  * - jd-intel's contract is "full JD text", so we must fetch each
15
19
  * posting's DETAIL endpoint to get jobAd.sections.
16
- * Large enterprise tenants with hundreds of openings will therefore be
17
- * slow against SmartRecruiters specifically. This is the API's shape,
18
- * not a bug here.
20
+ *
21
+ * The list does carry name, location and releasedDate, so the same
22
+ * pre-filter and detail budget Workday applies run here: list-evaluable
23
+ * filters narrow the candidates, then at most MAX_DETAIL_FETCHES of them
24
+ * are hydrated (see the budget note below). Without a filterContext the
25
+ * cap still holds, so a direct call on a 400-posting tenant reads 100.
19
26
  *
20
27
  * @param {string} slug - SmartRecruiters company identifier (e.g., 'Visa')
28
+ * @param {object} [ctx] - { filterContext, report }; report is called once
29
+ * with { ats, listed, prefiltered, hydrated, capped, org_name, org_url }
30
+ * when given
21
31
  * @returns {Promise<Array>} Normalized job objects
22
32
  */
23
- export async function fetchSmartrecruiters(slug) {
33
+ export async function fetchSmartrecruiters(slug, ctx = {}) {
34
+ const fc = ctx.filterContext || {};
35
+
24
36
  // 1. Page through the postings list.
25
37
  const postings = [];
26
38
  let offset = 0;
27
39
 
28
40
  while (true) {
29
41
  const listUrl = `${BASE_URL}/${slug}/postings?limit=${PAGE_SIZE}&offset=${offset}`;
30
- const resp = await fetch(listUrl);
42
+ const resp = await atsFetch(listUrl);
31
43
 
32
44
  if (!resp.ok) {
33
45
  if (resp.status === 404) return []; // Company not found
@@ -42,20 +54,89 @@ export async function fetchSmartrecruiters(slug) {
42
54
  if (content.length === 0 || offset >= (data.totalFound || 0)) break;
43
55
  }
44
56
 
45
- // 2. Fetch detail per posting for the description.
46
- const jobs = await Promise.all(postings.map(async (p) => {
57
+ // 2. Filter-aware candidate selection BEFORE the N+1 detail cost.
58
+ // The list row carries name, location and releasedDate, and the
59
+ // library re-applies every filter after this returns, so a keep here
60
+ // is never final. The detail adds no location (unlike Workday's
61
+ // additionalLocations), so a row with none follows the library's
62
+ // rule now: out under includes, kept under excludes.
63
+ let candidates = postings;
64
+
65
+ if (fc.titleFilter) {
66
+ const re = new RegExp(fc.titleFilter, 'i');
67
+ candidates = candidates.filter(p => re.test(p.name || ''));
68
+ }
69
+ if (Array.isArray(fc.locationIncludes) && fc.locationIncludes.length > 0) {
70
+ const matchers = fc.locationIncludes.map(makeLocationMatcher);
71
+ candidates = candidates.filter(p => {
72
+ const loc = listLocation(p).location.toLowerCase();
73
+ return matchers.some(m => m(loc));
74
+ });
75
+ }
76
+ if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
77
+ const matchers = fc.locationExcludes.map(makeLocationMatcher);
78
+ candidates = candidates.filter(p => {
79
+ const loc = listLocation(p).location.toLowerCase();
80
+ return !loc || !matchers.some(m => m(loc));
81
+ });
82
+ }
83
+ if (typeof fc.postedWithinDays === 'number') {
84
+ // postedAt comes from releasedDate alone, so the library's rule can
85
+ // run here in full: a missing or unparseable date is out either way.
86
+ const cutoff = Date.now() - fc.postedWithinDays * 86400000;
87
+ candidates = candidates.filter(p => {
88
+ const released = new Date(p.releasedDate || '').getTime();
89
+ return Number.isFinite(released) && released >= cutoff;
90
+ });
91
+ }
92
+
93
+ // 3. Bound the detail-fetch set, Workday's reasoning verbatim: a
94
+ // description `filter` is applied by the library AFTER this returns,
95
+ // so that case keeps the full backstop instead of truncating to
96
+ // `limit` (which could hydrate jobs that all fail the regex while
97
+ // better matches go unscanned). The library pages with `offset`
98
+ // after this returns, so the budget covers the page plus what
99
+ // precedes it. Candidates keep list order.
100
+ const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
101
+ const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
102
+ const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
103
+ const hydrate = candidates.slice(0, cap);
104
+
105
+ // Every list row carries company { identifier, name }. Neither the list
106
+ // nor the detail has a company website, and postingUrl is always on
107
+ // jobs.smartrecruiters.com, so org_url stays null (issue #58).
108
+ if (typeof ctx.report === 'function') {
109
+ ctx.report({
110
+ ats: 'smartrecruiters',
111
+ listed: postings.length,
112
+ prefiltered: candidates.length,
113
+ hydrated: hydrate.length,
114
+ capped: hydrate.length < candidates.length,
115
+ org_name: postings.find(p => p.company?.name)?.company.name || null,
116
+ org_url: null,
117
+ });
118
+ }
119
+
120
+ // 4. Fetch detail per candidate for the description. atsFetch's per-host
121
+ // queue keeps this fan-out to 4 requests at a time (a 412-posting
122
+ // tenant measured 53s unbounded), so the cap also holds the detail
123
+ // step to roughly 13s.
124
+ const jobs = await Promise.all(hydrate.map(async (p) => {
47
125
  let sections = {};
48
126
  let postingUrl = '';
127
+ let salary = null;
49
128
 
50
129
  try {
51
- const detailResp = await fetch(`${BASE_URL}/${slug}/postings/${p.id}`);
130
+ const detailResp = await atsFetch(`${BASE_URL}/${slug}/postings/${p.id}`);
52
131
  if (detailResp.ok) {
53
132
  const detail = await detailResp.json();
54
133
  sections = detail.jobAd?.sections || {};
55
134
  postingUrl = detail.postingUrl || detail.applyUrl || '';
135
+ salary = parseCompensation(detail.compensation);
56
136
  }
57
137
  } catch {
58
- // Detail fetch failed: fall back to list-only fields (no description).
138
+ // Detail fetch failed, retries included: fall back to list-only
139
+ // fields (no description). Reporting this is #85.
59
140
  }
60
141
 
61
142
  const description = [
@@ -64,12 +145,7 @@ export async function fetchSmartrecruiters(slug) {
64
145
  sections.additionalInformation?.text,
65
146
  ].filter(Boolean).join('\n\n');
66
147
 
67
- const loc = p.location || {};
68
- const place = loc.fullLocation
69
- || [loc.city, loc.region, loc.country].filter(Boolean).join(', ');
70
- let location = place;
71
- if (loc.remote) location = `Remote - ${place}`.replace(/ - $/, ' ');
72
- else if (loc.hybrid) location = `Hybrid - ${place}`.replace(/ - $/, ' ');
148
+ const { location, workplace } = listLocation(p);
73
149
 
74
150
  return normalize({
75
151
  companySlug: slug,
@@ -77,10 +153,11 @@ export async function fetchSmartrecruiters(slug) {
77
153
  title: p.name || '',
78
154
  department: p.department?.label || p.function?.label || '',
79
155
  location,
80
- description: stripHtml(description),
156
+ workplace,
157
+ description,
81
158
  url: postingUrl,
82
159
  postedAt: p.releasedDate || null,
83
- salary: null, // SmartRecruiters has no structured salary; normalizer parses text
160
+ salary, // null when the detail has no compensation; normalize() then parses text
84
161
  metadata: {
85
162
  smartRecruitersId: p.id,
86
163
  refNumber: p.refNumber || '',
@@ -95,20 +172,55 @@ export async function fetchSmartrecruiters(slug) {
95
172
  }
96
173
 
97
174
  /**
98
- * Check if a company exists on SmartRecruiters.
99
- * (HEAD isn't reliably supported on the postings endpoint, so use a
100
- * minimal GET.)
175
+ * The location string and workplace type a list row yields. Built once
176
+ * here so the pre-filter matches exactly what the normalized job carries,
177
+ * "Remote - " and "Hybrid - " prefixes included.
178
+ */
179
+ function listLocation(p) {
180
+ const loc = p.location || {};
181
+ const place = loc.fullLocation
182
+ || [loc.city, loc.region, loc.country].filter(Boolean).join(', ');
183
+ if (loc.remote) return { location: `Remote - ${place}`.replace(/ - $/, ' '), workplace: 'remote' };
184
+ if (loc.hybrid) return { location: `Hybrid - ${place}`.replace(/ - $/, ' '), workplace: 'hybrid' };
185
+ return { location: place, workplace: null };
186
+ }
187
+
188
+ const PERIODS = { YEARLY: 'year', MONTHLY: 'month', HOURLY: 'hour' };
189
+
190
+ /**
191
+ * Map the detail response's `compensation` to the shared salary shape.
192
+ *
193
+ * SmartRecruiters publishes `{min?, max?, currency, period}`, and both
194
+ * one-sided cases occur (a "max only" cap, a "from" floor), so each bound
195
+ * is passed through as null when absent rather than dropping the whole
196
+ * range. The period is kept as published: a MONTHLY figure is not
197
+ * annualized because tenants occasionally mislabel it (issue #70).
198
+ */
199
+ function parseCompensation(comp) {
200
+ if (!comp || !comp.currency) return null;
201
+ const min = Number.isFinite(comp.min) ? comp.min : null;
202
+ const max = Number.isFinite(comp.max) ? comp.max : null;
203
+ if (min === null && max === null) return null;
204
+ return {
205
+ min,
206
+ max,
207
+ currency: comp.currency,
208
+ period: PERIODS[comp.period] ?? null,
209
+ source: 'ats',
210
+ };
211
+ }
212
+
213
+ /**
214
+ * Check if a company exists on SmartRecruiters. See probeResult for the
215
+ * outcomes. (HEAD isn't reliably supported on the postings endpoint, so
216
+ * use a minimal GET.)
101
217
  */
102
218
  export async function hasSmartrecruiters(slug) {
103
- try {
104
- const resp = await fetch(`${BASE_URL}/${slug}/postings?limit=1`);
105
- if (!resp.ok) return false;
106
- // SmartRecruiters returns 200 with an empty page (not 404) for unknown
107
- // companies, so resp.ok alone false-positives on any slug. Confirm at
108
- // least one real posting exists before claiming a match.
109
- const data = await resp.json();
110
- return (data.totalFound || 0) > 0 || (data.content || []).length > 0;
111
- } catch {
112
- return false;
113
- }
219
+ const resp = await atsFetch(`${BASE_URL}/${slug}/postings?limit=1`);
220
+ if (!probeResult(resp, `SmartRecruiters probe for ${slug}`)) return false;
221
+ // SmartRecruiters returns 200 with an empty page (not 404) for unknown
222
+ // companies, so resp.ok alone false-positives on any slug. Confirm at
223
+ // least one real posting exists before claiming a match.
224
+ const data = await resp.json();
225
+ return (data.totalFound || 0) > 0 || (data.content || []).length > 0;
114
226
  }
@@ -1,5 +1,7 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize, decodeEntities } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
+ import { atsFetch } from '../http.js';
4
+ import { orgHost } from '../boards.js';
3
5
 
4
6
  /**
5
7
  * Fetch jobs from a TeamTailor career site via its public RSS feed.
@@ -16,29 +18,40 @@ import { atsErrorFromStatus } from '../errors.js';
16
18
  * to a custom domain (e.g. jobs.tibber.com).
17
19
  *
18
20
  * RSS quirk: descriptions are HTML-entity-encoded inside the XML
19
- * (`&lt;p&gt;...`). We decode that outer layer to real HTML, then
20
- * hand it to stripHtml() which strips tags and resolves the inner
21
- * entities. Decode order matters — `&amp;` resolves LAST so that
22
- * double-encoded sequences (`&amp;amp;`) collapse correctly.
21
+ * (`&lt;p&gt;...`). We decode that outer layer to real HTML with the
22
+ * shared decodeEntities() and hand the HTML to normalize(), which
23
+ * strips tags and resolves the inner entities. Decode order matters —
24
+ * `&amp;` resolves LAST so double-encoded sequences (`&amp;amp;`)
25
+ * collapse by one layer per pass.
23
26
  *
24
27
  * @param {string} slug - TeamTailor career-site slug (e.g., 'tibber')
28
+ * @param {object} [ctx] - { report }; report is called once with
29
+ * { ats, org_name, org_url } when given
25
30
  * @returns {Promise<Array>} Normalized job objects
26
31
  */
27
32
  // Most sites are {slug}.teamtailor.com, but some sit on a regional
28
- // segment, e.g. crunchbase.na.teamtailor.com. '' is the base host.
29
- const TT_REGIONS = ['', 'na', 'eu'];
33
+ // segment, e.g. crunchbase.na.teamtailor.com. '' is the base host. There
34
+ // is no reachable eu segment: {slug}.eu.teamtailor.com fails TLS for every
35
+ // slug, known or not, because the wildcard certificate covers one label
36
+ // only (live check 2026-09-27). Probing it was a guaranteed failure that
37
+ // the has() contract would now report as an outage.
38
+ const TT_REGIONS = ['', 'na'];
39
+
40
+ // Feeds send `none`, `hybrid`, `fully` or `onsite`. `none` is no signal.
41
+ const REMOTE_STATUS = { hybrid: 'hybrid', fully: 'remote', onsite: 'onsite' };
30
42
 
31
43
  /**
32
44
  * Resolve which TeamTailor host actually serves this slug's feed.
33
- * Returns the first 200 Response, throws on a non-404 error, or
34
- * returns null if no region has a feed.
45
+ * Returns the first 200 Response, throws on a non-404 error (atsFetch
46
+ * throws the 429, 5xx and network cases itself), or returns null if no
47
+ * region has a feed.
35
48
  */
36
49
  async function resolveFeed(slug, method = 'GET') {
37
50
  for (const region of TT_REGIONS) {
38
51
  const host = region
39
52
  ? `${slug}.${region}.teamtailor.com`
40
53
  : `${slug}.teamtailor.com`;
41
- const resp = await fetch(`https://${host}/jobs.rss`, {
54
+ const resp = await atsFetch(`https://${host}/jobs.rss`, {
42
55
  method,
43
56
  redirect: 'follow',
44
57
  });
@@ -51,21 +64,36 @@ async function resolveFeed(slug, method = 'GET') {
51
64
  return null;
52
65
  }
53
66
 
54
- export async function fetchTeamtailor(slug) {
67
+ export async function fetchTeamtailor(slug, ctx = {}) {
55
68
  const resp = await resolveFeed(slug, 'GET');
56
69
  if (!resp) return []; // No TeamTailor site in any known region
57
70
 
58
71
  const xml = await resp.text();
59
72
 
60
- const company = (
61
- xml.match(/<channel>[\s\S]*?<title>([\s\S]*?)<\/title>/)?.[1] || slug
62
- ).trim();
73
+ const channelTitle = (xml.match(/<channel>[\s\S]*?<title>([\s\S]*?)<\/title>/)?.[1] || '').trim();
74
+ const company = channelTitle || slug;
63
75
 
64
76
  const items = [...xml.matchAll(/<item>([\s\S]*?)<\/item>/g)].map(m => m[1]);
65
77
 
78
+ // The channel title is the company as the site names itself. The channel
79
+ // <link> always sits on {slug}.teamtailor.com, but item links follow the
80
+ // site's custom domain when it has one (jobs.tibber.com on a feed served
81
+ // from tibber.teamtailor.com), so the first item's link is the host that
82
+ // can say something; the channel link is the fallback for an empty feed.
83
+ if (typeof ctx.report === 'function') {
84
+ const link = items[0]?.match(/<link>([\s\S]*?)<\/link>/)?.[1]
85
+ || xml.match(/<channel>[\s\S]*?<link>([\s\S]*?)<\/link>/)?.[1]
86
+ || '';
87
+ ctx.report({
88
+ ats: 'teamtailor',
89
+ org_name: decodeEntities(channelTitle) || null,
90
+ org_url: orgHost(link.trim()),
91
+ });
92
+ }
93
+
66
94
  return items.map(item => {
67
- const pick = (tag) => {
68
- const m = item.match(new RegExp(`<${tag}[^>]*>([\\s\\S]*?)</${tag}>`));
95
+ const pick = (tag, src = item) => {
96
+ const m = src.match(new RegExp(`<${tag}[^>]*>([\\s\\S]*?)</${tag}>`));
69
97
  return m ? m[1].trim() : '';
70
98
  };
71
99
 
@@ -78,6 +106,12 @@ export async function fetchTeamtailor(slug) {
78
106
  const country = decodeEntities(pick('tt:country'));
79
107
  const remoteStatus = decodeEntities(pick('remoteStatus'));
80
108
 
109
+ // One <tt:location> per office the posting is open in, read the same
110
+ // way as the primary above so the entries line up.
111
+ const locations = [...item.matchAll(/<tt:location>([\s\S]*?)<\/tt:location>/g)].map(m =>
112
+ [decodeEntities(pick('tt:city', m[1])), decodeEntities(pick('tt:country', m[1]))].filter(Boolean).join(', ')
113
+ );
114
+
81
115
  let location = [city, country].filter(Boolean).join(', ');
82
116
  if (/remote/i.test(remoteStatus)) {
83
117
  location = location ? `Remote - ${location}` : 'Remote';
@@ -95,7 +129,9 @@ export async function fetchTeamtailor(slug) {
95
129
  title,
96
130
  department,
97
131
  location,
98
- description: stripHtml(decodeEntities(pick('description'))),
132
+ locations,
133
+ workplace: REMOTE_STATUS[remoteStatus.toLowerCase()] || null,
134
+ description: decodeEntities(pick('description')),
99
135
  url: link,
100
136
  postedAt,
101
137
  salary: null, // No structured salary; normalizer parses from text
@@ -108,28 +144,10 @@ export async function fetchTeamtailor(slug) {
108
144
  }
109
145
 
110
146
  /**
111
- * Decode the RSS entity/CDATA layer to real HTML.
112
- * `&amp;` is intentionally resolved LAST.
113
- */
114
- function decodeEntities(s) {
115
- if (!s) return '';
116
- return s
117
- .replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1')
118
- .replace(/&lt;/g, '<')
119
- .replace(/&gt;/g, '>')
120
- .replace(/&quot;/g, '"')
121
- .replace(/&#39;/g, "'")
122
- .replace(/&apos;/g, "'")
123
- .replace(/&amp;/g, '&');
124
- }
125
-
126
- /**
127
- * Check if a company has a TeamTailor career site.
147
+ * Check if a company has a TeamTailor career site: true when a regional
148
+ * host serves the feed, false when every host answers 404, and the
149
+ * AtsError from resolveFeed for anything else.
128
150
  */
129
151
  export async function hasTeamtailor(slug) {
130
- try {
131
- return (await resolveFeed(slug, 'HEAD')) !== null;
132
- } catch {
133
- return false;
134
- }
152
+ return (await resolveFeed(slug, 'HEAD')) !== null;
135
153
  }