jd-intel 0.8.2 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,4 +1,4 @@
1
- import { normalize } from '../normalizer.js';
1
+ import { normalize, extractSalaryFromText } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
 
4
4
  const API_URL = 'https://jobs.ashbyhq.com/api/non-user-graphql';
@@ -36,23 +36,33 @@ async function fetchAshbyRest(slug) {
36
36
  const jobs = data.jobs || [];
37
37
 
38
38
  return jobs.map(job => {
39
- const salary = parseAshbyCompensation(job.compensation);
39
+ const comp = job.compensation || {};
40
40
 
41
41
  return normalize({
42
42
  companySlug: slug,
43
43
  company: data.organizationName || slug,
44
44
  title: job.title || '',
45
- department: job.departmentName || '',
45
+ department: job.department || '',
46
46
  location: job.location || '',
47
+ locations: (job.secondaryLocations || []).map(l => l?.location || ''),
48
+ workplace: parseAshbyWorkplace(job),
47
49
  description: job.descriptionHtml || job.descriptionPlain || '',
48
50
  url: `https://jobs.ashbyhq.com/${slug}/${job.id}`,
49
51
  postedAt: job.publishedAt || null,
50
- salary,
52
+ salary: parseAshbyCompensation(comp),
51
53
  metadata: {
52
54
  ashbyId: job.id,
53
55
  employmentType: job.employmentType || '',
54
56
  isRemote: job.isRemote || false,
55
- team: job.teamName || '',
57
+ team: job.team || '',
58
+ // The rendered summaries keep what min/max drop: "Offers Equity",
59
+ // "Multiple Ranges", and per-location tiers labelled OTE.
60
+ compensationSummary: comp.compensationTierSummary || '',
61
+ compensationTiers: (comp.compensationTiers || []).map(tier => ({
62
+ title: tier.title || '',
63
+ summary: tier.tierSummary || '',
64
+ additionalInformation: tier.additionalInformation || '',
65
+ })),
56
66
  },
57
67
  }, 'ashby');
58
68
  });
@@ -110,22 +120,47 @@ async function fetchAshbyGraphQL(slug) {
110
120
  }, 'ashby'));
111
121
  }
112
122
 
123
+ const WORKPLACE_TYPES = { remote: 'remote', hybrid: 'hybrid', onsite: 'onsite' };
124
+
125
+ /**
126
+ * `workplaceType` is 'Remote', 'Hybrid' or 'OnSite'. `isRemote` is the
127
+ * older flag and can only say remote, so it is the fallback when the type
128
+ * is absent. false means nothing: the role may be hybrid or onsite.
129
+ */
130
+ function parseAshbyWorkplace(job) {
131
+ const type = WORKPLACE_TYPES[String(job.workplaceType || '').toLowerCase()];
132
+ if (type) return type;
133
+ return job.isRemote === true ? 'remote' : null;
134
+ }
135
+
136
+ const INTERVAL_PERIOD = { '1 YEAR': 'year', '1 MONTH': 'month', '1 HOUR': 'hour' };
137
+
138
+ /**
139
+ * Read pay from Ashby's `compensation` object (issue #67).
140
+ *
141
+ * `summaryComponents` carries one structured entry per component type
142
+ * (Salary, Bonus, Commission, Equity); the Salary entry spans every tier.
143
+ * `scrapeableCompensationSalarySummary` and `compensationTierSummary` are
144
+ * the rendered strings. A board that publishes no pay still sends the
145
+ * object, with null summaries and empty arrays, so a miss here has to
146
+ * return null for the normalizer's text fallback to run.
147
+ */
113
148
  function parseAshbyCompensation(comp) {
114
- if (!comp) return null;
115
- // Ashby compensation can be a string or structured object
116
- if (typeof comp === 'string') {
117
- const match = comp.match(/\$?([\d,]+)\s*[-–]\s*\$?([\d,]+)/);
118
- if (!match) return null;
149
+ const salary = (comp.summaryComponents || []).find(c => c.compensationType === 'Salary');
150
+ if (salary && (salary.minValue != null || salary.maxValue != null)) {
119
151
  return {
120
- min: parseInt(match[1].replace(/,/g, '')),
121
- max: parseInt(match[2].replace(/,/g, '')),
122
- currency: 'USD',
152
+ min: salary.minValue ?? null,
153
+ max: salary.maxValue ?? null,
154
+ currency: salary.currencyCode || 'USD',
155
+ period: INTERVAL_PERIOD[salary.interval] || null,
156
+ source: 'ats',
123
157
  };
124
158
  }
125
- if (comp.min && comp.max) {
126
- return { min: comp.min, max: comp.max, currency: comp.currency || 'USD' };
127
- }
128
- return null;
159
+ // The summaries are still the ATS's own compensation field, so a range
160
+ // read out of one counts as source 'ats'.
161
+ const parsed = extractSalaryFromText(comp.scrapeableCompensationSalarySummary)
162
+ || extractSalaryFromText(comp.compensationTierSummary);
163
+ return parsed ? { ...parsed, source: 'ats' } : null;
129
164
  }
130
165
 
131
166
  export async function hasAshby(slug) {
@@ -1,4 +1,4 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize, decodeEntities } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
 
4
4
  const BASE_URL = 'https://boards-api.greenhouse.io/v1/boards';
@@ -29,19 +29,42 @@ export async function fetchGreenhouse(slug) {
29
29
  title: job.title || '',
30
30
  department: job.departments?.[0]?.name || '',
31
31
  location: job.location?.name || '',
32
- description: stripHtml(job.content || ''),
32
+ workplace: parseGreenhouseWorkplace(job.metadata),
33
+ // `content` arrives HTML-escaped (`<p>`). Decode that outer layer
34
+ // once so normalize() sees real tags; it strips and decodes the rest.
35
+ description: decodeEntities(job.content || ''),
33
36
  url: job.absolute_url || '',
34
- postedAt: job.updated_at || null,
35
- salary: null, // Greenhouse doesn't expose salary in public API
37
+ // updated_at is an edit time that many boards bulk-refresh, so it is not
38
+ // a posting date. first_published is. Fallback covers boards without it (#69).
39
+ postedAt: job.first_published || job.updated_at || null,
40
+ salary: null, // list endpoint has no structured pay; normalizer parses the pay transparency text
36
41
  metadata: {
37
42
  greenhouseId: job.id,
38
43
  internal_job_id: job.internal_job_id,
39
44
  departments: job.departments?.map(d => d.name) || [],
40
45
  offices: job.offices?.map(o => o.name) || [],
46
+ updatedAt: job.updated_at,
41
47
  },
42
48
  }, 'greenhouse'));
43
49
  }
44
50
 
51
+ /**
52
+ * Greenhouse has no native workplace field. Boards that track it define a
53
+ * custom field ("Location Type", "Workplace Type") that arrives in the
54
+ * job's `metadata[]`, with `value` a string for single-select fields and
55
+ * an array for multi-select. Values seen: On-Site, Hybrid (Travel-Required),
56
+ * Remote. Anything else is no signal and the location string decides.
57
+ */
58
+ function parseGreenhouseWorkplace(metadata) {
59
+ const field = (metadata || []).find(m => /location type|workplace type/i.test(m?.name || ''));
60
+ if (!field) return null;
61
+ const value = [].concat(field.value ?? []).join(' ').toLowerCase();
62
+ if (/remote/.test(value)) return 'remote';
63
+ if (/hybrid/.test(value)) return 'hybrid';
64
+ if (/on-?site/.test(value)) return 'onsite';
65
+ return null;
66
+ }
67
+
45
68
  /**
46
69
  * Check if a company has a Greenhouse board.
47
70
  */
@@ -1,11 +1,16 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize, extractSalaryFromText } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
 
4
4
  const BASE_URL = 'https://api.lever.co/v0/postings';
5
5
 
6
+ const PERIODS = { 'per-year-salary': 'year', 'per-month-salary': 'month', 'per-hour-wage': 'hour' };
7
+ // Lever's workplaceType is one of these or 'unspecified'.
8
+ const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
9
+
6
10
  /**
7
11
  * Fetch all jobs from a Lever job board.
8
12
  * Public API, no auth required.
13
+ * Docs: https://github.com/lever/postings-api
9
14
  *
10
15
  * @param {string} slug - Company slug (e.g., 'stripe', 'figma')
11
16
  * @returns {Promise<Array>} Normalized job objects
@@ -22,30 +27,72 @@ export async function fetchLever(slug) {
22
27
  const jobs = await resp.json();
23
28
  if (!Array.isArray(jobs)) return [];
24
29
 
25
- return jobs.map(job => {
26
- const salary = parseLeverSalary(job.categories?.commitment, job.text);
30
+ return jobs.map(job => normalize({
31
+ companySlug: slug,
32
+ // Lever's API doesn't return the company name at the board or job level,
33
+ // so the slug is the honest fallback. `categories.team` is the team within
34
+ // the company ("Payments Platform"), not the company itself.
35
+ company: titleCaseSlug(slug),
36
+ title: job.text || '',
37
+ department: job.categories?.department || job.categories?.team || '',
38
+ location: job.categories?.location || '',
39
+ locations: job.categories?.allLocations || [],
40
+ workplace: WORKPLACE_TYPES.has(job.workplaceType) ? job.workplaceType : null,
41
+ description: buildDescription(job),
42
+ url: job.hostedUrl || '',
43
+ postedAt: job.createdAt ? new Date(job.createdAt).toISOString() : null,
44
+ salary: parseLeverSalary(job.salaryRange, job.text),
45
+ metadata: {
46
+ leverId: job.id,
47
+ team: job.categories?.team || '',
48
+ commitment: job.categories?.commitment || '', // Full-time, Part-time, etc.
49
+ workplaceType: job.workplaceType || '',
50
+ salaryDescription: job.salaryDescriptionPlain || '',
51
+ },
52
+ }, 'lever'));
53
+ }
54
+
55
+ /**
56
+ * Lever splits a posting across `description` (company intro plus overview),
57
+ * `lists` (one `{text, content}` per section: responsibilities, requirements,
58
+ * location details) and `additional` (benefits, EEO). Only the first used to
59
+ * reach the description, so requirements were invisible to filters and to
60
+ * the assistant (issue #64). Reassemble the whole posting as HTML and let
61
+ * normalize() render the headings and bullets.
62
+ */
63
+ function buildDescription(job) {
64
+ const parts = [job.description || job.descriptionPlain || ''];
65
+ for (const list of job.lists || []) {
66
+ const heading = (list.text || '').trim();
67
+ parts.push((heading ? `<h3>${escapeHtml(heading)}</h3>` : '') + (list.content || ''));
68
+ }
69
+ parts.push(job.additional || job.additionalPlain || '');
70
+ return parts.filter(Boolean).join('\n');
71
+ }
72
+
73
+ // `lists[].text` is plain text ("Skills & Experience"). Escaped, normalize()
74
+ // decodes it back; raw, a stray `<` would be stripped as a tag.
75
+ function escapeHtml(s) {
76
+ return s.replace(/&/g, '&amp;').replace(/</g, '&lt;').replace(/>/g, '&gt;');
77
+ }
27
78
 
28
- return normalize({
29
- companySlug: slug,
30
- // Lever's API doesn't return the company name at the board or job level,
31
- // so the slug is the honest fallback. `categories.team` is the team within
32
- // the company ("Payments Platform"), not the company itself.
33
- company: titleCaseSlug(slug),
34
- title: job.text || '',
35
- department: job.categories?.department || job.categories?.team || '',
36
- location: job.categories?.location || '',
37
- description: stripHtml(job.descriptionPlain || job.description || ''),
38
- url: job.hostedUrl || '',
39
- postedAt: job.createdAt ? new Date(job.createdAt).toISOString() : null,
40
- salary,
41
- metadata: {
42
- leverId: job.id,
43
- team: job.categories?.team || '',
44
- commitment: job.categories?.commitment || '', // Full-time, Part-time, etc.
45
- workplaceType: job.workplaceType || '',
46
- },
47
- }, 'lever');
48
- });
79
+ /**
80
+ * Lever publishes `salaryRange: {min, max, currency, interval}` on boards
81
+ * that state pay. Boards that don't sometimes put the range in the title.
82
+ */
83
+ function parseLeverSalary(range, title) {
84
+ const min = range?.min || null;
85
+ const max = range?.max || null;
86
+ if (min || max) {
87
+ return {
88
+ min,
89
+ max,
90
+ currency: range.currency || 'USD',
91
+ period: PERIODS[range.interval] || null,
92
+ source: 'ats',
93
+ };
94
+ }
95
+ return extractSalaryFromText(title || '');
49
96
  }
50
97
 
51
98
  function titleCaseSlug(slug) {
@@ -55,19 +102,6 @@ function titleCaseSlug(slug) {
55
102
  return slug.charAt(0).toUpperCase() + slug.slice(1);
56
103
  }
57
104
 
58
- function parseLeverSalary(commitment, title) {
59
- // Lever doesn't have a salary field, but sometimes it's in the title
60
- const match = (title || '').match(/\$[\d,]+\s*[-–]\s*\$[\d,]+/);
61
- if (!match) return null;
62
- const nums = match[0].match(/[\d,]+/g);
63
- if (!nums || nums.length < 2) return null;
64
- return {
65
- min: parseInt(nums[0].replace(/,/g, '')),
66
- max: parseInt(nums[1].replace(/,/g, '')),
67
- currency: 'USD',
68
- };
69
- }
70
-
71
105
  export async function hasLever(slug) {
72
106
  try {
73
107
  const resp = await fetch(`${BASE_URL}/${slug}?mode=json`, { method: 'HEAD' });
@@ -1,4 +1,4 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
 
4
4
  /**
@@ -6,9 +6,16 @@ import { atsErrorFromStatus } from '../errors.js';
6
6
  * Public API, no auth required.
7
7
  * Docs: https://docs.recruitee.com/reference/offers
8
8
  *
9
- * Single GET returns every offer with the full HTML description
10
- * inline — no N+1 (unlike SmartRecruiters), no XML (unlike
11
- * TeamTailor/Personio). The simplest adapter shape in the toolkit.
9
+ * Single GET returns every offer inline — no N+1 (unlike SmartRecruiters),
10
+ * no XML (unlike TeamTailor/Personio). The simplest adapter shape in the
11
+ * toolkit.
12
+ *
13
+ * Each offer carries two HTML fields, `description` and `requirements`.
14
+ * Which one holds the role depends on the tenant's template (and sometimes
15
+ * the posting): some keep the duties in `description` and the candidate
16
+ * profile in `requirements`, others put a company intro in `description`
17
+ * and everything else in `requirements`. Neither alone is the posting, so
18
+ * both are joined before normalize() strips them (issue #65).
12
19
  *
13
20
  * @param {string} slug - Recruitee company subdomain (e.g., 'vandebron')
14
21
  * @returns {Promise<Array>} Normalized job objects
@@ -30,13 +37,7 @@ export async function fetchRecruitee(slug) {
30
37
  let location = place;
31
38
  if (offer.remote) location = place ? `Remote - ${place}` : 'Remote';
32
39
 
33
- let postedAt = null;
34
- if (offer.created_at) {
35
- // Recruitee returns "2026-05-13 07:38:11 UTC"; coerce to ISO.
36
- const iso = offer.created_at.replace(' UTC', 'Z').replace(' ', 'T');
37
- const d = new Date(iso);
38
- if (!Number.isNaN(d.getTime())) postedAt = d.toISOString();
39
- }
40
+ const createdAt = toIso(offer.created_at);
40
41
 
41
42
  return normalize({
42
43
  companySlug: slug,
@@ -44,19 +45,71 @@ export async function fetchRecruitee(slug) {
44
45
  title: offer.title || '',
45
46
  department: offer.department || '',
46
47
  location,
47
- description: stripHtml(offer.description || ''),
48
+ locations: (offer.locations || []).map(l => [l.city, l.country].filter(Boolean).join(', ')),
49
+ workplace: parseRecruiteeWorkplace(offer),
50
+ description: [offer.description, offer.requirements].filter(Boolean).join('\n'),
48
51
  url: offer.careers_url || offer.careers_apply_url || '',
49
- postedAt,
50
- salary: null, // No structured salary; normalizer parses from text
52
+ // created_at can predate publication by years on long-lived offers,
53
+ // so it is not a posting date. published_at is.
54
+ postedAt: toIso(offer.published_at) || createdAt,
55
+ salary: parseRecruiteeSalary(offer.salary),
51
56
  metadata: {
52
57
  recruiteeId: offer.guid || offer.id,
53
58
  employmentType: offer.employment_type_code || '',
54
59
  category: offer.category_code || '',
60
+ createdAt,
55
61
  },
56
62
  }, 'recruitee');
57
63
  });
58
64
  }
59
65
 
66
+ /**
67
+ * Recruitee returns "2026-05-13 07:38:11 UTC"; coerce to ISO.
68
+ */
69
+ function toIso(ts) {
70
+ if (!ts) return null;
71
+ const d = new Date(ts.replace(' UTC', 'Z').replace(' ', 'T'));
72
+ return Number.isNaN(d.getTime()) ? null : d.toISOString();
73
+ }
74
+
75
+ /**
76
+ * Recruitee sends three booleans, not one enum. Hybrid wins when remote is
77
+ * also set, and on_site alone is onsite. All false is no signal.
78
+ */
79
+ function parseRecruiteeWorkplace(offer) {
80
+ if (offer.hybrid) return 'hybrid';
81
+ if (offer.remote) return 'remote';
82
+ if (offer.on_site) return 'onsite';
83
+ return null;
84
+ }
85
+
86
+ const PERIODS = new Set(['year', 'month', 'hour']);
87
+
88
+ /**
89
+ * Recruitee sends `salary` as `{min, max, period, currency}` with string
90
+ * amounts. Offers without pay still carry the object, either all-null or
91
+ * as a "0"/"0" placeholder, so anything without a positive side returns
92
+ * null and normalize() falls back to the posting text.
93
+ */
94
+ function parseRecruiteeSalary(salary) {
95
+ if (!salary) return null;
96
+ const min = toAmount(salary.min);
97
+ const max = toAmount(salary.max);
98
+ if (min === null && max === null) return null;
99
+ return {
100
+ min,
101
+ max,
102
+ currency: (salary.currency || '').toUpperCase(),
103
+ period: PERIODS.has(salary.period) ? salary.period : null,
104
+ source: 'ats',
105
+ };
106
+ }
107
+
108
+ function toAmount(value) {
109
+ const n = parseFloat(value);
110
+ return Number.isFinite(n) && n > 0 ? n : null;
111
+ }
112
+
60
113
  /**
61
114
  * Check if a company has a Recruitee career site.
62
115
  */
@@ -1,4 +1,4 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
 
4
4
  const BASE_URL = 'https://api.smartrecruiters.com/v1/companies';
@@ -10,7 +10,8 @@ const PAGE_SIZE = 100;
10
10
  * Docs: https://developers.smartrecruiters.com/reference/postingsget-1
11
11
  *
12
12
  * Two-step flow (unavoidable N+1):
13
- * - The postings LIST endpoint omits the job description entirely.
13
+ * - The postings LIST endpoint omits the job description entirely,
14
+ * and the structured `compensation` block with it.
14
15
  * - jd-intel's contract is "full JD text", so we must fetch each
15
16
  * posting's DETAIL endpoint to get jobAd.sections.
16
17
  * Large enterprise tenants with hundreds of openings will therefore be
@@ -46,6 +47,7 @@ export async function fetchSmartrecruiters(slug) {
46
47
  const jobs = await Promise.all(postings.map(async (p) => {
47
48
  let sections = {};
48
49
  let postingUrl = '';
50
+ let salary = null;
49
51
 
50
52
  try {
51
53
  const detailResp = await fetch(`${BASE_URL}/${slug}/postings/${p.id}`);
@@ -53,6 +55,7 @@ export async function fetchSmartrecruiters(slug) {
53
55
  const detail = await detailResp.json();
54
56
  sections = detail.jobAd?.sections || {};
55
57
  postingUrl = detail.postingUrl || detail.applyUrl || '';
58
+ salary = parseCompensation(detail.compensation);
56
59
  }
57
60
  } catch {
58
61
  // Detail fetch failed: fall back to list-only fields (no description).
@@ -68,8 +71,14 @@ export async function fetchSmartrecruiters(slug) {
68
71
  const place = loc.fullLocation
69
72
  || [loc.city, loc.region, loc.country].filter(Boolean).join(', ');
70
73
  let location = place;
71
- if (loc.remote) location = `Remote - ${place}`.replace(/ - $/, ' ');
72
- else if (loc.hybrid) location = `Hybrid - ${place}`.replace(/ - $/, ' ');
74
+ let workplace = null;
75
+ if (loc.remote) {
76
+ location = `Remote - ${place}`.replace(/ - $/, ' ');
77
+ workplace = 'remote';
78
+ } else if (loc.hybrid) {
79
+ location = `Hybrid - ${place}`.replace(/ - $/, ' ');
80
+ workplace = 'hybrid';
81
+ }
73
82
 
74
83
  return normalize({
75
84
  companySlug: slug,
@@ -77,10 +86,11 @@ export async function fetchSmartrecruiters(slug) {
77
86
  title: p.name || '',
78
87
  department: p.department?.label || p.function?.label || '',
79
88
  location,
80
- description: stripHtml(description),
89
+ workplace,
90
+ description,
81
91
  url: postingUrl,
82
92
  postedAt: p.releasedDate || null,
83
- salary: null, // SmartRecruiters has no structured salary; normalizer parses text
93
+ salary, // null when the detail has no compensation; normalize() then parses text
84
94
  metadata: {
85
95
  smartRecruitersId: p.id,
86
96
  refNumber: p.refNumber || '',
@@ -94,6 +104,31 @@ export async function fetchSmartrecruiters(slug) {
94
104
  return jobs;
95
105
  }
96
106
 
107
+ const PERIODS = { YEARLY: 'year', MONTHLY: 'month', HOURLY: 'hour' };
108
+
109
+ /**
110
+ * Map the detail response's `compensation` to the shared salary shape.
111
+ *
112
+ * SmartRecruiters publishes `{min?, max?, currency, period}`, and both
113
+ * one-sided cases occur (a "max only" cap, a "from" floor), so each bound
114
+ * is passed through as null when absent rather than dropping the whole
115
+ * range. The period is kept as published: a MONTHLY figure is not
116
+ * annualized because tenants occasionally mislabel it (issue #70).
117
+ */
118
+ function parseCompensation(comp) {
119
+ if (!comp || !comp.currency) return null;
120
+ const min = Number.isFinite(comp.min) ? comp.min : null;
121
+ const max = Number.isFinite(comp.max) ? comp.max : null;
122
+ if (min === null && max === null) return null;
123
+ return {
124
+ min,
125
+ max,
126
+ currency: comp.currency,
127
+ period: PERIODS[comp.period] ?? null,
128
+ source: 'ats',
129
+ };
130
+ }
131
+
97
132
  /**
98
133
  * Check if a company exists on SmartRecruiters.
99
134
  * (HEAD isn't reliably supported on the postings endpoint, so use a
@@ -1,4 +1,4 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize, decodeEntities } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
 
4
4
  /**
@@ -16,10 +16,11 @@ import { atsErrorFromStatus } from '../errors.js';
16
16
  * to a custom domain (e.g. jobs.tibber.com).
17
17
  *
18
18
  * RSS quirk: descriptions are HTML-entity-encoded inside the XML
19
- * (`&lt;p&gt;...`). We decode that outer layer to real HTML, then
20
- * hand it to stripHtml() which strips tags and resolves the inner
21
- * entities. Decode order matters — `&amp;` resolves LAST so that
22
- * double-encoded sequences (`&amp;amp;`) collapse correctly.
19
+ * (`&lt;p&gt;...`). We decode that outer layer to real HTML with the
20
+ * shared decodeEntities() and hand the HTML to normalize(), which
21
+ * strips tags and resolves the inner entities. Decode order matters —
22
+ * `&amp;` resolves LAST so double-encoded sequences (`&amp;amp;`)
23
+ * collapse by one layer per pass.
23
24
  *
24
25
  * @param {string} slug - TeamTailor career-site slug (e.g., 'tibber')
25
26
  * @returns {Promise<Array>} Normalized job objects
@@ -28,6 +29,9 @@ import { atsErrorFromStatus } from '../errors.js';
28
29
  // segment, e.g. crunchbase.na.teamtailor.com. '' is the base host.
29
30
  const TT_REGIONS = ['', 'na', 'eu'];
30
31
 
32
+ // Feeds send `none`, `hybrid`, `fully` or `onsite`. `none` is no signal.
33
+ const REMOTE_STATUS = { hybrid: 'hybrid', fully: 'remote', onsite: 'onsite' };
34
+
31
35
  /**
32
36
  * Resolve which TeamTailor host actually serves this slug's feed.
33
37
  * Returns the first 200 Response, throws on a non-404 error, or
@@ -64,8 +68,8 @@ export async function fetchTeamtailor(slug) {
64
68
  const items = [...xml.matchAll(/<item>([\s\S]*?)<\/item>/g)].map(m => m[1]);
65
69
 
66
70
  return items.map(item => {
67
- const pick = (tag) => {
68
- const m = item.match(new RegExp(`<${tag}[^>]*>([\\s\\S]*?)</${tag}>`));
71
+ const pick = (tag, src = item) => {
72
+ const m = src.match(new RegExp(`<${tag}[^>]*>([\\s\\S]*?)</${tag}>`));
69
73
  return m ? m[1].trim() : '';
70
74
  };
71
75
 
@@ -78,6 +82,12 @@ export async function fetchTeamtailor(slug) {
78
82
  const country = decodeEntities(pick('tt:country'));
79
83
  const remoteStatus = decodeEntities(pick('remoteStatus'));
80
84
 
85
+ // One <tt:location> per office the posting is open in, read the same
86
+ // way as the primary above so the entries line up.
87
+ const locations = [...item.matchAll(/<tt:location>([\s\S]*?)<\/tt:location>/g)].map(m =>
88
+ [decodeEntities(pick('tt:city', m[1])), decodeEntities(pick('tt:country', m[1]))].filter(Boolean).join(', ')
89
+ );
90
+
81
91
  let location = [city, country].filter(Boolean).join(', ');
82
92
  if (/remote/i.test(remoteStatus)) {
83
93
  location = location ? `Remote - ${location}` : 'Remote';
@@ -95,7 +105,9 @@ export async function fetchTeamtailor(slug) {
95
105
  title,
96
106
  department,
97
107
  location,
98
- description: stripHtml(decodeEntities(pick('description'))),
108
+ locations,
109
+ workplace: REMOTE_STATUS[remoteStatus.toLowerCase()] || null,
110
+ description: decodeEntities(pick('description')),
99
111
  url: link,
100
112
  postedAt,
101
113
  salary: null, // No structured salary; normalizer parses from text
@@ -107,22 +119,6 @@ export async function fetchTeamtailor(slug) {
107
119
  });
108
120
  }
109
121
 
110
- /**
111
- * Decode the RSS entity/CDATA layer to real HTML.
112
- * `&amp;` is intentionally resolved LAST.
113
- */
114
- function decodeEntities(s) {
115
- if (!s) return '';
116
- return s
117
- .replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1')
118
- .replace(/&lt;/g, '<')
119
- .replace(/&gt;/g, '>')
120
- .replace(/&quot;/g, '"')
121
- .replace(/&#39;/g, "'")
122
- .replace(/&apos;/g, "'")
123
- .replace(/&amp;/g, '&');
124
- }
125
-
126
122
  /**
127
123
  * Check if a company has a TeamTailor career site.
128
124
  */