jd-intel 0.8.3 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -131,6 +131,19 @@ const jobs = await fetchJobs({
131
131
  });
132
132
  ```
133
133
 
134
+ Results come back newest first by `postedAt`, undated last (`order: 'board'` keeps the ATS's own order). `fetchJobs` returns the page as an array. `fetchJobsDetailed` returns the same page plus `total_matched`, the number of matches before `offset` and `limit`, so you can tell a small board from a cut and page through the rest:
135
+
136
+ ```js
137
+ import { fetchJobsDetailed } from 'jd-intel';
138
+
139
+ const { jobs, total_matched } = await fetchJobsDetailed({
140
+ company: '<your-target-company>',
141
+ titleFilter: 'engineer',
142
+ limit: 20,
143
+ offset: 20, // second page
144
+ });
145
+ ```
146
+
134
147
  CLI usage: `npx jd-intel fetch <company-slug> --title-filter "engineer" --posted-within-days 14`. Full filter reference [below](#filters-quick-reference).
135
148
 
136
149
  Node.js 18+. No API keys. No configuration.
@@ -189,8 +202,10 @@ Every job normalizes to one schema, across every platform:
189
202
  "title": "Senior Software Engineer, Platform",
190
203
  "department": "Engineering",
191
204
  "location": "Remote - US",
205
+ "locations": ["Remote - US", "Toronto, Canada"],
192
206
  "locationType": "remote",
193
- "salary": { "min": 180000, "max": 240000, "currency": "USD" },
207
+ "workplace": { "type": "remote", "source": "ats" },
208
+ "salary": { "min": 180000, "max": 240000, "currency": "USD", "period": "year", "source": "text" },
194
209
  "description": "Design and build the API surface our customers integrate against...",
195
210
  "url": "https://boards.example.com/jobs/12345",
196
211
  "postedAt": "2026-04-10T14:30:00Z"
@@ -206,9 +221,11 @@ No custom parsing per company.
206
221
  | `title` | Full job title |
207
222
  | `company` | Normalized company name |
208
223
  | `department` | Team or department (when provided) |
209
- | `location` | City, state, country, or remote |
210
- | `locationType` | `remote`, `hybrid`, or `onsite` |
211
- | `salary` | Min-max range with currency (when available) |
224
+ | `location` | Primary location: city, state, country, or remote |
225
+ | `locations` | Every location the posting is open in, primary first. The location filters check each entry |
226
+ | `locationType` | `remote`, `hybrid`, `onsite`, or `unknown` when neither the platform nor the location text says |
227
+ | `workplace` | `{ type, source }`. `type` repeats `locationType`; `source` is `ats` when the platform stated it, `text` when read from the location string, null when unknown |
228
+ | `salary` | Min-max range with `currency`, plus `period` (`year`, `month`, `hour`, or null) and `source` (`ats` when the platform supplied it, `text` when parsed from the posting). Null when nothing is stated |
212
229
  | `description` | Full JD in clean markdown |
213
230
  | `url` | Direct link to the posting |
214
231
  | `postedAt` | Publication date (when provided) |
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "jd-intel",
3
- "version": "0.8.3",
3
+ "version": "0.9.0",
4
4
  "description": "Fetch and normalize job descriptions across seven major ATS (Greenhouse, Lever, Ashby, Workday, and more), for your AI assistant. No copy-paste.",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
@@ -1,4 +1,4 @@
1
- import { normalize } from '../normalizer.js';
1
+ import { normalize, extractSalaryFromText } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
 
4
4
  const API_URL = 'https://jobs.ashbyhq.com/api/non-user-graphql';
@@ -36,23 +36,33 @@ async function fetchAshbyRest(slug) {
36
36
  const jobs = data.jobs || [];
37
37
 
38
38
  return jobs.map(job => {
39
- const salary = parseAshbyCompensation(job.compensation);
39
+ const comp = job.compensation || {};
40
40
 
41
41
  return normalize({
42
42
  companySlug: slug,
43
43
  company: data.organizationName || slug,
44
44
  title: job.title || '',
45
- department: job.departmentName || '',
45
+ department: job.department || '',
46
46
  location: job.location || '',
47
+ locations: (job.secondaryLocations || []).map(l => l?.location || ''),
48
+ workplace: parseAshbyWorkplace(job),
47
49
  description: job.descriptionHtml || job.descriptionPlain || '',
48
50
  url: `https://jobs.ashbyhq.com/${slug}/${job.id}`,
49
51
  postedAt: job.publishedAt || null,
50
- salary,
52
+ salary: parseAshbyCompensation(comp),
51
53
  metadata: {
52
54
  ashbyId: job.id,
53
55
  employmentType: job.employmentType || '',
54
56
  isRemote: job.isRemote || false,
55
- team: job.teamName || '',
57
+ team: job.team || '',
58
+ // The rendered summaries keep what min/max drop: "Offers Equity",
59
+ // "Multiple Ranges", and per-location tiers labelled OTE.
60
+ compensationSummary: comp.compensationTierSummary || '',
61
+ compensationTiers: (comp.compensationTiers || []).map(tier => ({
62
+ title: tier.title || '',
63
+ summary: tier.tierSummary || '',
64
+ additionalInformation: tier.additionalInformation || '',
65
+ })),
56
66
  },
57
67
  }, 'ashby');
58
68
  });
@@ -110,22 +120,47 @@ async function fetchAshbyGraphQL(slug) {
110
120
  }, 'ashby'));
111
121
  }
112
122
 
123
+ const WORKPLACE_TYPES = { remote: 'remote', hybrid: 'hybrid', onsite: 'onsite' };
124
+
125
+ /**
126
+ * `workplaceType` is 'Remote', 'Hybrid' or 'OnSite'. `isRemote` is the
127
+ * older flag and can only say remote, so it is the fallback when the type
128
+ * is absent. false means nothing: the role may be hybrid or onsite.
129
+ */
130
+ function parseAshbyWorkplace(job) {
131
+ const type = WORKPLACE_TYPES[String(job.workplaceType || '').toLowerCase()];
132
+ if (type) return type;
133
+ return job.isRemote === true ? 'remote' : null;
134
+ }
135
+
136
+ const INTERVAL_PERIOD = { '1 YEAR': 'year', '1 MONTH': 'month', '1 HOUR': 'hour' };
137
+
138
+ /**
139
+ * Read pay from Ashby's `compensation` object (issue #67).
140
+ *
141
+ * `summaryComponents` carries one structured entry per component type
142
+ * (Salary, Bonus, Commission, Equity); the Salary entry spans every tier.
143
+ * `scrapeableCompensationSalarySummary` and `compensationTierSummary` are
144
+ * the rendered strings. A board that publishes no pay still sends the
145
+ * object, with null summaries and empty arrays, so a miss here has to
146
+ * return null for the normalizer's text fallback to run.
147
+ */
113
148
  function parseAshbyCompensation(comp) {
114
- if (!comp) return null;
115
- // Ashby compensation can be a string or structured object
116
- if (typeof comp === 'string') {
117
- const match = comp.match(/\$?([\d,]+)\s*[-–]\s*\$?([\d,]+)/);
118
- if (!match) return null;
149
+ const salary = (comp.summaryComponents || []).find(c => c.compensationType === 'Salary');
150
+ if (salary && (salary.minValue != null || salary.maxValue != null)) {
119
151
  return {
120
- min: parseInt(match[1].replace(/,/g, '')),
121
- max: parseInt(match[2].replace(/,/g, '')),
122
- currency: 'USD',
152
+ min: salary.minValue ?? null,
153
+ max: salary.maxValue ?? null,
154
+ currency: salary.currencyCode || 'USD',
155
+ period: INTERVAL_PERIOD[salary.interval] || null,
156
+ source: 'ats',
123
157
  };
124
158
  }
125
- if (comp.min && comp.max) {
126
- return { min: comp.min, max: comp.max, currency: comp.currency || 'USD' };
127
- }
128
- return null;
159
+ // The summaries are still the ATS's own compensation field, so a range
160
+ // read out of one counts as source 'ats'.
161
+ const parsed = extractSalaryFromText(comp.scrapeableCompensationSalarySummary)
162
+ || extractSalaryFromText(comp.compensationTierSummary);
163
+ return parsed ? { ...parsed, source: 'ats' } : null;
129
164
  }
130
165
 
131
166
  export async function hasAshby(slug) {
@@ -1,4 +1,4 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize, decodeEntities } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
 
4
4
  const BASE_URL = 'https://boards-api.greenhouse.io/v1/boards';
@@ -29,19 +29,42 @@ export async function fetchGreenhouse(slug) {
29
29
  title: job.title || '',
30
30
  department: job.departments?.[0]?.name || '',
31
31
  location: job.location?.name || '',
32
- description: stripHtml(job.content || ''),
32
+ workplace: parseGreenhouseWorkplace(job.metadata),
33
+ // `content` arrives HTML-escaped (`&lt;p&gt;`). Decode that outer layer
34
+ // once so normalize() sees real tags; it strips and decodes the rest.
35
+ description: decodeEntities(job.content || ''),
33
36
  url: job.absolute_url || '',
34
- postedAt: job.updated_at || null,
35
- salary: null, // Greenhouse doesn't expose salary in public API
37
+ // updated_at is an edit time that many boards bulk-refresh, so it is not
38
+ // a posting date. first_published is. Fallback covers boards without it (#69).
39
+ postedAt: job.first_published || job.updated_at || null,
40
+ salary: null, // list endpoint has no structured pay; normalizer parses the pay transparency text
36
41
  metadata: {
37
42
  greenhouseId: job.id,
38
43
  internal_job_id: job.internal_job_id,
39
44
  departments: job.departments?.map(d => d.name) || [],
40
45
  offices: job.offices?.map(o => o.name) || [],
46
+ updatedAt: job.updated_at,
41
47
  },
42
48
  }, 'greenhouse'));
43
49
  }
44
50
 
51
+ /**
52
+ * Greenhouse has no native workplace field. Boards that track it define a
53
+ * custom field ("Location Type", "Workplace Type") that arrives in the
54
+ * job's `metadata[]`, with `value` a string for single-select fields and
55
+ * an array for multi-select. Values seen: On-Site, Hybrid (Travel-Required),
56
+ * Remote. Anything else is no signal and the location string decides.
57
+ */
58
+ function parseGreenhouseWorkplace(metadata) {
59
+ const field = (metadata || []).find(m => /location type|workplace type/i.test(m?.name || ''));
60
+ if (!field) return null;
61
+ const value = [].concat(field.value ?? []).join(' ').toLowerCase();
62
+ if (/remote/.test(value)) return 'remote';
63
+ if (/hybrid/.test(value)) return 'hybrid';
64
+ if (/on-?site/.test(value)) return 'onsite';
65
+ return null;
66
+ }
67
+
45
68
  /**
46
69
  * Check if a company has a Greenhouse board.
47
70
  */
@@ -1,11 +1,16 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize, extractSalaryFromText } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
 
4
4
  const BASE_URL = 'https://api.lever.co/v0/postings';
5
5
 
6
+ const PERIODS = { 'per-year-salary': 'year', 'per-month-salary': 'month', 'per-hour-wage': 'hour' };
7
+ // Lever's workplaceType is one of these or 'unspecified'.
8
+ const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
9
+
6
10
  /**
7
11
  * Fetch all jobs from a Lever job board.
8
12
  * Public API, no auth required.
13
+ * Docs: https://github.com/lever/postings-api
9
14
  *
10
15
  * @param {string} slug - Company slug (e.g., 'stripe', 'figma')
11
16
  * @returns {Promise<Array>} Normalized job objects
@@ -22,30 +27,72 @@ export async function fetchLever(slug) {
22
27
  const jobs = await resp.json();
23
28
  if (!Array.isArray(jobs)) return [];
24
29
 
25
- return jobs.map(job => {
26
- const salary = parseLeverSalary(job.categories?.commitment, job.text);
30
+ return jobs.map(job => normalize({
31
+ companySlug: slug,
32
+ // Lever's API doesn't return the company name at the board or job level,
33
+ // so the slug is the honest fallback. `categories.team` is the team within
34
+ // the company ("Payments Platform"), not the company itself.
35
+ company: titleCaseSlug(slug),
36
+ title: job.text || '',
37
+ department: job.categories?.department || job.categories?.team || '',
38
+ location: job.categories?.location || '',
39
+ locations: job.categories?.allLocations || [],
40
+ workplace: WORKPLACE_TYPES.has(job.workplaceType) ? job.workplaceType : null,
41
+ description: buildDescription(job),
42
+ url: job.hostedUrl || '',
43
+ postedAt: job.createdAt ? new Date(job.createdAt).toISOString() : null,
44
+ salary: parseLeverSalary(job.salaryRange, job.text),
45
+ metadata: {
46
+ leverId: job.id,
47
+ team: job.categories?.team || '',
48
+ commitment: job.categories?.commitment || '', // Full-time, Part-time, etc.
49
+ workplaceType: job.workplaceType || '',
50
+ salaryDescription: job.salaryDescriptionPlain || '',
51
+ },
52
+ }, 'lever'));
53
+ }
54
+
55
+ /**
56
+ * Lever splits a posting across `description` (company intro plus overview),
57
+ * `lists` (one `{text, content}` per section: responsibilities, requirements,
58
+ * location details) and `additional` (benefits, EEO). Only the first used to
59
+ * reach the description, so requirements were invisible to filters and to
60
+ * the assistant (issue #64). Reassemble the whole posting as HTML and let
61
+ * normalize() render the headings and bullets.
62
+ */
63
+ function buildDescription(job) {
64
+ const parts = [job.description || job.descriptionPlain || ''];
65
+ for (const list of job.lists || []) {
66
+ const heading = (list.text || '').trim();
67
+ parts.push((heading ? `<h3>${escapeHtml(heading)}</h3>` : '') + (list.content || ''));
68
+ }
69
+ parts.push(job.additional || job.additionalPlain || '');
70
+ return parts.filter(Boolean).join('\n');
71
+ }
72
+
73
+ // `lists[].text` is plain text ("Skills & Experience"). Escaped, normalize()
74
+ // decodes it back; raw, a stray `<` would be stripped as a tag.
75
+ function escapeHtml(s) {
76
+ return s.replace(/&/g, '&amp;').replace(/</g, '&lt;').replace(/>/g, '&gt;');
77
+ }
27
78
 
28
- return normalize({
29
- companySlug: slug,
30
- // Lever's API doesn't return the company name at the board or job level,
31
- // so the slug is the honest fallback. `categories.team` is the team within
32
- // the company ("Payments Platform"), not the company itself.
33
- company: titleCaseSlug(slug),
34
- title: job.text || '',
35
- department: job.categories?.department || job.categories?.team || '',
36
- location: job.categories?.location || '',
37
- description: stripHtml(job.descriptionPlain || job.description || ''),
38
- url: job.hostedUrl || '',
39
- postedAt: job.createdAt ? new Date(job.createdAt).toISOString() : null,
40
- salary,
41
- metadata: {
42
- leverId: job.id,
43
- team: job.categories?.team || '',
44
- commitment: job.categories?.commitment || '', // Full-time, Part-time, etc.
45
- workplaceType: job.workplaceType || '',
46
- },
47
- }, 'lever');
48
- });
79
+ /**
80
+ * Lever publishes `salaryRange: {min, max, currency, interval}` on boards
81
+ * that state pay. Boards that don't sometimes put the range in the title.
82
+ */
83
+ function parseLeverSalary(range, title) {
84
+ const min = range?.min || null;
85
+ const max = range?.max || null;
86
+ if (min || max) {
87
+ return {
88
+ min,
89
+ max,
90
+ currency: range.currency || 'USD',
91
+ period: PERIODS[range.interval] || null,
92
+ source: 'ats',
93
+ };
94
+ }
95
+ return extractSalaryFromText(title || '');
49
96
  }
50
97
 
51
98
  function titleCaseSlug(slug) {
@@ -55,19 +102,6 @@ function titleCaseSlug(slug) {
55
102
  return slug.charAt(0).toUpperCase() + slug.slice(1);
56
103
  }
57
104
 
58
- function parseLeverSalary(commitment, title) {
59
- // Lever doesn't have a salary field, but sometimes it's in the title
60
- const match = (title || '').match(/\$[\d,]+\s*[-–]\s*\$[\d,]+/);
61
- if (!match) return null;
62
- const nums = match[0].match(/[\d,]+/g);
63
- if (!nums || nums.length < 2) return null;
64
- return {
65
- min: parseInt(nums[0].replace(/,/g, '')),
66
- max: parseInt(nums[1].replace(/,/g, '')),
67
- currency: 'USD',
68
- };
69
- }
70
-
71
105
  export async function hasLever(slug) {
72
106
  try {
73
107
  const resp = await fetch(`${BASE_URL}/${slug}?mode=json`, { method: 'HEAD' });
@@ -1,4 +1,4 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
 
4
4
  /**
@@ -6,9 +6,16 @@ import { atsErrorFromStatus } from '../errors.js';
6
6
  * Public API, no auth required.
7
7
  * Docs: https://docs.recruitee.com/reference/offers
8
8
  *
9
- * Single GET returns every offer with the full HTML description
10
- * inline — no N+1 (unlike SmartRecruiters), no XML (unlike
11
- * TeamTailor/Personio). The simplest adapter shape in the toolkit.
9
+ * Single GET returns every offer inline — no N+1 (unlike SmartRecruiters),
10
+ * no XML (unlike TeamTailor/Personio). The simplest adapter shape in the
11
+ * toolkit.
12
+ *
13
+ * Each offer carries two HTML fields, `description` and `requirements`.
14
+ * Which one holds the role depends on the tenant's template (and sometimes
15
+ * the posting): some keep the duties in `description` and the candidate
16
+ * profile in `requirements`, others put a company intro in `description`
17
+ * and everything else in `requirements`. Neither alone is the posting, so
18
+ * both are joined before normalize() strips them (issue #65).
12
19
  *
13
20
  * @param {string} slug - Recruitee company subdomain (e.g., 'vandebron')
14
21
  * @returns {Promise<Array>} Normalized job objects
@@ -30,13 +37,7 @@ export async function fetchRecruitee(slug) {
30
37
  let location = place;
31
38
  if (offer.remote) location = place ? `Remote - ${place}` : 'Remote';
32
39
 
33
- let postedAt = null;
34
- if (offer.created_at) {
35
- // Recruitee returns "2026-05-13 07:38:11 UTC"; coerce to ISO.
36
- const iso = offer.created_at.replace(' UTC', 'Z').replace(' ', 'T');
37
- const d = new Date(iso);
38
- if (!Number.isNaN(d.getTime())) postedAt = d.toISOString();
39
- }
40
+ const createdAt = toIso(offer.created_at);
40
41
 
41
42
  return normalize({
42
43
  companySlug: slug,
@@ -44,19 +45,71 @@ export async function fetchRecruitee(slug) {
44
45
  title: offer.title || '',
45
46
  department: offer.department || '',
46
47
  location,
47
- description: stripHtml(offer.description || ''),
48
+ locations: (offer.locations || []).map(l => [l.city, l.country].filter(Boolean).join(', ')),
49
+ workplace: parseRecruiteeWorkplace(offer),
50
+ description: [offer.description, offer.requirements].filter(Boolean).join('\n'),
48
51
  url: offer.careers_url || offer.careers_apply_url || '',
49
- postedAt,
50
- salary: null, // No structured salary; normalizer parses from text
52
+ // created_at can predate publication by years on long-lived offers,
53
+ // so it is not a posting date. published_at is.
54
+ postedAt: toIso(offer.published_at) || createdAt,
55
+ salary: parseRecruiteeSalary(offer.salary),
51
56
  metadata: {
52
57
  recruiteeId: offer.guid || offer.id,
53
58
  employmentType: offer.employment_type_code || '',
54
59
  category: offer.category_code || '',
60
+ createdAt,
55
61
  },
56
62
  }, 'recruitee');
57
63
  });
58
64
  }
59
65
 
66
+ /**
67
+ * Recruitee returns "2026-05-13 07:38:11 UTC"; coerce to ISO.
68
+ */
69
+ function toIso(ts) {
70
+ if (!ts) return null;
71
+ const d = new Date(ts.replace(' UTC', 'Z').replace(' ', 'T'));
72
+ return Number.isNaN(d.getTime()) ? null : d.toISOString();
73
+ }
74
+
75
+ /**
76
+ * Recruitee sends three booleans, not one enum. Hybrid wins when remote is
77
+ * also set, and on_site alone is onsite. All false is no signal.
78
+ */
79
+ function parseRecruiteeWorkplace(offer) {
80
+ if (offer.hybrid) return 'hybrid';
81
+ if (offer.remote) return 'remote';
82
+ if (offer.on_site) return 'onsite';
83
+ return null;
84
+ }
85
+
86
+ const PERIODS = new Set(['year', 'month', 'hour']);
87
+
88
+ /**
89
+ * Recruitee sends `salary` as `{min, max, period, currency}` with string
90
+ * amounts. Offers without pay still carry the object, either all-null or
91
+ * as a "0"/"0" placeholder, so anything without a positive side returns
92
+ * null and normalize() falls back to the posting text.
93
+ */
94
+ function parseRecruiteeSalary(salary) {
95
+ if (!salary) return null;
96
+ const min = toAmount(salary.min);
97
+ const max = toAmount(salary.max);
98
+ if (min === null && max === null) return null;
99
+ return {
100
+ min,
101
+ max,
102
+ currency: (salary.currency || '').toUpperCase(),
103
+ period: PERIODS.has(salary.period) ? salary.period : null,
104
+ source: 'ats',
105
+ };
106
+ }
107
+
108
+ function toAmount(value) {
109
+ const n = parseFloat(value);
110
+ return Number.isFinite(n) && n > 0 ? n : null;
111
+ }
112
+
60
113
  /**
61
114
  * Check if a company has a Recruitee career site.
62
115
  */
@@ -1,4 +1,4 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
 
4
4
  const BASE_URL = 'https://api.smartrecruiters.com/v1/companies';
@@ -10,7 +10,8 @@ const PAGE_SIZE = 100;
10
10
  * Docs: https://developers.smartrecruiters.com/reference/postingsget-1
11
11
  *
12
12
  * Two-step flow (unavoidable N+1):
13
- * - The postings LIST endpoint omits the job description entirely.
13
+ * - The postings LIST endpoint omits the job description entirely,
14
+ * and the structured `compensation` block with it.
14
15
  * - jd-intel's contract is "full JD text", so we must fetch each
15
16
  * posting's DETAIL endpoint to get jobAd.sections.
16
17
  * Large enterprise tenants with hundreds of openings will therefore be
@@ -46,6 +47,7 @@ export async function fetchSmartrecruiters(slug) {
46
47
  const jobs = await Promise.all(postings.map(async (p) => {
47
48
  let sections = {};
48
49
  let postingUrl = '';
50
+ let salary = null;
49
51
 
50
52
  try {
51
53
  const detailResp = await fetch(`${BASE_URL}/${slug}/postings/${p.id}`);
@@ -53,6 +55,7 @@ export async function fetchSmartrecruiters(slug) {
53
55
  const detail = await detailResp.json();
54
56
  sections = detail.jobAd?.sections || {};
55
57
  postingUrl = detail.postingUrl || detail.applyUrl || '';
58
+ salary = parseCompensation(detail.compensation);
56
59
  }
57
60
  } catch {
58
61
  // Detail fetch failed: fall back to list-only fields (no description).
@@ -68,8 +71,14 @@ export async function fetchSmartrecruiters(slug) {
68
71
  const place = loc.fullLocation
69
72
  || [loc.city, loc.region, loc.country].filter(Boolean).join(', ');
70
73
  let location = place;
71
- if (loc.remote) location = `Remote - ${place}`.replace(/ - $/, ' ');
72
- else if (loc.hybrid) location = `Hybrid - ${place}`.replace(/ - $/, ' ');
74
+ let workplace = null;
75
+ if (loc.remote) {
76
+ location = `Remote - ${place}`.replace(/ - $/, ' ');
77
+ workplace = 'remote';
78
+ } else if (loc.hybrid) {
79
+ location = `Hybrid - ${place}`.replace(/ - $/, ' ');
80
+ workplace = 'hybrid';
81
+ }
73
82
 
74
83
  return normalize({
75
84
  companySlug: slug,
@@ -77,10 +86,11 @@ export async function fetchSmartrecruiters(slug) {
77
86
  title: p.name || '',
78
87
  department: p.department?.label || p.function?.label || '',
79
88
  location,
80
- description: stripHtml(description),
89
+ workplace,
90
+ description,
81
91
  url: postingUrl,
82
92
  postedAt: p.releasedDate || null,
83
- salary: null, // SmartRecruiters has no structured salary; normalizer parses text
93
+ salary, // null when the detail has no compensation; normalize() then parses text
84
94
  metadata: {
85
95
  smartRecruitersId: p.id,
86
96
  refNumber: p.refNumber || '',
@@ -94,6 +104,31 @@ export async function fetchSmartrecruiters(slug) {
94
104
  return jobs;
95
105
  }
96
106
 
107
+ const PERIODS = { YEARLY: 'year', MONTHLY: 'month', HOURLY: 'hour' };
108
+
109
+ /**
110
+ * Map the detail response's `compensation` to the shared salary shape.
111
+ *
112
+ * SmartRecruiters publishes `{min?, max?, currency, period}`, and both
113
+ * one-sided cases occur (a "max only" cap, a "from" floor), so each bound
114
+ * is passed through as null when absent rather than dropping the whole
115
+ * range. The period is kept as published: a MONTHLY figure is not
116
+ * annualized because tenants occasionally mislabel it (issue #70).
117
+ */
118
+ function parseCompensation(comp) {
119
+ if (!comp || !comp.currency) return null;
120
+ const min = Number.isFinite(comp.min) ? comp.min : null;
121
+ const max = Number.isFinite(comp.max) ? comp.max : null;
122
+ if (min === null && max === null) return null;
123
+ return {
124
+ min,
125
+ max,
126
+ currency: comp.currency,
127
+ period: PERIODS[comp.period] ?? null,
128
+ source: 'ats',
129
+ };
130
+ }
131
+
97
132
  /**
98
133
  * Check if a company exists on SmartRecruiters.
99
134
  * (HEAD isn't reliably supported on the postings endpoint, so use a
@@ -1,4 +1,4 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize, decodeEntities } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
 
4
4
  /**
@@ -16,10 +16,11 @@ import { atsErrorFromStatus } from '../errors.js';
16
16
  * to a custom domain (e.g. jobs.tibber.com).
17
17
  *
18
18
  * RSS quirk: descriptions are HTML-entity-encoded inside the XML
19
- * (`&lt;p&gt;...`). We decode that outer layer to real HTML, then
20
- * hand it to stripHtml() which strips tags and resolves the inner
21
- * entities. Decode order matters — `&amp;` resolves LAST so that
22
- * double-encoded sequences (`&amp;amp;`) collapse correctly.
19
+ * (`&lt;p&gt;...`). We decode that outer layer to real HTML with the
20
+ * shared decodeEntities() and hand the HTML to normalize(), which
21
+ * strips tags and resolves the inner entities. Decode order matters —
22
+ * `&amp;` resolves LAST so double-encoded sequences (`&amp;amp;`)
23
+ * collapse by one layer per pass.
23
24
  *
24
25
  * @param {string} slug - TeamTailor career-site slug (e.g., 'tibber')
25
26
  * @returns {Promise<Array>} Normalized job objects
@@ -28,6 +29,9 @@ import { atsErrorFromStatus } from '../errors.js';
28
29
  // segment, e.g. crunchbase.na.teamtailor.com. '' is the base host.
29
30
  const TT_REGIONS = ['', 'na', 'eu'];
30
31
 
32
+ // Feeds send `none`, `hybrid`, `fully` or `onsite`. `none` is no signal.
33
+ const REMOTE_STATUS = { hybrid: 'hybrid', fully: 'remote', onsite: 'onsite' };
34
+
31
35
  /**
32
36
  * Resolve which TeamTailor host actually serves this slug's feed.
33
37
  * Returns the first 200 Response, throws on a non-404 error, or
@@ -64,8 +68,8 @@ export async function fetchTeamtailor(slug) {
64
68
  const items = [...xml.matchAll(/<item>([\s\S]*?)<\/item>/g)].map(m => m[1]);
65
69
 
66
70
  return items.map(item => {
67
- const pick = (tag) => {
68
- const m = item.match(new RegExp(`<${tag}[^>]*>([\\s\\S]*?)</${tag}>`));
71
+ const pick = (tag, src = item) => {
72
+ const m = src.match(new RegExp(`<${tag}[^>]*>([\\s\\S]*?)</${tag}>`));
69
73
  return m ? m[1].trim() : '';
70
74
  };
71
75
 
@@ -78,6 +82,12 @@ export async function fetchTeamtailor(slug) {
78
82
  const country = decodeEntities(pick('tt:country'));
79
83
  const remoteStatus = decodeEntities(pick('remoteStatus'));
80
84
 
85
+ // One <tt:location> per office the posting is open in, read the same
86
+ // way as the primary above so the entries line up.
87
+ const locations = [...item.matchAll(/<tt:location>([\s\S]*?)<\/tt:location>/g)].map(m =>
88
+ [decodeEntities(pick('tt:city', m[1])), decodeEntities(pick('tt:country', m[1]))].filter(Boolean).join(', ')
89
+ );
90
+
81
91
  let location = [city, country].filter(Boolean).join(', ');
82
92
  if (/remote/i.test(remoteStatus)) {
83
93
  location = location ? `Remote - ${location}` : 'Remote';
@@ -95,7 +105,9 @@ export async function fetchTeamtailor(slug) {
95
105
  title,
96
106
  department,
97
107
  location,
98
- description: stripHtml(decodeEntities(pick('description'))),
108
+ locations,
109
+ workplace: REMOTE_STATUS[remoteStatus.toLowerCase()] || null,
110
+ description: decodeEntities(pick('description')),
99
111
  url: link,
100
112
  postedAt,
101
113
  salary: null, // No structured salary; normalizer parses from text
@@ -107,22 +119,6 @@ export async function fetchTeamtailor(slug) {
107
119
  });
108
120
  }
109
121
 
110
- /**
111
- * Decode the RSS entity/CDATA layer to real HTML.
112
- * `&amp;` is intentionally resolved LAST.
113
- */
114
- function decodeEntities(s) {
115
- if (!s) return '';
116
- return s
117
- .replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1')
118
- .replace(/&lt;/g, '<')
119
- .replace(/&gt;/g, '>')
120
- .replace(/&quot;/g, '"')
121
- .replace(/&#39;/g, "'")
122
- .replace(/&apos;/g, "'")
123
- .replace(/&amp;/g, '&');
124
- }
125
-
126
122
  /**
127
123
  * Check if a company has a TeamTailor career site.
128
124
  */
@@ -1,9 +1,14 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
 
4
4
  const MAX_DETAIL_FETCHES = 100;
5
5
  const LIST_PAGE_SIZE = 20;
6
- const LIST_PAGE_HARD_CAP = 100; // <= 2000 list items scanned per request
6
+ // Upper bound on list pages per call: 100 pages of 20 = at most 2000
7
+ // postings scanned. Paging usually stops sooner, on a short page or when
8
+ // offset reaches the first page's total (see the loop below).
9
+ const LIST_PAGE_HARD_CAP = 100;
10
+ // A multi-location posting's list row reads "2 Locations", "14 Locations".
11
+ const MULTI_LOCATION = /^\s*\d+\s+locations?\s*$/;
7
12
 
8
13
  /**
9
14
  * Fetch jobs from a Workday tenant via the public "CXS" JSON API.
@@ -40,6 +45,7 @@ export async function fetchWorkday(slug, ctx = {}) {
40
45
  const postings = [];
41
46
  let offset = 0;
42
47
  let pages = 0;
48
+ let firstTotal = 0;
43
49
  while (pages < LIST_PAGE_HARD_CAP) {
44
50
  const resp = await fetch(`${base}/jobs`, {
45
51
  method: 'POST',
@@ -48,19 +54,24 @@ export async function fetchWorkday(slug, ctx = {}) {
48
54
  });
49
55
 
50
56
  if (!resp.ok) {
51
- if (resp.status === 404) return []; // wrong site / no such board
52
57
  if (offset === 0) {
58
+ if (resp.status === 404) return []; // wrong site / no such board
53
59
  throw atsErrorFromStatus(resp.status, `Workday API error for ${slug} (${tenant}/${env}/${site}): ${resp.status}`);
54
60
  }
55
- break; // mid-paging failure: keep what we have
61
+ break; // mid-paging failure (any status): keep what we have
56
62
  }
57
63
 
58
64
  const data = await resp.json();
59
65
  const page = data.jobPostings || [];
66
+ // Some tenants report the real `total` only at offset 0 and send
67
+ // `total: 0` on every later page, so only the first page's figure
68
+ // is trusted. A short page is the other stop signal.
69
+ if (pages === 0) firstTotal = data.total || 0;
60
70
  postings.push(...page);
61
71
  pages += 1;
62
72
  offset += LIST_PAGE_SIZE;
63
- if (page.length === 0 || offset >= (data.total || 0)) break;
73
+ if (page.length < LIST_PAGE_SIZE) break;
74
+ if (firstTotal > 0 && offset >= firstTotal) break;
64
75
  }
65
76
 
66
77
  // 2. Filter-aware candidate selection BEFORE the N+1 detail cost.
@@ -76,6 +87,9 @@ export async function fetchWorkday(slug, ctx = {}) {
76
87
  const inc = fc.locationIncludes.map(s => String(s).toLowerCase());
77
88
  candidates = candidates.filter(p => {
78
89
  const loc = (p.locationsText || '').toLowerCase();
90
+ // "2 Locations" says nothing about where. The row stays a candidate
91
+ // and the pass after hydration decides on the detail's location list.
92
+ if (MULTI_LOCATION.test(loc)) return true;
79
93
  return inc.some(s => loc.includes(s));
80
94
  });
81
95
  }
@@ -91,16 +105,21 @@ export async function fetchWorkday(slug, ctx = {}) {
91
105
  }
92
106
 
93
107
  // 3. Bound the detail-fetch set.
94
- // NOTE: huge-tenant coverage is intentionally capped for v1
95
- // (Salesforce ~1398 postings). A description `filter` is applied
96
- // by the library AFTER this returns, so for that case we keep the
97
- // full backstop instead of truncating tightly to `limit` (which
98
- // could hydrate jobs that all fail the regex while better matches
99
- // go unscanned). Proper fix (smart pagination / rate-limited
100
- // concurrency / surfaced truncation) is tracked in #26, to be
101
- // designed alongside retry/rate-limit work (#7).
108
+ // NOTE: huge-tenant coverage is intentionally capped for v1. Two
109
+ // caps apply: the list scan above stops at LIST_PAGE_HARD_CAP pages
110
+ // (2000 postings, enough for Salesforce's ~1398), and the detail set
111
+ // is cut to MAX_DETAIL_FETCHES here. A description `filter` is
112
+ // applied by the library AFTER this returns, so for that case we
113
+ // keep the full backstop instead of truncating tightly to `limit`
114
+ // (which could hydrate jobs that all fail the regex while better
115
+ // matches go unscanned). The library pages with `offset` after this
116
+ // returns, so the budget covers the page plus what precedes it.
117
+ // Proper fix (smart pagination / rate-limited concurrency / surfaced
118
+ // truncation) is tracked in #26, to be designed alongside
119
+ // retry/rate-limit work (#7).
102
120
  const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
103
- const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(limit, MAX_DETAIL_FETCHES);
121
+ const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
122
+ const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
104
123
  candidates = candidates.slice(0, cap);
105
124
 
106
125
  // 4. Hydrate descriptions via the per-posting detail endpoint.
@@ -126,7 +145,9 @@ export async function fetchWorkday(slug, ctx = {}) {
126
145
  title: p.title || info.title || '',
127
146
  department: '',
128
147
  location: info.location || p.locationsText || '',
129
- description: stripHtml(info.jobDescription || ''),
148
+ locations: info.additionalLocations || [],
149
+ workplace: parseWorkdayRemoteType(info.remoteType),
150
+ description: info.jobDescription || '',
130
151
  url: `https://${tenant}.${env}.myworkdayjobs.com/${site}${externalPath}`,
131
152
  postedAt: parseWorkdayDate(info.startDate) || normalizePostedOn(p.postedOn),
132
153
  salary: null, // normalizer extracts from description text
@@ -142,6 +163,20 @@ export async function fetchWorkday(slug, ctx = {}) {
142
163
  return jobs;
143
164
  }
144
165
 
166
+ /**
167
+ * Detail `remoteType` is free text set per tenant: "Remote", "Hybrid",
168
+ * "Office - Flexible", "On-site". A flexible office arrangement counts as
169
+ * hybrid, so that check runs before the office one. Some tenants send no
170
+ * value at all; the location string decides then.
171
+ */
172
+ function parseWorkdayRemoteType(remoteType) {
173
+ const s = String(remoteType || '').toLowerCase();
174
+ if (/remote/.test(s)) return 'remote';
175
+ if (/hybrid|flexible/.test(s)) return 'hybrid';
176
+ if (/office|on-?site/.test(s)) return 'onsite';
177
+ return null;
178
+ }
179
+
145
180
  /**
146
181
  * Workday list `postedOn` is a relative string ("Posted Today",
147
182
  * "Posted 5 Days Ago", "Posted 30+ Days Ago"). Decide membership in
package/src/cli.js CHANGED
@@ -9,6 +9,8 @@
9
9
  * jd-intel registry search <query>
10
10
  */
11
11
 
12
+ import { realpathSync } from 'node:fs';
13
+ import { fileURLToPath } from 'node:url';
12
14
  import { fetchJobs } from './index.js';
13
15
  import { detectAts, searchRegistry } from './registry.js';
14
16
 
@@ -85,7 +87,7 @@ async function main() {
85
87
  console.log(`Found ${jobs.length} jobs\n`);
86
88
 
87
89
  for (const job of jobs.slice(0, 20)) {
88
- const salary = job.salary ? ` | $${job.salary.min?.toLocaleString()}-$${job.salary.max?.toLocaleString()}` : '';
90
+ const salary = job.salary ? ` | ${formatSalary(job.salary)}` : '';
89
91
  const loc = job.location ? ` | ${job.location}` : '';
90
92
  const dept = job.department ? ` [${job.department}]` : '';
91
93
  console.log(` ${job.title}${dept}${loc}${salary}`);
@@ -162,8 +164,8 @@ Fetch options:
162
164
  --title-filter pattern Regex matched against TITLE only (role identity)
163
165
  --filter pattern Regex matched across title, department, description (topic/scope)
164
166
  --posted-within-days N Only jobs posted in the last N days
165
- --location-include "A,B,C" Keep jobs whose location contains any of these
166
- --location-exclude "A,B,C" Drop jobs whose location contains any of these
167
+ --location-include "A,B,C" Keep jobs where any listed location contains one of these
168
+ --location-exclude "A,B,C" Drop jobs only when every listed location contains one of these
167
169
  --limit N Cap results (default 100)
168
170
  --json Output full JSON
169
171
 
@@ -184,7 +186,32 @@ Examples:
184
186
  }
185
187
  }
186
188
 
187
- main().catch(err => {
188
- console.error('Error:', err.message);
189
- process.exit(1);
190
- });
189
+ export function formatSalary({ min, max, currency, period }) {
190
+ const hasMin = min != null;
191
+ const hasMax = max != null;
192
+ let range;
193
+ if (hasMin && hasMax) range = `${min.toLocaleString()}-${max.toLocaleString()}`;
194
+ else if (hasMin) range = `from ${min.toLocaleString()}`;
195
+ else range = `up to ${max.toLocaleString()}`;
196
+ const unit = period === 'hour' ? '/hr' : period === 'month' ? '/mo' : '';
197
+ return `${range} ${currency}${unit}`;
198
+ }
199
+
200
+ // Boot only when this file is the script Node was started with, so a test
201
+ // can import formatSalary without running a command. argv[1] is resolved
202
+ // through realpath because npm installs the bin as a symlink into .bin/,
203
+ // while import.meta.url already points at the real file.
204
+ function isEntrypoint() {
205
+ try {
206
+ return realpathSync(process.argv[1]) === fileURLToPath(import.meta.url);
207
+ } catch {
208
+ return false;
209
+ }
210
+ }
211
+
212
+ if (isEntrypoint()) {
213
+ main().catch(err => {
214
+ console.error('Error:', err.message);
215
+ process.exit(1);
216
+ });
217
+ }
package/src/filters.js CHANGED
@@ -4,14 +4,33 @@
4
4
  * Facts go here (deterministic field matches). Interpretations stay with the
5
5
  * caller — this module does substring matching on structured fields, nothing
6
6
  * semantic.
7
+ *
8
+ * Returns the page as an array. applyFiltersDetailed returns the same page
9
+ * plus total_matched, the match count before offset and limit.
7
10
  */
8
11
  export function applyFilters(jobs, options = {}) {
12
+ return applyFiltersDetailed(jobs, options).jobs;
13
+ }
14
+
15
+ /**
16
+ * Filter, sort, then page.
17
+ *
18
+ * Order is applied after the filters and before offset and limit, so a cut
19
+ * drops the oldest matches first. 'newest' sorts by postedAt descending with
20
+ * undated jobs last and ties broken by id, which keeps pages deterministic.
21
+ * 'board' keeps the order the adapter returned.
22
+ *
23
+ * @returns {{ jobs: Array, total_matched: number }}
24
+ */
25
+ export function applyFiltersDetailed(jobs, options = {}) {
9
26
  const {
10
27
  titleFilter,
11
28
  filter,
12
29
  postedWithinDays,
13
30
  locationIncludes,
14
31
  locationExcludes,
32
+ order = 'newest',
33
+ offset = 0,
15
34
  limit = 100,
16
35
  } = options;
17
36
 
@@ -42,25 +61,60 @@ export function applyFilters(jobs, options = {}) {
42
61
 
43
62
  if (Array.isArray(locationIncludes) && locationIncludes.length > 0) {
44
63
  const matchers = locationIncludes.map(makeLocationMatcher);
45
- result = result.filter(j => {
46
- const loc = (j.location || '').toLowerCase();
47
- return matchers.some(m => m(loc));
48
- });
64
+ result = result.filter(j => jobLocations(j).some(loc => matchers.some(m => m(loc))));
49
65
  }
50
66
 
51
67
  if (Array.isArray(locationExcludes) && locationExcludes.length > 0) {
52
68
  const matchers = locationExcludes.map(makeLocationMatcher);
53
- result = result.filter(j => {
54
- const loc = (j.location || '').toLowerCase();
55
- return !matchers.some(m => m(loc));
56
- });
69
+ result = result.filter(j => !jobLocations(j).every(loc => matchers.some(m => m(loc))));
57
70
  }
58
71
 
59
- if (typeof limit === 'number' && result.length > limit) {
60
- result = result.slice(0, limit);
72
+ const total_matched = result.length;
73
+
74
+ if (order !== 'board') {
75
+ result = [...result].sort(byNewest);
61
76
  }
62
77
 
63
- return result;
78
+ const start = typeof offset === 'number' && offset > 0 ? offset : 0;
79
+ const end = typeof limit === 'number' ? start + limit : undefined;
80
+ if (start > 0 || (end !== undefined && result.length > end)) {
81
+ result = result.slice(start, end);
82
+ }
83
+
84
+ return { jobs: result, total_matched };
85
+ }
86
+
87
+ /**
88
+ * Every location a job is open in, lowercased. A job passes an include when
89
+ * any of them matches and is dropped by an exclude only when all of them
90
+ * match: a role open in Berlin and New York is still open in Berlin for
91
+ * someone excluding the US (issue #68). Jobs from before `locations`
92
+ * existed fall back to the single `location` string.
93
+ */
94
+ function jobLocations(job) {
95
+ const list = Array.isArray(job.locations) && job.locations.length > 0
96
+ ? job.locations
97
+ : [job.location || ''];
98
+ return list.map(loc => String(loc).toLowerCase());
99
+ }
100
+
101
+ function postedTime(job) {
102
+ if (!job.postedAt) return null;
103
+ const t = new Date(job.postedAt).getTime();
104
+ return Number.isFinite(t) ? t : null;
105
+ }
106
+
107
+ function byNewest(a, b) {
108
+ const ta = postedTime(a);
109
+ const tb = postedTime(b);
110
+ if (ta !== tb) {
111
+ if (ta === null) return 1;
112
+ if (tb === null) return -1;
113
+ return tb - ta;
114
+ }
115
+ const ia = a.id || '';
116
+ const ib = b.id || '';
117
+ return ia < ib ? -1 : ia > ib ? 1 : 0;
64
118
  }
65
119
 
66
120
  /**
package/src/index.js CHANGED
@@ -8,11 +8,23 @@
8
8
 
9
9
  import { ADAPTERS, ATS_NAMES } from './adapters/index.js';
10
10
  import { loadRegistry, searchRegistry, detectAts, findAtsBySlug, findEntryBySlug, getRegistrySource } from './registry.js';
11
- import { applyFilters } from './filters.js';
11
+ import { applyFiltersDetailed } from './filters.js';
12
12
 
13
13
  /**
14
14
  * Fetch jobs from a company's ATS board.
15
15
  *
16
+ * Same options as fetchJobsDetailed; returns the page as an array.
17
+ *
18
+ * @returns {Promise<Array>} Normalized, filtered job objects
19
+ */
20
+ export async function fetchJobs(options = {}) {
21
+ const { jobs } = await fetchJobsDetailed(options);
22
+ return jobs;
23
+ }
24
+
25
+ /**
26
+ * Fetch jobs from a company's ATS board, with the match count.
27
+ *
16
28
  * @param {Object} options
17
29
  * @param {string} options.company - Company slug or name
18
30
  * @param {string} [options.ats] - Specific ATS platform. If omitted, auto-detects.
@@ -20,12 +32,14 @@ import { applyFilters } from './filters.js';
20
32
  * @param {string} [options.titleFilter] - Regex matched against title only. Use for role identity ("product manager", "staff engineer").
21
33
  * @param {string} [options.filter] - Regex matched across title, department, description. Use for topic/scope.
22
34
  * @param {number} [options.postedWithinDays] - Only return jobs posted within N days.
23
- * @param {string[]} [options.locationIncludes] - Keep jobs whose location contains any of these (case-insensitive).
24
- * @param {string[]} [options.locationExcludes] - Drop jobs whose location contains any of these (case-insensitive).
25
- * @param {number} [options.limit=100] - Maximum jobs to return after filtering.
26
- * @returns {Promise<Array>} Normalized, filtered job objects
35
+ * @param {string[]} [options.locationIncludes] - Keep jobs where any listed location contains any of these (case-insensitive).
36
+ * @param {string[]} [options.locationExcludes] - Drop jobs only when every listed location contains one of these (case-insensitive).
37
+ * @param {'newest'|'board'} [options.order='newest'] - 'newest': by postedAt descending, undated last, ties by id. 'board': the adapter's own order.
38
+ * @param {number} [options.offset=0] - Matches to skip after sorting (paging).
39
+ * @param {number} [options.limit=100] - Maximum jobs to return after offset.
40
+ * @returns {Promise<{ jobs: Array, total_matched: number }>} The page, plus the match count before offset and limit
27
41
  */
28
- export async function fetchJobs({
42
+ export async function fetchJobsDetailed({
29
43
  company,
30
44
  ats,
31
45
  config,
@@ -34,6 +48,8 @@ export async function fetchJobs({
34
48
  postedWithinDays,
35
49
  locationIncludes,
36
50
  locationExcludes,
51
+ order = 'newest',
52
+ offset = 0,
37
53
  limit = 100,
38
54
  } = {}) {
39
55
  if (!company) throw new Error('Company slug required');
@@ -45,7 +61,7 @@ export async function fetchJobs({
45
61
  // adapters declare fetch{Name}(slug) and ignore extra positional args
46
62
  // (JS no-op), so this is backward-compatible. Filter-aware adapters
47
63
  // (e.g. Workday) use it to avoid mass detail-fetching on huge tenants.
48
- const filterContext = { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes, limit };
64
+ const filterContext = { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes, offset, limit };
49
65
 
50
66
  let jobs;
51
67
  if (ats) {
@@ -93,7 +109,7 @@ export async function fetchJobs({
93
109
  }
94
110
  }
95
111
 
96
- return applyFilters(jobs, { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes, limit });
112
+ return applyFiltersDetailed(jobs, { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes, order, offset, limit });
97
113
  }
98
114
 
99
115
  /**
@@ -134,7 +150,7 @@ export { fetchLever } from './adapters/lever.js';
134
150
  export { fetchAshby } from './adapters/ashby.js';
135
151
 
136
152
  // Re-export filter logic for reuse (e.g., by the MCP server)
137
- export { applyFilters } from './filters.js';
153
+ export { applyFilters, applyFiltersDetailed } from './filters.js';
138
154
 
139
155
  // Re-export the list of supported ATS names (e.g. so the MCP layer can report
140
156
  // the full set detectAts probes, instead of hardcoding a stale subset).
package/src/normalizer.js CHANGED
@@ -13,20 +13,35 @@ export function jobId(company, title, ats, location = '') {
13
13
 
14
14
  /**
15
15
  * Normalize a raw ATS job object into the unified schema.
16
+ *
17
+ * Adapters pass `description` as HTML. This is the one place it is
18
+ * stripped and decoded (issue #66): a second pass would delete text the
19
+ * author escaped on purpose (`&lt;5 years`) and leave entities the first
20
+ * pass exposed (`&amp;mdash;` -> `&mdash;`) as literal noise.
21
+ *
22
+ * `raw.workplace` is the ATS's own arrangement, already mapped by the
23
+ * adapter to 'remote' | 'hybrid' | 'onsite', or null when the platform
24
+ * gives no signal. `raw.locations` lists every place the posting is open
25
+ * in; `location` stays the primary because it feeds the id (issue #68).
16
26
  */
17
27
  export function normalize(raw, ats) {
18
28
  const now = new Date().toISOString();
29
+ const description = stripHtml(raw.description || '');
30
+ const location = raw.location || '';
31
+ const workplace = resolveWorkplace(raw.workplace, location);
19
32
  return {
20
- id: jobId(raw.company || raw.companySlug, raw.title, ats, raw.location || ''),
33
+ id: jobId(raw.company || raw.companySlug, raw.title, ats, location),
21
34
  company: raw.company || raw.companySlug || '',
22
35
  companySlug: raw.companySlug || '',
23
36
  ats,
24
37
  title: raw.title || '',
25
38
  department: raw.department || '',
26
- location: raw.location || '',
27
- locationType: detectLocationType(raw.location || ''),
28
- salary: raw.salary || extractSalaryFromText(raw.description || ''),
29
- description: stripHtml(raw.description || ''),
39
+ location,
40
+ locations: uniqueLocations(location, raw.locations),
41
+ locationType: workplace.type,
42
+ workplace,
43
+ salary: raw.salary || extractSalaryFromText(description),
44
+ description,
30
45
  url: raw.url || '',
31
46
  postedAt: raw.postedAt || null,
32
47
  firstSeen: now,
@@ -36,62 +51,203 @@ export function normalize(raw, ats) {
36
51
  };
37
52
  }
38
53
 
54
+ const CURRENCY_CODES = 'USD|EUR|GBP|CAD|AUD|NZD|CHF|SEK|NOK|DKK|PLN|CZK|HUF|INR|SGD|HKD|JPY|CNY|BRL|MXN|ZAR|AED|ILS';
55
+ const SYMBOL_CURRENCY = { $: 'USD', '€': 'EUR', '£': 'GBP' };
56
+
57
+ // A number as job posts write it: 1,234,567 / 1.234.567 / 1234, with an
58
+ // optional one- or two-digit decimal part (211.4, 40.50, 60.000,50).
59
+ // Exactly three digits after a dot are a thousands group, the way Dutch
60
+ // and German boards write it: "€60.000" is sixty thousand, not sixty.
61
+ const NUMBER =
62
+ '\\d{1,3}(?:,\\d{3})+(?:\\.\\d{1,2})?' +
63
+ '|\\d{1,3}(?:\\.\\d{3})+(?:,\\d{1,2})?' +
64
+ '|\\d+(?:[.,]\\d{1,2})?';
65
+
66
+ // One side of a range: optional code before, optional symbol, the number,
67
+ // optional K, optional code after. The lookarounds keep the number from
68
+ // starting or ending inside a longer one ("234.567" out of "1.234.567").
69
+ const amountPattern = (p) =>
70
+ `(?:\\b(?<${p}CodeBefore>${CURRENCY_CODES})\\s?)?` +
71
+ `(?<${p}Sym>[$€£])?\\s?` +
72
+ `(?<![\\d.,])(?<${p}Num>${NUMBER})(?!\\d|[.,]\\d)\\s?` +
73
+ `(?<${p}K>[kK]\\b)?` +
74
+ `(?:\\s?(?<${p}CodeAfter>${CURRENCY_CODES})\\b)?`;
75
+
76
+ const SALARY_RANGE = new RegExp(
77
+ `${amountPattern('lo')}\\s*(?:[-–—]|\\bto\\b)\\s*${amountPattern('hi')}`,
78
+ 'g'
79
+ );
80
+
81
+ const HOUR_RE = /\b(?:per|an|each)\s+hour\b|\/\s*(?:hr|hour)\b|\bhourly\b/i;
82
+ const MONTH_RE = /\b(?:per|a|each)\s+month\b|\/\s*(?:mo|month)\b|\bmonthly\b/i;
83
+ const YEAR_RE = /\b(?:per|a|each)\s+(?:year|annum)\b|\/\s*(?:yr|year)\b|\b(?:annual(?:ly|ized)?|yearly)\b/i;
84
+
39
85
  /**
40
- * Detect location type from location string.
41
- */
42
- /**
43
- * Extract salary range from job description text.
44
- * Matches patterns like: $162,400 - $243,600 or $150K-$200K
86
+ * Extract a salary range from decoded job text.
87
+ *
88
+ * Accepts hyphen, en dash, em dash or "to" between the two amounts, an
89
+ * optional ISO currency code before, between or after them, `$` / EUR /
90
+ * GBP symbols, decimal K shorthand ($211.4K), and thousands grouped with
91
+ * either a comma or a dot (60,000 and 60.000 are both sixty thousand).
92
+ * A code wins over a symbol, so "$120,000 - $150,000 CAD" is CAD. Ranges
93
+ * with no currency marker at all (years, headcounts) are ignored.
94
+ *
95
+ * @returns {{min:number,max:number,currency:string,period:('year'|'month'|'hour'|null),source:'text'}|null}
45
96
  */
46
- function extractSalaryFromText(text) {
97
+ export function extractSalaryFromText(text) {
47
98
  if (!text) return null;
48
- // Match: $162,400 - $243,600 (full numbers)
49
- const fullMatch = text.match(/\$([\d,]+)\s*[-–]\s*\$([\d,]+)/);
50
- if (fullMatch) {
51
- return {
52
- min: parseInt(fullMatch[1].replace(/,/g, '')),
53
- max: parseInt(fullMatch[2].replace(/,/g, '')),
54
- currency: 'USD',
55
- };
56
- }
57
- // Match: $150K - $200K (shorthand)
58
- const kMatch = text.match(/\$(\d+)[kK]\s*[-–]\s*\$(\d+)[kK]/);
59
- if (kMatch) {
99
+ for (const m of text.matchAll(SALARY_RANGE)) {
100
+ const g = m.groups;
101
+ const end = m.index + m[0].length;
102
+ const after = text.slice(end, end + 40);
103
+ // "$20 - $30 million" is a revenue figure, not pay.
104
+ if (/^\s*(?:million|billion|m|bn?)\b/i.test(after)) continue;
105
+
106
+ const code = g.loCodeBefore || g.loCodeAfter || g.hiCodeBefore || g.hiCodeAfter;
107
+ const sym = g.loSym || g.hiSym;
108
+ if (!code && !sym) continue;
109
+
110
+ let min = parseAmount(g.loNum);
111
+ let max = parseAmount(g.hiNum);
112
+ if (g.loK || g.hiK) {
113
+ // "$150-200K" carries the K once for both sides.
114
+ if (min < 1000) min = Math.round(min * 1000);
115
+ if (max < 1000) max = Math.round(max * 1000);
116
+ }
117
+ if (!(min > 0) || !(max > 0)) continue;
118
+
119
+ const before = text.slice(Math.max(0, m.index - 40), m.index);
60
120
  return {
61
- min: parseInt(kMatch[1]) * 1000,
62
- max: parseInt(kMatch[2]) * 1000,
63
- currency: 'USD',
121
+ min,
122
+ max,
123
+ currency: (code || SYMBOL_CURRENCY[sym]).toUpperCase(),
124
+ period: detectPeriod(before, after, min),
125
+ source: 'text',
64
126
  };
65
127
  }
66
128
  return null;
67
129
  }
68
130
 
69
- function detectLocationType(location) {
70
- const lower = location.toLowerCase();
71
- if (/remote/i.test(lower)) return 'remote';
72
- if (/hybrid/i.test(lower)) return 'hybrid';
73
- if (/on-?site/i.test(lower)) return 'onsite';
74
- return location ? 'onsite' : 'unknown';
131
+ // Dots grouping thousands mean a comma is the decimal mark, and vice versa.
132
+ function parseAmount(s) {
133
+ if (/^\d{1,3}(?:\.\d{3})+/.test(s)) return Number(s.replace(/\./g, '').replace(',', '.'));
134
+ return Number(s.replace(/,(?=\d{3})/g, '').replace(',', '.'));
135
+ }
136
+
137
+ function detectPeriod(before, after, min) {
138
+ const explicit = periodWord(after) || periodWord(before);
139
+ if (explicit) return explicit;
140
+ return min >= 10000 ? 'year' : null;
141
+ }
142
+
143
+ function periodWord(text) {
144
+ if (HOUR_RE.test(text)) return 'hour';
145
+ if (MONTH_RE.test(text)) return 'month';
146
+ if (YEAR_RE.test(text)) return 'year';
147
+ return null;
148
+ }
149
+
150
+ const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
151
+
152
+ /**
153
+ * The platform's own value wins. Without one, a keyword in the location
154
+ * string is the next best signal. Without either the type is 'unknown':
155
+ * a city name alone does not say the role is onsite, and a guessed
156
+ * 'onsite' reads as a fact to whoever consumes it.
157
+ *
158
+ * @returns {{type:('remote'|'hybrid'|'onsite'|'unknown'), source:('ats'|'text'|null)}}
159
+ */
160
+ function resolveWorkplace(native, location) {
161
+ if (WORKPLACE_TYPES.has(native)) return { type: native, source: 'ats' };
162
+ const guessed = workplaceFromText(location);
163
+ if (guessed) return { type: guessed, source: 'text' };
164
+ return { type: 'unknown', source: null };
165
+ }
166
+
167
+ function workplaceFromText(location) {
168
+ const lower = (location || '').toLowerCase();
169
+ if (/remote/.test(lower)) return 'remote';
170
+ if (/hybrid/.test(lower)) return 'hybrid';
171
+ if (/on-?site/.test(lower)) return 'onsite';
172
+ return null;
173
+ }
174
+
175
+ function uniqueLocations(primary, extra) {
176
+ const out = [];
177
+ const seen = new Set();
178
+ for (const loc of [primary, ...(Array.isArray(extra) ? extra : [])]) {
179
+ const s = typeof loc === 'string' ? loc.trim() : '';
180
+ if (!s || seen.has(s.toLowerCase())) continue;
181
+ seen.add(s.toLowerCase());
182
+ out.push(s);
183
+ }
184
+ return out;
75
185
  }
76
186
 
77
187
  /**
78
188
  * Strip HTML tags and convert to clean text.
189
+ *
190
+ * Block closers become line breaks and list items become bullets before
191
+ * the remaining tags are removed. Entities are decoded LAST, so a literal
192
+ * `&lt;` in the source text never turns into a tag that gets stripped.
79
193
  */
80
194
  export function stripHtml(html) {
81
195
  if (!html) return '';
82
- return html
196
+ const text = html
83
197
  .replace(/<br\s*\/?>/gi, '\n')
84
- .replace(/<\/p>/gi, '\n\n')
85
- .replace(/<\/li>/gi, '\n')
86
- .replace(/<li>/gi, '- ')
87
- .replace(/<\/h[1-6]>/gi, '\n\n')
88
- .replace(/<h[1-6][^>]*>/gi, '## ')
89
- .replace(/<[^>]+>/g, '')
90
- .replace(/&amp;/g, '&')
91
- .replace(/&lt;/g, '<')
92
- .replace(/&gt;/g, '>')
93
- .replace(/&nbsp;/g, ' ')
94
- .replace(/&#\d+;/g, '')
198
+ .replace(/<\/(?:p|h[1-6])\s*>/gi, '\n\n')
199
+ .replace(/<\/(?:li|div|td|tr|ul|ol|table|section)\s*>/gi, '\n')
200
+ // Word-pasted markup (Lever lists, issue #64) opens a <p> inside each
201
+ // <li>; swallowing it keeps the item text on the bullet's line.
202
+ .replace(/<li\b[^>]*>\s*(?:<p\b[^>]*>\s*)?/gi, '- ')
203
+ .replace(/<h[1-6]\b[^>]*>/gi, '## ')
204
+ .replace(/<[^>]+>/g, '');
205
+ return decodeEntities(text)
206
+ .replace(/\u00a0/g, ' ')
95
207
  .replace(/\n{3,}/g, '\n\n')
96
208
  .trim();
97
209
  }
210
+
211
+ const NAMED_ENTITIES = {
212
+ lt: '<', gt: '>', quot: '"', apos: "'", nbsp: '\u00a0',
213
+ mdash: '—', ndash: '–', hellip: '…',
214
+ lsquo: '‘', rsquo: '’', ldquo: '“', rdquo: '”',
215
+ sbquo: '‚', bdquo: '„', laquo: '«', raquo: '»',
216
+ bull: '•', middot: '·', copy: '©', reg: '®', trade: '™',
217
+ deg: '°', times: '×', euro: '€', pound: '£', yen: '¥', cent: '¢',
218
+ agrave: 'à', aacute: 'á', acirc: 'â', atilde: 'ã', auml: 'ä', aring: 'å', aelig: 'æ',
219
+ ccedil: 'ç', egrave: 'è', eacute: 'é', ecirc: 'ê', euml: 'ë',
220
+ igrave: 'ì', iacute: 'í', icirc: 'î', iuml: 'ï', ntilde: 'ñ',
221
+ ograve: 'ò', oacute: 'ó', ocirc: 'ô', otilde: 'õ', ouml: 'ö', oslash: 'ø',
222
+ ugrave: 'ù', uacute: 'ú', ucirc: 'û', uuml: 'ü', yacute: 'ý', yuml: 'ÿ', szlig: 'ß',
223
+ Agrave: 'À', Aacute: 'Á', Acirc: 'Â', Atilde: 'Ã', Auml: 'Ä', Aring: 'Å', AElig: 'Æ',
224
+ Ccedil: 'Ç', Egrave: 'È', Eacute: 'É', Ecirc: 'Ê', Euml: 'Ë',
225
+ Igrave: 'Ì', Iacute: 'Í', Icirc: 'Î', Iuml: 'Ï', Ntilde: 'Ñ',
226
+ Ograve: 'Ò', Oacute: 'Ó', Ocirc: 'Ô', Otilde: 'Õ', Ouml: 'Ö', Oslash: 'Ø',
227
+ Ugrave: 'Ù', Uacute: 'Ú', Ucirc: 'Û', Uuml: 'Ü', Yacute: 'Ý',
228
+ };
229
+
230
+ /**
231
+ * Decode one layer of entity encoding (and unwrap CDATA) to real text.
232
+ *
233
+ * Decimal and hex references go through String.fromCodePoint, named
234
+ * references through the table above. `&amp;` is intentionally resolved
235
+ * LAST so double-encoded sequences (`&amp;mdash;`, `&amp;amp;`) collapse
236
+ * by exactly one layer per call. Used for the outer escaping Greenhouse
237
+ * and the Teamtailor RSS feed apply, and as stripHtml's final step.
238
+ */
239
+ export function decodeEntities(s) {
240
+ if (!s) return '';
241
+ return s
242
+ .replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1')
243
+ .replace(/&#(\d+);/g, (m, dec) => codePointToString(parseInt(dec, 10), m))
244
+ .replace(/&#[xX]([0-9a-fA-F]+);/g, (m, hex) => codePointToString(parseInt(hex, 16), m))
245
+ .replace(/&([A-Za-z][A-Za-z0-9]*);/g, (m, name) =>
246
+ (name !== 'amp' && Object.hasOwn(NAMED_ENTITIES, name)) ? NAMED_ENTITIES[name] : m)
247
+ .replace(/&amp;/g, '&');
248
+ }
249
+
250
+ function codePointToString(cp, fallback) {
251
+ if (!cp || cp > 0x10ffff || (cp >= 0xd800 && cp <= 0xdfff)) return fallback;
252
+ return String.fromCodePoint(cp);
253
+ }