jd-intel 0.8.2 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,9 +1,14 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
 
4
4
  const MAX_DETAIL_FETCHES = 100;
5
5
  const LIST_PAGE_SIZE = 20;
6
- const LIST_PAGE_HARD_CAP = 100; // <= 2000 list items scanned per request
6
+ // Upper bound on list pages per call: 100 pages of 20 = at most 2000
7
+ // postings scanned. Paging usually stops sooner, on a short page or when
8
+ // offset reaches the first page's total (see the loop below).
9
+ const LIST_PAGE_HARD_CAP = 100;
10
+ // A multi-location posting's list row reads "2 Locations", "14 Locations".
11
+ const MULTI_LOCATION = /^\s*\d+\s+locations?\s*$/;
7
12
 
8
13
  /**
9
14
  * Fetch jobs from a Workday tenant via the public "CXS" JSON API.
@@ -40,6 +45,7 @@ export async function fetchWorkday(slug, ctx = {}) {
40
45
  const postings = [];
41
46
  let offset = 0;
42
47
  let pages = 0;
48
+ let firstTotal = 0;
43
49
  while (pages < LIST_PAGE_HARD_CAP) {
44
50
  const resp = await fetch(`${base}/jobs`, {
45
51
  method: 'POST',
@@ -48,19 +54,24 @@ export async function fetchWorkday(slug, ctx = {}) {
48
54
  });
49
55
 
50
56
  if (!resp.ok) {
51
- if (resp.status === 404) return []; // wrong site / no such board
52
57
  if (offset === 0) {
58
+ if (resp.status === 404) return []; // wrong site / no such board
53
59
  throw atsErrorFromStatus(resp.status, `Workday API error for ${slug} (${tenant}/${env}/${site}): ${resp.status}`);
54
60
  }
55
- break; // mid-paging failure: keep what we have
61
+ break; // mid-paging failure (any status): keep what we have
56
62
  }
57
63
 
58
64
  const data = await resp.json();
59
65
  const page = data.jobPostings || [];
66
+ // Some tenants report the real `total` only at offset 0 and send
67
+ // `total: 0` on every later page, so only the first page's figure
68
+ // is trusted. A short page is the other stop signal.
69
+ if (pages === 0) firstTotal = data.total || 0;
60
70
  postings.push(...page);
61
71
  pages += 1;
62
72
  offset += LIST_PAGE_SIZE;
63
- if (page.length === 0 || offset >= (data.total || 0)) break;
73
+ if (page.length < LIST_PAGE_SIZE) break;
74
+ if (firstTotal > 0 && offset >= firstTotal) break;
64
75
  }
65
76
 
66
77
  // 2. Filter-aware candidate selection BEFORE the N+1 detail cost.
@@ -76,6 +87,9 @@ export async function fetchWorkday(slug, ctx = {}) {
76
87
  const inc = fc.locationIncludes.map(s => String(s).toLowerCase());
77
88
  candidates = candidates.filter(p => {
78
89
  const loc = (p.locationsText || '').toLowerCase();
90
+ // "2 Locations" says nothing about where. The row stays a candidate
91
+ // and the pass after hydration decides on the detail's location list.
92
+ if (MULTI_LOCATION.test(loc)) return true;
79
93
  return inc.some(s => loc.includes(s));
80
94
  });
81
95
  }
@@ -91,16 +105,21 @@ export async function fetchWorkday(slug, ctx = {}) {
91
105
  }
92
106
 
93
107
  // 3. Bound the detail-fetch set.
94
- // NOTE: huge-tenant coverage is intentionally capped for v1
95
- // (Salesforce ~1398 postings). A description `filter` is applied
96
- // by the library AFTER this returns, so for that case we keep the
97
- // full backstop instead of truncating tightly to `limit` (which
98
- // could hydrate jobs that all fail the regex while better matches
99
- // go unscanned). Proper fix (smart pagination / rate-limited
100
- // concurrency / surfaced truncation) is tracked in #26, to be
101
- // designed alongside retry/rate-limit work (#7).
108
+ // NOTE: huge-tenant coverage is intentionally capped for v1. Two
109
+ // caps apply: the list scan above stops at LIST_PAGE_HARD_CAP pages
110
+ // (2000 postings, enough for Salesforce's ~1398), and the detail set
111
+ // is cut to MAX_DETAIL_FETCHES here. A description `filter` is
112
+ // applied by the library AFTER this returns, so for that case we
113
+ // keep the full backstop instead of truncating tightly to `limit`
114
+ // (which could hydrate jobs that all fail the regex while better
115
+ // matches go unscanned). The library pages with `offset` after this
116
+ // returns, so the budget covers the page plus what precedes it.
117
+ // Proper fix (smart pagination / rate-limited concurrency / surfaced
118
+ // truncation) is tracked in #26, to be designed alongside
119
+ // retry/rate-limit work (#7).
102
120
  const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
103
- const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(limit, MAX_DETAIL_FETCHES);
121
+ const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
122
+ const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
104
123
  candidates = candidates.slice(0, cap);
105
124
 
106
125
  // 4. Hydrate descriptions via the per-posting detail endpoint.
@@ -126,7 +145,9 @@ export async function fetchWorkday(slug, ctx = {}) {
126
145
  title: p.title || info.title || '',
127
146
  department: '',
128
147
  location: info.location || p.locationsText || '',
129
- description: stripHtml(info.jobDescription || ''),
148
+ locations: info.additionalLocations || [],
149
+ workplace: parseWorkdayRemoteType(info.remoteType),
150
+ description: info.jobDescription || '',
130
151
  url: `https://${tenant}.${env}.myworkdayjobs.com/${site}${externalPath}`,
131
152
  postedAt: parseWorkdayDate(info.startDate) || normalizePostedOn(p.postedOn),
132
153
  salary: null, // normalizer extracts from description text
@@ -142,6 +163,20 @@ export async function fetchWorkday(slug, ctx = {}) {
142
163
  return jobs;
143
164
  }
144
165
 
166
+ /**
167
+ * Detail `remoteType` is free text set per tenant: "Remote", "Hybrid",
168
+ * "Office - Flexible", "On-site". A flexible office arrangement counts as
169
+ * hybrid, so that check runs before the office one. Some tenants send no
170
+ * value at all; the location string decides then.
171
+ */
172
+ function parseWorkdayRemoteType(remoteType) {
173
+ const s = String(remoteType || '').toLowerCase();
174
+ if (/remote/.test(s)) return 'remote';
175
+ if (/hybrid|flexible/.test(s)) return 'hybrid';
176
+ if (/office|on-?site/.test(s)) return 'onsite';
177
+ return null;
178
+ }
179
+
145
180
  /**
146
181
  * Workday list `postedOn` is a relative string ("Posted Today",
147
182
  * "Posted 5 Days Ago", "Posted 30+ Days Ago"). Decide membership in
package/src/cli.js CHANGED
@@ -9,6 +9,8 @@
9
9
  * jd-intel registry search <query>
10
10
  */
11
11
 
12
+ import { realpathSync } from 'node:fs';
13
+ import { fileURLToPath } from 'node:url';
12
14
  import { fetchJobs } from './index.js';
13
15
  import { detectAts, searchRegistry } from './registry.js';
14
16
 
@@ -85,7 +87,7 @@ async function main() {
85
87
  console.log(`Found ${jobs.length} jobs\n`);
86
88
 
87
89
  for (const job of jobs.slice(0, 20)) {
88
- const salary = job.salary ? ` | $${job.salary.min?.toLocaleString()}-$${job.salary.max?.toLocaleString()}` : '';
90
+ const salary = job.salary ? ` | ${formatSalary(job.salary)}` : '';
89
91
  const loc = job.location ? ` | ${job.location}` : '';
90
92
  const dept = job.department ? ` [${job.department}]` : '';
91
93
  console.log(` ${job.title}${dept}${loc}${salary}`);
@@ -162,8 +164,8 @@ Fetch options:
162
164
  --title-filter pattern Regex matched against TITLE only (role identity)
163
165
  --filter pattern Regex matched across title, department, description (topic/scope)
164
166
  --posted-within-days N Only jobs posted in the last N days
165
- --location-include "A,B,C" Keep jobs whose location contains any of these
166
- --location-exclude "A,B,C" Drop jobs whose location contains any of these
167
+ --location-include "A,B,C" Keep jobs where any listed location contains one of these
168
+ --location-exclude "A,B,C" Drop jobs only when every listed location contains one of these
167
169
  --limit N Cap results (default 100)
168
170
  --json Output full JSON
169
171
 
@@ -184,7 +186,32 @@ Examples:
184
186
  }
185
187
  }
186
188
 
187
- main().catch(err => {
188
- console.error('Error:', err.message);
189
- process.exit(1);
190
- });
189
+ export function formatSalary({ min, max, currency, period }) {
190
+ const hasMin = min != null;
191
+ const hasMax = max != null;
192
+ let range;
193
+ if (hasMin && hasMax) range = `${min.toLocaleString()}-${max.toLocaleString()}`;
194
+ else if (hasMin) range = `from ${min.toLocaleString()}`;
195
+ else range = `up to ${max.toLocaleString()}`;
196
+ const unit = period === 'hour' ? '/hr' : period === 'month' ? '/mo' : '';
197
+ return `${range} ${currency}${unit}`;
198
+ }
199
+
200
+ // Boot only when this file is the script Node was started with, so a test
201
+ // can import formatSalary without running a command. argv[1] is resolved
202
+ // through realpath because npm installs the bin as a symlink into .bin/,
203
+ // while import.meta.url already points at the real file.
204
+ function isEntrypoint() {
205
+ try {
206
+ return realpathSync(process.argv[1]) === fileURLToPath(import.meta.url);
207
+ } catch {
208
+ return false;
209
+ }
210
+ }
211
+
212
+ if (isEntrypoint()) {
213
+ main().catch(err => {
214
+ console.error('Error:', err.message);
215
+ process.exit(1);
216
+ });
217
+ }
package/src/filters.js CHANGED
@@ -4,14 +4,33 @@
4
4
  * Facts go here (deterministic field matches). Interpretations stay with the
5
5
  * caller — this module does substring matching on structured fields, nothing
6
6
  * semantic.
7
+ *
8
+ * Returns the page as an array. applyFiltersDetailed returns the same page
9
+ * plus total_matched, the match count before offset and limit.
7
10
  */
8
11
  export function applyFilters(jobs, options = {}) {
12
+ return applyFiltersDetailed(jobs, options).jobs;
13
+ }
14
+
15
+ /**
16
+ * Filter, sort, then page.
17
+ *
18
+ * Order is applied after the filters and before offset and limit, so a cut
19
+ * drops the oldest matches first. 'newest' sorts by postedAt descending with
20
+ * undated jobs last and ties broken by id, which keeps pages deterministic.
21
+ * 'board' keeps the order the adapter returned.
22
+ *
23
+ * @returns {{ jobs: Array, total_matched: number }}
24
+ */
25
+ export function applyFiltersDetailed(jobs, options = {}) {
9
26
  const {
10
27
  titleFilter,
11
28
  filter,
12
29
  postedWithinDays,
13
30
  locationIncludes,
14
31
  locationExcludes,
32
+ order = 'newest',
33
+ offset = 0,
15
34
  limit = 100,
16
35
  } = options;
17
36
 
@@ -42,25 +61,60 @@ export function applyFilters(jobs, options = {}) {
42
61
 
43
62
  if (Array.isArray(locationIncludes) && locationIncludes.length > 0) {
44
63
  const matchers = locationIncludes.map(makeLocationMatcher);
45
- result = result.filter(j => {
46
- const loc = (j.location || '').toLowerCase();
47
- return matchers.some(m => m(loc));
48
- });
64
+ result = result.filter(j => jobLocations(j).some(loc => matchers.some(m => m(loc))));
49
65
  }
50
66
 
51
67
  if (Array.isArray(locationExcludes) && locationExcludes.length > 0) {
52
68
  const matchers = locationExcludes.map(makeLocationMatcher);
53
- result = result.filter(j => {
54
- const loc = (j.location || '').toLowerCase();
55
- return !matchers.some(m => m(loc));
56
- });
69
+ result = result.filter(j => !jobLocations(j).every(loc => matchers.some(m => m(loc))));
57
70
  }
58
71
 
59
- if (typeof limit === 'number' && result.length > limit) {
60
- result = result.slice(0, limit);
72
+ const total_matched = result.length;
73
+
74
+ if (order !== 'board') {
75
+ result = [...result].sort(byNewest);
61
76
  }
62
77
 
63
- return result;
78
+ const start = typeof offset === 'number' && offset > 0 ? offset : 0;
79
+ const end = typeof limit === 'number' ? start + limit : undefined;
80
+ if (start > 0 || (end !== undefined && result.length > end)) {
81
+ result = result.slice(start, end);
82
+ }
83
+
84
+ return { jobs: result, total_matched };
85
+ }
86
+
87
+ /**
88
+ * Every location a job is open in, lowercased. A job passes an include when
89
+ * any of them matches and is dropped by an exclude only when all of them
90
+ * match: a role open in Berlin and New York is still open in Berlin for
91
+ * someone excluding the US (issue #68). Jobs from before `locations`
92
+ * existed fall back to the single `location` string.
93
+ */
94
+ function jobLocations(job) {
95
+ const list = Array.isArray(job.locations) && job.locations.length > 0
96
+ ? job.locations
97
+ : [job.location || ''];
98
+ return list.map(loc => String(loc).toLowerCase());
99
+ }
100
+
101
+ function postedTime(job) {
102
+ if (!job.postedAt) return null;
103
+ const t = new Date(job.postedAt).getTime();
104
+ return Number.isFinite(t) ? t : null;
105
+ }
106
+
107
+ function byNewest(a, b) {
108
+ const ta = postedTime(a);
109
+ const tb = postedTime(b);
110
+ if (ta !== tb) {
111
+ if (ta === null) return 1;
112
+ if (tb === null) return -1;
113
+ return tb - ta;
114
+ }
115
+ const ia = a.id || '';
116
+ const ib = b.id || '';
117
+ return ia < ib ? -1 : ia > ib ? 1 : 0;
64
118
  }
65
119
 
66
120
  /**
package/src/index.js CHANGED
@@ -8,11 +8,23 @@
8
8
 
9
9
  import { ADAPTERS, ATS_NAMES } from './adapters/index.js';
10
10
  import { loadRegistry, searchRegistry, detectAts, findAtsBySlug, findEntryBySlug, getRegistrySource } from './registry.js';
11
- import { applyFilters } from './filters.js';
11
+ import { applyFiltersDetailed } from './filters.js';
12
12
 
13
13
  /**
14
14
  * Fetch jobs from a company's ATS board.
15
15
  *
16
+ * Same options as fetchJobsDetailed; returns the page as an array.
17
+ *
18
+ * @returns {Promise<Array>} Normalized, filtered job objects
19
+ */
20
+ export async function fetchJobs(options = {}) {
21
+ const { jobs } = await fetchJobsDetailed(options);
22
+ return jobs;
23
+ }
24
+
25
+ /**
26
+ * Fetch jobs from a company's ATS board, with the match count.
27
+ *
16
28
  * @param {Object} options
17
29
  * @param {string} options.company - Company slug or name
18
30
  * @param {string} [options.ats] - Specific ATS platform. If omitted, auto-detects.
@@ -20,12 +32,14 @@ import { applyFilters } from './filters.js';
20
32
  * @param {string} [options.titleFilter] - Regex matched against title only. Use for role identity ("product manager", "staff engineer").
21
33
  * @param {string} [options.filter] - Regex matched across title, department, description. Use for topic/scope.
22
34
  * @param {number} [options.postedWithinDays] - Only return jobs posted within N days.
23
- * @param {string[]} [options.locationIncludes] - Keep jobs whose location contains any of these (case-insensitive).
24
- * @param {string[]} [options.locationExcludes] - Drop jobs whose location contains any of these (case-insensitive).
25
- * @param {number} [options.limit=100] - Maximum jobs to return after filtering.
26
- * @returns {Promise<Array>} Normalized, filtered job objects
35
+ * @param {string[]} [options.locationIncludes] - Keep jobs where any listed location contains any of these (case-insensitive).
36
+ * @param {string[]} [options.locationExcludes] - Drop jobs only when every listed location contains one of these (case-insensitive).
37
+ * @param {'newest'|'board'} [options.order='newest'] - 'newest': by postedAt descending, undated last, ties by id. 'board': the adapter's own order.
38
+ * @param {number} [options.offset=0] - Matches to skip after sorting (paging).
39
+ * @param {number} [options.limit=100] - Maximum jobs to return after offset.
40
+ * @returns {Promise<{ jobs: Array, total_matched: number }>} The page, plus the match count before offset and limit
27
41
  */
28
- export async function fetchJobs({
42
+ export async function fetchJobsDetailed({
29
43
  company,
30
44
  ats,
31
45
  config,
@@ -34,6 +48,8 @@ export async function fetchJobs({
34
48
  postedWithinDays,
35
49
  locationIncludes,
36
50
  locationExcludes,
51
+ order = 'newest',
52
+ offset = 0,
37
53
  limit = 100,
38
54
  } = {}) {
39
55
  if (!company) throw new Error('Company slug required');
@@ -45,7 +61,7 @@ export async function fetchJobs({
45
61
  // adapters declare fetch{Name}(slug) and ignore extra positional args
46
62
  // (JS no-op), so this is backward-compatible. Filter-aware adapters
47
63
  // (e.g. Workday) use it to avoid mass detail-fetching on huge tenants.
48
- const filterContext = { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes, limit };
64
+ const filterContext = { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes, offset, limit };
49
65
 
50
66
  let jobs;
51
67
  if (ats) {
@@ -93,7 +109,7 @@ export async function fetchJobs({
93
109
  }
94
110
  }
95
111
 
96
- return applyFilters(jobs, { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes, limit });
112
+ return applyFiltersDetailed(jobs, { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes, order, offset, limit });
97
113
  }
98
114
 
99
115
  /**
@@ -134,7 +150,7 @@ export { fetchLever } from './adapters/lever.js';
134
150
  export { fetchAshby } from './adapters/ashby.js';
135
151
 
136
152
  // Re-export filter logic for reuse (e.g., by the MCP server)
137
- export { applyFilters } from './filters.js';
153
+ export { applyFilters, applyFiltersDetailed } from './filters.js';
138
154
 
139
155
  // Re-export the list of supported ATS names (e.g. so the MCP layer can report
140
156
  // the full set detectAts probes, instead of hardcoding a stale subset).
package/src/normalizer.js CHANGED
@@ -13,20 +13,35 @@ export function jobId(company, title, ats, location = '') {
13
13
 
14
14
  /**
15
15
  * Normalize a raw ATS job object into the unified schema.
16
+ *
17
+ * Adapters pass `description` as HTML. This is the one place it is
18
+ * stripped and decoded (issue #66): a second pass would delete text the
19
+ * author escaped on purpose (`&lt;5 years`) and leave entities the first
20
+ * pass exposed (`&amp;mdash;` -> `&mdash;`) as literal noise.
21
+ *
22
+ * `raw.workplace` is the ATS's own arrangement, already mapped by the
23
+ * adapter to 'remote' | 'hybrid' | 'onsite', or null when the platform
24
+ * gives no signal. `raw.locations` lists every place the posting is open
25
+ * in; `location` stays the primary because it feeds the id (issue #68).
16
26
  */
17
27
  export function normalize(raw, ats) {
18
28
  const now = new Date().toISOString();
29
+ const description = stripHtml(raw.description || '');
30
+ const location = raw.location || '';
31
+ const workplace = resolveWorkplace(raw.workplace, location);
19
32
  return {
20
- id: jobId(raw.company || raw.companySlug, raw.title, ats, raw.location || ''),
33
+ id: jobId(raw.company || raw.companySlug, raw.title, ats, location),
21
34
  company: raw.company || raw.companySlug || '',
22
35
  companySlug: raw.companySlug || '',
23
36
  ats,
24
37
  title: raw.title || '',
25
38
  department: raw.department || '',
26
- location: raw.location || '',
27
- locationType: detectLocationType(raw.location || ''),
28
- salary: raw.salary || extractSalaryFromText(raw.description || ''),
29
- description: stripHtml(raw.description || ''),
39
+ location,
40
+ locations: uniqueLocations(location, raw.locations),
41
+ locationType: workplace.type,
42
+ workplace,
43
+ salary: raw.salary || extractSalaryFromText(description),
44
+ description,
30
45
  url: raw.url || '',
31
46
  postedAt: raw.postedAt || null,
32
47
  firstSeen: now,
@@ -36,62 +51,203 @@ export function normalize(raw, ats) {
36
51
  };
37
52
  }
38
53
 
54
+ const CURRENCY_CODES = 'USD|EUR|GBP|CAD|AUD|NZD|CHF|SEK|NOK|DKK|PLN|CZK|HUF|INR|SGD|HKD|JPY|CNY|BRL|MXN|ZAR|AED|ILS';
55
+ const SYMBOL_CURRENCY = { $: 'USD', '€': 'EUR', '£': 'GBP' };
56
+
57
+ // A number as job posts write it: 1,234,567 / 1.234.567 / 1234, with an
58
+ // optional one- or two-digit decimal part (211.4, 40.50, 60.000,50).
59
+ // Exactly three digits after a dot are a thousands group, the way Dutch
60
+ // and German boards write it: "€60.000" is sixty thousand, not sixty.
61
+ const NUMBER =
62
+ '\\d{1,3}(?:,\\d{3})+(?:\\.\\d{1,2})?' +
63
+ '|\\d{1,3}(?:\\.\\d{3})+(?:,\\d{1,2})?' +
64
+ '|\\d+(?:[.,]\\d{1,2})?';
65
+
66
+ // One side of a range: optional code before, optional symbol, the number,
67
+ // optional K, optional code after. The lookarounds keep the number from
68
+ // starting or ending inside a longer one ("234.567" out of "1.234.567").
69
+ const amountPattern = (p) =>
70
+ `(?:\\b(?<${p}CodeBefore>${CURRENCY_CODES})\\s?)?` +
71
+ `(?<${p}Sym>[$€£])?\\s?` +
72
+ `(?<![\\d.,])(?<${p}Num>${NUMBER})(?!\\d|[.,]\\d)\\s?` +
73
+ `(?<${p}K>[kK]\\b)?` +
74
+ `(?:\\s?(?<${p}CodeAfter>${CURRENCY_CODES})\\b)?`;
75
+
76
+ const SALARY_RANGE = new RegExp(
77
+ `${amountPattern('lo')}\\s*(?:[-–—]|\\bto\\b)\\s*${amountPattern('hi')}`,
78
+ 'g'
79
+ );
80
+
81
+ const HOUR_RE = /\b(?:per|an|each)\s+hour\b|\/\s*(?:hr|hour)\b|\bhourly\b/i;
82
+ const MONTH_RE = /\b(?:per|a|each)\s+month\b|\/\s*(?:mo|month)\b|\bmonthly\b/i;
83
+ const YEAR_RE = /\b(?:per|a|each)\s+(?:year|annum)\b|\/\s*(?:yr|year)\b|\b(?:annual(?:ly|ized)?|yearly)\b/i;
84
+
39
85
  /**
40
- * Detect location type from location string.
41
- */
42
- /**
43
- * Extract salary range from job description text.
44
- * Matches patterns like: $162,400 - $243,600 or $150K-$200K
86
+ * Extract a salary range from decoded job text.
87
+ *
88
+ * Accepts hyphen, en dash, em dash or "to" between the two amounts, an
89
+ * optional ISO currency code before, between or after them, `$` / EUR /
90
+ * GBP symbols, decimal K shorthand ($211.4K), and thousands grouped with
91
+ * either a comma or a dot (60,000 and 60.000 are both sixty thousand).
92
+ * A code wins over a symbol, so "$120,000 - $150,000 CAD" is CAD. Ranges
93
+ * with no currency marker at all (years, headcounts) are ignored.
94
+ *
95
+ * @returns {{min:number,max:number,currency:string,period:('year'|'month'|'hour'|null),source:'text'}|null}
45
96
  */
46
- function extractSalaryFromText(text) {
97
+ export function extractSalaryFromText(text) {
47
98
  if (!text) return null;
48
- // Match: $162,400 - $243,600 (full numbers)
49
- const fullMatch = text.match(/\$([\d,]+)\s*[-–]\s*\$([\d,]+)/);
50
- if (fullMatch) {
51
- return {
52
- min: parseInt(fullMatch[1].replace(/,/g, '')),
53
- max: parseInt(fullMatch[2].replace(/,/g, '')),
54
- currency: 'USD',
55
- };
56
- }
57
- // Match: $150K - $200K (shorthand)
58
- const kMatch = text.match(/\$(\d+)[kK]\s*[-–]\s*\$(\d+)[kK]/);
59
- if (kMatch) {
99
+ for (const m of text.matchAll(SALARY_RANGE)) {
100
+ const g = m.groups;
101
+ const end = m.index + m[0].length;
102
+ const after = text.slice(end, end + 40);
103
+ // "$20 - $30 million" is a revenue figure, not pay.
104
+ if (/^\s*(?:million|billion|m|bn?)\b/i.test(after)) continue;
105
+
106
+ const code = g.loCodeBefore || g.loCodeAfter || g.hiCodeBefore || g.hiCodeAfter;
107
+ const sym = g.loSym || g.hiSym;
108
+ if (!code && !sym) continue;
109
+
110
+ let min = parseAmount(g.loNum);
111
+ let max = parseAmount(g.hiNum);
112
+ if (g.loK || g.hiK) {
113
+ // "$150-200K" carries the K once for both sides.
114
+ if (min < 1000) min = Math.round(min * 1000);
115
+ if (max < 1000) max = Math.round(max * 1000);
116
+ }
117
+ if (!(min > 0) || !(max > 0)) continue;
118
+
119
+ const before = text.slice(Math.max(0, m.index - 40), m.index);
60
120
  return {
61
- min: parseInt(kMatch[1]) * 1000,
62
- max: parseInt(kMatch[2]) * 1000,
63
- currency: 'USD',
121
+ min,
122
+ max,
123
+ currency: (code || SYMBOL_CURRENCY[sym]).toUpperCase(),
124
+ period: detectPeriod(before, after, min),
125
+ source: 'text',
64
126
  };
65
127
  }
66
128
  return null;
67
129
  }
68
130
 
69
- function detectLocationType(location) {
70
- const lower = location.toLowerCase();
71
- if (/remote/i.test(lower)) return 'remote';
72
- if (/hybrid/i.test(lower)) return 'hybrid';
73
- if (/on-?site/i.test(lower)) return 'onsite';
74
- return location ? 'onsite' : 'unknown';
131
+ // Dots grouping thousands mean a comma is the decimal mark, and vice versa.
132
+ function parseAmount(s) {
133
+ if (/^\d{1,3}(?:\.\d{3})+/.test(s)) return Number(s.replace(/\./g, '').replace(',', '.'));
134
+ return Number(s.replace(/,(?=\d{3})/g, '').replace(',', '.'));
135
+ }
136
+
137
+ function detectPeriod(before, after, min) {
138
+ const explicit = periodWord(after) || periodWord(before);
139
+ if (explicit) return explicit;
140
+ return min >= 10000 ? 'year' : null;
141
+ }
142
+
143
+ function periodWord(text) {
144
+ if (HOUR_RE.test(text)) return 'hour';
145
+ if (MONTH_RE.test(text)) return 'month';
146
+ if (YEAR_RE.test(text)) return 'year';
147
+ return null;
148
+ }
149
+
150
+ const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
151
+
152
+ /**
153
+ * The platform's own value wins. Without one, a keyword in the location
154
+ * string is the next best signal. Without either the type is 'unknown':
155
+ * a city name alone does not say the role is onsite, and a guessed
156
+ * 'onsite' reads as a fact to whoever consumes it.
157
+ *
158
+ * @returns {{type:('remote'|'hybrid'|'onsite'|'unknown'), source:('ats'|'text'|null)}}
159
+ */
160
+ function resolveWorkplace(native, location) {
161
+ if (WORKPLACE_TYPES.has(native)) return { type: native, source: 'ats' };
162
+ const guessed = workplaceFromText(location);
163
+ if (guessed) return { type: guessed, source: 'text' };
164
+ return { type: 'unknown', source: null };
165
+ }
166
+
167
+ function workplaceFromText(location) {
168
+ const lower = (location || '').toLowerCase();
169
+ if (/remote/.test(lower)) return 'remote';
170
+ if (/hybrid/.test(lower)) return 'hybrid';
171
+ if (/on-?site/.test(lower)) return 'onsite';
172
+ return null;
173
+ }
174
+
175
+ function uniqueLocations(primary, extra) {
176
+ const out = [];
177
+ const seen = new Set();
178
+ for (const loc of [primary, ...(Array.isArray(extra) ? extra : [])]) {
179
+ const s = typeof loc === 'string' ? loc.trim() : '';
180
+ if (!s || seen.has(s.toLowerCase())) continue;
181
+ seen.add(s.toLowerCase());
182
+ out.push(s);
183
+ }
184
+ return out;
75
185
  }
76
186
 
77
187
  /**
78
188
  * Strip HTML tags and convert to clean text.
189
+ *
190
+ * Block closers become line breaks and list items become bullets before
191
+ * the remaining tags are removed. Entities are decoded LAST, so a literal
192
+ * `&lt;` in the source text never turns into a tag that gets stripped.
79
193
  */
80
194
  export function stripHtml(html) {
81
195
  if (!html) return '';
82
- return html
196
+ const text = html
83
197
  .replace(/<br\s*\/?>/gi, '\n')
84
- .replace(/<\/p>/gi, '\n\n')
85
- .replace(/<\/li>/gi, '\n')
86
- .replace(/<li>/gi, '- ')
87
- .replace(/<\/h[1-6]>/gi, '\n\n')
88
- .replace(/<h[1-6][^>]*>/gi, '## ')
89
- .replace(/<[^>]+>/g, '')
90
- .replace(/&amp;/g, '&')
91
- .replace(/&lt;/g, '<')
92
- .replace(/&gt;/g, '>')
93
- .replace(/&nbsp;/g, ' ')
94
- .replace(/&#\d+;/g, '')
198
+ .replace(/<\/(?:p|h[1-6])\s*>/gi, '\n\n')
199
+ .replace(/<\/(?:li|div|td|tr|ul|ol|table|section)\s*>/gi, '\n')
200
+ // Word-pasted markup (Lever lists, issue #64) opens a <p> inside each
201
+ // <li>; swallowing it keeps the item text on the bullet's line.
202
+ .replace(/<li\b[^>]*>\s*(?:<p\b[^>]*>\s*)?/gi, '- ')
203
+ .replace(/<h[1-6]\b[^>]*>/gi, '## ')
204
+ .replace(/<[^>]+>/g, '');
205
+ return decodeEntities(text)
206
+ .replace(/\u00a0/g, ' ')
95
207
  .replace(/\n{3,}/g, '\n\n')
96
208
  .trim();
97
209
  }
210
+
211
+ const NAMED_ENTITIES = {
212
+ lt: '<', gt: '>', quot: '"', apos: "'", nbsp: '\u00a0',
213
+ mdash: '—', ndash: '–', hellip: '…',
214
+ lsquo: '‘', rsquo: '’', ldquo: '“', rdquo: '”',
215
+ sbquo: '‚', bdquo: '„', laquo: '«', raquo: '»',
216
+ bull: '•', middot: '·', copy: '©', reg: '®', trade: '™',
217
+ deg: '°', times: '×', euro: '€', pound: '£', yen: '¥', cent: '¢',
218
+ agrave: 'à', aacute: 'á', acirc: 'â', atilde: 'ã', auml: 'ä', aring: 'å', aelig: 'æ',
219
+ ccedil: 'ç', egrave: 'è', eacute: 'é', ecirc: 'ê', euml: 'ë',
220
+ igrave: 'ì', iacute: 'í', icirc: 'î', iuml: 'ï', ntilde: 'ñ',
221
+ ograve: 'ò', oacute: 'ó', ocirc: 'ô', otilde: 'õ', ouml: 'ö', oslash: 'ø',
222
+ ugrave: 'ù', uacute: 'ú', ucirc: 'û', uuml: 'ü', yacute: 'ý', yuml: 'ÿ', szlig: 'ß',
223
+ Agrave: 'À', Aacute: 'Á', Acirc: 'Â', Atilde: 'Ã', Auml: 'Ä', Aring: 'Å', AElig: 'Æ',
224
+ Ccedil: 'Ç', Egrave: 'È', Eacute: 'É', Ecirc: 'Ê', Euml: 'Ë',
225
+ Igrave: 'Ì', Iacute: 'Í', Icirc: 'Î', Iuml: 'Ï', Ntilde: 'Ñ',
226
+ Ograve: 'Ò', Oacute: 'Ó', Ocirc: 'Ô', Otilde: 'Õ', Ouml: 'Ö', Oslash: 'Ø',
227
+ Ugrave: 'Ù', Uacute: 'Ú', Ucirc: 'Û', Uuml: 'Ü', Yacute: 'Ý',
228
+ };
229
+
230
+ /**
231
+ * Decode one layer of entity encoding (and unwrap CDATA) to real text.
232
+ *
233
+ * Decimal and hex references go through String.fromCodePoint, named
234
+ * references through the table above. `&amp;` is intentionally resolved
235
+ * LAST so double-encoded sequences (`&amp;mdash;`, `&amp;amp;`) collapse
236
+ * by exactly one layer per call. Used for the outer escaping Greenhouse
237
+ * and the Teamtailor RSS feed apply, and as stripHtml's final step.
238
+ */
239
+ export function decodeEntities(s) {
240
+ if (!s) return '';
241
+ return s
242
+ .replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1')
243
+ .replace(/&#(\d+);/g, (m, dec) => codePointToString(parseInt(dec, 10), m))
244
+ .replace(/&#[xX]([0-9a-fA-F]+);/g, (m, hex) => codePointToString(parseInt(hex, 16), m))
245
+ .replace(/&([A-Za-z][A-Za-z0-9]*);/g, (m, name) =>
246
+ (name !== 'amp' && Object.hasOwn(NAMED_ENTITIES, name)) ? NAMED_ENTITIES[name] : m)
247
+ .replace(/&amp;/g, '&');
248
+ }
249
+
250
+ function codePointToString(cp, fallback) {
251
+ if (!cp || cp > 0x10ffff || (cp >= 0xd800 && cp <= 0xdfff)) return fallback;
252
+ return String.fromCodePoint(cp);
253
+ }