jd-intel 0.8.3 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/normalizer.js CHANGED
@@ -13,20 +13,35 @@ export function jobId(company, title, ats, location = '') {
13
13
 
14
14
  /**
15
15
  * Normalize a raw ATS job object into the unified schema.
16
+ *
17
+ * Adapters pass `description` as HTML. This is the one place it is
18
+ * stripped and decoded (issue #66): a second pass would delete text the
19
+ * author escaped on purpose (`<5 years`) and leave entities the first
20
+ * pass exposed (`—` -> `—`) as literal noise.
21
+ *
22
+ * `raw.workplace` is the ATS's own arrangement, already mapped by the
23
+ * adapter to 'remote' | 'hybrid' | 'onsite', or null when the platform
24
+ * gives no signal. `raw.locations` lists every place the posting is open
25
+ * in; `location` stays the primary because it feeds the id (issue #68).
16
26
  */
17
27
  export function normalize(raw, ats) {
18
28
  const now = new Date().toISOString();
29
+ const description = stripHtml(raw.description || '');
30
+ const location = raw.location || '';
31
+ const workplace = resolveWorkplace(raw.workplace, location);
19
32
  return {
20
- id: jobId(raw.company || raw.companySlug, raw.title, ats, raw.location || ''),
33
+ id: jobId(raw.company || raw.companySlug, raw.title, ats, location),
21
34
  company: raw.company || raw.companySlug || '',
22
35
  companySlug: raw.companySlug || '',
23
36
  ats,
24
37
  title: raw.title || '',
25
38
  department: raw.department || '',
26
- location: raw.location || '',
27
- locationType: detectLocationType(raw.location || ''),
28
- salary: raw.salary || extractSalaryFromText(raw.description || ''),
29
- description: stripHtml(raw.description || ''),
39
+ location,
40
+ locations: uniqueLocations(location, raw.locations),
41
+ locationType: workplace.type,
42
+ workplace,
43
+ salary: raw.salary || extractSalaryFromText(description),
44
+ description,
30
45
  url: raw.url || '',
31
46
  postedAt: raw.postedAt || null,
32
47
  firstSeen: now,
@@ -36,62 +51,203 @@ export function normalize(raw, ats) {
36
51
  };
37
52
  }
38
53
 
54
+ const CURRENCY_CODES = 'USD|EUR|GBP|CAD|AUD|NZD|CHF|SEK|NOK|DKK|PLN|CZK|HUF|INR|SGD|HKD|JPY|CNY|BRL|MXN|ZAR|AED|ILS';
55
+ const SYMBOL_CURRENCY = { $: 'USD', '€': 'EUR', '£': 'GBP' };
56
+
57
+ // A number as job posts write it: 1,234,567 / 1.234.567 / 1234, with an
58
+ // optional one- or two-digit decimal part (211.4, 40.50, 60.000,50).
59
+ // Exactly three digits after a dot are a thousands group, the way Dutch
60
+ // and German boards write it: "€60.000" is sixty thousand, not sixty.
61
+ const NUMBER =
62
+ '\\d{1,3}(?:,\\d{3})+(?:\\.\\d{1,2})?' +
63
+ '|\\d{1,3}(?:\\.\\d{3})+(?:,\\d{1,2})?' +
64
+ '|\\d+(?:[.,]\\d{1,2})?';
65
+
66
+ // One side of a range: optional code before, optional symbol, the number,
67
+ // optional K, optional code after. The lookarounds keep the number from
68
+ // starting or ending inside a longer one ("234.567" out of "1.234.567").
69
+ const amountPattern = (p) =>
70
+ `(?:\\b(?<${p}CodeBefore>${CURRENCY_CODES})\\s?)?` +
71
+ `(?<${p}Sym>[$€£])?\\s?` +
72
+ `(?<![\\d.,])(?<${p}Num>${NUMBER})(?!\\d|[.,]\\d)\\s?` +
73
+ `(?<${p}K>[kK]\\b)?` +
74
+ `(?:\\s?(?<${p}CodeAfter>${CURRENCY_CODES})\\b)?`;
75
+
76
+ const SALARY_RANGE = new RegExp(
77
+ `${amountPattern('lo')}\\s*(?:[-–—]|\\bto\\b)\\s*${amountPattern('hi')}`,
78
+ 'g'
79
+ );
80
+
81
+ const HOUR_RE = /\b(?:per|an|each)\s+hour\b|\/\s*(?:hr|hour)\b|\bhourly\b/i;
82
+ const MONTH_RE = /\b(?:per|a|each)\s+month\b|\/\s*(?:mo|month)\b|\bmonthly\b/i;
83
+ const YEAR_RE = /\b(?:per|a|each)\s+(?:year|annum)\b|\/\s*(?:yr|year)\b|\b(?:annual(?:ly|ized)?|yearly)\b/i;
84
+
39
85
  /**
40
- * Detect location type from location string.
41
- */
42
- /**
43
- * Extract salary range from job description text.
44
- * Matches patterns like: $162,400 - $243,600 or $150K-$200K
86
+ * Extract a salary range from decoded job text.
87
+ *
88
+ * Accepts hyphen, en dash, em dash or "to" between the two amounts, an
89
+ * optional ISO currency code before, between or after them, `$` / EUR /
90
+ * GBP symbols, decimal K shorthand ($211.4K), and thousands grouped with
91
+ * either a comma or a dot (60,000 and 60.000 are both sixty thousand).
92
+ * A code wins over a symbol, so "$120,000 - $150,000 CAD" is CAD. Ranges
93
+ * with no currency marker at all (years, headcounts) are ignored.
94
+ *
95
+ * @returns {{min:number,max:number,currency:string,period:('year'|'month'|'hour'|null),source:'text'}|null}
45
96
  */
46
- function extractSalaryFromText(text) {
97
+ export function extractSalaryFromText(text) {
47
98
  if (!text) return null;
48
- // Match: $162,400 - $243,600 (full numbers)
49
- const fullMatch = text.match(/\$([\d,]+)\s*[-–]\s*\$([\d,]+)/);
50
- if (fullMatch) {
51
- return {
52
- min: parseInt(fullMatch[1].replace(/,/g, '')),
53
- max: parseInt(fullMatch[2].replace(/,/g, '')),
54
- currency: 'USD',
55
- };
56
- }
57
- // Match: $150K - $200K (shorthand)
58
- const kMatch = text.match(/\$(\d+)[kK]\s*[-–]\s*\$(\d+)[kK]/);
59
- if (kMatch) {
99
+ for (const m of text.matchAll(SALARY_RANGE)) {
100
+ const g = m.groups;
101
+ const end = m.index + m[0].length;
102
+ const after = text.slice(end, end + 40);
103
+ // "$20 - $30 million" is a revenue figure, not pay.
104
+ if (/^\s*(?:million|billion|m|bn?)\b/i.test(after)) continue;
105
+
106
+ const code = g.loCodeBefore || g.loCodeAfter || g.hiCodeBefore || g.hiCodeAfter;
107
+ const sym = g.loSym || g.hiSym;
108
+ if (!code && !sym) continue;
109
+
110
+ let min = parseAmount(g.loNum);
111
+ let max = parseAmount(g.hiNum);
112
+ if (g.loK || g.hiK) {
113
+ // "$150-200K" carries the K once for both sides.
114
+ if (min < 1000) min = Math.round(min * 1000);
115
+ if (max < 1000) max = Math.round(max * 1000);
116
+ }
117
+ if (!(min > 0) || !(max > 0)) continue;
118
+
119
+ const before = text.slice(Math.max(0, m.index - 40), m.index);
60
120
  return {
61
- min: parseInt(kMatch[1]) * 1000,
62
- max: parseInt(kMatch[2]) * 1000,
63
- currency: 'USD',
121
+ min,
122
+ max,
123
+ currency: (code || SYMBOL_CURRENCY[sym]).toUpperCase(),
124
+ period: detectPeriod(before, after, min),
125
+ source: 'text',
64
126
  };
65
127
  }
66
128
  return null;
67
129
  }
68
130
 
69
- function detectLocationType(location) {
70
- const lower = location.toLowerCase();
71
- if (/remote/i.test(lower)) return 'remote';
72
- if (/hybrid/i.test(lower)) return 'hybrid';
73
- if (/on-?site/i.test(lower)) return 'onsite';
74
- return location ? 'onsite' : 'unknown';
131
+ // Dots grouping thousands mean a comma is the decimal mark, and vice versa.
132
+ function parseAmount(s) {
133
+ if (/^\d{1,3}(?:\.\d{3})+/.test(s)) return Number(s.replace(/\./g, '').replace(',', '.'));
134
+ return Number(s.replace(/,(?=\d{3})/g, '').replace(',', '.'));
135
+ }
136
+
137
+ function detectPeriod(before, after, min) {
138
+ const explicit = periodWord(after) || periodWord(before);
139
+ if (explicit) return explicit;
140
+ return min >= 10000 ? 'year' : null;
141
+ }
142
+
143
+ function periodWord(text) {
144
+ if (HOUR_RE.test(text)) return 'hour';
145
+ if (MONTH_RE.test(text)) return 'month';
146
+ if (YEAR_RE.test(text)) return 'year';
147
+ return null;
148
+ }
149
+
150
+ const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
151
+
152
+ /**
153
+ * The platform's own value wins. Without one, a keyword in the location
154
+ * string is the next best signal. Without either the type is 'unknown':
155
+ * a city name alone does not say the role is onsite, and a guessed
156
+ * 'onsite' reads as a fact to whoever consumes it.
157
+ *
158
+ * @returns {{type:('remote'|'hybrid'|'onsite'|'unknown'), source:('ats'|'text'|null)}}
159
+ */
160
+ function resolveWorkplace(native, location) {
161
+ if (WORKPLACE_TYPES.has(native)) return { type: native, source: 'ats' };
162
+ const guessed = workplaceFromText(location);
163
+ if (guessed) return { type: guessed, source: 'text' };
164
+ return { type: 'unknown', source: null };
165
+ }
166
+
167
+ function workplaceFromText(location) {
168
+ const lower = (location || '').toLowerCase();
169
+ if (/remote/.test(lower)) return 'remote';
170
+ if (/hybrid/.test(lower)) return 'hybrid';
171
+ if (/on-?site/.test(lower)) return 'onsite';
172
+ return null;
173
+ }
174
+
175
+ function uniqueLocations(primary, extra) {
176
+ const out = [];
177
+ const seen = new Set();
178
+ for (const loc of [primary, ...(Array.isArray(extra) ? extra : [])]) {
179
+ const s = typeof loc === 'string' ? loc.trim() : '';
180
+ if (!s || seen.has(s.toLowerCase())) continue;
181
+ seen.add(s.toLowerCase());
182
+ out.push(s);
183
+ }
184
+ return out;
75
185
  }
76
186
 
77
187
  /**
78
188
  * Strip HTML tags and convert to clean text.
189
+ *
190
+ * Block closers become line breaks and list items become bullets before
191
+ * the remaining tags are removed. Entities are decoded LAST, so a literal
192
+ * `&lt;` in the source text never turns into a tag that gets stripped.
79
193
  */
80
194
  export function stripHtml(html) {
81
195
  if (!html) return '';
82
- return html
196
+ const text = html
83
197
  .replace(/<br\s*\/?>/gi, '\n')
84
- .replace(/<\/p>/gi, '\n\n')
85
- .replace(/<\/li>/gi, '\n')
86
- .replace(/<li>/gi, '- ')
87
- .replace(/<\/h[1-6]>/gi, '\n\n')
88
- .replace(/<h[1-6][^>]*>/gi, '## ')
89
- .replace(/<[^>]+>/g, '')
90
- .replace(/&amp;/g, '&')
91
- .replace(/&lt;/g, '<')
92
- .replace(/&gt;/g, '>')
93
- .replace(/&nbsp;/g, ' ')
94
- .replace(/&#\d+;/g, '')
198
+ .replace(/<\/(?:p|h[1-6])\s*>/gi, '\n\n')
199
+ .replace(/<\/(?:li|div|td|tr|ul|ol|table|section)\s*>/gi, '\n')
200
+ // Word-pasted markup (Lever lists, issue #64) opens a <p> inside each
201
+ // <li>; swallowing it keeps the item text on the bullet's line.
202
+ .replace(/<li\b[^>]*>\s*(?:<p\b[^>]*>\s*)?/gi, '- ')
203
+ .replace(/<h[1-6]\b[^>]*>/gi, '## ')
204
+ .replace(/<[^>]+>/g, '');
205
+ return decodeEntities(text)
206
+ .replace(/\u00a0/g, ' ')
95
207
  .replace(/\n{3,}/g, '\n\n')
96
208
  .trim();
97
209
  }
210
+
211
+ const NAMED_ENTITIES = {
212
+ lt: '<', gt: '>', quot: '"', apos: "'", nbsp: '\u00a0',
213
+ mdash: '—', ndash: '–', hellip: '…',
214
+ lsquo: '‘', rsquo: '’', ldquo: '“', rdquo: '”',
215
+ sbquo: '‚', bdquo: '„', laquo: '«', raquo: '»',
216
+ bull: '•', middot: '·', copy: '©', reg: '®', trade: '™',
217
+ deg: '°', times: '×', euro: '€', pound: '£', yen: '¥', cent: '¢',
218
+ agrave: 'à', aacute: 'á', acirc: 'â', atilde: 'ã', auml: 'ä', aring: 'å', aelig: 'æ',
219
+ ccedil: 'ç', egrave: 'è', eacute: 'é', ecirc: 'ê', euml: 'ë',
220
+ igrave: 'ì', iacute: 'í', icirc: 'î', iuml: 'ï', ntilde: 'ñ',
221
+ ograve: 'ò', oacute: 'ó', ocirc: 'ô', otilde: 'õ', ouml: 'ö', oslash: 'ø',
222
+ ugrave: 'ù', uacute: 'ú', ucirc: 'û', uuml: 'ü', yacute: 'ý', yuml: 'ÿ', szlig: 'ß',
223
+ Agrave: 'À', Aacute: 'Á', Acirc: 'Â', Atilde: 'Ã', Auml: 'Ä', Aring: 'Å', AElig: 'Æ',
224
+ Ccedil: 'Ç', Egrave: 'È', Eacute: 'É', Ecirc: 'Ê', Euml: 'Ë',
225
+ Igrave: 'Ì', Iacute: 'Í', Icirc: 'Î', Iuml: 'Ï', Ntilde: 'Ñ',
226
+ Ograve: 'Ò', Oacute: 'Ó', Ocirc: 'Ô', Otilde: 'Õ', Ouml: 'Ö', Oslash: 'Ø',
227
+ Ugrave: 'Ù', Uacute: 'Ú', Ucirc: 'Û', Uuml: 'Ü', Yacute: 'Ý',
228
+ };
229
+
230
+ /**
231
+ * Decode one layer of entity encoding (and unwrap CDATA) to real text.
232
+ *
233
+ * Decimal and hex references go through String.fromCodePoint, named
234
+ * references through the table above. `&amp;` is intentionally resolved
235
+ * LAST so double-encoded sequences (`&amp;mdash;`, `&amp;amp;`) collapse
236
+ * by exactly one layer per call. Used for the outer escaping Greenhouse
237
+ * and the Teamtailor RSS feed apply, and as stripHtml's final step.
238
+ */
239
+ export function decodeEntities(s) {
240
+ if (!s) return '';
241
+ return s
242
+ .replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1')
243
+ .replace(/&#(\d+);/g, (m, dec) => codePointToString(parseInt(dec, 10), m))
244
+ .replace(/&#[xX]([0-9a-fA-F]+);/g, (m, hex) => codePointToString(parseInt(hex, 16), m))
245
+ .replace(/&([A-Za-z][A-Za-z0-9]*);/g, (m, name) =>
246
+ (name !== 'amp' && Object.hasOwn(NAMED_ENTITIES, name)) ? NAMED_ENTITIES[name] : m)
247
+ .replace(/&amp;/g, '&');
248
+ }
249
+
250
+ function codePointToString(cp, fallback) {
251
+ if (!cp || cp > 0x10ffff || (cp >= 0xd800 && cp <= 0xdfff)) return fallback;
252
+ return String.fromCodePoint(cp);
253
+ }
package/src/registry.js CHANGED
@@ -1,10 +1,14 @@
1
1
  import { readFile } from 'node:fs/promises';
2
2
  import { join, dirname } from 'node:path';
3
3
  import { fileURLToPath } from 'node:url';
4
+ import { AtsError } from './errors.js';
4
5
 
5
6
  const __dirname = dirname(fileURLToPath(import.meta.url));
6
7
  const REGISTRY_DIR = join(__dirname, '..', 'registry');
7
8
 
9
+ // The one order the registry is ever walked in. Lookups, detectAts and the
10
+ // loaded object all follow it, so which file answers for a slug does not
11
+ // depend on which file's load finished first (issue #87).
8
12
  const PLATFORMS = ['greenhouse', 'lever', 'ashby', 'smartrecruiters', 'teamtailor', 'recruitee', 'workday'];
9
13
 
10
14
  // Network-first registry. A hosted copy lets installed bundles AND npx users
@@ -68,11 +72,8 @@ async function loadPlatform(platform) {
68
72
  */
69
73
  export async function loadRegistry(ats) {
70
74
  if (ats) return loadPlatform(ats);
71
- const all = {};
72
- await Promise.all(PLATFORMS.map(async (platform) => {
73
- all[platform] = await loadPlatform(platform);
74
- }));
75
- return all;
75
+ const lists = await Promise.all(PLATFORMS.map(loadPlatform));
76
+ return Object.fromEntries(PLATFORMS.map((platform, i) => [platform, lists[i]]));
76
77
  }
77
78
 
78
79
  /**
@@ -117,34 +118,43 @@ export async function searchRegistry(query) {
117
118
  // in each ATS's canonical form (SmartRecruiters uses PascalCase, e.g.
118
119
  // "Visa"), but callers pass a lowercased/alnum-stripped slug. Comparing
119
120
  // normalized forms keeps registry-first routing working for those.
120
- const normSlug = (s) => String(s).toLowerCase().replace(/[^a-z0-9]/g, '');
121
+ export const normSlug = (s) => String(s).toLowerCase().replace(/[^a-z0-9]/g, '');
122
+
123
+ // The loaded registry as [ats, companies] pairs in PLATFORMS order, whatever
124
+ // order the object's keys are in.
125
+ function platformEntries(all) {
126
+ return PLATFORMS.map(ats => [ats, all[ats] || []]);
127
+ }
128
+
129
+ function platformIndex(ats) {
130
+ const i = PLATFORMS.indexOf(ats);
131
+ return i === -1 ? PLATFORMS.length : i;
132
+ }
133
+
134
+ const byPlatform = (a, b) => platformIndex(a.ats) - platformIndex(b.ats);
121
135
 
122
136
  /**
123
137
  * Look up which ATS a slug belongs to in the registry.
124
138
  * Returns the ATS name (e.g., "greenhouse") or null if not in registry.
125
139
  */
126
140
  export async function findAtsBySlug(slug) {
127
- const all = await loadRegistry();
128
- const key = normSlug(slug);
129
- for (const [ats, companies] of Object.entries(all)) {
130
- if (companies.some(c => normSlug(c.slug) === key)) return ats;
131
- }
132
- return null;
141
+ const hit = await findEntryBySlug(slug);
142
+ return hit ? hit.ats : null;
133
143
  }
134
144
 
135
145
  /**
136
146
  * Look up the full registry entry for a slug, with its ATS.
137
147
  * Unlike findAtsBySlug (returns just the ats name), this returns the
138
148
  * whole entry so callers can read adapter-specific config (e.g. the
139
- * Workday {tenant, env, site} triple). Additive — does not change
140
- * findAtsBySlug, which has other callers.
149
+ * Workday {tenant, env, site} triple). The files are searched in
150
+ * PLATFORMS order, so the first match is the same on every call.
141
151
  *
142
152
  * @returns {Promise<{ats: string, entry: object}|null>}
143
153
  */
144
154
  export async function findEntryBySlug(slug) {
145
155
  const all = await loadRegistry();
146
156
  const key = normSlug(slug);
147
- for (const [ats, companies] of Object.entries(all)) {
157
+ for (const [ats, companies] of platformEntries(all)) {
148
158
  const entry = companies.find(c => normSlug(c.slug) === key);
149
159
  if (entry) return { ats, entry };
150
160
  }
@@ -152,18 +162,62 @@ export async function findEntryBySlug(slug) {
152
162
  }
153
163
 
154
164
  /**
155
- * Auto-detect which ATS a company uses.
165
+ * Where a company answers: the registry first, then a live probe of every
166
+ * adapter the registry did not already answer for.
167
+ *
168
+ * A slug the registry knows is listed with source 'registry' and not probed
169
+ * (Workday included, whose boards cannot be probed at all). Every remaining
170
+ * adapter's has() then runs: true adds a board with source 'probe', false
171
+ * adds nothing, and an AtsError (429, 5xx, 401, network) goes to `failed`
172
+ * with its code, so a board the probe could not check never reads as
173
+ * absent (issue #55). Any other error is a bug and is rethrown. Both lists
174
+ * come back in PLATFORMS order, never in completion order.
175
+ *
176
+ * @returns {Promise<{
177
+ * boards: Array<{ ats: string, slug: string, source: 'registry'|'probe' }>,
178
+ * failed: Array<{ ats: string, slug: string, code: string, message: string }>,
179
+ * }>}
156
180
  */
157
- export async function detectAts(companyName) {
181
+ export async function detectAtsDetailed(companyName) {
158
182
  const { ADAPTERS } = await import('./adapters/index.js');
159
- const slug = companyName.toLowerCase().replace(/[^a-z0-9]/g, '');
183
+ const slug = normSlug(companyName);
184
+ const all = await loadRegistry();
160
185
 
161
- const results = [];
162
- const checks = Object.entries(ADAPTERS).map(async ([ats, adapter]) => {
163
- const found = await adapter.has(slug);
164
- if (found) results.push({ ats, slug });
165
- });
186
+ const boards = [];
187
+ const failed = [];
188
+ const known = new Set();
189
+ for (const [ats, companies] of platformEntries(all)) {
190
+ const entry = companies.find(c => normSlug(c.slug) === slug);
191
+ if (entry) {
192
+ boards.push({ ats, slug: entry.slug, source: 'registry' });
193
+ known.add(ats);
194
+ }
195
+ }
166
196
 
167
- await Promise.allSettled(checks);
168
- return results;
197
+ const probes = Object.entries(ADAPTERS).filter(([ats]) => !known.has(ats));
198
+ const outcomes = await Promise.all(probes.map(async ([ats, adapter]) => {
199
+ try {
200
+ return { ats, found: await adapter.has(slug) };
201
+ } catch (err) {
202
+ if (!(err instanceof AtsError)) throw err;
203
+ return { ats, error: err };
204
+ }
205
+ }));
206
+ for (const { ats, found, error } of outcomes) {
207
+ if (error) failed.push({ ats, slug, code: error.code, message: error.message });
208
+ else if (found) boards.push({ ats, slug, source: 'probe' });
209
+ }
210
+
211
+ return { boards: boards.sort(byPlatform), failed: failed.sort(byPlatform) };
212
+ }
213
+
214
+ /**
215
+ * Auto-detect which ATS a company uses: the boards from detectAtsDetailed
216
+ * as [{ ats, slug }]. A failed probe never rejects this; the detailed
217
+ * variant is where those are reported. An adapter throwing anything but an
218
+ * AtsError is a bug and propagates, as it does through fetchJobs.
219
+ */
220
+ export async function detectAts(companyName) {
221
+ const { boards } = await detectAtsDetailed(companyName);
222
+ return boards.map(({ ats, slug }) => ({ ats, slug }));
169
223
  }