jd-intel 0.10.0 → 0.11.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -240,7 +240,8 @@ No custom parsing per company.
240
240
  | `locationType` | `remote`, `hybrid`, `onsite`, or `unknown` when neither the platform nor the location text says |
241
241
  | `workplace` | `{ type, source }`. `type` repeats `locationType`; `source` is `ats` when the platform stated it, `text` when read from the location string, null when unknown |
242
242
  | `salary` | Min-max range with `currency`, plus `period` (`year`, `month`, `hour`, or null) and `source` (`ats` when the platform supplied it, `text` when parsed from the posting). Null when nothing is stated |
243
- | `description` | Full JD in clean markdown |
243
+ | `description` | Full JD in clean markdown. Empty when `content.status` is `missing` |
244
+ | `content` | `{ status, reason }`. `complete` when the posting was read. `missing` when Workday or SmartRecruiters listed the job but its detail request failed (`reason`: `http_503`, `http_429`, `network_error`), so `description` and `salary` are unknown |
244
245
  | `url` | Direct link to the posting |
245
246
  | `postedAt` | Publication date (when provided) |
246
247
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "jd-intel",
3
- "version": "0.10.0",
3
+ "version": "0.11.1",
4
4
  "description": "Fetch and normalize job descriptions across seven major ATS (Greenhouse, Lever, Ashby, Workday, and more), for your AI assistant. No copy-paste.",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
@@ -15,7 +15,6 @@
15
15
  "scripts": {
16
16
  "test": "node --test test/*.test.js",
17
17
  "fetch": "node src/cli.js fetch",
18
- "search": "node src/cli.js search",
19
18
  "verify:registry": "node scripts/verify-registry.mjs",
20
19
  "sync:registry-pages": "node scripts/sync-pages-registry.mjs",
21
20
  "pack:mcpb": "node scripts/build-mcpb.mjs",
@@ -1,4 +1,4 @@
1
- import { normalize, extractSalaryFromText } from '../normalizer.js';
1
+ import { normalize, extractSalaryFromText, WORKPLACE_TYPES } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
  import { atsFetch, probeResult } from '../http.js';
4
4
 
@@ -14,11 +14,9 @@ const BOARD_URL = 'https://api.ashbyhq.com/posting-api/job-board';
14
14
  * into a silent empty result (issue #55).
15
15
  *
16
16
  * @param {string} slug - Company slug (e.g., 'notion', 'linear')
17
- * @param {object} [ctx] - { report }; report is called once with
18
- * { ats, org_name, org_url } when given
19
17
  * @returns {Promise<Array>} Normalized job objects
20
18
  */
21
- export async function fetchAshby(slug, ctx = {}) {
19
+ export async function fetchAshby(slug) {
22
20
  const url = `${BOARD_URL}/${slug}?includeCompensation=true`;
23
21
  const resp = await atsFetch(url);
24
22
 
@@ -31,10 +29,8 @@ export async function fetchAshby(slug, ctx = {}) {
31
29
  const jobs = data.jobs || [];
32
30
 
33
31
  // The REST response is { jobs, apiVersion }: no organization name, and
34
- // every link is on jobs.ashbyhq.com. Both null (issue #58).
35
- if (typeof ctx.report === 'function') {
36
- ctx.report({ ats: 'ashby', org_name: null, org_url: null });
37
- }
32
+ // every link is on jobs.ashbyhq.com. Nothing to report, so the board's
33
+ // org_name and org_url stay null (issue #58).
38
34
 
39
35
  return jobs.map(job => {
40
36
  const comp = job.compensation || {};
@@ -69,16 +65,14 @@ export async function fetchAshby(slug, ctx = {}) {
69
65
  });
70
66
  }
71
67
 
72
- const WORKPLACE_TYPES = { remote: 'remote', hybrid: 'hybrid', onsite: 'onsite' };
73
-
74
68
  /**
75
69
  * `workplaceType` is 'Remote', 'Hybrid' or 'OnSite'. `isRemote` is the
76
70
  * older flag and can only say remote, so it is the fallback when the type
77
71
  * is absent. false means nothing: the role may be hybrid or onsite.
78
72
  */
79
73
  function parseAshbyWorkplace(job) {
80
- const type = WORKPLACE_TYPES[String(job.workplaceType || '').toLowerCase()];
81
- if (type) return type;
74
+ const type = String(job.workplaceType || '').toLowerCase();
75
+ if (WORKPLACE_TYPES.has(type)) return type;
82
76
  return job.isRemote === true ? 'remote' : null;
83
77
  }
84
78
 
@@ -1,4 +1,4 @@
1
- import { normalize, decodeEntities } from '../normalizer.js';
1
+ import { normalize, decodeEntities, periodWord } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
  import { atsFetch, probeResult } from '../http.js';
4
4
 
@@ -11,11 +11,12 @@ const BASE_URL = 'https://boards-api.greenhouse.io/v1/boards';
11
11
  *
12
12
  * @param {string} slug - Company slug (e.g., 'stripe', 'notion')
13
13
  * @param {object} [ctx] - { report }; report is called once with
14
- * { ats, org_name, org_url } when given
14
+ * { org_name, org_url } when given
15
15
  * @returns {Promise<Array>} Normalized job objects
16
16
  */
17
17
  export async function fetchGreenhouse(slug, ctx = {}) {
18
- const url = `${BASE_URL}/${slug}/jobs?content=true`;
18
+ // pay_transparency adds pay_input_ranges to each row (issue #86).
19
+ const url = `${BASE_URL}/${slug}/jobs?content=true&pay_transparency=true`;
19
20
  const resp = await atsFetch(url);
20
21
 
21
22
  if (!resp.ok) {
@@ -31,35 +32,86 @@ export async function fetchGreenhouse(slug, ctx = {}) {
31
32
  // there is no company host to report (issue #58).
32
33
  if (typeof ctx.report === 'function') {
33
34
  ctx.report({
34
- ats: 'greenhouse',
35
35
  org_name: jobs.find(j => j.company_name)?.company_name || null,
36
36
  org_url: null,
37
37
  });
38
38
  }
39
39
 
40
- return jobs.map(job => normalize({
41
- companySlug: slug,
42
- company: data.name || slug,
43
- title: job.title || '',
44
- department: job.departments?.[0]?.name || '',
45
- location: job.location?.name || '',
46
- workplace: parseGreenhouseWorkplace(job.metadata),
47
- // `content` arrives HTML-escaped (`&lt;p&gt;`). Decode that outer layer
48
- // once so normalize() sees real tags; it strips and decodes the rest.
49
- description: decodeEntities(job.content || ''),
50
- url: job.absolute_url || '',
51
- // updated_at is an edit time that many boards bulk-refresh, so it is not
52
- // a posting date. first_published is. Fallback covers boards without it (#69).
53
- postedAt: job.first_published || job.updated_at || null,
54
- salary: null, // list endpoint has no structured pay; normalizer parses the pay transparency text
55
- metadata: {
56
- greenhouseId: job.id,
57
- internal_job_id: job.internal_job_id,
58
- departments: job.departments?.map(d => d.name) || [],
59
- offices: job.offices?.map(o => o.name) || [],
60
- updatedAt: job.updated_at,
61
- },
62
- }, 'greenhouse'));
40
+ return jobs.map(job => {
41
+ const payRanges = parsePayRanges(job.pay_input_ranges);
42
+ return normalize({
43
+ companySlug: slug,
44
+ company: data.name || slug,
45
+ title: job.title || '',
46
+ department: job.departments?.[0]?.name || '',
47
+ location: job.location?.name || '',
48
+ workplace: parseGreenhouseWorkplace(job.metadata),
49
+ // `content` arrives HTML-escaped (`&lt;p&gt;`). Decode that outer layer
50
+ // once so normalize() sees real tags; it strips and decodes the rest.
51
+ description: decodeEntities(job.content || ''),
52
+ url: job.absolute_url || '',
53
+ // updated_at is an edit time that many boards bulk-refresh, so it is not
54
+ // a posting date. first_published is. Fallback covers boards without it (#69).
55
+ postedAt: job.first_published || job.updated_at || null,
56
+ salary: salaryFromRanges(payRanges), // null without structured pay; normalize() then parses the text
57
+ metadata: {
58
+ greenhouseId: job.id,
59
+ internal_job_id: job.internal_job_id,
60
+ departments: job.departments?.map(d => d.name) || [],
61
+ offices: job.offices?.map(o => o.name) || [],
62
+ updatedAt: job.updated_at,
63
+ payRanges,
64
+ },
65
+ }, 'greenhouse');
66
+ });
67
+ }
68
+
69
+ /**
70
+ * `pay_input_ranges` is the board's pay transparency data:
71
+ * [{ min_cents, max_cents, currency_type, title, blurb }]. The blurb is
72
+ * boilerplate already rendered in the description, so it is dropped, and
73
+ * the platform can send the same entry twice, so entries are deduplicated.
74
+ * A board that does not publish pay sends an empty array or no key.
75
+ */
76
+ function parsePayRanges(ranges) {
77
+ const seen = new Set();
78
+ const out = [];
79
+ for (const r of Array.isArray(ranges) ? ranges : []) {
80
+ const range = {
81
+ title: r.title || '',
82
+ min: Number.isFinite(r.min_cents) ? r.min_cents / 100 : null,
83
+ max: Number.isFinite(r.max_cents) ? r.max_cents / 100 : null,
84
+ currency: r.currency_type || 'USD',
85
+ };
86
+ const key = JSON.stringify(range);
87
+ if ((range.min === null && range.max === null) || seen.has(key)) continue;
88
+ seen.add(key);
89
+ out.push(range);
90
+ }
91
+ return out;
92
+ }
93
+
94
+ /**
95
+ * One salary from the ranges. Ranges in one currency span (lowest min,
96
+ * highest max), as the Ashby adapter does for its tiers; with mixed
97
+ * currencies the first range stands alone. Every range stays in
98
+ * metadata.payRanges. The field has no period, so the period is the one
99
+ * the range titles state ("Annual", "Hourly") and null when they state
100
+ * none or disagree: an 'ats' value carries no guessed period.
101
+ */
102
+ function salaryFromRanges(ranges) {
103
+ if (ranges.length === 0) return null;
104
+ const used = ranges.every(r => r.currency === ranges[0].currency) ? ranges : [ranges[0]];
105
+ const mins = used.map(r => r.min).filter(v => v !== null);
106
+ const maxes = used.map(r => r.max).filter(v => v !== null);
107
+ const periods = new Set(used.map(r => periodWord(r.title)));
108
+ return {
109
+ min: mins.length ? Math.min(...mins) : null,
110
+ max: maxes.length ? Math.max(...maxes) : null,
111
+ currency: used[0].currency,
112
+ period: periods.size === 1 ? [...periods][0] : null,
113
+ source: 'ats',
114
+ };
63
115
  }
64
116
 
65
117
  /**
@@ -1,19 +1,29 @@
1
- export { fetchGreenhouse, hasGreenhouse } from './greenhouse.js';
2
- export { fetchLever, hasLever } from './lever.js';
3
- export { fetchAshby, hasAshby } from './ashby.js';
4
- export { fetchSmartrecruiters, hasSmartrecruiters } from './smartrecruiters.js';
5
- export { fetchTeamtailor, hasTeamtailor } from './teamtailor.js';
6
- export { fetchRecruitee, hasRecruitee } from './recruitee.js';
7
- export { fetchWorkday, hasWorkday } from './workday.js';
1
+ import { fetchGreenhouse, hasGreenhouse } from './greenhouse.js';
2
+ import { fetchLever, hasLever } from './lever.js';
3
+ import { fetchAshby, hasAshby } from './ashby.js';
4
+ import { fetchSmartrecruiters, hasSmartrecruiters } from './smartrecruiters.js';
5
+ import { fetchTeamtailor, hasTeamtailor } from './teamtailor.js';
6
+ import { fetchRecruitee, hasRecruitee } from './recruitee.js';
7
+ import { fetchWorkday, hasWorkday } from './workday.js';
8
+
9
+ export {
10
+ fetchGreenhouse, hasGreenhouse,
11
+ fetchLever, hasLever,
12
+ fetchAshby, hasAshby,
13
+ fetchSmartrecruiters, hasSmartrecruiters,
14
+ fetchTeamtailor, hasTeamtailor,
15
+ fetchRecruitee, hasRecruitee,
16
+ fetchWorkday, hasWorkday,
17
+ };
8
18
 
9
19
  export const ADAPTERS = {
10
- greenhouse: { fetch: (...args) => import('./greenhouse.js').then(m => m.fetchGreenhouse(...args)), has: (...args) => import('./greenhouse.js').then(m => m.hasGreenhouse(...args)) },
11
- lever: { fetch: (...args) => import('./lever.js').then(m => m.fetchLever(...args)), has: (...args) => import('./lever.js').then(m => m.hasLever(...args)) },
12
- ashby: { fetch: (...args) => import('./ashby.js').then(m => m.fetchAshby(...args)), has: (...args) => import('./ashby.js').then(m => m.hasAshby(...args)) },
13
- smartrecruiters: { fetch: (...args) => import('./smartrecruiters.js').then(m => m.fetchSmartrecruiters(...args)), has: (...args) => import('./smartrecruiters.js').then(m => m.hasSmartrecruiters(...args)) },
14
- teamtailor: { fetch: (...args) => import('./teamtailor.js').then(m => m.fetchTeamtailor(...args)), has: (...args) => import('./teamtailor.js').then(m => m.hasTeamtailor(...args)) },
15
- recruitee: { fetch: (...args) => import('./recruitee.js').then(m => m.fetchRecruitee(...args)), has: (...args) => import('./recruitee.js').then(m => m.hasRecruitee(...args)) },
16
- workday: { fetch: (...args) => import('./workday.js').then(m => m.fetchWorkday(...args)), has: (...args) => import('./workday.js').then(m => m.hasWorkday(...args)) },
20
+ greenhouse: { fetch: fetchGreenhouse, has: hasGreenhouse },
21
+ lever: { fetch: fetchLever, has: hasLever },
22
+ ashby: { fetch: fetchAshby, has: hasAshby },
23
+ smartrecruiters: { fetch: fetchSmartrecruiters, has: hasSmartrecruiters },
24
+ teamtailor: { fetch: fetchTeamtailor, has: hasTeamtailor },
25
+ recruitee: { fetch: fetchRecruitee, has: hasRecruitee },
26
+ workday: { fetch: fetchWorkday, has: hasWorkday },
17
27
  };
18
28
 
19
29
  export const ATS_NAMES = Object.keys(ADAPTERS);
@@ -1,12 +1,10 @@
1
- import { normalize, extractSalaryFromText } from '../normalizer.js';
1
+ import { normalize, extractSalaryFromText, toIso } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
  import { atsFetch, probeResult } from '../http.js';
4
4
 
5
5
  const BASE_URL = 'https://api.lever.co/v0/postings';
6
6
 
7
7
  const PERIODS = { 'per-year-salary': 'year', 'per-month-salary': 'month', 'per-hour-wage': 'hour' };
8
- // Lever's workplaceType is one of these or 'unspecified'.
9
- const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
10
8
 
11
9
  /**
12
10
  * Fetch all jobs from a Lever job board.
@@ -14,11 +12,9 @@ const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
14
12
  * Docs: https://github.com/lever/postings-api
15
13
  *
16
14
  * @param {string} slug - Company slug (e.g., 'stripe', 'figma')
17
- * @param {object} [ctx] - { report }; report is called once with
18
- * { ats, org_name, org_url } when given
19
15
  * @returns {Promise<Array>} Normalized job objects
20
16
  */
21
- export async function fetchLever(slug, ctx = {}) {
17
+ export async function fetchLever(slug) {
22
18
  const url = `${BASE_URL}/${slug}?mode=json`;
23
19
  const resp = await atsFetch(url);
24
20
 
@@ -31,10 +27,8 @@ export async function fetchLever(slug, ctx = {}) {
31
27
  if (!Array.isArray(jobs)) return [];
32
28
 
33
29
  // The postings response is a bare array of jobs: no organization name
34
- // anywhere, and every link is on jobs.lever.co. Both null (issue #58).
35
- if (typeof ctx.report === 'function') {
36
- ctx.report({ ats: 'lever', org_name: null, org_url: null });
37
- }
30
+ // anywhere, and every link is on jobs.lever.co. Nothing to report, so
31
+ // the board's org_name and org_url stay null (issue #58).
38
32
 
39
33
  return jobs.map(job => normalize({
40
34
  companySlug: slug,
@@ -46,10 +40,11 @@ export async function fetchLever(slug, ctx = {}) {
46
40
  department: job.categories?.department || job.categories?.team || '',
47
41
  location: job.categories?.location || '',
48
42
  locations: job.categories?.allLocations || [],
49
- workplace: WORKPLACE_TYPES.has(job.workplaceType) ? job.workplaceType : null,
43
+ // 'remote', 'hybrid', 'onsite' or 'unspecified'; normalize() ignores the last.
44
+ workplace: job.workplaceType,
50
45
  description: buildDescription(job),
51
46
  url: job.hostedUrl || '',
52
- postedAt: job.createdAt ? new Date(job.createdAt).toISOString() : null,
47
+ postedAt: toIso(job.createdAt),
53
48
  salary: parseLeverSalary(job.salaryRange, job.text),
54
49
  metadata: {
55
50
  leverId: job.id,
@@ -1,4 +1,4 @@
1
- import { normalize } from '../normalizer.js';
1
+ import { normalize, toIso } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
  import { atsFetch, probeResult } from '../http.js';
4
4
  import { orgHost } from '../boards.js';
@@ -21,7 +21,7 @@ import { orgHost } from '../boards.js';
21
21
  *
22
22
  * @param {string} slug - Recruitee company subdomain (e.g., 'vandebron')
23
23
  * @param {object} [ctx] - { report }; report is called once with
24
- * { ats, org_name, org_url } when given
24
+ * { org_name, org_url } when given
25
25
  * @returns {Promise<Array>} Normalized job objects
26
26
  */
27
27
  export async function fetchRecruitee(slug, ctx = {}) {
@@ -41,7 +41,6 @@ export async function fetchRecruitee(slug, ctx = {}) {
41
41
  // domain when the site has one and on {slug}.recruitee.com otherwise.
42
42
  if (typeof ctx.report === 'function') {
43
43
  ctx.report({
44
- ats: 'recruitee',
45
44
  org_name: offers.find(o => o.company_name)?.company_name || null,
46
45
  org_url: orgHost(offers.find(o => o.careers_url)?.careers_url),
47
46
  });
@@ -52,7 +51,7 @@ export async function fetchRecruitee(slug, ctx = {}) {
52
51
  let location = place;
53
52
  if (offer.remote) location = place ? `Remote - ${place}` : 'Remote';
54
53
 
55
- const createdAt = toIso(offer.created_at);
54
+ const createdAt = recruiteeDate(offer.created_at);
56
55
 
57
56
  return normalize({
58
57
  companySlug: slug,
@@ -66,7 +65,7 @@ export async function fetchRecruitee(slug, ctx = {}) {
66
65
  url: offer.careers_url || offer.careers_apply_url || '',
67
66
  // created_at can predate publication by years on long-lived offers,
68
67
  // so it is not a posting date. published_at is.
69
- postedAt: toIso(offer.published_at) || createdAt,
68
+ postedAt: recruiteeDate(offer.published_at) || createdAt,
70
69
  salary: parseRecruiteeSalary(offer.salary),
71
70
  metadata: {
72
71
  recruiteeId: offer.guid || offer.id,
@@ -78,14 +77,8 @@ export async function fetchRecruitee(slug, ctx = {}) {
78
77
  });
79
78
  }
80
79
 
81
- /**
82
- * Recruitee returns "2026-05-13 07:38:11 UTC"; coerce to ISO.
83
- */
84
- function toIso(ts) {
85
- if (!ts) return null;
86
- const d = new Date(ts.replace(' UTC', 'Z').replace(' ', 'T'));
87
- return Number.isNaN(d.getTime()) ? null : d.toISOString();
88
- }
80
+ // Recruitee returns "2026-05-13 07:38:11 UTC".
81
+ const recruiteeDate = (ts) => toIso(ts?.replace(' UTC', 'Z').replace(' ', 'T'));
89
82
 
90
83
  /**
91
84
  * Recruitee sends three booleans, not one enum. Hybrid wins when remote is
@@ -1,7 +1,7 @@
1
- import { normalize } from '../normalizer.js';
1
+ import { normalize, missingContent } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
  import { atsFetch, probeResult } from '../http.js';
4
- import { makeLocationMatcher } from '../filters.js';
4
+ import { prefilterRows } from '../filters.js';
5
5
 
6
6
  const BASE_URL = 'https://api.smartrecruiters.com/v1/companies';
7
7
  const PAGE_SIZE = 100;
@@ -21,12 +21,12 @@ const MAX_DETAIL_FETCHES = 100;
21
21
  * The list does carry name, location and releasedDate, so the same
22
22
  * pre-filter and detail budget Workday applies run here: list-evaluable
23
23
  * filters narrow the candidates, then at most MAX_DETAIL_FETCHES of them
24
- * are hydrated (see the budget note below). Without a filterContext the
24
+ * are hydrated (see prefilterRows). Without a filterContext the
25
25
  * cap still holds, so a direct call on a 400-posting tenant reads 100.
26
26
  *
27
27
  * @param {string} slug - SmartRecruiters company identifier (e.g., 'Visa')
28
28
  * @param {object} [ctx] - { filterContext, report }; report is called once
29
- * with { ats, listed, prefiltered, hydrated, capped, org_name, org_url }
29
+ * with { listed, prefiltered, hydrated, capped, org_name, org_url }
30
30
  * when given
31
31
  * @returns {Promise<Array>} Normalized job objects
32
32
  */
@@ -54,60 +54,28 @@ export async function fetchSmartrecruiters(slug, ctx = {}) {
54
54
  if (content.length === 0 || offset >= (data.totalFound || 0)) break;
55
55
  }
56
56
 
57
- // 2. Filter-aware candidate selection BEFORE the N+1 detail cost.
58
- // The list row carries name, location and releasedDate, and the
59
- // library re-applies every filter after this returns, so a keep here
60
- // is never final. The detail adds no location (unlike Workday's
61
- // additionalLocations), so a row with none follows the library's
62
- // rule now: out under includes, kept under excludes.
63
- let candidates = postings;
64
-
65
- if (fc.titleFilter) {
66
- const re = new RegExp(fc.titleFilter, 'i');
67
- candidates = candidates.filter(p => re.test(p.name || ''));
68
- }
69
- if (Array.isArray(fc.locationIncludes) && fc.locationIncludes.length > 0) {
70
- const matchers = fc.locationIncludes.map(makeLocationMatcher);
71
- candidates = candidates.filter(p => {
72
- const loc = listLocation(p).location.toLowerCase();
73
- return matchers.some(m => m(loc));
74
- });
75
- }
76
- if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
77
- const matchers = fc.locationExcludes.map(makeLocationMatcher);
78
- candidates = candidates.filter(p => {
79
- const loc = listLocation(p).location.toLowerCase();
80
- return !loc || !matchers.some(m => m(loc));
81
- });
82
- }
83
- if (typeof fc.postedWithinDays === 'number') {
84
- // postedAt comes from releasedDate alone, so the library's rule can
85
- // run here in full: a missing or unparseable date is out either way.
86
- const cutoff = Date.now() - fc.postedWithinDays * 86400000;
87
- candidates = candidates.filter(p => {
57
+ // 2. Filter-aware candidate selection BEFORE the N+1 detail cost, then
58
+ // the detail budget (see prefilterRows). The list row carries name,
59
+ // location and releasedDate. The detail adds no location (unlike
60
+ // Workday's additionalLocations), so a row with none follows the
61
+ // library's rule now: out under includes, kept under excludes.
62
+ // postedAt comes from releasedDate alone, so a missing or unparseable
63
+ // date is out, as it is in the library.
64
+ const { candidates, hydrate } = prefilterRows(postings, fc, {
65
+ title: p => p.name || '',
66
+ location: p => listLocation(p).location.toLowerCase(),
67
+ postedWithin: (p, days) => {
88
68
  const released = new Date(p.releasedDate || '').getTime();
89
- return Number.isFinite(released) && released >= cutoff;
90
- });
91
- }
92
-
93
- // 3. Bound the detail-fetch set, Workday's reasoning verbatim: a
94
- // description `filter` is applied by the library AFTER this returns,
95
- // so that case keeps the full backstop instead of truncating to
96
- // `limit` (which could hydrate jobs that all fail the regex while
97
- // better matches go unscanned). The library pages with `offset`
98
- // after this returns, so the budget covers the page plus what
99
- // precedes it. Candidates keep list order.
100
- const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
101
- const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
102
- const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
103
- const hydrate = candidates.slice(0, cap);
69
+ return Number.isFinite(released) && released >= Date.now() - days * 86400000;
70
+ },
71
+ max: MAX_DETAIL_FETCHES,
72
+ });
104
73
 
105
74
  // Every list row carries company { identifier, name }. Neither the list
106
75
  // nor the detail has a company website, and postingUrl is always on
107
76
  // jobs.smartrecruiters.com, so org_url stays null (issue #58).
108
77
  if (typeof ctx.report === 'function') {
109
78
  ctx.report({
110
- ats: 'smartrecruiters',
111
79
  listed: postings.length,
112
80
  prefiltered: candidates.length,
113
81
  hydrated: hydrate.length,
@@ -125,6 +93,7 @@ export async function fetchSmartrecruiters(slug, ctx = {}) {
125
93
  let sections = {};
126
94
  let postingUrl = '';
127
95
  let salary = null;
96
+ let content; // set only when the detail could not be read (issue #85)
128
97
 
129
98
  try {
130
99
  const detailResp = await atsFetch(`${BASE_URL}/${slug}/postings/${p.id}`);
@@ -133,10 +102,13 @@ export async function fetchSmartrecruiters(slug, ctx = {}) {
133
102
  sections = detail.jobAd?.sections || {};
134
103
  postingUrl = detail.postingUrl || detail.applyUrl || '';
135
104
  salary = parseCompensation(detail.compensation);
105
+ } else {
106
+ content = missingContent(detailResp);
136
107
  }
137
- } catch {
138
- // Detail fetch failed, retries included: fall back to list-only
139
- // fields (no description). Reporting this is #85.
108
+ } catch (err) {
109
+ // Detail fetch failed, retries included: list-only fields (no
110
+ // description, no url), marked missing.
111
+ content = missingContent(err);
140
112
  }
141
113
 
142
114
  const description = [
@@ -158,6 +130,7 @@ export async function fetchSmartrecruiters(slug, ctx = {}) {
158
130
  url: postingUrl,
159
131
  postedAt: p.releasedDate || null,
160
132
  salary, // null when the detail has no compensation; normalize() then parses text
133
+ content,
161
134
  metadata: {
162
135
  smartRecruitersId: p.id,
163
136
  refNumber: p.refNumber || '',
@@ -1,4 +1,4 @@
1
- import { normalize, decodeEntities } from '../normalizer.js';
1
+ import { normalize, decodeEntities, toIso } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
3
  import { atsFetch } from '../http.js';
4
4
  import { orgHost } from '../boards.js';
@@ -26,7 +26,7 @@ import { orgHost } from '../boards.js';
26
26
  *
27
27
  * @param {string} slug - TeamTailor career-site slug (e.g., 'tibber')
28
28
  * @param {object} [ctx] - { report }; report is called once with
29
- * { ats, org_name, org_url } when given
29
+ * { org_name, org_url } when given
30
30
  * @returns {Promise<Array>} Normalized job objects
31
31
  */
32
32
  // Most sites are {slug}.teamtailor.com, but some sit on a regional
@@ -85,7 +85,6 @@ export async function fetchTeamtailor(slug, ctx = {}) {
85
85
  || xml.match(/<channel>[\s\S]*?<link>([\s\S]*?)<\/link>/)?.[1]
86
86
  || '';
87
87
  ctx.report({
88
- ats: 'teamtailor',
89
88
  org_name: decodeEntities(channelTitle) || null,
90
89
  org_url: orgHost(link.trim()),
91
90
  });
@@ -117,12 +116,6 @@ export async function fetchTeamtailor(slug, ctx = {}) {
117
116
  location = location ? `Remote - ${location}` : 'Remote';
118
117
  }
119
118
 
120
- let postedAt = null;
121
- if (pubDateRaw) {
122
- const d = new Date(pubDateRaw);
123
- if (!Number.isNaN(d.getTime())) postedAt = d.toISOString();
124
- }
125
-
126
119
  return normalize({
127
120
  companySlug: slug,
128
121
  company,
@@ -133,7 +126,7 @@ export async function fetchTeamtailor(slug, ctx = {}) {
133
126
  workplace: REMOTE_STATUS[remoteStatus.toLowerCase()] || null,
134
127
  description: decodeEntities(pick('description')),
135
128
  url: link,
136
- postedAt,
129
+ postedAt: toIso(pubDateRaw),
137
130
  salary: null, // No structured salary; normalizer parses from text
138
131
  metadata: {
139
132
  teamtailorId: guid,
@@ -1,6 +1,6 @@
1
- import { normalize } from '../normalizer.js';
1
+ import { normalize, missingContent, toIso } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
- import { makeLocationMatcher } from '../filters.js';
3
+ import { prefilterRows } from '../filters.js';
4
4
  import { atsFetch } from '../http.js';
5
5
  import { orgHost } from '../boards.js';
6
6
 
@@ -35,7 +35,7 @@ const MULTI_LOCATION = /^\s*\d+\s+locations?\s*$/;
35
35
  * @param {string} slug - normalized company slug (registry routing key)
36
36
  * @param {object} [ctx] - { config:{tenant,env,site}, companyName, filterContext, report };
37
37
  * report is called once, after hydration, with
38
- * { ats, listed, prefiltered, hydrated, capped, org_name, org_url } when given
38
+ * { listed, prefiltered, hydrated, capped, org_name, org_url } when given
39
39
  * @returns {Promise<Array>} Normalized job objects
40
40
  */
41
41
  export async function fetchWorkday(slug, ctx = {}) {
@@ -92,55 +92,27 @@ export async function fetchWorkday(slug, ctx = {}) {
92
92
  if (firstTotal > 0 && offset >= firstTotal) break;
93
93
  }
94
94
 
95
- // 2. Filter-aware candidate selection BEFORE the N+1 detail cost.
96
- // The list carries title/locationsText/postedOn — enough to apply
97
- // titleFilter, location, and recency without descriptions.
98
- let candidates = postings;
99
-
100
- if (fc.titleFilter) {
101
- const re = new RegExp(fc.titleFilter, 'i');
102
- candidates = candidates.filter(p => re.test(p.title || ''));
103
- }
104
- // Location rows go through the applyFilters matcher, so the pre-filter
105
- // keeps exactly the rows the pass after hydration would (issue #61).
106
- // "2 Locations" says nothing about where: the row stays a candidate
107
- // through both filters and that later pass decides on the detail's
108
- // location list.
109
- if (Array.isArray(fc.locationIncludes) && fc.locationIncludes.length > 0) {
110
- const inc = fc.locationIncludes.map(makeLocationMatcher);
111
- candidates = candidates.filter(p => {
112
- const loc = (p.locationsText || '').toLowerCase();
113
- return MULTI_LOCATION.test(loc) || inc.some(m => m(loc));
114
- });
115
- }
116
- if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
117
- const exc = fc.locationExcludes.map(makeLocationMatcher);
118
- candidates = candidates.filter(p => {
95
+ // 2. Filter-aware candidate selection BEFORE the N+1 detail cost, then
96
+ // the detail budget (see prefilterRows). The list carries
97
+ // title/locationsText/postedOn, enough to apply titleFilter, location
98
+ // and recency without descriptions. "2 Locations" says nothing about
99
+ // where: the row stays a candidate through both location filters and
100
+ // the library's pass after hydration decides on the detail's location
101
+ // list (issue #61).
102
+ // NOTE: huge-tenant coverage is intentionally capped for v1. Two caps
103
+ // apply: the list scan above stops at LIST_PAGE_HARD_CAP pages (2000
104
+ // postings, enough for Salesforce's ~1398), and the detail set is cut
105
+ // to MAX_DETAIL_FETCHES here. Proper fix (smart pagination, surfaced
106
+ // truncation) is tracked in #26.
107
+ const { candidates, hydrate } = prefilterRows(postings, fc, {
108
+ title: p => p.title || '',
109
+ location: p => {
119
110
  const loc = (p.locationsText || '').toLowerCase();
120
- return MULTI_LOCATION.test(loc) || !exc.some(m => m(loc));
121
- });
122
- }
123
- if (typeof fc.postedWithinDays === 'number') {
124
- candidates = candidates.filter(p => withinDays(p.postedOn, fc.postedWithinDays));
125
- }
126
-
127
- // 3. Bound the detail-fetch set.
128
- // NOTE: huge-tenant coverage is intentionally capped for v1. Two
129
- // caps apply: the list scan above stops at LIST_PAGE_HARD_CAP pages
130
- // (2000 postings, enough for Salesforce's ~1398), and the detail set
131
- // is cut to MAX_DETAIL_FETCHES here. A description `filter` is
132
- // applied by the library AFTER this returns, so for that case we
133
- // keep the full backstop instead of truncating tightly to `limit`
134
- // (which could hydrate jobs that all fail the regex while better
135
- // matches go unscanned). The library pages with `offset` after this
136
- // returns, so the budget covers the page plus what precedes it.
137
- // Proper fix (smart pagination / rate-limited concurrency / surfaced
138
- // truncation) is tracked in #26, to be designed alongside
139
- // retry/rate-limit work (#7).
140
- const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
141
- const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
142
- const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
143
- const hydrate = candidates.slice(0, cap);
111
+ return MULTI_LOCATION.test(loc) ? null : loc;
112
+ },
113
+ postedWithin: (p, days) => withinDays(p.postedOn, days),
114
+ max: MAX_DETAIL_FETCHES,
115
+ });
144
116
 
145
117
  // 4. Hydrate descriptions via the per-posting detail endpoint. The detail
146
118
  // also carries `hiringOrganization: { name, url }` next to
@@ -151,6 +123,7 @@ export async function fetchWorkday(slug, ctx = {}) {
151
123
  const jobs = await Promise.all(hydrate.map(async (p, i) => {
152
124
  const externalPath = p.externalPath || ''; // already begins with '/job/...'
153
125
  let info = {};
126
+ let content; // set only when the detail could not be read (issue #85)
154
127
  try {
155
128
  // externalPath already carries the '/job/...' segment, so it is
156
129
  // concatenated directly onto the CXS base. Inserting another
@@ -160,9 +133,12 @@ export async function fetchWorkday(slug, ctx = {}) {
160
133
  const detail = await dResp.json();
161
134
  info = detail.jobPostingInfo || {};
162
135
  orgs[i] = detail.hiringOrganization || null;
136
+ } else {
137
+ content = missingContent(dResp);
163
138
  }
164
- } catch {
165
- // detail failed, retries included: fall back to list fields, empty description
139
+ } catch (err) {
140
+ // detail failed, retries included: list fields only, marked missing
141
+ content = missingContent(err);
166
142
  }
167
143
 
168
144
  return normalize({
@@ -175,8 +151,10 @@ export async function fetchWorkday(slug, ctx = {}) {
175
151
  workplace: parseWorkdayRemoteType(info.remoteType),
176
152
  description: info.jobDescription || '',
177
153
  url: `https://${tenant}.${env}.myworkdayjobs.com/${site}${externalPath}`,
178
- postedAt: parseWorkdayDate(info.startDate) || normalizePostedOn(p.postedOn),
154
+ // startDate is "2026-05-01" or "May 1, 2026".
155
+ postedAt: toIso(info.startDate) || normalizePostedOn(p.postedOn),
179
156
  salary: null, // normalizer extracts from description text
157
+ content,
180
158
  metadata: {
181
159
  workdayTenant: tenant,
182
160
  workdayEnv: env,
@@ -191,7 +169,6 @@ export async function fetchWorkday(slug, ctx = {}) {
191
169
  if (typeof ctx.report === 'function') {
192
170
  const org = orgs.find(Boolean) || {};
193
171
  ctx.report({
194
- ats: 'workday',
195
172
  listed: postings.length,
196
173
  prefiltered: candidates.length,
197
174
  hydrated: hydrate.length,
@@ -220,49 +197,37 @@ function parseWorkdayRemoteType(remoteType) {
220
197
 
221
198
  /**
222
199
  * Workday list `postedOn` is a relative string ("Posted Today",
223
- * "Posted 5 Days Ago", "Posted 30+ Days Ago"). Decide membership in
224
- * the last N days WITHOUT a network call. Unparseable -> keep (true);
225
- * the library re-filters authoritatively on the real postedAt after
226
- * hydration, so a false-keep here is corrected downstream.
200
+ * "Posted 5 Days Ago", "Posted 30+ Days Ago"). The days it names, or null
201
+ * when it names none.
227
202
  */
228
- function withinDays(postedOn, days) {
229
- if (!postedOn) return true;
230
- const s = String(postedOn).toLowerCase();
231
- if (/today/.test(s)) return days >= 0;
232
- if (/yesterday/.test(s)) return days >= 1;
203
+ function daysAgo(postedOn) {
204
+ const s = String(postedOn || '').toLowerCase();
205
+ if (/today/.test(s)) return 0;
206
+ if (/yesterday/.test(s)) return 1;
233
207
  const m = s.match(/(\d+)\+?\s*days?\s*ago/);
234
- if (m) return parseInt(m[1], 10) <= days;
235
- return true;
208
+ return m ? parseInt(m[1], 10) : null;
236
209
  }
237
210
 
238
211
  /**
239
- * Coerce a Workday list `postedOn` (relative) into an approx ISO date
240
- * so the library's postedWithinDays re-filter has a value to compare.
212
+ * Decide membership in the last N days WITHOUT a network call.
213
+ * Unparseable -> keep (true); the library re-filters authoritatively on
214
+ * the real postedAt after hydration, so a false-keep here is corrected
215
+ * downstream.
241
216
  */
242
- function normalizePostedOn(v) {
243
- if (!v) return null;
244
- const direct = new Date(v);
245
- if (Number.isFinite(direct.getTime())) return direct.toISOString();
246
- const s = String(v).toLowerCase();
247
- let daysAgo = null;
248
- if (/today/.test(s)) daysAgo = 0;
249
- else if (/yesterday/.test(s)) daysAgo = 1;
250
- else {
251
- const m = s.match(/(\d+)\+?\s*days?\s*ago/);
252
- if (m) daysAgo = parseInt(m[1], 10);
253
- }
254
- if (daysAgo === null) return null;
255
- return new Date(Date.now() - daysAgo * 86400000).toISOString();
217
+ function withinDays(postedOn, days) {
218
+ const n = daysAgo(postedOn);
219
+ return n === null || n <= days;
256
220
  }
257
221
 
258
222
  /**
259
- * Workday detail `startDate` ("2026-05-01" or "May 1, 2026"). Return
260
- * ISO, or null if unparseable.
223
+ * Coerce a Workday list `postedOn` (relative) into an approx ISO date
224
+ * so the library's postedWithinDays re-filter has a value to compare.
261
225
  */
262
- function parseWorkdayDate(s) {
263
- if (!s) return null;
264
- const d = new Date(s);
265
- return Number.isFinite(d.getTime()) ? d.toISOString() : null;
226
+ function normalizePostedOn(v) {
227
+ const direct = toIso(v);
228
+ if (direct) return direct;
229
+ const n = daysAgo(v);
230
+ return n === null ? null : new Date(Date.now() - n * 86400000).toISOString();
266
231
  }
267
232
 
268
233
  /**
package/src/cli.js CHANGED
@@ -11,6 +11,7 @@
11
11
 
12
12
  import { realpathSync } from 'node:fs';
13
13
  import { fileURLToPath } from 'node:url';
14
+ import { parseArgs } from 'node:util';
14
15
  import { fetchJobs } from './index.js';
15
16
  import { detectAtsDetailed, searchRegistry } from './registry.js';
16
17
 
@@ -19,30 +20,35 @@ const [,, command, ...args] = process.argv;
19
20
  async function main() {
20
21
  switch (command) {
21
22
  case 'fetch': {
22
- const company = args[0];
23
+ const string = { type: 'string' };
24
+ const { values: flags, positionals } = parseArgs({
25
+ args,
26
+ allowPositionals: true,
27
+ options: {
28
+ ats: string, 'title-filter': string, filter: string, 'posted-within-days': string,
29
+ 'location-include': string, 'location-exclude': string, limit: string,
30
+ 'workday-tenant': string, 'workday-env': string, 'workday-site': string,
31
+ json: { type: 'boolean' },
32
+ },
33
+ });
34
+ const company = positionals[0];
23
35
  if (!company) { console.error('Usage: jd-intel fetch <company> [--ats <platform>] (omit --ats to auto-detect; run "jd-intel" for the platform list)'); process.exit(1); }
24
- const getArg = (flag) => {
25
- const idx = args.indexOf(flag);
26
- return idx >= 0 ? args[idx + 1] : undefined;
27
- };
28
- let ats = getArg('--ats');
29
- const titleFilter = getArg('--title-filter');
30
- const filter = getArg('--filter');
31
- const postedWithinRaw = getArg('--posted-within-days');
32
- const postedWithinDays = postedWithinRaw !== undefined ? Number(postedWithinRaw) : undefined;
33
- const locIncludeRaw = getArg('--location-include');
34
- const locationIncludes = locIncludeRaw ? locIncludeRaw.split(',').map(s => s.trim()).filter(Boolean) : undefined;
35
- const locExcludeRaw = getArg('--location-exclude');
36
- const locationExcludes = locExcludeRaw ? locExcludeRaw.split(',').map(s => s.trim()).filter(Boolean) : undefined;
37
- const limitRaw = getArg('--limit');
38
- const limit = limitRaw !== undefined ? Number(limitRaw) : undefined;
36
+ const number = (v) => (v !== undefined ? Number(v) : undefined);
37
+ const list = (v) => (v ? v.split(',').map(s => s.trim()).filter(Boolean) : undefined);
38
+ let ats = flags.ats;
39
+ const titleFilter = flags['title-filter'];
40
+ const filter = flags.filter;
41
+ const postedWithinDays = number(flags['posted-within-days']);
42
+ const locationIncludes = list(flags['location-include']);
43
+ const locationExcludes = list(flags['location-exclude']);
44
+ const limit = number(flags.limit);
39
45
 
40
46
  // Workday is keyed by a {tenant, env, site} triple, not a slug.
41
47
  // Supplying it here makes a Workday board reachable without a
42
48
  // registry entry; presence of the flags infers --ats workday.
43
- const wdTenant = getArg('--workday-tenant');
44
- const wdEnv = getArg('--workday-env');
45
- const wdSite = getArg('--workday-site');
49
+ const wdTenant = flags['workday-tenant'];
50
+ const wdEnv = flags['workday-env'];
51
+ const wdSite = flags['workday-site'];
46
52
  let config;
47
53
  if (wdTenant || wdEnv || wdSite) {
48
54
  if (!wdTenant || !wdEnv || !wdSite) {
@@ -91,8 +97,10 @@ async function main() {
91
97
  const loc = job.location ? ` | ${job.location}` : '';
92
98
  const dept = job.department ? ` [${job.department}]` : '';
93
99
  console.log(` ${job.title}${dept}${loc}${salary}`);
94
- console.log(` ${job.url}`);
95
- if (job.description) {
100
+ console.log(` ${job.url || '(no URL: posting not read)'}`);
101
+ if (job.content?.status === 'missing') {
102
+ console.log(` posting not read: ${job.content.reason}`);
103
+ } else if (job.description) {
96
104
  const preview = job.description.substring(0, 120).replace(/\n/g, ' ');
97
105
  console.log(` ${preview}...`);
98
106
  }
@@ -103,7 +111,7 @@ async function main() {
103
111
  console.log(` ... and ${jobs.length - 20} more. Use --json for full output.`);
104
112
  }
105
113
 
106
- if (args.includes('--json')) {
114
+ if (flags.json) {
107
115
  console.log(JSON.stringify(jobs, null, 2));
108
116
  }
109
117
  break;
package/src/filters.js CHANGED
@@ -88,11 +88,7 @@ export function pageJobs(jobs, { order = 'newest', offset = 0, limit = 100 } = {
88
88
 
89
89
  const start = typeof offset === 'number' && offset > 0 ? offset : 0;
90
90
  const end = typeof limit === 'number' ? start + limit : undefined;
91
- if (start > 0 || (end !== undefined && result.length > end)) {
92
- result = result.slice(start, end);
93
- }
94
-
95
- return result;
91
+ return result.slice(start, end);
96
92
  }
97
93
 
98
94
  /**
@@ -174,3 +170,51 @@ export function makeLocationMatcher(needle) {
174
170
  }
175
171
  return (loc) => loc.includes(lower);
176
172
  }
173
+
174
+ /**
175
+ * The list pre-filter and detail budget the two-step adapters (Workday,
176
+ * SmartRecruiters) share: narrow the cheap list rows with the filters a row
177
+ * can answer, then bound how many get a detail request.
178
+ *
179
+ * The library re-applies every filter after hydration, so a keep here is
180
+ * never final. `location(row)` returns the lowercased location, or null
181
+ * when the row cannot say where it is (it then stays a candidate through
182
+ * both location filters). `postedWithin(row, days)` decides recency.
183
+ *
184
+ * A description `filter` runs only after hydration, so that case keeps the
185
+ * full `max` budget instead of truncating to the page (which could hydrate
186
+ * rows that all fail the regex while better matches go unscanned). Without
187
+ * one, the budget is the page plus the offset before it. List order is kept.
188
+ *
189
+ * @returns {{ candidates: Array, hydrate: Array }}
190
+ */
191
+ export function prefilterRows(rows, fc, { title, location, postedWithin, max }) {
192
+ let candidates = rows;
193
+
194
+ if (fc.titleFilter) {
195
+ const re = new RegExp(fc.titleFilter, 'i');
196
+ candidates = candidates.filter(p => re.test(title(p)));
197
+ }
198
+ if (Array.isArray(fc.locationIncludes) && fc.locationIncludes.length > 0) {
199
+ const inc = fc.locationIncludes.map(makeLocationMatcher);
200
+ candidates = candidates.filter(p => {
201
+ const loc = location(p);
202
+ return loc === null || inc.some(m => m(loc));
203
+ });
204
+ }
205
+ if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
206
+ const exc = fc.locationExcludes.map(makeLocationMatcher);
207
+ candidates = candidates.filter(p => {
208
+ const loc = location(p);
209
+ return loc === null || !exc.some(m => m(loc));
210
+ });
211
+ }
212
+ if (typeof fc.postedWithinDays === 'number') {
213
+ candidates = candidates.filter(p => postedWithin(p, fc.postedWithinDays));
214
+ }
215
+
216
+ const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
217
+ const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
218
+ const cap = fc.filter ? max : Math.min(skip + limit, max);
219
+ return { candidates, hydrate: candidates.slice(0, cap) };
220
+ }
package/src/index.js CHANGED
@@ -57,6 +57,7 @@ export async function fetchJobs(options = {}) {
57
57
  * jobs: Array,
58
58
  * total_matched: number,
59
59
  * total_before_filters: number,
60
+ * content_missing: number,
60
61
  * match: 'registry'|'probe'|'workday_override',
61
62
  * company: { key: string, name: string }|null,
62
63
  * boards: Array<object>,
@@ -65,6 +66,9 @@ export async function fetchJobs(options = {}) {
65
66
  * jobs: the page. total_matched: matches before offset and limit.
66
67
  * total_before_filters: rows every board listed before any filter (on
67
68
  * Workday and SmartRecruiters the list count, not the rows they hydrated).
69
+ * content_missing: jobs whose posting was not read because the detail
70
+ * request failed (Workday, SmartRecruiters), counted before the filters:
71
+ * a description filter cannot match text that never arrived.
68
72
  * match: how the company was resolved. company: the registry row's name
69
73
  * and its key (normalized name); null unless match is 'registry'.
70
74
  * boards: one entry per board that answered (see src/boards.js), with
@@ -194,6 +198,7 @@ export async function fetchJobsDetailed({
194
198
  jobs: pageJobs(matched, { order, offset, limit }),
195
199
  total_matched: matched.length,
196
200
  total_before_filters: boards.reduce((n, b) => n + b.jobs_found, 0),
201
+ content_missing: rows.filter(j => j.content?.status === 'missing').length,
197
202
  match,
198
203
  company: match === 'registry' ? { key: normSlug(targets[0].name), name: targets[0].name } : null,
199
204
  boards,
@@ -205,7 +210,13 @@ function registryTarget(hit, config) {
205
210
  return { ats: hit.ats, slug: hit.entry.slug, name: hit.entry.name, config: config || hit.entry.config };
206
211
  }
207
212
 
208
- function discoveryFailure(company, failed) {
213
+ /**
214
+ * The AtsError for a lookup where no board answered and at least one check
215
+ * failed: rate_limited when any failure was a 429, else ats_unreachable,
216
+ * with a message naming each failed check. `failed` is the list
217
+ * fetchJobsDetailed and detectAtsDetailed return.
218
+ */
219
+ export function discoveryFailure(company, failed) {
209
220
  const limited = failed.some(f => f.code === ERROR_CODES.RATE_LIMITED);
210
221
  const checks = failed.map(f => `${f.ats} (${f.message})`).join('; ');
211
222
  return new AtsError(
package/src/normalizer.js CHANGED
@@ -1,5 +1,15 @@
1
1
  import { createHash } from 'node:crypto';
2
2
 
3
+ /**
4
+ * A date value (string or epoch ms) as an ISO string, or null when it is
5
+ * missing or does not parse.
6
+ */
7
+ export function toIso(value) {
8
+ if (!value) return null;
9
+ const d = new Date(value);
10
+ return Number.isNaN(d.getTime()) ? null : d.toISOString();
11
+ }
12
+
3
13
  /**
4
14
  * Generate a stable ID for a job posting.
5
15
  */
@@ -23,6 +33,9 @@ export function jobId(company, title, ats, location = '') {
23
33
  * adapter to 'remote' | 'hybrid' | 'onsite', or null when the platform
24
34
  * gives no signal. `raw.locations` lists every place the posting is open
25
35
  * in; `location` stays the primary because it feeds the id (issue #68).
36
+ *
37
+ * `raw.content` is set only by a two-step adapter whose detail request
38
+ * failed (see missingContent). Every other job was read in full.
26
39
  */
27
40
  export function normalize(raw, ats) {
28
41
  const now = new Date().toISOString();
@@ -47,10 +60,21 @@ export function normalize(raw, ats) {
47
60
  firstSeen: now,
48
61
  lastSeen: now,
49
62
  status: 'open',
63
+ content: raw.content || { status: 'complete', reason: null },
50
64
  metadata: raw.metadata || {},
51
65
  };
52
66
  }
53
67
 
68
+ /**
69
+ * The `content` of a job whose detail request failed, so its description
70
+ * and pay were never read (issue #85). `failure` is the non-OK Response or
71
+ * the error atsFetch threw: an HTTP status gives "http_503", anything else
72
+ * "network_error".
73
+ */
74
+ export function missingContent(failure) {
75
+ return { status: 'missing', reason: failure?.status ? `http_${failure.status}` : 'network_error' };
76
+ }
77
+
54
78
  const CURRENCY_CODES = 'USD|EUR|GBP|CAD|AUD|NZD|CHF|SEK|NOK|DKK|PLN|CZK|HUF|INR|SGD|HKD|JPY|CNY|BRL|MXN|ZAR|AED|ILS';
55
79
  const SYMBOL_CURRENCY = { $: 'USD', '€': 'EUR', '£': 'GBP' };
56
80
 
@@ -140,14 +164,14 @@ function detectPeriod(before, after, min) {
140
164
  return min >= 10000 ? 'year' : null;
141
165
  }
142
166
 
143
- function periodWord(text) {
167
+ export function periodWord(text) {
144
168
  if (HOUR_RE.test(text)) return 'hour';
145
169
  if (MONTH_RE.test(text)) return 'month';
146
170
  if (YEAR_RE.test(text)) return 'year';
147
171
  return null;
148
172
  }
149
173
 
150
- const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
174
+ export const WORKPLACE_TYPES = new Set(['remote', 'hybrid', 'onsite']);
151
175
 
152
176
  /**
153
177
  * The platform's own value wins. Without one, a keyword in the location
package/src/registry.js CHANGED
@@ -2,15 +2,11 @@ import { readFile } from 'node:fs/promises';
2
2
  import { join, dirname } from 'node:path';
3
3
  import { fileURLToPath } from 'node:url';
4
4
  import { AtsError } from './errors.js';
5
+ import { ADAPTERS, ATS_NAMES } from './adapters/index.js';
5
6
 
6
7
  const __dirname = dirname(fileURLToPath(import.meta.url));
7
8
  const REGISTRY_DIR = join(__dirname, '..', 'registry');
8
9
 
9
- // The one order the registry is ever walked in. Lookups, detectAts and the
10
- // loaded object all follow it, so which file answers for a slug does not
11
- // depend on which file's load finished first (issue #87).
12
- const PLATFORMS = ['greenhouse', 'lever', 'ashby', 'smartrecruiters', 'teamtailor', 'recruitee', 'workday'];
13
-
14
10
  // Network-first registry. A hosted copy lets installed bundles AND npx users
15
11
  // pick up newly-added companies without reinstalling; the on-disk copy that
16
12
  // ships with the package is the guaranteed offline fallback. The base URL is
@@ -72,8 +68,11 @@ async function loadPlatform(platform) {
72
68
  */
73
69
  export async function loadRegistry(ats) {
74
70
  if (ats) return loadPlatform(ats);
75
- const lists = await Promise.all(PLATFORMS.map(loadPlatform));
76
- return Object.fromEntries(PLATFORMS.map((platform, i) => [platform, lists[i]]));
71
+ // ATS_NAMES is the one order the registry is ever walked in. Lookups,
72
+ // detectAts and this object all follow it, so which file answers for a
73
+ // slug does not depend on which file's load finished first (issue #87).
74
+ const lists = await Promise.all(ATS_NAMES.map(loadPlatform));
75
+ return Object.fromEntries(ATS_NAMES.map((platform, i) => [platform, lists[i]]));
77
76
  }
78
77
 
79
78
  /**
@@ -94,7 +93,10 @@ export function getRegistrySource() {
94
93
  }
95
94
 
96
95
  /**
97
- * Search registry for companies matching a query.
96
+ * Search registry for companies matching a query, best match first: an
97
+ * exact name or slug, then a name that starts with the query, then a name
98
+ * that contains it, then a sector-only match. Ties keep platform order, so
99
+ * a caller that cuts the list drops the weakest matches (issue #62).
98
100
  */
99
101
  export async function searchRegistry(query) {
100
102
  const all = await loadRegistry();
@@ -105,13 +107,16 @@ export async function searchRegistry(query) {
105
107
  for (const company of companies) {
106
108
  const name = (company.name || company.slug || '').toLowerCase();
107
109
  const sector = (company.sector || '').toLowerCase();
108
- if (name.includes(lower) || sector.includes(lower)) {
109
- results.push({ ...company, ats });
110
- }
110
+ const rank = name === lower || String(company.slug).toLowerCase() === lower ? 0
111
+ : name.startsWith(lower) ? 1
112
+ : name.includes(lower) ? 2
113
+ : sector.includes(lower) ? 3
114
+ : -1;
115
+ if (rank >= 0) results.push({ rank, row: { ...company, ats } });
111
116
  }
112
117
  }
113
118
 
114
- return results;
119
+ return results.sort((a, b) => a.rank - b.rank).map(r => r.row);
115
120
  }
116
121
 
117
122
  // Slug match is case/punctuation-insensitive: registry slugs are stored
@@ -120,18 +125,7 @@ export async function searchRegistry(query) {
120
125
  // normalized forms keeps registry-first routing working for those.
121
126
  export const normSlug = (s) => String(s).toLowerCase().replace(/[^a-z0-9]/g, '');
122
127
 
123
- // The loaded registry as [ats, companies] pairs in PLATFORMS order, whatever
124
- // order the object's keys are in.
125
- function platformEntries(all) {
126
- return PLATFORMS.map(ats => [ats, all[ats] || []]);
127
- }
128
-
129
- function platformIndex(ats) {
130
- const i = PLATFORMS.indexOf(ats);
131
- return i === -1 ? PLATFORMS.length : i;
132
- }
133
-
134
- const byPlatform = (a, b) => platformIndex(a.ats) - platformIndex(b.ats);
128
+ const byPlatform = (a, b) => ATS_NAMES.indexOf(a.ats) - ATS_NAMES.indexOf(b.ats);
135
129
 
136
130
  /**
137
131
  * Look up which ATS a slug belongs to in the registry.
@@ -147,14 +141,14 @@ export async function findAtsBySlug(slug) {
147
141
  * Unlike findAtsBySlug (returns just the ats name), this returns the
148
142
  * whole entry so callers can read adapter-specific config (e.g. the
149
143
  * Workday {tenant, env, site} triple). The files are searched in
150
- * PLATFORMS order, so the first match is the same on every call.
144
+ * ATS_NAMES order, so the first match is the same on every call.
151
145
  *
152
146
  * @returns {Promise<{ats: string, entry: object}|null>}
153
147
  */
154
148
  export async function findEntryBySlug(slug) {
155
149
  const all = await loadRegistry();
156
150
  const key = normSlug(slug);
157
- for (const [ats, companies] of platformEntries(all)) {
151
+ for (const [ats, companies] of Object.entries(all)) {
158
152
  const entry = companies.find(c => normSlug(c.slug) === key);
159
153
  if (entry) return { ats, entry };
160
154
  }
@@ -171,7 +165,7 @@ export async function findEntryBySlug(slug) {
171
165
  * adds nothing, and an AtsError (429, 5xx, 401, network) goes to `failed`
172
166
  * with its code, so a board the probe could not check never reads as
173
167
  * absent (issue #55). Any other error is a bug and is rethrown. Both lists
174
- * come back in PLATFORMS order, never in completion order.
168
+ * come back in ATS_NAMES order, never in completion order.
175
169
  *
176
170
  * @returns {Promise<{
177
171
  * boards: Array<{ ats: string, slug: string, source: 'registry'|'probe' }>,
@@ -179,14 +173,13 @@ export async function findEntryBySlug(slug) {
179
173
  * }>}
180
174
  */
181
175
  export async function detectAtsDetailed(companyName) {
182
- const { ADAPTERS } = await import('./adapters/index.js');
183
176
  const slug = normSlug(companyName);
184
177
  const all = await loadRegistry();
185
178
 
186
179
  const boards = [];
187
180
  const failed = [];
188
181
  const known = new Set();
189
- for (const [ats, companies] of platformEntries(all)) {
182
+ for (const [ats, companies] of Object.entries(all)) {
190
183
  const entry = companies.find(c => normSlug(c.slug) === slug);
191
184
  if (entry) {
192
185
  boards.push({ ats, slug: entry.slug, source: 'registry' });