jd-intel 0.8.3 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,9 +1,17 @@
1
- import { normalize, stripHtml } from '../normalizer.js';
1
+ import { normalize } from '../normalizer.js';
2
2
  import { atsErrorFromStatus } from '../errors.js';
3
+ import { makeLocationMatcher } from '../filters.js';
4
+ import { atsFetch } from '../http.js';
5
+ import { orgHost } from '../boards.js';
3
6
 
4
7
  const MAX_DETAIL_FETCHES = 100;
5
8
  const LIST_PAGE_SIZE = 20;
6
- const LIST_PAGE_HARD_CAP = 100; // <= 2000 list items scanned per request
9
+ // Upper bound on list pages per call: 100 pages of 20 = at most 2000
10
+ // postings scanned. Paging usually stops sooner, on a short page or when
11
+ // offset reaches the first page's total (see the loop below).
12
+ const LIST_PAGE_HARD_CAP = 100;
13
+ // A multi-location posting's list row reads "2 Locations", "14 Locations".
14
+ const MULTI_LOCATION = /^\s*\d+\s+locations?\s*$/;
7
15
 
8
16
  /**
9
17
  * Fetch jobs from a Workday tenant via the public "CXS" JSON API.
@@ -25,7 +33,9 @@ const LIST_PAGE_HARD_CAP = 100; // <= 2000 list items scanned per request
25
33
  * detail set.
26
34
  *
27
35
  * @param {string} slug - normalized company slug (registry routing key)
28
- * @param {object} [ctx] - { config:{tenant,env,site}, companyName, filterContext }
36
+ * @param {object} [ctx] - { config:{tenant,env,site}, companyName, filterContext, report };
37
+ * report is called once, after hydration, with
38
+ * { ats, listed, prefiltered, hydrated, capped, org_name, org_url } when given
29
39
  * @returns {Promise<Array>} Normalized job objects
30
40
  */
31
41
  export async function fetchWorkday(slug, ctx = {}) {
@@ -40,27 +50,46 @@ export async function fetchWorkday(slug, ctx = {}) {
40
50
  const postings = [];
41
51
  let offset = 0;
42
52
  let pages = 0;
43
- while (pages < LIST_PAGE_HARD_CAP) {
44
- const resp = await fetch(`${base}/jobs`, {
45
- method: 'POST',
46
- headers: { 'Content-Type': 'application/json' },
47
- body: JSON.stringify({ appliedFacets: {}, limit: LIST_PAGE_SIZE, offset, searchText: '' }),
48
- });
53
+ let firstTotal = 0;
54
+ let listCapped = false;
55
+ while (true) {
56
+ if (pages >= LIST_PAGE_HARD_CAP) {
57
+ listCapped = true;
58
+ break;
59
+ }
60
+ let resp;
61
+ try {
62
+ resp = await atsFetch(`${base}/jobs`, {
63
+ method: 'POST',
64
+ headers: { 'Content-Type': 'application/json' },
65
+ body: JSON.stringify({ appliedFacets: {}, limit: LIST_PAGE_SIZE, offset, searchText: '' }),
66
+ });
67
+ } catch (err) {
68
+ // A 429 or 5xx that outlasted the retries, or a network failure.
69
+ // After the first page, keep the postings already read.
70
+ if (offset === 0) throw err;
71
+ break;
72
+ }
49
73
 
50
74
  if (!resp.ok) {
51
- if (resp.status === 404) return []; // wrong site / no such board
52
75
  if (offset === 0) {
76
+ if (resp.status === 404) return []; // wrong site / no such board
53
77
  throw atsErrorFromStatus(resp.status, `Workday API error for ${slug} (${tenant}/${env}/${site}): ${resp.status}`);
54
78
  }
55
- break; // mid-paging failure: keep what we have
79
+ break; // mid-paging failure (any status): keep what we have
56
80
  }
57
81
 
58
82
  const data = await resp.json();
59
83
  const page = data.jobPostings || [];
84
+ // Some tenants report the real `total` only at offset 0 and send
85
+ // `total: 0` on every later page, so only the first page's figure
86
+ // is trusted. A short page is the other stop signal.
87
+ if (pages === 0) firstTotal = data.total || 0;
60
88
  postings.push(...page);
61
89
  pages += 1;
62
90
  offset += LIST_PAGE_SIZE;
63
- if (page.length === 0 || offset >= (data.total || 0)) break;
91
+ if (page.length < LIST_PAGE_SIZE) break;
92
+ if (firstTotal > 0 && offset >= firstTotal) break;
64
93
  }
65
94
 
66
95
  // 2. Filter-aware candidate selection BEFORE the N+1 detail cost.
@@ -72,18 +101,23 @@ export async function fetchWorkday(slug, ctx = {}) {
72
101
  const re = new RegExp(fc.titleFilter, 'i');
73
102
  candidates = candidates.filter(p => re.test(p.title || ''));
74
103
  }
104
+ // Location rows go through the applyFilters matcher, so the pre-filter
105
+ // keeps exactly the rows the pass after hydration would (issue #61).
106
+ // "2 Locations" says nothing about where: the row stays a candidate
107
+ // through both filters and that later pass decides on the detail's
108
+ // location list.
75
109
  if (Array.isArray(fc.locationIncludes) && fc.locationIncludes.length > 0) {
76
- const inc = fc.locationIncludes.map(s => String(s).toLowerCase());
110
+ const inc = fc.locationIncludes.map(makeLocationMatcher);
77
111
  candidates = candidates.filter(p => {
78
112
  const loc = (p.locationsText || '').toLowerCase();
79
- return inc.some(s => loc.includes(s));
113
+ return MULTI_LOCATION.test(loc) || inc.some(m => m(loc));
80
114
  });
81
115
  }
82
116
  if (Array.isArray(fc.locationExcludes) && fc.locationExcludes.length > 0) {
83
- const exc = fc.locationExcludes.map(s => String(s).toLowerCase());
117
+ const exc = fc.locationExcludes.map(makeLocationMatcher);
84
118
  candidates = candidates.filter(p => {
85
119
  const loc = (p.locationsText || '').toLowerCase();
86
- return !exc.some(s => loc.includes(s));
120
+ return MULTI_LOCATION.test(loc) || !exc.some(m => m(loc));
87
121
  });
88
122
  }
89
123
  if (typeof fc.postedWithinDays === 'number') {
@@ -91,33 +125,44 @@ export async function fetchWorkday(slug, ctx = {}) {
91
125
  }
92
126
 
93
127
  // 3. Bound the detail-fetch set.
94
- // NOTE: huge-tenant coverage is intentionally capped for v1
95
- // (Salesforce ~1398 postings). A description `filter` is applied
96
- // by the library AFTER this returns, so for that case we keep the
97
- // full backstop instead of truncating tightly to `limit` (which
98
- // could hydrate jobs that all fail the regex while better matches
99
- // go unscanned). Proper fix (smart pagination / rate-limited
100
- // concurrency / surfaced truncation) is tracked in #26, to be
101
- // designed alongside retry/rate-limit work (#7).
128
+ // NOTE: huge-tenant coverage is intentionally capped for v1. Two
129
+ // caps apply: the list scan above stops at LIST_PAGE_HARD_CAP pages
130
+ // (2000 postings, enough for Salesforce's ~1398), and the detail set
131
+ // is cut to MAX_DETAIL_FETCHES here. A description `filter` is
132
+ // applied by the library AFTER this returns, so for that case we
133
+ // keep the full backstop instead of truncating tightly to `limit`
134
+ // (which could hydrate jobs that all fail the regex while better
135
+ // matches go unscanned). The library pages with `offset` after this
136
+ // returns, so the budget covers the page plus what precedes it.
137
+ // Proper fix (smart pagination / rate-limited concurrency / surfaced
138
+ // truncation) is tracked in #26, to be designed alongside
139
+ // retry/rate-limit work (#7).
102
140
  const limit = typeof fc.limit === 'number' && fc.limit > 0 ? fc.limit : 100;
103
- const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(limit, MAX_DETAIL_FETCHES);
104
- candidates = candidates.slice(0, cap);
141
+ const skip = typeof fc.offset === 'number' && fc.offset > 0 ? fc.offset : 0;
142
+ const cap = fc.filter ? MAX_DETAIL_FETCHES : Math.min(skip + limit, MAX_DETAIL_FETCHES);
143
+ const hydrate = candidates.slice(0, cap);
105
144
 
106
- // 4. Hydrate descriptions via the per-posting detail endpoint.
107
- const jobs = await Promise.all(candidates.map(async (p) => {
145
+ // 4. Hydrate descriptions via the per-posting detail endpoint. The detail
146
+ // also carries `hiringOrganization: { name, url }` next to
147
+ // jobPostingInfo; the list does not. Kept per posting in list order so
148
+ // the one reported is the first hydrated posting's, not whichever
149
+ // detail answered first (a tenant can post under several entities).
150
+ const orgs = [];
151
+ const jobs = await Promise.all(hydrate.map(async (p, i) => {
108
152
  const externalPath = p.externalPath || ''; // already begins with '/job/...'
109
153
  let info = {};
110
154
  try {
111
155
  // externalPath already carries the '/job/...' segment, so it is
112
156
  // concatenated directly onto the CXS base. Inserting another
113
157
  // '/job' here yields '/job/job/...' which Workday rejects (422).
114
- const dResp = await fetch(`${base}${externalPath}`);
158
+ const dResp = await atsFetch(`${base}${externalPath}`);
115
159
  if (dResp.ok) {
116
160
  const detail = await dResp.json();
117
161
  info = detail.jobPostingInfo || {};
162
+ orgs[i] = detail.hiringOrganization || null;
118
163
  }
119
164
  } catch {
120
- // detail failed: fall back to list fields, empty description
165
+ // detail failed, retries included: fall back to list fields, empty description
121
166
  }
122
167
 
123
168
  return normalize({
@@ -126,7 +171,9 @@ export async function fetchWorkday(slug, ctx = {}) {
126
171
  title: p.title || info.title || '',
127
172
  department: '',
128
173
  location: info.location || p.locationsText || '',
129
- description: stripHtml(info.jobDescription || ''),
174
+ locations: info.additionalLocations || [],
175
+ workplace: parseWorkdayRemoteType(info.remoteType),
176
+ description: info.jobDescription || '',
130
177
  url: `https://${tenant}.${env}.myworkdayjobs.com/${site}${externalPath}`,
131
178
  postedAt: parseWorkdayDate(info.startDate) || normalizePostedOn(p.postedOn),
132
179
  salary: null, // normalizer extracts from description text
@@ -139,9 +186,38 @@ export async function fetchWorkday(slug, ctx = {}) {
139
186
  }, 'workday');
140
187
  }));
141
188
 
189
+ // Nothing hydrated (a filter miss, an empty site) means no detail was
190
+ // read, so the org is unknown rather than absent: null, null.
191
+ if (typeof ctx.report === 'function') {
192
+ const org = orgs.find(Boolean) || {};
193
+ ctx.report({
194
+ ats: 'workday',
195
+ listed: postings.length,
196
+ prefiltered: candidates.length,
197
+ hydrated: hydrate.length,
198
+ capped: listCapped || hydrate.length < candidates.length,
199
+ org_name: org.name || null,
200
+ org_url: orgHost(org.url),
201
+ });
202
+ }
203
+
142
204
  return jobs;
143
205
  }
144
206
 
207
+ /**
208
+ * Detail `remoteType` is free text set per tenant: "Remote", "Hybrid",
209
+ * "Office - Flexible", "On-site". A flexible office arrangement counts as
210
+ * hybrid, so that check runs before the office one. Some tenants send no
211
+ * value at all; the location string decides then.
212
+ */
213
+ function parseWorkdayRemoteType(remoteType) {
214
+ const s = String(remoteType || '').toLowerCase();
215
+ if (/remote/.test(s)) return 'remote';
216
+ if (/hybrid|flexible/.test(s)) return 'hybrid';
217
+ if (/office|on-?site/.test(s)) return 'onsite';
218
+ return null;
219
+ }
220
+
145
221
  /**
146
222
  * Workday list `postedOn` is a relative string ("Posted Today",
147
223
  * "Posted 5 Days Ago", "Posted 30+ Days Ago"). Decide membership in
package/src/boards.js ADDED
@@ -0,0 +1,80 @@
1
+ /**
2
+ * The boards[] entry of a fetchJobsDetailed result (issues #58, #60, #87).
3
+ *
4
+ * A board is one (ats, slug) the library fetched, with what came back. The
5
+ * fields fall in three groups: what the registry or the caller said about
6
+ * it (`name`, `site`), what the fetch found (`jobs_found`, `matched`,
7
+ * `scan`), and what the board says about itself (`org_name`, `org_url`).
8
+ * Only the adapters can read the third group, and they hand it over through
9
+ * ctx.report. Both fields are null where the platform exposes nothing, and
10
+ * they are never filled from the slug or the registry name: a slug is an
11
+ * address, not a confirmed identity.
12
+ */
13
+
14
+ const BOARD_URLS = {
15
+ greenhouse: (slug) => `https://boards.greenhouse.io/${slug}`,
16
+ lever: (slug) => `https://jobs.lever.co/${slug}`,
17
+ ashby: (slug) => `https://jobs.ashbyhq.com/${slug}`,
18
+ smartrecruiters: (slug) => `https://careers.smartrecruiters.com/${slug}`,
19
+ teamtailor: (slug) => `https://${slug}.teamtailor.com`,
20
+ recruitee: (slug) => `https://${slug}.recruitee.com`,
21
+ workday: (slug, config) => (config ? `https://${config.tenant}.${config.env}.myworkdayjobs.com/${config.site}` : null),
22
+ };
23
+
24
+ // Domains the platforms own. A link there (boards.greenhouse.io,
25
+ // jobs.lever.co, testco.recruitee.com, cisco.wd5.myworkdayjobs.com) says
26
+ // which ATS hosts the board, nothing about whose board it is.
27
+ const ATS_DOMAINS = ['greenhouse.io', 'lever.co', 'ashbyhq.com', 'smartrecruiters.com', 'teamtailor.com', 'recruitee.com', 'myworkdayjobs.com'];
28
+
29
+ /**
30
+ * The page a person opens to see the board, or null when the ATS is unknown
31
+ * or, for Workday, no {tenant, env, site} is at hand.
32
+ */
33
+ export function boardUrl(ats, slug, config) {
34
+ const build = BOARD_URLS[ats];
35
+ return build ? build(slug, config) : null;
36
+ }
37
+
38
+ /**
39
+ * The bare hostname a board's link points at ("jobs.example.com"), for
40
+ * org_url. Null when the link is missing or malformed, and null when the
41
+ * host belongs to an ATS, since that carries no signal about the company.
42
+ */
43
+ export function orgHost(link) {
44
+ let host;
45
+ try {
46
+ host = new URL(link).hostname.toLowerCase();
47
+ } catch {
48
+ return null;
49
+ }
50
+ if (!host || ATS_DOMAINS.some(d => host === d || host.endsWith(`.${d}`))) return null;
51
+ return host;
52
+ }
53
+
54
+ /**
55
+ * @param {object} board
56
+ * @param {string} board.ats
57
+ * @param {string} board.slug - The slug the adapter was called with (canonical casing on a registry hit)
58
+ * @param {string|null} board.name - Registry row name; null for a probe or an override
59
+ * @param {object} [board.config] - Workday {tenant, env, site}, when one was used
60
+ * @param {string|null} board.org_name - The organization name the ATS response states, else null
61
+ * @param {string|null} board.org_url - The careers or company host the board links to (see orgHost), else null
62
+ * @param {number} board.jobs_found - Rows the board listed before any filter: the list count an adapter reported through ctx.report when it filters before hydrating, else the rows it returned
63
+ * @param {number} board.matched - Rows left after filters, before offset and limit
64
+ * @param {object|null} board.scan - The { listed, prefiltered, hydrated, capped } counts the adapter reported through ctx.report, else null
65
+ */
66
+ export function describeBoard({ ats, slug, name = null, config, org_name = null, org_url = null, jobs_found, matched = 0, scan = null }) {
67
+ return {
68
+ ats,
69
+ slug,
70
+ name,
71
+ site: ats === 'workday' && config ? config.site : null,
72
+ board_url: boardUrl(ats, slug, config),
73
+ org_name,
74
+ org_url,
75
+ jobs_found,
76
+ matched,
77
+ selected: true,
78
+ scan,
79
+ };
80
+ }
package/src/cli.js CHANGED
@@ -9,8 +9,10 @@
9
9
  * jd-intel registry search <query>
10
10
  */
11
11
 
12
+ import { realpathSync } from 'node:fs';
13
+ import { fileURLToPath } from 'node:url';
12
14
  import { fetchJobs } from './index.js';
13
- import { detectAts, searchRegistry } from './registry.js';
15
+ import { detectAtsDetailed, searchRegistry } from './registry.js';
14
16
 
15
17
  const [,, command, ...args] = process.argv;
16
18
 
@@ -85,7 +87,7 @@ async function main() {
85
87
  console.log(`Found ${jobs.length} jobs\n`);
86
88
 
87
89
  for (const job of jobs.slice(0, 20)) {
88
- const salary = job.salary ? ` | $${job.salary.min?.toLocaleString()}-$${job.salary.max?.toLocaleString()}` : '';
90
+ const salary = job.salary ? ` | ${formatSalary(job.salary)}` : '';
89
91
  const loc = job.location ? ` | ${job.location}` : '';
90
92
  const dept = job.department ? ` [${job.department}]` : '';
91
93
  console.log(` ${job.title}${dept}${loc}${salary}`);
@@ -111,13 +113,17 @@ async function main() {
111
113
  const company = args[0];
112
114
  if (!company) { console.error('Usage: jd-intel detect <company>'); process.exit(1); }
113
115
  console.log(`Detecting ATS for ${company}...`);
114
- const results = await detectAts(company);
115
- if (results.length === 0) {
116
- console.log('No ATS board found for this company.');
117
- } else {
118
- for (const r of results) {
119
- console.log(` Found: ${r.ats} (slug: ${r.slug})`);
120
- }
116
+ const { boards, failed } = await detectAtsDetailed(company);
117
+ for (const b of boards) {
118
+ console.log(` Found: ${b.ats} (slug: ${b.slug}, ${b.source === 'registry' ? 'in the registry' : 'live probe'})`);
119
+ }
120
+ for (const f of failed) {
121
+ console.log(` Could not check ${f.ats}: ${f.message}`);
122
+ }
123
+ if (boards.length === 0) {
124
+ console.log(failed.length > 0
125
+ ? 'No ATS board confirmed. At least one check failed, so this is not a definite miss. Retry in a moment.'
126
+ : 'No ATS board found for this company.');
121
127
  }
122
128
  break;
123
129
  }
@@ -162,8 +168,8 @@ Fetch options:
162
168
  --title-filter pattern Regex matched against TITLE only (role identity)
163
169
  --filter pattern Regex matched across title, department, description (topic/scope)
164
170
  --posted-within-days N Only jobs posted in the last N days
165
- --location-include "A,B,C" Keep jobs whose location contains any of these
166
- --location-exclude "A,B,C" Drop jobs whose location contains any of these
171
+ --location-include "A,B,C" Keep jobs where any listed location contains one of these
172
+ --location-exclude "A,B,C" Drop jobs only when every listed location contains one of these
167
173
  --limit N Cap results (default 100)
168
174
  --json Output full JSON
169
175
 
@@ -184,7 +190,32 @@ Examples:
184
190
  }
185
191
  }
186
192
 
187
- main().catch(err => {
188
- console.error('Error:', err.message);
189
- process.exit(1);
190
- });
193
+ export function formatSalary({ min, max, currency, period }) {
194
+ const hasMin = min != null;
195
+ const hasMax = max != null;
196
+ let range;
197
+ if (hasMin && hasMax) range = `${min.toLocaleString()}-${max.toLocaleString()}`;
198
+ else if (hasMin) range = `from ${min.toLocaleString()}`;
199
+ else range = `up to ${max.toLocaleString()}`;
200
+ const unit = period === 'hour' ? '/hr' : period === 'month' ? '/mo' : '';
201
+ return `${range} ${currency}${unit}`;
202
+ }
203
+
204
+ // Boot only when this file is the script Node was started with, so a test
205
+ // can import formatSalary without running a command. argv[1] is resolved
206
+ // through realpath because npm installs the bin as a symlink into .bin/,
207
+ // while import.meta.url already points at the real file.
208
+ function isEntrypoint() {
209
+ try {
210
+ return realpathSync(process.argv[1]) === fileURLToPath(import.meta.url);
211
+ } catch {
212
+ return false;
213
+ }
214
+ }
215
+
216
+ if (isEntrypoint()) {
217
+ main().catch(err => {
218
+ console.error('Error:', err.message);
219
+ process.exit(1);
220
+ });
221
+ }
package/src/errors.js CHANGED
@@ -34,6 +34,20 @@ export class AtsError extends Error {
34
34
  }
35
35
  }
36
36
 
37
+ /**
38
+ * Thrown when a call cannot proceed because of its arguments: a missing
39
+ * company, an unknown ATS name, a regex that does not compile. Carries
40
+ * `code: 'invalid_args'` so callers tell a bad request from a failed fetch
41
+ * (AtsError) without reading the message.
42
+ */
43
+ export class ArgumentError extends Error {
44
+ constructor(message) {
45
+ super(message);
46
+ this.name = 'ArgumentError';
47
+ this.code = ERROR_CODES.INVALID_ARGS;
48
+ }
49
+ }
50
+
37
51
  /**
38
52
  * Helper for adapters: build an AtsError from an HTTP status (429 => rate
39
53
  * limited, anything else => unreachable) with the given message.
package/src/filters.js CHANGED
@@ -1,33 +1,56 @@
1
+ import { ArgumentError } from './errors.js';
2
+
1
3
  /**
2
4
  * Apply filters to a list of normalized jobs.
3
5
  *
4
6
  * Facts go here (deterministic field matches). Interpretations stay with the
5
7
  * caller — this module does substring matching on structured fields, nothing
6
8
  * semantic.
9
+ *
10
+ * Returns the page as an array. applyFiltersDetailed returns the same page
11
+ * plus total_matched, the match count before offset and limit.
7
12
  */
8
13
  export function applyFilters(jobs, options = {}) {
9
- const {
10
- titleFilter,
11
- filter,
12
- postedWithinDays,
13
- locationIncludes,
14
- locationExcludes,
15
- limit = 100,
16
- } = options;
14
+ return applyFiltersDetailed(jobs, options).jobs;
15
+ }
16
+
17
+ /**
18
+ * Filter, sort, then page.
19
+ *
20
+ * Order is applied after the filters and before offset and limit, so a cut
21
+ * drops the oldest matches first. 'newest' sorts by postedAt descending with
22
+ * undated jobs last and ties broken by id, which keeps pages deterministic.
23
+ * 'board' keeps the order the adapter returned.
24
+ *
25
+ * @returns {{ jobs: Array, total_matched: number }}
26
+ */
27
+ export function applyFiltersDetailed(jobs, options = {}) {
28
+ const { order = 'newest', offset = 0, limit = 100 } = options;
29
+ const matched = filterJobs(jobs, options);
30
+ return { jobs: pageJobs(matched, { order, offset, limit }), total_matched: matched.length };
31
+ }
32
+
33
+ /**
34
+ * The filter step on its own: every job that passes titleFilter, filter,
35
+ * postedWithinDays and the location filters, in the order given. No sort
36
+ * and no paging, so fetchJobsDetailed can count the matches per board
37
+ * before the page cut removes them.
38
+ */
39
+ export function filterJobs(jobs, options = {}) {
40
+ const { titleFilter, filter, postedWithinDays, locationIncludes, locationExcludes } = options;
41
+ const { title, topic } = compileFilterPatterns({ titleFilter, filter });
17
42
 
18
43
  let result = jobs;
19
44
 
20
- if (titleFilter) {
21
- const pattern = new RegExp(titleFilter, 'i');
22
- result = result.filter(j => pattern.test(j.title || ''));
45
+ if (title) {
46
+ result = result.filter(j => title.test(j.title || ''));
23
47
  }
24
48
 
25
- if (filter) {
26
- const pattern = new RegExp(filter, 'i');
49
+ if (topic) {
27
50
  result = result.filter(j =>
28
- pattern.test(j.title || '') ||
29
- pattern.test(j.department || '') ||
30
- pattern.test(j.description || '')
51
+ topic.test(j.title || '') ||
52
+ topic.test(j.department || '') ||
53
+ topic.test(j.description || '')
31
54
  );
32
55
  }
33
56
 
@@ -42,36 +65,106 @@ export function applyFilters(jobs, options = {}) {
42
65
 
43
66
  if (Array.isArray(locationIncludes) && locationIncludes.length > 0) {
44
67
  const matchers = locationIncludes.map(makeLocationMatcher);
45
- result = result.filter(j => {
46
- const loc = (j.location || '').toLowerCase();
47
- return matchers.some(m => m(loc));
48
- });
68
+ result = result.filter(j => jobLocations(j).some(loc => matchers.some(m => m(loc))));
49
69
  }
50
70
 
51
71
  if (Array.isArray(locationExcludes) && locationExcludes.length > 0) {
52
72
  const matchers = locationExcludes.map(makeLocationMatcher);
53
- result = result.filter(j => {
54
- const loc = (j.location || '').toLowerCase();
55
- return !matchers.some(m => m(loc));
56
- });
73
+ result = result.filter(j => !jobLocations(j).every(loc => matchers.some(m => m(loc))));
57
74
  }
58
75
 
59
- if (typeof limit === 'number' && result.length > limit) {
60
- result = result.slice(0, limit);
76
+ return result;
77
+ }
78
+
79
+ /**
80
+ * Sort, then cut the page (see applyFiltersDetailed for the order rules).
81
+ */
82
+ export function pageJobs(jobs, { order = 'newest', offset = 0, limit = 100 } = {}) {
83
+ let result = jobs;
84
+
85
+ if (order !== 'board') {
86
+ result = [...result].sort(byNewest);
87
+ }
88
+
89
+ const start = typeof offset === 'number' && offset > 0 ? offset : 0;
90
+ const end = typeof limit === 'number' ? start + limit : undefined;
91
+ if (start > 0 || (end !== undefined && result.length > end)) {
92
+ result = result.slice(start, end);
61
93
  }
62
94
 
63
95
  return result;
64
96
  }
65
97
 
66
98
  /**
67
- * Build a matcher for a single location keyword.
99
+ * Compile the two regex arguments, or throw ArgumentError naming the one
100
+ * that does not compile. Both are case-insensitive. fetchJobsDetailed calls
101
+ * this before its first request, so a bad pattern is reported as a bad
102
+ * argument and costs no upstream traffic.
103
+ *
104
+ * @returns {{ title: RegExp|null, topic: RegExp|null }}
105
+ */
106
+ export function compileFilterPatterns({ titleFilter, filter } = {}) {
107
+ return {
108
+ title: titleFilter ? compilePattern(titleFilter, 'titleFilter') : null,
109
+ topic: filter ? compilePattern(filter, 'filter') : null,
110
+ };
111
+ }
112
+
113
+ function compilePattern(source, name) {
114
+ try {
115
+ return new RegExp(source, 'i');
116
+ } catch (err) {
117
+ throw new ArgumentError(`${name}: ${err.message}`);
118
+ }
119
+ }
120
+
121
+ /**
122
+ * Every location a job is open in, lowercased. A job passes an include when
123
+ * any of them matches and is dropped by an exclude only when all of them
124
+ * match: a role open in Berlin and New York is still open in Berlin for
125
+ * someone excluding the US (issue #68). Jobs from before `locations`
126
+ * existed fall back to the single `location` string.
127
+ */
128
+ function jobLocations(job) {
129
+ const list = Array.isArray(job.locations) && job.locations.length > 0
130
+ ? job.locations
131
+ : [job.location || ''];
132
+ return list.map(loc => String(loc).toLowerCase());
133
+ }
134
+
135
+ function postedTime(job) {
136
+ if (!job.postedAt) return null;
137
+ const t = new Date(job.postedAt).getTime();
138
+ return Number.isFinite(t) ? t : null;
139
+ }
140
+
141
+ function byNewest(a, b) {
142
+ const ta = postedTime(a);
143
+ const tb = postedTime(b);
144
+ if (ta !== tb) {
145
+ if (ta === null) return 1;
146
+ if (tb === null) return -1;
147
+ return tb - ta;
148
+ }
149
+ const ia = a.id || '';
150
+ const ib = b.id || '';
151
+ return ia < ib ? -1 : ia > ib ? 1 : 0;
152
+ }
153
+
154
+ /**
155
+ * Build a matcher for a single location keyword. The matcher takes a
156
+ * lowercased location string.
68
157
  *
69
158
  * Short tokens (≤4 chars) use word-boundary matching to prevent substring
70
159
  * collisions like "US" matching "Australia", "Brussels", "Belarus", or "UK"
71
- * matching "Auckland". Longer tokens use substring matching so phrases like
160
+ * matching "Ukraine". Longer tokens use substring matching so phrases like
72
161
  * "United States" can match "United States of America".
162
+ *
163
+ * Exported for the Workday list pre-filter, so one rule (trim, empty
164
+ * keywords never match, word boundaries for short tokens) applies before
165
+ * and after detail hydration (issue #61).
73
166
  */
74
- function makeLocationMatcher(needle) {
167
+ export function makeLocationMatcher(needle) {
75
168
  const lower = (needle || '').toLowerCase().trim();
76
169
  if (!lower) return () => false;
77
170
  if (lower.length <= 4) {