jd-intel 0.11.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -8
- package/package.json +2 -3
- package/registry/personio.json +11 -0
- package/registry/workable.json +17 -0
- package/src/adapters/index.js +6 -0
- package/src/adapters/lever.js +2 -2
- package/src/adapters/personio.js +106 -0
- package/src/adapters/recruitee.js +5 -11
- package/src/adapters/teamtailor.js +2 -8
- package/src/adapters/workable.js +78 -0
- package/src/adapters/workday.js +5 -15
- package/src/boards.js +3 -1
- package/src/cli.js +2 -1
- package/src/index.js +1 -1
- package/src/normalizer.js +10 -0
- package/src/registry.js +8 -10
package/README.md
CHANGED
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
|
|
10
10
|
> **Stop pasting job descriptions into AI assistants. Let your AI fetch them directly.**
|
|
11
11
|
|
|
12
|
-
Full text. Clean structure. Across
|
|
12
|
+
Full text. Clean structure. Across nine major ATS. No copy-paste. No context loss.
|
|
13
13
|
|
|
14
14
|
---
|
|
15
15
|
|
|
@@ -40,9 +40,9 @@ Done.
|
|
|
40
40
|
Because scraping breaks where jd-intel doesn't:
|
|
41
41
|
|
|
42
42
|
- **Full JDs when browsing fails.** SPA-rendered boards, slow loads, auth walls, and geo-restrictions block a browser. They don't block a public API call.
|
|
43
|
-
- **Structured data, not HTML soup.** Salary, location type, department, and clean markdown, normalized across
|
|
43
|
+
- **Structured data, not HTML soup.** Salary, location type, department, and clean markdown, normalized across nine ATS.
|
|
44
44
|
- **No keys, no browser.** Public APIs only. Runs anywhere your AI does.
|
|
45
|
-
- **One schema, every platform.** Greenhouse, Lever, Ashby, SmartRecruiters, Teamtailor, Recruitee, Workday return the same shape.
|
|
45
|
+
- **One schema, every platform.** Greenhouse, Lever, Ashby, SmartRecruiters, Teamtailor, Recruitee, Personio, Workable, Workday return the same shape.
|
|
46
46
|
|
|
47
47
|
---
|
|
48
48
|
|
|
@@ -257,8 +257,9 @@ No custom parsing per company.
|
|
|
257
257
|
| SmartRecruiters | Shipped | Enterprise and mid-market |
|
|
258
258
|
| Teamtailor | Shipped | European startups and scale-ups |
|
|
259
259
|
| Recruitee | Shipped | Dutch / EU SMBs and scale-ups |
|
|
260
|
+
| Personio | Shipped | German / EU mid-market |
|
|
261
|
+
| Workable | Shipped | SMBs worldwide |
|
|
260
262
|
| Workday | Shipped | Large enterprises (registry-keyed) |
|
|
261
|
-
| Personio | Planned | German / EU mid-market |
|
|
262
263
|
|
|
263
264
|
Adding a new ATS is a single adapter file. See [Contributing](#contributing).
|
|
264
265
|
|
|
@@ -283,14 +284,12 @@ All filters AND together. Deep dive on patterns and gotchas: [docs/filters.md](d
|
|
|
283
284
|
|
|
284
285
|
**Shipped**
|
|
285
286
|
- Library, CLI, and MCP server (three surfaces of one toolkit)
|
|
286
|
-
- Greenhouse, Ashby, Lever, SmartRecruiters, Teamtailor, Recruitee, Workday adapters
|
|
287
|
+
- Greenhouse, Ashby, Lever, SmartRecruiters, Teamtailor, Recruitee, Personio, Workable, Workday adapters
|
|
287
288
|
- Title, topic, location, and date filters
|
|
288
289
|
- Salary extraction from JD text
|
|
289
|
-
- Verified company registry (
|
|
290
|
+
- Verified company registry (1,000+ companies)
|
|
290
291
|
|
|
291
292
|
**Next**
|
|
292
|
-
- Personio adapter (German / EU mid-market)
|
|
293
|
-
- Workable adapter (widget API; broad SMB coverage)
|
|
294
293
|
- Anthropic MCP marketplace submission
|
|
295
294
|
|
|
296
295
|
**Planned**
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "jd-intel",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"description": "Fetch and normalize job descriptions across
|
|
3
|
+
"version": "0.12.0",
|
|
4
|
+
"description": "Fetch and normalize job descriptions across nine major ATS (Greenhouse, Lever, Ashby, Workday, and more), for your AI assistant. No copy-paste.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "src/index.js",
|
|
7
7
|
"bin": {
|
|
@@ -15,7 +15,6 @@
|
|
|
15
15
|
"scripts": {
|
|
16
16
|
"test": "node --test test/*.test.js",
|
|
17
17
|
"fetch": "node src/cli.js fetch",
|
|
18
|
-
"search": "node src/cli.js search",
|
|
19
18
|
"verify:registry": "node scripts/verify-registry.mjs",
|
|
20
19
|
"sync:registry-pages": "node scripts/sync-pages-registry.mjs",
|
|
21
20
|
"pack:mcpb": "node scripts/build-mcpb.mjs",
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
[
|
|
2
|
+
{"slug": "1komma5grad", "name": "1KOMMA5°", "sector": "energy tech"},
|
|
3
|
+
{"slug": "thermondo", "name": "Thermondo", "sector": "energy tech"},
|
|
4
|
+
{"slug": "westwing", "name": "Westwing", "sector": "ecommerce"},
|
|
5
|
+
{"slug": "vivid", "name": "Vivid Money", "sector": "fintech (banking)"},
|
|
6
|
+
{"slug": "merantix", "name": "Merantix", "sector": "ai venture studio"},
|
|
7
|
+
{"slug": "ottonova", "name": "ottonova", "sector": "health insurance"},
|
|
8
|
+
{"slug": "finanzguru", "name": "Finanzguru", "sector": "fintech"},
|
|
9
|
+
{"slug": "holidu", "name": "Holidu", "sector": "travel tech"},
|
|
10
|
+
{"slug": "clark", "name": "CLARK", "sector": "insurtech"}
|
|
11
|
+
]
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
[
|
|
2
|
+
{"slug": "viva", "name": "Viva.com", "sector": "fintech / payments"},
|
|
3
|
+
{"slug": "spotawheel", "name": "Spotawheel", "sector": "automotive"},
|
|
4
|
+
{"slug": "learnworlds", "name": "LearnWorlds", "sector": "edtech"},
|
|
5
|
+
{"slug": "welcomepickups", "name": "Welcome Pickups", "sector": "travel tech"},
|
|
6
|
+
{"slug": "novibet", "name": "Novibet", "sector": "gaming"},
|
|
7
|
+
{"slug": "upstream", "name": "Upstream", "sector": "marketing tech"},
|
|
8
|
+
{"slug": "skroutz", "name": "Skroutz", "sector": "ecommerce"},
|
|
9
|
+
{"slug": "epignosis", "name": "Epignosis", "sector": "edtech"},
|
|
10
|
+
{"slug": "blueground", "name": "Blueground", "sector": "real estate tech"},
|
|
11
|
+
{"slug": "orfium", "name": "Orfium", "sector": "media"},
|
|
12
|
+
{"slug": "huggingface", "name": "Hugging Face", "sector": "ai / llms"},
|
|
13
|
+
{"slug": "hellasdirect", "name": "Hellas Direct", "sector": "insurtech"},
|
|
14
|
+
{"slug": "ferryhopper", "name": "Ferryhopper", "sector": "travel tech"},
|
|
15
|
+
{"slug": "persado", "name": "Persado", "sector": "marketing tech"},
|
|
16
|
+
{"slug": "natech", "name": "Natech", "sector": "banking software"}
|
|
17
|
+
]
|
package/src/adapters/index.js
CHANGED
|
@@ -4,6 +4,8 @@ import { fetchAshby, hasAshby } from './ashby.js';
|
|
|
4
4
|
import { fetchSmartrecruiters, hasSmartrecruiters } from './smartrecruiters.js';
|
|
5
5
|
import { fetchTeamtailor, hasTeamtailor } from './teamtailor.js';
|
|
6
6
|
import { fetchRecruitee, hasRecruitee } from './recruitee.js';
|
|
7
|
+
import { fetchPersonio, hasPersonio } from './personio.js';
|
|
8
|
+
import { fetchWorkable, hasWorkable } from './workable.js';
|
|
7
9
|
import { fetchWorkday, hasWorkday } from './workday.js';
|
|
8
10
|
|
|
9
11
|
export {
|
|
@@ -13,6 +15,8 @@ export {
|
|
|
13
15
|
fetchSmartrecruiters, hasSmartrecruiters,
|
|
14
16
|
fetchTeamtailor, hasTeamtailor,
|
|
15
17
|
fetchRecruitee, hasRecruitee,
|
|
18
|
+
fetchPersonio, hasPersonio,
|
|
19
|
+
fetchWorkable, hasWorkable,
|
|
16
20
|
fetchWorkday, hasWorkday,
|
|
17
21
|
};
|
|
18
22
|
|
|
@@ -23,6 +27,8 @@ export const ADAPTERS = {
|
|
|
23
27
|
smartrecruiters: { fetch: fetchSmartrecruiters, has: hasSmartrecruiters },
|
|
24
28
|
teamtailor: { fetch: fetchTeamtailor, has: hasTeamtailor },
|
|
25
29
|
recruitee: { fetch: fetchRecruitee, has: hasRecruitee },
|
|
30
|
+
personio: { fetch: fetchPersonio, has: hasPersonio },
|
|
31
|
+
workable: { fetch: fetchWorkable, has: hasWorkable },
|
|
26
32
|
workday: { fetch: fetchWorkday, has: hasWorkday },
|
|
27
33
|
};
|
|
28
34
|
|
package/src/adapters/lever.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize, extractSalaryFromText } from '../normalizer.js';
|
|
1
|
+
import { normalize, extractSalaryFromText, toIso } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
import { atsFetch, probeResult } from '../http.js';
|
|
4
4
|
|
|
@@ -44,7 +44,7 @@ export async function fetchLever(slug) {
|
|
|
44
44
|
workplace: job.workplaceType,
|
|
45
45
|
description: buildDescription(job),
|
|
46
46
|
url: job.hostedUrl || '',
|
|
47
|
-
postedAt:
|
|
47
|
+
postedAt: toIso(job.createdAt),
|
|
48
48
|
salary: parseLeverSalary(job.salaryRange, job.text),
|
|
49
49
|
metadata: {
|
|
50
50
|
leverId: job.id,
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
import { normalize, decodeEntities, toIso } from '../normalizer.js';
|
|
2
|
+
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { atsFetch } from '../http.js';
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Fetch jobs from a Personio career site via its public XML feed.
|
|
7
|
+
*
|
|
8
|
+
* `{slug}.jobs.personio.de/xml` is the unauthenticated feed Personio
|
|
9
|
+
* publishes for job-board integrations. One GET returns every open
|
|
10
|
+
* position with its full description, split into titled sections
|
|
11
|
+
* ("Your mission", "Your profile"), each an HTML fragment inside CDATA.
|
|
12
|
+
*
|
|
13
|
+
* Existence is read from the status, not the body: a career site with
|
|
14
|
+
* nothing open answers 200 with an empty <workzag-jobs>, and a slug with
|
|
15
|
+
* no site answers 307 to personio.com. So the request must not follow
|
|
16
|
+
* redirects, or every unknown slug would read as a marketing page.
|
|
17
|
+
*
|
|
18
|
+
* The feed names no company. `subcompany` is the hiring legal entity
|
|
19
|
+
* where the tenant set one, so it is what the board states about itself.
|
|
20
|
+
*
|
|
21
|
+
* @param {string} slug - Personio career-site subdomain (e.g., 'holidu')
|
|
22
|
+
* @param {object} [ctx] - { companyName, report }; report is called once with
|
|
23
|
+
* { org_name, org_url } when given
|
|
24
|
+
* @returns {Promise<Array>} Normalized job objects
|
|
25
|
+
*/
|
|
26
|
+
export async function fetchPersonio(slug, ctx = {}) {
|
|
27
|
+
const resp = await requestFeed(slug);
|
|
28
|
+
if (!resp) return []; // No Personio career site for this slug
|
|
29
|
+
|
|
30
|
+
const xml = await resp.text();
|
|
31
|
+
const positions = [...xml.matchAll(/<position>([\s\S]*?)<\/position>/g)].map(m => m[1]);
|
|
32
|
+
const pick = (src, tag) => decodeEntities(src.match(new RegExp(`<${tag}>([\\s\\S]*?)</${tag}>`))?.[1].trim() || '');
|
|
33
|
+
|
|
34
|
+
// Every link is on jobs.personio.de, so there is no company host to report.
|
|
35
|
+
if (typeof ctx.report === 'function') {
|
|
36
|
+
ctx.report({ org_name: positions.map(p => pick(p, 'subcompany')).find(Boolean) || null, org_url: null });
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
return positions.map(raw => {
|
|
40
|
+
// The posting's own fields, without the sections: each section has a
|
|
41
|
+
// <name> of its own.
|
|
42
|
+
const position = raw.replace(/<jobDescriptions>[\s\S]*<\/jobDescriptions>/, '');
|
|
43
|
+
const id = pick(position, 'id');
|
|
44
|
+
// The primary <office>, then any under <additionalOffices>.
|
|
45
|
+
const offices = [...position.matchAll(/<office>([\s\S]*?)<\/office>/g)].map(m => decodeEntities(m[1].trim()));
|
|
46
|
+
|
|
47
|
+
return normalize({
|
|
48
|
+
companySlug: slug,
|
|
49
|
+
// The feed names no company; subcompany is a legal entity that varies
|
|
50
|
+
// per posting, so it goes to metadata and the registry name labels the job.
|
|
51
|
+
company: ctx.companyName || slug,
|
|
52
|
+
title: pick(position, 'name'),
|
|
53
|
+
department: pick(position, 'department'),
|
|
54
|
+
location: offices[0] || '',
|
|
55
|
+
locations: offices,
|
|
56
|
+
workplace: null, // The feed has no remote/hybrid field; the location string decides.
|
|
57
|
+
description: buildDescription(raw),
|
|
58
|
+
url: id ? `https://${slug}.jobs.personio.de/job/${id}` : '',
|
|
59
|
+
postedAt: toIso(pick(position, 'createdAt')),
|
|
60
|
+
salary: null, // No structured salary; normalizer parses from text
|
|
61
|
+
metadata: {
|
|
62
|
+
personioId: id,
|
|
63
|
+
subcompany: pick(position, 'subcompany'),
|
|
64
|
+
recruitingCategory: pick(position, 'recruitingCategory'),
|
|
65
|
+
employmentType: pick(position, 'employmentType'),
|
|
66
|
+
seniority: pick(position, 'seniority'),
|
|
67
|
+
schedule: pick(position, 'schedule'),
|
|
68
|
+
yearsOfExperience: pick(position, 'yearsOfExperience'),
|
|
69
|
+
},
|
|
70
|
+
}, 'personio');
|
|
71
|
+
});
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* The feed response, or null when the slug has no career site (a redirect
|
|
76
|
+
* or a 404). atsFetch throws the 429, 5xx and network cases itself.
|
|
77
|
+
*/
|
|
78
|
+
async function requestFeed(slug) {
|
|
79
|
+
const resp = await atsFetch(`https://${slug}.jobs.personio.de/xml?language=en`, { redirect: 'manual' });
|
|
80
|
+
if (resp.ok) return resp;
|
|
81
|
+
if (resp.status === 404 || (resp.status >= 300 && resp.status < 400)) return null;
|
|
82
|
+
throw atsErrorFromStatus(resp.status, `Personio feed error for ${slug}: ${resp.status}`);
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* One posting's sections as HTML: each <jobDescription> is a plain-text
|
|
87
|
+
* <name> heading and a CDATA <value> that already holds HTML. Only the
|
|
88
|
+
* CDATA wrapper comes off here; normalize() strips and decodes the rest,
|
|
89
|
+
* once (see normalizer.js).
|
|
90
|
+
*/
|
|
91
|
+
function buildDescription(position) {
|
|
92
|
+
return [...position.matchAll(/<jobDescription>([\s\S]*?)<\/jobDescription>/g)].map(([, section]) => {
|
|
93
|
+
const heading = section.match(/<name>([\s\S]*?)<\/name>/)?.[1].trim() || '';
|
|
94
|
+
const value = (section.match(/<value>([\s\S]*?)<\/value>/)?.[1] || '').replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1').trim();
|
|
95
|
+
return (heading ? `<h3>${heading}</h3>` : '') + value;
|
|
96
|
+
}).filter(Boolean).join('\n');
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* Check if a company has a Personio career site: true when the feed
|
|
101
|
+
* answers, false on a redirect or a 404, and the AtsError from requestFeed
|
|
102
|
+
* for anything else.
|
|
103
|
+
*/
|
|
104
|
+
export async function hasPersonio(slug) {
|
|
105
|
+
return (await requestFeed(slug)) !== null;
|
|
106
|
+
}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize } from '../normalizer.js';
|
|
1
|
+
import { normalize, toIso } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
import { atsFetch, probeResult } from '../http.js';
|
|
4
4
|
import { orgHost } from '../boards.js';
|
|
@@ -51,7 +51,7 @@ export async function fetchRecruitee(slug, ctx = {}) {
|
|
|
51
51
|
let location = place;
|
|
52
52
|
if (offer.remote) location = place ? `Remote - ${place}` : 'Remote';
|
|
53
53
|
|
|
54
|
-
const createdAt =
|
|
54
|
+
const createdAt = recruiteeDate(offer.created_at);
|
|
55
55
|
|
|
56
56
|
return normalize({
|
|
57
57
|
companySlug: slug,
|
|
@@ -65,7 +65,7 @@ export async function fetchRecruitee(slug, ctx = {}) {
|
|
|
65
65
|
url: offer.careers_url || offer.careers_apply_url || '',
|
|
66
66
|
// created_at can predate publication by years on long-lived offers,
|
|
67
67
|
// so it is not a posting date. published_at is.
|
|
68
|
-
postedAt:
|
|
68
|
+
postedAt: recruiteeDate(offer.published_at) || createdAt,
|
|
69
69
|
salary: parseRecruiteeSalary(offer.salary),
|
|
70
70
|
metadata: {
|
|
71
71
|
recruiteeId: offer.guid || offer.id,
|
|
@@ -77,14 +77,8 @@ export async function fetchRecruitee(slug, ctx = {}) {
|
|
|
77
77
|
});
|
|
78
78
|
}
|
|
79
79
|
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
*/
|
|
83
|
-
function toIso(ts) {
|
|
84
|
-
if (!ts) return null;
|
|
85
|
-
const d = new Date(ts.replace(' UTC', 'Z').replace(' ', 'T'));
|
|
86
|
-
return Number.isNaN(d.getTime()) ? null : d.toISOString();
|
|
87
|
-
}
|
|
80
|
+
// Recruitee returns "2026-05-13 07:38:11 UTC".
|
|
81
|
+
const recruiteeDate = (ts) => toIso(ts?.replace(' UTC', 'Z').replace(' ', 'T'));
|
|
88
82
|
|
|
89
83
|
/**
|
|
90
84
|
* Recruitee sends three booleans, not one enum. Hybrid wins when remote is
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize, decodeEntities } from '../normalizer.js';
|
|
1
|
+
import { normalize, decodeEntities, toIso } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
import { atsFetch } from '../http.js';
|
|
4
4
|
import { orgHost } from '../boards.js';
|
|
@@ -116,12 +116,6 @@ export async function fetchTeamtailor(slug, ctx = {}) {
|
|
|
116
116
|
location = location ? `Remote - ${location}` : 'Remote';
|
|
117
117
|
}
|
|
118
118
|
|
|
119
|
-
let postedAt = null;
|
|
120
|
-
if (pubDateRaw) {
|
|
121
|
-
const d = new Date(pubDateRaw);
|
|
122
|
-
if (!Number.isNaN(d.getTime())) postedAt = d.toISOString();
|
|
123
|
-
}
|
|
124
|
-
|
|
125
119
|
return normalize({
|
|
126
120
|
companySlug: slug,
|
|
127
121
|
company,
|
|
@@ -132,7 +126,7 @@ export async function fetchTeamtailor(slug, ctx = {}) {
|
|
|
132
126
|
workplace: REMOTE_STATUS[remoteStatus.toLowerCase()] || null,
|
|
133
127
|
description: decodeEntities(pick('description')),
|
|
134
128
|
url: link,
|
|
135
|
-
postedAt,
|
|
129
|
+
postedAt: toIso(pubDateRaw),
|
|
136
130
|
salary: null, // No structured salary; normalizer parses from text
|
|
137
131
|
metadata: {
|
|
138
132
|
teamtailorId: guid,
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
import { normalize, toIso } from '../normalizer.js';
|
|
2
|
+
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { atsFetch, probeResult } from '../http.js';
|
|
4
|
+
|
|
5
|
+
const BASE_URL = 'https://apply.workable.com/api/v1/widget/accounts';
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Fetch jobs from a Workable careers page via its public widget endpoint.
|
|
9
|
+
*
|
|
10
|
+
* Why the widget, not the official API: Workable's REST API needs a
|
|
11
|
+
* per-account token. The widget endpoint is the unauthenticated one the
|
|
12
|
+
* hosted careers page itself calls, and `details=true` adds each job's
|
|
13
|
+
* full HTML description, so one GET returns the whole board.
|
|
14
|
+
*
|
|
15
|
+
* An account with nothing open answers 200 with `jobs: []`; a slug with no
|
|
16
|
+
* account answers 404.
|
|
17
|
+
*
|
|
18
|
+
* @param {string} slug - Workable account slug (e.g., 'epignosis')
|
|
19
|
+
* @param {object} [ctx] - { report }; report is called once with
|
|
20
|
+
* { org_name, org_url } when given
|
|
21
|
+
* @returns {Promise<Array>} Normalized job objects
|
|
22
|
+
*/
|
|
23
|
+
export async function fetchWorkable(slug, ctx = {}) {
|
|
24
|
+
const resp = await atsFetch(`${BASE_URL}/${slug}?details=true`);
|
|
25
|
+
|
|
26
|
+
if (!resp.ok) {
|
|
27
|
+
if (resp.status === 404) return []; // No Workable account for this slug
|
|
28
|
+
throw atsErrorFromStatus(resp.status, `Workable API error for ${slug}: ${resp.status}`);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
const data = await resp.json();
|
|
32
|
+
const jobs = data.jobs || [];
|
|
33
|
+
|
|
34
|
+
// The account's own name is at the top of the response. Every link is on
|
|
35
|
+
// apply.workable.com, so there is no company host to report.
|
|
36
|
+
if (typeof ctx.report === 'function') {
|
|
37
|
+
ctx.report({ org_name: data.name || null, org_url: null });
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
return jobs.map(job => {
|
|
41
|
+
const place = [job.city, job.state, job.country].filter(Boolean).join(', ');
|
|
42
|
+
return normalize({
|
|
43
|
+
companySlug: slug,
|
|
44
|
+
company: data.name || slug,
|
|
45
|
+
title: job.title || '',
|
|
46
|
+
department: job.department || '',
|
|
47
|
+
location: job.telecommuting ? (place ? `Remote - ${place}` : 'Remote') : place,
|
|
48
|
+
locations: (job.locations || [])
|
|
49
|
+
.filter(l => !l.hidden)
|
|
50
|
+
.map(l => [l.city, l.region, l.country].filter(Boolean).join(', ')),
|
|
51
|
+
// `telecommuting` can only say remote; false means nothing.
|
|
52
|
+
workplace: job.telecommuting === true ? 'remote' : null,
|
|
53
|
+
description: job.description || '',
|
|
54
|
+
url: job.url || job.shortlink || '',
|
|
55
|
+
// Date-only strings ("2026-09-07").
|
|
56
|
+
postedAt: toIso(job.published_on) || toIso(job.created_at),
|
|
57
|
+
salary: null, // No structured salary; normalizer parses from text
|
|
58
|
+
metadata: {
|
|
59
|
+
workableId: job.shortcode,
|
|
60
|
+
code: job.code || '',
|
|
61
|
+
employmentType: job.employment_type || '',
|
|
62
|
+
experience: job.experience || '',
|
|
63
|
+
education: job.education || '',
|
|
64
|
+
industry: job.industry || '',
|
|
65
|
+
function: job.function || '',
|
|
66
|
+
},
|
|
67
|
+
}, 'workable');
|
|
68
|
+
});
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Check if a company has a Workable account. See probeResult for the
|
|
73
|
+
* outcomes. The plain list (no details) is the cheap request.
|
|
74
|
+
*/
|
|
75
|
+
export async function hasWorkable(slug) {
|
|
76
|
+
const resp = await atsFetch(`${BASE_URL}/${slug}`);
|
|
77
|
+
return probeResult(resp, `Workable probe for ${slug}`);
|
|
78
|
+
}
|
package/src/adapters/workday.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { normalize, missingContent } from '../normalizer.js';
|
|
1
|
+
import { normalize, missingContent, toIso } from '../normalizer.js';
|
|
2
2
|
import { atsErrorFromStatus } from '../errors.js';
|
|
3
3
|
import { prefilterRows } from '../filters.js';
|
|
4
4
|
import { atsFetch } from '../http.js';
|
|
@@ -151,7 +151,8 @@ export async function fetchWorkday(slug, ctx = {}) {
|
|
|
151
151
|
workplace: parseWorkdayRemoteType(info.remoteType),
|
|
152
152
|
description: info.jobDescription || '',
|
|
153
153
|
url: `https://${tenant}.${env}.myworkdayjobs.com/${site}${externalPath}`,
|
|
154
|
-
|
|
154
|
+
// startDate is "2026-05-01" or "May 1, 2026".
|
|
155
|
+
postedAt: toIso(info.startDate) || normalizePostedOn(p.postedOn),
|
|
155
156
|
salary: null, // normalizer extracts from description text
|
|
156
157
|
content,
|
|
157
158
|
metadata: {
|
|
@@ -223,23 +224,12 @@ function withinDays(postedOn, days) {
|
|
|
223
224
|
* so the library's postedWithinDays re-filter has a value to compare.
|
|
224
225
|
*/
|
|
225
226
|
function normalizePostedOn(v) {
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
if (Number.isFinite(direct.getTime())) return direct.toISOString();
|
|
227
|
+
const direct = toIso(v);
|
|
228
|
+
if (direct) return direct;
|
|
229
229
|
const n = daysAgo(v);
|
|
230
230
|
return n === null ? null : new Date(Date.now() - n * 86400000).toISOString();
|
|
231
231
|
}
|
|
232
232
|
|
|
233
|
-
/**
|
|
234
|
-
* Workday detail `startDate` ("2026-05-01" or "May 1, 2026"). Return
|
|
235
|
-
* ISO, or null if unparseable.
|
|
236
|
-
*/
|
|
237
|
-
function parseWorkdayDate(s) {
|
|
238
|
-
if (!s) return null;
|
|
239
|
-
const d = new Date(s);
|
|
240
|
-
return Number.isFinite(d.getTime()) ? d.toISOString() : null;
|
|
241
|
-
}
|
|
242
|
-
|
|
243
233
|
/**
|
|
244
234
|
* Registry-only invariant: the {tenant,env,site} triple can't be
|
|
245
235
|
* probed from a company name. Always false so detect_ats never selects
|
package/src/boards.js
CHANGED
|
@@ -18,13 +18,15 @@ const BOARD_URLS = {
|
|
|
18
18
|
smartrecruiters: (slug) => `https://careers.smartrecruiters.com/${slug}`,
|
|
19
19
|
teamtailor: (slug) => `https://${slug}.teamtailor.com`,
|
|
20
20
|
recruitee: (slug) => `https://${slug}.recruitee.com`,
|
|
21
|
+
personio: (slug) => `https://${slug}.jobs.personio.de`,
|
|
22
|
+
workable: (slug) => `https://apply.workable.com/${slug}`,
|
|
21
23
|
workday: (slug, config) => (config ? `https://${config.tenant}.${config.env}.myworkdayjobs.com/${config.site}` : null),
|
|
22
24
|
};
|
|
23
25
|
|
|
24
26
|
// Domains the platforms own. A link there (boards.greenhouse.io,
|
|
25
27
|
// jobs.lever.co, testco.recruitee.com, cisco.wd5.myworkdayjobs.com) says
|
|
26
28
|
// which ATS hosts the board, nothing about whose board it is.
|
|
27
|
-
const ATS_DOMAINS = ['greenhouse.io', 'lever.co', 'ashbyhq.com', 'smartrecruiters.com', 'teamtailor.com', 'recruitee.com', 'myworkdayjobs.com'];
|
|
29
|
+
const ATS_DOMAINS = ['greenhouse.io', 'lever.co', 'ashbyhq.com', 'smartrecruiters.com', 'teamtailor.com', 'recruitee.com', 'personio.de', 'personio.com', 'workable.com', 'myworkdayjobs.com'];
|
|
28
30
|
|
|
29
31
|
/**
|
|
30
32
|
* The page a person opens to see the board, or null when the ATS is unknown
|
package/src/cli.js
CHANGED
|
@@ -163,7 +163,8 @@ Usage:
|
|
|
163
163
|
Fetch options:
|
|
164
164
|
--ats <platform> Skip auto-detect. One of: greenhouse, lever,
|
|
165
165
|
ashby, smartrecruiters, teamtailor, recruitee,
|
|
166
|
-
workday. Omit to
|
|
166
|
+
personio, workable, workday. Omit to
|
|
167
|
+
auto-detect (registry-backed).
|
|
167
168
|
--workday-tenant T Workday is keyed by a {tenant, env, site}
|
|
168
169
|
--workday-env wdN triple, not a slug. Registered Workday
|
|
169
170
|
--workday-site S companies work via auto-detect or --ats
|
package/src/index.js
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
*
|
|
4
4
|
* Fetches, normalizes, and enriches job data from public ATS APIs
|
|
5
5
|
* (Greenhouse, Lever, Ashby, SmartRecruiters, Teamtailor, Recruitee,
|
|
6
|
-
* Workday) into a unified schema.
|
|
6
|
+
* Personio, Workable, Workday) into a unified schema.
|
|
7
7
|
*/
|
|
8
8
|
|
|
9
9
|
import { ADAPTERS, ATS_NAMES } from './adapters/index.js';
|
package/src/normalizer.js
CHANGED
|
@@ -1,5 +1,15 @@
|
|
|
1
1
|
import { createHash } from 'node:crypto';
|
|
2
2
|
|
|
3
|
+
/**
|
|
4
|
+
* A date value (string or epoch ms) as an ISO string, or null when it is
|
|
5
|
+
* missing or does not parse.
|
|
6
|
+
*/
|
|
7
|
+
export function toIso(value) {
|
|
8
|
+
if (!value) return null;
|
|
9
|
+
const d = new Date(value);
|
|
10
|
+
return Number.isNaN(d.getTime()) ? null : d.toISOString();
|
|
11
|
+
}
|
|
12
|
+
|
|
3
13
|
/**
|
|
4
14
|
* Generate a stable ID for a job posting.
|
|
5
15
|
*/
|
package/src/registry.js
CHANGED
|
@@ -7,11 +7,6 @@ import { ADAPTERS, ATS_NAMES } from './adapters/index.js';
|
|
|
7
7
|
const __dirname = dirname(fileURLToPath(import.meta.url));
|
|
8
8
|
const REGISTRY_DIR = join(__dirname, '..', 'registry');
|
|
9
9
|
|
|
10
|
-
// The one order the registry is ever walked in. Lookups, detectAts and the
|
|
11
|
-
// loaded object all follow it, so which file answers for a slug does not
|
|
12
|
-
// depend on which file's load finished first (issue #87).
|
|
13
|
-
const PLATFORMS = ATS_NAMES;
|
|
14
|
-
|
|
15
10
|
// Network-first registry. A hosted copy lets installed bundles AND npx users
|
|
16
11
|
// pick up newly-added companies without reinstalling; the on-disk copy that
|
|
17
12
|
// ships with the package is the guaranteed offline fallback. The base URL is
|
|
@@ -73,8 +68,11 @@ async function loadPlatform(platform) {
|
|
|
73
68
|
*/
|
|
74
69
|
export async function loadRegistry(ats) {
|
|
75
70
|
if (ats) return loadPlatform(ats);
|
|
76
|
-
|
|
77
|
-
|
|
71
|
+
// ATS_NAMES is the one order the registry is ever walked in. Lookups,
|
|
72
|
+
// detectAts and this object all follow it, so which file answers for a
|
|
73
|
+
// slug does not depend on which file's load finished first (issue #87).
|
|
74
|
+
const lists = await Promise.all(ATS_NAMES.map(loadPlatform));
|
|
75
|
+
return Object.fromEntries(ATS_NAMES.map((platform, i) => [platform, lists[i]]));
|
|
78
76
|
}
|
|
79
77
|
|
|
80
78
|
/**
|
|
@@ -127,7 +125,7 @@ export async function searchRegistry(query) {
|
|
|
127
125
|
// normalized forms keeps registry-first routing working for those.
|
|
128
126
|
export const normSlug = (s) => String(s).toLowerCase().replace(/[^a-z0-9]/g, '');
|
|
129
127
|
|
|
130
|
-
const byPlatform = (a, b) =>
|
|
128
|
+
const byPlatform = (a, b) => ATS_NAMES.indexOf(a.ats) - ATS_NAMES.indexOf(b.ats);
|
|
131
129
|
|
|
132
130
|
/**
|
|
133
131
|
* Look up which ATS a slug belongs to in the registry.
|
|
@@ -143,7 +141,7 @@ export async function findAtsBySlug(slug) {
|
|
|
143
141
|
* Unlike findAtsBySlug (returns just the ats name), this returns the
|
|
144
142
|
* whole entry so callers can read adapter-specific config (e.g. the
|
|
145
143
|
* Workday {tenant, env, site} triple). The files are searched in
|
|
146
|
-
*
|
|
144
|
+
* ATS_NAMES order, so the first match is the same on every call.
|
|
147
145
|
*
|
|
148
146
|
* @returns {Promise<{ats: string, entry: object}|null>}
|
|
149
147
|
*/
|
|
@@ -167,7 +165,7 @@ export async function findEntryBySlug(slug) {
|
|
|
167
165
|
* adds nothing, and an AtsError (429, 5xx, 401, network) goes to `failed`
|
|
168
166
|
* with its code, so a board the probe could not check never reads as
|
|
169
167
|
* absent (issue #55). Any other error is a bug and is rethrown. Both lists
|
|
170
|
-
* come back in
|
|
168
|
+
* come back in ATS_NAMES order, never in completion order.
|
|
171
169
|
*
|
|
172
170
|
* @returns {Promise<{
|
|
173
171
|
* boards: Array<{ ats: string, slug: string, source: 'registry'|'probe' }>,
|