jd-intel 0.11.1 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -8
- package/package.json +2 -2
- package/registry/personio.json +11 -0
- package/registry/workable.json +17 -0
- package/src/adapters/index.js +6 -0
- package/src/adapters/personio.js +106 -0
- package/src/adapters/workable.js +78 -0
- package/src/boards.js +3 -1
- package/src/cli.js +2 -1
- package/src/index.js +1 -1
package/README.md
CHANGED
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
|
|
10
10
|
> **Stop pasting job descriptions into AI assistants. Let your AI fetch them directly.**
|
|
11
11
|
|
|
12
|
-
Full text. Clean structure. Across
|
|
12
|
+
Full text. Clean structure. Across nine major ATS. No copy-paste. No context loss.
|
|
13
13
|
|
|
14
14
|
---
|
|
15
15
|
|
|
@@ -40,9 +40,9 @@ Done.
|
|
|
40
40
|
Because scraping breaks where jd-intel doesn't:
|
|
41
41
|
|
|
42
42
|
- **Full JDs when browsing fails.** SPA-rendered boards, slow loads, auth walls, and geo-restrictions block a browser. They don't block a public API call.
|
|
43
|
-
- **Structured data, not HTML soup.** Salary, location type, department, and clean markdown, normalized across
|
|
43
|
+
- **Structured data, not HTML soup.** Salary, location type, department, and clean markdown, normalized across nine ATS.
|
|
44
44
|
- **No keys, no browser.** Public APIs only. Runs anywhere your AI does.
|
|
45
|
-
- **One schema, every platform.** Greenhouse, Lever, Ashby, SmartRecruiters, Teamtailor, Recruitee, Workday return the same shape.
|
|
45
|
+
- **One schema, every platform.** Greenhouse, Lever, Ashby, SmartRecruiters, Teamtailor, Recruitee, Personio, Workable, Workday return the same shape.
|
|
46
46
|
|
|
47
47
|
---
|
|
48
48
|
|
|
@@ -257,8 +257,9 @@ No custom parsing per company.
|
|
|
257
257
|
| SmartRecruiters | Shipped | Enterprise and mid-market |
|
|
258
258
|
| Teamtailor | Shipped | European startups and scale-ups |
|
|
259
259
|
| Recruitee | Shipped | Dutch / EU SMBs and scale-ups |
|
|
260
|
+
| Personio | Shipped | German / EU mid-market |
|
|
261
|
+
| Workable | Shipped | SMBs worldwide |
|
|
260
262
|
| Workday | Shipped | Large enterprises (registry-keyed) |
|
|
261
|
-
| Personio | Planned | German / EU mid-market |
|
|
262
263
|
|
|
263
264
|
Adding a new ATS is a single adapter file. See [Contributing](#contributing).
|
|
264
265
|
|
|
@@ -283,14 +284,12 @@ All filters AND together. Deep dive on patterns and gotchas: [docs/filters.md](d
|
|
|
283
284
|
|
|
284
285
|
**Shipped**
|
|
285
286
|
- Library, CLI, and MCP server (three surfaces of one toolkit)
|
|
286
|
-
- Greenhouse, Ashby, Lever, SmartRecruiters, Teamtailor, Recruitee, Workday adapters
|
|
287
|
+
- Greenhouse, Ashby, Lever, SmartRecruiters, Teamtailor, Recruitee, Personio, Workable, Workday adapters
|
|
287
288
|
- Title, topic, location, and date filters
|
|
288
289
|
- Salary extraction from JD text
|
|
289
|
-
- Verified company registry (
|
|
290
|
+
- Verified company registry (1,000+ companies)
|
|
290
291
|
|
|
291
292
|
**Next**
|
|
292
|
-
- Personio adapter (German / EU mid-market)
|
|
293
|
-
- Workable adapter (widget API; broad SMB coverage)
|
|
294
293
|
- Anthropic MCP marketplace submission
|
|
295
294
|
|
|
296
295
|
**Planned**
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "jd-intel",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"description": "Fetch and normalize job descriptions across
|
|
3
|
+
"version": "0.12.0",
|
|
4
|
+
"description": "Fetch and normalize job descriptions across nine major ATS (Greenhouse, Lever, Ashby, Workday, and more), for your AI assistant. No copy-paste.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "src/index.js",
|
|
7
7
|
"bin": {
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
[
|
|
2
|
+
{"slug": "1komma5grad", "name": "1KOMMA5°", "sector": "energy tech"},
|
|
3
|
+
{"slug": "thermondo", "name": "Thermondo", "sector": "energy tech"},
|
|
4
|
+
{"slug": "westwing", "name": "Westwing", "sector": "ecommerce"},
|
|
5
|
+
{"slug": "vivid", "name": "Vivid Money", "sector": "fintech (banking)"},
|
|
6
|
+
{"slug": "merantix", "name": "Merantix", "sector": "ai venture studio"},
|
|
7
|
+
{"slug": "ottonova", "name": "ottonova", "sector": "health insurance"},
|
|
8
|
+
{"slug": "finanzguru", "name": "Finanzguru", "sector": "fintech"},
|
|
9
|
+
{"slug": "holidu", "name": "Holidu", "sector": "travel tech"},
|
|
10
|
+
{"slug": "clark", "name": "CLARK", "sector": "insurtech"}
|
|
11
|
+
]
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
[
|
|
2
|
+
{"slug": "viva", "name": "Viva.com", "sector": "fintech / payments"},
|
|
3
|
+
{"slug": "spotawheel", "name": "Spotawheel", "sector": "automotive"},
|
|
4
|
+
{"slug": "learnworlds", "name": "LearnWorlds", "sector": "edtech"},
|
|
5
|
+
{"slug": "welcomepickups", "name": "Welcome Pickups", "sector": "travel tech"},
|
|
6
|
+
{"slug": "novibet", "name": "Novibet", "sector": "gaming"},
|
|
7
|
+
{"slug": "upstream", "name": "Upstream", "sector": "marketing tech"},
|
|
8
|
+
{"slug": "skroutz", "name": "Skroutz", "sector": "ecommerce"},
|
|
9
|
+
{"slug": "epignosis", "name": "Epignosis", "sector": "edtech"},
|
|
10
|
+
{"slug": "blueground", "name": "Blueground", "sector": "real estate tech"},
|
|
11
|
+
{"slug": "orfium", "name": "Orfium", "sector": "media"},
|
|
12
|
+
{"slug": "huggingface", "name": "Hugging Face", "sector": "ai / llms"},
|
|
13
|
+
{"slug": "hellasdirect", "name": "Hellas Direct", "sector": "insurtech"},
|
|
14
|
+
{"slug": "ferryhopper", "name": "Ferryhopper", "sector": "travel tech"},
|
|
15
|
+
{"slug": "persado", "name": "Persado", "sector": "marketing tech"},
|
|
16
|
+
{"slug": "natech", "name": "Natech", "sector": "banking software"}
|
|
17
|
+
]
|
package/src/adapters/index.js
CHANGED
|
@@ -4,6 +4,8 @@ import { fetchAshby, hasAshby } from './ashby.js';
|
|
|
4
4
|
import { fetchSmartrecruiters, hasSmartrecruiters } from './smartrecruiters.js';
|
|
5
5
|
import { fetchTeamtailor, hasTeamtailor } from './teamtailor.js';
|
|
6
6
|
import { fetchRecruitee, hasRecruitee } from './recruitee.js';
|
|
7
|
+
import { fetchPersonio, hasPersonio } from './personio.js';
|
|
8
|
+
import { fetchWorkable, hasWorkable } from './workable.js';
|
|
7
9
|
import { fetchWorkday, hasWorkday } from './workday.js';
|
|
8
10
|
|
|
9
11
|
export {
|
|
@@ -13,6 +15,8 @@ export {
|
|
|
13
15
|
fetchSmartrecruiters, hasSmartrecruiters,
|
|
14
16
|
fetchTeamtailor, hasTeamtailor,
|
|
15
17
|
fetchRecruitee, hasRecruitee,
|
|
18
|
+
fetchPersonio, hasPersonio,
|
|
19
|
+
fetchWorkable, hasWorkable,
|
|
16
20
|
fetchWorkday, hasWorkday,
|
|
17
21
|
};
|
|
18
22
|
|
|
@@ -23,6 +27,8 @@ export const ADAPTERS = {
|
|
|
23
27
|
smartrecruiters: { fetch: fetchSmartrecruiters, has: hasSmartrecruiters },
|
|
24
28
|
teamtailor: { fetch: fetchTeamtailor, has: hasTeamtailor },
|
|
25
29
|
recruitee: { fetch: fetchRecruitee, has: hasRecruitee },
|
|
30
|
+
personio: { fetch: fetchPersonio, has: hasPersonio },
|
|
31
|
+
workable: { fetch: fetchWorkable, has: hasWorkable },
|
|
26
32
|
workday: { fetch: fetchWorkday, has: hasWorkday },
|
|
27
33
|
};
|
|
28
34
|
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
import { normalize, decodeEntities, toIso } from '../normalizer.js';
|
|
2
|
+
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { atsFetch } from '../http.js';
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Fetch jobs from a Personio career site via its public XML feed.
|
|
7
|
+
*
|
|
8
|
+
* `{slug}.jobs.personio.de/xml` is the unauthenticated feed Personio
|
|
9
|
+
* publishes for job-board integrations. One GET returns every open
|
|
10
|
+
* position with its full description, split into titled sections
|
|
11
|
+
* ("Your mission", "Your profile"), each an HTML fragment inside CDATA.
|
|
12
|
+
*
|
|
13
|
+
* Existence is read from the status, not the body: a career site with
|
|
14
|
+
* nothing open answers 200 with an empty <workzag-jobs>, and a slug with
|
|
15
|
+
* no site answers 307 to personio.com. So the request must not follow
|
|
16
|
+
* redirects, or every unknown slug would read as a marketing page.
|
|
17
|
+
*
|
|
18
|
+
* The feed names no company. `subcompany` is the hiring legal entity
|
|
19
|
+
* where the tenant set one, so it is what the board states about itself.
|
|
20
|
+
*
|
|
21
|
+
* @param {string} slug - Personio career-site subdomain (e.g., 'holidu')
|
|
22
|
+
* @param {object} [ctx] - { companyName, report }; report is called once with
|
|
23
|
+
* { org_name, org_url } when given
|
|
24
|
+
* @returns {Promise<Array>} Normalized job objects
|
|
25
|
+
*/
|
|
26
|
+
export async function fetchPersonio(slug, ctx = {}) {
|
|
27
|
+
const resp = await requestFeed(slug);
|
|
28
|
+
if (!resp) return []; // No Personio career site for this slug
|
|
29
|
+
|
|
30
|
+
const xml = await resp.text();
|
|
31
|
+
const positions = [...xml.matchAll(/<position>([\s\S]*?)<\/position>/g)].map(m => m[1]);
|
|
32
|
+
const pick = (src, tag) => decodeEntities(src.match(new RegExp(`<${tag}>([\\s\\S]*?)</${tag}>`))?.[1].trim() || '');
|
|
33
|
+
|
|
34
|
+
// Every link is on jobs.personio.de, so there is no company host to report.
|
|
35
|
+
if (typeof ctx.report === 'function') {
|
|
36
|
+
ctx.report({ org_name: positions.map(p => pick(p, 'subcompany')).find(Boolean) || null, org_url: null });
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
return positions.map(raw => {
|
|
40
|
+
// The posting's own fields, without the sections: each section has a
|
|
41
|
+
// <name> of its own.
|
|
42
|
+
const position = raw.replace(/<jobDescriptions>[\s\S]*<\/jobDescriptions>/, '');
|
|
43
|
+
const id = pick(position, 'id');
|
|
44
|
+
// The primary <office>, then any under <additionalOffices>.
|
|
45
|
+
const offices = [...position.matchAll(/<office>([\s\S]*?)<\/office>/g)].map(m => decodeEntities(m[1].trim()));
|
|
46
|
+
|
|
47
|
+
return normalize({
|
|
48
|
+
companySlug: slug,
|
|
49
|
+
// The feed names no company; subcompany is a legal entity that varies
|
|
50
|
+
// per posting, so it goes to metadata and the registry name labels the job.
|
|
51
|
+
company: ctx.companyName || slug,
|
|
52
|
+
title: pick(position, 'name'),
|
|
53
|
+
department: pick(position, 'department'),
|
|
54
|
+
location: offices[0] || '',
|
|
55
|
+
locations: offices,
|
|
56
|
+
workplace: null, // The feed has no remote/hybrid field; the location string decides.
|
|
57
|
+
description: buildDescription(raw),
|
|
58
|
+
url: id ? `https://${slug}.jobs.personio.de/job/${id}` : '',
|
|
59
|
+
postedAt: toIso(pick(position, 'createdAt')),
|
|
60
|
+
salary: null, // No structured salary; normalizer parses from text
|
|
61
|
+
metadata: {
|
|
62
|
+
personioId: id,
|
|
63
|
+
subcompany: pick(position, 'subcompany'),
|
|
64
|
+
recruitingCategory: pick(position, 'recruitingCategory'),
|
|
65
|
+
employmentType: pick(position, 'employmentType'),
|
|
66
|
+
seniority: pick(position, 'seniority'),
|
|
67
|
+
schedule: pick(position, 'schedule'),
|
|
68
|
+
yearsOfExperience: pick(position, 'yearsOfExperience'),
|
|
69
|
+
},
|
|
70
|
+
}, 'personio');
|
|
71
|
+
});
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* The feed response, or null when the slug has no career site (a redirect
|
|
76
|
+
* or a 404). atsFetch throws the 429, 5xx and network cases itself.
|
|
77
|
+
*/
|
|
78
|
+
async function requestFeed(slug) {
|
|
79
|
+
const resp = await atsFetch(`https://${slug}.jobs.personio.de/xml?language=en`, { redirect: 'manual' });
|
|
80
|
+
if (resp.ok) return resp;
|
|
81
|
+
if (resp.status === 404 || (resp.status >= 300 && resp.status < 400)) return null;
|
|
82
|
+
throw atsErrorFromStatus(resp.status, `Personio feed error for ${slug}: ${resp.status}`);
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* One posting's sections as HTML: each <jobDescription> is a plain-text
|
|
87
|
+
* <name> heading and a CDATA <value> that already holds HTML. Only the
|
|
88
|
+
* CDATA wrapper comes off here; normalize() strips and decodes the rest,
|
|
89
|
+
* once (see normalizer.js).
|
|
90
|
+
*/
|
|
91
|
+
function buildDescription(position) {
|
|
92
|
+
return [...position.matchAll(/<jobDescription>([\s\S]*?)<\/jobDescription>/g)].map(([, section]) => {
|
|
93
|
+
const heading = section.match(/<name>([\s\S]*?)<\/name>/)?.[1].trim() || '';
|
|
94
|
+
const value = (section.match(/<value>([\s\S]*?)<\/value>/)?.[1] || '').replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1').trim();
|
|
95
|
+
return (heading ? `<h3>${heading}</h3>` : '') + value;
|
|
96
|
+
}).filter(Boolean).join('\n');
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* Check if a company has a Personio career site: true when the feed
|
|
101
|
+
* answers, false on a redirect or a 404, and the AtsError from requestFeed
|
|
102
|
+
* for anything else.
|
|
103
|
+
*/
|
|
104
|
+
export async function hasPersonio(slug) {
|
|
105
|
+
return (await requestFeed(slug)) !== null;
|
|
106
|
+
}
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
import { normalize, toIso } from '../normalizer.js';
|
|
2
|
+
import { atsErrorFromStatus } from '../errors.js';
|
|
3
|
+
import { atsFetch, probeResult } from '../http.js';
|
|
4
|
+
|
|
5
|
+
const BASE_URL = 'https://apply.workable.com/api/v1/widget/accounts';
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Fetch jobs from a Workable careers page via its public widget endpoint.
|
|
9
|
+
*
|
|
10
|
+
* Why the widget, not the official API: Workable's REST API needs a
|
|
11
|
+
* per-account token. The widget endpoint is the unauthenticated one the
|
|
12
|
+
* hosted careers page itself calls, and `details=true` adds each job's
|
|
13
|
+
* full HTML description, so one GET returns the whole board.
|
|
14
|
+
*
|
|
15
|
+
* An account with nothing open answers 200 with `jobs: []`; a slug with no
|
|
16
|
+
* account answers 404.
|
|
17
|
+
*
|
|
18
|
+
* @param {string} slug - Workable account slug (e.g., 'epignosis')
|
|
19
|
+
* @param {object} [ctx] - { report }; report is called once with
|
|
20
|
+
* { org_name, org_url } when given
|
|
21
|
+
* @returns {Promise<Array>} Normalized job objects
|
|
22
|
+
*/
|
|
23
|
+
export async function fetchWorkable(slug, ctx = {}) {
|
|
24
|
+
const resp = await atsFetch(`${BASE_URL}/${slug}?details=true`);
|
|
25
|
+
|
|
26
|
+
if (!resp.ok) {
|
|
27
|
+
if (resp.status === 404) return []; // No Workable account for this slug
|
|
28
|
+
throw atsErrorFromStatus(resp.status, `Workable API error for ${slug}: ${resp.status}`);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
const data = await resp.json();
|
|
32
|
+
const jobs = data.jobs || [];
|
|
33
|
+
|
|
34
|
+
// The account's own name is at the top of the response. Every link is on
|
|
35
|
+
// apply.workable.com, so there is no company host to report.
|
|
36
|
+
if (typeof ctx.report === 'function') {
|
|
37
|
+
ctx.report({ org_name: data.name || null, org_url: null });
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
return jobs.map(job => {
|
|
41
|
+
const place = [job.city, job.state, job.country].filter(Boolean).join(', ');
|
|
42
|
+
return normalize({
|
|
43
|
+
companySlug: slug,
|
|
44
|
+
company: data.name || slug,
|
|
45
|
+
title: job.title || '',
|
|
46
|
+
department: job.department || '',
|
|
47
|
+
location: job.telecommuting ? (place ? `Remote - ${place}` : 'Remote') : place,
|
|
48
|
+
locations: (job.locations || [])
|
|
49
|
+
.filter(l => !l.hidden)
|
|
50
|
+
.map(l => [l.city, l.region, l.country].filter(Boolean).join(', ')),
|
|
51
|
+
// `telecommuting` can only say remote; false means nothing.
|
|
52
|
+
workplace: job.telecommuting === true ? 'remote' : null,
|
|
53
|
+
description: job.description || '',
|
|
54
|
+
url: job.url || job.shortlink || '',
|
|
55
|
+
// Date-only strings ("2026-09-07").
|
|
56
|
+
postedAt: toIso(job.published_on) || toIso(job.created_at),
|
|
57
|
+
salary: null, // No structured salary; normalizer parses from text
|
|
58
|
+
metadata: {
|
|
59
|
+
workableId: job.shortcode,
|
|
60
|
+
code: job.code || '',
|
|
61
|
+
employmentType: job.employment_type || '',
|
|
62
|
+
experience: job.experience || '',
|
|
63
|
+
education: job.education || '',
|
|
64
|
+
industry: job.industry || '',
|
|
65
|
+
function: job.function || '',
|
|
66
|
+
},
|
|
67
|
+
}, 'workable');
|
|
68
|
+
});
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Check if a company has a Workable account. See probeResult for the
|
|
73
|
+
* outcomes. The plain list (no details) is the cheap request.
|
|
74
|
+
*/
|
|
75
|
+
export async function hasWorkable(slug) {
|
|
76
|
+
const resp = await atsFetch(`${BASE_URL}/${slug}`);
|
|
77
|
+
return probeResult(resp, `Workable probe for ${slug}`);
|
|
78
|
+
}
|
package/src/boards.js
CHANGED
|
@@ -18,13 +18,15 @@ const BOARD_URLS = {
|
|
|
18
18
|
smartrecruiters: (slug) => `https://careers.smartrecruiters.com/${slug}`,
|
|
19
19
|
teamtailor: (slug) => `https://${slug}.teamtailor.com`,
|
|
20
20
|
recruitee: (slug) => `https://${slug}.recruitee.com`,
|
|
21
|
+
personio: (slug) => `https://${slug}.jobs.personio.de`,
|
|
22
|
+
workable: (slug) => `https://apply.workable.com/${slug}`,
|
|
21
23
|
workday: (slug, config) => (config ? `https://${config.tenant}.${config.env}.myworkdayjobs.com/${config.site}` : null),
|
|
22
24
|
};
|
|
23
25
|
|
|
24
26
|
// Domains the platforms own. A link there (boards.greenhouse.io,
|
|
25
27
|
// jobs.lever.co, testco.recruitee.com, cisco.wd5.myworkdayjobs.com) says
|
|
26
28
|
// which ATS hosts the board, nothing about whose board it is.
|
|
27
|
-
const ATS_DOMAINS = ['greenhouse.io', 'lever.co', 'ashbyhq.com', 'smartrecruiters.com', 'teamtailor.com', 'recruitee.com', 'myworkdayjobs.com'];
|
|
29
|
+
const ATS_DOMAINS = ['greenhouse.io', 'lever.co', 'ashbyhq.com', 'smartrecruiters.com', 'teamtailor.com', 'recruitee.com', 'personio.de', 'personio.com', 'workable.com', 'myworkdayjobs.com'];
|
|
28
30
|
|
|
29
31
|
/**
|
|
30
32
|
* The page a person opens to see the board, or null when the ATS is unknown
|
package/src/cli.js
CHANGED
|
@@ -163,7 +163,8 @@ Usage:
|
|
|
163
163
|
Fetch options:
|
|
164
164
|
--ats <platform> Skip auto-detect. One of: greenhouse, lever,
|
|
165
165
|
ashby, smartrecruiters, teamtailor, recruitee,
|
|
166
|
-
workday. Omit to
|
|
166
|
+
personio, workable, workday. Omit to
|
|
167
|
+
auto-detect (registry-backed).
|
|
167
168
|
--workday-tenant T Workday is keyed by a {tenant, env, site}
|
|
168
169
|
--workday-env wdN triple, not a slug. Registered Workday
|
|
169
170
|
--workday-site S companies work via auto-detect or --ats
|
package/src/index.js
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
*
|
|
4
4
|
* Fetches, normalizes, and enriches job data from public ATS APIs
|
|
5
5
|
* (Greenhouse, Lever, Ashby, SmartRecruiters, Teamtailor, Recruitee,
|
|
6
|
-
* Workday) into a unified schema.
|
|
6
|
+
* Personio, Workable, Workday) into a unified schema.
|
|
7
7
|
*/
|
|
8
8
|
|
|
9
9
|
import { ADAPTERS, ATS_NAMES } from './adapters/index.js';
|