@timurtekb/tekjobs 0.24.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/CHANGELOG.md +131 -0
  2. package/LICENSE +21 -0
  3. package/README.md +145 -0
  4. package/app/README.md +22 -0
  5. package/app/dist/assets/index-DzGgV6NF.css +1 -0
  6. package/app/dist/assets/index-QteuouKv.js +77 -0
  7. package/app/dist/favicon.svg +1 -0
  8. package/app/dist/index.html +16 -0
  9. package/app/server/cover-letter.mjs +243 -0
  10. package/app/server/index.mjs +113 -0
  11. package/app/server/mail-check.mjs +234 -0
  12. package/app/server/mcp.mjs +129 -0
  13. package/app/server/people.mjs +164 -0
  14. package/app/server/store.mjs +1131 -0
  15. package/app/server/tailored-resume.mjs +213 -0
  16. package/cli.mjs +205 -0
  17. package/package.json +74 -0
  18. package/run.cmd +10 -0
  19. package/run.mjs +201 -0
  20. package/run.sh +12 -0
  21. package/samples/vault/Jobs/Basalt Systems - Senior Design Engineer, Growth (1010).md +56 -0
  22. package/samples/vault/Jobs/Brightline Studio - Design Engineer (1005).md +51 -0
  23. package/samples/vault/Jobs/Copperleaf - Principal Design Engineer (1006).md +53 -0
  24. package/samples/vault/Jobs/Example Studio - Senior Design Engineer (1013).md +52 -0
  25. package/samples/vault/Jobs/Fjord Analytics - Staff Frontend Engineer, Platform (1003).md +58 -0
  26. package/samples/vault/Jobs/Halcyon Robotics - Senior UX Engineer (1004).md +52 -0
  27. package/samples/vault/Jobs/Lumen Health - Design Systems Engineer (1002).md +51 -0
  28. package/samples/vault/Jobs/Meridian Pay - Staff Design Engineer (1009).md +67 -0
  29. package/samples/vault/Jobs/Northwind Labs - Staff Design Engineer, Design Systems (1000).md +57 -0
  30. package/samples/vault/Jobs/Orbital Software - Senior Design Engineer (1001).md +52 -0
  31. package/samples/vault/Jobs/Quill & Co - Design Engineer, Editor (1008).md +55 -0
  32. package/samples/vault/Jobs/Signalfire Design - Design Engineer (1012).md +53 -0
  33. package/samples/vault/Jobs/Tessellate - Senior Frontend Engineer, Design Systems (1007).md +52 -0
  34. package/samples/vault/Jobs/Verdant - UX Engineer (1011).md +51 -0
  35. package/samples/vault/Logs/2026-09-16.md +13 -0
  36. package/samples/vault/Logs/2026-09-23.md +13 -0
  37. package/samples/vault/People/Lee Example (Fjord Analytics).md +22 -0
  38. package/samples/vault/People/Priya Example (Meridian Pay).md +23 -0
  39. package/samples/vault/People/Sam Example (Northwind Labs).md +20 -0
  40. package/samples/vault/Profile/Positioning.md +16 -0
  41. package/samples/vault/Profile/Profile.md +41 -0
  42. package/samples/vault/Profile/Resume.md +32 -0
  43. package/samples/vault/Profile/Voice.md +15 -0
  44. package/samples/vault/README.md +11 -0
  45. package/samples/vault/Targets/Companies.md +318 -0
  46. package/samples/vault/Targets/Search Criteria.md +234 -0
  47. package/samples/vault/_Home.md +54 -0
  48. package/scraper/config.mjs +122 -0
  49. package/scraper/health.mjs +50 -0
  50. package/scraper/import-link.mjs +281 -0
  51. package/scraper/profile.mjs +186 -0
  52. package/scraper/rescore.mjs +236 -0
  53. package/scraper/resume-sync.mjs +224 -0
  54. package/scraper/resume.mjs +65 -0
  55. package/scraper/score.mjs +114 -0
  56. package/scraper/sources-email.mjs +242 -0
  57. package/scraper/sources-extra.mjs +558 -0
  58. package/scraper/sources-sites.mjs +280 -0
  59. package/scraper/sources.mjs +264 -0
  60. package/scraper/starter/companies-table.md +306 -0
  61. package/scraper/starter/criteria.json +56 -0
  62. package/scraper/starter/profile.md +38 -0
  63. package/scraper/vault.mjs +212 -0
@@ -0,0 +1,122 @@
1
+ import fs from 'node:fs';
2
+ import os from 'node:os';
3
+ import path from 'node:path';
4
+ import { fileURLToPath } from 'node:url';
5
+
6
+ export const ROOT = fileURLToPath(new URL('..', import.meta.url));
7
+ export const HOME_DIR = path.join(os.homedir(), '.tekjobs');
8
+ export const CONFIG_FILE = path.join(HOME_DIR, 'config.json');
9
+
10
+ /**
11
+ * The profile folder is the database: profile, resume, criteria, watchlist, job notes, logs.
12
+ * Resolution order: TEKJOBS_PROFILE (or the older TEKJOBS_VAULT) → ~/.tekjobs/config.json {"profile": …} → ~/.tekjobs/profile.
13
+ */
14
+ export function resolveProfileDir() {
15
+ if (process.env.TEKJOBS_PROFILE) return path.resolve(process.env.TEKJOBS_PROFILE);
16
+ if (process.env.TEKJOBS_VAULT) return path.resolve(process.env.TEKJOBS_VAULT);
17
+ try {
18
+ const cfg = JSON.parse(fs.readFileSync(CONFIG_FILE, 'utf8'));
19
+ if (cfg.profile) return path.resolve(cfg.profile);
20
+ } catch { /* no config yet */ }
21
+ return path.join(HOME_DIR, 'profile');
22
+ }
23
+ export function rememberProfileDir(dir) {
24
+ fs.mkdirSync(HOME_DIR, { recursive: true });
25
+ let cfg = {};
26
+ try { cfg = JSON.parse(fs.readFileSync(CONFIG_FILE, 'utf8')); } catch { /* fresh */ }
27
+ fs.writeFileSync(CONFIG_FILE, JSON.stringify({ ...cfg, profile: path.resolve(dir) }, null, 2));
28
+ }
29
+
30
+ function readConfig() {
31
+ try { return JSON.parse(fs.readFileSync(CONFIG_FILE, 'utf8')); } catch { return {}; }
32
+ }
33
+
34
+ /**
35
+ * The user agent on every request the tool makes: what it is, where the code lives, and how a site can reach
36
+ * the person running it when ~/.tekjobs/config.json sets {"contact": "you@example.com"}.
37
+ */
38
+ export const UA = `TekJobs/1.0 (personal job search tool; +https://github.com/Timurtek/tekjobs${readConfig().contact ? `; contact ${readConfig().contact}` : ''})`;
39
+
40
+ export const VAULT = resolveProfileDir();
41
+ export const PROFILE_DIR = VAULT;
42
+ /** Scan state lives inside the profile so a profile folder is self-contained and portable. */
43
+ export const DATA_DIR = path.join(VAULT, '.tekjobs');
44
+
45
+ export const P = {
46
+ profile: path.join(VAULT, 'Profile', 'Profile.md'),
47
+ profileDir: path.join(VAULT, 'Profile'),
48
+ criteria: path.join(VAULT, 'Targets', 'Search Criteria.md'),
49
+ // Named criteria sets. The scan runs with Search Criteria.md unless told `--criteria <name>`.
50
+ criteriaDir: path.join(VAULT, 'Targets', 'Criteria'),
51
+ companies: path.join(VAULT, 'Targets', 'Companies.md'),
52
+ jobs: path.join(VAULT, 'Jobs'),
53
+ inbox: path.join(VAULT, 'Inbox'),
54
+ logs: path.join(VAULT, 'Logs'),
55
+ home: path.join(VAULT, '_Home.md'),
56
+ seen: path.join(DATA_DIR, 'seen.json'),
57
+ lastRun: path.join(DATA_DIR, 'last-run.json'),
58
+ runsLog: path.join(DATA_DIR, 'runs.log'),
59
+ // What each board and feed did the last time the scan tried it (scraper/health.mjs).
60
+ health: path.join(DATA_DIR, 'board-health.json'),
61
+ };
62
+
63
+ export const ALL_ATS = ['greenhouse', 'lever', 'ashby', 'workday', 'rippling', 'smartrecruiters', 'workable', 'bamboohr', 'breezy', 'personio', 'teamtailor', 'eightfold', 'atlassian', 'github', 'spotify', 'amazon', 'google', 'apple'];
64
+
65
+ export function ensureDirs() {
66
+ for (const d of [DATA_DIR, P.jobs, P.inbox, P.logs, P.profileDir, path.dirname(P.criteria)]) fs.mkdirSync(d, { recursive: true });
67
+ }
68
+
69
+ /** Criteria live in a ```json fence inside Targets/Search Criteria.md */
70
+ export function parseCriteriaNote(md, label = 'Search Criteria') {
71
+ const m = md.match(/```json\s*\n([\s\S]*?)\n```/);
72
+ if (!m) throw new Error(`No \`\`\`json block found in ${label}`);
73
+ try {
74
+ return JSON.parse(m[1]);
75
+ } catch (e) {
76
+ throw new Error(`${label} JSON is invalid: ${e.message}`);
77
+ }
78
+ }
79
+ /** The active criteria, or any criteria note when given its path. */
80
+ export function loadCriteria(file = P.criteria) {
81
+ return parseCriteriaNote(fs.readFileSync(file, 'utf8'), file);
82
+ }
83
+ /** Where a named preset lives. The name is used as the file name, so the characters a file name cannot hold are dropped. */
84
+ export function criteriaPresetFile(name) {
85
+ const safe = String(name || '').replace(/[<>:"/\\|?*\x00-\x1f]/g, ' ').replace(/\s+/g, ' ').trim().slice(0, 80);
86
+ if (!safe) throw Object.assign(new Error('A preset needs a name.'), { status: 400 });
87
+ return path.join(P.criteriaDir, `${safe}.md`);
88
+ }
89
+
90
+ /** Companies live in a markdown table: | Company | ATS | Slug | Tier | Status | Notes | */
91
+ export function loadCompanies() {
92
+ const md = fs.readFileSync(P.companies, 'utf8');
93
+ const rows = [];
94
+ for (const line of md.split(/\r?\n/)) {
95
+ if (!line.trim().startsWith('|')) continue;
96
+ const cells = line.split('|').slice(1, -1).map((c) => c.trim());
97
+ if (cells.length < 3) continue;
98
+ const [name, ats, slug, tier = '', status = '', notes = ''] = cells;
99
+ if (!name || name.toLowerCase() === 'company') continue;
100
+ if (/^-+$/.test(name)) continue;
101
+ const atsNorm = ats.toLowerCase();
102
+ if (!ALL_ATS.includes(atsNorm)) continue;
103
+ rows.push({ name, ats: atsNorm, slug, tier, status, notes });
104
+ }
105
+ return rows;
106
+ }
107
+
108
+ /** Rewrite the Status column of the companies table with fetch results, in place. */
109
+ export function writeCompanyStatuses(statusBySlug) {
110
+ const md = fs.readFileSync(P.companies, 'utf8');
111
+ const out = md.split(/\r?\n/).map((line) => {
112
+ if (!line.trim().startsWith('|')) return line;
113
+ const cells = line.split('|').slice(1, -1).map((c) => c.trim());
114
+ if (cells.length < 6) return line;
115
+ const [name, ats, slug] = cells;
116
+ const key = `${ats.toLowerCase()}:${slug}`;
117
+ if (!(key in statusBySlug)) return line;
118
+ cells[4] = statusBySlug[key];
119
+ return `| ${cells.join(' | ')} |`;
120
+ });
121
+ fs.writeFileSync(P.companies, out.join('\n'));
122
+ }
@@ -0,0 +1,50 @@
1
+ // Source health: what each board and feed did the last time the scan tried it, kept beside the scan's other
2
+ // state in the profile folder's .tekjobs/ (not in the Companies table, whose one Status cell stays a short
3
+ // human line for Obsidian). The scan records; the app reads and derives a state per source.
4
+ import fs from 'node:fs';
5
+ import { P } from './config.mjs';
6
+
7
+ export const STALE_DAYS = 3;
8
+
9
+ export function loadHealth() {
10
+ try { return JSON.parse(fs.readFileSync(P.health, 'utf8')); } catch { return { boards: {}, feeds: {} }; }
11
+ }
12
+
13
+ export function saveHealth(h) {
14
+ fs.mkdirSync(require_dir(P.health), { recursive: true });
15
+ fs.writeFileSync(P.health, JSON.stringify(h, null, 2));
16
+ }
17
+ const require_dir = (file) => file.replace(/[\\/][^\\/]*$/, '');
18
+
19
+ /** Fold one fetch result into a source's record. `r` is the fetcher's own answer: ok, jobs, error, emptyBoard. */
20
+ export function record(prev = {}, r, now = new Date().toISOString()) {
21
+ const jobs = r.ok ? r.jobs.length : null;
22
+ if (!r.ok) return { ...prev, lastAttempt: now, lastError: r.error || 'failed', failStreak: (prev.failStreak || 0) + 1 };
23
+ return {
24
+ ...prev,
25
+ lastAttempt: now, lastOk: now, lastError: '', failStreak: 0,
26
+ lastJobs: jobs, lastOkJobs: jobs,
27
+ zeroStreak: jobs === 0 ? (prev.zeroStreak || 0) + 1 : 0,
28
+ everJobs: Math.max(prev.everJobs || 0, jobs),
29
+ };
30
+ }
31
+
32
+ /**
33
+ * One word for where a source stands: failed (its latest attempt failed), zero (answered, but with nothing,
34
+ * which for a board that once had jobs usually means the slug moved), stale (no success in STALE_DAYS),
35
+ * never (no attempt on record), ok. The Companies table's Status cell is the fallback for rows the health
36
+ * file has not seen, so an old profile folder is not all "never" on the first day.
37
+ */
38
+ export function healthState(h, statusCell = '', now = Date.now()) {
39
+ if (!h || !h.lastAttempt) {
40
+ if (/^bad-slug/.test(statusCell)) return 'failed';
41
+ if (/0 jobs/.test(statusCell)) return 'zero';
42
+ return statusCell ? 'ok' : 'never';
43
+ }
44
+ if (h.lastError && (!h.lastOk || h.lastAttempt >= h.lastOk)) return 'failed';
45
+ if (!h.lastOk || now - Date.parse(h.lastOk) > STALE_DAYS * 864e5) return 'stale';
46
+ if (h.lastOkJobs === 0) return 'zero';
47
+ return 'ok';
48
+ }
49
+
50
+ export const STATES = ['failed', 'zero', 'stale', 'never', 'ok'];
@@ -0,0 +1,281 @@
1
+ // One posting, from a link the person pasted. The scan watches boards; this is for the job seen somewhere
2
+ // else: a LinkedIn feed, a newsletter, a friend's message. The link is read once, turned into the same job
3
+ // shape the scan produces, scored with the same criteria, and written as a note unless a note for it exists.
4
+ //
5
+ // Where the link points at a board the scan understands (Greenhouse, Lever, Ashby) the board's API is used,
6
+ // so the note is as full as a scanned one. A LinkedIn link is fetched once as a signed-out visitor, the way a
7
+ // browser would open it; if the posting names the company's own apply page on one of those boards, that
8
+ // page is imported instead and the LinkedIn link is kept as where it was seen. Nothing here searches or
9
+ // crawls anything: one link in, one request or two out.
10
+ import { htmlToText, decodeEntities } from './sources.mjs';
11
+ import { scoreJob, parseSalary } from './score.mjs';
12
+ import { loadSeen, saveSeen, writeJobNote, allJobNotes, jobNotePath } from './vault.mjs';
13
+ import { loadCriteria, loadCompanies, UA } from './config.mjs';
14
+
15
+ async function get(url, { accept = 'text/html,application/json;q=0.9,*/*;q=0.8', timeoutMs = 25000 } = {}) {
16
+ try {
17
+ const r = await fetch(url, { headers: { 'user-agent': UA, accept, 'accept-language': 'en-US,en;q=0.9' }, redirect: 'follow', signal: AbortSignal.timeout(timeoutMs) });
18
+ const text = await r.text();
19
+ let data = null; try { data = JSON.parse(text); } catch { /* html */ }
20
+ return { status: r.status, ok: r.ok, data, text, url: r.url };
21
+ } catch (e) { return { status: 0, ok: false, data: null, text: '', error: e.name === 'TimeoutError' ? 'timeout' : e.message }; }
22
+ }
23
+ const isRemoteText = (s = '') => /\bremote\b|\bdistributed\b|\banywhere\b|\bwork from home\b/i.test(s) || /^\s*(united states|usa|u\.s\.|us|north america|americas)\s*$/i.test(s);
24
+ const iso = (v) => { if (!v) return null; const d = new Date(v); return isNaN(d) ? null : d.toISOString(); };
25
+ const job = (o) => ({ salary: '', department: '', employmentType: '', descriptionHtml: '', location: '', remote: false, posted: null, ...o });
26
+ const fail = (error, extra = {}) => ({ ok: false, error, ...extra });
27
+
28
+ /** Tracking parameters make every copy of a link unique; strip them so the same posting is one posting. */
29
+ export function cleanUrl(href) {
30
+ const u = new URL(String(href).trim());
31
+ if (!/^https?:$/.test(u.protocol)) throw Object.assign(new Error('Only http(s) links.'), { status: 400 });
32
+ for (const k of [...u.searchParams.keys()]) if (/^utm_/i.test(k) || /^(trk|tracking|trackingId|refId|midToken|mid|eid|lipi|licu|from|source|ref|src|gh_src|lever-source|position|pageNum|refresh|original_referer)$/i.test(k)) u.searchParams.delete(k);
33
+ u.hash = '';
34
+ return u.toString();
35
+ }
36
+
37
+ /** Which reader a link gets. Exported so the choice can be tested without a network. */
38
+ export function classify(url) {
39
+ const u = new URL(url);
40
+ const host = u.hostname.replace(/^www\./, '');
41
+ let m;
42
+ if ((m = u.href.match(/greenhouse\.io\/(?:embed\/job_app\?.*?for=)?([^/?#]+)\/jobs\/(\d+)/))) return { kind: 'greenhouse', slug: m[1], id: m[2] };
43
+ if ((m = u.href.match(/greenhouse\.io\/embed\/job_app\?(?:.*&)?for=([^&]+).*?(?:&|\?)token=(\d+)/))) return { kind: 'greenhouse', slug: m[1], id: m[2] };
44
+ if (u.searchParams.get('gh_jid')) return { kind: 'greenhouse', slug: '', id: u.searchParams.get('gh_jid') };
45
+ if ((m = u.href.match(/jobs\.lever\.co\/([^/?#]+)\/([0-9a-f-]{36})/i))) return { kind: 'lever', slug: m[1], id: m[2] };
46
+ if ((m = u.href.match(/jobs\.ashbyhq\.com\/([^/?#]+)\/([0-9a-f-]{36})/i))) return { kind: 'ashby', slug: m[1], id: m[2] };
47
+ // A company careers page embedding Ashby: ?ashby_jid=<uuid>. The board name is read off the page.
48
+ if (/^[0-9a-f-]{36}$/i.test(u.searchParams.get('ashby_jid') || '')) return { kind: 'ashby', slug: '', id: u.searchParams.get('ashby_jid') };
49
+ if (/(^|\.)linkedin\.com$/.test(host)) {
50
+ const id = (u.pathname.match(/\/jobs\/view\/(?:[^/]*?-)?(\d+)/) || [])[1] || u.searchParams.get('currentJobId');
51
+ if (id) return { kind: 'linkedin', id };
52
+ return { kind: 'unsupported', why: 'a LinkedIn link without a job id (open the posting itself and copy that link)' };
53
+ }
54
+ if (/(^|\.)indeed\.com$/.test(host)) return { kind: 'unsupported', why: 'Indeed blocks signed-out reads; open the posting and paste the company\'s own apply link instead' };
55
+ // Two career sites the scan already reads from their pages; a pasted job page goes through the same parser.
56
+ if ((m = u.href.match(/google\.com\/about\/careers\/applications\/jobs\/results\/(\d{9,})/))) return { kind: 'google', id: m[1] };
57
+ if ((m = u.href.match(/jobs\.apple\.com\/[^/]+\/details\/(\d+)/))) return { kind: 'apple', id: m[1] };
58
+ return { kind: 'page' };
59
+ }
60
+
61
+ const companyNameFor = (slug, fallback) => {
62
+ const c = loadCompanies().find((x) => x.slug.toLowerCase() === String(slug).toLowerCase());
63
+ return c ? c.name : fallback;
64
+ };
65
+ const titleCase = (s) => String(s).replace(/[-_]+/g, ' ').replace(/\b\w/g, (c) => c.toUpperCase());
66
+
67
+ async function fromGreenhouse({ slug, id }, url) {
68
+ if (!slug) {
69
+ // A company page with ?gh_jid=… usually embeds the Greenhouse board script, which names the board. When
70
+ // it renders the posting itself (AKQA does), the board token is almost always the company's domain name,
71
+ // so that is tried against the API; failing that, the page's own JSON-LD posting is the record.
72
+ const page = await get(url);
73
+ slug = (page.text.match(/greenhouse\.io\/embed\/job_board\/js\?for=([^"&']+)/) || page.text.match(/boards\.greenhouse\.io\/([^/"']+)/) || [])[1] || '';
74
+ if (!slug) {
75
+ const labels = new URL(url).hostname.replace(/^www\./, '').split('.');
76
+ const guess = labels.length >= 2 ? labels[labels.length - 2] : labels[0];
77
+ const probe = await get(`https://boards-api.greenhouse.io/v1/boards/${encodeURIComponent(guess)}/jobs/${id}?questions=false`, { accept: 'application/json' });
78
+ if (probe.data?.title) slug = guess;
79
+ else {
80
+ const p = jobPostingFromJsonLd(page.text);
81
+ if (!p) return fail('the page has a Greenhouse job id but never names its board, and carries no posting data of its own');
82
+ const j = jobFromJsonLd(p, url, { source: 'page', idPrefix: 'gh:page' });
83
+ j.id = `gh:page:${id}`;
84
+ return { ok: true, job: j };
85
+ }
86
+ }
87
+ }
88
+ const r = await get(`https://boards-api.greenhouse.io/v1/boards/${encodeURIComponent(slug)}/jobs/${id}?questions=false`, { accept: 'application/json' });
89
+ const j = r.data;
90
+ if (!j || !j.title) return fail(r.error || `Greenhouse says ${r.status} for ${slug}/${id}`);
91
+ const location = j.location?.name || (j.offices || []).map((o) => o.name).join('; ') || '';
92
+ return { ok: true, job: job({ id: `gh:${slug}:${j.id}`, source: 'greenhouse', company: companyNameFor(slug, j.company_name || titleCase(slug)), title: j.title, url: j.absolute_url || url, location, remote: isRemoteText(location) || isRemoteText(j.title), posted: iso(j.first_published || j.updated_at), descriptionHtml: j.content || '', department: (j.departments || []).map((d) => d.name).filter(Boolean).join(', ') }) };
93
+ }
94
+
95
+ async function fromLever({ slug, id }, url) {
96
+ const r = await get(`https://api.lever.co/v0/postings/${encodeURIComponent(slug)}/${id}`, { accept: 'application/json' });
97
+ const j = r.data;
98
+ if (!j || !j.text) return fail(r.error || `Lever says ${r.status} for ${slug}/${id}`);
99
+ const location = [j.categories?.location, j.categories?.allLocations?.join('; ')].filter(Boolean).join('; ');
100
+ const lists = (j.lists || []).map((l) => `<h3>${l.text}</h3>${l.content}`).join('');
101
+ const sr = j.salaryRange;
102
+ return { ok: true, job: job({ id: `lv:${slug}:${j.id}`, source: 'lever', company: companyNameFor(slug, titleCase(slug)), title: j.text, url: j.hostedUrl || url, location, remote: j.workplaceType === 'remote' || isRemoteText(location), posted: iso(j.createdAt), descriptionHtml: `${j.description || ''}${lists}${j.additional || ''}`, salary: sr && sr.min ? `${sr.currency || ''} ${sr.min}–${sr.max} / ${sr.interval || ''}`.trim() : '', department: [j.categories?.team, j.categories?.department].filter(Boolean).join(' / '), employmentType: j.categories?.commitment || '' }) };
103
+ }
104
+
105
+ async function fromAshby({ slug, id }, url) {
106
+ if (!slug) {
107
+ // The page's embed script or job links name the board; failing that, the domain label is the usual token.
108
+ const page = await get(url);
109
+ slug = (page.text.match(/jobs\.ashbyhq\.com\/([^/"'?#\s]+)/) || [])[1] || '';
110
+ if (!slug) {
111
+ const labels = new URL(url).hostname.replace(/^www\./, '').split('.');
112
+ const base = labels.length >= 2 ? labels[labels.length - 2] : labels[0];
113
+ for (const guess of [base, `${base}-${labels[labels.length - 1]}`]) {
114
+ const probe = await get(`https://api.ashbyhq.com/posting-api/job-board/${encodeURIComponent(guess)}`, { accept: 'application/json' });
115
+ if (probe.data?.jobs?.some((x) => x.id === id)) { slug = guess; break; }
116
+ }
117
+ }
118
+ if (!slug) return fail('the page has an Ashby job id but never names its board');
119
+ }
120
+ const r = await get(`https://api.ashbyhq.com/posting-api/job-board/${encodeURIComponent(slug)}?includeCompensation=true`, { accept: 'application/json' });
121
+ const j = (r.data?.jobs || []).find((x) => x.id === id);
122
+ if (!j) return fail(r.data ? `that posting is not on the ${slug} board any more (closed, or unlisted)` : r.error || `Ashby says ${r.status} for ${slug}`);
123
+ const location = [j.location, ...(j.secondaryLocations || []).map((l) => l.location)].filter(Boolean).join('; ');
124
+ return { ok: true, job: job({ id: `ab:${slug}:${j.id}`, source: 'ashby', company: companyNameFor(slug, titleCase(slug)), title: j.title, url: j.jobUrl || url, location, remote: j.workplaceType === 'Remote' || isRemoteText(location), workplaceType: j.workplaceType || '', posted: iso(j.publishedAt), descriptionHtml: j.descriptionHtml || j.descriptionPlain || '', salary: j.compensation?.compensationTierSummary || '', department: [j.department, j.team].filter(Boolean).join(' / '), employmentType: j.employmentType || '' }) };
125
+ }
126
+
127
+ /** The JobPosting from a page's JSON-LD, if it has one. Most career pages and LinkedIn's signed-out view do. */
128
+ export function jobPostingFromJsonLd(html) {
129
+ for (const m of html.matchAll(/<script[^>]*type=["']application\/ld(?:\+|&#x2B;|&#43;)json["'][^>]*>([\s\S]*?)<\/script>/gi)) {
130
+ try {
131
+ const d = JSON.parse(decodeEntities(m[1]).replace(new RegExp('[' + String.fromCharCode(0x2028, 0x2029) + ']', 'g'), ''));
132
+ const arr = [].concat(d).flatMap((x) => (x && x['@graph'] ? x['@graph'] : [x]));
133
+ const p = arr.find((x) => x && (x['@type'] === 'JobPosting' || (Array.isArray(x['@type']) && x['@type'].includes('JobPosting'))));
134
+ if (p) return p;
135
+ } catch { /* next block */ }
136
+ }
137
+ return null;
138
+ }
139
+ export function jobFromJsonLd(p, url, { source = 'page', idPrefix = 'link' } = {}) {
140
+ const locs = [].concat(p.jobLocation || []).map((l) => {
141
+ const a = l?.address || l; if (!a || typeof a !== 'object') return typeof l === 'string' ? l : '';
142
+ return [a.addressLocality, a.addressRegion, a.addressCountry?.name || a.addressCountry].filter(Boolean).join(', ');
143
+ }).filter(Boolean);
144
+ const location = [...new Set(locs)].join('; ') || (p.jobLocationType === 'TELECOMMUTE' ? 'Remote' : '');
145
+ const sal = p.baseSalary?.value || p.baseSalary;
146
+ const num = (v) => (v == null ? null : Number(String(v).replace(/[^\d.]/g, '')) || null);
147
+ const min = num(sal?.minValue ?? sal?.value), max = num(sal?.maxValue ?? sal?.value);
148
+ const unit = String(sal?.unitText || '').toUpperCase();
149
+ const salary = min && max && (!unit || unit === 'YEAR') ? `$${Math.round(min / 1000)}k–$${Math.round(max / 1000)}k` : '';
150
+ const org = p.hiringOrganization?.name || (typeof p.hiringOrganization === 'string' ? p.hiringOrganization : '');
151
+ const id = p.identifier?.value || p.identifier?.name || url;
152
+ return job({ id: `${idPrefix}:${String(id).slice(-80)}`, source, company: decodeEntities(String(org || '')).trim(), title: decodeEntities(String(p.title || '')).trim(), url, location, remote: p.jobLocationType === 'TELECOMMUTE' || isRemoteText(location) || isRemoteText(String(p.title || '')), posted: iso(p.datePosted), descriptionHtml: String(p.description || ''), salary, employmentType: [].concat(p.employmentType || []).join(', ') });
153
+ }
154
+
155
+ /** LinkedIn's signed-out job page hides the company's own apply link in a comment inside <code id="applyUrl">. */
156
+ export function linkedInApplyUrl(html) {
157
+ const m = html.match(/id="applyUrl"[^>]*>\s*<!--\s*"?([^"<]+?)"?\s*-->/);
158
+ if (!m) return '';
159
+ try {
160
+ const u = new URL(decodeEntities(m[1]));
161
+ // LinkedIn wraps the destination in its own redirect; the real link is the url parameter.
162
+ if (/linkedin\.com$/.test(u.hostname.replace(/^www\./, '')) && u.searchParams.get('url')) return u.searchParams.get('url');
163
+ return u.toString();
164
+ } catch { return ''; }
165
+ }
166
+
167
+ async function fromLinkedIn({ id }, seenAt) {
168
+ const url = `https://www.linkedin.com/jobs/view/${id}/`;
169
+ const r = await get(url);
170
+ if (r.status === 999 || /authwall|\/login|checkpoint/.test(r.url || '')) return fail('LinkedIn would not show that posting to a signed-out visitor. Open it, click Apply, and paste the company\'s own link.');
171
+ if (!r.ok) return fail(r.error || `LinkedIn says ${r.status} for job ${id}`);
172
+ const apply = linkedInApplyUrl(r.text);
173
+ if (apply) {
174
+ const via = classify(apply);
175
+ if (['greenhouse', 'lever', 'ashby'].includes(via.kind)) {
176
+ const inner = await READERS[via.kind](via, apply);
177
+ if (inner.ok) { inner.job.seenAt = seenAt; return inner; }
178
+ }
179
+ }
180
+ const p = jobPostingFromJsonLd(r.text);
181
+ if (!p) return fail('LinkedIn returned a page without the posting in it (it does that when it wants a sign-in). Open it, click Apply, and paste the company\'s own link.');
182
+ const j = jobFromJsonLd(p, apply || url, { source: 'linkedin', idPrefix: 'li' });
183
+ j.id = `li:${id}`;
184
+ j.seenAt = seenAt;
185
+ if (!j.company) j.company = decodeEntities((r.text.match(/<a[^>]*class="[^"]*topcard__org-name-link[^"]*"[^>]*>([\s\S]*?)<\/a>/) || [, ''])[1].replace(/<[^>]+>/g, '')).trim() || 'Unknown';
186
+ return { ok: true, job: j };
187
+ }
188
+
189
+ const meta = (html, name) => decodeEntities((html.match(new RegExp(`<meta[^>]+(?:property|name)="${name}"[^>]+content="([^"]*)"`, 'i')) || html.match(new RegExp(`<meta[^>]+content="([^"]*)"[^>]+(?:property|name)="${name}"`, 'i')) || [, ''])[1]).trim();
190
+ const GENERIC_HEADING = /^(job details?|job description|job opening|careers?|jobs?|open positions?|apply(?: now)?|position details?|overview)$/i;
191
+
192
+ /**
193
+ * The best title a plain page offers: the h1 unless it is a label like "Job details", then og:title, then the
194
+ * <title>, each cut at the site suffix (" | Acme Careers", " — Google Careers").
195
+ */
196
+ export function pageTitle(html) {
197
+ const h1 = decodeEntities((html.match(/<h1[^>]*>([\s\S]*?)<\/h1>/i) || [, ''])[1].replace(/<[^>]+>/g, '')).replace(/\s+/g, ' ').trim();
198
+ const candidates = [h1, meta(html, 'og:title'), decodeEntities((html.match(/<title[^>]*>([\s\S]*?)<\/title>/i) || [, ''])[1]).replace(/\s+/g, ' ').trim()];
199
+ const pick = candidates.find((t) => t && !GENERIC_HEADING.test(t)) || '';
200
+ return pick.split(/\s+[|–—-]\s+/)[0].trim();
201
+ }
202
+ /** The employer a plain page belongs to: og:site_name, else the domain's own name ("careers.example.com" is Example). */
203
+ export function siteName(html, url) {
204
+ const og = meta(html, 'og:site_name').replace(/\s*(careers?|jobs)\s*$/i, '').trim();
205
+ if (og) return og;
206
+ const host = new URL(url).hostname.replace(/^www\./, '').split('.');
207
+ // The registrable label: skip a country's second level ("co.uk", "com.au") the way a reader would.
208
+ const second = host.length >= 3 && /^(co|com|org|net|ac|gov|edu)$/.test(host[host.length - 2]) ? 3 : 2;
209
+ const label = host.length >= second ? host[host.length - second] : host[0];
210
+ return label.charAt(0).toUpperCase() + label.slice(1);
211
+ }
212
+
213
+ async function fromPage(_, url) {
214
+ const r = await get(url);
215
+ if (!r.ok) return fail(r.error || `the page says ${r.status}`);
216
+ const p = jobPostingFromJsonLd(r.text);
217
+ if (p) return { ok: true, job: jobFromJsonLd(p, url) };
218
+ // No structured data: the best title on the page and its text. Enough to score on the title and read later.
219
+ const title = pageTitle(r.text);
220
+ if (!title) return fail('the page has no job posting data and no title');
221
+ const body = r.text.replace(/<(header|nav|footer|aside)[\s\S]*?<\/\1>/gi, '');
222
+ return { ok: true, job: job({ id: `link:${url.slice(-80)}`, source: 'page', company: siteName(r.text, url), title, url, location: '', remote: isRemoteText(title), descriptionHtml: `<p>${htmlToText(body).slice(0, 8000).replace(/\n\n/g, '</p><p>')}</p>` }) };
223
+ }
224
+
225
+ async function fromGoogle(_, url) {
226
+ const r = await get(url);
227
+ if (!r.ok) return fail(r.error || `Google says ${r.status}`);
228
+ const { parseGoogleJobPage } = await import('./sources-sites.mjs');
229
+ const j = parseGoogleJobPage(r.text, url);
230
+ return j ? { ok: true, job: j } : fail('Google returned a page without the job on it');
231
+ }
232
+ async function fromApple(_, url) {
233
+ const r = await get(url);
234
+ if (!r.ok) return fail(r.error || `Apple says ${r.status}`);
235
+ const { parseAppleJobPage } = await import('./sources-sites.mjs');
236
+ const j = parseAppleJobPage(r.text, url);
237
+ return j ? { ok: true, job: j } : fail('Apple returned a page without the job data in it');
238
+ }
239
+
240
+ const READERS = { greenhouse: fromGreenhouse, lever: fromLever, ashby: fromAshby, linkedin: fromLinkedIn, google: fromGoogle, apple: fromApple, page: fromPage };
241
+
242
+ const norm = (s) => (s || '').toLowerCase().replace(/\(.*?\)|\[.*?\]/g, '').replace(/[^a-z0-9]+/g, ' ').trim();
243
+
244
+ /** Read and score one link without writing anything. Returns { ok, url, job, scored } or { ok: false, error }. */
245
+ export async function readLink(href, { criteria = loadCriteria() } = {}) {
246
+ let url;
247
+ try { url = cleanUrl(href); } catch (e) { return fail(e.message, { url: href }); }
248
+ const kind = classify(url);
249
+ if (kind.kind === 'unsupported') return fail(kind.why, { url });
250
+ const read = await READERS[kind.kind](kind, url);
251
+ if (!read.ok) return { ...read, url };
252
+ const j = read.job;
253
+ j.descriptionText = htmlToText(j.descriptionHtml || '');
254
+ const range = parseSalary(`${j.salary || ''}\n${j.descriptionText}`);
255
+ if (range) { j.salaryMin = range.min; j.salaryMax = range.max; if (!j.salary) j.salary = `$${Math.round(range.min / 1000)}k–$${Math.round(range.max / 1000)}k`; }
256
+ if (j.seenAt) j.descriptionHtml = `<p>Seen on LinkedIn: ${j.seenAt}</p>${j.descriptionHtml}`;
257
+ j.foundVia = `added from a link${j.seenAt ? ' seen on LinkedIn' : ''}`;
258
+ return { ok: true, url, job: j, scored: scoreJob(j, criteria) };
259
+ }
260
+
261
+ /** Import one link. Returns { ok, added, note, job, score, existing?, error? }. */
262
+ export async function importLink(href, { criteria = loadCriteria(), dry = false } = {}) {
263
+ const read = await readLink(href, { criteria });
264
+ if (!read.ok) return read;
265
+ const { url, job: j, scored } = read;
266
+
267
+ // Unique means: the scan has not seen this id, no note carries this URL, and no open note has the same
268
+ // company and title (the same posting reached through a different door).
269
+ const seen = loadSeen();
270
+ const notes = allJobNotes();
271
+ const dupe = seen[j.id] ? { why: 'the scan already has it', path: seen[j.id].path }
272
+ : (() => { const n = notes.find((f) => f.url === j.url); return n ? { why: 'a note already points at this link', path: n._file } : null; })()
273
+ || (() => { const n = notes.find((f) => norm(f.company) === norm(j.company) && norm(f.title) === norm(j.title)); return n ? { why: `a note already exists for ${j.company} / ${j.title}`, path: n._file } : null; })();
274
+ if (dupe) return { ok: true, added: false, url, job: j, score: scored.score, existing: dupe.path, reason: dupe.why };
275
+ if (dry) return { ok: true, added: false, dry: true, url, job: j, score: scored.score, note: jobNotePath(j) };
276
+
277
+ const file = writeJobNote(j, scored, criteria);
278
+ seen[j.id] = { path: file, firstSeen: new Date().toISOString(), company: j.company, title: j.title, companyKey: `link:${j.source}`, score: scored.score };
279
+ saveSeen(seen);
280
+ return { ok: true, added: true, url, job: j, score: scored.score, payBand: scored.payBand, reasons: scored.reasons, note: file, belowMin: scored.score < (criteria.minScore ?? 0), excluded: !!scored.excluded };
281
+ }