@timurtekb/tekjobs 0.24.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/CHANGELOG.md +131 -0
  2. package/LICENSE +21 -0
  3. package/README.md +145 -0
  4. package/app/README.md +22 -0
  5. package/app/dist/assets/index-DzGgV6NF.css +1 -0
  6. package/app/dist/assets/index-QteuouKv.js +77 -0
  7. package/app/dist/favicon.svg +1 -0
  8. package/app/dist/index.html +16 -0
  9. package/app/server/cover-letter.mjs +243 -0
  10. package/app/server/index.mjs +113 -0
  11. package/app/server/mail-check.mjs +234 -0
  12. package/app/server/mcp.mjs +129 -0
  13. package/app/server/people.mjs +164 -0
  14. package/app/server/store.mjs +1131 -0
  15. package/app/server/tailored-resume.mjs +213 -0
  16. package/cli.mjs +205 -0
  17. package/package.json +74 -0
  18. package/run.cmd +10 -0
  19. package/run.mjs +201 -0
  20. package/run.sh +12 -0
  21. package/samples/vault/Jobs/Basalt Systems - Senior Design Engineer, Growth (1010).md +56 -0
  22. package/samples/vault/Jobs/Brightline Studio - Design Engineer (1005).md +51 -0
  23. package/samples/vault/Jobs/Copperleaf - Principal Design Engineer (1006).md +53 -0
  24. package/samples/vault/Jobs/Example Studio - Senior Design Engineer (1013).md +52 -0
  25. package/samples/vault/Jobs/Fjord Analytics - Staff Frontend Engineer, Platform (1003).md +58 -0
  26. package/samples/vault/Jobs/Halcyon Robotics - Senior UX Engineer (1004).md +52 -0
  27. package/samples/vault/Jobs/Lumen Health - Design Systems Engineer (1002).md +51 -0
  28. package/samples/vault/Jobs/Meridian Pay - Staff Design Engineer (1009).md +67 -0
  29. package/samples/vault/Jobs/Northwind Labs - Staff Design Engineer, Design Systems (1000).md +57 -0
  30. package/samples/vault/Jobs/Orbital Software - Senior Design Engineer (1001).md +52 -0
  31. package/samples/vault/Jobs/Quill & Co - Design Engineer, Editor (1008).md +55 -0
  32. package/samples/vault/Jobs/Signalfire Design - Design Engineer (1012).md +53 -0
  33. package/samples/vault/Jobs/Tessellate - Senior Frontend Engineer, Design Systems (1007).md +52 -0
  34. package/samples/vault/Jobs/Verdant - UX Engineer (1011).md +51 -0
  35. package/samples/vault/Logs/2026-09-16.md +13 -0
  36. package/samples/vault/Logs/2026-09-23.md +13 -0
  37. package/samples/vault/People/Lee Example (Fjord Analytics).md +22 -0
  38. package/samples/vault/People/Priya Example (Meridian Pay).md +23 -0
  39. package/samples/vault/People/Sam Example (Northwind Labs).md +20 -0
  40. package/samples/vault/Profile/Positioning.md +16 -0
  41. package/samples/vault/Profile/Profile.md +41 -0
  42. package/samples/vault/Profile/Resume.md +32 -0
  43. package/samples/vault/Profile/Voice.md +15 -0
  44. package/samples/vault/README.md +11 -0
  45. package/samples/vault/Targets/Companies.md +318 -0
  46. package/samples/vault/Targets/Search Criteria.md +234 -0
  47. package/samples/vault/_Home.md +54 -0
  48. package/scraper/config.mjs +122 -0
  49. package/scraper/health.mjs +50 -0
  50. package/scraper/import-link.mjs +281 -0
  51. package/scraper/profile.mjs +186 -0
  52. package/scraper/rescore.mjs +236 -0
  53. package/scraper/resume-sync.mjs +224 -0
  54. package/scraper/resume.mjs +65 -0
  55. package/scraper/score.mjs +114 -0
  56. package/scraper/sources-email.mjs +242 -0
  57. package/scraper/sources-extra.mjs +558 -0
  58. package/scraper/sources-sites.mjs +280 -0
  59. package/scraper/sources.mjs +264 -0
  60. package/scraper/starter/companies-table.md +306 -0
  61. package/scraper/starter/criteria.json +56 -0
  62. package/scraper/starter/profile.md +38 -0
  63. package/scraper/vault.mjs +212 -0
@@ -0,0 +1,65 @@
1
+ // Resume text extraction: PDF (pdf-parse), DOCX (a minimal ZIP reader + word/document.xml), Markdown and plain text.
2
+ import fs from 'node:fs';
3
+ import path from 'node:path';
4
+ import zlib from 'node:zlib';
5
+
6
+ export async function extractText(file) {
7
+ const ext = path.extname(file).toLowerCase();
8
+ const buf = fs.readFileSync(file);
9
+ if (ext === '.pdf') return pdfText(buf);
10
+ if (ext === '.docx') return docxText(buf);
11
+ if (['.md', '.txt', '.markdown'].includes(ext)) return buf.toString('utf8');
12
+ throw new Error(`Unsupported resume format "${ext}". Use PDF, DOCX, Markdown or plain text.`);
13
+ }
14
+
15
+ async function pdfText(buf) {
16
+ let mod;
17
+ try { mod = await import('pdf-parse'); } catch { throw new Error('PDF support needs the pdf-parse package: run `npm install` in the TekJobs folder, or import the resume as DOCX, Markdown or text.'); }
18
+ if (mod.PDFParse) { // pdf-parse v2
19
+ const parser = new mod.PDFParse({ data: buf });
20
+ try { const r = await parser.getText(); return normalize(r.text || ''); } finally { await parser.destroy?.(); }
21
+ }
22
+ const fn = mod.default || mod; // pdf-parse v1
23
+ const r = await fn(buf);
24
+ return normalize(r.text || '');
25
+ }
26
+
27
+ /** DOCX is a ZIP; the body is word/document.xml. Paragraphs become lines, runs are joined, tags stripped. */
28
+ function docxText(buf) {
29
+ const xml = readZipEntry(buf, 'word/document.xml');
30
+ if (!xml) throw new Error('Not a DOCX file (no word/document.xml inside).');
31
+ const text = xml
32
+ .replace(/<w:tab\/>/g, '\t')
33
+ .replace(/<\/w:p>/g, '\n')
34
+ .replace(/<w:br[^>]*\/>/g, '\n')
35
+ .replace(/<[^>]+>/g, '')
36
+ .replace(/&amp;/g, '&').replace(/&lt;/g, '<').replace(/&gt;/g, '>').replace(/&quot;/g, '"').replace(/&apos;/g, "'");
37
+ return normalize(text);
38
+ }
39
+
40
+ function readZipEntry(buf, wanted) {
41
+ // Walk the central directory (signature 0x02014b50) to find the entry, then read it from its local header.
42
+ const eocd = buf.lastIndexOf(Buffer.from([0x50, 0x4b, 0x05, 0x06]));
43
+ if (eocd < 0) return null;
44
+ const cdOffset = buf.readUInt32LE(eocd + 16);
45
+ const count = buf.readUInt16LE(eocd + 10);
46
+ let p = cdOffset;
47
+ for (let i = 0; i < count; i++) {
48
+ if (buf.readUInt32LE(p) !== 0x02014b50) break;
49
+ const method = buf.readUInt16LE(p + 10);
50
+ const csize = buf.readUInt32LE(p + 20);
51
+ const nameLen = buf.readUInt16LE(p + 28), extraLen = buf.readUInt16LE(p + 30), commentLen = buf.readUInt16LE(p + 32);
52
+ const localOffset = buf.readUInt32LE(p + 42);
53
+ const name = buf.toString('utf8', p + 46, p + 46 + nameLen);
54
+ if (name === wanted) {
55
+ const lnameLen = buf.readUInt16LE(localOffset + 26), lextraLen = buf.readUInt16LE(localOffset + 28);
56
+ const start = localOffset + 30 + lnameLen + lextraLen;
57
+ const data = buf.subarray(start, start + csize);
58
+ return (method === 8 ? zlib.inflateRawSync(data) : data).toString('utf8');
59
+ }
60
+ p += 46 + nameLen + extraLen + commentLen;
61
+ }
62
+ return null;
63
+ }
64
+
65
+ const normalize = (s) => s.replace(/\r/g, '').replace(/[ \t]+\n/g, '\n').replace(/\n{3,}/g, '\n\n').trim();
@@ -0,0 +1,114 @@
1
+ const esc = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
2
+
3
+ /** Pull an annual USD pay range out of free text. Returns { min, max } in dollars or null. */
4
+ export function parseSalary(text = '') {
5
+ const t = text.replace(/–|—/g, '-');
6
+ const num = (s, k) => { let n = parseFloat(s.replace(/,/g, '')); if (k) n *= 1000; return n; };
7
+ const re = /\$\s?(\d{2,3}(?:,\d{3})?(?:\.\d+)?)\s?(k)?\s?(?:-|to|–)\s?\$?\s?(\d{2,3}(?:,\d{3})?(?:\.\d+)?)\s?(k)?/gi;
8
+ let best = null, m;
9
+ while ((m = re.exec(t))) {
10
+ let lo = num(m[1], m[2]), hi = num(m[3], m[4]);
11
+ if (lo < 1000) lo *= 1000; if (hi < 1000) hi *= 1000; // "$180-250" style
12
+ if (lo < 60000 || hi > 1500000 || hi < lo) continue; // hourly rates, equity %, junk
13
+ if (!best || hi > best.max) best = { min: lo, max: hi };
14
+ }
15
+ return best;
16
+ }
17
+ const wordHit = (hay, term) => new RegExp(`(^|[^a-z0-9])${esc(term)}([^a-z0-9]|$)`, 'i').test(hay);
18
+
19
+ /**
20
+ * Returns { score, reasons[], excluded } for one normalized job against the criteria JSON.
21
+ * `now` is the moment recency is judged from: the scan passes nothing (today); a rescore of an existing note
22
+ * passes the day the note was found, so the posting keeps the freshness it had when it was scored.
23
+ */
24
+ export function scoreJob(job, c, { now = Date.now() } = {}) {
25
+ const title = (job.title || '').toLowerCase();
26
+ const desc = (job.descriptionText || '').toLowerCase();
27
+ const loc = (job.location || '').toLowerCase();
28
+ const reasons = [];
29
+ let score = 0;
30
+
31
+ for (const ex of c.titleExclude || []) {
32
+ if (title.includes(ex.toLowerCase())) return { score: -999, reasons: [`excluded by title: "${ex}"`], excluded: true };
33
+ }
34
+
35
+ const titleHits = Object.entries(c.titleTerms || {})
36
+ .filter(([t]) => title.includes(t.toLowerCase()))
37
+ .sort((a, b) => b[1] - a[1]);
38
+ if (titleHits.length) {
39
+ // The best-matching term carries the weight; extra hits add a little and then stop. Uncapped, a title
40
+ // that happens to contain four of your terms outranks a better job whose title contains one, which is
41
+ // rewarding vocabulary rather than fit.
42
+ const extra = Math.min((titleHits.length - 1) * (c.titleExtraPer ?? 5), c.titleExtraCap ?? 10);
43
+ const pts = titleHits[0][1] + extra;
44
+ score += pts;
45
+ reasons.push(`title +${pts}: ${titleHits.map((h) => h[0]).join(', ')}`);
46
+ } else {
47
+ const pen = c.noTitleMatchPenalty ?? -40;
48
+ score += pen;
49
+ reasons.push(`no title match ${pen}`);
50
+ }
51
+
52
+ for (const [t, w] of Object.entries(c.seniority?.boost || {})) {
53
+ if (wordHit(title, t)) { score += w; reasons.push(`seniority +${w} (${t})`); break; }
54
+ }
55
+ for (const [t, w] of Object.entries(c.seniority?.penalty || {})) {
56
+ if (wordHit(title, t)) { score += w; reasons.push(`seniority ${w} (${t})`); break; }
57
+ }
58
+
59
+ let d = 0; const dhits = [];
60
+ for (const [t, w] of Object.entries(c.descTerms || {})) {
61
+ if (desc.includes(t.toLowerCase())) { d += w; dhits.push(t); }
62
+ }
63
+ d = Math.min(d, c.descCap ?? 35);
64
+ if (dhits.length) { score += d; reasons.push(`description +${d}: ${dhits.join(', ')}`); }
65
+
66
+ const L = c.location || {};
67
+ const isRemote = !!job.remote || /\bremote\b/.test(loc);
68
+ const bay = (L.bayAreaTerms || []).some((t) => loc.includes(t));
69
+ const us = (L.usTerms || []).some((t) => wordHit(loc, t));
70
+ const nonUs = (L.nonUsTerms || []).some((t) => wordHit(loc, t));
71
+ if (isRemote) { score += L.remoteBoost ?? 0; reasons.push(`remote +${L.remoteBoost ?? 0}`); }
72
+ else if (L.requireRemote) { score += L.notRemotePenalty ?? -60; reasons.push(`not remote ${L.notRemotePenalty ?? -60}`); }
73
+ if (bay) { score += L.bayAreaBoost ?? 0; reasons.push(`bay area +${L.bayAreaBoost ?? 0}`); }
74
+ if (nonUs && !us && !bay) { score += L.nonUsPenalty ?? 0; reasons.push(`non-US location ${L.nonUsPenalty ?? 0}`); }
75
+
76
+ const S = c.salary || {};
77
+ let payBand = 'unknown';
78
+ if (S.minAnnual && job.salaryMax) {
79
+ const top = `$${Math.round(job.salaryMax / 1000)}k`;
80
+ if (job.salaryMax >= S.minAnnual) {
81
+ payBand = 'floor';
82
+ // Clearing the floor is worth a fixed amount; clearing it by a lot is worth more. A flat bonus made
83
+ // a job topping out at the floor and one topping out 40% above it score identically, which is the
84
+ // opposite of how a person reads the same two numbers. Capped, so pay cannot dominate fit.
85
+ const over = Math.max(0, job.salaryMax - S.minAnnual);
86
+ const bonus = Math.min(Math.round((over / 10000) * (S.abovePer10k ?? 1)), S.aboveCap ?? 15);
87
+ score += (S.meetsBonus ?? 10) + bonus;
88
+ reasons.push(`pay range tops out at ${top}, at or above floor (+${S.meetsBonus ?? 10}${bonus ? `, +${bonus} for ${Math.round(over / 1000)}k above` : ''})`);
89
+ }
90
+ else if (S.stretchAnnual && job.salaryMax >= S.stretchAnnual) { payBand = 'stretch'; score += S.stretchPenalty ?? -8; reasons.push(`pay range tops out at ${top}, stretch band (${S.stretchPenalty ?? -8})`); }
91
+ else { payBand = 'below'; score += S.belowPenalty ?? -30; reasons.push(`pay range tops out at ${top}, under floor (${S.belowPenalty ?? -30})`); }
92
+ }
93
+ job.payBand = payBand;
94
+
95
+ if (job.posted) {
96
+ const days = (now - new Date(job.posted).getTime()) / 86400000;
97
+ if (Number.isFinite(days)) {
98
+ // Measured on a real vault: every listing that closed did so within seven days of being found, median
99
+ // two. A posting is perishable, so the fresh end of this scale needs more resolution than the stale
100
+ // end — without a days2 tier, something posted today and something posted four weeks ago differ by
101
+ // five points, which is less than one description keyword.
102
+ const r = c.recency || {};
103
+ const add = days <= 2 ? r.days2 ?? r.days7 ?? 0
104
+ : days <= 7 ? r.days7 ?? 0
105
+ : days <= 30 ? r.days30 ?? 0
106
+ : days <= 90 ? r.days90 ?? 0
107
+ : r.older ?? 0;
108
+ score += add;
109
+ reasons.push(`posted ${Math.max(0, Math.round(days))}d ago (${add >= 0 ? '+' : ''}${add})`);
110
+ }
111
+ }
112
+
113
+ return { score: Math.round(score), reasons, excluded: false, payBand };
114
+ }
@@ -0,0 +1,242 @@
1
+ // Job alert emails, parsed from a folder of saved messages. Same normalized job shape as sources.mjs.
2
+ //
3
+ // This is the front door to the boards that have no usable public API — LinkedIn, Indeed, Otta and the rest
4
+ // all send alert emails, and an email in your own mailbox is your own data. Nothing here contacts those
5
+ // sites: it reads files you put in a folder, and the links it produces are for you to open yourself.
6
+ //
7
+ // Drop `.eml` files into <profile>/Inbox/ (most mail clients save a message with File > Save As, and most
8
+ // can be given a rule that does it automatically). Files are never moved or deleted; the scan's own
9
+ // deduplication by job id stops a message being imported twice.
10
+ //
11
+ // What you get is thin by design. An alert email carries a title, a company, usually a location and rarely
12
+ // a sentence of description, so these rows score on their title almost alone and will sit below the same
13
+ // job fetched from an ATS. That is the right order: this source exists to catch what the others cannot see.
14
+ import fs from 'node:fs';
15
+ import path from 'node:path';
16
+ import { htmlToText, decodeEntities } from './sources.mjs';
17
+
18
+ // ---------------- a small MIME reader ----------------
19
+ // Enough of RFC 2045 to get the HTML out of a saved alert email, and no more. A full parser is a dependency
20
+ // this project does not want for one job.
21
+
22
+ function splitHeaders(raw) {
23
+ const end = raw.search(/\r?\n\r?\n/);
24
+ const head = end === -1 ? raw : raw.slice(0, end);
25
+ const body = end === -1 ? '' : raw.slice(end).replace(/^\r?\n\r?\n/, '');
26
+ const headers = {};
27
+ // Unfold: a header continues while the next line starts with whitespace.
28
+ for (const line of head.replace(/\r?\n[ \t]+/g, ' ').split(/\r?\n/)) {
29
+ const i = line.indexOf(':');
30
+ if (i > 0) headers[line.slice(0, i).trim().toLowerCase()] = line.slice(i + 1).trim();
31
+ }
32
+ return { headers, body };
33
+ }
34
+
35
+ const decodeQuotedPrintable = (s) =>
36
+ s.replace(/=\r?\n/g, '').replace(/=([0-9A-Fa-f]{2})/g, (_, h) => String.fromCharCode(parseInt(h, 16)));
37
+
38
+ function decodeBody(body, encoding = '') {
39
+ const enc = encoding.toLowerCase();
40
+ if (enc.includes('base64')) {
41
+ try { return Buffer.from(body.replace(/\s+/g, ''), 'base64').toString('utf8'); } catch { return body; }
42
+ }
43
+ if (enc.includes('quoted-printable')) {
44
+ // Decode to bytes first, then read as UTF-8: =C3=A9 is two bytes, not two characters.
45
+ return Buffer.from(decodeQuotedPrintable(body), 'binary').toString('utf8');
46
+ }
47
+ return body;
48
+ }
49
+
50
+ /** The best body part to read: HTML if the message has one, otherwise plain text. */
51
+ function bestPart(raw, depth = 0) {
52
+ const { headers, body } = splitHeaders(raw);
53
+ const type = (headers['content-type'] || 'text/plain').toLowerCase();
54
+
55
+ if (type.startsWith('multipart/') && depth < 6) {
56
+ const boundary = (type.match(/boundary="?([^";]+)"?/) || [])[1];
57
+ if (boundary) {
58
+ // Non-capturing: a capture group would put its own matches into the split result as undefined entries.
59
+ const parts = body.split(new RegExp(`--${boundary.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}(?:--)?\\r?\\n`)).slice(1);
60
+ const found = parts.filter((p) => p && p.trim()).map((p) => bestPart(p, depth + 1)).filter(Boolean);
61
+ return found.find((p) => p.type.includes('html')) || found.find((p) => p.type.includes('text')) || null;
62
+ }
63
+ }
64
+ if (!type.startsWith('text/')) return null;
65
+ return { type, text: decodeBody(body, headers['content-transfer-encoding'] || '') };
66
+ }
67
+
68
+ /** `=?utf-8?B?...?=` encoded-words, as they appear in Subject and From. */
69
+ function decodeWords(s = '') {
70
+ return s.replace(/=\?([^?]+)\?([BbQq])\?([^?]*)\?=/g, (_, charset, enc, text) => {
71
+ try {
72
+ if (enc.toUpperCase() === 'B') return Buffer.from(text, 'base64').toString('utf8');
73
+ return Buffer.from(decodeQuotedPrintable(text.replace(/_/g, ' ')), 'binary').toString('utf8');
74
+ } catch { return text; }
75
+ });
76
+ }
77
+
78
+ export function readEmail(raw) {
79
+ const { headers } = splitHeaders(raw);
80
+ const part = bestPart(raw);
81
+ return {
82
+ from: decodeWords(headers.from || ''),
83
+ subject: decodeWords(headers.subject || ''),
84
+ date: headers.date || '',
85
+ html: part && part.type.includes('html') ? part.text : '',
86
+ text: part ? part.text : '',
87
+ };
88
+ }
89
+
90
+ // ---------------- turning a message into jobs ----------------
91
+
92
+ /** Tracking parameters make every copy of a link unique, which would defeat deduplication. */
93
+ function cleanUrl(href) {
94
+ try {
95
+ const u = new URL(decodeEntities(href.replace(/&amp;/g, '&')));
96
+ for (const k of [...u.searchParams.keys()]) {
97
+ // utm_ is a prefix (utm_source, utm_campaign, …); the rest are whole names.
98
+ if (/^utm_/i.test(k) || /^(trk|tracking|trackingId|refId|midToken|mid|eid|lipi|licu|from|source|ref)$/i.test(k)) u.searchParams.delete(k);
99
+ }
100
+ u.hash = '';
101
+ return u.toString();
102
+ } catch { return ''; }
103
+ }
104
+
105
+ /** Anchors, in document order, as { href, text }. */
106
+ function anchors(html) {
107
+ const out = [];
108
+ for (const m of html.matchAll(/<a\b[^>]*href=["']([^"']+)["'][^>]*>([\s\S]*?)<\/a>/gi)) {
109
+ // `end` matters: the context for a job is what comes *after* its link. Starting at `index` instead
110
+ // drags the anchor's own text into the first line whenever the markup has no block break after it,
111
+ // and the company is then read as part of the title.
112
+ out.push({ href: m[1], text: decodeEntities(m[2].replace(/<[^>]+>/g, ' ')).replace(/\s+/g, ' ').trim(), index: m.index ?? 0, end: (m.index ?? 0) + m[0].length });
113
+ }
114
+ return out;
115
+ }
116
+
117
+ /**
118
+ * Each board's alert email is laid out differently, but all of them repeat one block per job around one
119
+ * link. `id` pulls a stable identifier out of that link; `clean` rebuilds the canonical URL from it, so the
120
+ * same job in two different alerts collapses to one row.
121
+ */
122
+ const BOARDS = [
123
+ {
124
+ name: 'linkedin',
125
+ from: /linkedin\.com/i,
126
+ link: /linkedin\.com\/(?:comm\/)?jobs\/view\/(\d+)/i,
127
+ clean: (id) => `https://www.linkedin.com/jobs/view/${id}/`,
128
+ },
129
+ {
130
+ name: 'indeed',
131
+ from: /indeed\.com/i,
132
+ link: /indeed\.com\/(?:viewjob|rc\/clk|pagead\/clk)[^"']*?[?&]jk=([0-9a-f]+)/i,
133
+ clean: (id) => `https://www.indeed.com/viewjob?jk=${id}`,
134
+ },
135
+ {
136
+ // Otta was acquired by Welcome to the Jungle; app.otta.com now redirects to app.welcometothejungle.com.
137
+ // Both senders and both link shapes are matched, because old alerts keep working and new ones arrive
138
+ // under the new name.
139
+ name: 'otta',
140
+ from: /(otta\.com|welcometothejungle\.com)/i,
141
+ link: /(?:otta\.com|welcometothejungle\.com)\/jobs\/([A-Za-z0-9_-]+)/i,
142
+ clean: (id) => `https://app.welcometothejungle.com/jobs/${id}`,
143
+ },
144
+ {
145
+ name: 'wellfound',
146
+ from: /(wellfound|angel)\.co/i,
147
+ link: /wellfound\.com\/jobs\/(\d+)/i,
148
+ clean: (id) => `https://wellfound.com/jobs/${id}`,
149
+ },
150
+ ];
151
+
152
+ /** Anything else: keep links that look like a job posting on a board we already understand. */
153
+ const GENERIC_LINK = /(greenhouse\.io|lever\.co|ashbyhq\.com|myworkdayjobs\.com|smartrecruiters\.com|workable\.com|breezy\.hr|bamboohr\.com)\/[^"']*/i;
154
+
155
+ /**
156
+ * The text that follows a job link, up to the next one — where the company and location live in every
157
+ * layout examined. Read as plain text so a change of markup does not break it.
158
+ */
159
+ function contextAfter(html, from, to) {
160
+ return htmlToText(html.slice(from, to === -1 ? from + 1200 : to))
161
+ .split('\n')
162
+ .map((l) => l.trim())
163
+ .filter(Boolean);
164
+ }
165
+
166
+ const LOCATION_HINT = /remote|hybrid|on-?site|,\s*[A-Z]{2}\b|United States|United Kingdom|Canada|Germany|India|Ireland|Netherlands|Australia|Singapore/i;
167
+
168
+ export function jobsFromEmail(raw, file = '') {
169
+ const mail = readEmail(raw);
170
+ const html = mail.html || mail.text;
171
+ if (!html) return [];
172
+ const board = BOARDS.find((b) => b.from.test(mail.from)) || null;
173
+ const links = anchors(html);
174
+ const posted = mail.date && !isNaN(Date.parse(mail.date)) ? new Date(mail.date).toISOString() : null;
175
+
176
+ const jobs = [];
177
+ const seen = new Set();
178
+ for (let i = 0; i < links.length; i++) {
179
+ const a = links[i];
180
+ const hit = board ? a.href.match(board.link) : a.href.match(GENERIC_LINK);
181
+ if (!hit) continue;
182
+ const url = board ? board.clean(hit[1]) : cleanUrl(a.href);
183
+ if (!url || seen.has(url)) continue;
184
+
185
+ // The anchor text is the title in every layout examined; anchors that are buttons ("Apply", "View job")
186
+ // wrap the same link, so the first one with real words wins.
187
+ const title = a.text.replace(/\s+/g, ' ').trim();
188
+ if (!title || title.length < 3 || /^(apply|view|see|show)\b/i.test(title)) continue;
189
+
190
+ const next = links.slice(i + 1).find((l) => (board ? l.href.match(board.link) : l.href.match(GENERIC_LINK)));
191
+ const lines = contextAfter(html, a.end, next ? next.index : -1).filter((l) => l && l !== title);
192
+ const location = lines.find((l) => LOCATION_HINT.test(l) && l.length < 80) || '';
193
+ const company = lines.find((l) => l !== location && l.length < 60 && !/^\W/.test(l)) || '';
194
+
195
+ seen.add(url);
196
+ jobs.push({
197
+ id: `mail:${board ? board.name : 'link'}:${url}`,
198
+ source: `email:${board ? board.name : 'link'}`,
199
+ company: company || 'Unknown (from an alert email)',
200
+ title,
201
+ url,
202
+ location,
203
+ remote: /\bremote\b/i.test(`${location} ${title}`),
204
+ posted,
205
+ // Alert emails carry a snippet at best. The subject is kept so the note says where the row came from.
206
+ descriptionHtml: `<p>Imported from an alert email${board ? ` (${board.name})` : ''}${file ? `: <code>${path.basename(file)}</code>` : ''}.</p><p>${decodeEntities(mail.subject)}</p>`,
207
+ salary: '',
208
+ department: '',
209
+ employmentType: '',
210
+ });
211
+ }
212
+ return jobs;
213
+ }
214
+
215
+ /**
216
+ * Every `.eml` in the inbox folder, as jobs. Never writes, moves or deletes a message: the folder is the
217
+ * person's, and the scan already refuses to rewrite a note it has seen before.
218
+ */
219
+ export async function fetchEmailInbox(criteria = {}) {
220
+ const cfg = criteria.email || {};
221
+ const dir = cfg.dir || (criteria._inbox ?? '');
222
+ if (!dir) return { ok: false, jobs: [], error: 'no inbox folder configured' };
223
+ if (!fs.existsSync(dir)) return { ok: false, jobs: [], error: `no folder at ${dir} — create it and save alert emails into it as .eml` };
224
+
225
+ const files = fs.readdirSync(dir).filter((f) => /\.(eml|txt|mht|mhtml)$/i.test(f));
226
+ if (!files.length) return { ok: false, jobs: [], error: `no .eml files in ${dir}` };
227
+
228
+ const all = [];
229
+ const failures = [];
230
+ for (const f of files.slice(0, cfg.maxFiles ?? 200)) {
231
+ const full = path.join(dir, f);
232
+ try {
233
+ all.push(...jobsFromEmail(fs.readFileSync(full, 'utf8'), full));
234
+ } catch (e) {
235
+ failures.push(`${f}: ${e.message}`);
236
+ }
237
+ }
238
+ const byUrl = new Map();
239
+ for (const j of all) if (!byUrl.has(j.url)) byUrl.set(j.url, j);
240
+ if (!byUrl.size) return { ok: false, jobs: [], error: failures.length ? `no jobs found; ${failures[0]}` : `no job links found in ${files.length} message(s)` };
241
+ return { ok: true, jobs: [...byUrl.values()] };
242
+ }