openings 0.1.32 → 0.1.34

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openings",
3
- "version": "0.1.32",
3
+ "version": "0.1.34",
4
4
  "description": "Find evidence-grounded jobs, including relevant roles you may not have searched for, without accounts or API keys.",
5
5
  "author": { "name": "Openings contributors" },
6
6
  "license": "MIT",
package/README.md CHANGED
@@ -2,7 +2,18 @@
2
2
 
3
3
  A free, candidate-safe job search for AI agents.
4
4
 
5
- Openings indexes public company job boards into a private index on your machine and exposes it to any MCP client. Your agent can find roles that fit a resume, explain the fit with evidence, and propose truthful resume improvements. There are no accounts, no API keys, no model calls, and no way to submit an application.
5
+ Openings indexes public company job boards into a private index on your machine and exposes it to any MCP client. Your agent can find roles that fit a resume, explain the fit with evidence, and propose truthful resume improvements. Installed this way there are no accounts and no API keys; there are never model calls and never a way to submit an application.
6
+
7
+ Two ways to run it, with different data rules:
8
+
9
+ | | On your machine (this package) | Hosted connector at `openings.avagama.co/mcp` |
10
+ |---|---|---|
11
+ | Sign-in | None | Email code, so an account exists |
12
+ | Job index | Built and stored on your machine | Ours, shared |
13
+ | Resume | Parsed in memory for one request, never written to disk | Sent to our server, processed in memory for that request, then discarded |
14
+ | What we receive | Crawl reports, plus anonymous usage events unless you turn them off | The same usage record, keyed to your account |
15
+
16
+ Both are covered in [Sharing crawls and usage](#sharing-crawls-and-usage) and on the [privacy page](https://avagama.co/privacy/).
6
17
 
7
18
  - **Eleven providers plus company sites.** Greenhouse, Lever, Ashby, Workday, Recruitee, SmartRecruiters, Workable, Breezy, Freshteam, Keka, and Zoho Recruit, crawled from their public structured endpoints, plus employer career sites read only through the schema.org JobPosting markup they publish for search engines. No free-form HTML scraping.
8
19
  - **Verified sources only.** Every company in the catalog passed an identity check against its own board.
@@ -99,7 +110,11 @@ On startup the server makes one request to the npm registry to learn the latest
99
110
 
100
111
  ## Privacy
101
112
 
102
- Job data comes straight from public ATS endpoints and is stored only on your machine. Your MCP client reads the resume file and passes its content to a tool; Openings never sees the path and never persists the content or anything derived from it. Every proposed change stays subject to your review.
113
+ Job data comes straight from public ATS endpoints and is stored only on your machine. Your MCP client reads the resume file and passes its content to a tool; Openings never sees the path, and it never writes the resume to disk, logs it, or sends it anywhere.
114
+
115
+ One thing derived from a resume does leave your machine while usage reporting is on, which is the default: the skill and title words the parser extracted, in the anonymous event described in [Sharing crawls and usage](#sharing-crawls-and-usage). Never the resume text, the quoted evidence, your name, or your contact details. `OPENINGS_USAGE=off` stops it, and an empty `OPENINGS_AGGREGATOR_URL` keeps everything local.
116
+
117
+ Every proposed change stays subject to your review.
103
118
 
104
119
  ## Develop
105
120
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openings",
3
- "version": "0.1.32",
3
+ "version": "0.1.34",
4
4
  "description": "A free, candidate-safe job-search substrate for AI agents",
5
5
  "license": "MIT",
6
6
  "repository": { "type": "git", "url": "git+https://github.com/abhay-avagama/hiring-agent.git" },
package/src/catalog.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  import { CASCADE_MINIMUM, CASCADE_WINDOWS, type Company, type Job, type JobAge, type JobSummary, type SearchQuery, type SearchWindow } from "./types.ts";
2
- import { providerSpec, type JsonGet } from "./providers.ts";
2
+ import { decodeEntities, providerSpec, type JsonGet } from "./providers.ts";
3
3
  import { crawlSite, sitePostingsToJobs } from "./jobposting-site.ts";
4
4
  import { classifyJob, isEligibleForCountry, normalizeLocation } from "./locations.ts";
5
5
 
@@ -518,10 +518,7 @@ function stripHtml(value: string): string {
518
518
  .replace(/<br\s*\/?\s*>/gi, "\n")
519
519
  .replace(/<\/p>/gi, "\n\n")
520
520
  .replace(/<[^>]+>/g, "")
521
- .replace(/&nbsp;/g, " ")
522
- .replace(/&amp;/g, "&")
523
- .replace(/&lt;/g, "<")
524
- .replace(/&gt;/g, ">")
521
+ .replace(/&[#a-z0-9]+;/gi, (entity) => decodeEntities(entity))
525
522
  .replace(/\n{3,}/g, "\n\n")
526
523
  .trim();
527
524
  }
@@ -36,7 +36,8 @@ export function createCrawlReporter(options: { url: string; fetcher?: Fetch; tim
36
36
  method: "POST",
37
37
  headers: { "content-type": "application/json", "content-encoding": "gzip" },
38
38
  body: Bun.gzipSync(JSON.stringify(payload)),
39
- signal: AbortSignal.timeout(options.timeoutMs ?? 10_000),
39
+ // Big employers send tens of megabytes and the aggregator ingests one report at a time; give them room.
40
+ signal: AbortSignal.timeout(options.timeoutMs ?? 120_000),
40
41
  });
41
42
  if (!response.ok) throw new Error(`Aggregator rejected crawl report: HTTP ${response.status}`);
42
43
  };
package/src/crawler.ts CHANGED
@@ -30,7 +30,7 @@ interface CrawlerOptions {
30
30
  }
31
31
 
32
32
  export interface Crawler {
33
- crawl(sources: Company[], settings?: { prune?: boolean }): Promise<CrawlReport>;
33
+ crawl(sources: Company[], settings?: { prune?: boolean; workdayCountries?: string[] }): Promise<CrawlReport>;
34
34
  }
35
35
 
36
36
  export function createCrawler(options: CrawlerOptions): Crawler {
@@ -124,7 +124,8 @@ export function createCrawler(options: CrawlerOptions): Crawler {
124
124
  const listed = await options.fetchJobs(source, controller.signal, {
125
125
  onBackoff: ({ status, delayMs }) => { metric.backoffMs += delayMs; if (status === 429) metric.throttles += 1; },
126
126
  workdayPageDelayMs: options.workdayPageDelayMs,
127
- workdayCountries: options.workdayCountries,
127
+ // The countries this crawl asked for drive Workday's supplementary passes, not just the configured default.
128
+ workdayCountries: settings?.workdayCountries ?? options.workdayCountries,
128
129
  });
129
130
  const jobs = await withExperience(source, listed, partitionFor(partitions, source.slug)?.jobs);
130
131
  partitions[source.slug] = { fetchedAt: now().toISOString(), jobs };
@@ -159,7 +160,10 @@ export function createCrawler(options: CrawlerOptions): Crawler {
159
160
  await options.store.write({ version: 1, updatedAt: finishedAt, partitions, lastCrawl: report });
160
161
  if (options.onCrawled) {
161
162
  const crawled = selectedSources.filter((source) => metrics.get(source.slug)!.status === "succeeded");
162
- await Promise.all(crawled.map((source) => Promise.resolve().then(() => options.onCrawled!(source, partitions[source.slug]!)).catch(() => undefined)));
163
+ // Four at a time: sending a thousand reports at once made the largest ones time out while the aggregator queued them.
164
+ let next = 0;
165
+ const send = async () => { while (next < crawled.length) { const source = crawled[next++]!; await Promise.resolve().then(() => options.onCrawled!(source, partitions[source.slug]!)).catch(() => undefined); } };
166
+ await Promise.all(Array.from({ length: Math.min(4, crawled.length) }, send));
163
167
  }
164
168
  return report;
165
169
  },
package/src/experience.ts CHANGED
@@ -1,14 +1,18 @@
1
+ import { decodeEntities } from "./providers.ts";
2
+
1
3
  /** Years of experience a posting asks for, read from its own text. */
2
4
  export interface Experience { min: number; max?: number }
3
5
 
4
6
  const WORDS: Record<string, number> = { one: 1, two: 2, three: 3, four: 4, five: 5, six: 6, seven: 7, eight: 8, nine: 9, ten: 10, twelve: 12, fifteen: 15 };
5
7
  const NUM = String.raw`(\d{1,2}(?:\.\d)?|${Object.keys(WORDS).join("|")})`;
6
- const RANGE = String.raw`(?<![\d.])${NUM}\s*\+?\s*(?:(?:-|–|—|to)\s*${NUM}\s*)?\+?\s*(?:years?|yrs?)(?:'|’)?`;
8
+ const RANGE = String.raw`(?<![\d.])${NUM}\s*\+?\s*(?:(?:-|–|—|to)\s*${NUM}\s*)?\+?\s*(?:years?|yrs?)\+?(?:['’]s?)?`;
7
9
  // "5+ years of hands-on Java experience" or "Experience: 2-5 years" / "Years of experience 6 to 8 years".
8
- const YEARS_THEN_EXPERIENCE = new RegExp(String.raw`${RANGE}(?:\s+of)?(?:\s+[\w/&,'’()-]+){0,6}?\s+(?:experience|exp)\b`, "gi");
9
- const EXPERIENCE_THEN_YEARS = new RegExp(String.raw`\b(?:experience|exp)\b[^.\n\d]{0,30}?${RANGE}`, "gi");
10
+ const YEARS_THEN_EXPERIENCE = new RegExp(String.raw`${RANGE}(?:\s+of)?(?:\s+[\w/&,'’()-]+){0,10}?\s+(?:experiences?|exp)\b`, "gi");
11
+ const EXPERIENCE_THEN_YEARS = new RegExp(String.raw`\b(?:experiences?|exp)\b[^.\n\d]{0,30}?${RANGE}`, "gi");
12
+ // "minimum 15+ years" with no word "experience" after it.
13
+ const MINIMUM_YEARS = new RegExp(String.raw`\b(?:minimum|min\.?|at least|atleast)\s*(?:of\s*)?${RANGE}`, "gi");
10
14
  // Company history reads the same way ("our 135 years of experience"); skip matches that talk about the employer.
11
- const ABOUT_EMPLOYER = /\b(?:we|our|company|firm|founded|established|history|legacy|heritage)\b[^.\n?!:;•·-]*$/i;
15
+ const ABOUT_EMPLOYER = /\b(?:our (?:company|firm|group|organi[sz]ation|brands?|history|legacy)|(?:company|firm|group) (?:has|with)|founded|established|history|legacy|heritage)\b[^.\n?!:;•·-]*$/i;
12
16
  const MIN_PREFIX = /\b(?:minimum|min\.?|at least|atleast)\s*(?:of\s*)?$/i;
13
17
 
14
18
  function value(text: string | undefined): number | undefined {
@@ -17,16 +21,17 @@ function value(text: string | undefined): number | undefined {
17
21
  }
18
22
 
19
23
  /** The first experience requirement the text states, or null when it states none. */
20
- export function statedExperience(text: string): Experience | null {
24
+ export function statedExperience(description: string): Experience | null {
25
+ const text = decodeEntities(description);
21
26
  const found: Array<{ index: number; experience: Experience }> = [];
22
- for (const pattern of [YEARS_THEN_EXPERIENCE, EXPERIENCE_THEN_YEARS]) {
27
+ for (const pattern of [YEARS_THEN_EXPERIENCE, EXPERIENCE_THEN_YEARS, MINIMUM_YEARS]) {
23
28
  for (const match of text.matchAll(pattern)) {
24
29
  const [low, high] = [value(match[1]), value(match[2])];
25
30
  if (low === undefined || !Number.isFinite(low)) continue;
26
31
  // Requirements rarely pass 20 years; bigger numbers are nearly always an employer's history.
27
32
  if (low > 20 || (high !== undefined && (high < low || high > 40))) continue;
28
33
  const before = text.slice(Math.max(0, match.index - 60), match.index);
29
- if (ABOUT_EMPLOYER.test(before) && !MIN_PREFIX.test(before)) continue;
34
+ if (pattern !== MINIMUM_YEARS && ABOUT_EMPLOYER.test(before) && !MIN_PREFIX.test(before)) continue;
30
35
  found.push({ index: match.index, experience: high !== undefined && high !== low ? { min: low, max: high } : { min: low } });
31
36
  }
32
37
  }
@@ -211,7 +211,7 @@ function refreshReason(policy: RefreshPolicy, snapshot: JobSnapshot | null, matc
211
211
  }
212
212
 
213
213
  function refreshScope(intent: CandidateIntent, snapshot: JobSnapshot | null, sources: Company[]): CrawlScope {
214
- return intent.countries?.length ? { slugs: relevantSourceSlugs(snapshot, intent, sources) } : {};
214
+ return intent.countries?.length ? { slugs: relevantSourceSlugs(snapshot, intent, sources), countries: intent.countries } : {};
215
215
  }
216
216
 
217
217
  function relevantSnapshot(snapshot: JobSnapshot, intent: CandidateIntent, sources: Company[]): JobSnapshot {
@@ -38,7 +38,7 @@ export function createJobSearchPreparer(options: {
38
38
  }
39
39
  const pendingBefore = pendingSources(options.sources, snapshot, now(), freshnessMs);
40
40
  const selected = pendingBefore.filter((source) => !input.attempted.has(source.slug)).slice(0, batchSize).map((source) => source.slug);
41
- const crawl = selected.length ? await options.crawl({ slugs: selected }) : undefined;
41
+ const crawl = selected.length ? await options.crawl({ slugs: selected, countries }) : undefined;
42
42
  if (crawl) snapshot = await options.store.read();
43
43
  if (!snapshot) throw new Error("Job search preparation did not produce a local snapshot");
44
44
 
@@ -1,5 +1,5 @@
1
1
  import { classifyJob } from "./locations.ts";
2
- import { countryLabel } from "./providers.ts";
2
+ import { countryLabel, decodeEntities } from "./providers.ts";
3
3
  import { extractLinks, fetchSafePage, robotsAllows, type PageTransport, type ResolveHost } from "./safe-head.ts";
4
4
  import type { Job } from "./types.ts";
5
5
 
@@ -158,5 +158,5 @@ export function sitePostingsToJobs(company: { slug: string; name: string }, post
158
158
  function hash(value: string): string { let h = 2166136261; for (const ch of value) { h ^= ch.charCodeAt(0); h = Math.imul(h, 16777619) >>> 0; } return h.toString(16); }
159
159
  function text(value: unknown): string { return typeof value === "string" ? value.trim() : typeof value === "number" ? String(value) : ""; }
160
160
  function isoDate(value: unknown): string | undefined { const time = Date.parse(text(value)); return Number.isFinite(time) ? new Date(time).toISOString() : undefined; }
161
- function stripHtml(value: string): string { return value.replace(/<[^>]+>/g, " ").replace(/&nbsp;/g, " ").replace(/&amp;/g, "&").replace(/&lt;/g, "<").replace(/&gt;/g, ">").replace(/\s+/g, " ").trim(); }
161
+ function stripHtml(value: string): string { return decodeEntities(value.replace(/<[^>]+>/g, " ")).replace(/\s+/g, " ").trim(); }
162
162
  function isRecord(value: unknown): value is Record<string, unknown> { return typeof value === "object" && value !== null && !Array.isArray(value); }
package/src/local-jobs.ts CHANGED
@@ -44,7 +44,10 @@ export function createLocalJobs(options: LocalJobsOptions) {
44
44
 
45
45
  async function crawl(scope: CrawlScope = {}): Promise<CrawlReport> {
46
46
  const selected = selectSources(options.sources, scope);
47
- return crawler.crawl(selected, { prune: !scope.country && !scope.countries && !scope.slugs });
47
+ // Workday's country passes cover the countries this crawl is for, on top of the configured ones.
48
+ const asked = [...(scope.countries ?? []), ...(scope.country ? [scope.country] : [])].map((code) => code.toUpperCase());
49
+ const workdayCountries = asked.length ? [...new Set([...(options.workdayCountries ?? []), ...asked])] : undefined;
50
+ return crawler.crawl(selected, { prune: !scope.country && !scope.countries && !scope.slugs, ...(workdayCountries ? { workdayCountries } : {}) });
48
51
  }
49
52
 
50
53
  async function ensureFresh(offline: boolean, staleDays: number): Promise<{ snapshot: JobSnapshot; refreshed: boolean }> {
package/src/providers.ts CHANGED
@@ -294,10 +294,20 @@ function validToken(value: string | undefined): string | null {
294
294
  return value && /^[A-Za-z0-9][A-Za-z0-9._-]*$/.test(value) && !RESERVED_TOKENS.has(value.toLowerCase()) ? value : null;
295
295
  }
296
296
 
297
+ const NAMED_ENTITIES: Record<string, string> = { nbsp: " ", amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", ndash: "–", mdash: "—", lsquo: "‘", rsquo: "’", ldquo: "“", rdquo: "”", bull: "•", hellip: "…" };
298
+ /** Decode HTML entities in one pass (Workday sends "4&#43; years"), so "&amp;#43;" stays "&#43;" instead of becoming "+". */
299
+ export function decodeEntities(value: string): string {
300
+ return value.replace(/&(#\d{1,7}|#x[0-9a-f]{1,6}|[a-z]{2,8});/gi, (entity, code: string) => {
301
+ if (code[0] !== "#") return NAMED_ENTITIES[code.toLowerCase()] ?? entity;
302
+ const point = code[1] === "x" || code[1] === "X" ? parseInt(code.slice(2), 16) : Number(code.slice(1));
303
+ return point > 0 && point <= 0x10ffff ? String.fromCodePoint(point) : entity;
304
+ });
305
+ }
306
+
297
307
  export function plainText(value: string): string {
298
308
  return value
299
309
  .replace(/<br\s*\/?\s*>/gi, "\n").replace(/<\/(p|li|div|h[1-6])>/gi, "\n").replace(/<[^>]+>/g, "")
300
- .replace(/&nbsp;/g, " ").replace(/&amp;/g, "&").replace(/&lt;/g, "<").replace(/&gt;/g, ">").replace(/&quot;/g, '"').replace(/&#39;/g, "'")
310
+ .replace(/&[#a-z0-9]+;/gi, (entity) => decodeEntities(entity))
301
311
  .replace(/[ \t]+\n/g, "\n").replace(/\n{3,}/g, "\n\n").trim();
302
312
  }
303
313
 
package/src/version.ts CHANGED
@@ -1 +1 @@
1
- export const VERSION = "0.1.32";
1
+ export const VERSION = "0.1.34";