openings 0.1.45 → 0.1.47

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openings",
3
- "version": "0.1.45",
3
+ "version": "0.1.46",
4
4
  "description": "Find evidence-grounded jobs, including relevant roles you may not have searched for, without accounts or API keys.",
5
5
  "author": { "name": "Openings contributors" },
6
6
  "license": "MIT",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openings",
3
- "version": "0.1.45",
3
+ "version": "0.1.47",
4
4
  "description": "A free, candidate-safe job-search substrate for AI agents",
5
5
  "license": "MIT",
6
6
  "repository": { "type": "git", "url": "git+https://github.com/abhay-avagama/hiring-agent.git" },
package/src/catalog.ts CHANGED
@@ -1,6 +1,7 @@
1
1
  import { CASCADE_MINIMUM, CASCADE_WINDOWS, type Company, type Job, type JobAge, type JobSummary, type SearchQuery, type SearchWindow } from "./types.ts";
2
2
  import { decodeEntities, providerSpec, type JsonGet } from "./providers.ts";
3
3
  import { matchesSearchTerms, searchHaystack, searchTerms } from "./text-match.ts";
4
+ import { extractSkills } from "./skills.ts";
4
5
  import { crawlSite, sitePostingsToJobs } from "./jobposting-site.ts";
5
6
  import { classifyJob, isEligibleForCountry, normalizeLocation } from "./locations.ts";
6
7
  import { experienceMatches, normalizeJobExperience } from "./experience.ts";
@@ -518,7 +519,9 @@ function matches(job: Job, query: SearchQuery): boolean {
518
519
  const terms = searchTerms(query.query);
519
520
  const location = query.location ? normalizeLocation(query.location) : undefined;
520
521
  // Whole tokens only: a substring match once made "ios" find "Axio Biosolutions".
521
- const searchable = searchHaystack(`${job.title} ${job.company}`);
522
+ // A skill stated only in the description is what most searches are for, so the description's vocabulary
523
+ // terms join the title and employer. The hosted index sends them precomputed; a local crawl reads them here.
524
+ const searchable = searchHaystack(`${job.title} ${job.company} ${job.skills ?? extractSkills(job.description)}`);
522
525
  return (terms.length === 0 || matchesSearchTerms(searchable, terms))
523
526
  && (!location || normalizeLocation(job.location).includes(location))
524
527
  && (!query.country || isEligibleForCountry(job, query.country))
@@ -27,19 +27,37 @@ export function resolveAggregatorUrl(value: string | undefined): string | undefi
27
27
  }
28
28
 
29
29
  /** Posts each successfully crawled partition to the aggregator. Only job data is sent, never resume content. */
30
- export function createCrawlReporter(options: { url: string; fetcher?: Fetch; timeoutMs?: number }) {
30
+ export function createCrawlReporter(options: { url: string; fetcher?: Fetch; timeoutMs?: number; attempts?: number; sleep?: (ms: number) => Promise<void> }) {
31
31
  const fetcher = options.fetcher ?? globalThis.fetch;
32
32
  const target = endpoint(options.url, "v1/crawls");
33
+ const attempts = Math.max(1, options.attempts ?? 3);
34
+ const sleep = options.sleep ?? ((ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms)));
33
35
  return async (source: Company, partition: JobPartition): Promise<void> => {
34
36
  const payload: CrawlReportPayload = { version: 1, source: { slug: source.slug, ats: source.ats, token: source.token }, fetchedAt: partition.fetchedAt, jobs: partition.jobs };
35
- const response = await fetcher(target, {
36
- method: "POST",
37
- headers: { "content-type": "application/json", "content-encoding": "gzip" },
38
- body: Bun.gzipSync(JSON.stringify(payload)),
39
- // Big employers send tens of megabytes and the aggregator ingests one report at a time; give them room.
40
- signal: AbortSignal.timeout(options.timeoutMs ?? 120_000),
41
- });
42
- if (!response.ok) throw new Error(`Aggregator rejected crawl report: HTTP ${response.status}`);
37
+ const body = Bun.gzipSync(JSON.stringify(payload));
38
+ let last: Error | undefined;
39
+ // A crawl that ends while the aggregator is restarting used to lose its whole report in silence.
40
+ // Retry the failures worth retrying: a refused connection, a timeout, or the server being briefly unwell.
41
+ for (let attempt = 1; attempt <= attempts; attempt += 1) {
42
+ try {
43
+ const response = await fetcher(target, {
44
+ method: "POST",
45
+ headers: { "content-type": "application/json", "content-encoding": "gzip" },
46
+ body,
47
+ // Big employers send tens of megabytes and the aggregator ingests one report at a time; give them room.
48
+ signal: AbortSignal.timeout(options.timeoutMs ?? 120_000),
49
+ });
50
+ if (response.ok) return;
51
+ // A rejected report will be rejected again: the catalog disagrees, or the payload is too large.
52
+ if (response.status < 500) throw new Error(`Aggregator rejected crawl report: HTTP ${response.status}`);
53
+ last = new Error(`Aggregator returned HTTP ${response.status}`);
54
+ } catch (error) {
55
+ if (error instanceof Error && error.message.startsWith("Aggregator rejected")) throw error;
56
+ last = error instanceof Error ? error : new Error(String(error));
57
+ }
58
+ if (attempt < attempts) await sleep(attempt * 5_000);
59
+ }
60
+ throw last ?? new Error("Crawl report failed");
43
61
  };
44
62
  }
45
63
 
package/src/index.ts CHANGED
@@ -19,6 +19,7 @@ export const catalog = createCatalog({ companies });
19
19
  export { createCatalog, searchJobs } from "./catalog.ts";
20
20
  export { EXPERIENCE_VERSION, experienceLabel, experienceMatches, normalizeJobExperience, statedExperience, titleExperience, type Experience } from "./experience.ts";
21
21
  export { matchesSearchTerms, searchHaystack, searchTerms, searchTokens } from "./text-match.ts";
22
+ export { extractSkills } from "./skills.ts";
22
23
  export type { Catalog } from "./catalog.ts";
23
24
  export type { Ats, Company, CrawlReport, Job, JobPartition, JobSnapshot, JobSummary, SearchQuery } from "./types.ts";
24
25
  export { createCrawlReporter, fetchSeedSnapshot, resolveAggregatorUrl } from "./crawl-reporting.ts";
package/src/skills.ts ADDED
@@ -0,0 +1,77 @@
1
+ /**
2
+ * Skill tokens lifted out of a job description so search can find them.
3
+ *
4
+ * Search matches a job by its title and employer only, which is why a "react native" search returned 78 India
5
+ * roles while 171 more carried the term only in their description. Holding the descriptions themselves in the
6
+ * index is not an option (334 MB across live India and US roles, against 11 MB of titles), so each role keeps
7
+ * the handful of vocabulary terms its description mentions: a few dozen bytes, and none of the boilerplate.
8
+ */
9
+ import { searchTokens } from "./text-match.ts";
10
+
11
+ /**
12
+ * Terms a candidate actually types. Deliberately narrow: a term earns its place by being a technology or
13
+ * practice someone searches for, not by appearing often. Two-word entries match only as adjacent words.
14
+ * Left out on purpose: bare "c", "go" and ".net", which collide with ordinary prose, and employer names such
15
+ * as Oracle and Workday, which would tag every role at that employer.
16
+ */
17
+ const VOCABULARY = [
18
+ // languages
19
+ "java", "python", "javascript", "typescript", "golang", "ruby", "php", "scala", "kotlin", "swift", "rust",
20
+ "perl", "matlab", "c++", "c#", "objective c", "dart", "elixir", "haskell", "groovy", "sql", "plsql", "pl sql",
21
+ // web and frontend
22
+ "react", "angular", "vue", "svelte", "next.js", "redux", "jquery", "html", "css", "sass", "tailwind",
23
+ "webpack", "bootstrap", "wordpress", "shopify",
24
+ // mobile
25
+ "android", "ios", "react native", "flutter", "swiftui", "jetpack compose", "xcode", "ionic", "cordova", "xamarin",
26
+ // backend
27
+ "node.js", "express", "django", "flask", "fastapi", "spring", "spring boot", "hibernate", "laravel", "rails",
28
+ "asp.net", "graphql", "grpc", "microservices", "kafka", "rabbitmq", "celery", "websocket",
29
+ // data
30
+ "mysql", "postgresql", "postgres", "mongodb", "cassandra", "redis", "elasticsearch", "snowflake", "databricks",
31
+ "hadoop", "spark", "hive", "airflow", "etl", "dbt", "tableau", "power bi", "looker", "bigquery", "redshift",
32
+ "data warehouse", "data pipeline",
33
+ // machine learning
34
+ "machine learning", "deep learning", "tensorflow", "pytorch", "keras", "nlp", "computer vision", "llm",
35
+ "generative ai", "huggingface", "scikit", "pandas", "numpy", "opencv", "mlops",
36
+ // cloud and infrastructure
37
+ "aws", "azure", "gcp", "kubernetes", "docker", "terraform", "ansible", "jenkins", "gitlab", "github",
38
+ "ci cd", "cicd", "helm", "prometheus", "grafana", "datadog", "splunk", "linux", "nginx", "openshift",
39
+ "cloudformation", "serverless", "devops", "kibana",
40
+ // testing
41
+ "selenium", "cypress", "playwright", "appium", "junit", "testng", "pytest", "jmeter", "postman",
42
+ "test automation", "automation testing",
43
+ // security
44
+ "owasp", "penetration testing", "cryptography", "iam", "soc 2",
45
+ // embedded and hardware
46
+ "embedded", "rtos", "verilog", "vhdl", "firmware", "autosar", "fpga", "plc",
47
+ // enterprise platforms
48
+ "sap", "abap", "salesforce", "servicenow", "sharepoint", "mulesoft", "sapui5",
49
+ // practice
50
+ "agile", "scrum", "jira", "git", "kubernetes", "figma",
51
+ ] as const;
52
+
53
+ /** Vocabulary reduced to the same stemmed tokens the matcher produces, so stored terms and queries agree. */
54
+ const UNIGRAMS = new Set<string>();
55
+ const BIGRAMS = new Set<string>();
56
+ for (const term of VOCABULARY) {
57
+ const tokens = searchTokens(term);
58
+ if (tokens.length === 1) UNIGRAMS.add(tokens[0]!);
59
+ else if (tokens.length === 2) BIGRAMS.add(`${tokens[0]} ${tokens[1]}`);
60
+ }
61
+
62
+ /**
63
+ * The vocabulary terms this text mentions, as a space-separated string ready to append to the haystack.
64
+ * Phrases contribute their own words: matching is term by term, so "react native" is stored as both.
65
+ */
66
+ export function extractSkills(text: string): string {
67
+ if (!text) return "";
68
+ const tokens = searchTokens(text);
69
+ const hits = new Set<string>();
70
+ for (let index = 0; index < tokens.length; index += 1) {
71
+ const token = tokens[index]!;
72
+ if (UNIGRAMS.has(token)) hits.add(token);
73
+ const next = tokens[index + 1];
74
+ if (next !== undefined && BIGRAMS.has(`${token} ${next}`)) { hits.add(token); hits.add(next); }
75
+ }
76
+ return [...hits].join(" ");
77
+ }
package/src/types.ts CHANGED
@@ -102,6 +102,9 @@ export const CASCADE_MINIMUM = 5;
102
102
 
103
103
  export interface Job extends JobSummary {
104
104
  description: string;
105
+ /** Vocabulary terms read out of the description, so search finds a skill the title never mentions. Set by the
106
+ * hosted index, which carries these in place of the descriptions themselves; absent means read the description. */
107
+ skills?: string;
105
108
  }
106
109
 
107
110
  /** Partition lookup that ignores inherited properties, so a slug such as "constructor" never resolves to Object.prototype. */
package/src/version.ts CHANGED
@@ -1 +1 @@
1
- export const VERSION = "0.1.45";
1
+ export const VERSION = "0.1.46";