openings 0.1.46 → 0.1.48

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openings",
3
- "version": "0.1.46",
3
+ "version": "0.1.48",
4
4
  "description": "A free, candidate-safe job-search substrate for AI agents",
5
5
  "license": "MIT",
6
6
  "repository": { "type": "git", "url": "git+https://github.com/abhay-avagama/hiring-agent.git" },
package/src/catalog.ts CHANGED
@@ -1,6 +1,7 @@
1
1
  import { CASCADE_MINIMUM, CASCADE_WINDOWS, type Company, type Job, type JobAge, type JobSummary, type SearchQuery, type SearchWindow } from "./types.ts";
2
2
  import { decodeEntities, providerSpec, type JsonGet } from "./providers.ts";
3
3
  import { matchesSearchTerms, searchHaystack, searchTerms } from "./text-match.ts";
4
+ import { extractSkills } from "./skills.ts";
4
5
  import { crawlSite, sitePostingsToJobs } from "./jobposting-site.ts";
5
6
  import { classifyJob, isEligibleForCountry, normalizeLocation } from "./locations.ts";
6
7
  import { experienceMatches, normalizeJobExperience } from "./experience.ts";
@@ -518,7 +519,9 @@ function matches(job: Job, query: SearchQuery): boolean {
518
519
  const terms = searchTerms(query.query);
519
520
  const location = query.location ? normalizeLocation(query.location) : undefined;
520
521
  // Whole tokens only: a substring match once made "ios" find "Axio Biosolutions".
521
- const searchable = searchHaystack(`${job.title} ${job.company}`);
522
+ // A skill stated only in the description is what most searches are for, so the description's vocabulary
523
+ // terms join the title and employer. The hosted index sends them precomputed; a local crawl reads them here.
524
+ const searchable = searchHaystack(`${job.title} ${job.company} ${job.skills ?? extractSkills(job.description)}`);
522
525
  return (terms.length === 0 || matchesSearchTerms(searchable, terms))
523
526
  && (!location || normalizeLocation(job.location).includes(location))
524
527
  && (!query.country || isEligibleForCountry(job, query.country))
package/src/index.ts CHANGED
@@ -19,6 +19,7 @@ export const catalog = createCatalog({ companies });
19
19
  export { createCatalog, searchJobs } from "./catalog.ts";
20
20
  export { EXPERIENCE_VERSION, experienceLabel, experienceMatches, normalizeJobExperience, statedExperience, titleExperience, type Experience } from "./experience.ts";
21
21
  export { matchesSearchTerms, searchHaystack, searchTerms, searchTokens } from "./text-match.ts";
22
+ export { extractSkills } from "./skills.ts";
22
23
  export type { Catalog } from "./catalog.ts";
23
24
  export type { Ats, Company, CrawlReport, Job, JobPartition, JobSnapshot, JobSummary, SearchQuery } from "./types.ts";
24
25
  export { createCrawlReporter, fetchSeedSnapshot, resolveAggregatorUrl } from "./crawl-reporting.ts";
package/src/skills.ts ADDED
@@ -0,0 +1,93 @@
1
+ /**
2
+ * Skill tokens lifted out of a job description so search can find them.
3
+ *
4
+ * Search matches a job by its title and employer only, which is why a "react native" search returned 78 India
5
+ * roles while 171 more carried the term only in their description. Holding the descriptions themselves in the
6
+ * index is not an option (334 MB across live India and US roles, against 11 MB of titles), so each role keeps
7
+ * the handful of vocabulary terms its description mentions: a few dozen bytes, and none of the boilerplate.
8
+ */
9
+ import { searchTokens } from "./text-match.ts";
10
+
11
+ /**
12
+ * Terms a candidate actually types. Deliberately narrow: a term earns its place by being a technology or
13
+ * practice someone searches for, not by appearing often. Two-word entries match only as adjacent words.
14
+ * Left out on purpose: bare "c", "go" and ".net", which collide with ordinary prose, and employer names such
15
+ * as Oracle and Workday, which would tag every role at that employer.
16
+ *
17
+ * Measured and withdrawn after the first panel run: "data warehouse", "data pipeline", "test automation" and
18
+ * "automation testing" each contributed an ordinary word ("data", "test", "automation") that a candidate types
19
+ * as a modifier, so "data engineer" reached any Software Engineer whose description mentioned a data pipeline.
20
+ * A term whose words a query uses as modifiers cannot come from the description. Gone for the same reason:
21
+ * "express", "spring", "embedded", "plc" and "agile"/"scrum"/"jira", which are ordinary words or boilerplate.
22
+ *
23
+ * The vocabulary holds tools, never the name of a role. "DevOps" and "Android" in a description say the job
24
+ * touches them; a sample of what they reached was support and management roles, not DevOps or Android jobs.
25
+ * A role's own title is the evidence that it is that role, and search ranks a title match above a mention.
26
+ */
27
+ const VOCABULARY = [
28
+ // languages
29
+ "java", "python", "javascript", "typescript", "golang", "ruby", "php", "scala", "kotlin", "swift", "rust",
30
+ "perl", "matlab", "c++", "c#", "objective c", "dart", "elixir", "haskell", "groovy", "sql", "plsql", "pl sql",
31
+ // web and frontend
32
+ "react", "angular", "vue", "svelte", "next.js", "redux", "jquery", "html", "css", "sass", "tailwind",
33
+ "webpack", "bootstrap", "wordpress", "shopify",
34
+ // mobile
35
+ "react native", "flutter", "swiftui", "jetpack compose", "xcode", "ionic", "cordova", "xamarin",
36
+ // backend
37
+ "node.js", "django", "flask", "fastapi", "spring boot", "hibernate", "laravel", "rails",
38
+ "asp.net", "graphql", "grpc", "microservices", "kafka", "rabbitmq", "celery", "websocket",
39
+ // data
40
+ "mysql", "postgresql", "postgres", "mongodb", "cassandra", "redis", "elasticsearch", "snowflake", "databricks",
41
+ "hadoop", "spark", "hive", "airflow", "etl", "dbt", "tableau", "power bi", "looker", "bigquery", "redshift",
42
+
43
+ // machine learning
44
+ "machine learning", "deep learning", "tensorflow", "pytorch", "keras", "nlp", "computer vision", "llm",
45
+ "generative ai", "huggingface", "scikit", "pandas", "numpy", "opencv",
46
+ // cloud and infrastructure
47
+ "aws", "azure", "gcp", "kubernetes", "docker", "terraform", "ansible", "jenkins", "gitlab", "github",
48
+ "ci cd", "cicd", "helm", "prometheus", "grafana", "datadog", "splunk", "linux", "nginx", "openshift",
49
+ "cloudformation", "serverless", "kibana",
50
+ // testing
51
+ "selenium", "cypress", "playwright", "appium", "junit", "testng", "pytest", "jmeter", "postman",
52
+
53
+ // security
54
+ "owasp", "penetration testing", "cryptography", "iam", "soc 2",
55
+ // embedded and hardware
56
+ "rtos", "verilog", "vhdl", "firmware", "autosar", "fpga",
57
+ // enterprise platforms
58
+ "sap", "abap", "salesforce", "servicenow", "sharepoint", "mulesoft", "sapui5",
59
+ // practice
60
+ "git", "figma",
61
+ ] as const;
62
+
63
+ /** Vocabulary reduced to the same stemmed tokens the matcher produces, so stored terms and queries agree. */
64
+ const UNIGRAMS = new Set<string>();
65
+ const BIGRAMS = new Set<string>();
66
+ for (const term of VOCABULARY) {
67
+ const tokens = searchTokens(term);
68
+ if (tokens.length === 1) UNIGRAMS.add(tokens[0]!);
69
+ else if (tokens.length === 2) BIGRAMS.add(`${tokens[0]} ${tokens[1]}`);
70
+ }
71
+
72
+ /**
73
+ * The vocabulary terms this text mentions, as a space-separated string ready to append to the haystack.
74
+ * Phrases contribute their own words: matching is term by term, so "react native" is stored as both.
75
+ */
76
+ /** A term named once is a passing reference; a requirement is restated. The count was measured, not guessed. */
77
+ const MENTIONS_REQUIRED = 2;
78
+
79
+ export function extractSkills(text: string): string {
80
+ if (!text) return "";
81
+ const tokens = searchTokens(text);
82
+ const counts = new Map<string, number>();
83
+ const count = (term: string) => counts.set(term, (counts.get(term) ?? 0) + 1);
84
+ for (let index = 0; index < tokens.length; index += 1) {
85
+ const token = tokens[index]!;
86
+ if (UNIGRAMS.has(token)) count(token);
87
+ const next = tokens[index + 1];
88
+ if (next !== undefined && BIGRAMS.has(`${token} ${next}`)) { count(token); count(next); }
89
+ }
90
+ const hits: string[] = [];
91
+ for (const [term, seen] of counts) if (seen >= MENTIONS_REQUIRED) hits.push(term);
92
+ return hits.join(" ");
93
+ }
package/src/types.ts CHANGED
@@ -102,6 +102,9 @@ export const CASCADE_MINIMUM = 5;
102
102
 
103
103
  export interface Job extends JobSummary {
104
104
  description: string;
105
+ /** Vocabulary terms read out of the description, so search finds a skill the title never mentions. Set by the
106
+ * hosted index, which carries these in place of the descriptions themselves; absent means read the description. */
107
+ skills?: string;
105
108
  }
106
109
 
107
110
  /** Partition lookup that ignores inherited properties, so a slug such as "constructor" never resolves to Object.prototype. */