openings 0.1.47 → 0.1.48

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/package.json +1 -1
  2. package/src/skills.ts +28 -12
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openings",
3
- "version": "0.1.47",
3
+ "version": "0.1.48",
4
4
  "description": "A free, candidate-safe job-search substrate for AI agents",
5
5
  "license": "MIT",
6
6
  "repository": { "type": "git", "url": "git+https://github.com/abhay-avagama/hiring-agent.git" },
package/src/skills.ts CHANGED
@@ -13,6 +13,16 @@ import { searchTokens } from "./text-match.ts";
13
13
  * practice someone searches for, not by appearing often. Two-word entries match only as adjacent words.
14
14
  * Left out on purpose: bare "c", "go" and ".net", which collide with ordinary prose, and employer names such
15
15
  * as Oracle and Workday, which would tag every role at that employer.
16
+ *
17
+ * Measured and withdrawn after the first panel run: "data warehouse", "data pipeline", "test automation" and
18
+ * "automation testing" each contributed an ordinary word ("data", "test", "automation") that a candidate types
19
+ * as a modifier, so "data engineer" reached any Software Engineer whose description mentioned a data pipeline.
20
+ * A term whose words a query uses as modifiers cannot come from the description. Gone for the same reason:
21
+ * "express", "spring", "embedded", "plc" and "agile"/"scrum"/"jira", which are ordinary words or boilerplate.
22
+ *
23
+ * The vocabulary holds tools, never the name of a role. "DevOps" and "Android" in a description say the job
24
+ * touches them; a sample of what they reached was support and management roles, not DevOps or Android jobs.
25
+ * A role's own title is the evidence that it is that role, and search ranks a title match above a mention.
16
26
  */
17
27
  const VOCABULARY = [
18
28
  // languages
@@ -22,32 +32,32 @@ const VOCABULARY = [
22
32
  "react", "angular", "vue", "svelte", "next.js", "redux", "jquery", "html", "css", "sass", "tailwind",
23
33
  "webpack", "bootstrap", "wordpress", "shopify",
24
34
  // mobile
25
- "android", "ios", "react native", "flutter", "swiftui", "jetpack compose", "xcode", "ionic", "cordova", "xamarin",
35
+ "react native", "flutter", "swiftui", "jetpack compose", "xcode", "ionic", "cordova", "xamarin",
26
36
  // backend
27
- "node.js", "express", "django", "flask", "fastapi", "spring", "spring boot", "hibernate", "laravel", "rails",
37
+ "node.js", "django", "flask", "fastapi", "spring boot", "hibernate", "laravel", "rails",
28
38
  "asp.net", "graphql", "grpc", "microservices", "kafka", "rabbitmq", "celery", "websocket",
29
39
  // data
30
40
  "mysql", "postgresql", "postgres", "mongodb", "cassandra", "redis", "elasticsearch", "snowflake", "databricks",
31
41
  "hadoop", "spark", "hive", "airflow", "etl", "dbt", "tableau", "power bi", "looker", "bigquery", "redshift",
32
- "data warehouse", "data pipeline",
42
+
33
43
  // machine learning
34
44
  "machine learning", "deep learning", "tensorflow", "pytorch", "keras", "nlp", "computer vision", "llm",
35
- "generative ai", "huggingface", "scikit", "pandas", "numpy", "opencv", "mlops",
45
+ "generative ai", "huggingface", "scikit", "pandas", "numpy", "opencv",
36
46
  // cloud and infrastructure
37
47
  "aws", "azure", "gcp", "kubernetes", "docker", "terraform", "ansible", "jenkins", "gitlab", "github",
38
48
  "ci cd", "cicd", "helm", "prometheus", "grafana", "datadog", "splunk", "linux", "nginx", "openshift",
39
- "cloudformation", "serverless", "devops", "kibana",
49
+ "cloudformation", "serverless", "kibana",
40
50
  // testing
41
51
  "selenium", "cypress", "playwright", "appium", "junit", "testng", "pytest", "jmeter", "postman",
42
- "test automation", "automation testing",
52
+
43
53
  // security
44
54
  "owasp", "penetration testing", "cryptography", "iam", "soc 2",
45
55
  // embedded and hardware
46
- "embedded", "rtos", "verilog", "vhdl", "firmware", "autosar", "fpga", "plc",
56
+ "rtos", "verilog", "vhdl", "firmware", "autosar", "fpga",
47
57
  // enterprise platforms
48
58
  "sap", "abap", "salesforce", "servicenow", "sharepoint", "mulesoft", "sapui5",
49
59
  // practice
50
- "agile", "scrum", "jira", "git", "kubernetes", "figma",
60
+ "git", "figma",
51
61
  ] as const;
52
62
 
53
63
  /** Vocabulary reduced to the same stemmed tokens the matcher produces, so stored terms and queries agree. */
@@ -63,15 +73,21 @@ for (const term of VOCABULARY) {
63
73
  * The vocabulary terms this text mentions, as a space-separated string ready to append to the haystack.
64
74
  * Phrases contribute their own words: matching is term by term, so "react native" is stored as both.
65
75
  */
76
+ /** A term named once is a passing reference; a requirement is restated. The count was measured, not guessed. */
77
+ const MENTIONS_REQUIRED = 2;
78
+
66
79
  export function extractSkills(text: string): string {
67
80
  if (!text) return "";
68
81
  const tokens = searchTokens(text);
69
- const hits = new Set<string>();
82
+ const counts = new Map<string, number>();
83
+ const count = (term: string) => counts.set(term, (counts.get(term) ?? 0) + 1);
70
84
  for (let index = 0; index < tokens.length; index += 1) {
71
85
  const token = tokens[index]!;
72
- if (UNIGRAMS.has(token)) hits.add(token);
86
+ if (UNIGRAMS.has(token)) count(token);
73
87
  const next = tokens[index + 1];
74
- if (next !== undefined && BIGRAMS.has(`${token} ${next}`)) { hits.add(token); hits.add(next); }
88
+ if (next !== undefined && BIGRAMS.has(`${token} ${next}`)) { count(token); count(next); }
75
89
  }
76
- return [...hits].join(" ");
90
+ const hits: string[] = [];
91
+ for (const [term, seen] of counts) if (seen >= MENTIONS_REQUIRED) hits.push(term);
92
+ return hits.join(" ");
77
93
  }