openings 0.1.47 → 0.1.48
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/skills.ts +28 -12
package/package.json
CHANGED
package/src/skills.ts
CHANGED
|
@@ -13,6 +13,16 @@ import { searchTokens } from "./text-match.ts";
|
|
|
13
13
|
* practice someone searches for, not by appearing often. Two-word entries match only as adjacent words.
|
|
14
14
|
* Left out on purpose: bare "c", "go" and ".net", which collide with ordinary prose, and employer names such
|
|
15
15
|
* as Oracle and Workday, which would tag every role at that employer.
|
|
16
|
+
*
|
|
17
|
+
* Measured and withdrawn after the first panel run: "data warehouse", "data pipeline", "test automation" and
|
|
18
|
+
* "automation testing" each contributed an ordinary word ("data", "test", "automation") that a candidate types
|
|
19
|
+
* as a modifier, so "data engineer" reached any Software Engineer whose description mentioned a data pipeline.
|
|
20
|
+
* A term whose words a query uses as modifiers cannot come from the description. Gone for the same reason:
|
|
21
|
+
* "express", "spring", "embedded", "plc" and "agile"/"scrum"/"jira", which are ordinary words or boilerplate.
|
|
22
|
+
*
|
|
23
|
+
* The vocabulary holds tools, never the name of a role. "DevOps" and "Android" in a description say the job
|
|
24
|
+
* touches them; a sample of what they reached was support and management roles, not DevOps or Android jobs.
|
|
25
|
+
* A role's own title is the evidence that it is that role, and search ranks a title match above a mention.
|
|
16
26
|
*/
|
|
17
27
|
const VOCABULARY = [
|
|
18
28
|
// languages
|
|
@@ -22,32 +32,32 @@ const VOCABULARY = [
|
|
|
22
32
|
"react", "angular", "vue", "svelte", "next.js", "redux", "jquery", "html", "css", "sass", "tailwind",
|
|
23
33
|
"webpack", "bootstrap", "wordpress", "shopify",
|
|
24
34
|
// mobile
|
|
25
|
-
"
|
|
35
|
+
"react native", "flutter", "swiftui", "jetpack compose", "xcode", "ionic", "cordova", "xamarin",
|
|
26
36
|
// backend
|
|
27
|
-
"node.js", "
|
|
37
|
+
"node.js", "django", "flask", "fastapi", "spring boot", "hibernate", "laravel", "rails",
|
|
28
38
|
"asp.net", "graphql", "grpc", "microservices", "kafka", "rabbitmq", "celery", "websocket",
|
|
29
39
|
// data
|
|
30
40
|
"mysql", "postgresql", "postgres", "mongodb", "cassandra", "redis", "elasticsearch", "snowflake", "databricks",
|
|
31
41
|
"hadoop", "spark", "hive", "airflow", "etl", "dbt", "tableau", "power bi", "looker", "bigquery", "redshift",
|
|
32
|
-
|
|
42
|
+
|
|
33
43
|
// machine learning
|
|
34
44
|
"machine learning", "deep learning", "tensorflow", "pytorch", "keras", "nlp", "computer vision", "llm",
|
|
35
|
-
"generative ai", "huggingface", "scikit", "pandas", "numpy", "opencv",
|
|
45
|
+
"generative ai", "huggingface", "scikit", "pandas", "numpy", "opencv",
|
|
36
46
|
// cloud and infrastructure
|
|
37
47
|
"aws", "azure", "gcp", "kubernetes", "docker", "terraform", "ansible", "jenkins", "gitlab", "github",
|
|
38
48
|
"ci cd", "cicd", "helm", "prometheus", "grafana", "datadog", "splunk", "linux", "nginx", "openshift",
|
|
39
|
-
"cloudformation", "serverless", "
|
|
49
|
+
"cloudformation", "serverless", "kibana",
|
|
40
50
|
// testing
|
|
41
51
|
"selenium", "cypress", "playwright", "appium", "junit", "testng", "pytest", "jmeter", "postman",
|
|
42
|
-
|
|
52
|
+
|
|
43
53
|
// security
|
|
44
54
|
"owasp", "penetration testing", "cryptography", "iam", "soc 2",
|
|
45
55
|
// embedded and hardware
|
|
46
|
-
"
|
|
56
|
+
"rtos", "verilog", "vhdl", "firmware", "autosar", "fpga",
|
|
47
57
|
// enterprise platforms
|
|
48
58
|
"sap", "abap", "salesforce", "servicenow", "sharepoint", "mulesoft", "sapui5",
|
|
49
59
|
// practice
|
|
50
|
-
"
|
|
60
|
+
"git", "figma",
|
|
51
61
|
] as const;
|
|
52
62
|
|
|
53
63
|
/** Vocabulary reduced to the same stemmed tokens the matcher produces, so stored terms and queries agree. */
|
|
@@ -63,15 +73,21 @@ for (const term of VOCABULARY) {
|
|
|
63
73
|
* The vocabulary terms this text mentions, as a space-separated string ready to append to the haystack.
|
|
64
74
|
* Phrases contribute their own words: matching is term by term, so "react native" is stored as both.
|
|
65
75
|
*/
|
|
76
|
+
/** A term named once is a passing reference; a requirement is restated. The count was measured, not guessed. */
|
|
77
|
+
const MENTIONS_REQUIRED = 2;
|
|
78
|
+
|
|
66
79
|
export function extractSkills(text: string): string {
|
|
67
80
|
if (!text) return "";
|
|
68
81
|
const tokens = searchTokens(text);
|
|
69
|
-
const
|
|
82
|
+
const counts = new Map<string, number>();
|
|
83
|
+
const count = (term: string) => counts.set(term, (counts.get(term) ?? 0) + 1);
|
|
70
84
|
for (let index = 0; index < tokens.length; index += 1) {
|
|
71
85
|
const token = tokens[index]!;
|
|
72
|
-
if (UNIGRAMS.has(token))
|
|
86
|
+
if (UNIGRAMS.has(token)) count(token);
|
|
73
87
|
const next = tokens[index + 1];
|
|
74
|
-
if (next !== undefined && BIGRAMS.has(`${token} ${next}`)) {
|
|
88
|
+
if (next !== undefined && BIGRAMS.has(`${token} ${next}`)) { count(token); count(next); }
|
|
75
89
|
}
|
|
76
|
-
|
|
90
|
+
const hits: string[] = [];
|
|
91
|
+
for (const [term, seen] of counts) if (seen >= MENTIONS_REQUIRED) hits.push(term);
|
|
92
|
+
return hits.join(" ");
|
|
77
93
|
}
|