openings 0.1.46 → 0.1.47
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/catalog.ts +4 -1
- package/src/index.ts +1 -0
- package/src/skills.ts +77 -0
- package/src/types.ts +3 -0
package/package.json
CHANGED
package/src/catalog.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { CASCADE_MINIMUM, CASCADE_WINDOWS, type Company, type Job, type JobAge, type JobSummary, type SearchQuery, type SearchWindow } from "./types.ts";
|
|
2
2
|
import { decodeEntities, providerSpec, type JsonGet } from "./providers.ts";
|
|
3
3
|
import { matchesSearchTerms, searchHaystack, searchTerms } from "./text-match.ts";
|
|
4
|
+
import { extractSkills } from "./skills.ts";
|
|
4
5
|
import { crawlSite, sitePostingsToJobs } from "./jobposting-site.ts";
|
|
5
6
|
import { classifyJob, isEligibleForCountry, normalizeLocation } from "./locations.ts";
|
|
6
7
|
import { experienceMatches, normalizeJobExperience } from "./experience.ts";
|
|
@@ -518,7 +519,9 @@ function matches(job: Job, query: SearchQuery): boolean {
|
|
|
518
519
|
const terms = searchTerms(query.query);
|
|
519
520
|
const location = query.location ? normalizeLocation(query.location) : undefined;
|
|
520
521
|
// Whole tokens only: a substring match once made "ios" find "Axio Biosolutions".
|
|
521
|
-
|
|
522
|
+
// A skill stated only in the description is what most searches are for, so the description's vocabulary
|
|
523
|
+
// terms join the title and employer. The hosted index sends them precomputed; a local crawl reads them here.
|
|
524
|
+
const searchable = searchHaystack(`${job.title} ${job.company} ${job.skills ?? extractSkills(job.description)}`);
|
|
522
525
|
return (terms.length === 0 || matchesSearchTerms(searchable, terms))
|
|
523
526
|
&& (!location || normalizeLocation(job.location).includes(location))
|
|
524
527
|
&& (!query.country || isEligibleForCountry(job, query.country))
|
package/src/index.ts
CHANGED
|
@@ -19,6 +19,7 @@ export const catalog = createCatalog({ companies });
|
|
|
19
19
|
export { createCatalog, searchJobs } from "./catalog.ts";
|
|
20
20
|
export { EXPERIENCE_VERSION, experienceLabel, experienceMatches, normalizeJobExperience, statedExperience, titleExperience, type Experience } from "./experience.ts";
|
|
21
21
|
export { matchesSearchTerms, searchHaystack, searchTerms, searchTokens } from "./text-match.ts";
|
|
22
|
+
export { extractSkills } from "./skills.ts";
|
|
22
23
|
export type { Catalog } from "./catalog.ts";
|
|
23
24
|
export type { Ats, Company, CrawlReport, Job, JobPartition, JobSnapshot, JobSummary, SearchQuery } from "./types.ts";
|
|
24
25
|
export { createCrawlReporter, fetchSeedSnapshot, resolveAggregatorUrl } from "./crawl-reporting.ts";
|
package/src/skills.ts
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Skill tokens lifted out of a job description so search can find them.
|
|
3
|
+
*
|
|
4
|
+
* Search matches a job by its title and employer only, which is why a "react native" search returned 78 India
|
|
5
|
+
* roles while 171 more carried the term only in their description. Holding the descriptions themselves in the
|
|
6
|
+
* index is not an option (334 MB across live India and US roles, against 11 MB of titles), so each role keeps
|
|
7
|
+
* the handful of vocabulary terms its description mentions: a few dozen bytes, and none of the boilerplate.
|
|
8
|
+
*/
|
|
9
|
+
import { searchTokens } from "./text-match.ts";
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Terms a candidate actually types. Deliberately narrow: a term earns its place by being a technology or
|
|
13
|
+
* practice someone searches for, not by appearing often. Two-word entries match only as adjacent words.
|
|
14
|
+
* Left out on purpose: bare "c", "go" and ".net", which collide with ordinary prose, and employer names such
|
|
15
|
+
* as Oracle and Workday, which would tag every role at that employer.
|
|
16
|
+
*/
|
|
17
|
+
const VOCABULARY = [
|
|
18
|
+
// languages
|
|
19
|
+
"java", "python", "javascript", "typescript", "golang", "ruby", "php", "scala", "kotlin", "swift", "rust",
|
|
20
|
+
"perl", "matlab", "c++", "c#", "objective c", "dart", "elixir", "haskell", "groovy", "sql", "plsql", "pl sql",
|
|
21
|
+
// web and frontend
|
|
22
|
+
"react", "angular", "vue", "svelte", "next.js", "redux", "jquery", "html", "css", "sass", "tailwind",
|
|
23
|
+
"webpack", "bootstrap", "wordpress", "shopify",
|
|
24
|
+
// mobile
|
|
25
|
+
"android", "ios", "react native", "flutter", "swiftui", "jetpack compose", "xcode", "ionic", "cordova", "xamarin",
|
|
26
|
+
// backend
|
|
27
|
+
"node.js", "express", "django", "flask", "fastapi", "spring", "spring boot", "hibernate", "laravel", "rails",
|
|
28
|
+
"asp.net", "graphql", "grpc", "microservices", "kafka", "rabbitmq", "celery", "websocket",
|
|
29
|
+
// data
|
|
30
|
+
"mysql", "postgresql", "postgres", "mongodb", "cassandra", "redis", "elasticsearch", "snowflake", "databricks",
|
|
31
|
+
"hadoop", "spark", "hive", "airflow", "etl", "dbt", "tableau", "power bi", "looker", "bigquery", "redshift",
|
|
32
|
+
"data warehouse", "data pipeline",
|
|
33
|
+
// machine learning
|
|
34
|
+
"machine learning", "deep learning", "tensorflow", "pytorch", "keras", "nlp", "computer vision", "llm",
|
|
35
|
+
"generative ai", "huggingface", "scikit", "pandas", "numpy", "opencv", "mlops",
|
|
36
|
+
// cloud and infrastructure
|
|
37
|
+
"aws", "azure", "gcp", "kubernetes", "docker", "terraform", "ansible", "jenkins", "gitlab", "github",
|
|
38
|
+
"ci cd", "cicd", "helm", "prometheus", "grafana", "datadog", "splunk", "linux", "nginx", "openshift",
|
|
39
|
+
"cloudformation", "serverless", "devops", "kibana",
|
|
40
|
+
// testing
|
|
41
|
+
"selenium", "cypress", "playwright", "appium", "junit", "testng", "pytest", "jmeter", "postman",
|
|
42
|
+
"test automation", "automation testing",
|
|
43
|
+
// security
|
|
44
|
+
"owasp", "penetration testing", "cryptography", "iam", "soc 2",
|
|
45
|
+
// embedded and hardware
|
|
46
|
+
"embedded", "rtos", "verilog", "vhdl", "firmware", "autosar", "fpga", "plc",
|
|
47
|
+
// enterprise platforms
|
|
48
|
+
"sap", "abap", "salesforce", "servicenow", "sharepoint", "mulesoft", "sapui5",
|
|
49
|
+
// practice
|
|
50
|
+
"agile", "scrum", "jira", "git", "kubernetes", "figma",
|
|
51
|
+
] as const;
|
|
52
|
+
|
|
53
|
+
/** Vocabulary reduced to the same stemmed tokens the matcher produces, so stored terms and queries agree. */
|
|
54
|
+
const UNIGRAMS = new Set<string>();
|
|
55
|
+
const BIGRAMS = new Set<string>();
|
|
56
|
+
for (const term of VOCABULARY) {
|
|
57
|
+
const tokens = searchTokens(term);
|
|
58
|
+
if (tokens.length === 1) UNIGRAMS.add(tokens[0]!);
|
|
59
|
+
else if (tokens.length === 2) BIGRAMS.add(`${tokens[0]} ${tokens[1]}`);
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* The vocabulary terms this text mentions, as a space-separated string ready to append to the haystack.
|
|
64
|
+
* Phrases contribute their own words: matching is term by term, so "react native" is stored as both.
|
|
65
|
+
*/
|
|
66
|
+
export function extractSkills(text: string): string {
|
|
67
|
+
if (!text) return "";
|
|
68
|
+
const tokens = searchTokens(text);
|
|
69
|
+
const hits = new Set<string>();
|
|
70
|
+
for (let index = 0; index < tokens.length; index += 1) {
|
|
71
|
+
const token = tokens[index]!;
|
|
72
|
+
if (UNIGRAMS.has(token)) hits.add(token);
|
|
73
|
+
const next = tokens[index + 1];
|
|
74
|
+
if (next !== undefined && BIGRAMS.has(`${token} ${next}`)) { hits.add(token); hits.add(next); }
|
|
75
|
+
}
|
|
76
|
+
return [...hits].join(" ");
|
|
77
|
+
}
|
package/src/types.ts
CHANGED
|
@@ -102,6 +102,9 @@ export const CASCADE_MINIMUM = 5;
|
|
|
102
102
|
|
|
103
103
|
export interface Job extends JobSummary {
|
|
104
104
|
description: string;
|
|
105
|
+
/** Vocabulary terms read out of the description, so search finds a skill the title never mentions. Set by the
|
|
106
|
+
* hosted index, which carries these in place of the descriptions themselves; absent means read the description. */
|
|
107
|
+
skills?: string;
|
|
105
108
|
}
|
|
106
109
|
|
|
107
110
|
/** Partition lookup that ignores inherited properties, so a slug such as "constructor" never resolves to Object.prototype. */
|