openings 0.1.42 → 0.1.44

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openings",
3
- "version": "0.1.42",
3
+ "version": "0.1.44",
4
4
  "description": "Find evidence-grounded jobs, including relevant roles you may not have searched for, without accounts or API keys.",
5
5
  "author": { "name": "Openings contributors" },
6
6
  "license": "MIT",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openings",
3
- "version": "0.1.42",
3
+ "version": "0.1.44",
4
4
  "description": "A free, candidate-safe job-search substrate for AI agents",
5
5
  "license": "MIT",
6
6
  "repository": { "type": "git", "url": "git+https://github.com/abhay-avagama/hiring-agent.git" },
package/src/text-match.ts CHANGED
@@ -6,12 +6,38 @@
6
6
  /** Everything outside this set separates tokens, so "full-stack" and "full stack" agree. */
7
7
  const SEPARATORS = /[^a-z0-9.+#]+/g;
8
8
 
9
- /** A trailing plural "s" is dropped on both sides, so "engineers" and "engineer" meet in the middle. */
9
+ /**
10
+ * Word forms are reduced on both sides, because a candidate typing "engineer" means "Engineering" too.
11
+ * Only plurals and "-ing" come off. A wider rule was measured and withdrawn: stripping "er" and "ment" merged
12
+ * agent nouns with activity nouns, so "developer" reached 378 Business and Corporate Development titles and
13
+ * "engineer" reached Search Engine Optimization. Word-form overlap helps retrieval, but these are different jobs.
14
+ * Technology names carrying + or # are never touched.
15
+ */
16
+ const SUFFIXES = ["ing", "s"];
10
17
  function stem(token: string): string {
11
18
  if (token.length < 4 || /[+#]/.test(token)) return token;
12
- return token.endsWith("s") && !token.endsWith("ss") ? token.slice(0, -1) : token;
19
+ let word = token;
20
+ // Reduce repeatedly, or "engineerings" stops one form short of "engineer".
21
+ for (let pass = 0; pass < 3; pass += 1) {
22
+ const before = word;
23
+ for (const suffix of SUFFIXES) {
24
+ if (word.length - suffix.length >= 4 && word.endsWith(suffix) && !(suffix === "s" && word.endsWith("ss"))) {
25
+ word = word.slice(0, -suffix.length);
26
+ break;
27
+ }
28
+ }
29
+ if (word === before) break;
30
+ }
31
+ return word;
13
32
  }
14
33
 
34
+ /** Words written as one in some titles and two in others; the index carries both readings. */
35
+ /** Written as one word, these are read as two as well, but never joined the other way: joining "Dev" and "Ops" in
36
+ * "Clin Dev Ops" produced a DevOps match for Clinical Development Operations. */
37
+ const SPLIT_ONLY = [["dev", "ops"], ["web", "ops"], ["fin", "tech"]].map(([left, right]) => ({ left: left!, right: right!, joined: `${left}${right}` }));
38
+
39
+ const COMPOUNDS = [["full", "stack"], ["front", "end"], ["back", "end"], ["data", "base"], ["work", "flow"]].map(([left, right]) => ({ left: left!, right: right!, joined: `${left}${right}` }));
40
+
15
41
  export function searchTokens(text: string): string[] {
16
42
  return text.toLocaleLowerCase().replace(SEPARATORS, " ").split(" ")
17
43
  .map((token) => token.replace(/^\.+/, "").replace(/\.+$/, ""))
@@ -22,10 +48,18 @@ export function searchTokens(text: string): string[] {
22
48
  /** The text a job is matched against, padded so a term only matches a whole token. A dotted name is also
23
49
  * indexed by its parts, so "node" finds "Node.js" while "node.js" still does. */
24
50
  export function searchHaystack(text: string): string {
25
- const indexed = new Set<string>();
26
- for (const token of searchTokens(text)) {
27
- indexed.add(token);
28
- if (token.includes(".")) for (const part of token.split(".")) if (part) indexed.add(stem(part));
51
+ const tokens = searchTokens(text);
52
+ const indexed = new Set<string>(tokens);
53
+ for (const token of tokens) if (token.includes(".")) for (const part of token.split(".")) if (part) indexed.add(stem(part));
54
+ // A word written as one is also read as two, which is always safe: the title really does contain both parts.
55
+ for (const { left, right, joined } of SPLIT_ONLY) if (indexed.has(stem(joined))) { indexed.add(stem(left)); indexed.add(stem(right)); }
56
+ // "Fullstack" and "Full Stack" are the same job: index each reading so either spelling finds both.
57
+ for (const { left, right, joined } of COMPOUNDS) {
58
+ const leftStem = stem(left); const rightStem = stem(right); const joinedStem = stem(joined);
59
+ if (indexed.has(joinedStem)) { indexed.add(leftStem); indexed.add(rightStem); }
60
+ for (let index = 0; index + 1 < tokens.length; index += 1) {
61
+ if (tokens[index] === leftStem && tokens[index + 1] === rightStem) indexed.add(joinedStem);
62
+ }
29
63
  }
30
64
  return ` ${[...indexed].join(" ")} `;
31
65
  }
package/src/version.ts CHANGED
@@ -1 +1 @@
1
- export const VERSION = "0.1.42";
1
+ export const VERSION = "0.1.44";