openings 0.1.43 → 0.1.44

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openings",
3
- "version": "0.1.43",
3
+ "version": "0.1.44",
4
4
  "description": "Find evidence-grounded jobs, including relevant roles you may not have searched for, without accounts or API keys.",
5
5
  "author": { "name": "Openings contributors" },
6
6
  "license": "MIT",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openings",
3
- "version": "0.1.43",
3
+ "version": "0.1.44",
4
4
  "description": "A free, candidate-safe job-search substrate for AI agents",
5
5
  "license": "MIT",
6
6
  "repository": { "type": "git", "url": "git+https://github.com/abhay-avagama/hiring-agent.git" },
package/src/text-match.ts CHANGED
@@ -8,16 +8,16 @@ const SEPARATORS = /[^a-z0-9.+#]+/g;
8
8
 
9
9
  /**
10
10
  * Word forms are reduced on both sides, because a candidate typing "engineer" means "Engineering" too.
11
- * A first version dropped only plurals, and a coverage panel measured the cost: of 598 results it removed,
12
- * roughly 42 were genuine false matches and the rest were real roles whose titles used another form.
13
- * Suffixes come off longest first, then a trailing "e", so engineer/engineering and develop/developer/development
14
- * all land on one stem. Technology names carrying + or # are never touched.
11
+ * Only plurals and "-ing" come off. A wider rule was measured and withdrawn: stripping "er" and "ment" merged
12
+ * agent nouns with activity nouns, so "developer" reached 378 Business and Corporate Development titles and
13
+ * "engineer" reached Search Engine Optimization. Word-form overlap helps retrieval, but these are different jobs.
14
+ * Technology names carrying + or # are never touched.
15
15
  */
16
- const SUFFIXES = ["ment", "ing", "or", "er", "s"];
16
+ const SUFFIXES = ["ing", "s"];
17
17
  function stem(token: string): string {
18
18
  if (token.length < 4 || /[+#]/.test(token)) return token;
19
19
  let word = token;
20
- // Reduce repeatedly, or "engineering" stops at "engineer" while "engineer" goes on to "engine" and the two never meet.
20
+ // Reduce repeatedly, or "engineerings" stops one form short of "engineer".
21
21
  for (let pass = 0; pass < 3; pass += 1) {
22
22
  const before = word;
23
23
  for (const suffix of SUFFIXES) {
@@ -28,11 +28,15 @@ function stem(token: string): string {
28
28
  }
29
29
  if (word === before) break;
30
30
  }
31
- return word.length > 4 && word.endsWith("e") ? word.slice(0, -1) : word;
31
+ return word;
32
32
  }
33
33
 
34
34
  /** Words written as one in some titles and two in others; the index carries both readings. */
35
- const COMPOUNDS = [["full", "stack"], ["front", "end"], ["back", "end"], ["dev", "ops"], ["data", "base"], ["work", "flow"]].map(([left, right]) => ({ left: left!, right: right!, joined: `${left}${right}` }));
35
+ /** Written as one word, these are read as two as well, but never joined the other way: joining "Dev" and "Ops" in
36
+ * "Clin Dev Ops" produced a DevOps match for Clinical Development Operations. */
37
+ const SPLIT_ONLY = [["dev", "ops"], ["web", "ops"], ["fin", "tech"]].map(([left, right]) => ({ left: left!, right: right!, joined: `${left}${right}` }));
38
+
39
+ const COMPOUNDS = [["full", "stack"], ["front", "end"], ["back", "end"], ["data", "base"], ["work", "flow"]].map(([left, right]) => ({ left: left!, right: right!, joined: `${left}${right}` }));
36
40
 
37
41
  export function searchTokens(text: string): string[] {
38
42
  return text.toLocaleLowerCase().replace(SEPARATORS, " ").split(" ")
@@ -47,6 +51,8 @@ export function searchHaystack(text: string): string {
47
51
  const tokens = searchTokens(text);
48
52
  const indexed = new Set<string>(tokens);
49
53
  for (const token of tokens) if (token.includes(".")) for (const part of token.split(".")) if (part) indexed.add(stem(part));
54
+ // A word written as one is also read as two, which is always safe: the title really does contain both parts.
55
+ for (const { left, right, joined } of SPLIT_ONLY) if (indexed.has(stem(joined))) { indexed.add(stem(left)); indexed.add(stem(right)); }
50
56
  // "Fullstack" and "Full Stack" are the same job: index each reading so either spelling finds both.
51
57
  for (const { left, right, joined } of COMPOUNDS) {
52
58
  const leftStem = stem(left); const rightStem = stem(right); const joinedStem = stem(joined);
package/src/version.ts CHANGED
@@ -1 +1 @@
1
- export const VERSION = "0.1.43";
1
+ export const VERSION = "0.1.44";