cry-search 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/CLAUDE.md +285 -0
  2. package/LICENSE.md +67 -0
  3. package/README.md +283 -0
  4. package/UNIVERSE.md +818 -0
  5. package/dist/common/SearchUniverse.d.ts +210 -0
  6. package/dist/common/SearchUniverse.d.ts.map +1 -0
  7. package/dist/common/findInArray.d.ts +37 -0
  8. package/dist/common/findInArray.d.ts.map +1 -0
  9. package/dist/common/findInArrayReturnDataAndMeta.d.ts +52 -0
  10. package/dist/common/findInArrayReturnDataAndMeta.d.ts.map +1 -0
  11. package/dist/common/findInLinkedArrays.d.ts +66 -0
  12. package/dist/common/findInLinkedArrays.d.ts.map +1 -0
  13. package/dist/common/numeric/createSearchArrayMetadata.d.ts +65 -0
  14. package/dist/common/numeric/createSearchArrayMetadata.d.ts.map +1 -0
  15. package/dist/common/numeric/createSearchLinkedMetadata.d.ts +46 -0
  16. package/dist/common/numeric/createSearchLinkedMetadata.d.ts.map +1 -0
  17. package/dist/common/numeric/findInLinkedArrays.d.ts +32 -0
  18. package/dist/common/numeric/findInLinkedArrays.d.ts.map +1 -0
  19. package/dist/common/numeric/index.d.ts +9 -0
  20. package/dist/common/numeric/index.d.ts.map +1 -0
  21. package/dist/common/numeric/searchInData.d.ts +31 -0
  22. package/dist/common/numeric/searchInData.d.ts.map +1 -0
  23. package/dist/common/numeric/updateSearchMetadata.d.ts +60 -0
  24. package/dist/common/numeric/updateSearchMetadata.d.ts.map +1 -0
  25. package/dist/common/string/createSearchArrayMetadata.d.ts +117 -0
  26. package/dist/common/string/createSearchArrayMetadata.d.ts.map +1 -0
  27. package/dist/common/string/createSearchLinkedMetadata.d.ts +36 -0
  28. package/dist/common/string/createSearchLinkedMetadata.d.ts.map +1 -0
  29. package/dist/common/string/index.d.ts +10 -0
  30. package/dist/common/string/index.d.ts.map +1 -0
  31. package/dist/common/string/updateSearchMetadata.d.ts +103 -0
  32. package/dist/common/string/updateSearchMetadata.d.ts.map +1 -0
  33. package/dist/common/syncSearchArrayMetadata.d.ts +64 -0
  34. package/dist/common/syncSearchArrayMetadata.d.ts.map +1 -0
  35. package/dist/common/updateSearchLinkedMetadata.d.ts +129 -0
  36. package/dist/common/updateSearchLinkedMetadata.d.ts.map +1 -0
  37. package/dist/index.cjs +1827 -0
  38. package/dist/index.d.cts +32 -0
  39. package/dist/index.d.ts +32 -0
  40. package/dist/index.d.ts.map +1 -0
  41. package/dist/index.js +1795 -0
  42. package/dist/types/index.d.ts +300 -0
  43. package/dist/types/index.d.ts.map +1 -0
  44. package/dist/types.d.ts +6 -0
  45. package/dist/types.d.ts.map +1 -0
  46. package/dist/utils/matchToken.d.ts +65 -0
  47. package/dist/utils/matchToken.d.ts.map +1 -0
  48. package/dist/utils/matchTokenNumeric.d.ts +94 -0
  49. package/dist/utils/matchTokenNumeric.d.ts.map +1 -0
  50. package/dist/utils/normalizeDates.d.ts +39 -0
  51. package/dist/utils/normalizeDates.d.ts.map +1 -0
  52. package/dist/utils/prepareStringForSearch.d.ts +28 -0
  53. package/dist/utils/prepareStringForSearch.d.ts.map +1 -0
  54. package/dist/utils/prepareStringForSearchNumeric.d.ts +42 -0
  55. package/dist/utils/prepareStringForSearchNumeric.d.ts.map +1 -0
  56. package/dist/utils/preprocessString.d.ts +26 -0
  57. package/dist/utils/preprocessString.d.ts.map +1 -0
  58. package/dist/utils/sanitiseString.d.ts +26 -0
  59. package/dist/utils/sanitiseString.d.ts.map +1 -0
  60. package/dist/utils/tokenRegistry.d.ts +70 -0
  61. package/dist/utils/tokenRegistry.d.ts.map +1 -0
  62. package/dist/utils/tokenize.d.ts +46 -0
  63. package/dist/utils/tokenize.d.ts.map +1 -0
  64. package/package.json +67 -0
@@ -0,0 +1,26 @@
1
+ /**
2
+ * Sanitises a string for search by normalizing case, removing diacritics, and filtering characters.
3
+ *
4
+ * Processing steps:
5
+ * 1. Converts to lowercase (case-insensitive search)
6
+ * 2. Removes diacritics using Unicode normalization (č,ć → c, š → s, etc.)
7
+ * 3. Keeps only letters (a-z), digits (0-9), periods (.), minuses (-), and spaces
8
+ * 4. Replaces all other characters with space
9
+ *
10
+ * @param input - The preprocessed string to sanitise
11
+ * @returns Sanitised string ready for tokenization
12
+ *
13
+ * @example
14
+ * ```typescript
15
+ * sanitiseString('Čokolada Milka!')
16
+ * // Returns: 'cokolada milka '
17
+ *
18
+ * sanitiseString('Žaba ČAČA')
19
+ * // Returns: 'zaba caca'
20
+ *
21
+ * sanitiseString('price: $12.50')
22
+ * // Returns: 'price 12.50'
23
+ * ```
24
+ */
25
+ export declare function sanitiseString(input: string): string;
26
+ //# sourceMappingURL=sanitiseString.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"sanitiseString.d.ts","sourceRoot":"","sources":["../../src/utils/sanitiseString.ts"],"names":[],"mappings":"AAIA;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,wBAAgB,cAAc,CAAC,KAAK,EAAE,MAAM,GAAG,MAAM,CAepD"}
@@ -0,0 +1,70 @@
1
+ /**
2
+ * Global token registry for numeric token optimization.
3
+ *
4
+ * This module provides a global array of tokens that allows numeric token IDs
5
+ * to be used instead of string tokens. The array is accessed by index for
6
+ * string lookups during search, avoiding the overhead of Map.get().
7
+ *
8
+ * During metadata creation, a temporary Map<string, number> (tokensMap) is used
9
+ * for fast string->number lookups, then cleared after use.
10
+ */
11
+ /** Global array of all registered tokens. Index = token ID. */
12
+ export declare const tokens: string[];
13
+ /**
14
+ * Gets the current number of registered tokens.
15
+ * @returns The count of registered tokens
16
+ */
17
+ export declare function getTokenCount(): number;
18
+ /**
19
+ * Gets a token string by its numeric ID.
20
+ * Fast O(1) array access.
21
+ *
22
+ * @param id - The numeric token ID
23
+ * @returns The token string, or undefined if ID is out of range
24
+ */
25
+ export declare function getTokenById(id: number): string | undefined;
26
+ /**
27
+ * Finds a token ID by its string value.
28
+ * Slow O(n) linear search - use tokensMap during bulk operations.
29
+ *
30
+ * @param token - The token string to find
31
+ * @returns The numeric token ID, or -1 if not found
32
+ */
33
+ export declare function findTokenId(token: string): number;
34
+ /**
35
+ * Registers a new token and returns its ID.
36
+ * Should only be called when the token doesn't exist (checked via tokensMap).
37
+ *
38
+ * @param token - The token string to register
39
+ * @returns The newly assigned numeric token ID
40
+ */
41
+ export declare function registerToken(token: string): number;
42
+ /**
43
+ * Gets or creates a token ID using a tokensMap for fast lookups.
44
+ * This is the primary method for bulk operations during metadata creation.
45
+ *
46
+ * @param token - The token string
47
+ * @param tokensMap - Map from string to numeric ID (for fast lookup)
48
+ * @returns The numeric token ID (existing or newly created)
49
+ */
50
+ export declare function getOrCreateTokenId(token: string, tokensMap: Map<string, number>): number;
51
+ /**
52
+ * Creates a new temporary tokensMap populated with all currently registered tokens.
53
+ * Use this at the start of bulk metadata creation operations.
54
+ *
55
+ * @returns A new Map<string, number> with all existing token mappings
56
+ */
57
+ export declare function createTokensMap(): Map<string, number>;
58
+ /**
59
+ * Clears a tokensMap after use to free memory.
60
+ * The map is cleared and should be set to null by the caller.
61
+ *
62
+ * @param tokensMap - The map to clear
63
+ */
64
+ export declare function clearTokensMap(tokensMap: Map<string, number>): void;
65
+ /**
66
+ * Resets the token registry to empty state.
67
+ * WARNING: Only use for testing or complete reinitialization.
68
+ */
69
+ export declare function resetTokenRegistry(): void;
70
+ //# sourceMappingURL=tokenRegistry.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"tokenRegistry.d.ts","sourceRoot":"","sources":["../../src/utils/tokenRegistry.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AAKH,+DAA+D;AAC/D,eAAO,MAAM,MAAM,EAAE,MAAM,EAAgC,CAAC;AAK5D;;;GAGG;AACH,wBAAgB,aAAa,IAAI,MAAM,CAEtC;AAED;;;;;;GAMG;AACH,wBAAgB,YAAY,CAAC,EAAE,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAE3D;AAED;;;;;;GAMG;AACH,wBAAgB,WAAW,CAAC,KAAK,EAAE,MAAM,GAAG,MAAM,CAOjD;AAED;;;;;;GAMG;AACH,wBAAgB,aAAa,CAAC,KAAK,EAAE,MAAM,GAAG,MAAM,CAKnD;AAED;;;;;;;GAOG;AACH,wBAAgB,kBAAkB,CAAC,KAAK,EAAE,MAAM,EAAE,SAAS,EAAE,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,GAAG,MAAM,CAQxF;AAED;;;;;GAKG;AACH,wBAAgB,eAAe,IAAI,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,CAMrD;AAED;;;;;GAKG;AACH,wBAAgB,cAAc,CAAC,SAAS,EAAE,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,GAAG,IAAI,CAEnE;AAED;;;GAGG;AACH,wBAAgB,kBAAkB,IAAI,IAAI,CAKzC"}
@@ -0,0 +1,46 @@
1
+ export interface TokenizeConfig {
2
+ /** Map of tokens to replace (key -> value) */
3
+ replaceTokens?: Map<string, string>;
4
+ /** Set of tokens to remove */
5
+ removeTokens?: Set<string>;
6
+ /**
7
+ * Optional map for string->number token ID conversion.
8
+ * If provided, tokenize will use this for fast lookups during bulk operations.
9
+ * If not provided, tokenize returns string tokens as before.
10
+ */
11
+ tokensMap?: Map<string, number>;
12
+ }
13
+ /** Default token replacements for common terms */
14
+ export declare const DEFAULT_REPLACE_TOKENS: Map<string, string>;
15
+ /** Default tokens to remove (common short words with low search value) */
16
+ export declare const DEFAULT_REMOVE_TOKENS: Set<string>;
17
+ /** Default tokenize config using the default replacements and removals */
18
+ export declare const DEFAULT_TOKENIZE_CONFIG: TokenizeConfig;
19
+ /**
20
+ * Tokenizes a string for search by splitting on spaces.
21
+ *
22
+ * Processing steps:
23
+ * 1. Splits on spaces
24
+ * 2. Trims all substrings
25
+ * 3. Filters out empty strings
26
+ * 4. Replaces tokens according to replaceTokens map (if provided)
27
+ * 5. Removes tokens in removeTokens set (if provided)
28
+ *
29
+ * @param input - The sanitized string to tokenize
30
+ * @param config - Optional configuration for token replacement and removal
31
+ * @returns Array of non-empty tokens
32
+ *
33
+ * @example
34
+ * ```typescript
35
+ * tokenize('hello world')
36
+ * // Returns: ['hello', 'world']
37
+ *
38
+ * tokenize(' foo bar ')
39
+ * // Returns: ['foo', 'bar']
40
+ *
41
+ * tokenize('zdravilo za psi', DEFAULT_TOKENIZE_CONFIG)
42
+ * // Returns: ['zdr', 'ps'] (za is removed, zdravilo->zdr, psi->ps)
43
+ * ```
44
+ */
45
+ export declare function tokenize(input: string, config?: TokenizeConfig): string[];
46
+ //# sourceMappingURL=tokenize.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"tokenize.d.ts","sourceRoot":"","sources":["../../src/utils/tokenize.ts"],"names":[],"mappings":"AAAA,MAAM,WAAW,cAAc;IAC7B,8CAA8C;IAC9C,aAAa,CAAC,EAAE,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACpC,8BAA8B;IAC9B,YAAY,CAAC,EAAE,GAAG,CAAC,MAAM,CAAC,CAAC;IAC3B;;;;OAIG;IACH,SAAS,CAAC,EAAE,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;CACjC;AAED,kDAAkD;AAClD,eAAO,MAAM,sBAAsB,qBAcjC,CAAC;AAEH,0EAA0E;AAC1E,eAAO,MAAM,qBAAqB,aAUhC,CAAC;AAEH,0EAA0E;AAC1E,eAAO,MAAM,uBAAuB,EAAE,cAGrC,CAAC;AAEF;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AACH,wBAAgB,QAAQ,CAAC,KAAK,EAAE,MAAM,EAAE,MAAM,CAAC,EAAE,cAAc,GAAG,MAAM,EAAE,CAuBzE"}
package/package.json ADDED
@@ -0,0 +1,67 @@
1
+ {
2
+ "name": "cry-search",
3
+ "version": "1.0.0",
4
+ "description": "A fast, memory-efficient search library for large datasets with support for tokenized matching, linked collections, and per-field match modes.",
5
+ "type": "module",
6
+ "main": "./dist/index.js",
7
+ "module": "./dist/index.js",
8
+ "types": "./dist/index.d.ts",
9
+ "exports": {
10
+ ".": {
11
+ "import": {
12
+ "types": "./dist/index.d.ts",
13
+ "default": "./dist/index.js"
14
+ },
15
+ "require": {
16
+ "types": "./dist/index.d.cts",
17
+ "default": "./dist/index.cjs"
18
+ }
19
+ }
20
+ },
21
+ "files": [
22
+ "dist",
23
+ "README.md",
24
+ "CLAUDE.md",
25
+ "UNIVERSE.md",
26
+ "LICENSE.md"
27
+ ],
28
+ "scripts": {
29
+ "build": "bun run build:js && bun run build:types",
30
+ "build:esm": "bun build ./src/index.ts --outdir ./dist --target browser --format esm",
31
+ "build:cjs": "bun build ./src/index.ts --target browser --format cjs --outfile ./dist/index.cjs",
32
+ "build:js": "bun run build:esm && bun run build:cjs",
33
+ "build:types": "tsc -p tsconfig.build.json && cp ./dist/index.d.ts ./dist/index.d.cts",
34
+ "clean": "rm -rf dist",
35
+ "test": "bun test",
36
+ "prepublishOnly": "bun run clean && bun run build"
37
+ },
38
+ "keywords": [
39
+ "search",
40
+ "fuzzy-search",
41
+ "tokenize",
42
+ "metadata",
43
+ "fast-search",
44
+ "linked-collections",
45
+ "typescript"
46
+ ],
47
+ "author": "Primož Krajnik",
48
+ "license": "SEE LICENSE IN LICENSE.md",
49
+ "repository": {
50
+ "type": "git",
51
+ "url": ""
52
+ },
53
+ "devDependencies": {
54
+ "@types/bun": "latest"
55
+ },
56
+ "peerDependencies": {
57
+ "typescript": "^5"
58
+ },
59
+ "directories": {
60
+ "test": "test"
61
+ },
62
+ "dependencies": {
63
+ "bun-types": "^1.3.6",
64
+ "typescript": "^5.9.3",
65
+ "undici-types": "^7.16.0"
66
+ }
67
+ }