cry-search 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +285 -0
- package/LICENSE.md +67 -0
- package/README.md +283 -0
- package/UNIVERSE.md +818 -0
- package/dist/common/SearchUniverse.d.ts +210 -0
- package/dist/common/SearchUniverse.d.ts.map +1 -0
- package/dist/common/findInArray.d.ts +37 -0
- package/dist/common/findInArray.d.ts.map +1 -0
- package/dist/common/findInArrayReturnDataAndMeta.d.ts +52 -0
- package/dist/common/findInArrayReturnDataAndMeta.d.ts.map +1 -0
- package/dist/common/findInLinkedArrays.d.ts +66 -0
- package/dist/common/findInLinkedArrays.d.ts.map +1 -0
- package/dist/common/numeric/createSearchArrayMetadata.d.ts +65 -0
- package/dist/common/numeric/createSearchArrayMetadata.d.ts.map +1 -0
- package/dist/common/numeric/createSearchLinkedMetadata.d.ts +46 -0
- package/dist/common/numeric/createSearchLinkedMetadata.d.ts.map +1 -0
- package/dist/common/numeric/findInLinkedArrays.d.ts +32 -0
- package/dist/common/numeric/findInLinkedArrays.d.ts.map +1 -0
- package/dist/common/numeric/index.d.ts +9 -0
- package/dist/common/numeric/index.d.ts.map +1 -0
- package/dist/common/numeric/searchInData.d.ts +31 -0
- package/dist/common/numeric/searchInData.d.ts.map +1 -0
- package/dist/common/numeric/updateSearchMetadata.d.ts +60 -0
- package/dist/common/numeric/updateSearchMetadata.d.ts.map +1 -0
- package/dist/common/string/createSearchArrayMetadata.d.ts +117 -0
- package/dist/common/string/createSearchArrayMetadata.d.ts.map +1 -0
- package/dist/common/string/createSearchLinkedMetadata.d.ts +36 -0
- package/dist/common/string/createSearchLinkedMetadata.d.ts.map +1 -0
- package/dist/common/string/index.d.ts +10 -0
- package/dist/common/string/index.d.ts.map +1 -0
- package/dist/common/string/updateSearchMetadata.d.ts +103 -0
- package/dist/common/string/updateSearchMetadata.d.ts.map +1 -0
- package/dist/common/syncSearchArrayMetadata.d.ts +64 -0
- package/dist/common/syncSearchArrayMetadata.d.ts.map +1 -0
- package/dist/common/updateSearchLinkedMetadata.d.ts +129 -0
- package/dist/common/updateSearchLinkedMetadata.d.ts.map +1 -0
- package/dist/index.cjs +1827 -0
- package/dist/index.d.cts +32 -0
- package/dist/index.d.ts +32 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +1795 -0
- package/dist/types/index.d.ts +300 -0
- package/dist/types/index.d.ts.map +1 -0
- package/dist/types.d.ts +6 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/utils/matchToken.d.ts +65 -0
- package/dist/utils/matchToken.d.ts.map +1 -0
- package/dist/utils/matchTokenNumeric.d.ts +94 -0
- package/dist/utils/matchTokenNumeric.d.ts.map +1 -0
- package/dist/utils/normalizeDates.d.ts +39 -0
- package/dist/utils/normalizeDates.d.ts.map +1 -0
- package/dist/utils/prepareStringForSearch.d.ts +28 -0
- package/dist/utils/prepareStringForSearch.d.ts.map +1 -0
- package/dist/utils/prepareStringForSearchNumeric.d.ts +42 -0
- package/dist/utils/prepareStringForSearchNumeric.d.ts.map +1 -0
- package/dist/utils/preprocessString.d.ts +26 -0
- package/dist/utils/preprocessString.d.ts.map +1 -0
- package/dist/utils/sanitiseString.d.ts +26 -0
- package/dist/utils/sanitiseString.d.ts.map +1 -0
- package/dist/utils/tokenRegistry.d.ts +70 -0
- package/dist/utils/tokenRegistry.d.ts.map +1 -0
- package/dist/utils/tokenize.d.ts +46 -0
- package/dist/utils/tokenize.d.ts.map +1 -0
- package/package.json +67 -0
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Sanitises a string for search by normalizing case, removing diacritics, and filtering characters.
|
|
3
|
+
*
|
|
4
|
+
* Processing steps:
|
|
5
|
+
* 1. Converts to lowercase (case-insensitive search)
|
|
6
|
+
* 2. Removes diacritics using Unicode normalization (č,ć → c, š → s, etc.)
|
|
7
|
+
* 3. Keeps only letters (a-z), digits (0-9), periods (.), minuses (-), and spaces
|
|
8
|
+
* 4. Replaces all other characters with space
|
|
9
|
+
*
|
|
10
|
+
* @param input - The preprocessed string to sanitise
|
|
11
|
+
* @returns Sanitised string ready for tokenization
|
|
12
|
+
*
|
|
13
|
+
* @example
|
|
14
|
+
* ```typescript
|
|
15
|
+
* sanitiseString('Čokolada Milka!')
|
|
16
|
+
* // Returns: 'cokolada milka '
|
|
17
|
+
*
|
|
18
|
+
* sanitiseString('Žaba ČAČA')
|
|
19
|
+
* // Returns: 'zaba caca'
|
|
20
|
+
*
|
|
21
|
+
* sanitiseString('price: $12.50')
|
|
22
|
+
* // Returns: 'price 12.50'
|
|
23
|
+
* ```
|
|
24
|
+
*/
|
|
25
|
+
export declare function sanitiseString(input: string): string;
|
|
26
|
+
//# sourceMappingURL=sanitiseString.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"sanitiseString.d.ts","sourceRoot":"","sources":["../../src/utils/sanitiseString.ts"],"names":[],"mappings":"AAIA;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,wBAAgB,cAAc,CAAC,KAAK,EAAE,MAAM,GAAG,MAAM,CAepD"}
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Global token registry for numeric token optimization.
|
|
3
|
+
*
|
|
4
|
+
* This module provides a global array of tokens that allows numeric token IDs
|
|
5
|
+
* to be used instead of string tokens. The array is accessed by index for
|
|
6
|
+
* string lookups during search, avoiding the overhead of Map.get().
|
|
7
|
+
*
|
|
8
|
+
* During metadata creation, a temporary Map<string, number> (tokensMap) is used
|
|
9
|
+
* for fast string->number lookups, then cleared after use.
|
|
10
|
+
*/
|
|
11
|
+
/** Global array of all registered tokens. Index = token ID. */
|
|
12
|
+
export declare const tokens: string[];
|
|
13
|
+
/**
|
|
14
|
+
* Gets the current number of registered tokens.
|
|
15
|
+
* @returns The count of registered tokens
|
|
16
|
+
*/
|
|
17
|
+
export declare function getTokenCount(): number;
|
|
18
|
+
/**
|
|
19
|
+
* Gets a token string by its numeric ID.
|
|
20
|
+
* Fast O(1) array access.
|
|
21
|
+
*
|
|
22
|
+
* @param id - The numeric token ID
|
|
23
|
+
* @returns The token string, or undefined if ID is out of range
|
|
24
|
+
*/
|
|
25
|
+
export declare function getTokenById(id: number): string | undefined;
|
|
26
|
+
/**
|
|
27
|
+
* Finds a token ID by its string value.
|
|
28
|
+
* Slow O(n) linear search - use tokensMap during bulk operations.
|
|
29
|
+
*
|
|
30
|
+
* @param token - The token string to find
|
|
31
|
+
* @returns The numeric token ID, or -1 if not found
|
|
32
|
+
*/
|
|
33
|
+
export declare function findTokenId(token: string): number;
|
|
34
|
+
/**
|
|
35
|
+
* Registers a new token and returns its ID.
|
|
36
|
+
* Should only be called when the token doesn't exist (checked via tokensMap).
|
|
37
|
+
*
|
|
38
|
+
* @param token - The token string to register
|
|
39
|
+
* @returns The newly assigned numeric token ID
|
|
40
|
+
*/
|
|
41
|
+
export declare function registerToken(token: string): number;
|
|
42
|
+
/**
|
|
43
|
+
* Gets or creates a token ID using a tokensMap for fast lookups.
|
|
44
|
+
* This is the primary method for bulk operations during metadata creation.
|
|
45
|
+
*
|
|
46
|
+
* @param token - The token string
|
|
47
|
+
* @param tokensMap - Map from string to numeric ID (for fast lookup)
|
|
48
|
+
* @returns The numeric token ID (existing or newly created)
|
|
49
|
+
*/
|
|
50
|
+
export declare function getOrCreateTokenId(token: string, tokensMap: Map<string, number>): number;
|
|
51
|
+
/**
|
|
52
|
+
* Creates a new temporary tokensMap populated with all currently registered tokens.
|
|
53
|
+
* Use this at the start of bulk metadata creation operations.
|
|
54
|
+
*
|
|
55
|
+
* @returns A new Map<string, number> with all existing token mappings
|
|
56
|
+
*/
|
|
57
|
+
export declare function createTokensMap(): Map<string, number>;
|
|
58
|
+
/**
|
|
59
|
+
* Clears a tokensMap after use to free memory.
|
|
60
|
+
* The map is cleared and should be set to null by the caller.
|
|
61
|
+
*
|
|
62
|
+
* @param tokensMap - The map to clear
|
|
63
|
+
*/
|
|
64
|
+
export declare function clearTokensMap(tokensMap: Map<string, number>): void;
|
|
65
|
+
/**
|
|
66
|
+
* Resets the token registry to empty state.
|
|
67
|
+
* WARNING: Only use for testing or complete reinitialization.
|
|
68
|
+
*/
|
|
69
|
+
export declare function resetTokenRegistry(): void;
|
|
70
|
+
//# sourceMappingURL=tokenRegistry.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"tokenRegistry.d.ts","sourceRoot":"","sources":["../../src/utils/tokenRegistry.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AAKH,+DAA+D;AAC/D,eAAO,MAAM,MAAM,EAAE,MAAM,EAAgC,CAAC;AAK5D;;;GAGG;AACH,wBAAgB,aAAa,IAAI,MAAM,CAEtC;AAED;;;;;;GAMG;AACH,wBAAgB,YAAY,CAAC,EAAE,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAE3D;AAED;;;;;;GAMG;AACH,wBAAgB,WAAW,CAAC,KAAK,EAAE,MAAM,GAAG,MAAM,CAOjD;AAED;;;;;;GAMG;AACH,wBAAgB,aAAa,CAAC,KAAK,EAAE,MAAM,GAAG,MAAM,CAKnD;AAED;;;;;;;GAOG;AACH,wBAAgB,kBAAkB,CAAC,KAAK,EAAE,MAAM,EAAE,SAAS,EAAE,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,GAAG,MAAM,CAQxF;AAED;;;;;GAKG;AACH,wBAAgB,eAAe,IAAI,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,CAMrD;AAED;;;;;GAKG;AACH,wBAAgB,cAAc,CAAC,SAAS,EAAE,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,GAAG,IAAI,CAEnE;AAED;;;GAGG;AACH,wBAAgB,kBAAkB,IAAI,IAAI,CAKzC"}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
export interface TokenizeConfig {
|
|
2
|
+
/** Map of tokens to replace (key -> value) */
|
|
3
|
+
replaceTokens?: Map<string, string>;
|
|
4
|
+
/** Set of tokens to remove */
|
|
5
|
+
removeTokens?: Set<string>;
|
|
6
|
+
/**
|
|
7
|
+
* Optional map for string->number token ID conversion.
|
|
8
|
+
* If provided, tokenize will use this for fast lookups during bulk operations.
|
|
9
|
+
* If not provided, tokenize returns string tokens as before.
|
|
10
|
+
*/
|
|
11
|
+
tokensMap?: Map<string, number>;
|
|
12
|
+
}
|
|
13
|
+
/** Default token replacements for common terms */
|
|
14
|
+
export declare const DEFAULT_REPLACE_TOKENS: Map<string, string>;
|
|
15
|
+
/** Default tokens to remove (common short words with low search value) */
|
|
16
|
+
export declare const DEFAULT_REMOVE_TOKENS: Set<string>;
|
|
17
|
+
/** Default tokenize config using the default replacements and removals */
|
|
18
|
+
export declare const DEFAULT_TOKENIZE_CONFIG: TokenizeConfig;
|
|
19
|
+
/**
|
|
20
|
+
* Tokenizes a string for search by splitting on spaces.
|
|
21
|
+
*
|
|
22
|
+
* Processing steps:
|
|
23
|
+
* 1. Splits on spaces
|
|
24
|
+
* 2. Trims all substrings
|
|
25
|
+
* 3. Filters out empty strings
|
|
26
|
+
* 4. Replaces tokens according to replaceTokens map (if provided)
|
|
27
|
+
* 5. Removes tokens in removeTokens set (if provided)
|
|
28
|
+
*
|
|
29
|
+
* @param input - The sanitized string to tokenize
|
|
30
|
+
* @param config - Optional configuration for token replacement and removal
|
|
31
|
+
* @returns Array of non-empty tokens
|
|
32
|
+
*
|
|
33
|
+
* @example
|
|
34
|
+
* ```typescript
|
|
35
|
+
* tokenize('hello world')
|
|
36
|
+
* // Returns: ['hello', 'world']
|
|
37
|
+
*
|
|
38
|
+
* tokenize(' foo bar ')
|
|
39
|
+
* // Returns: ['foo', 'bar']
|
|
40
|
+
*
|
|
41
|
+
* tokenize('zdravilo za psi', DEFAULT_TOKENIZE_CONFIG)
|
|
42
|
+
* // Returns: ['zdr', 'ps'] (za is removed, zdravilo->zdr, psi->ps)
|
|
43
|
+
* ```
|
|
44
|
+
*/
|
|
45
|
+
export declare function tokenize(input: string, config?: TokenizeConfig): string[];
|
|
46
|
+
//# sourceMappingURL=tokenize.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"tokenize.d.ts","sourceRoot":"","sources":["../../src/utils/tokenize.ts"],"names":[],"mappings":"AAAA,MAAM,WAAW,cAAc;IAC7B,8CAA8C;IAC9C,aAAa,CAAC,EAAE,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACpC,8BAA8B;IAC9B,YAAY,CAAC,EAAE,GAAG,CAAC,MAAM,CAAC,CAAC;IAC3B;;;;OAIG;IACH,SAAS,CAAC,EAAE,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;CACjC;AAED,kDAAkD;AAClD,eAAO,MAAM,sBAAsB,qBAcjC,CAAC;AAEH,0EAA0E;AAC1E,eAAO,MAAM,qBAAqB,aAUhC,CAAC;AAEH,0EAA0E;AAC1E,eAAO,MAAM,uBAAuB,EAAE,cAGrC,CAAC;AAEF;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AACH,wBAAgB,QAAQ,CAAC,KAAK,EAAE,MAAM,EAAE,MAAM,CAAC,EAAE,cAAc,GAAG,MAAM,EAAE,CAuBzE"}
|
package/package.json
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "cry-search",
|
|
3
|
+
"version": "1.0.0",
|
|
4
|
+
"description": "A fast, memory-efficient search library for large datasets with support for tokenized matching, linked collections, and per-field match modes.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"main": "./dist/index.js",
|
|
7
|
+
"module": "./dist/index.js",
|
|
8
|
+
"types": "./dist/index.d.ts",
|
|
9
|
+
"exports": {
|
|
10
|
+
".": {
|
|
11
|
+
"import": {
|
|
12
|
+
"types": "./dist/index.d.ts",
|
|
13
|
+
"default": "./dist/index.js"
|
|
14
|
+
},
|
|
15
|
+
"require": {
|
|
16
|
+
"types": "./dist/index.d.cts",
|
|
17
|
+
"default": "./dist/index.cjs"
|
|
18
|
+
}
|
|
19
|
+
}
|
|
20
|
+
},
|
|
21
|
+
"files": [
|
|
22
|
+
"dist",
|
|
23
|
+
"README.md",
|
|
24
|
+
"CLAUDE.md",
|
|
25
|
+
"UNIVERSE.md",
|
|
26
|
+
"LICENSE.md"
|
|
27
|
+
],
|
|
28
|
+
"scripts": {
|
|
29
|
+
"build": "bun run build:js && bun run build:types",
|
|
30
|
+
"build:esm": "bun build ./src/index.ts --outdir ./dist --target browser --format esm",
|
|
31
|
+
"build:cjs": "bun build ./src/index.ts --target browser --format cjs --outfile ./dist/index.cjs",
|
|
32
|
+
"build:js": "bun run build:esm && bun run build:cjs",
|
|
33
|
+
"build:types": "tsc -p tsconfig.build.json && cp ./dist/index.d.ts ./dist/index.d.cts",
|
|
34
|
+
"clean": "rm -rf dist",
|
|
35
|
+
"test": "bun test",
|
|
36
|
+
"prepublishOnly": "bun run clean && bun run build"
|
|
37
|
+
},
|
|
38
|
+
"keywords": [
|
|
39
|
+
"search",
|
|
40
|
+
"fuzzy-search",
|
|
41
|
+
"tokenize",
|
|
42
|
+
"metadata",
|
|
43
|
+
"fast-search",
|
|
44
|
+
"linked-collections",
|
|
45
|
+
"typescript"
|
|
46
|
+
],
|
|
47
|
+
"author": "Primož Krajnik",
|
|
48
|
+
"license": "SEE LICENSE IN LICENSE.md",
|
|
49
|
+
"repository": {
|
|
50
|
+
"type": "git",
|
|
51
|
+
"url": ""
|
|
52
|
+
},
|
|
53
|
+
"devDependencies": {
|
|
54
|
+
"@types/bun": "latest"
|
|
55
|
+
},
|
|
56
|
+
"peerDependencies": {
|
|
57
|
+
"typescript": "^5"
|
|
58
|
+
},
|
|
59
|
+
"directories": {
|
|
60
|
+
"test": "test"
|
|
61
|
+
},
|
|
62
|
+
"dependencies": {
|
|
63
|
+
"bun-types": "^1.3.6",
|
|
64
|
+
"typescript": "^5.9.3",
|
|
65
|
+
"undici-types": "^7.16.0"
|
|
66
|
+
}
|
|
67
|
+
}
|