@njinlabs/njin 0.10.0 → 0.10.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/core/model/index.ts +46 -12
- package/src/modules/surreal.ts +26 -7
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@njinlabs/njin",
|
|
3
|
-
"version": "0.10.
|
|
3
|
+
"version": "0.10.2",
|
|
4
4
|
"description": "A modern framework for building company profiles, landing pages, and content-driven websites.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"keywords": ["bun", "elysia", "surrealdb", "edgejs", "cms", "framework"],
|
package/src/core/model/index.ts
CHANGED
|
@@ -4,6 +4,10 @@ import { RecordId, Table, type Values } from "surrealdb";
|
|
|
4
4
|
import { z } from "zod";
|
|
5
5
|
import { runAfterHooks, runBeforeDestroyHooks, runBeforeHooks } from "./hooks";
|
|
6
6
|
|
|
7
|
+
// See the containment-boost comment in read()'s relevanceSelect below for why this exists and
|
|
8
|
+
// why the value just needs to clear typical BM25 scores (~0-10), not be precisely tuned.
|
|
9
|
+
const CONTAINMENT_BOOST = 100;
|
|
10
|
+
|
|
7
11
|
export type FormMeta = {
|
|
8
12
|
label: string;
|
|
9
13
|
unique?: boolean;
|
|
@@ -262,25 +266,55 @@ export const makeModel = <Rules extends z.ZodObject>(
|
|
|
262
266
|
// user-defined schema shape (they're injected in create()/update()) — allow sorting by them too.
|
|
263
267
|
const sortableFields = new Set([...Object.keys(config.schema.shape), "id", "createdAt", "updatedAt"]);
|
|
264
268
|
const hasExplicitSort = Boolean(sort && sortableFields.has(sort));
|
|
265
|
-
// An explicit sort always wins; otherwise, when searching, rank by
|
|
266
|
-
//
|
|
267
|
-
//
|
|
268
|
-
//
|
|
269
|
-
//
|
|
270
|
-
//
|
|
271
|
-
//
|
|
272
|
-
//
|
|
273
|
-
//
|
|
274
|
-
|
|
269
|
+
// An explicit sort always wins; otherwise, when searching, rank by relevance instead of
|
|
270
|
+
// leaving result order unspecified. `ORDER BY` only accepts a bare identifier here, not a
|
|
271
|
+
// function call — so relevance is projected as an aliased field below (SELECT ... AS
|
|
272
|
+
// __relevance) and stripped back out of each returned record afterwards, since it isn't
|
|
273
|
+
// part of the model's schema.
|
|
274
|
+
//
|
|
275
|
+
// Every field — flat or nested — contributes raw BM25 (search::score(N)) plus a containment
|
|
276
|
+
// boost. search::score(N) only works against a FULLTEXT index on the table actually being
|
|
277
|
+
// queried, and a nested match happened on a different table entirely — so its score is
|
|
278
|
+
// pulled back via a correlated subquery using SurrealDB's $parent (the current outer row),
|
|
279
|
+
// re-running the same `targetField @N@ $search` match scoped to just that one linked record
|
|
280
|
+
// (`id = $parent.<local>`, or `id IN $parent.<local>` for a multi relation) and summing the
|
|
281
|
+
// result with math::sum (0 rows -> 0, exactly what an unmatched/absent relation should
|
|
282
|
+
// contribute). Nested entries used to be excluded from __relevance altogether, which meant a
|
|
283
|
+
// row matched *only* through its relation (e.g. brand.name === "Red Wing", normally the most
|
|
284
|
+
// precise signal available) scored exactly 0 — tied with "didn't match" and ranked below any
|
|
285
|
+
// row that merely shared a few ngrams with the query on a flat field.
|
|
286
|
+
//
|
|
287
|
+
// The containment check itself: the ngram analyzer (see SEARCH_ANALYZER_DEFINITION in
|
|
288
|
+
// ../../modules/surreal) matches on shared n-grams, which means a document can match without
|
|
289
|
+
// ever containing the search string as a whole — two unrelated titles can share enough short
|
|
290
|
+
// n-grams to both "match", and short/generic field values are disproportionately likely to
|
|
291
|
+
// do so. A field that truly contains the (lowercased) search string is a much stronger
|
|
292
|
+
// signal than raw BM25 alone, so it's boosted well above the normal BM25 range (empirically
|
|
293
|
+
// small, well under 10) to consistently outrank n-gram-only matches, without removing those
|
|
294
|
+
// matches from the result set (fuzzy/typo recall from ngram is unaffected — this only changes
|
|
295
|
+
// ordering). Both sides also have their spaces stripped before comparing, since real product
|
|
296
|
+
// data routinely writes a multi-word term as one run-together token (e.g. "REDWING 2415..."
|
|
297
|
+
// for "Red Wing") — a plain substring check against "red wing" (with the space) would miss
|
|
298
|
+
// that despite it being a stronger match than most ngram overlaps.
|
|
299
|
+
const useRelevance = !hasExplicitSort && Boolean(search && searchPlan.length);
|
|
275
300
|
const orderBy = hasExplicitSort
|
|
276
301
|
? `ORDER BY ${sort} ${order === "desc" ? "DESC" : "ASC"}`
|
|
277
302
|
: useRelevance
|
|
278
303
|
? "ORDER BY __relevance DESC"
|
|
279
304
|
: "";
|
|
305
|
+
const containmentCheck = (field: string) =>
|
|
306
|
+
`string::contains(string::replace(string::lowercase(${field}), " ", ""), string::replace(string::lowercase($search), " ", ""))`;
|
|
280
307
|
const relevanceSelect = useRelevance
|
|
281
308
|
? `, (${searchPlan
|
|
282
|
-
.map((e, i) =>
|
|
283
|
-
|
|
309
|
+
.map((e, i) => {
|
|
310
|
+
const n = i + 1;
|
|
311
|
+
const boost = (field: string) => `(IF ${containmentCheck(field)} THEN ${CONTAINMENT_BOOST} ELSE 0 END)`;
|
|
312
|
+
if (e.kind === "flat") {
|
|
313
|
+
return `(search::score(${n}) + ${boost(e.field)})`;
|
|
314
|
+
}
|
|
315
|
+
const idFilter = e.multi ? `id IN $parent.${e.local}` : `id = $parent.${e.local}`;
|
|
316
|
+
return `math::sum((SELECT VALUE (search::score(${n}) + ${boost(e.targetField)}) FROM ${e.targetPrefix} WHERE ${idFilter} AND ${e.targetField} @${n}@ $search))`;
|
|
317
|
+
})
|
|
284
318
|
.join(" + ")}) AS __relevance`
|
|
285
319
|
: "";
|
|
286
320
|
|
package/src/modules/surreal.ts
CHANGED
|
@@ -15,10 +15,20 @@ export const isRemotePath = (path: string) => REMOTE_SCHEMES.some((scheme) => pa
|
|
|
15
15
|
// tokenizer splits on whitespace only (unlike `class`, which also splits on punctuation:
|
|
16
16
|
// "Next.js" would become "next" / "." / "js", and a lone "." can't form any 2-char ngram,
|
|
17
17
|
// so a query for "next.js" — tokenized the same way — would never match). The `ngram`
|
|
18
|
-
// filter then indexes overlapping
|
|
18
|
+
// filter then indexes overlapping 3-10 char slices of each whitespace-delimited token so
|
|
19
19
|
// the `@N@` match operator can find a term anywhere inside a field (not just a whole-field
|
|
20
|
-
// match) and still tolerate minor typos, similar to trigram search.
|
|
20
|
+
// match) and still tolerate minor typos, similar to trigram search. Minimum is 3, not 2 —
|
|
21
|
+
// 2-char slices ("si", "ze", digit pairs, ...) are common enough across unrelated documents
|
|
22
|
+
// that they matched all over the place on real-sized catalogs (a 2-char generic word turns
|
|
23
|
+
// into near-universal noise); 3+ cuts that collision rate drastically. core/model/index.ts's
|
|
24
|
+
// relevanceSelect adds a containment-based score boost on top of this, so a document that
|
|
25
|
+
// truly contains the query substring still outranks one that only shares a few ngrams with it.
|
|
21
26
|
const SEARCH_ANALYZER = "njin_search";
|
|
27
|
+
// Single source of truth for the analyzer body — reused both in the DEFINE below and in
|
|
28
|
+
// schemaHash, so any future change here (e.g. a different ngram range) automatically
|
|
29
|
+
// invalidates the stored hash and re-runs the DEFINE on next boot instead of silently
|
|
30
|
+
// leaving an already-migrated DB on the old analyzer definition.
|
|
31
|
+
const SEARCH_ANALYZER_DEFINITION = "TOKENIZERS blank FILTERS lowercase,ngram(3,10)";
|
|
22
32
|
|
|
23
33
|
// Records the hash of the last schema this DB was migrated to, so a worker booting against
|
|
24
34
|
// an already-migrated DB (idle-evict/crash respawn — every DEFINE below is idempotent but
|
|
@@ -70,7 +80,13 @@ const ensureTables = async (db: Surreal) => {
|
|
|
70
80
|
}
|
|
71
81
|
|
|
72
82
|
const schemaHash = createHash("sha256")
|
|
73
|
-
.update(
|
|
83
|
+
.update(
|
|
84
|
+
JSON.stringify({
|
|
85
|
+
prefixes: [...prefixes].sort(),
|
|
86
|
+
searchIndexes: [...searchIndexes.keys()].sort(),
|
|
87
|
+
analyzer: SEARCH_ANALYZER_DEFINITION,
|
|
88
|
+
}),
|
|
89
|
+
)
|
|
74
90
|
.digest("hex");
|
|
75
91
|
|
|
76
92
|
// Must be DEFINE'd before the SELECT below can even run — unlike a table that exists but
|
|
@@ -87,13 +103,16 @@ const ensureTables = async (db: Surreal) => {
|
|
|
87
103
|
await db.query(`DEFINE TABLE IF NOT EXISTS ${prefix} SCHEMALESS;`);
|
|
88
104
|
}
|
|
89
105
|
|
|
90
|
-
|
|
106
|
+
// OVERWRITE, not IF NOT EXISTS — this block only runs when schemaHash just changed (see the
|
|
107
|
+
// early return above), so an analyzer/index that already exists under this name here means its
|
|
108
|
+
// definition is stale and needs replacing, not skipping. FULLTEXT, not SEARCH — this SurrealDB
|
|
109
|
+
// version renamed the index-type keyword; SEARCH ANALYZER ... is a parse error here even though
|
|
110
|
+
// older docs/examples use it.
|
|
111
|
+
await db.query(`DEFINE ANALYZER OVERWRITE ${SEARCH_ANALYZER} ${SEARCH_ANALYZER_DEFINITION};`);
|
|
91
112
|
|
|
92
113
|
for (const { prefix: targetPrefix, field } of searchIndexes.values()) {
|
|
93
|
-
// FULLTEXT, not SEARCH — this SurrealDB version renamed the index-type keyword;
|
|
94
|
-
// SEARCH ANALYZER ... is a parse error here even though older docs/examples use it.
|
|
95
114
|
await db.query(
|
|
96
|
-
`DEFINE INDEX
|
|
115
|
+
`DEFINE INDEX OVERWRITE idx_search_${targetPrefix}_${field} ON TABLE ${targetPrefix} FIELDS ${field} FULLTEXT ANALYZER ${SEARCH_ANALYZER} BM25 HIGHLIGHTS;`,
|
|
97
116
|
);
|
|
98
117
|
}
|
|
99
118
|
|