@njinlabs/njin 0.10.1 → 0.10.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/core/model/index.ts +41 -24
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@njinlabs/njin",
|
|
3
|
-
"version": "0.10.
|
|
3
|
+
"version": "0.10.2",
|
|
4
4
|
"description": "A modern framework for building company profiles, landing pages, and content-driven websites.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"keywords": ["bun", "elysia", "surrealdb", "edgejs", "cms", "framework"],
|
package/src/core/model/index.ts
CHANGED
|
@@ -266,38 +266,55 @@ export const makeModel = <Rules extends z.ZodObject>(
|
|
|
266
266
|
// user-defined schema shape (they're injected in create()/update()) — allow sorting by them too.
|
|
267
267
|
const sortableFields = new Set([...Object.keys(config.schema.shape), "id", "createdAt", "updatedAt"]);
|
|
268
268
|
const hasExplicitSort = Boolean(sort && sortableFields.has(sort));
|
|
269
|
-
// An explicit sort always wins; otherwise, when searching, rank by
|
|
270
|
-
//
|
|
271
|
-
//
|
|
272
|
-
//
|
|
273
|
-
//
|
|
274
|
-
// there's no score to rank by, so relevance ordering is skipped (same as no searchFields).
|
|
275
|
-
// `ORDER BY` only accepts a bare identifier here, not a function call — so relevance is
|
|
276
|
-
// projected as an aliased field below (SELECT ... AS __relevance) and stripped back out
|
|
277
|
-
// of each returned record afterwards, since it isn't part of the model's schema.
|
|
269
|
+
// An explicit sort always wins; otherwise, when searching, rank by relevance instead of
|
|
270
|
+
// leaving result order unspecified. `ORDER BY` only accepts a bare identifier here, not a
|
|
271
|
+
// function call — so relevance is projected as an aliased field below (SELECT ... AS
|
|
272
|
+
// __relevance) and stripped back out of each returned record afterwards, since it isn't
|
|
273
|
+
// part of the model's schema.
|
|
278
274
|
//
|
|
279
|
-
//
|
|
280
|
-
//
|
|
281
|
-
//
|
|
282
|
-
//
|
|
283
|
-
//
|
|
284
|
-
//
|
|
285
|
-
//
|
|
286
|
-
//
|
|
287
|
-
|
|
275
|
+
// Every field — flat or nested — contributes raw BM25 (search::score(N)) plus a containment
|
|
276
|
+
// boost. search::score(N) only works against a FULLTEXT index on the table actually being
|
|
277
|
+
// queried, and a nested match happened on a different table entirely — so its score is
|
|
278
|
+
// pulled back via a correlated subquery using SurrealDB's $parent (the current outer row),
|
|
279
|
+
// re-running the same `targetField @N@ $search` match scoped to just that one linked record
|
|
280
|
+
// (`id = $parent.<local>`, or `id IN $parent.<local>` for a multi relation) and summing the
|
|
281
|
+
// result with math::sum (0 rows -> 0, exactly what an unmatched/absent relation should
|
|
282
|
+
// contribute). Nested entries used to be excluded from __relevance altogether, which meant a
|
|
283
|
+
// row matched *only* through its relation (e.g. brand.name === "Red Wing", normally the most
|
|
284
|
+
// precise signal available) scored exactly 0 — tied with "didn't match" and ranked below any
|
|
285
|
+
// row that merely shared a few ngrams with the query on a flat field.
|
|
286
|
+
//
|
|
287
|
+
// The containment check itself: the ngram analyzer (see SEARCH_ANALYZER_DEFINITION in
|
|
288
|
+
// ../../modules/surreal) matches on shared n-grams, which means a document can match without
|
|
289
|
+
// ever containing the search string as a whole — two unrelated titles can share enough short
|
|
290
|
+
// n-grams to both "match", and short/generic field values are disproportionately likely to
|
|
291
|
+
// do so. A field that truly contains the (lowercased) search string is a much stronger
|
|
292
|
+
// signal than raw BM25 alone, so it's boosted well above the normal BM25 range (empirically
|
|
293
|
+
// small, well under 10) to consistently outrank n-gram-only matches, without removing those
|
|
294
|
+
// matches from the result set (fuzzy/typo recall from ngram is unaffected — this only changes
|
|
295
|
+
// ordering). Both sides also have their spaces stripped before comparing, since real product
|
|
296
|
+
// data routinely writes a multi-word term as one run-together token (e.g. "REDWING 2415..."
|
|
297
|
+
// for "Red Wing") — a plain substring check against "red wing" (with the space) would miss
|
|
298
|
+
// that despite it being a stronger match than most ngram overlaps.
|
|
299
|
+
const useRelevance = !hasExplicitSort && Boolean(search && searchPlan.length);
|
|
288
300
|
const orderBy = hasExplicitSort
|
|
289
301
|
? `ORDER BY ${sort} ${order === "desc" ? "DESC" : "ASC"}`
|
|
290
302
|
: useRelevance
|
|
291
303
|
? "ORDER BY __relevance DESC"
|
|
292
304
|
: "";
|
|
305
|
+
const containmentCheck = (field: string) =>
|
|
306
|
+
`string::contains(string::replace(string::lowercase(${field}), " ", ""), string::replace(string::lowercase($search), " ", ""))`;
|
|
293
307
|
const relevanceSelect = useRelevance
|
|
294
308
|
? `, (${searchPlan
|
|
295
|
-
.map((e, i) =>
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
309
|
+
.map((e, i) => {
|
|
310
|
+
const n = i + 1;
|
|
311
|
+
const boost = (field: string) => `(IF ${containmentCheck(field)} THEN ${CONTAINMENT_BOOST} ELSE 0 END)`;
|
|
312
|
+
if (e.kind === "flat") {
|
|
313
|
+
return `(search::score(${n}) + ${boost(e.field)})`;
|
|
314
|
+
}
|
|
315
|
+
const idFilter = e.multi ? `id IN $parent.${e.local}` : `id = $parent.${e.local}`;
|
|
316
|
+
return `math::sum((SELECT VALUE (search::score(${n}) + ${boost(e.targetField)}) FROM ${e.targetPrefix} WHERE ${idFilter} AND ${e.targetField} @${n}@ $search))`;
|
|
317
|
+
})
|
|
301
318
|
.join(" + ")}) AS __relevance`
|
|
302
319
|
: "";
|
|
303
320
|
|