@njinlabs/njin 0.9.0 → 0.10.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/core/model/index.ts +18 -1
- package/src/modules/surreal.ts +69 -26
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@njinlabs/njin",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.10.1",
|
|
4
4
|
"description": "A modern framework for building company profiles, landing pages, and content-driven websites.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"keywords": ["bun", "elysia", "surrealdb", "edgejs", "cms", "framework"],
|
package/src/core/model/index.ts
CHANGED
|
@@ -4,6 +4,10 @@ import { RecordId, Table, type Values } from "surrealdb";
|
|
|
4
4
|
import { z } from "zod";
|
|
5
5
|
import { runAfterHooks, runBeforeDestroyHooks, runBeforeHooks } from "./hooks";
|
|
6
6
|
|
|
7
|
+
// See the containment-boost comment in read()'s relevanceSelect below for why this exists and
|
|
8
|
+
// why the value just needs to clear typical BM25 scores (~0-10), not be precisely tuned.
|
|
9
|
+
const CONTAINMENT_BOOST = 100;
|
|
10
|
+
|
|
7
11
|
export type FormMeta = {
|
|
8
12
|
label: string;
|
|
9
13
|
unique?: boolean;
|
|
@@ -271,6 +275,15 @@ export const makeModel = <Rules extends z.ZodObject>(
|
|
|
271
275
|
// `ORDER BY` only accepts a bare identifier here, not a function call — so relevance is
|
|
272
276
|
// projected as an aliased field below (SELECT ... AS __relevance) and stripped back out
|
|
273
277
|
// of each returned record afterwards, since it isn't part of the model's schema.
|
|
278
|
+
//
|
|
279
|
+
// Each flat field's score also gets a containment boost: the ngram analyzer (see
|
|
280
|
+
// SEARCH_ANALYZER_DEFINITION in ../../modules/surreal) matches on shared n-grams, which
|
|
281
|
+
// means a document can match without ever containing the search string as a whole — two
|
|
282
|
+
// unrelated titles can share enough short n-grams to both "match". A document whose field
|
|
283
|
+
// literally contains the (lowercased) search string is a much stronger relevance signal than
|
|
284
|
+
// raw BM25 alone, so it's boosted well above the normal BM25 range (empirically small, well
|
|
285
|
+
// under 10) to consistently outrank n-gram-only matches, while leaving those matches in the
|
|
286
|
+
// result set (fuzzy/typo recall from ngram is unaffected — this only changes ordering).
|
|
274
287
|
const useRelevance = !hasExplicitSort && Boolean(search && searchPlan.some((e) => e.kind === "flat"));
|
|
275
288
|
const orderBy = hasExplicitSort
|
|
276
289
|
? `ORDER BY ${sort} ${order === "desc" ? "DESC" : "ASC"}`
|
|
@@ -279,7 +292,11 @@ export const makeModel = <Rules extends z.ZodObject>(
|
|
|
279
292
|
: "";
|
|
280
293
|
const relevanceSelect = useRelevance
|
|
281
294
|
? `, (${searchPlan
|
|
282
|
-
.map((e, i) =>
|
|
295
|
+
.map((e, i) =>
|
|
296
|
+
e.kind === "flat"
|
|
297
|
+
? `(search::score(${i + 1}) + (IF string::contains(string::lowercase(${e.field}), string::lowercase($search)) THEN ${CONTAINMENT_BOOST} ELSE 0 END))`
|
|
298
|
+
: null,
|
|
299
|
+
)
|
|
283
300
|
.filter((s): s is string => s !== null)
|
|
284
301
|
.join(" + ")}) AS __relevance`
|
|
285
302
|
: "";
|
package/src/modules/surreal.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
1
2
|
import { getConfig } from "../core/config";
|
|
2
3
|
import { resolveSearchPlan } from "../core/model";
|
|
3
4
|
import { makeModule } from "../core/module";
|
|
@@ -14,10 +15,27 @@ export const isRemotePath = (path: string) => REMOTE_SCHEMES.some((scheme) => pa
|
|
|
14
15
|
// tokenizer splits on whitespace only (unlike `class`, which also splits on punctuation:
|
|
15
16
|
// "Next.js" would become "next" / "." / "js", and a lone "." can't form any 2-char ngram,
|
|
16
17
|
// so a query for "next.js" — tokenized the same way — would never match). The `ngram`
|
|
17
|
-
// filter then indexes overlapping
|
|
18
|
+
// filter then indexes overlapping 3-10 char slices of each whitespace-delimited token so
|
|
18
19
|
// the `@N@` match operator can find a term anywhere inside a field (not just a whole-field
|
|
19
|
-
// match) and still tolerate minor typos, similar to trigram search.
|
|
20
|
+
// match) and still tolerate minor typos, similar to trigram search. Minimum is 3, not 2 —
|
|
21
|
+
// 2-char slices ("si", "ze", digit pairs, ...) are common enough across unrelated documents
|
|
22
|
+
// that they matched all over the place on real-sized catalogs (a 2-char generic word turns
|
|
23
|
+
// into near-universal noise); 3+ cuts that collision rate drastically. core/model/index.ts's
|
|
24
|
+
// relevanceSelect adds a containment-based score boost on top of this, so a document that
|
|
25
|
+
// truly contains the query substring still outranks one that only shares a few ngrams with it.
|
|
20
26
|
const SEARCH_ANALYZER = "njin_search";
|
|
27
|
+
// Single source of truth for the analyzer body — reused both in the DEFINE below and in
|
|
28
|
+
// schemaHash, so any future change here (e.g. a different ngram range) automatically
|
|
29
|
+
// invalidates the stored hash and re-runs the DEFINE on next boot instead of silently
|
|
30
|
+
// leaving an already-migrated DB on the old analyzer definition.
|
|
31
|
+
const SEARCH_ANALYZER_DEFINITION = "TOKENIZERS blank FILTERS lowercase,ngram(3,10)";
|
|
32
|
+
|
|
33
|
+
// Records the hash of the last schema this DB was migrated to, so a worker booting against
|
|
34
|
+
// an already-migrated DB (idle-evict/crash respawn — every DEFINE below is idempotent but
|
|
35
|
+
// still a full network round trip per statement) can skip straight past ensureTables(). Keyed
|
|
36
|
+
// off the DB itself rather than the process/build: a worker pointed at a fresh or different DB
|
|
37
|
+
// (e.g. a per-client env change) finds no matching record here and still runs the DEFINEs.
|
|
38
|
+
const SCHEMA_META_TABLE = "njin_schema_meta";
|
|
21
39
|
|
|
22
40
|
// SurrealDB's `GROUP ALL` aggregate (used for count queries) throws NotFoundError
|
|
23
41
|
// on a table that has never had a row created, unlike plain SELECT. Defining every
|
|
@@ -36,25 +54,11 @@ const ensureTables = async (db: Surreal) => {
|
|
|
36
54
|
const prefixes = new Set<string>(models.map((model) => model.prefix));
|
|
37
55
|
prefixes.add("vars");
|
|
38
56
|
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
const defineSearchIndex = async (targetPrefix: string, field: string) => {
|
|
46
|
-
const key = `${targetPrefix}.${field}`;
|
|
47
|
-
if (definedIndexes.has(key)) return;
|
|
48
|
-
definedIndexes.add(key);
|
|
49
|
-
|
|
50
|
-
// FULLTEXT, not SEARCH — this SurrealDB version renamed the index-type keyword;
|
|
51
|
-
// SEARCH ANALYZER ... is a parse error here even though older docs/examples use it.
|
|
52
|
-
await db.query(
|
|
53
|
-
`DEFINE INDEX IF NOT EXISTS idx_search_${targetPrefix}_${field} ON TABLE ${targetPrefix} FIELDS ${field} FULLTEXT ANALYZER ${SEARCH_ANALYZER} BM25 HIGHLIGHTS;`,
|
|
54
|
-
);
|
|
55
|
-
};
|
|
56
|
-
|
|
57
|
-
const definedIndexes = new Set<string>(); // dedupe prefix+field in case two factories share a prefix, or a nested reference targets an already-indexed field
|
|
57
|
+
// Resolve every search index's target table+field up front (also dedupes prefix+field in
|
|
58
|
+
// case two factories share a prefix, or a nested reference targets an already-indexed
|
|
59
|
+
// field) — needed both to compute the schema hash below and to drive the DEFINE INDEX loop
|
|
60
|
+
// further down.
|
|
61
|
+
const searchIndexes = new Map<string, { prefix: string; field: string }>();
|
|
58
62
|
for (const model of models) {
|
|
59
63
|
// A nested searchFields entry (e.g. "author.name") needs its index defined on the
|
|
60
64
|
// *target* table/field instead — the local relation field holds a record link, not
|
|
@@ -67,13 +71,52 @@ const ensureTables = async (db: Surreal) => {
|
|
|
67
71
|
: (model.searchFields ?? []).map((field) => ({ kind: "flat" as const, field }));
|
|
68
72
|
|
|
69
73
|
for (const entry of plan) {
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
}
|
|
74
|
+
const target =
|
|
75
|
+
entry.kind === "flat"
|
|
76
|
+
? { prefix: model.prefix, field: entry.field }
|
|
77
|
+
: { prefix: entry.targetPrefix, field: entry.targetField };
|
|
78
|
+
searchIndexes.set(`${target.prefix}.${target.field}`, target);
|
|
75
79
|
}
|
|
76
80
|
}
|
|
81
|
+
|
|
82
|
+
const schemaHash = createHash("sha256")
|
|
83
|
+
.update(
|
|
84
|
+
JSON.stringify({
|
|
85
|
+
prefixes: [...prefixes].sort(),
|
|
86
|
+
searchIndexes: [...searchIndexes.keys()].sort(),
|
|
87
|
+
analyzer: SEARCH_ANALYZER_DEFINITION,
|
|
88
|
+
}),
|
|
89
|
+
)
|
|
90
|
+
.digest("hex");
|
|
91
|
+
|
|
92
|
+
// Must be DEFINE'd before the SELECT below can even run — unlike a table that exists but
|
|
93
|
+
// has no rows (see the GROUP ALL note above), a table SurrealDB has never seen DEFINE'd at
|
|
94
|
+
// all makes ANY query against a specific record id in it throw NotFoundError, plain SELECT
|
|
95
|
+
// included. One extra round trip on every boot (skip path too), still far cheaper than the
|
|
96
|
+
// N DEFINEs it's gating.
|
|
97
|
+
await db.query(`DEFINE TABLE IF NOT EXISTS ${SCHEMA_META_TABLE} SCHEMALESS;`);
|
|
98
|
+
|
|
99
|
+
const [rows] = await db.query<[{ hash: string }[]]>(`SELECT hash FROM ${SCHEMA_META_TABLE}:current;`);
|
|
100
|
+
if (rows?.[0]?.hash === schemaHash) return;
|
|
101
|
+
|
|
102
|
+
for (const prefix of prefixes) {
|
|
103
|
+
await db.query(`DEFINE TABLE IF NOT EXISTS ${prefix} SCHEMALESS;`);
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
// OVERWRITE, not IF NOT EXISTS — this block only runs when schemaHash just changed (see the
|
|
107
|
+
// early return above), so an analyzer/index that already exists under this name here means its
|
|
108
|
+
// definition is stale and needs replacing, not skipping. FULLTEXT, not SEARCH — this SurrealDB
|
|
109
|
+
// version renamed the index-type keyword; SEARCH ANALYZER ... is a parse error here even though
|
|
110
|
+
// older docs/examples use it.
|
|
111
|
+
await db.query(`DEFINE ANALYZER OVERWRITE ${SEARCH_ANALYZER} ${SEARCH_ANALYZER_DEFINITION};`);
|
|
112
|
+
|
|
113
|
+
for (const { prefix: targetPrefix, field } of searchIndexes.values()) {
|
|
114
|
+
await db.query(
|
|
115
|
+
`DEFINE INDEX OVERWRITE idx_search_${targetPrefix}_${field} ON TABLE ${targetPrefix} FIELDS ${field} FULLTEXT ANALYZER ${SEARCH_ANALYZER} BM25 HIGHLIGHTS;`,
|
|
116
|
+
);
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
await db.query(`UPSERT ${SCHEMA_META_TABLE}:current SET hash = '${schemaHash}';`);
|
|
77
120
|
};
|
|
78
121
|
|
|
79
122
|
const surreal = makeModule(() => {
|