@njinlabs/njin 0.10.1 → 0.10.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@njinlabs/njin",
3
- "version": "0.10.1",
3
+ "version": "0.10.3",
4
4
  "description": "A modern framework for building company profiles, landing pages, and content-driven websites.",
5
5
  "type": "module",
6
6
  "keywords": ["bun", "elysia", "surrealdb", "edgejs", "cms", "framework"],
@@ -266,38 +266,55 @@ export const makeModel = <Rules extends z.ZodObject>(
266
266
  // user-defined schema shape (they're injected in create()/update()) — allow sorting by them too.
267
267
  const sortableFields = new Set([...Object.keys(config.schema.shape), "id", "createdAt", "updatedAt"]);
268
268
  const hasExplicitSort = Boolean(sort && sortableFields.has(sort));
269
- // An explicit sort always wins; otherwise, when searching, rank by BM25 relevance
270
- // (summed across every matched flat search field) instead of leaving result order
271
- // unspecified. Nested (relation) entries are excluded from this sum — a match found via
272
- // the IN/CONTAINSANY subquery above has no per-record score in this query's context, since
273
- // it happened on a different table entirely. If a model's searchFields are all nested,
274
- // there's no score to rank by, so relevance ordering is skipped (same as no searchFields).
275
- // `ORDER BY` only accepts a bare identifier here, not a function call — so relevance is
276
- // projected as an aliased field below (SELECT ... AS __relevance) and stripped back out
277
- // of each returned record afterwards, since it isn't part of the model's schema.
269
+ // An explicit sort always wins; otherwise, when searching, rank by relevance instead of
270
+ // leaving result order unspecified. `ORDER BY` only accepts a bare identifier here, not a
271
+ // function call — so relevance is projected as an aliased field below (SELECT ... AS
272
+ // __relevance) and stripped back out of each returned record afterwards, since it isn't
273
+ // part of the model's schema.
278
274
  //
279
- // Each flat field's score also gets a containment boost: the ngram analyzer (see
280
- // SEARCH_ANALYZER_DEFINITION in ../../modules/surreal) matches on shared n-grams, which
281
- // means a document can match without ever containing the search string as a whole — two
282
- // unrelated titles can share enough short n-grams to both "match". A document whose field
283
- // literally contains the (lowercased) search string is a much stronger relevance signal than
284
- // raw BM25 alone, so it's boosted well above the normal BM25 range (empirically small, well
285
- // under 10) to consistently outrank n-gram-only matches, while leaving those matches in the
286
- // result set (fuzzy/typo recall from ngram is unaffected — this only changes ordering).
287
- const useRelevance = !hasExplicitSort && Boolean(search && searchPlan.some((e) => e.kind === "flat"));
275
+ // Every field — flat or nested — contributes raw BM25 (search::score(N)) plus a containment
276
+ // boost. search::score(N) only works against a FULLTEXT index on the table actually being
277
+ // queried, and a nested match happened on a different table entirely — so its score is
278
+ // pulled back via a correlated subquery using SurrealDB's $parent (the current outer row),
279
+ // re-running the same `targetField @N@ $search` match scoped to just that one linked record
280
+ // (`id = $parent.<local>`, or `id IN $parent.<local>` for a multi relation) and summing the
281
+ // result with math::sum (0 rows -> 0, exactly what an unmatched/absent relation should
282
+ // contribute). Nested entries used to be excluded from __relevance altogether, which meant a
283
+ // row matched *only* through its relation (e.g. brand.name === "Red Wing", normally the most
284
+ // precise signal available) scored exactly 0 — tied with "didn't match" and ranked below any
285
+ // row that merely shared a few ngrams with the query on a flat field.
286
+ //
287
+ // The containment check itself: the ngram analyzer (see SEARCH_ANALYZER_DEFINITION in
288
+ // ../../modules/surreal) matches on shared n-grams, which means a document can match without
289
+ // ever containing the search string as a whole — two unrelated titles can share enough short
290
+ // n-grams to both "match", and short/generic field values are disproportionately likely to
291
+ // do so. A field that truly contains the (lowercased) search string is a much stronger
292
+ // signal than raw BM25 alone, so it's boosted well above the normal BM25 range (empirically
293
+ // small, well under 10) to consistently outrank n-gram-only matches, without removing those
294
+ // matches from the result set (fuzzy/typo recall from ngram is unaffected — this only changes
295
+ // ordering). Both sides also have their spaces stripped before comparing, since real product
296
+ // data routinely writes a multi-word term as one run-together token (e.g. "REDWING 2415..."
297
+ // for "Red Wing") — a plain substring check against "red wing" (with the space) would miss
298
+ // that despite it being a stronger match than most ngram overlaps.
299
+ const useRelevance = !hasExplicitSort && Boolean(search && searchPlan.length);
288
300
  const orderBy = hasExplicitSort
289
301
  ? `ORDER BY ${sort} ${order === "desc" ? "DESC" : "ASC"}`
290
302
  : useRelevance
291
303
  ? "ORDER BY __relevance DESC"
292
304
  : "";
305
+ const containmentCheck = (field: string) =>
306
+ `string::contains(string::replace(string::lowercase(${field}), " ", ""), string::replace(string::lowercase($search), " ", ""))`;
293
307
  const relevanceSelect = useRelevance
294
308
  ? `, (${searchPlan
295
- .map((e, i) =>
296
- e.kind === "flat"
297
- ? `(search::score(${i + 1}) + (IF string::contains(string::lowercase(${e.field}), string::lowercase($search)) THEN ${CONTAINMENT_BOOST} ELSE 0 END))`
298
- : null,
299
- )
300
- .filter((s): s is string => s !== null)
309
+ .map((e, i) => {
310
+ const n = i + 1;
311
+ const boost = (field: string) => `(IF ${containmentCheck(field)} THEN ${CONTAINMENT_BOOST} ELSE 0 END)`;
312
+ if (e.kind === "flat") {
313
+ return `(search::score(${n}) + ${boost(e.field)})`;
314
+ }
315
+ const idFilter = e.multi ? `id IN $parent.${e.local}` : `id = $parent.${e.local}`;
316
+ return `math::sum((SELECT VALUE (search::score(${n}) + ${boost(e.targetField)}) FROM ${e.targetPrefix} WHERE ${idFilter} AND ${e.targetField} @${n}@ $search))`;
317
+ })
301
318
  .join(" + ")}) AS __relevance`
302
319
  : "";
303
320
 
@@ -46,6 +46,21 @@ export const resolveClientIp = (
46
46
  return server?.requestIP(request)?.address ?? null;
47
47
  };
48
48
 
49
+ // `request.url` reflects what the backend itself sees — behind a reverse proxy that
50
+ // terminates TLS (the common case), that's `http://` even though visitors hit the site
51
+ // over `https://`. That scheme mismatch alone breaks isSameOrigin's exact comparison,
52
+ // so every internal navigation gets misclassified as an external referrer. Prefer the
53
+ // headers a trusted proxy sets; fall back to request.url for direct connections.
54
+ export const resolveRequestOrigin = (request: Request) => {
55
+ const forwardedHost = request.headers.get("x-forwarded-host");
56
+ if (forwardedHost) {
57
+ const proto = request.headers.get("x-forwarded-proto")?.split(",")[0]!.trim() || "http";
58
+ return `${proto}://${forwardedHost.split(",")[0]!.trim()}`;
59
+ }
60
+
61
+ return new URL(request.url).origin;
62
+ };
63
+
49
64
  const hashVisitor = (ip: string, userAgent: string) => {
50
65
  const today = moment().format("YYYY-MM-DD");
51
66
  return new Bun.CryptoHasher("sha256").update(`${ip}:${userAgent}:${today}`).digest("hex");
@@ -155,13 +155,13 @@ const view = makeModule(() => {
155
155
  });
156
156
 
157
157
  // Fire-and-forget — never awaited, so it can't add latency to the response.
158
- import("./analytics").then(({ default: analytics, resolveClientIp }) =>
158
+ import("./analytics").then(({ default: analytics, resolveClientIp, resolveRequestOrigin }) =>
159
159
  analytics().track({
160
160
  path,
161
161
  referrer: request.headers.get("referer"),
162
162
  userAgent: request.headers.get("user-agent"),
163
163
  ip: resolveClientIp(request, server),
164
- requestUrl: request.url,
164
+ requestUrl: resolveRequestOrigin(request),
165
165
  }),
166
166
  );
167
167