@decocms/blocks 7.58.0 → 7.59.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/cms/index.ts +1 -0
- package/src/cms/resolve.ts +29 -3
package/package.json
CHANGED
package/src/cms/index.ts
CHANGED
package/src/cms/resolve.ts
CHANGED
|
@@ -310,9 +310,35 @@ export function getDeferredRawProps(
|
|
|
310
310
|
// Bot detection — bots always receive fully eager pages for SEO
|
|
311
311
|
// ---------------------------------------------------------------------------
|
|
312
312
|
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
313
|
+
/**
|
|
314
|
+
* User-Agent substrings that classify a request as a bot.
|
|
315
|
+
*
|
|
316
|
+
* Exported as data, not just baked into the regex, because the CDN cache rules
|
|
317
|
+
* have to bypass exactly the same requests this list makes eager — see
|
|
318
|
+
* `@decocms/blocks-cli/scripts/cdn-rules.ts`. Two hand-maintained copies would
|
|
319
|
+
* drift, and the failure mode is a crawler's eager HTML (~10x larger) being
|
|
320
|
+
* served to humans from the CDN, or the deferred one being served to Google.
|
|
321
|
+
*/
|
|
322
|
+
export const BOT_UA_SUBSTRINGS = [
|
|
323
|
+
"bot",
|
|
324
|
+
"crawl",
|
|
325
|
+
"spider",
|
|
326
|
+
"slurp",
|
|
327
|
+
"facebookexternalhit",
|
|
328
|
+
"mediapartners",
|
|
329
|
+
"google",
|
|
330
|
+
"bing",
|
|
331
|
+
"yandex",
|
|
332
|
+
"baidu",
|
|
333
|
+
"duckduck",
|
|
334
|
+
"teoma",
|
|
335
|
+
"ia_archiver",
|
|
336
|
+
"semrush",
|
|
337
|
+
"ahrefs",
|
|
338
|
+
"lighthouse",
|
|
339
|
+
] as const;
|
|
340
|
+
|
|
341
|
+
const botPatterns: RegExp[] = [new RegExp(BOT_UA_SUBSTRINGS.join("|"), "i")];
|
|
316
342
|
|
|
317
343
|
/**
|
|
318
344
|
* Add a custom bot detection regex.
|