@sonordev/site-kit 7.0.1 → 7.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. package/CHANGELOG.md +3539 -0
  2. package/README.md +12 -13
  3. package/agent-manifest.json +1 -1
  4. package/dist/{AnalyticsProvider-EMM2TKRE.js → AnalyticsProvider-XXWTFKJH.js} +4 -4
  5. package/dist/{ArticleViewTracker-RA64BGL6.js → ArticleViewTracker-V4KZB6QN.js} +3 -3
  6. package/dist/{BlocksPopup-D25RFNOV.js → BlocksPopup-JHGHB6XW.js} +4 -4
  7. package/dist/{ChatWidget-RYI7BMJJ.js → ChatWidget-CG32POI3.js} +5 -5
  8. package/dist/{EngageWidget-UKFCN33M.js → EngageWidget-LQMR4LEX.js} +4 -4
  9. package/dist/{FileField-MUHA7LZR.js → FileField-KUG3CKXG.js} +3 -3
  10. package/dist/{FormSpotlight-TCLPWPLL.js → FormSpotlight-FVNPOCU3.js} +1 -1
  11. package/dist/{FormStage-CNYLP6I6.js → FormStage-C7VKRURJ.js} +1 -1
  12. package/dist/{ManagedForm-7ZL5SKTO.js → ManagedForm-VLNJKV65.js} +6 -6
  13. package/dist/{ManagedNewsletterForm-33B4JLX7.js → ManagedNewsletterForm-KJEU23BV.js} +4 -4
  14. package/dist/{SignalCore-L5FVDHFE.js → SignalCore-K2O46QG7.js} +3 -3
  15. package/dist/{SiteDesignReporter-4JOFL4FP.js → SiteDesignReporter-D7MD66GI.js} +5 -5
  16. package/dist/SitemapSync-NMXGMPCQ.js +8 -0
  17. package/dist/_client/booking-widget.js +5 -5
  18. package/dist/affiliates/index.js +3 -3
  19. package/dist/analytics/index.js +4 -4
  20. package/dist/articles/index.js +1 -1
  21. package/dist/articles/server-ui.js +1 -1
  22. package/dist/chat/index.js +5 -5
  23. package/dist/{chunk-QANVUXKH.js → chunk-42OXY4JV.js} +1 -1
  24. package/dist/{chunk-GYESATRY.js → chunk-56JNI463.js} +1 -1
  25. package/dist/{chunk-MV2MBTC3.js → chunk-5FBY2ZIH.js} +1 -1
  26. package/dist/{chunk-BMO3VGMR.js → chunk-7JIKGKWD.js} +7 -7
  27. package/dist/{chunk-QGHSMJKW.js → chunk-B6RZ2NRH.js} +1 -1
  28. package/dist/{chunk-OFOAHPUV.js → chunk-BEL7YFMC.js} +1 -1
  29. package/dist/{chunk-WATH55UY.js → chunk-BS7FWUOY.js} +1 -1
  30. package/dist/{chunk-FL4EPUWA.js → chunk-DKTSGYLM.js} +2 -2
  31. package/dist/{chunk-HGCK465A.js → chunk-GGD4P7UW.js} +1 -1
  32. package/dist/{chunk-FYBZ5SNP.js → chunk-GYY6ETGB.js} +1 -1
  33. package/dist/{chunk-CVTVNC2U.js → chunk-K5WZX776.js} +2 -2
  34. package/dist/{chunk-4IQ52CXL.js → chunk-LJZ3SUET.js} +2 -2
  35. package/dist/{chunk-P5J7VMQ3.js → chunk-O52CH273.js} +1 -1
  36. package/dist/{chunk-3KUUH2YP.js → chunk-OIETJKIL.js} +1 -1
  37. package/dist/{chunk-V6LSQRTH.js → chunk-P2GIIQH5.js} +1 -1
  38. package/dist/{chunk-4RMVXRBO.js → chunk-P72ZJRSX.js} +3 -3
  39. package/dist/{chunk-EGOD74PP.js → chunk-RU2RMTGT.js} +2 -2
  40. package/dist/{chunk-QZZIKMAT.js → chunk-SAUTJMK6.js} +1 -1
  41. package/dist/{chunk-T3MC4HOD.js → chunk-SWP36NCB.js} +1 -1
  42. package/dist/{chunk-5SEM2V4A.js → chunk-T4SY3FMN.js} +3 -3
  43. package/dist/{chunk-P4GRY6QP.js → chunk-ZRE4ZYEG.js} +1 -1
  44. package/dist/{chunk-UZN4ZYR2.js → chunk-ZSLRAMCK.js} +1 -1
  45. package/dist/client/index.js +3 -3
  46. package/dist/commerce/index.js +4 -4
  47. package/dist/engage/index.js +6 -6
  48. package/dist/fleet/index.js +4 -4
  49. package/dist/forms/index.js +7 -7
  50. package/dist/forms/server.js +2 -2
  51. package/dist/forms/types.d.ts +3 -1
  52. package/dist/images/index.js +4 -4
  53. package/dist/index.js +1 -1
  54. package/dist/layout/client.js +7 -7
  55. package/dist/layout/index.js +8 -8
  56. package/dist/maps/index.js +3 -3
  57. package/dist/mcp/sonor.js +6 -6
  58. package/dist/seo/client.js +4 -4
  59. package/dist/seo/index.js +4 -4
  60. package/dist/server/index.js +2 -2
  61. package/dist/shared/version.d.ts +1 -1
  62. package/dist/signal/index.js +2 -2
  63. package/dist/sync/index.js +5 -5
  64. package/dist/website/images.js +4 -4
  65. package/dist/website/index.js +5 -5
  66. package/dist/website/popups.js +4 -4
  67. package/docs/MIGRATING-TO-7.md +146 -0
  68. package/docs.json +67 -0
  69. package/package.json +9 -4
  70. package/src/admin-auth/README.md +88 -0
  71. package/src/analytics/README.md +264 -0
  72. package/src/articles/README.md +325 -0
  73. package/src/commerce/README.md +109 -0
  74. package/src/cta-bar/README.md +154 -0
  75. package/src/engage/README.md +241 -0
  76. package/src/forms/README.md +219 -0
  77. package/src/images/README.md +74 -0
  78. package/src/layout/README.md +66 -0
  79. package/src/llms/README.md +723 -0
  80. package/src/mcp/README.md +376 -0
  81. package/src/motion/README.md +372 -0
  82. package/src/og/README.md +304 -0
  83. package/src/proxy/README.md +152 -0
  84. package/src/redirects/README.md +74 -0
  85. package/src/reputation/README.md +64 -0
  86. package/src/seo/README.md +359 -0
  87. package/src/signal/README.md +115 -0
  88. package/src/sitemap/README.md +127 -0
  89. package/src/sync/README.md +115 -0
  90. package/dist/SitemapSync-7WKY4HXI.js +0 -8
@@ -0,0 +1,723 @@
1
+ # GEO Module — `@sonordev/site-kit/llms`
2
+
3
+ Generative Engine Optimization (GEO) and Answer Engine Optimization (AEO) for Sonor-powered Next.js sites. Makes businesses visible to ChatGPT, Claude, Perplexity, Google AI Overviews, and voice assistants.
4
+
5
+ ---
6
+
7
+ ## Architecture
8
+
9
+ ```
10
+ Sonor API (api.sonor.io) Signal API (signal.sonor.io)
11
+ ├─ GET /api/public/llms/data ◄──────────── AI generates managed_llm_schema
12
+ ├─ GET /api/public/llms/txt + llms_public_summary per page
13
+ ├─ GET /seo/llms/preview (auth'd) via seo-meta-optimization pipeline
14
+ └─ GET /seo/llms/analytics (auth'd)
15
+ │
16
+ ▼
17
+ site-kit (npm: @sonordev/site-kit/llms)
18
+ ├─ generateLLMsTxt() ← builds markdown from Sonor data
19
+ ├─ createLLMsTxtHandler() ← zero-config Next.js route handler
20
+ ├─ buildAiDiscoveryHeaders() ← Link header for crawler discovery
21
+ ├─ createProxy({ llmsDiscovery }) ← sends the Link header from proxy.ts
22
+ ├─ buildAiCrawlerRules() / createRobotsTxtHandler() ← robots.txt for AI crawlers
23
+ ├─ writeLLMsTxtToPublic() ← build-time static file generation
24
+ ├─ createLlmsRevalidateHandler() ← on-demand ISR cache bust
25
+ ├─ LLMSchema (RSC) ← JSON-LD with managed_llm_schema + isPartOf
26
+ ├─ SpeakableSchema ← JSON-LD for voice assistants
27
+ └─ AEO* components ← semantic HTML for AI extraction
28
+ │
29
+ ▼
30
+ Next.js Site
31
+ ├─ /llms.txt ← route handler (dynamic or static)
32
+ ├─ /llms-full.txt ← extended version (more pages/FAQ)
33
+ ├─ Link header ← rel="describedby" on all HTML responses
34
+ ├─ robots.txt ← allows AI crawlers to access /llms.txt
35
+ └─ JSON-LD in <head> ← managed_llm_schema per page
36
+ ```
37
+
38
+ ### Contract
39
+
40
+ The GEO system uses a shared contract (`@sonordev/site-kit/llms/contract`) consumed by site-kit, Sonor API, and Signal API. The contract defines:
41
+
42
+ - **`LLM_GEO_CONTRACT_VERSION`** (currently `1`) — increment on breaking payload changes
43
+ - **`LLMS_PUBLIC_SUMMARY_MAX_LENGTH`** (`400`) — max chars for page link notes
44
+ - **Sanitizers** — `sanitizeLlmsPublicSummary()`, `sanitizeLlmsDisclaimerLine()`, `sanitizePrimaryLanguageTag()`, `pickManagedLlmSchemaForJsonLd()`
45
+
46
+ See `LLM_GEO_CONTRACT.md` for the full spec.
47
+
48
+ ---
49
+
50
+ ## Implementation Guide
51
+
52
+ ### Step 1: Route Handlers (required)
53
+
54
+ ```ts
55
+ // app/llms.txt/route.ts
56
+ import { createLLMsTxtHandler } from '@sonordev/site-kit/llms'
57
+ export const GET = createLLMsTxtHandler()
58
+ // Next 15+: GET handlers are dynamic by default — opt into prerendering
59
+ export const revalidate = 3600
60
+ ```
61
+
62
+ ```ts
63
+ // app/llms-full.txt/route.ts
64
+ import { createLLMsFullTxtHandler } from '@sonordev/site-kit/llms'
65
+ export const GET = createLLMsFullTxtHandler()
66
+ export const revalidate = 3600
67
+ ```
68
+
69
+ Both handlers:
70
+ - Serve static `public/llms.txt` if it exists (build-time optimized, `preferStatic: true` default)
71
+ - Fall back to dynamic generation from Sonor API
72
+ - Set `Cache-Control` with `s-maxage=3600` and `stale-while-revalidate=86400`
73
+ - Return a weak `ETag` (conditional `If-None-Match`/304 handling happens at the Next static layer / CDN)
74
+ - Include `X-Generated-At` and `X-Sections` response headers
75
+ - Never read the incoming `Request`, so the route prerenders statically (○) instead of
76
+ bailing out with a "Dynamic server usage" error during `next build`
77
+
78
+ ### Step 2: Discovery header (required)
79
+
80
+ **Proxy (recommended)**
81
+
82
+ ```ts
83
+ // proxy.ts
84
+ import { createProxy } from '@sonordev/site-kit/proxy'
85
+
86
+ export default createProxy({
87
+ llmsDiscovery: { siteUrl: 'https://example.com' },
88
+ })
89
+
90
+ // Inlined: an imported `config.matcher` is a build error. See the proxy README.
91
+ export const config = {
92
+ matcher: [
93
+ '/((?!_next/static|_next/image|favicon\\.ico|.*\\.(?:ico|png|jpg|jpeg|gif|webp|svg|woff2?)$).*)',
94
+ ],
95
+ }
96
+ ```
97
+
98
+ This adds `Link: <https://example.com/llms.txt>; rel="describedby"; type="text/markdown"`
99
+ to every request that could be for a page: GET or HEAD, with an Accept header
100
+ that allows HTML or no Accept header at all. Requests with no Accept header
101
+ count because that's how curl and many crawlers and AI fetchers ask, and
102
+ they're who the header is for (they used to be skipped). Skipped:
103
+ file-like paths (`/llms.txt`, `/sitemap.xml`), `/api/` routes, Accept headers
104
+ that name only non-HTML types, and Next's RSC navigation requests. The rule is
105
+ `wantsLlmsDiscoveryLink`, exported from `@sonordev/site-kit/llms`.
106
+
107
+ **No proxy? next.config**
108
+
109
+ ```ts
110
+ // next.config.ts
111
+ import { withSiteKitConfig } from '@sonordev/site-kit/config'
112
+
113
+ export default withSiteKitConfig({ llmsTxtDiscoveryLink: true })
114
+ ```
115
+
116
+ This sends the same header from next.config `headers()`, on every route. It
117
+ reads the origin from `NEXT_PUBLIC_SITE_URL`. If you also set
118
+ `nextConfig.headers`, yours replaces the helper's, so merge
119
+ `buildAiDiscoveryHeaders({ siteUrl })` into it yourself.
120
+
121
+ **Not from the root layout.** Older versions of this README showed
122
+ `export async function headers()` in `app/layout.tsx`. That isn't a Next API:
123
+ `headers()` is a next.config option, and exported from a layout it's an
124
+ ordinary function nobody calls, so it sends nothing. `sonor-setup geo` fails a
125
+ site wired that way and says so. Move it to the proxy.
126
+
127
+ `buildAiDiscoveryHeaders` builds the header value for either place, including
128
+ multilingual alternates:
129
+
130
+ ```ts
131
+ buildAiDiscoveryHeaders({
132
+ siteUrl: 'https://example.com',
133
+ languageAlternates: [
134
+ { hreflang: 'fr', href: 'https://example.com/fr/llms.txt' },
135
+ ],
136
+ })
137
+ ```
138
+
139
+ ### Step 3: robots.txt (required)
140
+
141
+ Name the AI crawlers explicitly with the kit's curated lists rather than a
142
+ hand-rolled one, so a crawler the kit adds reaches every site on its next
143
+ upgrade:
144
+
145
+ ```ts
146
+ // app/robots.ts
147
+ import type { MetadataRoute } from 'next'
148
+ import { buildAiCrawlerRules } from '@sonordev/site-kit/llms'
149
+
150
+ export default function robots(): MetadataRoute.Robots {
151
+ return {
152
+ rules: [
153
+ { userAgent: '*', allow: '/', disallow: ['/api/'] },
154
+ // Retrieval crawlers (they cite you) and training crawlers, each in its
155
+ // own group. training: 'block' keeps retrieval and opts out of training.
156
+ ...buildAiCrawlerRules({ disallow: ['/api/'] }),
157
+ ],
158
+ sitemap: 'https://example.com/sitemap.xml',
159
+ }
160
+ }
161
+ ```
162
+
163
+ A crawler that matches a named group ignores the `*` group, so pass the same
164
+ `disallow` to both. The lists are `AI_RETRIEVAL_CRAWLERS` and
165
+ `AI_TRAINING_CRAWLERS`. Don't keep a local list beside them; if an agent is
166
+ missing, add it to `src/llms/aiRobots.ts`.
167
+
168
+ Don't add a `Content-Signal` line (contentsignals.org). Google's robots.txt
169
+ parser reports it as an "Unknown directive" error in Search Console, and
170
+ robots.txt is the one file every crawler has to parse cleanly. Express the
171
+ same preference with `buildAiCrawlerRules({ training: 'allow' | 'block' })`.
172
+ `createRobotsTxtHandler` (a plain-text route handler) ignores
173
+ `contentSignals` since 6.3.4; remove it from a site's config when you next
174
+ touch the file.
175
+
176
+ These helpers are exported from `@sonordev/site-kit/llms`, their home.
177
+ `@sonordev/site-kit/robots` re-exports them too, beside `createRobots`: same
178
+ functions, either import works.
179
+
180
+ ### Step 4: Keep llms.txt out of the sitemap and the index
181
+
182
+ An XML sitemap lists indexable HTML pages. llms.txt is a plain-text restatement
183
+ of those pages, so listing it invites search engines to index a thin duplicate
184
+ of the site. AI crawlers don't need it listed: they fetch `/llms.txt` by
185
+ convention, and the `Link: rel="describedby"` discovery header points at it from
186
+ every page. Leave `includeLlmsTxtInSitemap` / `includeLlmsFullTxtInSitemap` off
187
+ (the default); `sonor-setup geo` warns when a sitemap lists them.
188
+
189
+ `createLLMsTxtHandler` and `createLLMsFullTxtHandler` send
190
+ `X-Robots-Tag: noindex` by default, which keeps the file out of search indexes
191
+ without stopping AI crawlers from fetching it. Pass `noindex: false` only if you
192
+ want it in search results.
193
+
194
+ A static `public/llms.txt` (the build-time write below) is served straight from
195
+ the CDN, so neither the handler nor next.config `headers()` ever sees it. Set the
196
+ header in host config instead:
197
+
198
+ ```toml
199
+ # netlify.toml
200
+ [[headers]]
201
+ for = "/llms.txt"
202
+ [headers.values]
203
+ X-Robots-Tag = "noindex"
204
+
205
+ [[headers]]
206
+ for = "/llms-full.txt"
207
+ [headers.values]
208
+ X-Robots-Tag = "noindex"
209
+ ```
210
+
211
+ ```ts
212
+ createSitemap({
213
+ baseUrl: 'https://example.com',
214
+ optimizedLLMsTxt: true, // Write build-time static file (default: true)
215
+ })
216
+ ```
217
+
218
+ The build-time write never persists a failure stub: if the Sonor fetch fails and no local data is available, `writeLLMsTxtToPublic()` skips the write (warning logged) so an existing `public/llms.txt` / `public/llms-full.txt` keeps serving via `preferStatic`.
219
+
220
+ Sonor generates this file with an LLM call that can take about a minute, and a
221
+ prerendered route only gets 60s, so the in-route write often times out and keeps
222
+ the previous file. For a refresh on every build, let the postbuild own it:
223
+
224
+ ```jsonc
225
+ "scripts": { "postbuild": "sonor-register-sitemap --write-llms" }
226
+ ```
227
+
228
+ with `optimizedLLMsTxt: false` in `createSitemap`, so one writer owns the file.
229
+ Reads behind this write always bypass Next's Data Cache — hosts persist it
230
+ between builds, and a cached read is how a site shipped a months-old llms.txt.
231
+
232
+ ### Step 5: On-Demand Revalidation (optional)
233
+
234
+ When Sonor data changes (page summaries, FAQs, etc.), bust the ISR cache instantly instead of waiting for `s-maxage` to expire:
235
+
236
+ ```ts
237
+ // app/api/revalidate-llms/route.ts
238
+ import { createLlmsRevalidateHandler } from '@sonordev/site-kit/llms'
239
+ export const POST = createLlmsRevalidateHandler(process.env.REVALIDATION_SECRET!)
240
+ ```
241
+
242
+ Sonor can POST to this endpoint with `Authorization: Bearer <secret>` to revalidate `/llms.txt` and `/llms-full.txt`.
243
+
244
+ #### Sonor's SEO webhook (`/api/seo-revalidate`)
245
+
246
+ When a title, description or schema changes in Sonor, Sonor POSTs the
247
+ affected paths and cache tags to the site with the project key.
248
+ `createSeoRevalidationHandler` wraps the handler above and regenerates those
249
+ pages, `/sitemap.xml` and both llms files without a rebuild:
250
+
251
+ ```ts
252
+ // app/api/seo-revalidate/route.ts
253
+ import { revalidatePath, revalidateTag } from 'next/cache'
254
+ import { createSeoRevalidationHandler } from '@sonordev/site-kit/llms'
255
+
256
+ export const runtime = 'nodejs'
257
+
258
+ export async function POST(request: Request) {
259
+ return createSeoRevalidationHandler({
260
+ secret: process.env.SONOR_API_KEY || '',
261
+ revalidatePath,
262
+ revalidateTag,
263
+ publicationBasePath: '/insights', // only when the site has a publication
264
+ })(request)
265
+ }
266
+ ```
267
+
268
+ - Auth is `Authorization: Bearer <SONOR_API_KEY>` only, compared in constant
269
+ time. No `?secret=` query form.
270
+ - Body: `{ paths?, path?, tags?, tag?, revalidateAll? }`, at most 16 KB and
271
+ 100 paths/tags. Every path must be a same-site local path (no scheme, `//`,
272
+ query, fragment, `[segment]` or dot segment, even percent-encoded). One bad
273
+ entry refuses the whole call (400) before any cache is touched.
274
+ - `revalidateAll`, or a tag-only call carrying `seo`, also revalidates the root
275
+ layout. Tags expire immediately (`{ expire: 0 }`).
276
+ - `publicationBasePath` refreshes the publication index plus `rss.xml` and
277
+ `feed.xml` on every call, and maps legacy `/blog/...` paths to the same URL
278
+ under the publication root, refreshing both.
279
+ - `secret` can be a getter (`() => process.env.SONOR_API_KEY`), read on every
280
+ call, so the handler can be created once at module scope.
281
+ - `extraPaths` are regenerated on every call. `extendPayload(payload, body)`
282
+ adds paths or tags from body fields the handler doesn't read; its result is
283
+ validated like the body. `@sonordev/agency-site-kit/revalidate` uses both
284
+ for portfolio hubs, `slug`/`slugs` and its default `portfolio` tag.
285
+
286
+ ---
287
+
288
+ ## llms.txt Generation
289
+
290
+ ### How `generateLLMsTxt()` Works
291
+
292
+ 1. Fetches all data via `getLLMsData()` → `GET /api/public/llms/data`
293
+ 2. Optionally merges with local data (`getLocalData` callback) when Sonor returns empty
294
+ 3. Builds markdown sections in order: **Header → About → Services → Portfolio → Contact → FAQ → Pages → Optional → Full Context Link → Knowledge Graph → Topic Clusters → Custom Sections**
295
+ 4. Portfolio, Entity, and Topic Cluster sections are fetched **in parallel** via `Promise.allSettled()`
296
+ 5. Returns `{ markdown, metadata }` where metadata includes `sections`, `attempted_sections`, and `failed_sections`
297
+
298
+ ### llms.txt Spec Compliance (llmstxt.org)
299
+
300
+ ```markdown
301
+ # Business Name
302
+
303
+ > Tagline or summary
304
+ > Content index last updated: 2026-04-08T00:00:00Z
305
+ > Primary language: en
306
+ > This information is provided for reference purposes only.
307
+
308
+ ## About
309
+
310
+ Business description...
311
+
312
+ ## Services
313
+
314
+ - [Service Name](/services/slug): Brief description
315
+
316
+ ## Portfolio & Case Studies
317
+
318
+ ### [Project Title](https://live-url.com)
319
+
320
+ Description of the project.
321
+
322
+ ## Contact Information
323
+
324
+ - **Phone:** 555-1234
325
+ - **Email:** info@example.com
326
+ - **Address:** 123 Main St, City, State
327
+
328
+ ## Frequently Asked Questions
329
+
330
+ ### How do you work?
331
+
332
+ We follow a proven process...
333
+
334
+ ## Site Pages
335
+
336
+ - [Home](https://example.com/): Public-safe summary of the page content
337
+ - [About](https://example.com/about): Learn about our team and mission
338
+
339
+ ## Optional
340
+
341
+ - [Privacy Policy](https://example.com/privacy): How we collect and use your information
342
+
343
+ ## Full context
344
+
345
+ - [llms-full.txt](https://example.com/llms-full.txt): Expanded index for large-context systems.
346
+
347
+ ## Knowledge Graph
348
+
349
+ ### Business Name (Primary)
350
+
351
+ - **Type:** Organization
352
+ - **Schema:** LocalBusiness
353
+
354
+ ## Topic Clusters
355
+
356
+ ### Family Law Basics
357
+ Topic: Family Law
358
+ Area: Cincinnati, OH
359
+ Articles: 8
360
+ Service page: https://example.com/services/family-law
361
+ Pillar: [Complete Guide to Family Law](/article/family-law-guide)
362
+ - [How to File for Divorce](/article/filing-divorce) (Mar 15, 2026)
363
+ - [Child Custody Laws](/article/custody-laws) (Mar 10, 2026)
364
+ ```
365
+
366
+ ### Configuration Options
367
+
368
+ ```ts
369
+ generateLLMsTxt({
370
+ // Section toggles (all default to true)
371
+ includeBusinessInfo: true,
372
+ includeServices: true,
373
+ includeFAQ: true,
374
+ includePages: true,
375
+ includeContact: true,
376
+ includePortfolio: true,
377
+ includeEntities: true,
378
+
379
+ // Limits
380
+ maxFAQItems: 20, // default 20 (100 in full mode)
381
+ maxPages: 50, // default 50 (200 in full mode)
382
+ maxPortfolioItems: 20, // default 20 (50 in full mode)
383
+ maxEntities: 50, // default 50 (200 in full mode)
384
+ maxArticlesPerCluster: 5, // default 5
385
+
386
+ // Spec features
387
+ linkToFullLlms: true, // Append "Full context" section
388
+ optionalPagePaths: ['/privacy', '/terms'], // Moved from Site Pages to ## Optional
389
+ pageListNotesFromPublicSummaryOnly: false, // Strict mode: no description fallback
390
+
391
+ // Overrides (take precedence over Sonor settings)
392
+ headerPrimaryLanguage: 'en',
393
+ headerDisclaimer: 'This information is for reference purposes only.',
394
+
395
+ // Custom sections
396
+ customSections: [{ title: 'Specializations', content: 'We specialize in...' }],
397
+
398
+ // Local data fallback when Sonor is empty
399
+ getLocalData: async () => ({ business: {...}, services: [...], ... }),
400
+ })
401
+ ```
402
+
403
+ ### Optional pages
404
+
405
+ `optionalPagePaths` demotes pages to `## Optional`, the section llmstxt.org lets a
406
+ short-context parser skip. A matching page **moves**: it's removed from
407
+ `## Site Pages` and listed once under Optional with the same URL and note. The
408
+ filter runs before `maxPages`, so demoting a page frees an index slot for the next
409
+ one. `/privacy`, `privacy` and `/privacy/` all match the same page. A path with no
410
+ matching page is still listed under Optional. The section needs a resolvable base
411
+ URL (see `baseUrl`); without one it's skipped and the pages stay in the index
412
+ rather than disappearing.
413
+
414
+ ### Metadata Returned
415
+
416
+ ```ts
417
+ const { markdown, metadata } = await generateLLMsTxt({})
418
+
419
+ metadata.generated_at // ISO timestamp
420
+ metadata.project_id // From API key
421
+ metadata.sections // ['header', 'about', 'services', 'faq', 'pages', ...]
422
+ metadata.attempted_sections // ['portfolio', 'knowledge-graph', 'topic-clusters']
423
+ metadata.failed_sections // ['knowledge-graph'] — if entity API was unavailable
424
+ ```
425
+
426
+ ---
427
+
428
+ ## HTTP Caching
429
+
430
+ All llms.txt responses use this header strategy:
431
+
432
+ ```
433
+ Content-Type: text/plain; charset=utf-8
434
+ Cache-Control: public, max-age=300, s-maxage=3600, stale-while-revalidate=86400, stale-if-error=86400
435
+ ETag: W/"<sha1-base64url>"
436
+ Vary: Accept-Encoding
437
+ ```
438
+
439
+ - **Browser:** caches 5 minutes, then revalidates
440
+ - **CDN:** caches 1 hour, serves stale for 24 hours while revalidating
441
+ - **304 support:** The Next static layer / CDN matches `If-None-Match` against the emitted
442
+ ETag. The handlers themselves never read request headers — doing so would force the
443
+ route dynamic and break static prerendering
444
+
445
+ The `llmsResponseHeaders(body, extra?)` helper is exported for custom route handlers.
446
+
447
+ ---
448
+
449
+ ## JSON-LD Integration
450
+
451
+ ### LLMSchema (via ManagedSchema)
452
+
453
+ The `LLMSchema` component in `@sonordev/site-kit/seo` emits `managed_llm_schema` from Sonor as JSON-LD:
454
+
455
+ ```html
456
+ <script type="application/ld+json">
457
+ {
458
+ "@context": "https://schema.org",
459
+ "@type": "WebPage",
460
+ "name": "Divorce Law",
461
+ "description": "Comprehensive divorce representation...",
462
+ "url": "https://example.com/divorce",
463
+ "additionalType": "https://sonor.io/ns/LLMOptimizedContent",
464
+ "isPartOf": {
465
+ "@type": "WebSite",
466
+ "@id": "https://example.com/#website",
467
+ "url": "https://example.com"
468
+ }
469
+ }
470
+ </script>
471
+ ```
472
+
473
+ - `managed_llm_schema` is generated by Signal AI during SEO meta optimization
474
+ - Only **known keys** from `MANAGED_LLM_SCHEMA_KNOWN_KEYS` are emitted (safety filter)
475
+ - `isPartOf` uses a stable `@id` pattern: `${siteUrl}/#website`
476
+
477
+ ### Organization Stub
478
+
479
+ `createWebSiteOrganizationStub()` from `@sonordev/site-kit/seo` creates paired Organization + WebSite entities for `@graph` injection with stable `@id` anchors (`/#organization`, `/#website`).
480
+
481
+ ---
482
+
483
+ ## AEO Components
484
+
485
+ Semantic HTML components with schema.org microdata and `data-sonor-*` attributes for AI extraction.
486
+
487
+ | Component | Schema Type | Use Case |
488
+ |-----------|------------|----------|
489
+ | `AEOBlock` | Question/Answer | FAQ-style Q&A content |
490
+ | `AEOSummary` | — | Key points lists (speakable) |
491
+ | `AEODefinition` | DefinedTerm | Term/definition pairs |
492
+ | `AEOSteps` / `AEOStep` | HowTo | Step-by-step processes |
493
+ | `AEOComparison` | — | Feature/option comparison tables |
494
+ | `AEOClaim` | Claim | Source-attributed factual claims |
495
+ | `AEOEntity` | — | Inline entity annotations |
496
+ | `AEOProvenanceList` | — | Citation source lists |
497
+ | `AEOCitedContent` | — | Content with numbered citations |
498
+
499
+ All components support:
500
+ - `speakable` prop → adds `data-speakable="true"` for voice assistants
501
+ - `entityId` prop → links to knowledge graph via `data-sonor-entity`
502
+ - `className` prop → custom styling
503
+
504
+ ### Example: Service Page with AEO
505
+
506
+ ```tsx
507
+ import { AEOSummary, AEOSteps, AEOStep, AEOBlock } from '@sonordev/site-kit/llms'
508
+
509
+ export default function DivorcePage() {
510
+ return (
511
+ <article>
512
+ <h1>Divorce Law Services</h1>
513
+
514
+ <AEOSummary
515
+ title="Key Facts"
516
+ points={[
517
+ '25+ years of family law experience',
518
+ 'Serving Northern Kentucky and Cincinnati',
519
+ 'Free initial consultation available',
520
+ ]}
521
+ speakable
522
+ />
523
+
524
+ <AEOSteps title="The Divorce Process" speakable>
525
+ <AEOStep name="Consultation" text="Meet with attorney to discuss your situation" position={1} />
526
+ <AEOStep name="File Petition" text="Submit divorce petition to circuit court" position={2} />
527
+ <AEOStep name="Negotiate" text="Work toward fair settlement" position={3} />
528
+ <AEOStep name="Finalize" text="Court issues final decree" position={4} />
529
+ </AEOSteps>
530
+
531
+ <AEOBlock type="answer" question="How long does a divorce take?" speakable>
532
+ Uncontested divorces typically take 60-90 days. Contested cases take 6-12 months.
533
+ </AEOBlock>
534
+ </article>
535
+ )
536
+ }
537
+ ```
538
+
539
+ ---
540
+
541
+ ## Speakable Schema
542
+
543
+ Marks page sections for voice assistant extraction via `SpeakableSpecification` JSON-LD.
544
+
545
+ ```tsx
546
+ import { SpeakableSchema } from '@sonordev/site-kit/llms'
547
+
548
+ <SpeakableSchema
549
+ type="WebPage"
550
+ name="Divorce Law"
551
+ url="https://example.com/divorce"
552
+ speakable={{ cssSelectors: ['h1', '[data-speakable="summary"]'] }}
553
+ />
554
+ ```
555
+
556
+ Default selectors by page type:
557
+
558
+ | Type | Selectors |
559
+ |------|-----------|
560
+ | `page` | `h1`, `[data-speakable]`, `.intro`, `[role="main"] > p:first-of-type` |
561
+ | `article` | `h1`, `.article-summary`, `article > p:first-of-type`, `[data-speakable]` |
562
+ | `service` | `h1`, `.service-description`, `[data-speakable="summary"]` |
563
+ | `faq` | `.faq-question`, `[data-speakable]` |
564
+ | `contact` | `h1`, `.contact-info`, `[itemprop="address"]` |
565
+
566
+ ---
567
+
568
+ ## Sonor API Endpoints
569
+
570
+ ### Public (API key auth via `x-api-key`)
571
+
572
+ | Endpoint | Purpose |
573
+ |----------|---------|
574
+ | `GET /api/public/llms/data` | All LLM visibility data (business, services, FAQ, pages, meta) |
575
+ | `GET /api/public/llms/txt` | AI-optimized llms.txt markdown (calls Signal AEO, with fallback) |
576
+ | `GET /api/public/llms/txt?full=true` | Extended version (200 pages, 100 FAQ) |
577
+ | `GET /api/public/llms/txt?publicSummaryOnly=true` | Strict mode: page notes from `llms_public_summary` only |
578
+ | `GET /api/public/llms/business` | Business info only |
579
+ | `GET /api/public/llms/services` | Services list |
580
+ | `GET /api/public/llms/faq` | FAQ items (`getFAQItems(projectId?, limit?, site?)` sends `?site=`, so a microsite gets its own FAQs plus project-wide ones) |
581
+ | `GET /api/public/llms/pages` | Page summaries |
582
+
583
+ ### Authenticated (dashboard)
584
+
585
+ | Endpoint | Purpose |
586
+ |----------|---------|
587
+ | `GET /seo/projects/:id/llms/preview` | Preview llms.txt markdown + stats |
588
+ | `GET /seo/projects/:id/llms/analytics?period=7` | AI crawler request breakdown |
589
+
590
+ ### Key Database Fields
591
+
592
+ | Table | Column | Purpose |
593
+ |-------|--------|---------|
594
+ | `seo_pages` | `llms_public_summary` | Public-safe summary for llms.txt link notes (max 400 chars) |
595
+ | `seo_pages` | `managed_llm_schema` | JSON-LD object for per-page LLM optimization |
596
+ | `seo_pages` | `language_alternates` | Optional hreflang map (locale → URL) |
597
+ | `seo_pages` | `llm_schema_generated_at` | When Signal last generated the schema |
598
+ | `projects.settings` | `primary_language` | BCP 47 tag for llms.txt blockquote |
599
+ | `projects.settings` | `llms_disclaimer` | Optional disclaimer line in blockquote |
600
+ | `llms_request_log` | `bot_class` | AI crawler classification for analytics |
601
+
602
+ ---
603
+
604
+ ## Environment Variables
605
+
606
+ ```bash
607
+ # Required (server-only — SiteKitLayout injects into client automatically):
608
+ SONOR_API_KEY=sonor_xxxxxxxx_xxxxx
609
+
610
+ # Optional:
611
+ SONOR_API_URL=https://api.sonor.io # Default
612
+ NEXT_PUBLIC_SITE_URL=https://example.com # For CLI status checks
613
+ REVALIDATION_SECRET=your_secret # For on-demand revalidation endpoint
614
+ ```
615
+
616
+ ---
617
+
618
+ ## CLI Validation
619
+
620
+ `npx sonor-setup status` runs a comprehensive llms.txt health check when `NEXT_PUBLIC_SITE_URL` is set:
621
+
622
+ - **HTTP status** — must be 200
623
+ - **Content structure** — H1 title, blockquote summary, H2 section count
624
+ - **Response headers** — Content-Type, Cache-Control (s-maxage), ETag presence
625
+ - **Freshness** — parses `last_updated` from blockquote, warns if > 7 days old
626
+
627
+ ---
628
+
629
+ ## Exports
630
+
631
+ ```ts
632
+ // Types
633
+ import type {
634
+ LLMBusinessInfo, LLMContactInfo, LLMService, LLMFAQItem,
635
+ LLMPageSummary, LLMPortfolioItem, LLMsDataResponse, LLMsPayloadMeta,
636
+ GenerateLLMSTxtOptions, LLMSTxtContent, WriteLLMsTxtOptions,
637
+ SpeakableConfig, SpeakableSchemaProps,
638
+ AEOBlockProps, AEOSummaryProps, AEODefinitionProps,
639
+ AEOClaimProps, AEOEntityProps, ContentProvenance,
640
+ AEOProvenanceListProps, AEOCitedContentProps,
641
+ AiDiscoveryHeadersOptions, LlmsLanguageAlternate,
642
+ } from '@sonordev/site-kit/llms'
643
+
644
+ // Contract (lightweight — safe for API imports)
645
+ import {
646
+ LLM_GEO_CONTRACT_VERSION, LLMS_PUBLIC_SUMMARY_MAX_LENGTH,
647
+ LLMS_DISCLAIMER_MAX_LENGTH, MANAGED_LLM_SCHEMA_KNOWN_KEYS,
648
+ sanitizeLlmsPublicSummary, sanitizeLlmsDisclaimerLine,
649
+ sanitizePrimaryLanguageTag, pickManagedLlmSchemaForJsonLd,
650
+ } from '@sonordev/site-kit/llms/contract'
651
+
652
+ // Generation
653
+ import { generateLLMsTxt, generateLLMsFullTxt } from '@sonordev/site-kit/llms'
654
+
655
+ // Route handlers
656
+ import {
657
+ createLLMsTxtHandler, createLLMsFullTxtHandler, llmsResponseHeaders,
658
+ } from '@sonordev/site-kit/llms'
659
+
660
+ // Discovery
661
+ import { buildAiDiscoveryHeaders } from '@sonordev/site-kit/llms'
662
+
663
+ // API data fetchers (React cache()-wrapped)
664
+ import {
665
+ getLLMsData, getBusinessInfo, getServices, getFAQItems,
666
+ getPageSummaries, getOptimizedLLMsTxt,
667
+ } from '@sonordev/site-kit/llms'
668
+
669
+ // Build-time
670
+ import { writeLLMsTxtToPublic } from '@sonordev/site-kit/llms'
671
+
672
+ // Revalidation
673
+ import { createLlmsRevalidateHandler } from '@sonordev/site-kit/llms'
674
+
675
+ // Speakable
676
+ import {
677
+ SpeakableSchema, createSpeakableSchema,
678
+ getSpeakableSelectorsForPage, DEFAULT_SPEAKABLE_SELECTORS,
679
+ } from '@sonordev/site-kit/llms'
680
+
681
+ // AEO Components
682
+ import {
683
+ AEOBlock, AEOSummary, AEODefinition, AEOSteps, AEOStep,
684
+ AEOComparison, AEOClaim, AEOEntity, AEOProvenanceList, AEOCitedContent,
685
+ } from '@sonordev/site-kit/llms'
686
+
687
+ // Proxy (separate import path)
688
+ import { createProxy } from '@sonordev/site-kit/proxy'
689
+ // Config type: { llmsDiscovery?: { siteUrl: string; llmsPath?: string } | false }
690
+ ```
691
+
692
+ ---
693
+
694
+ ## Testing
695
+
696
+ ```bash
697
+ # Run all llms tests (42 tests across 4 suites)
698
+ pnpm test src/llms/
699
+
700
+ # Test suites:
701
+ # - contract.test.ts — sanitizers, PII filtering, schema key allowlist
702
+ # - generateLLMsTxt.test.ts — generation: sections, limits, flags, fallbacks, metadata
703
+ # - handlers.test.ts — response headers, ETag, Cache-Control
704
+ # - discovery-headers.test.ts — Link header format, hreflang validation
705
+ ```
706
+
707
+ ### Manual verification
708
+
709
+ ```bash
710
+ # Check llms.txt
711
+ curl -s https://example.com/llms.txt | head -20
712
+
713
+ # Check headers
714
+ curl -sI https://example.com/llms.txt | grep -E 'ETag|Cache-Control|Content-Type|Link'
715
+
716
+ # Check discovery Link header on HTML page
717
+ curl -sI -H 'Accept: text/html' https://example.com/ | grep Link
718
+
719
+ # Check 304 support (served by the CDN/static layer, not the handler)
720
+ ETAG=$(curl -sI https://example.com/llms.txt | grep ETag | awk '{print $2}' | tr -d '\r')
721
+ curl -sI -H "If-None-Match: $ETAG" https://example.com/llms.txt | head -1
722
+ # Should return: HTTP/2 304
723
+ ```