@ultimat3/seo 27.8.1 → 27.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CLAUDE.md CHANGED
@@ -46,6 +46,15 @@ Tier 1. May import `@ultimat3/core`, `@ultimat3/schema`, `@ultimat3/i18n`. Nothi
46
46
  which costs the alternates that WERE right. `defaultLocaleUrl` answers `undefined` unless the
47
47
  default locale's own URL is among the ones this route emits — the same explicit-fallback shape
48
48
  `meta.ts`'s `hreflangSet` takes. Never re-derive it from `path`.
49
+ - **No `sections`, no byte changes.** The unsectioned sitemap is pinned by hash in
50
+ `sitemap-sections.test.ts`. Sorting, a child's own index `<lastmod>` and `<name>-<n>` naming
51
+ belong to the sectioned path only; a fix that must reach both paths re-pins the hash on purpose.
52
+ - **A page's `lastmod` never reads the ambient zone, and a future one is clamped.**
53
+ `sitemapDatetime` refuses a zoneless time before `Date.parse` can read it in the process's zone.
54
+ Refusing a future date would fail a whole enumeration for one skewed row; clamping to `now` is
55
+ the true upper bound. A route's `lastmod` string is the caller's and is written as given.
56
+ - **The hreflang cluster, `x-default` and the `<loc>`s read ONE trimmed list** — the page's
57
+ `locales`, minus what `disallow` blocks. Never trim one of the three alone.
49
58
  - **Errors name the file, not the URL.** `RouteRecord.file` is in every cause and every fix; an agent must be able to open the source without guessing.
50
59
  - **Fail closed, and core reads the key.** `isIndexable()` is `environment === 'production'` and
51
60
  nothing else — `staging`, a laptop, a typo and an unset variable all disallow. `ULTIMATE_ENV` has
package/README.md CHANGED
@@ -18,6 +18,7 @@ enforced where they are built: each builder throws.
18
18
  | `X_SEO_CANONICAL_MISMATCH` | `meta.canonical` does not resolve to the route's own URL | a wrong canonical de-indexes the page in favour of another |
19
19
  | `X_LD_INVALID` | a JSON-LD node missing a required schema.org field | invalid structured data drops the rich result silently |
20
20
  | `X_SITEMAP_TOO_LARGE` | the sitemap index exceeds 50,000 files | past the protocol limit the whole sitemap is discarded |
21
+ | `X_SITEMAP_ENTRY_INVALID` | a page's `lastmod` is not a `Date` or ISO 8601 with a zone, or its `locales` names a locale the sitemap is not built for | a wrong date teaches a crawler to distrust every date in the file; a typo'd locale drops the page from one silently |
21
22
 
22
23
  Performance budgets are **not** here: `x verify`'s `budgets` step is `@ultimat3/cli`'s
23
24
  `checkBudgets`, over the route manifest and the build's own stats, and it throws
@@ -47,8 +48,11 @@ a job boundary the class is gone and the `code` is what survives — match on th
47
48
  | `meta.ts` | the metadata model, `renderMeta()` → head tags: title template, canonical, robots, `og:*`, `twitter:*`, hreflang + `x-default`, `theme-color` per colour scheme. **`robots: { index: false }` withdraws every `og:*`, `article:*` and `twitter:*` tag** — declared or derived; there is no `social` switch |
48
49
  | `validate.ts` | the build gate — `validateMeta()` (`--json`-shaped) and `assertMeta()` |
49
50
  | `ld.ts` | typed JSON-LD builders; required fields are required in the **input type** |
50
- | `sitemap.ts` | `buildSitemap()` from the route table + each route's `prerender()`, per-locale alternates, automatic index splitting past 50k — the index stays `/sitemap.xml`, the parts are `/sitemaps/<n>.xml` (`SITEMAP_PARTS_DIR`) |
51
- | `robots.ts` | `buildRobots()`, environment-aware and fail-closed; `disallow` reaches **every** group, and a `User-agent: *` group is emitted for it when none is declared |
51
+ | `sitemap.ts` | `buildSitemap()` from the route table + each route's `prerender()`, per-locale alternates. One urlset at `/sitemap.xml`; past a cap (`SITEMAP_MAX_URLS` 50,000, `SITEMAP_MAX_BYTES` 50 MB) the index stays `/sitemap.xml` and the parts are `/sitemaps/<n>.xml` (`SITEMAP_PARTS_DIR`). With `sections` it is always an index of `/sitemaps/<name>-<n>.xml` — see below |
52
+ | `sitemap-sections.ts` | `sectionOf(path, sections)`: which named section a page files under (`/blog/**`, `/blog/*`, exact), first match wins, else `pages` |
53
+ | `sitemap-datetime.ts` | `sitemapDatetime(value, now)`: a page's `lastmod` as W3C Datetime in UTC, or `undefined` — a zoneless time is not a date here |
54
+ | `sitemap-xml.ts` | the `<urlset>` and `<sitemapindex>` XML, and `paginate()` — files that each fit both caps |
55
+ | `robots.ts` | `buildRobots()`, environment-aware and fail-closed; `disallow` reaches **every** group, and a `User-agent: *` group is emitted for it when none is declared |. `disallowMatcher(rules)` reads those rules as a crawler does — what `buildSitemap({ disallow })` leaves out
52
56
  | `rss.ts` | `buildFeed()` → RSS 2.0 + Atom + JSON Feed from one item list. Channel `author`/`copyright`/`icon` → Atom `<author>`/`<rights>`/`<icon>`; an item `author` → Atom `<author>`, RSS `<author>` when it has an email, `<dc:creator>` when it does not; an item `image` → Atom `<link rel="enclosure">`, RSS `<media:content medium="image">`. Extra RSS namespaces are declared only when used |
53
57
  | `feed-dates.ts` | the one place a feed timestamp is parsed or formatted — an item date that will not parse is *absent*, never `Invalid Date` and never a crash |
54
58
  | `images.ts` | `srcset` widths (never past the intrinsic width or `MAX_IMAGE_WIDTH`, 8192), modern formats before the original — `DEFAULT_FORMATS` is what the built-in driver encodes (`webp`); pass `formats: FORMAT_ORDER` for AVIF behind a CDN driver — inlined intrinsic dimensions, and `parseImageQuery()` — reads a minted URL back into a transform request |
@@ -64,6 +68,33 @@ ld.Article({ headline: 'Ship it', author: { name: 'Ada' } });
64
68
  A missing `datePublished` is a compile error, not a Search Console warning three
65
69
  weeks later.
66
70
 
71
+ ## Sitemap: pages, sections, caps
72
+
73
+ ```ts
74
+ import { buildSitemap, type RouteRecord } from '@ultimat3/seo';
75
+
76
+ declare const routes: readonly RouteRecord[];
77
+
78
+ await buildSitemap(routes, {
79
+ baseUrl: 'https://notificado.co',
80
+ locales: ['es-co', 'en'],
81
+ defaultLocale: 'es-co',
82
+ maxUrls: 1000, // page size; default 50,000
83
+ sections: [{ name: 'blog', match: '/blog/**' }], // omitted: one urlset
84
+ disallow: ['/panel'], // robots.txt rules: never listed
85
+ });
86
+ // → index /sitemap.xml · files /sitemaps/pages-1.xml, /sitemaps/blog-1.xml, /sitemaps/blog-2.xml …
87
+ ```
88
+
89
+ | Rule | |
90
+ |---|---|
91
+ | a page | `RouteRecord.prerender()` answers `'/blog/a'` or a `SitemapPage` — `{ path, lastmod?, locales? }`. `expandRoutePages()` reads both; `expandRoute()` still answers paths |
92
+ | `lastmod` | a `Date` or ISO 8601 with a zone, written in UTC. No zone, or no date: `X_SITEMAP_ENTRY_INVALID`. After `now`: clamped to `now` — one row from a clock a second ahead must not fail every page, and a future date is one a crawler stops believing |
93
+ | `locales` | per page, else `RouteRecord.locales`, else all. One `<url>` per listed locale, hreflang among those only; `x-default` only when the default is listed; a page left in one locale of several carries no alternates; `[]` lists it nowhere. An unknown name is `X_SITEMAP_ENTRY_INVALID` |
94
+ | sections | matched on the unprefixed path, so a page and its locale copies share a file. `pages` (reserved) first, then declaration order; inside one, sorted by path. Numbered from 1; an empty one writes nothing; the index `<lastmod>` of a child is the newest inside it (`SitemapFile.lastmod`) |
95
+ | no sections | byte-for-byte the output before sections existed — `sitemap-sections.test.ts` pins the hash |
96
+ | names | `sections` and `maxUrls` are refused by `@ultimat3/core`'s `sitemapSectionIssues`, the validator `defineConfig` runs for `seo.sitemap` |
97
+
67
98
  ## robots.txt is fail-closed
68
99
 
69
100
  Only the literal string `production` in `ULTIMATE_ENV` / `NODE_ENV` opts a deploy
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ultimat3/seo",
3
- "version": "27.8.1",
3
+ "version": "27.9.0",
4
4
  "description": "Enforced SEO: typed meta, JSON-LD, sitemap, robots, feeds, responsive images, perf budgets",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -32,7 +32,7 @@
32
32
  "test": "bun test"
33
33
  },
34
34
  "dependencies": {
35
- "@ultimat3/core": "27.8.1",
36
- "@ultimat3/schema": "27.8.1"
35
+ "@ultimat3/core": "27.9.0",
36
+ "@ultimat3/schema": "27.9.0"
37
37
  }
38
38
  }
package/src/errors.ts CHANGED
@@ -19,6 +19,7 @@ export const SEO_ERROR_CODES = {
19
19
  sitemapTooLarge: 'X_SITEMAP_TOO_LARGE',
20
20
  imageQueryInvalid: 'X_IMAGE_QUERY_INVALID',
21
21
  linkInvalid: 'X_SEO_LINK_INVALID',
22
+ sitemapEntryInvalid: 'X_SITEMAP_ENTRY_INVALID',
22
23
  } as const;
23
24
 
24
25
  export type SeoErrorCode = (typeof SEO_ERROR_CODES)[keyof typeof SEO_ERROR_CODES];
@@ -42,6 +43,9 @@ registerErrorCodes({
42
43
  X_SITEMAP_TOO_LARGE: { title: 'sitemap exceeds the 50,000-entry protocol limit' },
43
44
  X_IMAGE_QUERY_INVALID: { title: 'an image transform query parameter is present but unusable' },
44
45
  X_SEO_LINK_INVALID: { title: 'a route meta link is unsafe or would fetch nothing' },
46
+ X_SITEMAP_ENTRY_INVALID: {
47
+ title: 'a sitemap entry carries a lastmod or a locale the sitemap cannot write',
48
+ },
45
49
  });
46
50
 
47
51
  export interface SeoErrorInit {
@@ -148,6 +152,19 @@ export function sitemapTooLarge(count: number, max: number): SeoError {
148
152
  });
149
153
  }
150
154
 
155
+ /**
156
+ * A page a sitemap was asked to list with a `lastmod` or a locale it cannot write. `file` is the
157
+ * route's source — where the `prerender()` that answered the entry lives.
158
+ */
159
+ export function sitemapEntryInvalid(file: string, path: string, problem: string): SeoError {
160
+ return new SeoError({
161
+ code: SEO_ERROR_CODES.sitemapEntryInvalid,
162
+ cause: `${file} lists "${path}" in the sitemap with ${problem}`,
163
+ fix: `x verify --json # after prerender() in ${file} answers lastmod as a Date or an ISO 8601 string with a zone (row.updatedAt), and locales naming only locales the app routes`,
164
+ meta: { file, path, problem },
165
+ });
166
+ }
167
+
151
168
  /**
152
169
  * `parseImageQuery`'s only refusal: a `?w=`/`?q=`/`?f=` value present but not usable — serving
153
170
  * the untransformed original against a URL that asked for a size would be the layout shift this
package/src/index.ts CHANGED
@@ -13,6 +13,7 @@ export {
13
13
  notImplementedDriver,
14
14
  SEO_ERROR_CODES,
15
15
  SeoError,
16
+ sitemapEntryInvalid,
16
17
  sitemapTooLarge,
17
18
  } from './errors';
18
19
  export type {
@@ -98,9 +99,9 @@ export {
98
99
  TITLE_MAX_LENGTH,
99
100
  } from './meta';
100
101
  export type { RobotsConfig, RobotsGroup } from './robots';
101
- export { buildRobots, isIndexable } from './robots';
102
- export type { ChangeFreq, RouteRecord, Surface } from './routes';
103
- export { expandRoute, indexableRoutes, isDynamic } from './routes';
102
+ export { buildRobots, disallowMatcher, isIndexable } from './robots';
103
+ export type { ChangeFreq, RouteRecord, SitemapPage, Surface } from './routes';
104
+ export { expandRoute, expandRoutePages, indexableRoutes, isDynamic } from './routes';
104
105
  export type { BuildFeedOptions, Feed, FeedAuthor, FeedChannel, FeedItem } from './rss';
105
106
  export { buildFeed } from './rss';
106
107
  export type {
@@ -114,10 +115,14 @@ export {
114
115
  buildSitemap,
115
116
  chunk,
116
117
  SITEMAP_INDEX_MAX_FILES,
118
+ SITEMAP_MAX_BYTES,
117
119
  SITEMAP_MAX_URLS,
118
120
  SITEMAP_PARTS_DIR,
119
121
  sitemapUrls,
120
122
  } from './sitemap';
123
+ export { sitemapDatetime } from './sitemap-datetime';
124
+ export type { SitemapSection } from './sitemap-sections';
125
+ export { sectionOf } from './sitemap-sections';
121
126
  export type { MetaIssue, MetaValidationReport, ValidateMetaOptions } from './validate';
122
127
  export { assertMeta, validateMeta } from './validate';
123
128
  export { absoluteUrl, escapeXml } from './xml';
package/src/robots.ts CHANGED
@@ -96,3 +96,27 @@ export function buildRobots(config: RobotsConfig): string {
96
96
 
97
97
  return `${lines.join('\n').trimEnd()}\n`;
98
98
  }
99
+
100
+ /**
101
+ * `rule` as a crawler reads it: a PREFIX of the path, `*` standing for any run of characters and a
102
+ * trailing `$` anchoring the end. An empty rule disallows nothing.
103
+ */
104
+ function disallowPattern(rule: string): RegExp | undefined {
105
+ if (rule === '') return undefined;
106
+ const anchored = rule.endsWith('$');
107
+ const body = (anchored ? rule.slice(0, -1) : rule)
108
+ .split('*')
109
+ .map((literal) => literal.replace(/[.+?^${}()|[\]\\]/g, '\\$&'))
110
+ .join('.*');
111
+ return new RegExp(`^${body}${anchored ? '$' : ''}`);
112
+ }
113
+
114
+ /**
115
+ * A test for "do these `Disallow:` rules keep a crawler out of this path". What the sitemap asks
116
+ * before it lists a URL: one that `robots.txt` blocks is reported as "submitted URL blocked by
117
+ * robots.txt", and the two files are written from one config.
118
+ */
119
+ export function disallowMatcher(rules: readonly string[]): (path: string) => boolean {
120
+ const patterns = rules.flatMap((rule) => disallowPattern(rule) ?? []);
121
+ return (path) => patterns.some((pattern) => pattern.test(path));
122
+ }
package/src/routes.ts CHANGED
@@ -10,6 +10,25 @@ export type Surface = 'site' | 'app' | 'api';
10
10
 
11
11
  export type ChangeFreq = 'always' | 'hourly' | 'daily' | 'weekly' | 'monthly' | 'yearly' | 'never';
12
12
 
13
+ /**
14
+ * One concrete page of a route, as a sitemap lists it. A bare string is the same thing with nothing
15
+ * to add: `'/blog/a'` is `{ path: '/blog/a' }`.
16
+ */
17
+ export interface SitemapPage {
18
+ /** The concrete path, WITHOUT a locale prefix: `/blog/a`. */
19
+ readonly path: string;
20
+ /**
21
+ * When this page last changed: a `Date`, or ISO 8601 with an explicit zone (`2026-10-01`,
22
+ * `2026-10-01T15:04:05Z`). Written as W3C Datetime in UTC; wins over the route's own `lastmod`.
23
+ */
24
+ readonly lastmod?: string | Date;
25
+ /**
26
+ * The locales this page is LISTED in — one `<url>` each, cross-linked with hreflang among
27
+ * themselves only. Omitted: every locale the sitemap is built for. Empty: listed nowhere.
28
+ */
29
+ readonly locales?: readonly string[];
30
+ }
31
+
13
32
  export interface RouteRecord {
14
33
  /** URL pattern, e.g. `/blog/:slug`. */
15
34
  path: string;
@@ -18,8 +37,13 @@ export interface RouteRecord {
18
37
  surface: Surface;
19
38
  render: RenderMode;
20
39
  meta?: RouteMeta;
21
- /** Concrete paths for a dynamic route, from `defineRoute({ prerender })`. */
22
- prerender?: () => readonly string[] | Promise<readonly string[]>;
40
+ /** The route's concrete pages, from `defineRoute({ prerender })` — required for a dynamic one. */
41
+ prerender?: () => readonly (string | SitemapPage)[] | Promise<readonly (string | SitemapPage)[]>;
42
+ /**
43
+ * The locales every page of this route is listed in, unless the page names its own
44
+ * (`SitemapPage.locales`). Omitted: every locale the sitemap is built for.
45
+ */
46
+ locales?: readonly string[];
23
47
  /** Keep out of the sitemap and emit `noindex`. */
24
48
  noindex?: boolean;
25
49
  /**
@@ -47,9 +71,20 @@ export function indexableRoutes(routes: readonly RouteRecord[]): readonly RouteR
47
71
  );
48
72
  }
49
73
 
74
+ /**
75
+ * Every concrete page a route resolves to, each in the one object form. A route that declares
76
+ * `prerender()` is the pages it answers — with or without params: a page with none may still say
77
+ * its own `lastmod` and `locales` there. Without one, a static path is its one page and a dynamic
78
+ * one is none.
79
+ */
80
+ export async function expandRoutePages(route: RouteRecord): Promise<readonly SitemapPage[]> {
81
+ if (route.prerender === undefined) return isDynamic(route.path) ? [] : [{ path: route.path }];
82
+ return (await route.prerender()).map((page) =>
83
+ typeof page === 'string' ? { path: page } : page,
84
+ );
85
+ }
86
+
50
87
  /** Every concrete URL a route resolves to, expanding `prerender()`. */
51
88
  export async function expandRoute(route: RouteRecord): Promise<readonly string[]> {
52
- if (!isDynamic(route.path)) return [route.path];
53
- if (route.prerender === undefined) return [];
54
- return await route.prerender();
89
+ return (await expandRoutePages(route)).map((page) => page.path);
55
90
  }
@@ -0,0 +1,33 @@
1
+ // Single responsibility: one page's `lastmod`, as the W3C Datetime a sitemap writes — or the
2
+ // answer that it is not a date. No ambient zone is ever read: a time without one is refused.
3
+
4
+ /**
5
+ * ISO 8601 as a sitemap accepts it, with the zone REQUIRED on a time. `Date.parse` alone is not the
6
+ * test: it reads `2026-10-01T10:00:00` and `October 1, 2026` in the process's own zone, so one row
7
+ * would be two instants on a laptop in Bogotá and in a container on UTC.
8
+ */
9
+ const ISO_ZONED =
10
+ /^\d{4}-\d{2}-\d{2}(?:T\d{2}:\d{2}(?::\d{2}(?:\.\d{1,9})?)?(?:Z|[+-]\d{2}:\d{2}))?$/;
11
+
12
+ /** What a refusal quotes back, so the fix names the forms that are accepted. */
13
+ export const SITEMAP_DATETIME_FORMS =
14
+ 'a Date or an ISO 8601 date with an explicit zone (2026-10-01, 2026-10-01T15:04:05Z, 2026-10-01T10:04:05-05:00)';
15
+
16
+ /**
17
+ * `value` in UTC (`2026-10-01T15:04:05.000Z`), or `undefined` when it names no instant.
18
+ *
19
+ * A moment after `now` is CLAMPED to `now`, never refused and never written as it stands. Written,
20
+ * it is a date a crawler learns to distrust for the whole file. Refused, one row stamped by a
21
+ * database a second ahead of this process fails the enumeration for every page. Clamped, it says
22
+ * the one thing that is certainly true of a change that already happened: no later than now.
23
+ */
24
+ export function sitemapDatetime(value: unknown, now: Date): string | undefined {
25
+ const time =
26
+ value instanceof Date
27
+ ? value.getTime()
28
+ : typeof value === 'string' && ISO_ZONED.test(value)
29
+ ? Date.parse(value)
30
+ : Number.NaN;
31
+ if (!Number.isFinite(time)) return undefined;
32
+ return new Date(Math.min(time, now.getTime())).toISOString();
33
+ }
@@ -0,0 +1,50 @@
1
+ // Single responsibility: which named section of a sitemap index a page files under. The shape and
2
+ // its refusals are `@ultimat3/core`'s (`seo.sitemap.sections`), so `defineConfig` and a direct
3
+ // `buildSitemap` call reject the same things in the same words.
4
+
5
+ import {
6
+ assert,
7
+ SITEMAP_DEFAULT_SECTION,
8
+ type SitemapSectionConfig,
9
+ sectionPatterns,
10
+ sitemapSectionIssues,
11
+ } from '@ultimat3/core';
12
+
13
+ export type SitemapSection = SitemapSectionConfig;
14
+
15
+ /** A page's path as a pattern reads it: no query, no fragment, no trailing slash. */
16
+ function segmentsOf(path: string): readonly string[] {
17
+ const bare = path.replace(/[?#].*$/, '').replace(/\/+$/, '');
18
+ return bare === '' ? [] : bare.slice(1).split('/');
19
+ }
20
+
21
+ function matches(pattern: string, page: readonly string[]): boolean {
22
+ const wanted = segmentsOf(pattern);
23
+ const subtree = wanted.at(-1) === '**';
24
+ const fixed = subtree ? wanted.slice(0, -1) : wanted;
25
+ // `/blog/**` is `/blog` and everything below it: a section's own landing page files with it.
26
+ if (subtree ? page.length < fixed.length : page.length !== fixed.length) return false;
27
+ return fixed.every((segment, index) => segment === '*' || segment === page[index]);
28
+ }
29
+
30
+ /**
31
+ * The section `path` files under: the first one, in declaration order, with a pattern that matches
32
+ * — else the default. `path` is UNPREFIXED, so a page and its locale copies share one file.
33
+ */
34
+ export function sectionOf(path: string, sections: readonly SitemapSection[]): string {
35
+ const page = segmentsOf(path);
36
+ for (const section of sections) {
37
+ if (sectionPatterns(section).some((pattern) => matches(pattern, page))) return section.name;
38
+ }
39
+ return SITEMAP_DEFAULT_SECTION;
40
+ }
41
+
42
+ /** Refuses what `defineConfig` refuses, for a caller that built the options by hand. */
43
+ export function assertSections(sections: readonly SitemapSection[], pageSize: number): void {
44
+ const issues = sitemapSectionIssues(sections, pageSize);
45
+ assert(
46
+ issues.length === 0,
47
+ issues.join('; '),
48
+ "pass sections as [{ name: 'blog', match: '/blog/**' }] — a lower-case slug and a path pattern — and maxUrls from 1 to 50000",
49
+ );
50
+ }
@@ -0,0 +1,134 @@
1
+ // Single responsibility: the XML of a sitemap — one `<urlset>`, one `<sitemapindex>` — and the
2
+ // split of a list of URLs into files that each fit BOTH protocol caps, entries and bytes.
3
+
4
+ import { assert } from '@ultimat3/core';
5
+ import type { ChangeFreq } from './routes';
6
+ import { attributes, escapeXml } from './xml';
7
+
8
+ export interface SitemapAlternate {
9
+ hreflang: string;
10
+ href: string;
11
+ }
12
+
13
+ export interface SitemapUrl {
14
+ loc: string;
15
+ lastmod?: string;
16
+ changefreq?: ChangeFreq;
17
+ priority?: number;
18
+ alternates?: readonly SitemapAlternate[];
19
+ }
20
+
21
+ const HEAD = '<?xml version="1.0" encoding="UTF-8"?>\n';
22
+ const SITEMAP_NS = ' xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"';
23
+ const XHTML_NS = ' xmlns:xhtml="http://www.w3.org/1999/xhtml"';
24
+ /** The envelope around the `<url>` entries, at its largest: every byte of it is ASCII. */
25
+ const ENVELOPE_BYTES = `${HEAD}<urlset${SITEMAP_NS}${XHTML_NS}>\n\n</urlset>\n`.length;
26
+
27
+ function urlXml(url: SitemapUrl): string {
28
+ const parts = [` <loc>${escapeXml(url.loc)}</loc>`];
29
+ if (url.lastmod !== undefined) parts.push(` <lastmod>${escapeXml(url.lastmod)}</lastmod>`);
30
+ if (url.changefreq !== undefined) parts.push(` <changefreq>${url.changefreq}</changefreq>`);
31
+ if (url.priority !== undefined) {
32
+ parts.push(` <priority>${url.priority.toFixed(1)}</priority>`);
33
+ }
34
+ for (const alternate of url.alternates ?? []) {
35
+ parts.push(
36
+ ` <xhtml:link${attributes({ rel: 'alternate', hreflang: alternate.hreflang, href: alternate.href })}/>`,
37
+ );
38
+ }
39
+ return ` <url>\n${parts.join('\n')}\n </url>`;
40
+ }
41
+
42
+ export function urlSetXml(urls: readonly SitemapUrl[]): string {
43
+ const needsXhtml = urls.some((url) => (url.alternates?.length ?? 0) > 0);
44
+ const ns = needsXhtml ? `${SITEMAP_NS}${XHTML_NS}` : SITEMAP_NS;
45
+ return `${HEAD}<urlset${ns}>\n${urls.map(urlXml).join('\n')}\n</urlset>\n`;
46
+ }
47
+
48
+ export interface SitemapIndexEntry {
49
+ /** Absolute. */
50
+ readonly loc: string;
51
+ readonly lastmod?: string | undefined;
52
+ }
53
+
54
+ export function indexXml(entries: readonly SitemapIndexEntry[]): string {
55
+ const body = entries
56
+ .map((entry) => {
57
+ const lastmod =
58
+ entry.lastmod === undefined ? '' : `\n <lastmod>${escapeXml(entry.lastmod)}</lastmod>`;
59
+ return ` <sitemap>\n <loc>${escapeXml(entry.loc)}</loc>${lastmod}\n </sitemap>`;
60
+ })
61
+ .join('\n');
62
+ return `${HEAD}<sitemapindex${SITEMAP_NS}>\n${body}\n</sitemapindex>\n`;
63
+ }
64
+
65
+ export function chunk<T>(items: readonly T[], size: number): T[][] {
66
+ // The loop advances by `size`, so a non-positive one never moves the cursor: `maxUrls: 0` in a
67
+ // route config turned a build into an infinite loop allocating empty slices until the box ran
68
+ // out of memory. A fractional size is refused for a quieter reason — `slice` truncates it, so
69
+ // the groups silently stop being the size that was asked for.
70
+ assert(
71
+ Number.isSafeInteger(size) && size > 0,
72
+ `a chunk size must be a positive integer, got ${String(size)}: a non-positive step never advances and the loop cannot end`,
73
+ 'pass a positive integer — buildSitemap(routes, { baseUrl, maxUrls: 50000 }), the sitemaps.org bound SITEMAP_MAX_URLS already carries',
74
+ );
75
+ const out: T[][] = [];
76
+ for (let index = 0; index < items.length; index += size) {
77
+ out.push(items.slice(index, index + size));
78
+ }
79
+ return out;
80
+ }
81
+
82
+ const encoder = new TextEncoder();
83
+ const bytesOf = (text: string): number => encoder.encode(text).length;
84
+
85
+ /** One group of at most `maxUrls`, cut again wherever its XML would pass `maxBytes`. */
86
+ function splitByBytes(urls: readonly SitemapUrl[], maxBytes: number): SitemapUrl[][] {
87
+ const out: SitemapUrl[][] = [];
88
+ let current: SitemapUrl[] = [];
89
+ let size = ENVELOPE_BYTES;
90
+ for (const url of urls) {
91
+ const cost = bytesOf(urlXml(url)) + 1;
92
+ // A lone URL larger than the cap still gets a file: there is no smaller unit to write.
93
+ if (current.length > 0 && size + cost > maxBytes) {
94
+ out.push(current);
95
+ current = [];
96
+ size = ENVELOPE_BYTES;
97
+ }
98
+ current.push(url);
99
+ size += cost;
100
+ }
101
+ if (current.length > 0) out.push(current);
102
+ return out;
103
+ }
104
+
105
+ export interface SitemapPageXml {
106
+ readonly urls: readonly SitemapUrl[];
107
+ readonly xml: string;
108
+ }
109
+
110
+ /**
111
+ * `urls` as the files that hold them: at most `maxUrls` entries and at most `maxBytes` of XML each,
112
+ * in order. No URL at all is ONE empty urlset — `/sitemap.xml` always answers a document.
113
+ */
114
+ export function paginate(
115
+ urls: readonly SitemapUrl[],
116
+ maxUrls: number,
117
+ maxBytes: number,
118
+ ): readonly SitemapPageXml[] {
119
+ if (urls.length === 0) return [{ urls, xml: urlSetXml(urls) }];
120
+ const pages: SitemapPageXml[] = [];
121
+ for (const group of chunk(urls, maxUrls)) {
122
+ const xml = urlSetXml(group);
123
+ // UTF-8 is at most three bytes per UTF-16 unit: a file this short cannot pass the cap, and the
124
+ // common case never pays for an encode.
125
+ if (xml.length * 3 <= maxBytes || bytesOf(xml) <= maxBytes) {
126
+ pages.push({ urls: group, xml });
127
+ continue;
128
+ }
129
+ for (const part of splitByBytes(group, maxBytes)) {
130
+ pages.push({ urls: part, xml: urlSetXml(part) });
131
+ }
132
+ }
133
+ return pages;
134
+ }
package/src/sitemap.ts CHANGED
@@ -1,47 +1,51 @@
1
- // Sitemap generation from the route table. Dynamic routes contribute the URLs
2
- // their `prerender()` enumerates, so the sitemap can never drift from what the
3
- // build actually produced. Splits into an index past the 50,000-URL protocol cap.
1
+ // Sitemap generation from the route table. Dynamic routes contribute the pages their `prerender()`
2
+ // enumerates, so the sitemap can never drift from what the build actually produced. One urlset —
3
+ // split into an index past a protocol cap — or, with `sections`, an index of named, paginated parts.
4
4
 
5
- import { assert } from '@ultimat3/core';
6
- import { sitemapTooLarge } from './errors';
5
+ import { assert, SITEMAP_DEFAULT_SECTION } from '@ultimat3/core';
6
+ import { sitemapEntryInvalid, sitemapTooLarge } from './errors';
7
7
  import { hreflangTag } from './locale-tags';
8
- import { type ChangeFreq, expandRoute, indexableRoutes, type RouteRecord } from './routes';
9
- import { absoluteUrl, attributes, escapeXml } from './xml';
8
+ import { disallowMatcher } from './robots';
9
+ import { expandRoutePages, indexableRoutes, type RouteRecord, type SitemapPage } from './routes';
10
+ import { SITEMAP_DATETIME_FORMS, sitemapDatetime } from './sitemap-datetime';
11
+ import { assertSections, type SitemapSection, sectionOf } from './sitemap-sections';
12
+ import {
13
+ indexXml,
14
+ paginate,
15
+ type SitemapAlternate,
16
+ type SitemapPageXml,
17
+ type SitemapUrl,
18
+ } from './sitemap-xml';
19
+ import { absoluteUrl } from './xml';
10
20
 
11
- /** Both limits are from the sitemaps.org protocol. */
21
+ export type { SitemapAlternate, SitemapUrl } from './sitemap-xml';
22
+ export { chunk } from './sitemap-xml';
23
+
24
+ /** All three limits are from the sitemaps.org protocol; the byte one is of the UNCOMPRESSED file. */
12
25
  export const SITEMAP_MAX_URLS = 50_000;
13
26
  export const SITEMAP_INDEX_MAX_FILES = 50_000;
27
+ export const SITEMAP_MAX_BYTES = 52_428_800;
14
28
 
15
29
  /**
16
- * Where the parts of a split sitemap live: `/sitemaps/1.xml`, `/sitemaps/2.xml`, … — one directory,
17
- * so a running server answers every part from ONE route (`/sitemaps/:file`). `/sitemap-N.xml` needed
18
- * a parameter inside a segment, which no router here has, and every part 404'd from the web role.
30
+ * Where the parts of a sitemap index live: `/sitemaps/1.xml`, … for a split, `/sitemaps/blog-1.xml`,
31
+ * … for a section — one directory, so a running server answers every part from ONE route
32
+ * (`/sitemaps/:file`). `/sitemap-N.xml` needed a parameter inside a segment, which no router here
33
+ * has, and every part 404'd from the web role.
19
34
  */
20
35
  export const SITEMAP_PARTS_DIR = '/sitemaps';
21
36
 
22
- export interface SitemapAlternate {
23
- hreflang: string;
24
- href: string;
25
- }
26
-
27
- export interface SitemapUrl {
28
- loc: string;
29
- lastmod?: string;
30
- changefreq?: ChangeFreq;
31
- priority?: number;
32
- alternates?: readonly SitemapAlternate[];
33
- }
34
-
35
37
  export interface SitemapFile {
36
- /** Path relative to the site root: `/sitemap.xml`, or a part, `/sitemaps/1.xml`. */
38
+ /** Path relative to the site root: `/sitemap.xml`, or a part, `/sitemaps/blog-1.xml`. */
37
39
  path: string;
38
40
  xml: string;
39
41
  urlCount: number;
42
+ /** The newest `<lastmod>` inside a urlset — what the index says of it. Absent when none has one. */
43
+ lastmod?: string;
40
44
  }
41
45
 
42
46
  export interface SitemapResult {
43
47
  readonly files: readonly SitemapFile[];
44
- /** Present only when the URLs did not fit in a single file. */
48
+ /** Present when `/sitemap.xml` is an index: sections were declared, or one file did not fit. */
45
49
  readonly index: SitemapFile | undefined;
46
50
  readonly urlCount: number;
47
51
  }
@@ -57,8 +61,23 @@ export interface BuildSitemapOptions {
57
61
  * emitted at all** — there is no unprefixed URL in the sitemap for it to name.
58
62
  */
59
63
  defaultLocale?: string;
64
+ /** `<url>` entries per file — the page size of a section. Defaults to the protocol's 50,000. */
60
65
  maxUrls?: number;
66
+ /** Bytes of XML per file. Defaults to the protocol's 50 MB; a test passes less. */
67
+ maxBytes?: number;
61
68
  lastmod?: string;
69
+ /**
70
+ * Named parts. Declared, `/sitemap.xml` is ALWAYS an index — of `/sitemaps/<name>-<n>.xml`, the
71
+ * pages no section matches under `pages` — whatever the URL count. Omitted or empty: one urlset.
72
+ */
73
+ sections?: readonly SitemapSection[];
74
+ /**
75
+ * `robots.txt` `Disallow:` rules. A URL one of them matches is not listed, and leaves its
76
+ * hreflang cluster: a crawler told to stay out of a URL must not be handed it to fetch.
77
+ */
78
+ disallow?: readonly string[];
79
+ /** The moment a page's `lastmod` may not be later than. Omitted: when the sitemap is built. */
80
+ now?: Date;
62
81
  }
63
82
 
64
83
  function localize(path: string, locale: string, options: BuildSitemapOptions): string {
@@ -85,29 +104,83 @@ function defaultLocaleUrl(
85
104
  return localised.find((entry) => entry.locale === options.defaultLocale)?.path;
86
105
  }
87
106
 
88
- /** Every concrete URL the route table produces, with per-locale alternates. */
89
- export async function sitemapUrls(
107
+ /** One page and every `<url>` it is listed as — its locale copies, which file and move together. */
108
+ interface Cluster {
109
+ /** The page's UNPREFIXED path: what a section matches and what a section sorts by. */
110
+ readonly page: string;
111
+ readonly urls: readonly SitemapUrl[];
112
+ }
113
+
114
+ /**
115
+ * The locales `page` is listed in, in the SITEMAP's order. A name outside `locales` is refused: a
116
+ * typo (`en-us` for `en`) would otherwise drop the page from a locale without a word.
117
+ */
118
+ function listedLocales(
119
+ route: RouteRecord,
120
+ page: SitemapPage,
121
+ locales: readonly string[],
122
+ ): readonly string[] {
123
+ const wanted = page.locales ?? route.locales;
124
+ if (wanted === undefined) return locales;
125
+ const unknown = wanted.find((locale) => !locales.includes(locale));
126
+ if (unknown !== undefined) {
127
+ throw sitemapEntryInvalid(
128
+ route.file,
129
+ page.path,
130
+ `the locale "${unknown}", which is not one of the routed locales (${locales.join(', ')})`,
131
+ );
132
+ }
133
+ return locales.filter((locale) => wanted.includes(locale));
134
+ }
135
+
136
+ function lastmodOf(
137
+ route: RouteRecord,
138
+ page: SitemapPage,
139
+ options: BuildSitemapOptions,
140
+ now: Date,
141
+ ): string | undefined {
142
+ if (page.lastmod === undefined) return route.lastmod ?? options.lastmod;
143
+ const lastmod = sitemapDatetime(page.lastmod, now);
144
+ if (lastmod !== undefined) return lastmod;
145
+ const written =
146
+ page.lastmod instanceof Date ? 'an invalid Date' : JSON.stringify(String(page.lastmod));
147
+ throw sitemapEntryInvalid(
148
+ route.file,
149
+ page.path,
150
+ `the lastmod ${written}, which is not ${SITEMAP_DATETIME_FORMS}`,
151
+ );
152
+ }
153
+
154
+ async function sitemapClusters(
90
155
  routes: readonly RouteRecord[],
91
156
  options: BuildSitemapOptions,
92
- ): Promise<readonly SitemapUrl[]> {
93
- const urls: SitemapUrl[] = [];
157
+ ): Promise<readonly Cluster[]> {
158
+ const clusters: Cluster[] = [];
94
159
  const locales = options.locales ?? [];
160
+ const disallowed = disallowMatcher(options.disallow ?? []);
161
+ const now = options.now ?? new Date();
95
162
 
96
163
  for (const route of indexableRoutes(routes)) {
97
- for (const path of await expandRoute(route)) {
164
+ for (const page of await expandRoutePages(route)) {
165
+ const path = page.path;
98
166
  // Localised once, then read three times — the alternates, the `x-default` candidate and the
99
167
  // `<loc>`s below are three questions with one answer, and they drifted apart when each
100
- // computed its own.
101
- const localised = locales.map((locale) => ({
102
- locale,
103
- path: localize(path, locale, options),
104
- }));
105
- const alternates: SitemapAlternate[] = localised.map((entry) => ({
106
- // BCP 47 region form (`es-CO`), the spelling `renderMeta` puts in the head: one cluster,
107
- // one vocabulary, whether a crawler reads the page or the sitemap.
108
- hreflang: hreflangTag(entry.locale),
109
- href: absoluteUrl(options.baseUrl, entry.path),
110
- }));
168
+ // computed its own. Trimmed HERE, to the locales the page is listed in and the URLs a
169
+ // crawler may fetch, so all three read the same trimmed list.
170
+ const localised = listedLocales(route, page, locales)
171
+ .map((locale) => ({ locale, path: localize(path, locale, options) }))
172
+ .filter((entry) => !disallowed(entry.path));
173
+ // A page left in ONE locale of several has no alternate to name: the cluster is the page.
174
+ // (A sitemap built for one locale keeps its cluster of one — it always wrote it.)
175
+ const alone = localised.length === 1 && locales.length > 1;
176
+ const alternates: SitemapAlternate[] = alone
177
+ ? []
178
+ : localised.map((entry) => ({
179
+ // BCP 47 region form (`es-CO`), the spelling `renderMeta` puts in the head: one cluster,
180
+ // one vocabulary, whether a crawler reads the page or the sitemap.
181
+ hreflang: hreflangTag(entry.locale),
182
+ href: absoluteUrl(options.baseUrl, entry.path),
183
+ }));
111
184
  // `x-default` only when it names a URL THIS sitemap lists. It was the bare `path` for every
112
185
  // route: with `locales` set and no `defaultLocale`, every `<loc>` is prefixed and the
113
186
  // unprefixed path is one the sitemap never mentions — an hreflang cluster pointing at a URL
@@ -119,77 +192,81 @@ export async function sitemapUrls(
119
192
  alternates.push({ hreflang: 'x-default', href: absoluteUrl(options.baseUrl, fallback) });
120
193
  }
121
194
 
122
- const lastmod = route.lastmod ?? options.lastmod;
123
- const emitFor = locales.length === 0 ? [path] : localised.map((entry) => entry.path);
124
- for (const emitted of emitFor) {
125
- urls.push({
126
- loc: absoluteUrl(options.baseUrl, emitted),
127
- ...(lastmod === undefined ? {} : { lastmod }),
128
- ...(route.changefreq === undefined ? {} : { changefreq: route.changefreq }),
129
- ...(route.priority === undefined ? {} : { priority: route.priority }),
130
- ...(alternates.length === 0 ? {} : { alternates }),
131
- });
132
- }
195
+ const lastmod = lastmodOf(route, page, options, now);
196
+ // No locales at all: the page is its one URL — unless it asked to be listed nowhere.
197
+ const nowhere = (page.locales ?? route.locales)?.length === 0;
198
+ const unlocalised = nowhere || disallowed(path) ? [] : [path];
199
+ const emitFor = locales.length === 0 ? unlocalised : localised.map((entry) => entry.path);
200
+ const urls = emitFor.map((emitted) => ({
201
+ loc: absoluteUrl(options.baseUrl, emitted),
202
+ ...(lastmod === undefined ? {} : { lastmod }),
203
+ ...(route.changefreq === undefined ? {} : { changefreq: route.changefreq }),
204
+ ...(route.priority === undefined ? {} : { priority: route.priority }),
205
+ ...(alternates.length === 0 ? {} : { alternates }),
206
+ }));
207
+ if (urls.length > 0) clusters.push({ page: path, urls });
133
208
  }
134
209
  }
135
- return urls;
210
+ return clusters;
136
211
  }
137
212
 
138
- function renderUrlSet(urls: readonly SitemapUrl[]): string {
139
- const needsXhtml = urls.some((url) => (url.alternates?.length ?? 0) > 0);
140
- const ns = needsXhtml
141
- ? ' xmlns="http://www.sitemaps.org/schemas/sitemap/0.9" xmlns:xhtml="http://www.w3.org/1999/xhtml"'
142
- : ' xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"';
143
- const body = urls
144
- .map((url) => {
145
- const parts = [` <loc>${escapeXml(url.loc)}</loc>`];
146
- if (url.lastmod !== undefined) parts.push(` <lastmod>${escapeXml(url.lastmod)}</lastmod>`);
147
- if (url.changefreq !== undefined) {
148
- parts.push(` <changefreq>${url.changefreq}</changefreq>`);
149
- }
150
- if (url.priority !== undefined) {
151
- parts.push(` <priority>${url.priority.toFixed(1)}</priority>`);
152
- }
153
- for (const alternate of url.alternates ?? []) {
154
- parts.push(
155
- ` <xhtml:link${attributes({ rel: 'alternate', hreflang: alternate.hreflang, href: alternate.href })}/>`,
156
- );
157
- }
158
- return ` <url>\n${parts.join('\n')}\n </url>`;
159
- })
160
- .join('\n');
161
- return `<?xml version="1.0" encoding="UTF-8"?>\n<urlset${ns}>\n${body}\n</urlset>\n`;
213
+ /** Every concrete URL the route table produces, with per-locale alternates. */
214
+ export async function sitemapUrls(
215
+ routes: readonly RouteRecord[],
216
+ options: BuildSitemapOptions,
217
+ ): Promise<readonly SitemapUrl[]> {
218
+ return (await sitemapClusters(routes, options)).flatMap((cluster) => cluster.urls);
162
219
  }
163
220
 
164
- function renderIndex(files: readonly SitemapFile[], options: BuildSitemapOptions): string {
165
- const body = files
166
- .map((file) => {
167
- const loc = escapeXml(absoluteUrl(options.baseUrl, file.path));
168
- const lastmod =
169
- options.lastmod === undefined
170
- ? ''
171
- : `\n <lastmod>${escapeXml(options.lastmod)}</lastmod>`;
172
- return ` <sitemap>\n <loc>${loc}</loc>${lastmod}\n </sitemap>`;
173
- })
174
- .join('\n');
175
- return `<?xml version="1.0" encoding="UTF-8"?>\n<sitemapindex xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">\n${body}\n</sitemapindex>\n`;
221
+ /** The newest `<lastmod>` among `urls`, as written — compared as instants, not as text. */
222
+ function newestLastmod(urls: readonly SitemapUrl[]): string | undefined {
223
+ let newest: string | undefined;
224
+ let at = Number.NEGATIVE_INFINITY;
225
+ for (const url of urls) {
226
+ const time = url.lastmod === undefined ? Number.NaN : Date.parse(url.lastmod);
227
+ if (Number.isFinite(time) && time > at) {
228
+ at = time;
229
+ newest = url.lastmod;
230
+ }
231
+ }
232
+ return newest;
176
233
  }
177
234
 
178
- export function chunk<T>(items: readonly T[], size: number): T[][] {
179
- // The loop advances by `size`, so a non-positive one never moves the cursor: `maxUrls: 0` in a
180
- // route config turned a build into an infinite loop allocating empty slices until the box ran
181
- // out of memory. A fractional size is refused for a quieter reason — `slice` truncates it, so
182
- // the groups silently stop being the size that was asked for.
183
- assert(
184
- Number.isSafeInteger(size) && size > 0,
185
- `a chunk size must be a positive integer, got ${String(size)}: a non-positive step never advances and the loop cannot end`,
186
- 'pass a positive integer — buildSitemap(routes, { baseUrl, maxUrls: 50000 }), the sitemaps.org bound SITEMAP_MAX_URLS already carries',
187
- );
188
- const out: T[][] = [];
189
- for (let index = 0; index < items.length; index += size) {
190
- out.push(items.slice(index, index + size));
191
- }
192
- return out;
235
+ const fileOf = (path: string, page: SitemapPageXml): SitemapFile => {
236
+ const lastmod = newestLastmod(page.urls);
237
+ return {
238
+ path,
239
+ xml: page.xml,
240
+ urlCount: page.urls.length,
241
+ ...(lastmod === undefined ? {} : { lastmod }),
242
+ };
243
+ };
244
+
245
+ const comparePages = (a: Cluster, b: Cluster): number =>
246
+ a.page < b.page ? -1 : a.page > b.page ? 1 : 0;
247
+
248
+ /**
249
+ * The files of a sectioned sitemap, the default section first and the rest in declaration order.
250
+ * Inside a section pages are SORTED by path: the bytes then depend on which pages exist and never
251
+ * on the order a query happened to return them in, so a crawler re-reading a page of a section
252
+ * diffs it against the same neighbours. A section with nothing in it writes no file.
253
+ */
254
+ function sectionFiles(
255
+ clusters: readonly Cluster[],
256
+ sections: readonly SitemapSection[],
257
+ maxUrls: number,
258
+ maxBytes: number,
259
+ ): readonly SitemapFile[] {
260
+ const names = [SITEMAP_DEFAULT_SECTION, ...sections.map((section) => section.name)];
261
+ const filed = new Map<string, Cluster[]>(names.map((name) => [name, []]));
262
+ for (const cluster of clusters) filed.get(sectionOf(cluster.page, sections))?.push(cluster);
263
+ return names.flatMap((name) => {
264
+ const urls = (filed.get(name) ?? []).sort(comparePages).flatMap((cluster) => cluster.urls);
265
+ if (urls.length === 0) return [];
266
+ return paginate(urls, maxUrls, maxBytes).map((page, position) =>
267
+ fileOf(`${SITEMAP_PARTS_DIR}/${name}-${String(position + 1)}.xml`, page),
268
+ );
269
+ });
193
270
  }
194
271
 
195
272
  export async function buildSitemap(
@@ -197,6 +274,7 @@ export async function buildSitemap(
197
274
  options: BuildSitemapOptions,
198
275
  ): Promise<SitemapResult> {
199
276
  const maxUrls = options.maxUrls ?? SITEMAP_MAX_URLS;
277
+ const maxBytes = options.maxBytes ?? SITEMAP_MAX_BYTES;
200
278
  // Refused here and not only in `chunk`, so the answer does not depend on how many URLs the site
201
279
  // happens to have today: `maxUrls: 2.5` is a typo whether or not this build has enough routes to
202
280
  // reach the split, exactly as a metric refuses `maxSeries: 1.5` at declaration.
@@ -205,34 +283,62 @@ export async function buildSitemap(
205
283
  `buildSitemap({ maxUrls }) must be a positive integer, got ${String(maxUrls)}`,
206
284
  'pass a positive integer — buildSitemap(routes, { baseUrl, maxUrls: 50000 }) — or omit it and take SITEMAP_MAX_URLS, the sitemaps.org bound',
207
285
  );
208
- const urls = await sitemapUrls(routes, options);
286
+ assert(
287
+ Number.isSafeInteger(maxBytes) && maxBytes > 0,
288
+ `buildSitemap({ maxBytes }) must be a positive integer, got ${String(maxBytes)}`,
289
+ 'pass a positive integer — buildSitemap(routes, { baseUrl, maxBytes: 52428800 }) — or omit it and take SITEMAP_MAX_BYTES, the sitemaps.org bound',
290
+ );
291
+ const sections = options.sections ?? [];
292
+ if (sections.length > 0) assertSections(sections, Math.min(maxUrls, SITEMAP_MAX_URLS));
293
+
294
+ const clusters = await sitemapClusters(routes, options);
295
+ const urlCount = clusters.reduce((count, cluster) => count + cluster.urls.length, 0);
296
+ // Counted before a single file is rendered: a site this far past the index cap would otherwise
297
+ // build tens of thousands of documents on the way to a refusal.
298
+ if (Math.ceil(urlCount / maxUrls) > SITEMAP_INDEX_MAX_FILES) {
299
+ throw sitemapTooLarge(Math.ceil(urlCount / maxUrls), SITEMAP_INDEX_MAX_FILES);
300
+ }
209
301
 
210
- if (urls.length <= maxUrls) {
302
+ if (sections.length > 0 && urlCount > 0) {
303
+ const files = sectionFiles(clusters, sections, maxUrls, maxBytes);
304
+ if (files.length > SITEMAP_INDEX_MAX_FILES) {
305
+ throw sitemapTooLarge(files.length, SITEMAP_INDEX_MAX_FILES);
306
+ }
307
+ const entries = files.map((file) => ({
308
+ loc: absoluteUrl(options.baseUrl, file.path),
309
+ lastmod: file.lastmod,
310
+ }));
211
311
  return {
212
- files: [{ path: '/sitemap.xml', xml: renderUrlSet(urls), urlCount: urls.length }],
213
- index: undefined,
214
- urlCount: urls.length,
312
+ files,
313
+ index: { path: '/sitemap.xml', xml: indexXml(entries), urlCount: files.length },
314
+ urlCount,
215
315
  };
216
316
  }
217
317
 
218
- const groups = chunk(urls, maxUrls);
219
- if (groups.length > SITEMAP_INDEX_MAX_FILES) {
220
- throw sitemapTooLarge(groups.length, SITEMAP_INDEX_MAX_FILES);
318
+ const pages = paginate(
319
+ clusters.flatMap((cluster) => cluster.urls),
320
+ maxUrls,
321
+ maxBytes,
322
+ );
323
+ const [only] = pages;
324
+ if (pages.length === 1 && only !== undefined) {
325
+ return { files: [fileOf('/sitemap.xml', only)], index: undefined, urlCount };
221
326
  }
222
-
223
- const files: SitemapFile[] = groups.map((group, position) => ({
224
- path: `${SITEMAP_PARTS_DIR}/${position + 1}.xml`,
225
- xml: renderUrlSet(group),
226
- urlCount: group.length,
327
+ if (pages.length > SITEMAP_INDEX_MAX_FILES) {
328
+ throw sitemapTooLarge(pages.length, SITEMAP_INDEX_MAX_FILES);
329
+ }
330
+ const files = pages.map((page, position) =>
331
+ fileOf(`${SITEMAP_PARTS_DIR}/${String(position + 1)}.xml`, page),
332
+ );
333
+ // An unsectioned split says of every part what the caller said of the build (`options.lastmod`):
334
+ // the bytes it always wrote. Sections are where a part's own newest date is read.
335
+ const entries = files.map((file) => ({
336
+ loc: absoluteUrl(options.baseUrl, file.path),
337
+ lastmod: options.lastmod,
227
338
  }));
228
-
229
339
  return {
230
340
  files,
231
- index: {
232
- path: '/sitemap.xml',
233
- xml: renderIndex(files, options),
234
- urlCount: files.length,
235
- },
236
- urlCount: urls.length,
341
+ index: { path: '/sitemap.xml', xml: indexXml(entries), urlCount: files.length },
342
+ urlCount,
237
343
  };
238
344
  }