@shuji-bonji/rfcxml-mcp 0.6.13 → 0.6.53

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/README.ja.md +178 -50
  2. package/README.md +181 -51
  3. package/dist/cli/prefetch.d.ts +6 -1
  4. package/dist/cli/prefetch.d.ts.map +1 -1
  5. package/dist/cli/prefetch.js +53 -23
  6. package/dist/cli/prefetch.js.map +1 -1
  7. package/dist/config.d.ts +0 -2
  8. package/dist/config.d.ts.map +1 -1
  9. package/dist/config.js +3 -2
  10. package/dist/config.js.map +1 -1
  11. package/dist/constants.d.ts +16 -0
  12. package/dist/constants.d.ts.map +1 -1
  13. package/dist/constants.js +80 -7
  14. package/dist/constants.js.map +1 -1
  15. package/dist/services/checklist-generator.d.ts.map +1 -1
  16. package/dist/services/checklist-generator.js +72 -7
  17. package/dist/services/checklist-generator.js.map +1 -1
  18. package/dist/services/rfc-fetcher.d.ts +19 -1
  19. package/dist/services/rfc-fetcher.d.ts.map +1 -1
  20. package/dist/services/rfc-fetcher.js +116 -19
  21. package/dist/services/rfc-fetcher.js.map +1 -1
  22. package/dist/services/rfc-service.d.ts +11 -1
  23. package/dist/services/rfc-service.d.ts.map +1 -1
  24. package/dist/services/rfc-service.js +45 -5
  25. package/dist/services/rfc-service.js.map +1 -1
  26. package/dist/services/rfc-text-parser.d.ts.map +1 -1
  27. package/dist/services/rfc-text-parser.js +1594 -88
  28. package/dist/services/rfc-text-parser.js.map +1 -1
  29. package/dist/services/rfcxml-parser.d.ts.map +1 -1
  30. package/dist/services/rfcxml-parser.js +366 -98
  31. package/dist/services/rfcxml-parser.js.map +1 -1
  32. package/dist/tools/definitions.d.ts +8 -0
  33. package/dist/tools/definitions.d.ts.map +1 -1
  34. package/dist/tools/definitions.js +17 -1
  35. package/dist/tools/definitions.js.map +1 -1
  36. package/dist/tools/handlers.d.ts +24 -12
  37. package/dist/tools/handlers.d.ts.map +1 -1
  38. package/dist/tools/handlers.js +161 -36
  39. package/dist/tools/handlers.js.map +1 -1
  40. package/dist/types/index.d.ts +22 -2
  41. package/dist/types/index.d.ts.map +1 -1
  42. package/dist/utils/cache.d.ts +19 -0
  43. package/dist/utils/cache.d.ts.map +1 -1
  44. package/dist/utils/cache.js +32 -0
  45. package/dist/utils/cache.js.map +1 -1
  46. package/dist/utils/disk-cache.d.ts +10 -6
  47. package/dist/utils/disk-cache.d.ts.map +1 -1
  48. package/dist/utils/disk-cache.js +13 -9
  49. package/dist/utils/disk-cache.js.map +1 -1
  50. package/dist/utils/logger.d.ts.map +1 -1
  51. package/dist/utils/logger.js +3 -1
  52. package/dist/utils/logger.js.map +1 -1
  53. package/dist/utils/requirement-extractor.d.ts.map +1 -1
  54. package/dist/utils/requirement-extractor.js +413 -21
  55. package/dist/utils/requirement-extractor.js.map +1 -1
  56. package/dist/utils/section.d.ts.map +1 -1
  57. package/dist/utils/section.js +11 -0
  58. package/dist/utils/section.js.map +1 -1
  59. package/dist/utils/statement-matcher.d.ts +69 -4
  60. package/dist/utils/statement-matcher.d.ts.map +1 -1
  61. package/dist/utils/statement-matcher.js +524 -48
  62. package/dist/utils/statement-matcher.js.map +1 -1
  63. package/dist/utils/text.d.ts +62 -0
  64. package/dist/utils/text.d.ts.map +1 -1
  65. package/dist/utils/text.js +273 -18
  66. package/dist/utils/text.js.map +1 -1
  67. package/package.json +5 -2
@@ -4,7 +4,7 @@
4
4
  */
5
5
  import { XMLParser } from 'fast-xml-parser';
6
6
  import { createRequirementRegex } from '../constants.js';
7
- import { extractCrossReferences, toArray } from '../utils/text.js';
7
+ import { extractCrossReferences, extractRequirementMarkers as extractMarkers, toArray, dropNonDefinitions, } from '../utils/text.js';
8
8
  import { compareSectionNumbers, normalizeSectionNumber } from '../utils/section.js';
9
9
  import { extractRequirementsFromSections, } from '../utils/requirement-extractor.js';
10
10
  /**
@@ -203,14 +203,26 @@ export function parseRFCXML(xml) {
203
203
  // xref を先に解くのは、`<em><xref/></em>` のような入れ子で内側から
204
204
  // 組み立てるため。
205
205
  const inlineRendered = renderInlineTags(renderXrefTags(normalizeBcp14Tags(xml)));
206
- const normalizedXml = stripNonPrinting(inlineRendered);
206
+ // `<table>` の `pn` は節の中での位置を持たないので、パース前に位置を書き込む。
207
+ const normalizedXml = annotateTableOrder(stripNonPrinting(inlineRendered));
207
208
  const parsed = parser.parse(normalizedXml);
208
209
  const rfc = parsed.rfc || parsed;
209
210
  return {
210
211
  metadata: extractMetadata(rfc),
211
- sections: extractSections(rfc.middle?.section || []),
212
+ // 後付録は `<back>` に置かれる。`<middle>` だけを見ていたため、
213
+ // `get_rfc_structure` に後付録が 1 つも出ていなかった。RFC 9114 の
214
+ // Appendix A.2.5 には本物の定義があり、`get_definitions` はそれを
215
+ // §A.2.5 として返すのに、その節が構造に無い状態だった。
216
+ sections: [
217
+ ...extractSections(rfc.middle?.section || []),
218
+ // 参考文献の欄も RFC の節である。テキスト経路は §19 References を
219
+ // 節として返すのに、XML 経路は返さず、同じ RFC の目次が経路によって
220
+ // 食い違っていた(RFC 9110 §19 / 9112 §13 / 9114 §12)。
221
+ ...extractReferenceSections(rfc.back?.references || []),
222
+ ...extractSections(rfc.back?.section || []),
223
+ ],
212
224
  references: extractReferences(rfc.back?.references || []),
213
- definitions: mergeDefinitions(extractDefinitions(rfc), extractIrefDefinitions(inlineRendered)),
225
+ definitions: dropNonDefinitions(mergeDefinitions(extractDefinitions(rfc), extractIrefDefinitions(inlineRendered))),
214
226
  };
215
227
  }
216
228
  /**
@@ -268,76 +280,258 @@ export function extractPublicationDate(dateNode) {
268
280
  /**
269
281
  * セクション構造の抽出
270
282
  */
283
+ /**
284
+ * `<references>` を節として返す。
285
+ *
286
+ * 中身は `<reference>` なので本文は無い。番号と題名だけを持つ節になる。
287
+ * 参照そのものは `get_rfc_dependencies` が返す。
288
+ */
289
+ function extractReferenceSections(references) {
290
+ if (!references)
291
+ return [];
292
+ const list = Array.isArray(references) ? references : [references];
293
+ return list.map((node) => ({
294
+ anchor: node['@_anchor'],
295
+ number: node['@_pn'],
296
+ title: extractProse(node.name) || 'References',
297
+ content: [],
298
+ subsections: extractReferenceSections(node.references),
299
+ }));
300
+ }
301
+ /** 索引の節。中身は語の並びであって文ではない。 */
302
+ const INDEX_SECTION_TITLE = /^index$/i;
271
303
  function extractSections(sections) {
272
304
  if (!sections)
273
305
  return [];
274
306
  const sectionArray = Array.isArray(sections) ? sections : [sections];
275
- return sectionArray.map((sec) => ({
276
- anchor: sec['@_anchor'],
277
- number: sec['@_pn'] || sec['@_numbered'],
278
- title: extractProse(sec.name) || 'Untitled Section',
279
- content: extractContent(sec),
280
- subsections: extractSections(sec.section),
281
- }));
307
+ return sectionArray.map((sec) => {
308
+ const title = extractProse(sec.name) || 'Untitled Section';
309
+ return {
310
+ anchor: sec['@_anchor'],
311
+ number: sec['@_pn'] || sec['@_numbered'],
312
+ title,
313
+ // 索引は語の並びであって文ではない。RFC 9051 の付録 H は
314
+ // `MUST (specification requirement term)` のような項目を並べており、
315
+ // これを本文として読み、`R-H-1` … `R-H-9` の 9 件の要件を立てていた。
316
+ // 要件文は `M MAX (search result option) MAX (search return item name)
317
+ // MAY (specification requirement term) …` である。
318
+ // 節そのものは目次に残す。中身だけを空にする。
319
+ content: INDEX_SECTION_TITLE.test(title) ? [] : extractContent(sec),
320
+ subsections: extractSections(sec.section),
321
+ };
322
+ });
282
323
  }
283
324
  /**
284
325
  * コンテンツブロックの抽出
285
326
  */
286
327
  /** 文の続きとして繋ぐ要素を、いくつ先まで見るか。 */
287
328
  const MAX_CONTINUATION_BLOCKS = 3;
329
+ /** 文の続きとして取り込む箇条書きの項目数の上限。これを超えるものは表とみなす。 */
330
+ const MAX_MERGED_LIST_ITEMS = 20;
288
331
  /** 文の続きとして取り込む表示例の最大の長さ。これを超えるものは独立した図とみなす。 */
289
332
  const INLINE_EXAMPLE_MAX_LENGTH = 120;
290
- /** `pn="section-9.3.5-4"` の末尾の連番。 */
333
+ /**
334
+ * `pn` の末尾の連番。節の中での位置を表す。
335
+ *
336
+ * 直下の要素は `pn="section-9.3.5-4"` で `[4]`。入れ子の要素は
337
+ * `pn="section-4.1-4.2.1"`(節 4.1・4 番目の塊・2 番目の項目・その 1 番目の段落)で
338
+ * `[4, 2, 1]`。末尾の 1 つだけを見ると入れ子の段落が `null` になり、`<dd>` の中の
339
+ * `<t>` を持つ節が丸ごと並べ直しをあきらめていた。
340
+ */
291
341
  function paragraphOrder(pn) {
292
- const match = /-(\d+)$/.exec(pn ?? '');
293
- return match ? Number(match[1]) : null;
342
+ const match = /-(\d+(?:\.\d+)*)$/.exec(pn ?? '');
343
+ return match ? match[1].split('.').map(Number) : null;
344
+ }
345
+ /** `[4, 2, 1]` と `[4, 2, 1, 0.5]` のような並び順の比較。前から順に比べる。 */
346
+ function compareOrder(a, b) {
347
+ const length = Math.min(a.length, b.length);
348
+ for (let i = 0; i < length; i++) {
349
+ if (a[i] !== b[i])
350
+ return a[i] - b[i];
351
+ }
352
+ return a.length - b.length;
294
353
  }
295
354
  /**
296
- * 節の直下の要素を `pn` の連番順に並べる。
355
+ * `<table>` の `pn` は `table-1` で、節の中での位置を持たない。
356
+ *
357
+ * パースの前に、直前の `pn="section-…"` から位置を作って `x-order` 属性に書いておく。
358
+ * 直前の要素が `[4, 2, 1]` なら表は `[4, 2, 1, 0.5]` で、その要素の直後・次の要素の前に
359
+ * 並ぶ。節の直下で表より前に何も無ければ `[0.5]`。
360
+ */
361
+ const TABLE_ORDER_ATTRIBUTE = 'x-order';
362
+ function annotateTableOrder(xml) {
363
+ return xml.replace(/<table\b([^>]*)>/gi, (tag, attrs, offset) => {
364
+ const before = xml.lastIndexOf('pn="section-', offset);
365
+ let order = '0.5';
366
+ if (before !== -1) {
367
+ const pn = /^pn="([^"]*)"/.exec(xml.slice(before))?.[1];
368
+ const parsed = paragraphOrder(pn);
369
+ if (parsed)
370
+ order = `${parsed.join('.')}.0.5`;
371
+ }
372
+ return `<table${attrs} ${TABLE_ORDER_ATTRIBUTE}="${order}">`;
373
+ });
374
+ }
375
+ /**
376
+ * 節の中の要素を `pn` の連番順に並べる。
297
377
  *
298
378
  * `preserveOrder: false` で動かしているため、木からは `<t>` と `<ul>` の並び順が
299
379
  * 失われる。公開版 RFCXML は `pn="section-9.3.5-4"` の形で連番を持つので、
300
380
  * それで並べ直す。連番を持たない要素が 1 つでもあれば並べ直しをあきらめて
301
381
  * `null` を返す(公開前の RFCXML)。
382
+ *
383
+ * 節の直下の `<t>` / `<ul>` / `<ol>` / `<sourcecode>` / `<artwork>` だけを見ていた。
384
+ * `<dl>` の `<dd>`、`<aside>` / `<blockquote>` の中の `<t>`、`<figure>` の中の
385
+ * `<artwork>` / `<sourcecode>`、`<table>` は content block にならず、その中の
386
+ * BCP 14 キーワードがどのツールにも出ていなかった。RFC 9113 §4.1(Frame Format)は
387
+ * フレームヘッダの各フィールドを `<dl>` で書き、`<dd>` の中に `<bcp14>` が 6 個あるが、
388
+ * `get_requirements` は 0 件だった。
302
389
  */
303
390
  function orderedElements(section) {
391
+ const elements = collectElements(section, sectionScope(section));
392
+ if (elements.some((element) => element.order === null))
393
+ return null;
394
+ return elements.sort((a, b) => compareOrder(a.order, b.order));
395
+ }
396
+ function sectionScope(section) {
397
+ return section['@_pn'] || section['@_anchor'] || 'section';
398
+ }
399
+ /**
400
+ * 入れ物(節・`<dd>` / `<aside>` / `<blockquote>` / `<figure>`)の中の要素を集める。
401
+ *
402
+ * 集めるものと、集めないもの。
403
+ *
404
+ * | 要素 | 扱い |
405
+ * |---|---|
406
+ * | `<t>` | 散文(`extractProse`) |
407
+ * | `<ul>` / `<ol>` | 箇条書き。項目は `extractProse(li)` で入れ子ごと 1 つの項目にする |
408
+ * | `<sourcecode>` / `<artwork>` | 空白を畳まない。`type="svg"` は印字されないので出さない |
409
+ * | `<artset>` | svg 以外の `<artwork>` を 1 つ採る |
410
+ * | `<figure>` | 中の `<artwork>` / `<sourcecode>` を親と同じ入れ物として出す |
411
+ * | `<dl>` | `<dd>` の直下のテキストと `<t>` を散文にする。`<dt>` の用語は要件文に混ぜない |
412
+ * | `<aside>` / `<blockquote>` | 直下のテキストと中の要素を、独立した入れ物として出す |
413
+ * | `<table>` | 見出しの行と本文の行 |
414
+ *
415
+ * `<li>` の中は再帰しない。`extractProse(li)` が入れ子の `<t>` / `<ul>` / `<dl>` を
416
+ * 含めて 1 つの項目にしているためで、再帰すると同じ文が 2 回出る。
417
+ */
418
+ function collectElements(container, scope) {
304
419
  const elements = [];
305
420
  const push = (node, element) => {
306
- const order = paragraphOrder(node['@_pn']);
307
- if (order === null)
308
- throw new Error('no pn');
309
- elements.push({ order, node, ...element });
421
+ elements.push({ order: paragraphOrder(node['@_pn']), node, scope, ...element });
310
422
  };
311
- try {
312
- for (const t of toArray(section.t))
313
- push(t, { kind: 'text', text: extractProse(t) });
314
- for (const list of toArray(section.ul)) {
315
- push(list, {
316
- kind: 'list',
317
- text: '',
318
- style: 'symbols',
319
- items: toArray(list.li).map((li) => extractProse(li)),
320
- });
321
- }
322
- for (const list of toArray(section.ol)) {
323
- push(list, {
324
- kind: 'list',
325
- text: '',
326
- style: 'numbers',
327
- items: toArray(list.li).map((li) => extractProse(li)),
328
- });
329
- }
330
- for (const code of toArray(section.sourcecode)) {
331
- push(code, { kind: 'sourcecode', text: extractText(code), language: code['@_type'] });
423
+ for (const t of toArray(container.t))
424
+ push(t, { kind: 'text', text: extractProse(t) });
425
+ for (const list of toArray(container.ul)) {
426
+ push(list, {
427
+ kind: 'list',
428
+ text: '',
429
+ style: 'symbols',
430
+ items: toArray(list.li).map((li) => extractProse(li)),
431
+ });
432
+ }
433
+ for (const list of toArray(container.ol)) {
434
+ push(list, {
435
+ kind: 'list',
436
+ text: '',
437
+ style: 'numbers',
438
+ items: toArray(list.li).map((li) => extractProse(li)),
439
+ });
440
+ }
441
+ const pushArtwork = (art) => {
442
+ if (isSvgArtwork(art))
443
+ return;
444
+ push(art, { kind: 'artwork', text: extractText(art) });
445
+ };
446
+ const pushSourcecode = (code) => {
447
+ push(code, { kind: 'sourcecode', text: extractText(code), language: code['@_type'] });
448
+ };
449
+ const pushArtset = (artset) => {
450
+ const printable = toArray(artset.artwork).find((art) => !isSvgArtwork(art));
451
+ if (printable)
452
+ pushArtwork(printable);
453
+ };
454
+ for (const code of toArray(container.sourcecode))
455
+ pushSourcecode(code);
456
+ for (const art of toArray(container.artwork))
457
+ pushArtwork(art);
458
+ for (const artset of toArray(container.artset))
459
+ pushArtset(artset);
460
+ // `<figure>` は表示上の囲みで、中の図は親の流れの一部である。`<t>… MAY send</t>`
461
+ // のあとの表示例と同じ扱いにするため、入れ物は親のまま。
462
+ for (const figure of toArray(container.figure)) {
463
+ for (const code of toArray(figure.sourcecode))
464
+ pushSourcecode(code);
465
+ for (const art of toArray(figure.artwork))
466
+ pushArtwork(art);
467
+ for (const artset of toArray(figure.artset))
468
+ pushArtset(artset);
469
+ }
470
+ // `<dd>` は `<dd>text</dd>` と `<dd><t>…</t><t>…</t></dd>` の両方の形がある。
471
+ // 直下のテキストは `<dd>` 自身の `pn` の位置、`<t>` はそれぞれの `pn` の位置に置く。
472
+ for (const dl of toArray(container.dl)) {
473
+ for (const dd of toArray(dl.dd)) {
474
+ elements.push(...collectQuotedElements(dd));
332
475
  }
333
- for (const art of toArray(section.artwork)) {
334
- push(art, { kind: 'artwork', text: extractText(art) });
476
+ }
477
+ for (const key of ['aside', 'blockquote']) {
478
+ for (const node of toArray(container[key])) {
479
+ elements.push(...collectQuotedElements(node));
335
480
  }
336
481
  }
337
- catch {
338
- return null;
482
+ for (const table of toArray(container.table)) {
483
+ const order = tableOrder(table);
484
+ elements.push({
485
+ order,
486
+ node: table,
487
+ scope,
488
+ kind: 'table',
489
+ text: '',
490
+ headers: tableHeaders(table),
491
+ rows: tableRows(table),
492
+ });
493
+ }
494
+ return elements;
495
+ }
496
+ /**
497
+ * `<dd>` / `<aside>` / `<blockquote>` の中身。直下のテキストと、入れ子の要素。
498
+ *
499
+ * 自身の `pn` を入れ物の名前にする。`pn` が無ければ(公開前の RFCXML)
500
+ * 並べ直しはどのみち行われないので、入れ物の区別も要らない。
501
+ */
502
+ function collectQuotedElements(node) {
503
+ const scope = node['@_pn'] || 'quoted';
504
+ const elements = [];
505
+ const direct = typeof node === 'string' ? node : node['#text'];
506
+ const text = extractProse(direct);
507
+ if (text) {
508
+ elements.push({ order: paragraphOrder(node['@_pn']), node, scope, kind: 'text', text });
339
509
  }
340
- return elements.sort((a, b) => a.order - b.order);
510
+ if (typeof node === 'object')
511
+ elements.push(...collectElements(node, scope));
512
+ return elements;
513
+ }
514
+ /** `<artwork type="svg">` は印字されない。`<artset>` の中では ascii-art の側が印字される。 */
515
+ function isSvgArtwork(art) {
516
+ return (art['@_type'] ?? '').toLowerCase() === 'svg' || Boolean(art.svg);
517
+ }
518
+ function tableOrder(table) {
519
+ const raw = table[`@_${TABLE_ORDER_ATTRIBUTE}`];
520
+ if (typeof raw !== 'string' || raw.length === 0)
521
+ return null;
522
+ return raw.split('.').map(Number);
523
+ }
524
+ function tableCells(tr) {
525
+ return [...toArray(tr.th), ...toArray(tr.td)].map((cell) => extractProse(cell));
526
+ }
527
+ function tableHeaders(table) {
528
+ const head = toArray(table.thead).flatMap((thead) => toArray(thead.tr));
529
+ return head.length > 0 ? tableCells(head[0]) : [];
530
+ }
531
+ function tableRows(table) {
532
+ const bodies = toArray(table.tbody);
533
+ const rows = bodies.length > 0 ? bodies.flatMap((tbody) => toArray(tbody.tr)) : [];
534
+ return [...rows, ...toArray(table.tr)].map(tableCells);
341
535
  }
342
536
  /**
343
537
  * 文の途中で終わる段落に、直後の要素を取り込む。
@@ -391,6 +585,10 @@ function joinListItems(items) {
391
585
  }, '');
392
586
  return /[.!?]$/.test(joined) ? joined : `${joined}.`;
393
587
  }
588
+ /** 箇条書きの項目が、それ自体で BCP 14 の要件になっているか。 */
589
+ function listCarriesRequirement(element) {
590
+ return (element.items ?? []).some((item) => createRequirementRegex().test(item));
591
+ }
394
592
  function mergeContinuations(elements) {
395
593
  const merged = [];
396
594
  const consumed = new Set();
@@ -405,15 +603,42 @@ function mergeContinuations(elements) {
405
603
  let text = element.text;
406
604
  const hasKeyword = createRequirementRegex().test(text);
407
605
  for (let step = 1; hasKeyword && step <= MAX_CONTINUATION_BLOCKS; step++) {
408
- if (/[.!?:;]$/.test(text))
409
- break;
410
606
  const next = elements[i + step];
411
607
  if (!next || consumed.has(i + step))
412
608
  break;
609
+ // 入れ物をまたいで繋がない。`<dd>` の段落の続きは次の `<dd>` にはないし、
610
+ // `<aside>` の注記は本文の続きではない。
611
+ if (next.scope !== element.scope)
612
+ break;
613
+ // 表は文の続きではない。行ごとに切り出す。
614
+ if (next.kind === 'table')
615
+ break;
616
+ // コロンで終わる文は、続く箇条書きで完結することがある。
617
+ //
618
+ // <t>… an origin server <bcp14>MUST</bcp14> send either:</t>
619
+ // <ul><li>an immediate response with a final status code, …</li>
620
+ // <li>an immediate 100 (Continue) response …</li></ul>
621
+ //
622
+ // `<t>` だけを要件文にすると `MUST send either:` で終わり、**何を選ぶのかが
623
+ // 書かれていない**(RFC 9110 §10.1.1)。
624
+ //
625
+ // ただし項目自身がキーワードを持つなら取り込まない。その項目は独立した
626
+ // 要件であり、取り込むと項目の側の要件文が失われる。
627
+ const completesWithList = /:$/.test(text) &&
628
+ (next.kind === 'list' ? !listCarriesRequirement(next) : next.kind !== 'text');
629
+ if (/[.!?:;]$/.test(text) && !completesWithList)
630
+ break;
413
631
  if (next.kind === 'list') {
414
632
  const items = (next.items ?? []).filter((item) => item.length > 0);
415
633
  if (items.length === 0)
416
634
  break;
635
+ // 項目が多い箇条書きは、文の続きではなく表である。RFC 9113 の
636
+ // Appendix A は "… as a connection error of type INADEQUATE_SECURITY:"
637
+ // のあとに禁止する暗号スイートを約 300 件並べる。繋ぐと 1 件の要件が
638
+ // 9,992 文字になり、`generate_checklist` の項目として読めない。
639
+ // 繋がなければ要件文はコロンで終わり、その節を見よという形になる。
640
+ if (items.length > MAX_MERGED_LIST_ITEMS)
641
+ break;
417
642
  text = `${text} ${joinListItems(items)}`;
418
643
  consumed.add(i + step);
419
644
  break;
@@ -437,8 +662,18 @@ function extractContent(section) {
437
662
  const ordered = orderedElements(section);
438
663
  if (!ordered)
439
664
  return extractContentUnordered(section);
665
+ return toContentBlocks(mergeContinuations(ordered));
666
+ }
667
+ /**
668
+ * `pn` を持たない RFCXML 用。並べ直さず、集めた順(入れ物ごとに種類の順)に出す。
669
+ * 文の続きを繋ぐことはしない。並び順が判らないためである。
670
+ */
671
+ function extractContentUnordered(section) {
672
+ return toContentBlocks(collectElements(section, sectionScope(section)));
673
+ }
674
+ function toContentBlocks(elements) {
440
675
  const blocks = [];
441
- for (const element of mergeContinuations(ordered)) {
676
+ for (const element of elements) {
442
677
  if (element.kind === 'text') {
443
678
  if (element.text)
444
679
  blocks.push(createTextBlock(element.text, element.node));
@@ -456,43 +691,15 @@ function extractContent(section) {
456
691
  else if (element.kind === 'sourcecode') {
457
692
  blocks.push({ type: 'sourcecode', language: element.language, content: element.text });
458
693
  }
694
+ else if (element.kind === 'table') {
695
+ blocks.push({ type: 'table', headers: element.headers ?? [], rows: element.rows ?? [] });
696
+ }
459
697
  else {
460
698
  blocks.push({ type: 'artwork', content: element.text });
461
699
  }
462
700
  }
463
701
  return blocks;
464
702
  }
465
- /** `pn` を持たない RFCXML 用。並べ直さず、種類ごとに出す。 */
466
- function extractContentUnordered(section) {
467
- const blocks = [];
468
- for (const t of toArray(section.t)) {
469
- const text = extractProse(t);
470
- if (text)
471
- blocks.push(createTextBlock(text, t));
472
- }
473
- for (const [lists, style] of [
474
- [toArray(section.ul), 'symbols'],
475
- [toArray(section.ol), 'numbers'],
476
- ]) {
477
- for (const list of lists) {
478
- blocks.push({
479
- type: 'list',
480
- style,
481
- items: toArray(list.li).map((li) => {
482
- const content = extractProse(li);
483
- return { content, requirements: extractRequirementMarkers(content) };
484
- }),
485
- });
486
- }
487
- }
488
- for (const code of toArray(section.sourcecode)) {
489
- blocks.push({ type: 'sourcecode', language: code['@_type'], content: extractText(code) });
490
- }
491
- for (const art of toArray(section.artwork)) {
492
- blocks.push({ type: 'artwork', content: extractText(art) });
493
- }
494
- return blocks;
495
- }
496
703
  /**
497
704
  * テキストブロックを作成
498
705
  * @param text - 抽出されたテキスト内容
@@ -580,18 +787,11 @@ function extractXrefReferences(node) {
580
787
  }
581
788
  /**
582
789
  * 要件マーカーの抽出(<bcp14> 要素またはテキストから)
790
+ *
791
+ * 実体は `utils/text.ts` にある。テキスト経路と同じものを使う。
583
792
  */
584
793
  function extractRequirementMarkers(text) {
585
- const markers = [];
586
- const regex = createRequirementRegex();
587
- let match;
588
- while ((match = regex.exec(text)) !== null) {
589
- markers.push({
590
- level: match[1],
591
- position: match.index,
592
- });
593
- }
594
- return markers;
794
+ return extractMarkers(text);
595
795
  }
596
796
  /**
597
797
  * 要件の完全抽出(文脈付き)
@@ -664,7 +864,10 @@ function parseReference(ref, type) {
664
864
  anchor: ref['@_anchor'] || '',
665
865
  type,
666
866
  rfcNumber,
667
- title: extractProse(front.title) || '',
867
+ // 題名の末尾の読点を落とす。テキスト経路と揃える。
868
+ // RFC 9180 の `<title>SEC 1: Elliptic Curve Cryptography,</title>` のように、
869
+ // 引用の書式をそのまま `<title>` に入れている RFC がある。
870
+ title: extractProse(front.title).replace(/[,;]$/, '') || '',
668
871
  target: ref['@_target'],
669
872
  };
670
873
  }
@@ -685,7 +888,7 @@ function extractDefinitions(rfc) {
685
888
  for (let i = 0; i < dts.length; i++) {
686
889
  const term = extractProse(dts[i]);
687
890
  const definition = extractProse(dds[i]);
688
- if (term && isMeaningfulDefinition(definition)) {
891
+ if (isMeaningfulTerm(term) && isMeaningfulDefinition(definition)) {
689
892
  definitions.push({
690
893
  term: normalizeTerm(term),
691
894
  definition,
@@ -795,6 +998,15 @@ const EMPTY_DEFINITIONS = new Set(['n/a', 'na', 'none', 'not applicable', '-', '
795
998
  * RFC 9110 §14.6 の登録票は `Optional parameters: N/A` のように空欄を埋める。
796
999
  * これを定義として返しても何も伝えていない。
797
1000
  */
1001
+ /**
1002
+ * 用語として採れるか。
1003
+ *
1004
+ * 記号だけの `<dt>` がある。RFC 9147 §3 は表示の約束ごとを `<dl>` で書き、
1005
+ * `'+'` `'*'` `'{}'` `'[]'` を項目にしている。これらは用語ではない。
1006
+ */
1007
+ function isMeaningfulTerm(term) {
1008
+ return /[A-Za-z0-9]/.test(term ?? '');
1009
+ }
798
1010
  function isMeaningfulDefinition(definition) {
799
1011
  const trimmed = definition.trim();
800
1012
  if (trimmed.length < 3)
@@ -871,21 +1083,74 @@ function definingParagraph(xml, paragraphs, irefStart, irefEnd, term) {
871
1083
  const first = take(start);
872
1084
  if (!first)
873
1085
  return undefined;
874
- const needle = term.toLowerCase();
875
- if (extractProse(stripTags(first.body)).toLowerCase().includes(needle))
876
- return first;
877
- // 起点に用語が出てこない。同じ節の中を数段落先まで見る。
1086
+ const proseOf = (candidate) => extractProse(stripTags(candidate.body)).toLowerCase();
1087
+ // 索引の項目は分類を括弧で足すことがある(`max-age (cache directive)`)。
1088
+ // 本文はその括弧を書かないので、括弧を外した形でも探す。外さずに探していたため、
1089
+ // RFC 9111 §5.2.1.1 の `max-age (cache directive)` は本文のどの段落にも当たらず、
1090
+ // 起点の段落 **`Argument syntax:`** をそのまま説明として返していた。§5.2 の
1091
+ // キャッシュ指示子 21 件がこの形である。
1092
+ const needles = [term.toLowerCase(), term.toLowerCase().replace(/\s*\([^)]*\)\s*$/, '')];
1093
+ const defines = quotedDefinitionPattern(term);
1094
+ // 同じ節の中の、起点から数段落ぶん。
1095
+ const scope = [first];
878
1096
  for (let i = start + 1; i < Math.min(start + 1 + DEFINITION_LOOKAHEAD, paragraphs.length); i++) {
879
1097
  const candidate = take(i);
880
1098
  if (!candidate)
881
1099
  break;
882
1100
  if (candidate.section !== first.section)
883
1101
  break;
884
- if (extractProse(stripTags(candidate.body)).toLowerCase().includes(needle))
885
- return candidate;
1102
+ scope.push(candidate);
1103
+ }
1104
+ // 1. 起点がその用語を **引用符付きで** 定義しているなら、それを採る。
1105
+ // 2. 無ければ、同じ節の中で引用符付きで定義している段落を探す。
1106
+ const defining = scope.find((candidate) => defines.test(proseOf(candidate)));
1107
+ if (defining)
1108
+ return defining;
1109
+ // 3. 用語が出てくる段落。起点を先に見るので、これまでの動きと変わらない。
1110
+ for (const needle of needles) {
1111
+ const mentioning = scope.find((candidate) => proseOf(candidate).includes(needle));
1112
+ if (mentioning)
1113
+ return mentioning;
886
1114
  }
887
1115
  return first;
888
1116
  }
1117
+ /**
1118
+ * その用語を **引用符付きで** 定義している文の形。
1119
+ *
1120
+ * `<iref>` の直後の段落を採ると、節の導入が定義として返る。RFC 9110 §3.3 は
1121
+ *
1122
+ * ```xml
1123
+ * <iref primary="true" item="client"/><iref primary="true" item="server"/>
1124
+ * <iref primary="true" item="connection"/>
1125
+ * <t>HTTP is a client/server protocol that operates over a reliable
1126
+ * transport- or session-layer "connection".</t>
1127
+ * <t>An HTTP "client" is a program that establishes a connection to a server …
1128
+ * An HTTP "server" is a program that accepts connections …</t>
1129
+ * ```
1130
+ *
1131
+ * と書く。導入の段落は 3 つの用語すべてを含むので「用語を含む段落」の規則で
1132
+ * 拾われ、**`client` と `server` と `connection` の説明が同じ 1 文**になっていた。
1133
+ * 定義は次の段落にある。
1134
+ *
1135
+ * **引用符が付いているものだけ**を定義とみなす。引用符を求めずに
1136
+ * `<用語> is|are|refers to` を探すと、定義ではない文が当たる。実測で
1137
+ * RFC 9110 §9.3.8 の `TRACE method` は `Responses to the TRACE method are not
1138
+ * cacheable.` に、§6.2 の `control data` は `control data is sent as the first
1139
+ * line of a message` に移り、どちらも定義の文から離れた(8 件中 6 件が悪化した)。
1140
+ *
1141
+ * 引用符と動詞の間は 3 語まで許す。RFC 9112 §9.6 は
1142
+ * `The "close" connection option is defined as a signal that …` と書く。
1143
+ *
1144
+ * 実測(RFC 67 本・定義 1,769 件): 説明が変わったもの 2 件
1145
+ * (RFC 9110 §3.3 の `client` と `server`)。
1146
+ */
1147
+ function quotedDefinitionPattern(term) {
1148
+ const escaped = term.toLowerCase().replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
1149
+ const quote = '["\u201c\u201d]';
1150
+ const gap = '(?:\\s+\\S+){0,3}\\s+';
1151
+ return new RegExp(`${quote}${escaped}${quote}${gap}(?:is|are|refers?\\s+to|means|denotes?)\\b` +
1152
+ `|\\b(?:called|known\\s+as|referred\\s+to\\s+as|termed|defined\\s+as)\\s+(?:an?\\s+|the\\s+)?${quote}${escaped}${quote}`, 'i');
1153
+ }
889
1154
  /**
890
1155
  * 段落が属する節の識別子。
891
1156
  *
@@ -893,9 +1158,12 @@ function definingParagraph(xml, paragraphs, irefStart, irefEnd, term) {
893
1158
  * 公開前の RFCXML では、直前の `<section>` の `anchor` / `pn` を使う。
894
1159
  */
895
1160
  function sectionOfParagraph(xml, paragraph) {
1161
+ // 箇条書きの中の `<t>` は `pn="section-7.1-8.1"`(節 7.1・8 番目の塊・1 番目の
1162
+ // 項目)になる。`-\d+$` だけを外すと `.1` が残り、節が "7.1-8.1" になっていた。
1163
+ // 実在しない節を指すので `get_definitions` の `section` が引けなくなる。
896
1164
  const pn = attributeOf(paragraph.tag, 'pn');
897
1165
  if (pn)
898
- return pn.replace(/-\d+$/, '');
1166
+ return pn.replace(/-\d+(?:\.\d+)*$/, '');
899
1167
  const before = xml.lastIndexOf('<section', paragraph.open);
900
1168
  if (before === -1)
901
1169
  return '';