@rr0/cms 0.3.59 → 0.3.61
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/CMSGenerator.d.ts +12 -0
- package/dist/CMSGenerator.js +14 -3
- package/dist/DataContentVisitor.test.js +3 -3
- package/dist/search/SearchIndexStep.d.ts +6 -1
- package/dist/search/SearchIndexStep.js +94 -19
- package/dist/search/SearchIndexStep.test.js +133 -3
- package/dist/search/SearchVisitor.d.ts +4 -1
- package/dist/search/SearchVisitor.js +33 -10
- package/dist/time/datasource/CsvMapper.js +2 -1
- package/package.json +2 -2
package/dist/CMSGenerator.d.ts
CHANGED
|
@@ -24,6 +24,18 @@ export interface CMSGeneratorOptions {
|
|
|
24
24
|
org: DataOptions;
|
|
25
25
|
};
|
|
26
26
|
siteBaseUrl: string;
|
|
27
|
+
/**
|
|
28
|
+
* A file holding the part of `netlify.toml` that is NOT derived from `.htaccess` — what a host
|
|
29
|
+
* needs that Apache cannot say: a redirect carrying its own status or `force`, one to an absolute
|
|
30
|
+
* URL on another domain, a header scoped to anything narrower than the whole site, a build or
|
|
31
|
+
* plugin section.
|
|
32
|
+
*
|
|
33
|
+
* Without it those lines have nowhere to live but the generated file itself, where every full
|
|
34
|
+
* build wipes them — which is exactly how this site's `ufoathome.org` redirects and its CORS
|
|
35
|
+
* headers disappeared twice. See ssg-api's HtAccessReplaceCommand for the mechanism; naming a file
|
|
36
|
+
* that is not there fails the build rather than dropping it in silence.
|
|
37
|
+
*/
|
|
38
|
+
netlifyPreambleFile?: string;
|
|
27
39
|
timeFormat: Intl.DateTimeFormatOptions;
|
|
28
40
|
directoryPages: string[];
|
|
29
41
|
ufoCaseDirectoryFile: string;
|
package/dist/CMSGenerator.js
CHANGED
|
@@ -103,7 +103,15 @@ export class CMSGenerator {
|
|
|
103
103
|
const peopleRenderer = new PeopleHtmlRenderer();
|
|
104
104
|
const { peopleService, peopleSteps } = await this.setupPeople(context, peopleRenderer, this.options.copies);
|
|
105
105
|
const timeTextBuilder = this.timeTextBuilder;
|
|
106
|
-
const searchVisitor = new SearchVisitor({
|
|
106
|
+
const searchVisitor = new SearchVisitor({
|
|
107
|
+
notIndexedUrls: [
|
|
108
|
+
"404.html",
|
|
109
|
+
"Referencement.html",
|
|
110
|
+
"org/ca/company/avro/Avrocar/index.html",
|
|
111
|
+
"org/renseignement.html",
|
|
112
|
+
"time/1/9/7/7/Poher_Matrice/app/dist/index.html"
|
|
113
|
+
], indexWords: false
|
|
114
|
+
}, timeTextBuilder);
|
|
107
115
|
const { sourceRenderer, sourceFactory, sourceReplacerFactory } = this.setupSources(timeTextBuilder, timeFormat);
|
|
108
116
|
const { noteRenderer, noteReplacerFactory } = this.setupNotes();
|
|
109
117
|
const caseRenderer = new CaseSummaryRenderer(noteRenderer, sourceFactory, sourceRenderer, timeElementFactory);
|
|
@@ -127,7 +135,9 @@ export class CMSGenerator {
|
|
|
127
135
|
}
|
|
128
136
|
}();
|
|
129
137
|
const htAccessToNetlifyConfig = {
|
|
130
|
-
replacements: [
|
|
138
|
+
replacements: [
|
|
139
|
+
new HtAccessToNetlifyConfigReplaceCommand(this.options.siteBaseUrl, this.options.netlifyPreambleFile)
|
|
140
|
+
],
|
|
131
141
|
roots: [".htaccess"],
|
|
132
142
|
getOutputPath: (_context) => "netlify.toml"
|
|
133
143
|
};
|
|
@@ -188,7 +198,7 @@ export class CMSGenerator {
|
|
|
188
198
|
}
|
|
189
199
|
const reindex = args.reindex;
|
|
190
200
|
if (reindex === null || reindex === void 0 ? void 0 : reindex.includes("search")) {
|
|
191
|
-
ssg.add(new SearchIndexStep("search/index.json", searchVisitor));
|
|
201
|
+
ssg.add(new SearchIndexStep("search/index.json", searchVisitor, outDir, (context, fileName) => timeService.setContextFromFile(context, fileName)));
|
|
192
202
|
}
|
|
193
203
|
if (reindex === null || reindex === void 0 ? void 0 : reindex.includes("sources")) {
|
|
194
204
|
ssg.add(new SourceIndexStep(this.options.sourceRegistryFileName, sourceFactory));
|
|
@@ -212,6 +222,7 @@ export class CMSGenerator {
|
|
|
212
222
|
catch (e) {
|
|
213
223
|
context.error(err);
|
|
214
224
|
}
|
|
225
|
+
throw err;
|
|
215
226
|
}
|
|
216
227
|
finally {
|
|
217
228
|
console.timeEnd("ssg");
|
|
@@ -18,7 +18,7 @@ describe("DataContentVisitor", () => {
|
|
|
18
18
|
};
|
|
19
19
|
test("insert portrait image when contents has no image", async () => {
|
|
20
20
|
var _a, _b;
|
|
21
|
-
const context = cmsTestUtil.newHtmlContext("people/v/VertongenJeanLuc/index.html",
|
|
21
|
+
const context = cmsTestUtil.newHtmlContext("people/v/VertongenJeanLuc/index.html", "<div class=\"contents\"><p>Biographie</p></div>");
|
|
22
22
|
const visitor = new TestDataContentVisitor();
|
|
23
23
|
await visitor.renderImage(context, portraitEvent);
|
|
24
24
|
const figure = context.file.document.querySelector(".contents > figure");
|
|
@@ -26,9 +26,9 @@ describe("DataContentVisitor", () => {
|
|
|
26
26
|
expect((_b = figure === null || figure === void 0 ? void 0 : figure.querySelector("figcaption")) === null || _b === void 0 ? void 0 : _b.textContent).toBe("Portrait");
|
|
27
27
|
});
|
|
28
28
|
test("does not insert a portrait image already present", async () => {
|
|
29
|
-
const context = cmsTestUtil.newHtmlContext("people/v/VertongenJeanLuc/index.html",
|
|
29
|
+
const context = cmsTestUtil.newHtmlContext("people/v/VertongenJeanLuc/index.html", "<div class=\"contents\"><img src=\"portrait.jpg\"><p>Biographie</p></div>");
|
|
30
30
|
const visitor = new TestDataContentVisitor();
|
|
31
31
|
await visitor.renderImage(context, portraitEvent);
|
|
32
|
-
expect(context.file.document.querySelectorAll(
|
|
32
|
+
expect(context.file.document.querySelectorAll(".contents img[src=\"portrait.jpg\"]").length).toBe(1);
|
|
33
33
|
});
|
|
34
34
|
});
|
|
@@ -1,21 +1,26 @@
|
|
|
1
1
|
import { SsgContext, SsgStep } from "ssg-api";
|
|
2
2
|
import { SearchVisitor } from "./SearchVisitor.js";
|
|
3
|
+
import { HtmlRR0Context } from "../RR0Context.js";
|
|
3
4
|
/**
|
|
4
5
|
* Saves the index file collected by the SearchCommand.
|
|
5
6
|
*/
|
|
6
7
|
export declare class SearchIndexStep implements SsgStep {
|
|
7
8
|
protected fileName: string;
|
|
8
9
|
protected searchCommand: SearchVisitor;
|
|
10
|
+
protected contentRoot?: string;
|
|
11
|
+
protected prepareContext?: (context: HtmlRR0Context, fileName: string) => void;
|
|
9
12
|
protected encoding: BufferEncoding;
|
|
10
13
|
/**
|
|
11
14
|
* @param fileName The index file path
|
|
12
15
|
* @param searchCommand The command that collected the pages info.
|
|
13
16
|
*/
|
|
14
|
-
constructor(fileName: string, searchCommand: SearchVisitor);
|
|
17
|
+
constructor(fileName: string, searchCommand: SearchVisitor, contentRoot?: string, prepareContext?: (context: HtmlRR0Context, fileName: string) => void);
|
|
15
18
|
/**
|
|
16
19
|
* Write the search index file.
|
|
17
20
|
*
|
|
18
21
|
* @param context
|
|
19
22
|
*/
|
|
20
23
|
execute(context: SsgContext): Promise<any>;
|
|
24
|
+
protected decodeTitle(title: string): string;
|
|
25
|
+
protected readSourceTitle(sourceFile: string): string;
|
|
21
26
|
}
|
|
@@ -1,6 +1,8 @@
|
|
|
1
|
+
import { SearchVisitor } from "./SearchVisitor.js";
|
|
1
2
|
import fs from "fs";
|
|
2
3
|
import { writeFile } from "@javarome/fileutil";
|
|
3
4
|
import path from "path";
|
|
5
|
+
import { glob } from "glob";
|
|
4
6
|
/**
|
|
5
7
|
* Saves the index file collected by the SearchCommand.
|
|
6
8
|
*/
|
|
@@ -9,9 +11,11 @@ export class SearchIndexStep {
|
|
|
9
11
|
* @param fileName The index file path
|
|
10
12
|
* @param searchCommand The command that collected the pages info.
|
|
11
13
|
*/
|
|
12
|
-
constructor(fileName, searchCommand) {
|
|
14
|
+
constructor(fileName, searchCommand, contentRoot, prepareContext) {
|
|
13
15
|
this.fileName = fileName;
|
|
14
16
|
this.searchCommand = searchCommand;
|
|
17
|
+
this.contentRoot = contentRoot;
|
|
18
|
+
this.prepareContext = prepareContext;
|
|
15
19
|
this.encoding = "utf8";
|
|
16
20
|
}
|
|
17
21
|
/**
|
|
@@ -19,33 +23,104 @@ export class SearchIndexStep {
|
|
|
19
23
|
*
|
|
20
24
|
* @param context
|
|
21
25
|
*/
|
|
22
|
-
execute(context) {
|
|
26
|
+
async execute(context) {
|
|
27
|
+
var _a;
|
|
23
28
|
const newIndex = this.searchCommand.index;
|
|
24
29
|
let existingIndex;
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
const
|
|
28
|
-
const
|
|
29
|
-
const
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
30
|
+
if (this.contentRoot) {
|
|
31
|
+
const pagesByUrl = new Map();
|
|
32
|
+
const pagesByResultTitle = new Map();
|
|
33
|
+
const duplicateTitles = [];
|
|
34
|
+
const outputFiles = (await glob(path.join(this.contentRoot, "**/*.html"))).sort();
|
|
35
|
+
const sourceRoot = path.dirname(path.dirname(this.fileName));
|
|
36
|
+
for (const outputFile of outputFiles) {
|
|
37
|
+
const url = path.relative(this.contentRoot, outputFile);
|
|
38
|
+
const sourceFile = path.resolve(sourceRoot, url);
|
|
39
|
+
if (!fs.existsSync(sourceFile)) {
|
|
40
|
+
continue;
|
|
41
|
+
}
|
|
42
|
+
const pageContext = context.clone();
|
|
43
|
+
(_a = this.prepareContext) === null || _a === void 0 ? void 0 : _a.call(this, pageContext, outputFile);
|
|
44
|
+
const html = fs.readFileSync(outputFile, this.encoding);
|
|
45
|
+
const titleMatch = /<title[^>]*>([\s\S]*?)<\/title>/i.exec(html);
|
|
46
|
+
const outputTitle = titleMatch ? this.decodeTitle(titleMatch[1]) : "";
|
|
47
|
+
const title = this.readSourceTitle(sourceFile) || outputTitle;
|
|
48
|
+
const pageInfo = this.searchCommand.pageInfoFrom(title, url.split(path.sep).join("/"), pageContext);
|
|
49
|
+
if (pageInfo) {
|
|
50
|
+
const resultTitle = SearchVisitor.resultTitle(pageInfo);
|
|
51
|
+
const titleIndexed = pagesByResultTitle.get(resultTitle);
|
|
52
|
+
if (titleIndexed && titleIndexed.url !== pageInfo.url) {
|
|
53
|
+
duplicateTitles.push(`Search result "${resultTitle}" with URL ${pageInfo.url} is already indexed with URL ${titleIndexed.url}`);
|
|
54
|
+
continue;
|
|
55
|
+
}
|
|
56
|
+
pagesByUrl.set(pageInfo.url, pageInfo);
|
|
57
|
+
pagesByResultTitle.set(resultTitle, pageInfo);
|
|
58
|
+
}
|
|
34
59
|
}
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
}
|
|
38
|
-
catch (e) {
|
|
39
|
-
if (e.errno !== -2) {
|
|
40
|
-
throw e;
|
|
60
|
+
if (duplicateTitles.length > 0) {
|
|
61
|
+
throw new Error(`${duplicateTitles.length} indistinguishable search results:\n${duplicateTitles.join("\n")}`);
|
|
41
62
|
}
|
|
42
|
-
|
|
43
|
-
|
|
63
|
+
const pages = Array.from(pagesByUrl.values());
|
|
64
|
+
pages.sort((pageInfo1, pageInfo2) => pageInfo1.title.localeCompare(pageInfo2.title));
|
|
65
|
+
existingIndex = { pages, words: newIndex.words };
|
|
44
66
|
}
|
|
67
|
+
else
|
|
68
|
+
try {
|
|
69
|
+
existingIndex = JSON.parse(fs.readFileSync(this.fileName, { encoding: this.encoding }));
|
|
70
|
+
const newPages = newIndex.pages;
|
|
71
|
+
const contentRoot = path.dirname(path.dirname(this.fileName));
|
|
72
|
+
const pagesByUrl = new Map(existingIndex.pages
|
|
73
|
+
.filter(page => fs.existsSync(path.resolve(contentRoot, page.url)))
|
|
74
|
+
.map(page => [page.url, page]));
|
|
75
|
+
for (const newPage of newPages) {
|
|
76
|
+
pagesByUrl.set(newPage.url, newPage);
|
|
77
|
+
}
|
|
78
|
+
existingIndex.pages = Array.from(pagesByUrl.values());
|
|
79
|
+
existingIndex.pages.sort((pageInfo1, pageInfo2) => pageInfo1.title > pageInfo2.title ? 1 : pageInfo1.title < pageInfo2.title ? -1 : 0);
|
|
80
|
+
}
|
|
81
|
+
catch (e) {
|
|
82
|
+
if (e.errno !== -2) {
|
|
83
|
+
throw e;
|
|
84
|
+
}
|
|
85
|
+
context.warn("Could not find", this.fileName, "Will create it");
|
|
86
|
+
existingIndex = newIndex;
|
|
87
|
+
}
|
|
45
88
|
const indexSize = existingIndex.pages.length;
|
|
46
89
|
context.setVar("indexSize", indexSize);
|
|
47
90
|
context.log("Saving search index of", indexSize, "pages at", this.fileName);
|
|
48
91
|
const indexJson = JSON.stringify(existingIndex);
|
|
49
92
|
return writeFile(this.fileName, indexJson, "utf-8");
|
|
50
93
|
}
|
|
94
|
+
decodeTitle(title) {
|
|
95
|
+
return title
|
|
96
|
+
.replace(/<[^>]+>/g, "")
|
|
97
|
+
.replace(/&#(\d+);/g, (_match, code) => String.fromCodePoint(Number(code)))
|
|
98
|
+
.replace(/&#x([\da-f]+);/gi, (_match, code) => String.fromCodePoint(Number.parseInt(code, 16)))
|
|
99
|
+
.replace(/ /gi, " ")
|
|
100
|
+
.replace(/&/gi, "&")
|
|
101
|
+
.replace(/"/gi, "\"")
|
|
102
|
+
.replace(/'|'/gi, "'")
|
|
103
|
+
.replace(/</gi, "<")
|
|
104
|
+
.replace(/>/gi, ">")
|
|
105
|
+
.replace(/\s+/g, " ")
|
|
106
|
+
.trim();
|
|
107
|
+
}
|
|
108
|
+
readSourceTitle(sourceFile) {
|
|
109
|
+
const source = fs.readFileSync(sourceFile, this.encoding);
|
|
110
|
+
const htmlTitleMatch = /<title[^>]*>([\s\S]*?)<\/title>/i.exec(source);
|
|
111
|
+
if (htmlTitleMatch) {
|
|
112
|
+
return this.decodeTitle(htmlTitleMatch[1]);
|
|
113
|
+
}
|
|
114
|
+
for (const directiveMatch of source.matchAll(/<!--#set\s+([\s\S]*?)-->/gi)) {
|
|
115
|
+
const attributes = directiveMatch[1];
|
|
116
|
+
if (!/\bvar\s*=\s*(["'])title\1/i.test(attributes)) {
|
|
117
|
+
continue;
|
|
118
|
+
}
|
|
119
|
+
const valueMatch = /\bvalue\s*=\s*(["'])([\s\S]*?)\1/i.exec(attributes);
|
|
120
|
+
if (valueMatch) {
|
|
121
|
+
return this.decodeTitle(valueMatch[2]);
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
return "";
|
|
125
|
+
}
|
|
51
126
|
}
|
|
@@ -2,6 +2,7 @@ import { describe, expect, test } from "@javarome/testscript";
|
|
|
2
2
|
import fs from "fs";
|
|
3
3
|
import os from "os";
|
|
4
4
|
import path from "path";
|
|
5
|
+
import { SsgContextImpl } from "ssg-api";
|
|
5
6
|
import { SearchIndexStep } from "./SearchIndexStep.js";
|
|
6
7
|
describe("SearchIndexStep", () => {
|
|
7
8
|
test("removes missing and duplicate pages while updating the current pages", async () => {
|
|
@@ -29,9 +30,12 @@ describe("SearchIndexStep", () => {
|
|
|
29
30
|
};
|
|
30
31
|
const visitor = { index: newIndex };
|
|
31
32
|
const context = {
|
|
32
|
-
setVar() {
|
|
33
|
-
|
|
34
|
-
|
|
33
|
+
setVar() {
|
|
34
|
+
},
|
|
35
|
+
log() {
|
|
36
|
+
},
|
|
37
|
+
warn() {
|
|
38
|
+
}
|
|
35
39
|
};
|
|
36
40
|
await new SearchIndexStep(fileName, visitor).execute(context);
|
|
37
41
|
const savedIndex = JSON.parse(fs.readFileSync(fileName, "utf8"));
|
|
@@ -44,4 +48,130 @@ describe("SearchIndexStep", () => {
|
|
|
44
48
|
fs.rmSync(root, { recursive: true, force: true });
|
|
45
49
|
}
|
|
46
50
|
});
|
|
51
|
+
test("rebuilds the index from every HTML page in the output directory", async () => {
|
|
52
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), "rr0-search-index-"));
|
|
53
|
+
try {
|
|
54
|
+
const outDir = path.join(root, "out");
|
|
55
|
+
const searchDir = path.join(root, "search");
|
|
56
|
+
fs.mkdirSync(path.join(outDir, "science"), { recursive: true });
|
|
57
|
+
fs.mkdirSync(searchDir);
|
|
58
|
+
fs.mkdirSync(path.join(root, "science"));
|
|
59
|
+
fs.writeFileSync(path.join(outDir, "science", "UfoAtHome.html"), "<html><head><title>UFO@home</title></head><body></body></html>");
|
|
60
|
+
fs.writeFileSync(path.join(outDir, "other.html"), "<html><head><title>Other page</title></head><body></body></html>");
|
|
61
|
+
fs.writeFileSync(path.join(root, "science", "UfoAtHome.html"), "");
|
|
62
|
+
fs.writeFileSync(path.join(root, "other.html"), "");
|
|
63
|
+
fs.writeFileSync(path.join(outDir, "stale.html"), "<html><head><title>Stale page</title></head><body></body></html>");
|
|
64
|
+
const fileName = path.join(searchDir, "index.json");
|
|
65
|
+
const newIndex = { pages: [], words: {} };
|
|
66
|
+
const visitor = {
|
|
67
|
+
index: newIndex,
|
|
68
|
+
pageInfoFrom(title, url) {
|
|
69
|
+
return {
|
|
70
|
+
title,
|
|
71
|
+
url,
|
|
72
|
+
time: ""
|
|
73
|
+
};
|
|
74
|
+
}
|
|
75
|
+
};
|
|
76
|
+
const context = new SsgContextImpl("fr");
|
|
77
|
+
await new SearchIndexStep(fileName, visitor, outDir).execute(context);
|
|
78
|
+
const savedIndex = JSON.parse(fs.readFileSync(fileName, "utf8"));
|
|
79
|
+
expect(savedIndex.pages).toEqual([
|
|
80
|
+
{ title: "Other page", url: "other.html", time: "" },
|
|
81
|
+
{ title: "UFO@home", url: "science/UfoAtHome.html", time: "" }
|
|
82
|
+
]);
|
|
83
|
+
}
|
|
84
|
+
finally {
|
|
85
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
86
|
+
}
|
|
87
|
+
});
|
|
88
|
+
test("rejects the same title at two different URLs", async () => {
|
|
89
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), "rr0-search-index-"));
|
|
90
|
+
try {
|
|
91
|
+
const outDir = path.join(root, "out");
|
|
92
|
+
const searchDir = path.join(root, "search");
|
|
93
|
+
fs.mkdirSync(outDir);
|
|
94
|
+
fs.mkdirSync(searchDir);
|
|
95
|
+
fs.writeFileSync(path.join(outDir, "first.html"), "<html><head><title>Same title</title></head><body></body></html>");
|
|
96
|
+
fs.writeFileSync(path.join(outDir, "second.html"), "<html><head><title>Same title</title></head><body></body></html>");
|
|
97
|
+
fs.writeFileSync(path.join(root, "first.html"), "");
|
|
98
|
+
fs.writeFileSync(path.join(root, "second.html"), "");
|
|
99
|
+
const visitor = {
|
|
100
|
+
index: { pages: [], words: {} },
|
|
101
|
+
pageInfoFrom(title, url) {
|
|
102
|
+
return {
|
|
103
|
+
title,
|
|
104
|
+
url,
|
|
105
|
+
time: ""
|
|
106
|
+
};
|
|
107
|
+
}
|
|
108
|
+
};
|
|
109
|
+
const context = new SsgContextImpl("fr");
|
|
110
|
+
let error;
|
|
111
|
+
try {
|
|
112
|
+
await new SearchIndexStep(path.join(searchDir, "index.json"), visitor, outDir).execute(context);
|
|
113
|
+
}
|
|
114
|
+
catch (e) {
|
|
115
|
+
error = e;
|
|
116
|
+
}
|
|
117
|
+
expect(error === null || error === void 0 ? void 0 : error.message.includes("Search result \"Same title\"")).toEqual(true);
|
|
118
|
+
}
|
|
119
|
+
finally {
|
|
120
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
121
|
+
}
|
|
122
|
+
});
|
|
123
|
+
test("allows identical document titles when their search result dates differ", async () => {
|
|
124
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), "rr0-search-index-"));
|
|
125
|
+
try {
|
|
126
|
+
const outDir = path.join(root, "out");
|
|
127
|
+
const searchDir = path.join(root, "search");
|
|
128
|
+
fs.mkdirSync(outDir);
|
|
129
|
+
fs.mkdirSync(searchDir);
|
|
130
|
+
for (const fileName of ["first.html", "second.html"]) {
|
|
131
|
+
fs.writeFileSync(path.join(outDir, fileName), "<html><head><title>Same title</title></head><body></body></html>");
|
|
132
|
+
fs.writeFileSync(path.join(root, fileName), "");
|
|
133
|
+
}
|
|
134
|
+
const visitor = {
|
|
135
|
+
index: { pages: [], words: {} },
|
|
136
|
+
pageInfoFrom(title, url) {
|
|
137
|
+
return { title, url, time: url === "first.html" ? "1900" : "1901" };
|
|
138
|
+
}
|
|
139
|
+
};
|
|
140
|
+
const context = new SsgContextImpl("fr");
|
|
141
|
+
const fileName = path.join(searchDir, "index.json");
|
|
142
|
+
await new SearchIndexStep(fileName, visitor, outDir).execute(context);
|
|
143
|
+
const savedIndex = JSON.parse(fs.readFileSync(fileName, "utf8"));
|
|
144
|
+
expect(savedIndex.pages.length).toEqual(2);
|
|
145
|
+
}
|
|
146
|
+
finally {
|
|
147
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
148
|
+
}
|
|
149
|
+
});
|
|
150
|
+
test("prefers an explicit source title over a generated chronological title", async () => {
|
|
151
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), "rr0-search-index-"));
|
|
152
|
+
try {
|
|
153
|
+
const outDir = path.join(root, "out");
|
|
154
|
+
const searchDir = path.join(root, "search");
|
|
155
|
+
fs.mkdirSync(outDir);
|
|
156
|
+
fs.mkdirSync(searchDir);
|
|
157
|
+
fs.writeFileSync(path.join(outDir, "article.html"), "<html><head><title>Mai 1954</title></head><body></body></html>");
|
|
158
|
+
fs.writeFileSync(path.join(root, "article.html"), "<!--#set var=\"title\" value=\"Canada Hunts for Saucers\" --><!--#include virtual=\"/header.html\" -->");
|
|
159
|
+
const visitor = {
|
|
160
|
+
index: { pages: [], words: {} },
|
|
161
|
+
pageInfoFrom(title, url) {
|
|
162
|
+
return { title, url, time: "mai 1954" };
|
|
163
|
+
}
|
|
164
|
+
};
|
|
165
|
+
const context = new SsgContextImpl("fr");
|
|
166
|
+
const fileName = path.join(searchDir, "index.json");
|
|
167
|
+
await new SearchIndexStep(fileName, visitor, outDir).execute(context);
|
|
168
|
+
const savedIndex = JSON.parse(fs.readFileSync(fileName, "utf8"));
|
|
169
|
+
expect(savedIndex.pages).toEqual([
|
|
170
|
+
{ title: "Canada Hunts for Saucers", url: "article.html", time: "mai 1954" }
|
|
171
|
+
]);
|
|
172
|
+
}
|
|
173
|
+
finally {
|
|
174
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
175
|
+
}
|
|
176
|
+
});
|
|
47
177
|
});
|
|
@@ -30,12 +30,15 @@ export type SearchCommandConfig = {
|
|
|
30
30
|
export declare class SearchVisitor implements FileVisitor {
|
|
31
31
|
protected config: SearchCommandConfig;
|
|
32
32
|
protected timeTextBuilder: TimeTextBuilder;
|
|
33
|
+
static resultTitle(pageInfo: PageInfo): string;
|
|
33
34
|
readonly index: SearchIndex;
|
|
34
35
|
protected readonly contentStream: fs.WriteStream | undefined;
|
|
35
36
|
constructor(config: SearchCommandConfig, timeTextBuilder: TimeTextBuilder);
|
|
36
37
|
contentStepEnd(): Promise<void>;
|
|
37
38
|
visit(context: HtmlRR0Context): Promise<void>;
|
|
38
|
-
|
|
39
|
+
pageInfo(context: HtmlRR0Context, contentRoot?: string): PageInfo | undefined;
|
|
40
|
+
pageInfoFrom(title: string, url: string, context: HtmlRR0Context): PageInfo | undefined;
|
|
41
|
+
protected handleAlreadyIndexed(resultTitle: string, url: string, titleIndexed: PageInfo): void;
|
|
39
42
|
protected getContents(doc: Document): string;
|
|
40
43
|
protected indexContent(context: HtmlRR0Context, outputFile: HtmlFileContents): void;
|
|
41
44
|
protected indexWords(context: HtmlRR0Context, outputFile: HtmlFileContents): void;
|
|
@@ -1,8 +1,13 @@
|
|
|
1
1
|
import fs from "fs";
|
|
2
|
+
import path from "path";
|
|
2
3
|
/**
|
|
3
4
|
* Builds an index of pages.
|
|
4
5
|
*/
|
|
5
6
|
export class SearchVisitor {
|
|
7
|
+
static resultTitle(pageInfo) {
|
|
8
|
+
const { title, time } = pageInfo;
|
|
9
|
+
return title + (time && time !== title.toLowerCase() ? ` (${time})` : "");
|
|
10
|
+
}
|
|
6
11
|
constructor(config, timeTextBuilder) {
|
|
7
12
|
this.config = config;
|
|
8
13
|
this.timeTextBuilder = timeTextBuilder;
|
|
@@ -24,17 +29,21 @@ export class SearchVisitor {
|
|
|
24
29
|
}
|
|
25
30
|
async visit(context) {
|
|
26
31
|
const file = context.file;
|
|
27
|
-
const
|
|
28
|
-
|
|
29
|
-
const url = file.name.startsWith(outDir) ? file.name.substring(outDir.length) : file.name;
|
|
30
|
-
if (title && !this.config.notIndexedUrls.includes(url)) {
|
|
32
|
+
const pageInfo = this.pageInfo(context);
|
|
33
|
+
if (pageInfo) {
|
|
31
34
|
const indexedPages = this.index.pages;
|
|
32
|
-
const
|
|
35
|
+
const pageIndex = indexedPages.findIndex(page => page.url === pageInfo.url);
|
|
36
|
+
const resultTitle = SearchVisitor.resultTitle(pageInfo);
|
|
37
|
+
const titleIndexed = indexedPages.find(page => SearchVisitor.resultTitle(page) === resultTitle && page.url !== pageInfo.url);
|
|
33
38
|
if (titleIndexed) {
|
|
34
|
-
this.handleAlreadyIndexed(
|
|
39
|
+
this.handleAlreadyIndexed(resultTitle, pageInfo.url, titleIndexed);
|
|
40
|
+
}
|
|
41
|
+
if (pageIndex >= 0) {
|
|
42
|
+
indexedPages[pageIndex] = pageInfo;
|
|
43
|
+
}
|
|
44
|
+
else {
|
|
45
|
+
indexedPages.push(pageInfo);
|
|
35
46
|
}
|
|
36
|
-
const time = this.timeTextBuilder.build(context, { year: "numeric", month: "short", day: "numeric" }).toLowerCase();
|
|
37
|
-
indexedPages.push({ title, url, time });
|
|
38
47
|
}
|
|
39
48
|
if (this.config.indexWords) {
|
|
40
49
|
this.indexWords(context, file);
|
|
@@ -43,8 +52,22 @@ export class SearchVisitor {
|
|
|
43
52
|
this.indexContent(context, file);
|
|
44
53
|
}
|
|
45
54
|
}
|
|
46
|
-
|
|
47
|
-
|
|
55
|
+
pageInfo(context, contentRoot) {
|
|
56
|
+
const file = context.file;
|
|
57
|
+
const url = contentRoot
|
|
58
|
+
? path.relative(contentRoot, file.name).split(path.sep).join("/")
|
|
59
|
+
: file.name.startsWith("out/") ? file.name.substring("out/".length) : file.name;
|
|
60
|
+
return this.pageInfoFrom(file.title, url, context);
|
|
61
|
+
}
|
|
62
|
+
pageInfoFrom(title, url, context) {
|
|
63
|
+
if (!title || this.config.notIndexedUrls.includes(url)) {
|
|
64
|
+
return undefined;
|
|
65
|
+
}
|
|
66
|
+
const time = this.timeTextBuilder.build(context, { year: "numeric", month: "short", day: "numeric" }).toLowerCase();
|
|
67
|
+
return { title, url, time };
|
|
68
|
+
}
|
|
69
|
+
handleAlreadyIndexed(resultTitle, url, titleIndexed) {
|
|
70
|
+
throw new Error(`Search result "${resultTitle}" with URL ${url} is already indexed with URL ${titleIndexed.url}`);
|
|
48
71
|
}
|
|
49
72
|
getContents(doc) {
|
|
50
73
|
const div = doc.createElement("div");
|
|
@@ -64,7 +64,8 @@ export class CsvMapper {
|
|
|
64
64
|
*/
|
|
65
65
|
mapAll(context, sourceCases, sourceTime) {
|
|
66
66
|
const values = sourceCases.map(c => this.map(context, c, sourceTime));
|
|
67
|
-
|
|
67
|
+
const header = Array.from(this.fields).sort((field1, field2) => field1.localeCompare(field2));
|
|
68
|
+
return header.join(this.sep) + "\n" + values.join("\n");
|
|
68
69
|
}
|
|
69
70
|
escape(value, force) {
|
|
70
71
|
if (this.escapeStr && (force || value.indexOf(this.sep) >= 0)) {
|
package/package.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"name": "@rr0/cms",
|
|
3
3
|
"type": "module",
|
|
4
4
|
"author": "Jérôme Beau <rr0@rr0.org> (https://rr0.org)",
|
|
5
|
-
"version": "0.3.
|
|
5
|
+
"version": "0.3.61",
|
|
6
6
|
"description": "RR0 Content Management System (CMS)",
|
|
7
7
|
"exports": "./dist/index.js",
|
|
8
8
|
"types": "./dist/index.d.ts",
|
|
@@ -41,7 +41,7 @@
|
|
|
41
41
|
"image-size": "^2.0.2",
|
|
42
42
|
"jsdom": "^27.4.0",
|
|
43
43
|
"selenium-webdriver": "^4.39.0",
|
|
44
|
-
"ssg-api": "^1.
|
|
44
|
+
"ssg-api": "^1.18.0"
|
|
45
45
|
},
|
|
46
46
|
"devDependencies": {
|
|
47
47
|
"@javarome/testscript": "^0.13.1",
|