@rr0/cms 0.3.58 → 0.3.60
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/CMSGenerator.js +11 -2
- package/dist/DataContentVisitor.test.js +3 -3
- package/dist/search/SearchIndexStep.d.ts +6 -1
- package/dist/search/SearchIndexStep.js +93 -19
- package/dist/search/SearchIndexStep.test.d.ts +1 -0
- package/dist/search/SearchIndexStep.test.js +177 -0
- package/dist/search/SearchVisitor.d.ts +4 -1
- package/dist/search/SearchVisitor.js +33 -10
- package/dist/time/datasource/CsvMapper.js +2 -1
- package/package.json +1 -1
package/dist/CMSGenerator.js
CHANGED
|
@@ -103,7 +103,15 @@ export class CMSGenerator {
|
|
|
103
103
|
const peopleRenderer = new PeopleHtmlRenderer();
|
|
104
104
|
const { peopleService, peopleSteps } = await this.setupPeople(context, peopleRenderer, this.options.copies);
|
|
105
105
|
const timeTextBuilder = this.timeTextBuilder;
|
|
106
|
-
const searchVisitor = new SearchVisitor({
|
|
106
|
+
const searchVisitor = new SearchVisitor({
|
|
107
|
+
notIndexedUrls: [
|
|
108
|
+
"404.html",
|
|
109
|
+
"Referencement.html",
|
|
110
|
+
"org/ca/company/avro/Avrocar/index.html",
|
|
111
|
+
"org/renseignement.html",
|
|
112
|
+
"time/1/9/7/7/Poher_Matrice/app/dist/index.html"
|
|
113
|
+
], indexWords: false
|
|
114
|
+
}, timeTextBuilder);
|
|
107
115
|
const { sourceRenderer, sourceFactory, sourceReplacerFactory } = this.setupSources(timeTextBuilder, timeFormat);
|
|
108
116
|
const { noteRenderer, noteReplacerFactory } = this.setupNotes();
|
|
109
117
|
const caseRenderer = new CaseSummaryRenderer(noteRenderer, sourceFactory, sourceRenderer, timeElementFactory);
|
|
@@ -188,7 +196,7 @@ export class CMSGenerator {
|
|
|
188
196
|
}
|
|
189
197
|
const reindex = args.reindex;
|
|
190
198
|
if (reindex === null || reindex === void 0 ? void 0 : reindex.includes("search")) {
|
|
191
|
-
ssg.add(new SearchIndexStep("search/index.json", searchVisitor));
|
|
199
|
+
ssg.add(new SearchIndexStep("search/index.json", searchVisitor, outDir, (context, fileName) => timeService.setContextFromFile(context, fileName)));
|
|
192
200
|
}
|
|
193
201
|
if (reindex === null || reindex === void 0 ? void 0 : reindex.includes("sources")) {
|
|
194
202
|
ssg.add(new SourceIndexStep(this.options.sourceRegistryFileName, sourceFactory));
|
|
@@ -212,6 +220,7 @@ export class CMSGenerator {
|
|
|
212
220
|
catch (e) {
|
|
213
221
|
context.error(err);
|
|
214
222
|
}
|
|
223
|
+
throw err;
|
|
215
224
|
}
|
|
216
225
|
finally {
|
|
217
226
|
console.timeEnd("ssg");
|
|
@@ -18,7 +18,7 @@ describe("DataContentVisitor", () => {
|
|
|
18
18
|
};
|
|
19
19
|
test("insert portrait image when contents has no image", async () => {
|
|
20
20
|
var _a, _b;
|
|
21
|
-
const context = cmsTestUtil.newHtmlContext("people/v/VertongenJeanLuc/index.html",
|
|
21
|
+
const context = cmsTestUtil.newHtmlContext("people/v/VertongenJeanLuc/index.html", "<div class=\"contents\"><p>Biographie</p></div>");
|
|
22
22
|
const visitor = new TestDataContentVisitor();
|
|
23
23
|
await visitor.renderImage(context, portraitEvent);
|
|
24
24
|
const figure = context.file.document.querySelector(".contents > figure");
|
|
@@ -26,9 +26,9 @@ describe("DataContentVisitor", () => {
|
|
|
26
26
|
expect((_b = figure === null || figure === void 0 ? void 0 : figure.querySelector("figcaption")) === null || _b === void 0 ? void 0 : _b.textContent).toBe("Portrait");
|
|
27
27
|
});
|
|
28
28
|
test("does not insert a portrait image already present", async () => {
|
|
29
|
-
const context = cmsTestUtil.newHtmlContext("people/v/VertongenJeanLuc/index.html",
|
|
29
|
+
const context = cmsTestUtil.newHtmlContext("people/v/VertongenJeanLuc/index.html", "<div class=\"contents\"><img src=\"portrait.jpg\"><p>Biographie</p></div>");
|
|
30
30
|
const visitor = new TestDataContentVisitor();
|
|
31
31
|
await visitor.renderImage(context, portraitEvent);
|
|
32
|
-
expect(context.file.document.querySelectorAll(
|
|
32
|
+
expect(context.file.document.querySelectorAll(".contents img[src=\"portrait.jpg\"]").length).toBe(1);
|
|
33
33
|
});
|
|
34
34
|
});
|
|
@@ -1,21 +1,26 @@
|
|
|
1
1
|
import { SsgContext, SsgStep } from "ssg-api";
|
|
2
2
|
import { SearchVisitor } from "./SearchVisitor.js";
|
|
3
|
+
import { HtmlRR0Context } from "../RR0Context.js";
|
|
3
4
|
/**
|
|
4
5
|
* Saves the index file collected by the SearchCommand.
|
|
5
6
|
*/
|
|
6
7
|
export declare class SearchIndexStep implements SsgStep {
|
|
7
8
|
protected fileName: string;
|
|
8
9
|
protected searchCommand: SearchVisitor;
|
|
10
|
+
protected contentRoot?: string;
|
|
11
|
+
protected prepareContext?: (context: HtmlRR0Context, fileName: string) => void;
|
|
9
12
|
protected encoding: BufferEncoding;
|
|
10
13
|
/**
|
|
11
14
|
* @param fileName The index file path
|
|
12
15
|
* @param searchCommand The command that collected the pages info.
|
|
13
16
|
*/
|
|
14
|
-
constructor(fileName: string, searchCommand: SearchVisitor);
|
|
17
|
+
constructor(fileName: string, searchCommand: SearchVisitor, contentRoot?: string, prepareContext?: (context: HtmlRR0Context, fileName: string) => void);
|
|
15
18
|
/**
|
|
16
19
|
* Write the search index file.
|
|
17
20
|
*
|
|
18
21
|
* @param context
|
|
19
22
|
*/
|
|
20
23
|
execute(context: SsgContext): Promise<any>;
|
|
24
|
+
protected decodeTitle(title: string): string;
|
|
25
|
+
protected readSourceTitle(sourceFile: string): string;
|
|
21
26
|
}
|
|
@@ -1,5 +1,8 @@
|
|
|
1
|
+
import { SearchVisitor } from "./SearchVisitor.js";
|
|
1
2
|
import fs from "fs";
|
|
2
3
|
import { writeFile } from "@javarome/fileutil";
|
|
4
|
+
import path from "path";
|
|
5
|
+
import { glob } from "glob";
|
|
3
6
|
/**
|
|
4
7
|
* Saves the index file collected by the SearchCommand.
|
|
5
8
|
*/
|
|
@@ -8,9 +11,11 @@ export class SearchIndexStep {
|
|
|
8
11
|
* @param fileName The index file path
|
|
9
12
|
* @param searchCommand The command that collected the pages info.
|
|
10
13
|
*/
|
|
11
|
-
constructor(fileName, searchCommand) {
|
|
14
|
+
constructor(fileName, searchCommand, contentRoot, prepareContext) {
|
|
12
15
|
this.fileName = fileName;
|
|
13
16
|
this.searchCommand = searchCommand;
|
|
17
|
+
this.contentRoot = contentRoot;
|
|
18
|
+
this.prepareContext = prepareContext;
|
|
14
19
|
this.encoding = "utf8";
|
|
15
20
|
}
|
|
16
21
|
/**
|
|
@@ -18,35 +23,104 @@ export class SearchIndexStep {
|
|
|
18
23
|
*
|
|
19
24
|
* @param context
|
|
20
25
|
*/
|
|
21
|
-
execute(context) {
|
|
26
|
+
async execute(context) {
|
|
27
|
+
var _a;
|
|
22
28
|
const newIndex = this.searchCommand.index;
|
|
23
29
|
let existingIndex;
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
const
|
|
27
|
-
const
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
30
|
+
if (this.contentRoot) {
|
|
31
|
+
const pagesByUrl = new Map();
|
|
32
|
+
const pagesByResultTitle = new Map();
|
|
33
|
+
const duplicateTitles = [];
|
|
34
|
+
const outputFiles = (await glob(path.join(this.contentRoot, "**/*.html"))).sort();
|
|
35
|
+
const sourceRoot = path.dirname(path.dirname(this.fileName));
|
|
36
|
+
for (const outputFile of outputFiles) {
|
|
37
|
+
const url = path.relative(this.contentRoot, outputFile);
|
|
38
|
+
const sourceFile = path.resolve(sourceRoot, url);
|
|
39
|
+
if (!fs.existsSync(sourceFile)) {
|
|
40
|
+
continue;
|
|
32
41
|
}
|
|
33
|
-
|
|
34
|
-
|
|
42
|
+
const pageContext = context.clone();
|
|
43
|
+
(_a = this.prepareContext) === null || _a === void 0 ? void 0 : _a.call(this, pageContext, outputFile);
|
|
44
|
+
const html = fs.readFileSync(outputFile, this.encoding);
|
|
45
|
+
const titleMatch = /<title[^>]*>([\s\S]*?)<\/title>/i.exec(html);
|
|
46
|
+
const outputTitle = titleMatch ? this.decodeTitle(titleMatch[1]) : "";
|
|
47
|
+
const title = this.readSourceTitle(sourceFile) || outputTitle;
|
|
48
|
+
const pageInfo = this.searchCommand.pageInfoFrom(title, url.split(path.sep).join("/"), pageContext);
|
|
49
|
+
if (pageInfo) {
|
|
50
|
+
const resultTitle = SearchVisitor.resultTitle(pageInfo);
|
|
51
|
+
const titleIndexed = pagesByResultTitle.get(resultTitle);
|
|
52
|
+
if (titleIndexed && titleIndexed.url !== pageInfo.url) {
|
|
53
|
+
duplicateTitles.push(`Search result "${resultTitle}" with URL ${pageInfo.url} is already indexed with URL ${titleIndexed.url}`);
|
|
54
|
+
continue;
|
|
55
|
+
}
|
|
56
|
+
pagesByUrl.set(pageInfo.url, pageInfo);
|
|
57
|
+
pagesByResultTitle.set(resultTitle, pageInfo);
|
|
35
58
|
}
|
|
36
59
|
}
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
catch (e) {
|
|
40
|
-
if (e.errno !== -2) {
|
|
41
|
-
throw e;
|
|
60
|
+
if (duplicateTitles.length > 0) {
|
|
61
|
+
throw new Error(`${duplicateTitles.length} indistinguishable search results:\n${duplicateTitles.join("\n")}`);
|
|
42
62
|
}
|
|
43
|
-
|
|
44
|
-
|
|
63
|
+
const pages = Array.from(pagesByUrl.values());
|
|
64
|
+
pages.sort((pageInfo1, pageInfo2) => pageInfo1.title.localeCompare(pageInfo2.title));
|
|
65
|
+
existingIndex = { pages, words: newIndex.words };
|
|
45
66
|
}
|
|
67
|
+
else
|
|
68
|
+
try {
|
|
69
|
+
existingIndex = JSON.parse(fs.readFileSync(this.fileName, { encoding: this.encoding }));
|
|
70
|
+
const newPages = newIndex.pages;
|
|
71
|
+
const contentRoot = path.dirname(path.dirname(this.fileName));
|
|
72
|
+
const pagesByUrl = new Map(existingIndex.pages
|
|
73
|
+
.filter(page => fs.existsSync(path.resolve(contentRoot, page.url)))
|
|
74
|
+
.map(page => [page.url, page]));
|
|
75
|
+
for (const newPage of newPages) {
|
|
76
|
+
pagesByUrl.set(newPage.url, newPage);
|
|
77
|
+
}
|
|
78
|
+
existingIndex.pages = Array.from(pagesByUrl.values());
|
|
79
|
+
existingIndex.pages.sort((pageInfo1, pageInfo2) => pageInfo1.title > pageInfo2.title ? 1 : pageInfo1.title < pageInfo2.title ? -1 : 0);
|
|
80
|
+
}
|
|
81
|
+
catch (e) {
|
|
82
|
+
if (e.errno !== -2) {
|
|
83
|
+
throw e;
|
|
84
|
+
}
|
|
85
|
+
context.warn("Could not find", this.fileName, "Will create it");
|
|
86
|
+
existingIndex = newIndex;
|
|
87
|
+
}
|
|
46
88
|
const indexSize = existingIndex.pages.length;
|
|
47
89
|
context.setVar("indexSize", indexSize);
|
|
48
90
|
context.log("Saving search index of", indexSize, "pages at", this.fileName);
|
|
49
91
|
const indexJson = JSON.stringify(existingIndex);
|
|
50
92
|
return writeFile(this.fileName, indexJson, "utf-8");
|
|
51
93
|
}
|
|
94
|
+
decodeTitle(title) {
|
|
95
|
+
return title
|
|
96
|
+
.replace(/<[^>]+>/g, "")
|
|
97
|
+
.replace(/&#(\d+);/g, (_match, code) => String.fromCodePoint(Number(code)))
|
|
98
|
+
.replace(/&#x([\da-f]+);/gi, (_match, code) => String.fromCodePoint(Number.parseInt(code, 16)))
|
|
99
|
+
.replace(/ /gi, " ")
|
|
100
|
+
.replace(/&/gi, "&")
|
|
101
|
+
.replace(/"/gi, "\"")
|
|
102
|
+
.replace(/'|'/gi, "'")
|
|
103
|
+
.replace(/</gi, "<")
|
|
104
|
+
.replace(/>/gi, ">")
|
|
105
|
+
.replace(/\s+/g, " ")
|
|
106
|
+
.trim();
|
|
107
|
+
}
|
|
108
|
+
readSourceTitle(sourceFile) {
|
|
109
|
+
const source = fs.readFileSync(sourceFile, this.encoding);
|
|
110
|
+
const htmlTitleMatch = /<title[^>]*>([\s\S]*?)<\/title>/i.exec(source);
|
|
111
|
+
if (htmlTitleMatch) {
|
|
112
|
+
return this.decodeTitle(htmlTitleMatch[1]);
|
|
113
|
+
}
|
|
114
|
+
for (const directiveMatch of source.matchAll(/<!--#set\s+([\s\S]*?)-->/gi)) {
|
|
115
|
+
const attributes = directiveMatch[1];
|
|
116
|
+
if (!/\bvar\s*=\s*(["'])title\1/i.test(attributes)) {
|
|
117
|
+
continue;
|
|
118
|
+
}
|
|
119
|
+
const valueMatch = /\bvalue\s*=\s*(["'])([\s\S]*?)\1/i.exec(attributes);
|
|
120
|
+
if (valueMatch) {
|
|
121
|
+
return this.decodeTitle(valueMatch[2]);
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
return "";
|
|
125
|
+
}
|
|
52
126
|
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
import { describe, expect, test } from "@javarome/testscript";
|
|
2
|
+
import fs from "fs";
|
|
3
|
+
import os from "os";
|
|
4
|
+
import path from "path";
|
|
5
|
+
import { SsgContextImpl } from "ssg-api";
|
|
6
|
+
import { SearchIndexStep } from "./SearchIndexStep.js";
|
|
7
|
+
describe("SearchIndexStep", () => {
|
|
8
|
+
test("removes missing and duplicate pages while updating the current pages", async () => {
|
|
9
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), "rr0-search-index-"));
|
|
10
|
+
try {
|
|
11
|
+
const searchDir = path.join(root, "search");
|
|
12
|
+
fs.mkdirSync(searchDir);
|
|
13
|
+
fs.writeFileSync(path.join(root, "current.html"), "");
|
|
14
|
+
const fileName = path.join(searchDir, "index.json");
|
|
15
|
+
const existingIndex = {
|
|
16
|
+
pages: [
|
|
17
|
+
{ title: "Old current title", url: "current.html", time: "" },
|
|
18
|
+
{ title: "Duplicate current title", url: "current.html", time: "" },
|
|
19
|
+
{ title: "Moved page", url: "old/path.html", time: "" }
|
|
20
|
+
],
|
|
21
|
+
words: {}
|
|
22
|
+
};
|
|
23
|
+
fs.writeFileSync(fileName, JSON.stringify(existingIndex));
|
|
24
|
+
const newIndex = {
|
|
25
|
+
pages: [
|
|
26
|
+
{ title: "Current title", url: "current.html", time: "" },
|
|
27
|
+
{ title: "New page", url: "new.html", time: "" }
|
|
28
|
+
],
|
|
29
|
+
words: {}
|
|
30
|
+
};
|
|
31
|
+
const visitor = { index: newIndex };
|
|
32
|
+
const context = {
|
|
33
|
+
setVar() {
|
|
34
|
+
},
|
|
35
|
+
log() {
|
|
36
|
+
},
|
|
37
|
+
warn() {
|
|
38
|
+
}
|
|
39
|
+
};
|
|
40
|
+
await new SearchIndexStep(fileName, visitor).execute(context);
|
|
41
|
+
const savedIndex = JSON.parse(fs.readFileSync(fileName, "utf8"));
|
|
42
|
+
expect(savedIndex.pages).toEqual([
|
|
43
|
+
{ title: "Current title", url: "current.html", time: "" },
|
|
44
|
+
{ title: "New page", url: "new.html", time: "" }
|
|
45
|
+
]);
|
|
46
|
+
}
|
|
47
|
+
finally {
|
|
48
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
49
|
+
}
|
|
50
|
+
});
|
|
51
|
+
test("rebuilds the index from every HTML page in the output directory", async () => {
|
|
52
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), "rr0-search-index-"));
|
|
53
|
+
try {
|
|
54
|
+
const outDir = path.join(root, "out");
|
|
55
|
+
const searchDir = path.join(root, "search");
|
|
56
|
+
fs.mkdirSync(path.join(outDir, "science"), { recursive: true });
|
|
57
|
+
fs.mkdirSync(searchDir);
|
|
58
|
+
fs.mkdirSync(path.join(root, "science"));
|
|
59
|
+
fs.writeFileSync(path.join(outDir, "science", "UfoAtHome.html"), "<html><head><title>UFO@home</title></head><body></body></html>");
|
|
60
|
+
fs.writeFileSync(path.join(outDir, "other.html"), "<html><head><title>Other page</title></head><body></body></html>");
|
|
61
|
+
fs.writeFileSync(path.join(root, "science", "UfoAtHome.html"), "");
|
|
62
|
+
fs.writeFileSync(path.join(root, "other.html"), "");
|
|
63
|
+
fs.writeFileSync(path.join(outDir, "stale.html"), "<html><head><title>Stale page</title></head><body></body></html>");
|
|
64
|
+
const fileName = path.join(searchDir, "index.json");
|
|
65
|
+
const newIndex = { pages: [], words: {} };
|
|
66
|
+
const visitor = {
|
|
67
|
+
index: newIndex,
|
|
68
|
+
pageInfoFrom(title, url) {
|
|
69
|
+
return {
|
|
70
|
+
title,
|
|
71
|
+
url,
|
|
72
|
+
time: ""
|
|
73
|
+
};
|
|
74
|
+
}
|
|
75
|
+
};
|
|
76
|
+
const context = new SsgContextImpl("fr");
|
|
77
|
+
await new SearchIndexStep(fileName, visitor, outDir).execute(context);
|
|
78
|
+
const savedIndex = JSON.parse(fs.readFileSync(fileName, "utf8"));
|
|
79
|
+
expect(savedIndex.pages).toEqual([
|
|
80
|
+
{ title: "Other page", url: "other.html", time: "" },
|
|
81
|
+
{ title: "UFO@home", url: "science/UfoAtHome.html", time: "" }
|
|
82
|
+
]);
|
|
83
|
+
}
|
|
84
|
+
finally {
|
|
85
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
86
|
+
}
|
|
87
|
+
});
|
|
88
|
+
test("rejects the same title at two different URLs", async () => {
|
|
89
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), "rr0-search-index-"));
|
|
90
|
+
try {
|
|
91
|
+
const outDir = path.join(root, "out");
|
|
92
|
+
const searchDir = path.join(root, "search");
|
|
93
|
+
fs.mkdirSync(outDir);
|
|
94
|
+
fs.mkdirSync(searchDir);
|
|
95
|
+
fs.writeFileSync(path.join(outDir, "first.html"), "<html><head><title>Same title</title></head><body></body></html>");
|
|
96
|
+
fs.writeFileSync(path.join(outDir, "second.html"), "<html><head><title>Same title</title></head><body></body></html>");
|
|
97
|
+
fs.writeFileSync(path.join(root, "first.html"), "");
|
|
98
|
+
fs.writeFileSync(path.join(root, "second.html"), "");
|
|
99
|
+
const visitor = {
|
|
100
|
+
index: { pages: [], words: {} },
|
|
101
|
+
pageInfoFrom(title, url) {
|
|
102
|
+
return {
|
|
103
|
+
title,
|
|
104
|
+
url,
|
|
105
|
+
time: ""
|
|
106
|
+
};
|
|
107
|
+
}
|
|
108
|
+
};
|
|
109
|
+
const context = new SsgContextImpl("fr");
|
|
110
|
+
let error;
|
|
111
|
+
try {
|
|
112
|
+
await new SearchIndexStep(path.join(searchDir, "index.json"), visitor, outDir).execute(context);
|
|
113
|
+
}
|
|
114
|
+
catch (e) {
|
|
115
|
+
error = e;
|
|
116
|
+
}
|
|
117
|
+
expect(error === null || error === void 0 ? void 0 : error.message.includes("Search result \"Same title\"")).toEqual(true);
|
|
118
|
+
}
|
|
119
|
+
finally {
|
|
120
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
121
|
+
}
|
|
122
|
+
});
|
|
123
|
+
test("allows identical document titles when their search result dates differ", async () => {
|
|
124
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), "rr0-search-index-"));
|
|
125
|
+
try {
|
|
126
|
+
const outDir = path.join(root, "out");
|
|
127
|
+
const searchDir = path.join(root, "search");
|
|
128
|
+
fs.mkdirSync(outDir);
|
|
129
|
+
fs.mkdirSync(searchDir);
|
|
130
|
+
for (const fileName of ["first.html", "second.html"]) {
|
|
131
|
+
fs.writeFileSync(path.join(outDir, fileName), "<html><head><title>Same title</title></head><body></body></html>");
|
|
132
|
+
fs.writeFileSync(path.join(root, fileName), "");
|
|
133
|
+
}
|
|
134
|
+
const visitor = {
|
|
135
|
+
index: { pages: [], words: {} },
|
|
136
|
+
pageInfoFrom(title, url) {
|
|
137
|
+
return { title, url, time: url === "first.html" ? "1900" : "1901" };
|
|
138
|
+
}
|
|
139
|
+
};
|
|
140
|
+
const context = new SsgContextImpl("fr");
|
|
141
|
+
const fileName = path.join(searchDir, "index.json");
|
|
142
|
+
await new SearchIndexStep(fileName, visitor, outDir).execute(context);
|
|
143
|
+
const savedIndex = JSON.parse(fs.readFileSync(fileName, "utf8"));
|
|
144
|
+
expect(savedIndex.pages.length).toEqual(2);
|
|
145
|
+
}
|
|
146
|
+
finally {
|
|
147
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
148
|
+
}
|
|
149
|
+
});
|
|
150
|
+
test("prefers an explicit source title over a generated chronological title", async () => {
|
|
151
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), "rr0-search-index-"));
|
|
152
|
+
try {
|
|
153
|
+
const outDir = path.join(root, "out");
|
|
154
|
+
const searchDir = path.join(root, "search");
|
|
155
|
+
fs.mkdirSync(outDir);
|
|
156
|
+
fs.mkdirSync(searchDir);
|
|
157
|
+
fs.writeFileSync(path.join(outDir, "article.html"), "<html><head><title>Mai 1954</title></head><body></body></html>");
|
|
158
|
+
fs.writeFileSync(path.join(root, "article.html"), "<!--#set var=\"title\" value=\"Canada Hunts for Saucers\" --><!--#include virtual=\"/header.html\" -->");
|
|
159
|
+
const visitor = {
|
|
160
|
+
index: { pages: [], words: {} },
|
|
161
|
+
pageInfoFrom(title, url) {
|
|
162
|
+
return { title, url, time: "mai 1954" };
|
|
163
|
+
}
|
|
164
|
+
};
|
|
165
|
+
const context = new SsgContextImpl("fr");
|
|
166
|
+
const fileName = path.join(searchDir, "index.json");
|
|
167
|
+
await new SearchIndexStep(fileName, visitor, outDir).execute(context);
|
|
168
|
+
const savedIndex = JSON.parse(fs.readFileSync(fileName, "utf8"));
|
|
169
|
+
expect(savedIndex.pages).toEqual([
|
|
170
|
+
{ title: "Canada Hunts for Saucers", url: "article.html", time: "mai 1954" }
|
|
171
|
+
]);
|
|
172
|
+
}
|
|
173
|
+
finally {
|
|
174
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
175
|
+
}
|
|
176
|
+
});
|
|
177
|
+
});
|
|
@@ -30,12 +30,15 @@ export type SearchCommandConfig = {
|
|
|
30
30
|
export declare class SearchVisitor implements FileVisitor {
|
|
31
31
|
protected config: SearchCommandConfig;
|
|
32
32
|
protected timeTextBuilder: TimeTextBuilder;
|
|
33
|
+
static resultTitle(pageInfo: PageInfo): string;
|
|
33
34
|
readonly index: SearchIndex;
|
|
34
35
|
protected readonly contentStream: fs.WriteStream | undefined;
|
|
35
36
|
constructor(config: SearchCommandConfig, timeTextBuilder: TimeTextBuilder);
|
|
36
37
|
contentStepEnd(): Promise<void>;
|
|
37
38
|
visit(context: HtmlRR0Context): Promise<void>;
|
|
38
|
-
|
|
39
|
+
pageInfo(context: HtmlRR0Context, contentRoot?: string): PageInfo | undefined;
|
|
40
|
+
pageInfoFrom(title: string, url: string, context: HtmlRR0Context): PageInfo | undefined;
|
|
41
|
+
protected handleAlreadyIndexed(resultTitle: string, url: string, titleIndexed: PageInfo): void;
|
|
39
42
|
protected getContents(doc: Document): string;
|
|
40
43
|
protected indexContent(context: HtmlRR0Context, outputFile: HtmlFileContents): void;
|
|
41
44
|
protected indexWords(context: HtmlRR0Context, outputFile: HtmlFileContents): void;
|
|
@@ -1,8 +1,13 @@
|
|
|
1
1
|
import fs from "fs";
|
|
2
|
+
import path from "path";
|
|
2
3
|
/**
|
|
3
4
|
* Builds an index of pages.
|
|
4
5
|
*/
|
|
5
6
|
export class SearchVisitor {
|
|
7
|
+
static resultTitle(pageInfo) {
|
|
8
|
+
const { title, time } = pageInfo;
|
|
9
|
+
return title + (time && time !== title.toLowerCase() ? ` (${time})` : "");
|
|
10
|
+
}
|
|
6
11
|
constructor(config, timeTextBuilder) {
|
|
7
12
|
this.config = config;
|
|
8
13
|
this.timeTextBuilder = timeTextBuilder;
|
|
@@ -24,17 +29,21 @@ export class SearchVisitor {
|
|
|
24
29
|
}
|
|
25
30
|
async visit(context) {
|
|
26
31
|
const file = context.file;
|
|
27
|
-
const
|
|
28
|
-
|
|
29
|
-
const url = file.name.startsWith(outDir) ? file.name.substring(outDir.length) : file.name;
|
|
30
|
-
if (title && !this.config.notIndexedUrls.includes(url)) {
|
|
32
|
+
const pageInfo = this.pageInfo(context);
|
|
33
|
+
if (pageInfo) {
|
|
31
34
|
const indexedPages = this.index.pages;
|
|
32
|
-
const
|
|
35
|
+
const pageIndex = indexedPages.findIndex(page => page.url === pageInfo.url);
|
|
36
|
+
const resultTitle = SearchVisitor.resultTitle(pageInfo);
|
|
37
|
+
const titleIndexed = indexedPages.find(page => SearchVisitor.resultTitle(page) === resultTitle && page.url !== pageInfo.url);
|
|
33
38
|
if (titleIndexed) {
|
|
34
|
-
this.handleAlreadyIndexed(
|
|
39
|
+
this.handleAlreadyIndexed(resultTitle, pageInfo.url, titleIndexed);
|
|
40
|
+
}
|
|
41
|
+
if (pageIndex >= 0) {
|
|
42
|
+
indexedPages[pageIndex] = pageInfo;
|
|
43
|
+
}
|
|
44
|
+
else {
|
|
45
|
+
indexedPages.push(pageInfo);
|
|
35
46
|
}
|
|
36
|
-
const time = this.timeTextBuilder.build(context, { year: "numeric", month: "short", day: "numeric" }).toLowerCase();
|
|
37
|
-
indexedPages.push({ title, url, time });
|
|
38
47
|
}
|
|
39
48
|
if (this.config.indexWords) {
|
|
40
49
|
this.indexWords(context, file);
|
|
@@ -43,8 +52,22 @@ export class SearchVisitor {
|
|
|
43
52
|
this.indexContent(context, file);
|
|
44
53
|
}
|
|
45
54
|
}
|
|
46
|
-
|
|
47
|
-
|
|
55
|
+
pageInfo(context, contentRoot) {
|
|
56
|
+
const file = context.file;
|
|
57
|
+
const url = contentRoot
|
|
58
|
+
? path.relative(contentRoot, file.name).split(path.sep).join("/")
|
|
59
|
+
: file.name.startsWith("out/") ? file.name.substring("out/".length) : file.name;
|
|
60
|
+
return this.pageInfoFrom(file.title, url, context);
|
|
61
|
+
}
|
|
62
|
+
pageInfoFrom(title, url, context) {
|
|
63
|
+
if (!title || this.config.notIndexedUrls.includes(url)) {
|
|
64
|
+
return undefined;
|
|
65
|
+
}
|
|
66
|
+
const time = this.timeTextBuilder.build(context, { year: "numeric", month: "short", day: "numeric" }).toLowerCase();
|
|
67
|
+
return { title, url, time };
|
|
68
|
+
}
|
|
69
|
+
handleAlreadyIndexed(resultTitle, url, titleIndexed) {
|
|
70
|
+
throw new Error(`Search result "${resultTitle}" with URL ${url} is already indexed with URL ${titleIndexed.url}`);
|
|
48
71
|
}
|
|
49
72
|
getContents(doc) {
|
|
50
73
|
const div = doc.createElement("div");
|
|
@@ -64,7 +64,8 @@ export class CsvMapper {
|
|
|
64
64
|
*/
|
|
65
65
|
mapAll(context, sourceCases, sourceTime) {
|
|
66
66
|
const values = sourceCases.map(c => this.map(context, c, sourceTime));
|
|
67
|
-
|
|
67
|
+
const header = Array.from(this.fields).sort((field1, field2) => field1.localeCompare(field2));
|
|
68
|
+
return header.join(this.sep) + "\n" + values.join("\n");
|
|
68
69
|
}
|
|
69
70
|
escape(value, force) {
|
|
70
71
|
if (this.escapeStr && (force || value.indexOf(this.sep) >= 0)) {
|
package/package.json
CHANGED