@opentermsarchive/engine 16.0.2 → 16.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/config/default.json +13 -0
  2. package/config/test.json +12 -0
  3. package/package.json +1 -1
  4. package/scripts/declarations/validate/index.mocha.js +7 -0
  5. package/scripts/import/index.js +1 -1
  6. package/scripts/import/loadCommits.js +1 -1
  7. package/scripts/rewrite/initializer/index.js +1 -1
  8. package/scripts/rewrite/rewrite-snapshots.js +1 -1
  9. package/scripts/rewrite/rewrite-versions.js +1 -1
  10. package/src/archivist/fetcher/htmlOnlyFetcher.js +8 -0
  11. package/src/archivist/fetcher/index.test.js +12 -0
  12. package/src/archivist/index.js +106 -23
  13. package/src/archivist/index.test.js +659 -9
  14. package/src/archivist/recorder/repositories/git/dataMapper.js +2 -1
  15. package/src/archivist/recorder/repositories/git/index.js +2 -13
  16. package/src/archivist/recorder/repositories/git/index.test.js +23 -1
  17. package/src/archivist/services/index.js +45 -1
  18. package/src/archivist/services/index.test.js +52 -1
  19. package/src/archivist/services/sourceDocument.js +8 -2
  20. package/src/archivist/services/sourceDocument.test.js +37 -0
  21. package/src/archivist/tracking-results/errors.js +5 -0
  22. package/src/archivist/tracking-results/index.js +233 -0
  23. package/src/archivist/tracking-results/index.test.js +406 -0
  24. package/src/archivist/tracking-results/recorder.js +156 -0
  25. package/src/archivist/tracking-results/recorder.test.js +559 -0
  26. package/src/archivist/tracking-results/repository.js +136 -0
  27. package/src/archivist/tracking-results/repository.test.js +763 -0
  28. package/src/archivist/tracking-results/run/dataMapper.js +51 -0
  29. package/src/archivist/tracking-results/run/dataMapper.test.js +168 -0
  30. package/src/archivist/tracking-results/run/index.js +115 -0
  31. package/src/archivist/tracking-results/run/index.test.js +221 -0
  32. package/src/archivist/tracking-results/terms-result/dataMapper.js +200 -0
  33. package/src/archivist/tracking-results/terms-result/dataMapper.test.js +575 -0
  34. package/src/archivist/tracking-results/terms-result/index.js +42 -0
  35. package/src/archivist/tracking-results/terms-result/index.test.js +228 -0
  36. package/src/git/errors.js +1 -0
  37. package/src/{archivist/recorder/repositories/git/git.js → git/index.js} +59 -9
  38. package/src/git/index.test.js +266 -0
  39. package/src/git/pathSegment.js +18 -0
  40. package/src/git/pathSegment.test.js +33 -0
  41. package/src/index.js +5 -4
  42. package/src/reporter/index.js +9 -4
  43. package/src/reporter/index.test.js +47 -9
  44. package/src/archivist/recorder/repositories/git/git.test.js +0 -114
  45. /package/src/{archivist/recorder/repositories/git → git}/trailers.js +0 -0
  46. /package/src/{archivist/recorder/repositories/git → git}/trailers.test.js +0 -0
@@ -67,7 +67,8 @@ export function toDomain(commit) {
67
67
  }
68
68
 
69
69
  const [relativeFilePath] = modifiedFilesInCommit;
70
- const snapshotIdsMatch = body.match(/\b[0-9a-f]{5,40}\b/g);
70
+ const bodyWithoutTrailers = Object.keys(trailers).length ? body.split(/\n\n+/).slice(0, -1).join('\n\n') : body; // Trailers, when present, are the last section of the body; their values, such as the hexadecimal segments of a run ID, must not be read as snapshot IDs
71
+ const snapshotIdsMatch = bodyWithoutTrailers.match(/\b[0-9a-f]{5,40}\b/g);
71
72
 
72
73
  const [ termsType, documentId ] = path.basename(relativeFilePath, path.extname(relativeFilePath)).split(TERMS_TYPE_AND_DOCUMENT_ID_SEPARATOR);
73
74
 
@@ -8,27 +8,16 @@ import path from 'path';
8
8
 
9
9
  import mime from 'mime';
10
10
 
11
+ import Git from '../../../../git/index.js';
12
+ import { isPlainPathSegment } from '../../../../git/pathSegment.js';
11
13
  import RepositoryInterface from '../interface.js';
12
14
 
13
15
  import * as DataMapper from './dataMapper.js';
14
- import Git from './git.js';
15
16
 
16
17
  const fs = fsApi.promises;
17
18
 
18
19
  const RECORD_ID_REGEXP = /^[0-9a-f]{7,40}$/i; // Git commit SHA 7 (abbreviated) to 40 (full) hexadecimal characters. Prevent value such as `--output=…` to be parsed as a command-line option
19
20
 
20
- const CONTROL_CHARACTERS_REGEXP = /\p{Cc}/u; // Matches any Unicode "control" character: the C0 range (U+0000 to U+001F), DEL (U+007F) and the C1 range (U+0080 to U+009F), i.e. 65 non-printable characters including NUL. The `u` flag is required for the `\p{...}` property escape to be recognised, otherwise the pattern would match the literal text `p{Cc}`. Legitimate service IDs, terms types and document IDs never contain these, and NUL in particular can truncate a value once it reaches git or the filesystem, so any segment holding one is rejected.
21
-
22
- // Keeps hostile values from reaching git, where a pathspec that resolves outside the repository (such as `../foo/*`) aborts with an error that exposes the repository location.
23
- function isPlainPathSegment(segment) {
24
- return segment.length > 0
25
- && segment !== '.'
26
- && segment !== '..'
27
- && !segment.includes('/')
28
- && !segment.includes('\\')
29
- && !CONTROL_CHARACTERS_REGEXP.test(segment);
30
- }
31
-
32
21
  function canMatchRecordFilePath(...pathSegments) {
33
22
  // A non-string segment means "not provided" (`undefined`, or `false` for an absent document ID) and constrains nothing
34
23
  return pathSegments.every(segment => typeof segment !== 'string' || isPlainPathSegment(segment));
@@ -7,11 +7,11 @@ import chaiAsPromised from 'chai-as-promised';
7
7
  import config from 'config';
8
8
  import mime from 'mime';
9
9
 
10
+ import Git from '../../../../git/index.js';
10
11
  import Snapshot from '../../snapshot.js';
11
12
  import Version from '../../version.js';
12
13
 
13
14
  import { TERMS_TYPE_AND_DOCUMENT_ID_SEPARATOR, SNAPSHOT_ID_MARKER, COMMIT_MESSAGE_PREFIXES } from './dataMapper.js';
14
- import Git from './git.js';
15
15
 
16
16
  import GitRepository from './index.js';
17
17
 
@@ -398,6 +398,28 @@ describe('GitRepository', () => {
398
398
  expect(record.metadata).to.deep.equal(METADATA);
399
399
  });
400
400
 
401
+ context('when a metadata value contains hexadecimal segments', () => {
402
+ let recordWithRunId;
403
+
404
+ before(async () => {
405
+ const { id: recordId } = await subject.save(new Version({
406
+ serviceId: SERVICE_PROVIDER_ID,
407
+ termsType: TERMS_TYPE,
408
+ content: `${CONTENT} (updated)`,
409
+ fetchDate: FETCH_DATE_LATER,
410
+ snapshotIds: [SNAPSHOT_ID],
411
+ mimeType: HTML_MIME_TYPE,
412
+ metadata: { ...METADATA, 'x-run-id': 'ota-run-189e7be3-60ef-40c7-ac28-81a44288e105' },
413
+ }));
414
+
415
+ recordWithRunId = await subject.findById(recordId);
416
+ });
417
+
418
+ it('returns only the snapshot ID', () => {
419
+ expect(recordWithRunId.snapshotIds).to.deep.equal([SNAPSHOT_ID]);
420
+ });
421
+ });
422
+
401
423
  context('when requested record does not exist', () => {
402
424
  it('returns null', async () => {
403
425
  expect(await subject.findById('inexistantID')).to.equal(null);
@@ -2,8 +2,10 @@ import fs from 'fs/promises';
2
2
  import path from 'path';
3
3
  import { pathToFileURL } from 'url';
4
4
 
5
+ import async from 'async';
5
6
  import config from 'config';
6
7
 
8
+ import Git from '../../git/index.js';
7
9
  import * as exposedFilters from '../extract/exposedFilters.js';
8
10
 
9
11
  import Service from './service.js';
@@ -13,6 +15,8 @@ import Terms from './terms.js';
13
15
  export const DECLARATIONS_PATH = './declarations';
14
16
  const declarationsPath = path.resolve(process.cwd(), config.get('@opentermsarchive/engine.collectionPath'), DECLARATIONS_PATH);
15
17
 
18
+ const MAX_PARALLEL_DECLARATIONS_READS = 5; // Reading declarations at a commit spawns one git process per service; left unbounded, a large collection would exhaust the file descriptors or processes allowed to the engine
19
+
16
20
  const JSON_EXT = '.json';
17
21
  const JS_EXT = '.js';
18
22
  const HISTORY_SUFFIX = '.history';
@@ -106,10 +110,11 @@ function createWrappedFilter(baseFunction, filterName, filterParams) {
106
110
  return;
107
111
  }
108
112
 
109
- if (filterParams || exposedFilters[filterName]) { // Built-in filters always receive their parameters before the context, even when none are declared
113
+ if (filterParams !== undefined || exposedFilters[filterName]) { // Built-in filters always receive their parameters before the context, even when none are declared
110
114
  const wrappedFilter = (webPageDOM, context) => baseFunction(webPageDOM, filterParams, context);
111
115
 
112
116
  Object.defineProperty(wrappedFilter, 'name', { value: filterName });
117
+ Object.defineProperty(wrappedFilter, 'declaration', { value: filterParams === undefined ? filterName : { [filterName]: filterParams } }); // Keep the declared form, as the parameters are otherwise only reachable through the closure
113
118
 
114
119
  return wrappedFilter;
115
120
  }
@@ -145,6 +150,45 @@ export function getServiceFilters(serviceFilters, declaredFilters) {
145
150
  export async function getDeclaredServicesIds() {
146
151
  const fileNames = await fs.readdir(declarationsPath);
147
152
 
153
+ return declaredServicesIdsFromFileNames(fileNames);
154
+ }
155
+
156
+ export function getDeclarationsCommit() { // Resolves to null when the declarations are not versioned with Git
157
+ return Git.getHeadSha(declarationsPath);
158
+ }
159
+
160
+ // Returns the [{ serviceId, termsType }] declared at the given commit of the declarations repository.
161
+ // Reads through git so the answer reflects the declarations exactly as they were at that commit, not as they are on disk; used by crash recovery to derive the coverage of a run that referenced this commit.
162
+ export async function getDeclaredTermsAtCommit(commit) {
163
+ const fileNames = await Git.listFilesAtCommit(declarationsPath, commit);
164
+ const serviceIds = declaredServicesIdsFromFileNames(fileNames);
165
+
166
+ return declaredTermsOf(serviceIds, async serviceId => {
167
+ const rawDeclaration = await Git.readFileAtCommit(declarationsPath, commit, `${serviceId}${JSON_EXT}`);
168
+
169
+ try {
170
+ return JSON.parse(rawDeclaration);
171
+ } catch (error) {
172
+ throw new Error(`The "${serviceId}" service declaration at commit ${commit} is malformed and cannot be parsed`);
173
+ }
174
+ });
175
+ }
176
+
177
+ export async function getDeclaredTerms() { // Working-tree counterpart of getDeclaredTermsAtCommit; used as approximation when a declarations commit is not reachable anymore
178
+ return declaredTermsOf(await getDeclaredServicesIds(), loadServiceDeclaration);
179
+ }
180
+
181
+ async function declaredTermsOf(serviceIds, loadDeclaration) {
182
+ const declaredTermsPerService = await async.mapLimit(serviceIds, MAX_PARALLEL_DECLARATIONS_READS, async serviceId => {
183
+ const declaration = await loadDeclaration(serviceId);
184
+
185
+ return Object.keys(declaration.terms ?? {}).map(termsType => ({ serviceId, termsType }));
186
+ });
187
+
188
+ return declaredTermsPerService.flat(); // Flattened in serviceIds order to keep the result deterministic regardless of read completion order
189
+ }
190
+
191
+ function declaredServicesIdsFromFileNames(fileNames) {
148
192
  return fileNames
149
193
  .filter(fileName => fileName.endsWith(JSON_EXT) && !fileName.includes(`${HISTORY_SUFFIX}${JSON_EXT}`))
150
194
  .map(fileName => path.basename(fileName, JSON_EXT));
@@ -2,10 +2,12 @@ import fs from 'fs/promises';
2
2
  import path from 'path';
3
3
 
4
4
  import { expect, use } from 'chai';
5
+ import config from 'config';
5
6
  import sinon from 'sinon';
6
7
  import sinonChai from 'sinon-chai';
7
8
 
8
9
  import expectedServices from '../../../test/fixtures/services.js';
10
+ import Git, { GitObjectNotFoundError } from '../../git/index.js';
9
11
  import createWebPageDOM from '../extract/dom.js';
10
12
  import * as exposedFilters from '../extract/exposedFilters.js';
11
13
 
@@ -13,7 +15,7 @@ import Service from './service.js';
13
15
  import SourceDocument from './sourceDocument.js';
14
16
  import Terms from './terms.js';
15
17
 
16
- import { getDeclaredServicesIds, loadServiceDeclaration, loadServiceFilters, getServiceFilters, createSourceDocuments, createServiceFromDeclaration, load, loadWithHistory } from './index.js';
18
+ import { getDeclaredServicesIds, getDeclaredTermsAtCommit, loadServiceDeclaration, loadServiceFilters, getServiceFilters, createSourceDocuments, createServiceFromDeclaration, load, loadWithHistory } from './index.js';
17
19
 
18
20
  use(sinonChai);
19
21
 
@@ -151,6 +153,36 @@ describe('Services', () => {
151
153
  });
152
154
  });
153
155
 
156
+ describe('#getDeclaredTermsAtCommit', () => {
157
+ const declarationsPath = path.resolve(process.cwd(), config.get('@opentermsarchive/engine.collectionPath'), './declarations');
158
+
159
+ afterEach(() => sinon.restore());
160
+
161
+ it('returns the terms declared at the given commit, whatever the working tree contains', async function () {
162
+ this.timeout(10000);
163
+
164
+ const commit = await Git.getHeadSha(declarationsPath); // The test declarations are committed in the engine repository itself, so its HEAD reflects them exactly
165
+ const expected = Object.entries(expectedServices).flatMap(([ serviceId, service ]) => service.getTermsTypes().map(termsType => ({ serviceId, termsType })));
166
+
167
+ sinon.stub(fs, 'readdir').resolves(['uncommitted-service.json']); // Make the working tree differ from the commit: it must not be read
168
+ sinon.stub(fs, 'readFile').resolves(JSON.stringify({ name: 'Uncommitted service', terms: { Imprint: {} } }));
169
+
170
+ expect(await getDeclaredTermsAtCommit(commit)).to.have.deep.members(expected);
171
+ });
172
+
173
+ it('throws a GitObjectNotFoundError for an unknown commit', async () => {
174
+ try {
175
+ await getDeclaredTermsAtCommit('deadbeefdeadbeefdeadbeefdeadbeefdeadbeef');
176
+ } catch (error) {
177
+ expect(error).to.be.an.instanceOf(GitObjectNotFoundError);
178
+
179
+ return;
180
+ }
181
+
182
+ expect.fail('No error was thrown');
183
+ });
184
+ });
185
+
154
186
  describe('#loadServiceDeclaration', () => {
155
187
  let readFile;
156
188
 
@@ -310,6 +342,19 @@ describe('Services', () => {
310
342
  expect(result[0](null, 'context')).to.equal('foo');
311
343
  });
312
344
 
345
+ it('keeps the declared form of filters declared with parameters', () => {
346
+ const [filter] = getServiceFilters({ paramFilter: (dom, param) => param }, [{ paramFilter: [ 'foo', 'bar' ] }]);
347
+
348
+ expect(filter.declaration).to.deep.equal({ paramFilter: [ 'foo', 'bar' ] });
349
+ });
350
+
351
+ it('keeps the name as the declared form of exposed filters declared by string name', () => {
352
+ const [filterName] = Object.keys(exposedFilters);
353
+ const [filter] = getServiceFilters({}, [filterName]);
354
+
355
+ expect(filter.declaration).to.equal(filterName);
356
+ });
357
+
313
358
  describe('parameters passed to filters', () => {
314
359
  let serviceLoadedFilters;
315
360
  let passedDOM;
@@ -348,6 +393,12 @@ describe('Services', () => {
348
393
  testParameterPassing({ param1: 'param1', param2: 'param2' });
349
394
  });
350
395
  });
396
+
397
+ context('as a falsy value', () => {
398
+ it('passes parameters correctly', () => {
399
+ testParameterPassing(false);
400
+ });
401
+ });
351
402
  });
352
403
  });
353
404
 
@@ -40,8 +40,14 @@ export default class SourceDocument {
40
40
  }
41
41
 
42
42
  clearContent() {
43
- this.content = null;
43
+ this.content = null; // Only the potentially large content is cleared: the MIME type is read after the extraction, to record the tracking results
44
+ }
45
+
46
+ resetObservations() {
47
+ // mimeType and snapshotId are observations of a single tracking attempt, but they are stored on declaration objects that live for the whole process: without this reset, a failed fetch would expose the previous run's values as if they belonged to the failed attempt.
48
+ // The proper pattern would be for the fetch and extract pipeline to return its observations instead of mutating the declarations, letting consumers build their records from run-scoped data; this reset contains that debt rather than fixing it.
44
49
  this.mimeType = null;
50
+ this.snapshotId = null;
45
51
  }
46
52
 
47
53
  static extractCssSelectorsFromProperty(property) {
@@ -83,7 +89,7 @@ export default class SourceDocument {
83
89
  fetch: this.location,
84
90
  select: this.contentSelectors,
85
91
  remove: this.insignificantContentSelectors,
86
- filter: this.filters ? this.filters.map(filter => filter.name) : undefined,
92
+ filter: this.filters ? this.filters.map(filter => filter.declaration ?? filter.name) : undefined, // Filters declared with parameters carry their declared form, so that a change of parameters is persisted
87
93
  executeClientScripts: this.executeClientScripts,
88
94
  };
89
95
  }
@@ -214,6 +214,29 @@ describe('SourceDocument', () => {
214
214
  });
215
215
  });
216
216
 
217
+ describe('#clearContent', () => {
218
+ it('clears the content but keeps the MIME type', () => {
219
+ const sourceDocument = new SourceDocument({ location: URL, content: '<html></html>', mimeType: 'text/html' });
220
+
221
+ sourceDocument.clearContent();
222
+
223
+ expect(sourceDocument.content).to.be.null;
224
+ expect(sourceDocument.mimeType).to.equal('text/html');
225
+ });
226
+ });
227
+
228
+ describe('#resetObservations', () => {
229
+ it('clears the MIME type and the snapshot ID observed by a previous tracking', () => {
230
+ const sourceDocument = new SourceDocument({ location: URL, mimeType: 'text/html' });
231
+
232
+ sourceDocument.snapshotId = 'abc123';
233
+ sourceDocument.resetObservations();
234
+
235
+ expect(sourceDocument.mimeType).to.be.null;
236
+ expect(sourceDocument.snapshotId).to.be.null;
237
+ });
238
+ });
239
+
217
240
  describe('#toPersistence', () => {
218
241
  it('converts basic source document declarations into JSON representation', () => {
219
242
  const result = new SourceDocument({
@@ -275,5 +298,19 @@ describe('SourceDocument', () => {
275
298
 
276
299
  expect(result).to.deep.equal(expectedResult);
277
300
  });
301
+
302
+ it('converts filters declared with parameters to their declared form', () => {
303
+ const filterWithParameters = () => {};
304
+
305
+ Object.defineProperty(filterWithParameters, 'declaration', { value: { removeQueryParams: ['utm_source'] } });
306
+
307
+ const result = new SourceDocument({
308
+ location: URL,
309
+ contentSelectors: 'body',
310
+ filters: [ filterWithParameters, function filterSomething() {} ],
311
+ }).toPersistence();
312
+
313
+ expect(result.filter).to.deep.equal([{ removeQueryParams: ['utm_source'] }, 'filterSomething' ]);
314
+ });
278
315
  });
279
316
  });
@@ -0,0 +1,5 @@
1
+ /* eslint-disable max-classes-per-file */ // Grouping the module's error types here is the point of an errors module
2
+
3
+ export class MissingCollectionIdError extends Error {} // The collection metadata does not provide the id that identifies the collection in every persisted run; tracking-results cannot record without it
4
+
5
+ export class UnreadableRunError extends Error {} // The persisted run.json cannot be read as a valid Run (corrupted JSON, incompatible schema from another engine version): recovery is impossible by construction, unlike infrastructure failures for which a retry is meaningful
@@ -0,0 +1,233 @@
1
+ import events from 'events';
2
+ import { createRequire } from 'module';
3
+
4
+ import config from 'config';
5
+ import mime from 'mime';
6
+
7
+ import { GitObjectNotFoundError } from '../../git/index.js';
8
+ import { getCollection } from '../collection/index.js';
9
+ import { ExtractDocumentError } from '../extract/index.js';
10
+ import { FetchDocumentError } from '../fetcher/index.js';
11
+ import * as declaredServices from '../services/index.js';
12
+ import Service from '../services/service.js';
13
+
14
+ import { MissingCollectionIdError, UnreadableRunError } from './errors.js';
15
+ import TrackingResultsRecorder from './recorder.js';
16
+ import TrackingResultsRepository from './repository.js';
17
+ import { STATUSES } from './terms-result/index.js';
18
+
19
+ export { MissingCollectionIdError } from './errors.js';
20
+ export { RUN_ID_TRAILER_KEY } from './recorder.js';
21
+
22
+ const require = createRequire(import.meta.url);
23
+ const { version: PACKAGE_VERSION } = require('../../../package.json');
24
+
25
+ export default class TrackingResults extends events.EventEmitter {
26
+ static async create(trackingResultsConfig) { // Resolves the engine-side wiring (collection identity, schedule, engine version) so the caller does not need to know it
27
+ if (trackingResultsConfig.storage.type !== 'git') { // Git is the only supported backend, as the audit trail relies on its tamper-evident properties
28
+ throw new Error(`Unsupported tracking-results storage type "${trackingResultsConfig.storage.type}"; only "git" is supported`);
29
+ }
30
+
31
+ const collection = await getCollection();
32
+ const collectionId = collection.metadata?.id; // getCollection always resolves to a Collection instance; metadata stays undefined when the metadata file is absent, which is exactly the missing-id case reported below
33
+
34
+ if (!collectionId) {
35
+ throw new MissingCollectionIdError('Collection metadata "id" is required to record tracking-results, as it identifies the collection in every persisted run. Add an "id" field to the collection metadata file.');
36
+ }
37
+
38
+ const repository = new TrackingResultsRepository(trackingResultsConfig.storage.git);
39
+ const recorder = new TrackingResultsRecorder({
40
+ repository,
41
+ collectionId,
42
+ schedule: config.get('@opentermsarchive/engine.trackingSchedule'),
43
+ engineVersion: PACKAGE_VERSION,
44
+ });
45
+
46
+ return new TrackingResults({ recorder });
47
+ }
48
+
49
+ constructor({ recorder }) {
50
+ super();
51
+ this.recorder = recorder;
52
+ this.ready = false;
53
+ }
54
+
55
+ async initialize() {
56
+ await this.ensureReady(); // Attempted as early as possible so readers of the repository see the previous run finalized without waiting for the next tracking run
57
+ }
58
+
59
+ async finalize() {
60
+ if (!this.ready) { // Nothing was recorded, and the repository may not even be initialized
61
+ return;
62
+ }
63
+
64
+ try {
65
+ await this.recorder.finalize();
66
+ } catch (error) {
67
+ this.emit('warn', { message: `Could not finalize the tracking-results repository: ${error.message}; recorded commits are kept locally and the finalization will be retried at the next run` }); // Everything is already committed locally, so a failed push or commit-graph update loses nothing: the next successful finalize pushes all accumulated commits
68
+ }
69
+ }
70
+
71
+ get hasRunInProgress() {
72
+ return Boolean(this.recorder.currentRun);
73
+ }
74
+
75
+ get currentRunId() {
76
+ return this.recorder.currentRun?.runId ?? null;
77
+ }
78
+
79
+ async getDeclarationsCommit() { // Meant to be called right before the declarations are loaded, so that the commit identifies the declarations applied by every run of this process, whatever happens to their repository afterwards
80
+ try {
81
+ const declarationsCommit = await declaredServices.getDeclarationsCommit();
82
+
83
+ if (!declarationsCommit) { // The audit trail premise (tamper-evident declarations commit) cannot be honoured
84
+ this.emit('warn', { message: 'The declarations directory is not a Git repository; tracking-results runs will not be recorded' });
85
+ }
86
+
87
+ return declarationsCommit;
88
+ } catch (error) { // An actual git failure, distinct from "declarations is not a Git repository" which is reported as null
89
+ this.emit('warn', { message: `Could not read the declarations commit: ${error.message}; tracking-results runs will not be recorded` });
90
+
91
+ return null;
92
+ }
93
+ }
94
+
95
+ async startRun({ services, declarationsCommit, selectedServicesIds, selectedTermsTypes }) {
96
+ if (!await this.ensureReady()) { // A new Start run commit would hide the crashed run's reference SHA and make its recovery impossible forever
97
+ return;
98
+ }
99
+
100
+ if (!declarationsCommit) { // Already reported by getDeclarationsCommit
101
+ return;
102
+ }
103
+
104
+ const servicesIds = Object.keys(services).sort((a, b) => a.localeCompare(b)); // Sorted so the persisted skipped list is deterministic
105
+
106
+ try {
107
+ await this.recorder.startRun({
108
+ declarationsCommit,
109
+ servicesCount: servicesIds.length, // Declared counts always cover the full declarations, whatever subset this run processes; coverage.skipped carries the difference so the coverage proof (declared = processed + skipped) stays derivable
110
+ termsCount: Service.getNumberOfTerms(services, servicesIds, []),
111
+ skippedTerms: unselectedTerms(services, servicesIds, { selectedServicesIds, selectedTermsTypes }),
112
+ });
113
+ } catch (error) {
114
+ this.emit('warn', { message: `Could not start the tracking-results run: ${error.message}; tracking-results is disabled for this run` }); // Like every other tracking-results failure, a failed run-start commit degrades the audit trail, never the tracking itself
115
+ }
116
+ }
117
+
118
+ recordSuccess(terms, { transientErrors } = {}) {
119
+ return this.record(terms, { status: STATUSES.ok, transientErrorReasons: transientErrors?.length ? categorizeReasons(transientErrors) : undefined });
120
+ }
121
+
122
+ recordFailure(terms, errors) {
123
+ return this.record(terms, { status: STATUSES.failed, reasons: categorizeReasons(errors) });
124
+ }
125
+
126
+ async record(terms, { status, reasons, transientErrorReasons }) {
127
+ if (!this.hasRunInProgress) { // No tracking-results run is active (technical upgrades, run start skipped or failed); nothing to record
128
+ return;
129
+ }
130
+
131
+ await this.recorder.recordTermsOutcome({
132
+ serviceId: terms.service.id,
133
+ termsType: terms.type,
134
+ serviceName: terms.service.name,
135
+ sourceDocuments: terms.sourceDocuments.map(sourceDocument => ({
136
+ id: sourceDocument.id,
137
+ ...sourceDocument.toPersistence(),
138
+ mimeType: normalizeMimeType(sourceDocument.mimeType),
139
+ snapshotId: sourceDocument.snapshotId ?? null,
140
+ })),
141
+ status,
142
+ reasons,
143
+ transientErrorReasons,
144
+ });
145
+ }
146
+
147
+ completeRun() {
148
+ return this.recorder.completeRun();
149
+ }
150
+
151
+ async ensureReady() { // Returns true when it is safe to write a new run-start commit: the repository is initialized and a pending in_progress run.json has been finalized. Success is memoised, and a failure is retried at the next call so a long-lived scheduled process self-heals without a restart
152
+ if (this.ready) {
153
+ return true;
154
+ }
155
+
156
+ try {
157
+ await this.recorder.initialize(); // Repeated at each attempt, as it also drops what a failed attempt may have left uncommitted
158
+
159
+ const recovered = await this.recorder.recoverCrashedRunIfAny({ getDeclaredTermsAtCommit: commit => this.declaredTermsAtCommit(commit) });
160
+
161
+ this.ready = true;
162
+
163
+ if (recovered) {
164
+ this.emit('warn', { message: `Recovered crashed run ${recovered.shortRunId}: persisted ${recovered.coverage.processed} processed terms and ${recovered.coverage.skipped.length} skipped terms before the new run starts` });
165
+ }
166
+
167
+ return true;
168
+ } catch (error) {
169
+ if (error instanceof UnreadableRunError) { // A stale or incompatible run.json cannot be recovered by construction: proceed, the next startRun will overwrite it with a valid one
170
+ this.ready = true;
171
+ this.emit('warn', { message: `Could not read previous tracking-results state: ${error.message}. The previous run.json may be from an incompatible engine version; it will be overwritten by the next run.` });
172
+
173
+ return true;
174
+ }
175
+
176
+ this.emit('warn', { message: `Could not prepare the tracking-results repository: ${error.message}; tracking-results is disabled for this run and the preparation will be retried at the next one. If this warning persists across runs, inspect or delete the repository at "${this.recorder.repository.path}" so it can be recreated` }); // An auxiliary audit trail must not prevent tracking itself, be it at initialization
177
+
178
+ return false;
179
+ }
180
+ }
181
+
182
+ async declaredTermsAtCommit(commit) {
183
+ try {
184
+ return await declaredServices.getDeclaredTermsAtCommit(commit);
185
+ } catch (error) {
186
+ if (error instanceof GitObjectNotFoundError) { // The commit is unreachable forever (shallow clone, rewritten declarations history): fall back to the currently declared terms so the crashed run can still be finalized, and surface the approximation
187
+ this.emit('warn', { message: `Declarations commit ${commit} is not reachable; crash recovery coverage falls back to the currently declared terms` });
188
+
189
+ return declaredServices.getDeclaredTerms();
190
+ }
191
+
192
+ throw error;
193
+ }
194
+ }
195
+ }
196
+
197
+ function normalizeMimeType(mimeType) { // Fetchers report the raw Content-Type header while Git snapshots report the type derived from their file extension; aligned on the latter, so that a parameter or an alias is not recorded as a MIME type change
198
+ if (!mimeType) {
199
+ return null;
200
+ }
201
+
202
+ return mime.getType(mime.getExtension(mimeType) || '') || mimeType.split(';')[0].trim().toLowerCase();
203
+ }
204
+
205
+ function categorizeReasons(errors) { // Tags each reason with [fetch], [extraction] or [internal] so that commit subjects and consumers of the persisted reasons can split by category without sniffing error types
206
+ return errors.map(error => {
207
+ if (error instanceof FetchDocumentError) {
208
+ return `[fetch] ${error.message}`;
209
+ }
210
+
211
+ if (error instanceof ExtractDocumentError) {
212
+ return `[extraction] ${error.message}`;
213
+ }
214
+
215
+ return '[internal] Unexpected engine error'; // The message of an unexpected error may expose server paths or git output, which must not enter a published history that cannot be rewritten; the details are in the logs
216
+ });
217
+ }
218
+
219
+ function unselectedTerms(services, servicesIds, { selectedServicesIds, selectedTermsTypes }) { // Enumerates the declared terms that this run will not process, so partial runs (CLI-filtered services or terms types) record them as explicitly skipped
220
+ const skipped = [];
221
+
222
+ for (const serviceId of servicesIds) {
223
+ const selectedTypes = new Set(selectedServicesIds.includes(serviceId) ? services[serviceId].getTermsTypes(selectedTermsTypes) : []);
224
+
225
+ for (const termsType of services[serviceId].getTermsTypes()) {
226
+ if (!selectedTypes.has(termsType)) {
227
+ skipped.push({ serviceId, termsType, reason: 'not selected for this run' });
228
+ }
229
+ }
230
+ }
231
+
232
+ return skipped;
233
+ }