@opentermsarchive/engine 15.3.0 → 16.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,395 @@
1
+ import { createHash } from 'crypto';
2
+ import fsApi from 'fs';
3
+ import fs from 'fs/promises';
4
+ import { Readable } from 'stream';
5
+
6
+ import { expect } from 'chai';
7
+ import config from 'config';
8
+ import sinon from 'sinon';
9
+ import supertest from 'supertest';
10
+
11
+ import DatasetStorage from '../../dataset/storage.js';
12
+ import app from '../server.js';
13
+
14
+ import { NO_DATASET_ERROR } from './dataset.js';
15
+
16
+ const basePath = config.get('@opentermsarchive/engine.collection-api.basePath');
17
+ const request = supertest(app);
18
+
19
+ function binaryParser(res, callback) { // superagent only buffers text, JSON and media types on its own, so ZIP bodies have to be collected explicitly
20
+ const chunks = [];
21
+
22
+ res.on('data', chunk => chunks.push(chunk));
23
+ res.on('end', () => callback(null, Buffer.concat(chunks)));
24
+ }
25
+
26
+ const METADATA_URL = `${basePath}/v1/dataset/latest`;
27
+ const DOWNLOAD_URL = `${basePath}/v1/dataset/latest/download`;
28
+ const ARCHIVE_FILENAME = 'sandbox-2026-01-01.zip';
29
+ const ARCHIVE_CONTENT = Buffer.from('Archive content standing for a ZIP file in tests');
30
+ const METADATA = {
31
+ filename: ARCHIVE_FILENAME,
32
+ title: 'sandbox',
33
+ license: 'ODbL-1.0',
34
+ releaseDate: '2026-01-01T08:30:12Z',
35
+ firstVersionDate: '2021-01-01T11:27:00Z',
36
+ lastVersionDate: '2022-01-06T11:32:47Z',
37
+ servicesCount: 2,
38
+ termsCount: 3,
39
+ versionsCount: 4,
40
+ size: ARCHIVE_CONTENT.length,
41
+ sha256: createHash('sha256').update(ARCHIVE_CONTENT).digest('hex'),
42
+ };
43
+
44
+ describe('Dataset API', () => {
45
+ const storage = new DatasetStorage(config.get('@opentermsarchive/engine.dataset.storagePath'));
46
+
47
+ async function storeDataset() {
48
+ await fs.mkdir(storage.path, { recursive: true });
49
+ await fs.writeFile(storage.archivePath(ARCHIVE_FILENAME), ARCHIVE_CONTENT);
50
+ await storage.save(METADATA);
51
+ }
52
+
53
+ async function removeDataset() {
54
+ await fs.rm(storage.path, { recursive: true, force: true });
55
+ }
56
+
57
+ function itRespondsWithNoDatasetError(getResponse) {
58
+ it('responds with 404 status code', () => {
59
+ expect(getResponse().status).to.equal(404);
60
+ });
61
+
62
+ it('responds with Content-Type application/json', () => {
63
+ expect(getResponse().type).to.equal('application/json');
64
+ });
65
+
66
+ it('returns an explicit error message', () => {
67
+ expect(getResponse().body).to.deep.equal({ error: NO_DATASET_ERROR });
68
+ });
69
+ }
70
+
71
+ describe('GET /dataset/latest', () => {
72
+ let response;
73
+
74
+ context('when no dataset has been generated', () => {
75
+ before(async () => {
76
+ await removeDataset();
77
+ response = await request.get(METADATA_URL);
78
+ });
79
+
80
+ itRespondsWithNoDatasetError(() => response);
81
+ });
82
+
83
+ context('when the archive described by the metadata is missing', () => {
84
+ before(async () => {
85
+ await storeDataset();
86
+ await fs.rm(storage.archivePath(ARCHIVE_FILENAME));
87
+ response = await request.get(METADATA_URL);
88
+ });
89
+
90
+ after(removeDataset);
91
+
92
+ itRespondsWithNoDatasetError(() => response);
93
+ });
94
+
95
+ context('when a dataset exists', () => {
96
+ before(async () => {
97
+ await storeDataset();
98
+ response = await request.get(METADATA_URL);
99
+ });
100
+
101
+ after(removeDataset);
102
+
103
+ it('responds with 200 status code', () => {
104
+ expect(response.status).to.equal(200);
105
+ });
106
+
107
+ it('responds with Content-Type application/json', () => {
108
+ expect(response.type).to.equal('application/json');
109
+ });
110
+
111
+ it('returns the dataset metadata along with its download URL', () => {
112
+ const { host } = new URL(response.request.url);
113
+
114
+ expect(response.body).to.deep.equal({
115
+ ...METADATA,
116
+ downloadURL: `http://${host}${DOWNLOAD_URL}`,
117
+ });
118
+ });
119
+ });
120
+
121
+ context('behind a reverse proxy', () => {
122
+ before(async () => {
123
+ await storeDataset();
124
+ response = await request
125
+ .get(METADATA_URL)
126
+ .set('X-Forwarded-Proto', 'https')
127
+ .set('X-Forwarded-Host', 'api.example.com');
128
+ });
129
+
130
+ after(removeDataset);
131
+
132
+ it('uses the forwarded protocol and host in the download URL', () => {
133
+ expect(response.body.downloadURL).to.equal(`https://api.example.com${DOWNLOAD_URL}`);
134
+ });
135
+ });
136
+
137
+ context('behind a chain of reverse proxies', () => {
138
+ before(async () => {
139
+ await storeDataset();
140
+ response = await request
141
+ .get(METADATA_URL)
142
+ .set('X-Forwarded-Proto', 'https')
143
+ .set('X-Forwarded-Host', 'api.example.com, edge.internal');
144
+ });
145
+
146
+ after(removeDataset);
147
+
148
+ it('uses the first host in the forwarded list in the download URL', () => {
149
+ expect(response.body.downloadURL).to.equal(`https://api.example.com${DOWNLOAD_URL}`);
150
+ });
151
+ });
152
+
153
+ context('when the stored metadata is corrupted', () => {
154
+ before(async () => {
155
+ await fs.mkdir(storage.path, { recursive: true });
156
+ await fs.writeFile(storage.metadataPath, '{ not json');
157
+ response = await request.get(METADATA_URL);
158
+ });
159
+
160
+ after(removeDataset);
161
+
162
+ it('responds with 500 status code', () => {
163
+ expect(response.status).to.equal(500);
164
+ });
165
+
166
+ it('responds with Content-Type application/json', () => {
167
+ expect(response.type).to.equal('application/json');
168
+ });
169
+
170
+ it('returns a generic error message', () => {
171
+ expect(response.body).to.deep.equal({ error: 'Internal Server Error' });
172
+ });
173
+ });
174
+ });
175
+
176
+ describe('GET /dataset/latest/download', () => {
177
+ let response;
178
+
179
+ context('when no dataset has been generated', () => {
180
+ before(async () => {
181
+ await removeDataset();
182
+ response = await request.get(DOWNLOAD_URL);
183
+ });
184
+
185
+ itRespondsWithNoDatasetError(() => response);
186
+ });
187
+
188
+ context('when the archive described by the metadata is missing', () => {
189
+ before(async () => {
190
+ await storeDataset();
191
+ await fs.rm(storage.archivePath(ARCHIVE_FILENAME));
192
+ response = await request.get(DOWNLOAD_URL);
193
+ });
194
+
195
+ after(removeDataset);
196
+
197
+ itRespondsWithNoDatasetError(() => response);
198
+ });
199
+
200
+ context('when a dataset exists', () => {
201
+ before(storeDataset);
202
+
203
+ after(removeDataset);
204
+
205
+ describe('without conditions', () => {
206
+ before(async () => {
207
+ response = await request.get(DOWNLOAD_URL).buffer(true).parse(binaryParser);
208
+ });
209
+
210
+ it('responds with 200 status code', () => {
211
+ expect(response.status).to.equal(200);
212
+ });
213
+
214
+ it('responds with Content-Type application/zip', () => {
215
+ expect(response.type).to.equal('application/zip');
216
+ });
217
+
218
+ it('exposes the archive as an attachment named after the archive file', () => {
219
+ expect(response.headers['content-disposition']).to.equal(`attachment; filename="${ARCHIVE_FILENAME}"`);
220
+ });
221
+
222
+ it('exposes the archive size as Content-Length', () => {
223
+ expect(response.headers['content-length']).to.equal(String(ARCHIVE_CONTENT.length));
224
+ });
225
+
226
+ it('exposes the archive checksum as a strong ETag', () => {
227
+ expect(response.headers.etag).to.equal(`"${METADATA.sha256}"`);
228
+ });
229
+
230
+ it('exposes the release date as Last-Modified', () => {
231
+ expect(response.headers['last-modified']).to.equal(new Date(METADATA.releaseDate).toUTCString());
232
+ });
233
+
234
+ it('advertises byte range support', () => {
235
+ expect(response.headers['accept-ranges']).to.equal('bytes');
236
+ });
237
+
238
+ it('returns the archive content', () => {
239
+ expect(response.body.equals(ARCHIVE_CONTENT)).to.be.true;
240
+ });
241
+ });
242
+
243
+ describe('with a conditional request', () => {
244
+ it('returns 304 with no body when If-None-Match matches the archive checksum', async () => {
245
+ const conditionalResponse = await request.get(DOWNLOAD_URL).set('If-None-Match', `"${METADATA.sha256}"`);
246
+
247
+ expect(conditionalResponse.status).to.equal(304);
248
+ expect(conditionalResponse.text).to.be.empty;
249
+ });
250
+
251
+ it('returns 200 with the archive when If-None-Match does not match', async () => {
252
+ const conditionalResponse = await request.get(DOWNLOAD_URL).set('If-None-Match', '"another-checksum"');
253
+
254
+ expect(conditionalResponse.status).to.equal(200);
255
+ });
256
+
257
+ it('returns 304 with no body when If-Modified-Since is at or after the release date', async () => {
258
+ const conditionalResponse = await request.get(DOWNLOAD_URL).set('If-Modified-Since', new Date(METADATA.releaseDate).toUTCString());
259
+
260
+ expect(conditionalResponse.status).to.equal(304);
261
+ expect(conditionalResponse.text).to.be.empty;
262
+ });
263
+
264
+ it('returns 200 with the archive when If-Modified-Since is before the release date', async () => {
265
+ const conditionalResponse = await request.get(DOWNLOAD_URL).set('If-Modified-Since', new Date('2025-12-31T00:00:00Z').toUTCString());
266
+
267
+ expect(conditionalResponse.status).to.equal(200);
268
+ });
269
+
270
+ it('returns a JSON error with 412 status code when If-Match does not match the archive checksum', async () => {
271
+ const conditionalResponse = await request.get(DOWNLOAD_URL).set('If-Match', '"another-checksum"');
272
+
273
+ expect(conditionalResponse.status).to.equal(412);
274
+ expect(conditionalResponse.type).to.equal('application/json');
275
+ expect(conditionalResponse.headers).to.not.have.any.keys('content-disposition', 'last-modified');
276
+ expect(conditionalResponse.body).to.deep.equal({ error: 'Precondition Failed' });
277
+ });
278
+ });
279
+
280
+ describe('with a range request', () => {
281
+ it('returns the requested bytes with 206 status code', async () => {
282
+ const rangeResponse = await request.get(DOWNLOAD_URL).set('Range', 'bytes=0-4').buffer(true).parse(binaryParser);
283
+
284
+ expect(rangeResponse.status).to.equal(206);
285
+ expect(rangeResponse.headers['content-range']).to.equal(`bytes 0-4/${ARCHIVE_CONTENT.length}`);
286
+ expect(rangeResponse.body.equals(ARCHIVE_CONTENT.subarray(0, 5))).to.be.true;
287
+ });
288
+
289
+ it('returns a JSON error with 416 status code when the range cannot be satisfied', async () => {
290
+ const rangeResponse = await request.get(DOWNLOAD_URL).set('Range', `bytes=${ARCHIVE_CONTENT.length}-`);
291
+
292
+ expect(rangeResponse.status).to.equal(416);
293
+ expect(rangeResponse.type).to.equal('application/json');
294
+ expect(rangeResponse.headers['content-range']).to.equal(`bytes */${ARCHIVE_CONTENT.length}`);
295
+ expect(rangeResponse.headers).to.not.have.any.keys('content-disposition', 'last-modified');
296
+ expect(rangeResponse.headers.etag).to.not.equal(`"${METADATA.sha256}"`);
297
+ expect(rangeResponse.body).to.deep.equal({ error: 'Range Not Satisfiable' });
298
+ });
299
+
300
+ it('returns 200 with the full archive when If-Range does not match the current archive', async () => {
301
+ const rangeResponse = await request.get(DOWNLOAD_URL).set('If-Range', '"stale-checksum"').set('Range', 'bytes=0-4').buffer(true)
302
+ .parse(binaryParser);
303
+
304
+ expect(rangeResponse.status).to.equal(200);
305
+ expect(rangeResponse.body.equals(ARCHIVE_CONTENT)).to.be.true;
306
+ });
307
+ });
308
+
309
+ describe('with a HEAD request', () => {
310
+ it('returns the archive headers without its content', async () => {
311
+ const headResponse = await request.head(DOWNLOAD_URL);
312
+
313
+ expect(headResponse.status).to.equal(200);
314
+ expect(headResponse.headers['content-length']).to.equal(String(ARCHIVE_CONTENT.length));
315
+ expect(headResponse.headers.etag).to.equal(`"${METADATA.sha256}"`);
316
+ expect(headResponse.text).to.be.oneOf([ undefined, '' ]);
317
+ });
318
+ });
319
+ });
320
+
321
+ context('when the archive file name starts with a dot', () => {
322
+ const DOTFILE_ARCHIVE_FILENAME = '.weekly-2026-01-01.zip';
323
+
324
+ before(async () => {
325
+ await fs.mkdir(storage.path, { recursive: true });
326
+ await fs.writeFile(storage.archivePath(DOTFILE_ARCHIVE_FILENAME), ARCHIVE_CONTENT);
327
+ await storage.save({ ...METADATA, filename: DOTFILE_ARCHIVE_FILENAME });
328
+ response = await request.get(DOWNLOAD_URL).buffer(true).parse(binaryParser);
329
+ });
330
+
331
+ after(removeDataset);
332
+
333
+ it('responds with 200 status code', () => {
334
+ expect(response.status).to.equal(200);
335
+ });
336
+
337
+ it('returns the archive content', () => {
338
+ expect(response.body.equals(ARCHIVE_CONTENT)).to.be.true;
339
+ });
340
+ });
341
+
342
+ context('when the archive cannot be read', () => {
343
+ before(async () => {
344
+ await storeDataset();
345
+ sinon.stub(fsApi, 'createReadStream').returns(new Readable({ read() { this.destroy(new Error('Disk failure')); } })); // `send` opens the archive through the `fs` module at transfer time, once it has already described the archive on the response
346
+ response = await request.get(DOWNLOAD_URL);
347
+ });
348
+
349
+ after(async () => {
350
+ sinon.restore();
351
+ await removeDataset();
352
+ });
353
+
354
+ it('responds with 500 status code', () => {
355
+ expect(response.status).to.equal(500);
356
+ });
357
+
358
+ it('responds with Content-Type application/json', () => {
359
+ expect(response.type).to.equal('application/json');
360
+ });
361
+
362
+ it('returns a generic error message', () => {
363
+ expect(response.body).to.deep.equal({ error: 'Internal Server Error' });
364
+ });
365
+
366
+ it('does not describe the archive', () => {
367
+ expect(response.headers).to.not.have.any.keys('content-disposition', 'last-modified', 'accept-ranges');
368
+ expect(response.headers.etag).to.not.equal(`"${METADATA.sha256}"`);
369
+ });
370
+ });
371
+
372
+ context('when the archive is replaced between the metadata lookup and the transfer', () => {
373
+ before(async () => {
374
+ await storeDataset();
375
+ sinon.stub(DatasetStorage.prototype, 'findLatest').resolves({ ...METADATA, filename: 'sandbox-2026-01-02.zip' }); // The metadata describes an archive that is no longer on disk, as when a generation completes right after the lookup
376
+ response = await request.get(DOWNLOAD_URL);
377
+ });
378
+
379
+ after(async () => {
380
+ sinon.restore();
381
+ await removeDataset();
382
+ });
383
+
384
+ itRespondsWithNoDatasetError(() => response);
385
+ });
386
+ });
387
+
388
+ describe('GET /dataset', () => {
389
+ it('responds with 404 status code', async () => {
390
+ const response = await request.get(`${basePath}/v1/dataset`);
391
+
392
+ expect(response.status).to.equal(404);
393
+ });
394
+ });
395
+ });
@@ -54,6 +54,14 @@ describe('Docs API', () => {
54
54
  it('/version/{serviceId}/{termsType}/{date}', () => {
55
55
  expect(subject).to.have.property('/version/{serviceId}/{termsType}/{date}');
56
56
  });
57
+
58
+ it('/dataset/latest', () => {
59
+ expect(subject).to.have.property('/dataset/latest');
60
+ });
61
+
62
+ it('/dataset/latest/download', () => {
63
+ expect(subject).to.have.property('/dataset/latest/download');
64
+ });
57
65
  });
58
66
  });
59
67
  });
@@ -3,6 +3,7 @@ import { js2xml } from 'xml-js';
3
3
 
4
4
  import { getCollection } from '../../archivist/collection/index.js';
5
5
  import { toISODateWithoutMilliseconds } from '../../archivist/utils/date.js';
6
+ import { buildAbsoluteBaseUrl } from '../utils/url.js';
6
7
 
7
8
  const RECORD_TYPES = {
8
9
  firstRecord: 'First record',
@@ -18,12 +19,6 @@ const SCHEMES = Object.freeze({
18
19
  recordType: `tag:${TAG_AUTHORITY}:scheme:record-type`,
19
20
  });
20
21
 
21
- function buildAbsoluteBaseUrl(req) {
22
- const host = req.get('X-Forwarded-Host') ?? req.get('host'); // Behind a trusted reverse proxy, the public host comes from X-Forwarded-Host. req.get('host') only sees the internal Host header, so we read the forwarded value explicitly and fall back to the direct host for non-proxied setups (dev, tests).
23
-
24
- return `${req.protocol}://${host}${req.baseUrl}`;
25
- }
26
-
27
22
  function classifyRecordType(version) {
28
23
  return version.isFirstRecord ? RECORD_TYPES.firstRecord : RECORD_TYPES.change;
29
24
  }
@@ -5,7 +5,9 @@ import helmet from 'helmet';
5
5
  import { getCollection } from '../../archivist/collection/index.js';
6
6
  import RepositoryFactory from '../../archivist/recorder/repositories/factory.js';
7
7
  import * as Services from '../../archivist/services/index.js';
8
+ import DatasetStorage from '../../dataset/storage.js';
8
9
 
10
+ import datasetRouter from './dataset.js';
9
11
  import docsRouter from './docs.js';
10
12
  import feedRouter from './feed.js';
11
13
  import metadataRouter from './metadata.js';
@@ -40,6 +42,7 @@ export default async function apiRouter(basePath) {
40
42
  const versionsRepository = await RepositoryFactory.create(versionsStorageConfig).initialize();
41
43
  const snapshotsRepository = await RepositoryFactory.create(config.get('@opentermsarchive/engine.recorder.snapshots.storage')).initialize();
42
44
  const feedConfig = config.get('@opentermsarchive/engine.collection-api.feed');
45
+ const datasetStorage = new DatasetStorage(config.get('@opentermsarchive/engine.dataset.storagePath'));
43
46
 
44
47
  if (!collection.metadata?.id) {
45
48
  throw new Error('Collection metadata "id" is required to expose feed endpoints, as it is used to build the tag URIs that uniquely identify the feed and its entries. Add an "id" field to the collection metadata file.');
@@ -53,6 +56,7 @@ export default async function apiRouter(basePath) {
53
56
  router.use(servicesRouter(services));
54
57
  router.use(versionsRouter(versionsRepository, snapshotsRepository));
55
58
  router.use(feedRouter(services, versionsRepository, versionsStorageConfig.type, feedConfig.limit, feedConfig.versionUrlTemplate));
59
+ router.use(datasetRouter(datasetStorage));
56
60
 
57
61
  return router;
58
62
  }
@@ -0,0 +1,3 @@
1
+ export function buildAbsoluteBaseUrl(req) {
2
+ return `${req.protocol}://${req.host}${req.baseUrl}`;
3
+ }
@@ -0,0 +1,4 @@
1
+ export const SPDX_ID = 'ODbL-1.0'; // Identifier of the license shipped in scripts/dataset/assets/LICENSE
2
+ export const DATAGOUV_ID = 'odc-odbl'; // data.gouv.fr's own identifier for the same license, distinct from the SPDX scheme
3
+ export const DISPLAY_NAME = 'Open Database (ODbL) License';
4
+ export const INFO_URL = 'https://opendatacommons.org/licenses/odbl/1.0/';
@@ -0,0 +1,66 @@
1
+ import fs from 'fs/promises';
2
+ import path from 'path';
3
+
4
+ export const TEMPORARY_SUFFIX = '.tmp';
5
+
6
+ const METADATA_FILENAME = 'metadata.json';
7
+ const ARCHIVE_EXTENSION = '.zip';
8
+
9
+ export default class DatasetStorage {
10
+ constructor(storagePath) {
11
+ this.path = path.resolve(process.cwd(), storagePath);
12
+ }
13
+
14
+ get metadataPath() {
15
+ return path.join(this.path, METADATA_FILENAME);
16
+ }
17
+
18
+ archivePath(filename) {
19
+ return path.join(this.path, filename);
20
+ }
21
+
22
+ async findLatest() {
23
+ let metadata;
24
+
25
+ try {
26
+ metadata = JSON.parse(await fs.readFile(this.metadataPath, 'utf8'));
27
+ await fs.access(this.archivePath(metadata.filename));
28
+ } catch (error) {
29
+ if (error.code === 'ENOENT') {
30
+ return null;
31
+ }
32
+
33
+ throw error;
34
+ }
35
+
36
+ return metadata;
37
+ }
38
+
39
+ async save(metadata) {
40
+ await fs.access(this.archivePath(metadata.filename)); // Never describe an archive that is not in place
41
+
42
+ const temporaryPath = `${this.metadataPath}${TEMPORARY_SUFFIX}`;
43
+
44
+ await fs.writeFile(temporaryPath, JSON.stringify(metadata, null, 2));
45
+ await fs.rename(temporaryPath, this.metadataPath); // Readers see either the previous or the new metadata, never a partial file
46
+ }
47
+
48
+ async removePreviousArchives() {
49
+ const metadata = await this.findLatest();
50
+
51
+ if (!metadata) {
52
+ return;
53
+ }
54
+
55
+ await this.#removeArchivesExcept(metadata.filename);
56
+ }
57
+
58
+ async #removeArchivesExcept(filename) {
59
+ const entries = await fs.readdir(this.path);
60
+ const isArchiveOrLeftover = entry => entry.endsWith(ARCHIVE_EXTENSION) || entry.endsWith(`${ARCHIVE_EXTENSION}${TEMPORARY_SUFFIX}`); // Cleanup is restricted to archives and their leftovers: the storage path is user-provided and may hold unrelated files
61
+
62
+ await Promise.all(entries
63
+ .filter(entry => entry !== filename && isArchiveOrLeftover(entry))
64
+ .map(entry => fs.rm(this.archivePath(entry), { force: true })));
65
+ }
66
+ }