@opentermsarchive/engine 15.3.0 → 16.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,32 @@
1
+ import { expect, use } from 'chai';
2
+ import sinon from 'sinon';
3
+ import sinonChai from 'sinon-chai';
4
+
5
+ import logger, { createModuleLogger } from './index.js';
6
+
7
+ use(sinonChai);
8
+
9
+ describe('Dataset logger', () => {
10
+ describe('#createModuleLogger', () => {
11
+ const MODULE_NAME = 'test-module';
12
+ let moduleLogger;
13
+
14
+ before(() => {
15
+ moduleLogger = createModuleLogger(MODULE_NAME);
16
+ });
17
+
18
+ afterEach(() => {
19
+ sinon.restore();
20
+ });
21
+
22
+ [ 'debug', 'info', 'warn', 'error' ].forEach(level => {
23
+ it(`forwards ${level} messages to the logger with the module name`, () => {
24
+ const stub = sinon.stub(logger, level);
25
+
26
+ moduleLogger[level]('message');
27
+
28
+ expect(stub).to.have.been.calledOnceWithExactly('message', { module: MODULE_NAME });
29
+ });
30
+ });
31
+ });
32
+ });
@@ -8,7 +8,6 @@ import { createModuleLogger } from '../../logger/index.js';
8
8
 
9
9
  const logger = createModuleLogger('datagouv');
10
10
 
11
- const DATASET_LICENSE = 'odc-odbl';
12
11
  const DEFAULT_RESOURCE_DESCRIPTION = 'See README.md inside the archive for dataset structure and usage information.';
13
12
 
14
13
  const routes = {
@@ -112,11 +111,11 @@ export async function createDataset({ apiBaseUrl, headers, organizationId, title
112
111
  return dataset;
113
112
  }
114
113
 
115
- export async function updateDatasetMetadata({ apiBaseUrl, headers, datasetId, title, description, stats, frequency }) {
114
+ export async function updateDatasetMetadata({ apiBaseUrl, headers, datasetId, title, description, license, stats, frequency }) {
116
115
  const updatePayload = {
117
116
  title,
118
117
  description,
119
- license: DATASET_LICENSE,
118
+ license,
120
119
  frequency,
121
120
  };
122
121
 
@@ -1,5 +1,6 @@
1
1
  import config from 'config';
2
2
 
3
+ import * as license from '../../../../src/dataset/license.js';
3
4
  import * as readme from '../../assets/README.template.js';
4
5
  import { createModuleLogger } from '../../logger/index.js';
5
6
 
@@ -9,7 +10,6 @@ const logger = createModuleLogger('datagouv');
9
10
 
10
11
  const PRODUCTION_API_BASE_URL = 'https://www.data.gouv.fr/api/1';
11
12
  const DEMO_API_BASE_URL = 'https://demo.data.gouv.fr/api/1';
12
- const DATASET_LICENSE = 'odc-odbl';
13
13
 
14
14
  export default async function publish({ archivePath, stats }) {
15
15
  const { datasetId, organizationIdOrSlug, apiBaseUrl, headers, datasetTitle, frequency } = loadConfiguration();
@@ -19,7 +19,7 @@ export default async function publish({ archivePath, stats }) {
19
19
  ? await getDataset({ apiBaseUrl, headers, datasetId })
20
20
  : await ensureDatasetExists({ apiBaseUrl, headers, organizationIdOrSlug, datasetTitle, description, frequency });
21
21
 
22
- await updateDatasetMetadata({ apiBaseUrl, headers, datasetId: dataset.id, title: datasetTitle, description, stats, frequency });
22
+ await updateDatasetMetadata({ apiBaseUrl, headers, datasetId: dataset.id, title: datasetTitle, description, license: license.DATAGOUV_ID, stats, frequency });
23
23
 
24
24
  const { resourceId, fileName } = await handleResourceUpload({ apiBaseUrl, headers, datasetId: dataset.id, dataset, archivePath });
25
25
 
@@ -73,7 +73,7 @@ async function ensureDatasetExists({ apiBaseUrl, headers, organizationIdOrSlug,
73
73
  let dataset = await findDatasetByTitle({ apiBaseUrl, headers, organizationId: organization.id, title: datasetTitle });
74
74
 
75
75
  if (!dataset) {
76
- dataset = await createDataset({ apiBaseUrl, headers, organizationId: organization.id, title: datasetTitle, description, license: DATASET_LICENSE, frequency });
76
+ dataset = await createDataset({ apiBaseUrl, headers, organizationId: organization.id, title: datasetTitle, description, license: license.DATAGOUV_ID, frequency });
77
77
  }
78
78
 
79
79
  return dataset;
@@ -51,7 +51,8 @@ export default async function publish({
51
51
  projectId = null;
52
52
  }
53
53
 
54
- const tagName = `${path.basename(archivePath, path.extname(archivePath))}`; // use archive filename as Git tag
54
+ const archiveFilename = path.basename(archivePath);
55
+ const tagName = path.basename(archiveFilename, path.extname(archiveFilename)); // use archive filename as Git tag
55
56
 
56
57
  try {
57
58
  let options = GitLab.baseOptionsHttpReq(process.env.OTA_ENGINE_GITLAB_RELEASES_TOKEN);
@@ -88,7 +89,7 @@ export default async function publish({
88
89
  // restrict characters to the ones allowed by GitLab APIs
89
90
  const packageName = config.get('@opentermsarchive/engine.dataset.title').replace(/[^a-zA-Z0-9.\-_]/g, '-');
90
91
  const packageVersion = tagName.replace(/[^a-zA-Z0-9.\-_]/g, '-');
91
- const packageFileName = archivePath.replace(/[^a-zA-Z0-9.\-_/]/g, '-');
92
+ const packageFileName = archiveFilename.replace(/[^a-zA-Z0-9.\-_]/g, '-');
92
93
 
93
94
  logger.debug(`packageName: ${packageName}, packageVersion: ${packageVersion} packageFileName: ${packageFileName}`);
94
95
 
@@ -108,9 +109,9 @@ export default async function publish({
108
109
  // Create the release and link the package
109
110
  const formData = new FormData();
110
111
 
111
- formData.append('name', archivePath);
112
+ formData.append('name', archiveFilename);
112
113
  formData.append('url', publishedPackageUrl);
113
- formData.append('file', fsApi.createReadStream(archivePath), { filename: path.basename(archivePath) });
114
+ formData.append('file', fsApi.createReadStream(archivePath), { filename: archiveFilename });
114
115
 
115
116
  options = GitLab.baseOptionsHttpReq(process.env.OTA_ENGINE_GITLAB_RELEASES_TOKEN);
116
117
  options.method = 'POST';
@@ -0,0 +1,90 @@
1
+ import fs from 'fs/promises';
2
+ import path from 'path';
3
+ import { fileURLToPath } from 'url';
4
+
5
+ import { expect } from 'chai';
6
+ import config from 'config';
7
+ import nock from 'nock';
8
+
9
+ import publish from './index.js';
10
+
11
+ const __dirname = path.dirname(fileURLToPath(import.meta.url));
12
+
13
+ const { origin: API_ORIGIN, pathname: API_PATH } = new URL(config.get('@opentermsarchive/engine.dataset.apiBaseURL'));
14
+ const PROJECT_PATH = 'OpenTermsArchive/sandbox-versions';
15
+ const PROJECT_ID = 1;
16
+ const ARCHIVE_NAME = 'sandbox-2026-01-01';
17
+ const ARCHIVE_FILENAME = `${ARCHIVE_NAME}.zip`;
18
+ const TMP_PATH = path.resolve(__dirname, './tmp');
19
+ const ARCHIVE_PATH = path.join(TMP_PATH, ARCHIVE_FILENAME);
20
+ const DIRECT_ASSET_URL = 'https://gitlab.example.test/download/42';
21
+ const STATS = {
22
+ servicesCount: 2,
23
+ firstVersionDate: new Date('2021-01-01T00:00:00Z'),
24
+ lastVersionDate: new Date('2022-01-06T00:00:00Z'),
25
+ };
26
+
27
+ describe('GitLab dataset publisher', () => {
28
+ describe('#publish', () => {
29
+ let packageUploadScope;
30
+ let assetLinkScope;
31
+ let result;
32
+ let previousToken;
33
+
34
+ before(async function () {
35
+ this.timeout(5000);
36
+ previousToken = process.env.OTA_ENGINE_GITLAB_RELEASES_TOKEN;
37
+ process.env.OTA_ENGINE_GITLAB_RELEASES_TOKEN = 'token';
38
+
39
+ await fs.mkdir(TMP_PATH, { recursive: true });
40
+ await fs.writeFile(ARCHIVE_PATH, 'archive content'); // Plain text keeps the multipart body inspectable: nock hands binary bodies to matchers as hex strings
41
+
42
+ nock(API_ORIGIN)
43
+ .get(`${API_PATH}/projects/${encodeURIComponent(PROJECT_PATH)}`)
44
+ .reply(200, { id: PROJECT_ID });
45
+
46
+ nock(API_ORIGIN)
47
+ .post(`${API_PATH}/projects/${PROJECT_ID}/releases`)
48
+ .reply(201, { commit: { id: 'sha' } });
49
+
50
+ packageUploadScope = nock(API_ORIGIN)
51
+ .put(`${API_PATH}/projects/${PROJECT_ID}/packages/generic/sandbox/${ARCHIVE_NAME}/${ARCHIVE_FILENAME}`)
52
+ .query({ status: 'default', select: 'package_file' })
53
+ .reply(201, { id: 42 });
54
+
55
+ assetLinkScope = nock(API_ORIGIN)
56
+ .post(`${API_PATH}/projects/${PROJECT_ID}/releases/${ARCHIVE_NAME}/assets/links`, body => body.includes(`name="name"\r\n\r\n${ARCHIVE_FILENAME}\r\n`))
57
+ .reply(201, { direct_asset_url: DIRECT_ASSET_URL });
58
+
59
+ result = await publish({
60
+ archivePath: ARCHIVE_PATH,
61
+ releaseDate: new Date('2026-01-01T08:30:00Z'),
62
+ stats: STATS,
63
+ });
64
+ });
65
+
66
+ after(async () => {
67
+ nock.cleanAll();
68
+
69
+ if (previousToken === undefined) {
70
+ delete process.env.OTA_ENGINE_GITLAB_RELEASES_TOKEN;
71
+ } else {
72
+ process.env.OTA_ENGINE_GITLAB_RELEASES_TOKEN = previousToken;
73
+ }
74
+
75
+ await fs.rm(TMP_PATH, { recursive: true, force: true });
76
+ });
77
+
78
+ it('uploads the package under the archive file name', () => {
79
+ expect(packageUploadScope.isDone()).to.be.true;
80
+ });
81
+
82
+ it('links the release asset under the archive file name', () => {
83
+ expect(assetLinkScope.isDone()).to.be.true;
84
+ });
85
+
86
+ it('returns the direct asset URL', () => {
87
+ expect(result).to.equal(DIRECT_ASSET_URL);
88
+ });
89
+ });
90
+ });
@@ -97,7 +97,7 @@ export default async options => {
97
97
  }
98
98
  });
99
99
 
100
- if (!schemaOnly && service) {
100
+ if (service) {
101
101
  service.getTermsTypes()
102
102
  .filter(termsType => {
103
103
  if (!service.terms[termsType]?.latest) { // If this terms type has been deleted and there is only a historical record for it, but no current valid declaration
@@ -126,6 +126,10 @@ export default async options => {
126
126
  });
127
127
  }
128
128
 
129
+ if (schemaOnly) {
130
+ return; // Remaining checks require fetching the source documents
131
+ }
132
+
129
133
  terms.sourceDocuments.forEach(sourceDocument => {
130
134
  let filteredContent;
131
135
 
@@ -2,5 +2,10 @@ import logger from '../logger.js';
2
2
 
3
3
  export default function errorsMiddleware(err, req, res, next) {
4
4
  logger.error(err.stack);
5
+
6
+ if (res.headersSent) { // A response that already started cannot be restarted with a JSON error; defer to Express' default handler, which ends the request instead of leaving the client waiting
7
+ return next(err);
8
+ }
9
+
5
10
  res.status(500).json({ error: 'Internal Server Error' }); // Never echo internal error details: they can expose server internals such as filesystem paths
6
11
  }
@@ -0,0 +1,56 @@
1
+ import { expect, use } from 'chai';
2
+ import sinon from 'sinon';
3
+ import sinonChai from 'sinon-chai';
4
+
5
+ import errorsMiddleware from './errors.js';
6
+
7
+ use(sinonChai);
8
+
9
+ describe('Errors middleware', () => {
10
+ const error = new Error('Boom');
11
+
12
+ let res;
13
+ let next;
14
+
15
+ beforeEach(() => {
16
+ res = {
17
+ status: sinon.stub().returnsThis(),
18
+ json: sinon.stub(),
19
+ };
20
+ next = sinon.stub();
21
+ });
22
+
23
+ context('when no response has been sent yet', () => {
24
+ beforeEach(() => {
25
+ res.headersSent = false;
26
+ errorsMiddleware(error, {}, res, next);
27
+ });
28
+
29
+ it('responds with 500 status code', () => {
30
+ expect(res.status).to.have.been.calledOnceWith(500);
31
+ });
32
+
33
+ it('responds with a generic JSON error', () => {
34
+ expect(res.json).to.have.been.calledOnceWith({ error: 'Internal Server Error' });
35
+ });
36
+
37
+ it('does not call next', () => {
38
+ expect(next).to.not.have.been.called;
39
+ });
40
+ });
41
+
42
+ context('when headers have already been sent', () => {
43
+ beforeEach(() => {
44
+ res.headersSent = true;
45
+ errorsMiddleware(error, {}, res, next);
46
+ });
47
+
48
+ it('forwards the error to the default Express handler instead of responding again', () => {
49
+ expect(next).to.have.been.calledOnceWith(error);
50
+ });
51
+
52
+ it('does not attempt to set a status code', () => {
53
+ expect(res.status).to.not.have.been.called;
54
+ });
55
+ });
56
+ });
@@ -0,0 +1,269 @@
1
+ import http from 'http';
2
+
3
+ import express from 'express';
4
+
5
+ import { buildAbsoluteBaseUrl } from '../utils/url.js';
6
+
7
+ export const NO_DATASET_ERROR = 'No dataset has been generated yet';
8
+
9
+ const TRANSFER_HEADERS = [ 'Accept-Ranges', 'Cache-Control', 'Content-Disposition', 'Content-Length', 'Content-Range', 'Content-Type', 'ETag', 'Last-Modified' ];
10
+ const METADATA_PATH = '/dataset/latest';
11
+ const DOWNLOAD_PATH = `${METADATA_PATH}/download`;
12
+
13
+ function handleTransferError(error, res, next) {
14
+ if (!error || error.code === 'ECONNABORTED') {
15
+ return;
16
+ }
17
+
18
+ if (res.headersSent) { // The transfer started before failing: `send` destroyed the read stream but never ended the response, so it must be forwarded to end it instead of leaving the client waiting
19
+ return next(error);
20
+ }
21
+
22
+ TRANSFER_HEADERS.forEach(header => res.removeHeader(header)); // `send` has already described the archive on the response when it fails, whether it rejects the request or cannot read the file; those headers must not describe the JSON error instead
23
+
24
+ if (error.status === 404) { // The archive was replaced under another name between the metadata lookup and the transfer; the new metadata is already in place, so a retry succeeds
25
+ return res.status(404).json({ error: NO_DATASET_ERROR });
26
+ }
27
+
28
+ if (error.status < 500) { // Client errors raised by `send`, such as an unsatisfiable range or a failed precondition
29
+ return res.status(error.status).set(error.headers ?? {}).json({ error: http.STATUS_CODES[error.status] });
30
+ }
31
+
32
+ return next(error);
33
+ }
34
+
35
+ /**
36
+ * @param {object} datasetStorage The storage holding the latest dataset of the collection
37
+ * @returns {express.Router} The router instance
38
+ * @swagger
39
+ * tags:
40
+ * name: Dataset
41
+ * description: Dataset API
42
+ * components:
43
+ * schemas:
44
+ * Dataset:
45
+ * type: object
46
+ * description: Metadata of the latest dataset generated on this instance. Counts and dates are frozen at generation time, unlike the live counters of the collection metadata.
47
+ * additionalProperties: false
48
+ * required:
49
+ * - filename
50
+ * - title
51
+ * - license
52
+ * - releaseDate
53
+ * - firstVersionDate
54
+ * - lastVersionDate
55
+ * - servicesCount
56
+ * - termsCount
57
+ * - versionsCount
58
+ * - size
59
+ * - sha256
60
+ * - downloadURL
61
+ * properties:
62
+ * filename:
63
+ * type: string
64
+ * description: Name of the archive file.
65
+ * example: demo-2026-07-06.zip
66
+ * title:
67
+ * type: string
68
+ * description: Dataset title.
69
+ * example: demo
70
+ * license:
71
+ * type: string
72
+ * description: SPDX identifier of the dataset license.
73
+ * example: ODbL-1.0
74
+ * releaseDate:
75
+ * type: string
76
+ * format: date-time
77
+ * description: Datetime when the dataset was generated.
78
+ * example: 2026-07-06T08:30:12Z
79
+ * firstVersionDate:
80
+ * type: string
81
+ * format: date-time
82
+ * description: Fetch date of the earliest version in the dataset.
83
+ * example: 2022-01-01T12:00:00Z
84
+ * lastVersionDate:
85
+ * type: string
86
+ * format: date-time
87
+ * description: Fetch date of the latest version in the dataset.
88
+ * example: 2026-07-05T23:12:00Z
89
+ * servicesCount:
90
+ * type: integer
91
+ * description: Number of services in the dataset.
92
+ * example: 262
93
+ * termsCount:
94
+ * type: integer
95
+ * description: Number of distinct terms (service and terms type pairs) in the dataset.
96
+ * example: 641
97
+ * versionsCount:
98
+ * type: integer
99
+ * description: Total number of version files in the dataset.
100
+ * example: 38294
101
+ * size:
102
+ * type: integer
103
+ * description: Size of the archive in bytes.
104
+ * example: 104857600
105
+ * sha256:
106
+ * type: string
107
+ * description: SHA-256 checksum of the archive, allowing to verify its integrity and to detect changes without downloading it.
108
+ * example: 9f86d081884c7d659a2feaa0c55ad015a3bf4f1b2b0b822cd15d6c15b0f00a08
109
+ * downloadURL:
110
+ * type: string
111
+ * format: uri
112
+ * description: Absolute URL of the archive download endpoint.
113
+ * example: https://example.org/collection-api/v1/dataset/latest/download
114
+ */
115
+ export default function datasetRouter(datasetStorage) {
116
+ const router = express.Router();
117
+
118
+ /**
119
+ * @swagger
120
+ * /dataset/latest:
121
+ * get:
122
+ * summary: Get the metadata of the latest dataset.
123
+ * tags: [Dataset]
124
+ * produces:
125
+ * - application/json
126
+ * responses:
127
+ * 200:
128
+ * description: A JSON object describing the latest dataset generated on this instance.
129
+ * content:
130
+ * application/json:
131
+ * schema:
132
+ * $ref: '#/components/schemas/Dataset'
133
+ * 404:
134
+ * $ref: '#/components/responses/NotFoundError'
135
+ */
136
+ router.get(METADATA_PATH, async (req, res) => {
137
+ const metadata = await datasetStorage.findLatest();
138
+
139
+ if (!metadata) {
140
+ return res.status(404).json({ error: NO_DATASET_ERROR });
141
+ }
142
+
143
+ return res.json({
144
+ ...metadata,
145
+ downloadURL: `${buildAbsoluteBaseUrl(req)}${DOWNLOAD_PATH}`,
146
+ });
147
+ });
148
+
149
+ /**
150
+ * @swagger
151
+ * /dataset/latest/download:
152
+ * get:
153
+ * summary: Download the latest dataset archive.
154
+ * description: Streams the ZIP archive of the latest dataset. Supports conditional requests through the `If-None-Match` and `If-Modified-Since` headers, and resumable downloads through `Range` requests.
155
+ * tags: [Dataset]
156
+ * parameters:
157
+ * - in: header
158
+ * name: If-None-Match
159
+ * description: ETag of an archive already held by the client, to receive `304 Not Modified` when it is still the latest one.
160
+ * schema:
161
+ * type: string
162
+ * - in: header
163
+ * name: If-Modified-Since
164
+ * description: Date of an archive already held by the client, to receive `304 Not Modified` when no dataset was released since.
165
+ * schema:
166
+ * type: string
167
+ * - in: header
168
+ * name: If-Match
169
+ * description: ETag the client expects the archive to still match, to receive `412 Precondition Failed` when it changed.
170
+ * schema:
171
+ * type: string
172
+ * - in: header
173
+ * name: If-Unmodified-Since
174
+ * description: Date the client expects the archive to still match, to receive `412 Precondition Failed` when a newer dataset was released since.
175
+ * schema:
176
+ * type: string
177
+ * - in: header
178
+ * name: Range
179
+ * description: Byte range to resume an interrupted download.
180
+ * schema:
181
+ * type: string
182
+ * example: bytes=1048576-
183
+ * - in: header
184
+ * name: If-Range
185
+ * description: ETag or date the requested `Range` applies to, to fall back to a full `200` response with the whole archive when it no longer matches.
186
+ * schema:
187
+ * type: string
188
+ * responses:
189
+ * 200:
190
+ * description: The dataset archive.
191
+ * headers:
192
+ * Content-Disposition:
193
+ * description: Attachment disposition carrying the archive file name.
194
+ * schema:
195
+ * type: string
196
+ * Content-Length:
197
+ * description: Size of the archive in bytes.
198
+ * schema:
199
+ * type: integer
200
+ * ETag:
201
+ * description: Quoted SHA-256 checksum of the archive.
202
+ * schema:
203
+ * type: string
204
+ * Last-Modified:
205
+ * description: Release date of the dataset.
206
+ * schema:
207
+ * type: string
208
+ * Accept-Ranges:
209
+ * description: Always `bytes`.
210
+ * schema:
211
+ * type: string
212
+ * content:
213
+ * application/zip:
214
+ * schema:
215
+ * type: string
216
+ * format: binary
217
+ * 206:
218
+ * description: The requested byte range of the dataset archive.
219
+ * headers:
220
+ * Content-Range:
221
+ * schema:
222
+ * type: string
223
+ * content:
224
+ * application/zip:
225
+ * schema:
226
+ * type: string
227
+ * format: binary
228
+ * 304:
229
+ * description: The client already holds the latest dataset archive.
230
+ * 404:
231
+ * $ref: '#/components/responses/NotFoundError'
232
+ * 412:
233
+ * description: The archive changed since the version referenced by `If-Match` or `If-Unmodified-Since`.
234
+ * content:
235
+ * application/json:
236
+ * schema:
237
+ * $ref: '#/components/schemas/ErrorResponse'
238
+ * 416:
239
+ * description: The requested byte range cannot be satisfied.
240
+ * headers:
241
+ * Content-Range:
242
+ * schema:
243
+ * type: string
244
+ * content:
245
+ * application/json:
246
+ * schema:
247
+ * $ref: '#/components/schemas/ErrorResponse'
248
+ */
249
+ router.get(DOWNLOAD_PATH, async (req, res, next) => {
250
+ const metadata = await datasetStorage.findLatest();
251
+
252
+ if (!metadata) {
253
+ return res.status(404).json({ error: NO_DATASET_ERROR });
254
+ }
255
+
256
+ const options = {
257
+ root: datasetStorage.path, // Confines the served file to the storage directory and keeps `send` from refusing paths that contain dot-directories
258
+ dotfiles: 'allow', // Archive names are chosen by operators through `--file`; a name starting with a dot, such as `.weekly`, must still be served
259
+ headers: { // Applied by `send` once the archive is found and kept over its own defaults, so that the validators describe the dataset rather than the file on disk
260
+ ETag: `"${metadata.sha256}"`,
261
+ 'Last-Modified': new Date(metadata.releaseDate).toUTCString(),
262
+ },
263
+ };
264
+
265
+ return res.download(metadata.filename, metadata.filename, options, error => handleTransferError(error, res, next));
266
+ });
267
+
268
+ return router;
269
+ }