@opentermsarchive/engine 15.3.1 → 16.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/ota-dataset.js +3 -5
- package/bin/ota.js +1 -1
- package/config/default.json +2 -1
- package/config/test.json +3 -1
- package/package.json +2 -2
- package/scripts/dataset/assets/README.template.js +3 -1
- package/scripts/dataset/export/index.js +69 -53
- package/scripts/dataset/export/index.test.js +73 -1
- package/scripts/dataset/export/test/fixtures/dataset/README.md +1 -1
- package/scripts/dataset/index.js +26 -12
- package/scripts/dataset/index.test.js +195 -0
- package/scripts/dataset/logger/index.js +1 -0
- package/scripts/dataset/logger/index.test.js +32 -0
- package/scripts/dataset/publish/datagouv/dataset.js +2 -3
- package/scripts/dataset/publish/datagouv/index.js +3 -3
- package/scripts/dataset/publish/gitlab/index.js +5 -4
- package/scripts/dataset/publish/gitlab/index.test.js +90 -0
- package/src/archivist/fetcher/fullDomFetcher.js +46 -12
- package/src/archivist/fetcher/fullDomFetcher.test.js +52 -1
- package/src/collection-api/middlewares/errors.js +5 -0
- package/src/collection-api/middlewares/errors.test.js +56 -0
- package/src/collection-api/routes/dataset.js +269 -0
- package/src/collection-api/routes/dataset.test.js +395 -0
- package/src/collection-api/routes/docs.test.js +8 -0
- package/src/collection-api/routes/feed.js +1 -6
- package/src/collection-api/routes/index.js +4 -0
- package/src/collection-api/utils/url.js +3 -0
- package/src/dataset/license.js +4 -0
- package/src/dataset/storage.js +66 -0
- package/src/dataset/storage.test.js +209 -0
|
@@ -0,0 +1,269 @@
|
|
|
1
|
+
import http from 'http';
|
|
2
|
+
|
|
3
|
+
import express from 'express';
|
|
4
|
+
|
|
5
|
+
import { buildAbsoluteBaseUrl } from '../utils/url.js';
|
|
6
|
+
|
|
7
|
+
export const NO_DATASET_ERROR = 'No dataset has been generated yet';
|
|
8
|
+
|
|
9
|
+
const TRANSFER_HEADERS = [ 'Accept-Ranges', 'Cache-Control', 'Content-Disposition', 'Content-Length', 'Content-Range', 'Content-Type', 'ETag', 'Last-Modified' ];
|
|
10
|
+
const METADATA_PATH = '/dataset/latest';
|
|
11
|
+
const DOWNLOAD_PATH = `${METADATA_PATH}/download`;
|
|
12
|
+
|
|
13
|
+
function handleTransferError(error, res, next) {
|
|
14
|
+
if (!error || error.code === 'ECONNABORTED') {
|
|
15
|
+
return;
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
if (res.headersSent) { // The transfer started before failing: `send` destroyed the read stream but never ended the response, so it must be forwarded to end it instead of leaving the client waiting
|
|
19
|
+
return next(error);
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
TRANSFER_HEADERS.forEach(header => res.removeHeader(header)); // `send` has already described the archive on the response when it fails, whether it rejects the request or cannot read the file; those headers must not describe the JSON error instead
|
|
23
|
+
|
|
24
|
+
if (error.status === 404) { // The archive was replaced under another name between the metadata lookup and the transfer; the new metadata is already in place, so a retry succeeds
|
|
25
|
+
return res.status(404).json({ error: NO_DATASET_ERROR });
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
if (error.status < 500) { // Client errors raised by `send`, such as an unsatisfiable range or a failed precondition
|
|
29
|
+
return res.status(error.status).set(error.headers ?? {}).json({ error: http.STATUS_CODES[error.status] });
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
return next(error);
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* @param {object} datasetStorage The storage holding the latest dataset of the collection
|
|
37
|
+
* @returns {express.Router} The router instance
|
|
38
|
+
* @swagger
|
|
39
|
+
* tags:
|
|
40
|
+
* name: Dataset
|
|
41
|
+
* description: Dataset API
|
|
42
|
+
* components:
|
|
43
|
+
* schemas:
|
|
44
|
+
* Dataset:
|
|
45
|
+
* type: object
|
|
46
|
+
* description: Metadata of the latest dataset generated on this instance. Counts and dates are frozen at generation time, unlike the live counters of the collection metadata.
|
|
47
|
+
* additionalProperties: false
|
|
48
|
+
* required:
|
|
49
|
+
* - filename
|
|
50
|
+
* - title
|
|
51
|
+
* - license
|
|
52
|
+
* - releaseDate
|
|
53
|
+
* - firstVersionDate
|
|
54
|
+
* - lastVersionDate
|
|
55
|
+
* - servicesCount
|
|
56
|
+
* - termsCount
|
|
57
|
+
* - versionsCount
|
|
58
|
+
* - size
|
|
59
|
+
* - sha256
|
|
60
|
+
* - downloadURL
|
|
61
|
+
* properties:
|
|
62
|
+
* filename:
|
|
63
|
+
* type: string
|
|
64
|
+
* description: Name of the archive file.
|
|
65
|
+
* example: demo-2026-07-06.zip
|
|
66
|
+
* title:
|
|
67
|
+
* type: string
|
|
68
|
+
* description: Dataset title.
|
|
69
|
+
* example: demo
|
|
70
|
+
* license:
|
|
71
|
+
* type: string
|
|
72
|
+
* description: SPDX identifier of the dataset license.
|
|
73
|
+
* example: ODbL-1.0
|
|
74
|
+
* releaseDate:
|
|
75
|
+
* type: string
|
|
76
|
+
* format: date-time
|
|
77
|
+
* description: Datetime when the dataset was generated.
|
|
78
|
+
* example: 2026-07-06T08:30:12Z
|
|
79
|
+
* firstVersionDate:
|
|
80
|
+
* type: string
|
|
81
|
+
* format: date-time
|
|
82
|
+
* description: Fetch date of the earliest version in the dataset.
|
|
83
|
+
* example: 2022-01-01T12:00:00Z
|
|
84
|
+
* lastVersionDate:
|
|
85
|
+
* type: string
|
|
86
|
+
* format: date-time
|
|
87
|
+
* description: Fetch date of the latest version in the dataset.
|
|
88
|
+
* example: 2026-07-05T23:12:00Z
|
|
89
|
+
* servicesCount:
|
|
90
|
+
* type: integer
|
|
91
|
+
* description: Number of services in the dataset.
|
|
92
|
+
* example: 262
|
|
93
|
+
* termsCount:
|
|
94
|
+
* type: integer
|
|
95
|
+
* description: Number of distinct terms (service and terms type pairs) in the dataset.
|
|
96
|
+
* example: 641
|
|
97
|
+
* versionsCount:
|
|
98
|
+
* type: integer
|
|
99
|
+
* description: Total number of version files in the dataset.
|
|
100
|
+
* example: 38294
|
|
101
|
+
* size:
|
|
102
|
+
* type: integer
|
|
103
|
+
* description: Size of the archive in bytes.
|
|
104
|
+
* example: 104857600
|
|
105
|
+
* sha256:
|
|
106
|
+
* type: string
|
|
107
|
+
* description: SHA-256 checksum of the archive, allowing to verify its integrity and to detect changes without downloading it.
|
|
108
|
+
* example: 9f86d081884c7d659a2feaa0c55ad015a3bf4f1b2b0b822cd15d6c15b0f00a08
|
|
109
|
+
* downloadURL:
|
|
110
|
+
* type: string
|
|
111
|
+
* format: uri
|
|
112
|
+
* description: Absolute URL of the archive download endpoint.
|
|
113
|
+
* example: https://example.org/collection-api/v1/dataset/latest/download
|
|
114
|
+
*/
|
|
115
|
+
export default function datasetRouter(datasetStorage) {
|
|
116
|
+
const router = express.Router();
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* @swagger
|
|
120
|
+
* /dataset/latest:
|
|
121
|
+
* get:
|
|
122
|
+
* summary: Get the metadata of the latest dataset.
|
|
123
|
+
* tags: [Dataset]
|
|
124
|
+
* produces:
|
|
125
|
+
* - application/json
|
|
126
|
+
* responses:
|
|
127
|
+
* 200:
|
|
128
|
+
* description: A JSON object describing the latest dataset generated on this instance.
|
|
129
|
+
* content:
|
|
130
|
+
* application/json:
|
|
131
|
+
* schema:
|
|
132
|
+
* $ref: '#/components/schemas/Dataset'
|
|
133
|
+
* 404:
|
|
134
|
+
* $ref: '#/components/responses/NotFoundError'
|
|
135
|
+
*/
|
|
136
|
+
router.get(METADATA_PATH, async (req, res) => {
|
|
137
|
+
const metadata = await datasetStorage.findLatest();
|
|
138
|
+
|
|
139
|
+
if (!metadata) {
|
|
140
|
+
return res.status(404).json({ error: NO_DATASET_ERROR });
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
return res.json({
|
|
144
|
+
...metadata,
|
|
145
|
+
downloadURL: `${buildAbsoluteBaseUrl(req)}${DOWNLOAD_PATH}`,
|
|
146
|
+
});
|
|
147
|
+
});
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* @swagger
|
|
151
|
+
* /dataset/latest/download:
|
|
152
|
+
* get:
|
|
153
|
+
* summary: Download the latest dataset archive.
|
|
154
|
+
* description: Streams the ZIP archive of the latest dataset. Supports conditional requests through the `If-None-Match` and `If-Modified-Since` headers, and resumable downloads through `Range` requests.
|
|
155
|
+
* tags: [Dataset]
|
|
156
|
+
* parameters:
|
|
157
|
+
* - in: header
|
|
158
|
+
* name: If-None-Match
|
|
159
|
+
* description: ETag of an archive already held by the client, to receive `304 Not Modified` when it is still the latest one.
|
|
160
|
+
* schema:
|
|
161
|
+
* type: string
|
|
162
|
+
* - in: header
|
|
163
|
+
* name: If-Modified-Since
|
|
164
|
+
* description: Date of an archive already held by the client, to receive `304 Not Modified` when no dataset was released since.
|
|
165
|
+
* schema:
|
|
166
|
+
* type: string
|
|
167
|
+
* - in: header
|
|
168
|
+
* name: If-Match
|
|
169
|
+
* description: ETag the client expects the archive to still match, to receive `412 Precondition Failed` when it changed.
|
|
170
|
+
* schema:
|
|
171
|
+
* type: string
|
|
172
|
+
* - in: header
|
|
173
|
+
* name: If-Unmodified-Since
|
|
174
|
+
* description: Date the client expects the archive to still match, to receive `412 Precondition Failed` when a newer dataset was released since.
|
|
175
|
+
* schema:
|
|
176
|
+
* type: string
|
|
177
|
+
* - in: header
|
|
178
|
+
* name: Range
|
|
179
|
+
* description: Byte range to resume an interrupted download.
|
|
180
|
+
* schema:
|
|
181
|
+
* type: string
|
|
182
|
+
* example: bytes=1048576-
|
|
183
|
+
* - in: header
|
|
184
|
+
* name: If-Range
|
|
185
|
+
* description: ETag or date the requested `Range` applies to, to fall back to a full `200` response with the whole archive when it no longer matches.
|
|
186
|
+
* schema:
|
|
187
|
+
* type: string
|
|
188
|
+
* responses:
|
|
189
|
+
* 200:
|
|
190
|
+
* description: The dataset archive.
|
|
191
|
+
* headers:
|
|
192
|
+
* Content-Disposition:
|
|
193
|
+
* description: Attachment disposition carrying the archive file name.
|
|
194
|
+
* schema:
|
|
195
|
+
* type: string
|
|
196
|
+
* Content-Length:
|
|
197
|
+
* description: Size of the archive in bytes.
|
|
198
|
+
* schema:
|
|
199
|
+
* type: integer
|
|
200
|
+
* ETag:
|
|
201
|
+
* description: Quoted SHA-256 checksum of the archive.
|
|
202
|
+
* schema:
|
|
203
|
+
* type: string
|
|
204
|
+
* Last-Modified:
|
|
205
|
+
* description: Release date of the dataset.
|
|
206
|
+
* schema:
|
|
207
|
+
* type: string
|
|
208
|
+
* Accept-Ranges:
|
|
209
|
+
* description: Always `bytes`.
|
|
210
|
+
* schema:
|
|
211
|
+
* type: string
|
|
212
|
+
* content:
|
|
213
|
+
* application/zip:
|
|
214
|
+
* schema:
|
|
215
|
+
* type: string
|
|
216
|
+
* format: binary
|
|
217
|
+
* 206:
|
|
218
|
+
* description: The requested byte range of the dataset archive.
|
|
219
|
+
* headers:
|
|
220
|
+
* Content-Range:
|
|
221
|
+
* schema:
|
|
222
|
+
* type: string
|
|
223
|
+
* content:
|
|
224
|
+
* application/zip:
|
|
225
|
+
* schema:
|
|
226
|
+
* type: string
|
|
227
|
+
* format: binary
|
|
228
|
+
* 304:
|
|
229
|
+
* description: The client already holds the latest dataset archive.
|
|
230
|
+
* 404:
|
|
231
|
+
* $ref: '#/components/responses/NotFoundError'
|
|
232
|
+
* 412:
|
|
233
|
+
* description: The archive changed since the version referenced by `If-Match` or `If-Unmodified-Since`.
|
|
234
|
+
* content:
|
|
235
|
+
* application/json:
|
|
236
|
+
* schema:
|
|
237
|
+
* $ref: '#/components/schemas/ErrorResponse'
|
|
238
|
+
* 416:
|
|
239
|
+
* description: The requested byte range cannot be satisfied.
|
|
240
|
+
* headers:
|
|
241
|
+
* Content-Range:
|
|
242
|
+
* schema:
|
|
243
|
+
* type: string
|
|
244
|
+
* content:
|
|
245
|
+
* application/json:
|
|
246
|
+
* schema:
|
|
247
|
+
* $ref: '#/components/schemas/ErrorResponse'
|
|
248
|
+
*/
|
|
249
|
+
router.get(DOWNLOAD_PATH, async (req, res, next) => {
|
|
250
|
+
const metadata = await datasetStorage.findLatest();
|
|
251
|
+
|
|
252
|
+
if (!metadata) {
|
|
253
|
+
return res.status(404).json({ error: NO_DATASET_ERROR });
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
const options = {
|
|
257
|
+
root: datasetStorage.path, // Confines the served file to the storage directory and keeps `send` from refusing paths that contain dot-directories
|
|
258
|
+
dotfiles: 'allow', // Archive names are chosen by operators through `--file`; a name starting with a dot, such as `.weekly`, must still be served
|
|
259
|
+
headers: { // Applied by `send` once the archive is found and kept over its own defaults, so that the validators describe the dataset rather than the file on disk
|
|
260
|
+
ETag: `"${metadata.sha256}"`,
|
|
261
|
+
'Last-Modified': new Date(metadata.releaseDate).toUTCString(),
|
|
262
|
+
},
|
|
263
|
+
};
|
|
264
|
+
|
|
265
|
+
return res.download(metadata.filename, metadata.filename, options, error => handleTransferError(error, res, next));
|
|
266
|
+
});
|
|
267
|
+
|
|
268
|
+
return router;
|
|
269
|
+
}
|
|
@@ -0,0 +1,395 @@
|
|
|
1
|
+
import { createHash } from 'crypto';
|
|
2
|
+
import fsApi from 'fs';
|
|
3
|
+
import fs from 'fs/promises';
|
|
4
|
+
import { Readable } from 'stream';
|
|
5
|
+
|
|
6
|
+
import { expect } from 'chai';
|
|
7
|
+
import config from 'config';
|
|
8
|
+
import sinon from 'sinon';
|
|
9
|
+
import supertest from 'supertest';
|
|
10
|
+
|
|
11
|
+
import DatasetStorage from '../../dataset/storage.js';
|
|
12
|
+
import app from '../server.js';
|
|
13
|
+
|
|
14
|
+
import { NO_DATASET_ERROR } from './dataset.js';
|
|
15
|
+
|
|
16
|
+
const basePath = config.get('@opentermsarchive/engine.collection-api.basePath');
|
|
17
|
+
const request = supertest(app);
|
|
18
|
+
|
|
19
|
+
function binaryParser(res, callback) { // superagent only buffers text, JSON and media types on its own, so ZIP bodies have to be collected explicitly
|
|
20
|
+
const chunks = [];
|
|
21
|
+
|
|
22
|
+
res.on('data', chunk => chunks.push(chunk));
|
|
23
|
+
res.on('end', () => callback(null, Buffer.concat(chunks)));
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
const METADATA_URL = `${basePath}/v1/dataset/latest`;
|
|
27
|
+
const DOWNLOAD_URL = `${basePath}/v1/dataset/latest/download`;
|
|
28
|
+
const ARCHIVE_FILENAME = 'sandbox-2026-01-01.zip';
|
|
29
|
+
const ARCHIVE_CONTENT = Buffer.from('Archive content standing for a ZIP file in tests');
|
|
30
|
+
const METADATA = {
|
|
31
|
+
filename: ARCHIVE_FILENAME,
|
|
32
|
+
title: 'sandbox',
|
|
33
|
+
license: 'ODbL-1.0',
|
|
34
|
+
releaseDate: '2026-01-01T08:30:12Z',
|
|
35
|
+
firstVersionDate: '2021-01-01T11:27:00Z',
|
|
36
|
+
lastVersionDate: '2022-01-06T11:32:47Z',
|
|
37
|
+
servicesCount: 2,
|
|
38
|
+
termsCount: 3,
|
|
39
|
+
versionsCount: 4,
|
|
40
|
+
size: ARCHIVE_CONTENT.length,
|
|
41
|
+
sha256: createHash('sha256').update(ARCHIVE_CONTENT).digest('hex'),
|
|
42
|
+
};
|
|
43
|
+
|
|
44
|
+
describe('Dataset API', () => {
|
|
45
|
+
const storage = new DatasetStorage(config.get('@opentermsarchive/engine.dataset.storagePath'));
|
|
46
|
+
|
|
47
|
+
async function storeDataset() {
|
|
48
|
+
await fs.mkdir(storage.path, { recursive: true });
|
|
49
|
+
await fs.writeFile(storage.archivePath(ARCHIVE_FILENAME), ARCHIVE_CONTENT);
|
|
50
|
+
await storage.save(METADATA);
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
async function removeDataset() {
|
|
54
|
+
await fs.rm(storage.path, { recursive: true, force: true });
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
function itRespondsWithNoDatasetError(getResponse) {
|
|
58
|
+
it('responds with 404 status code', () => {
|
|
59
|
+
expect(getResponse().status).to.equal(404);
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
it('responds with Content-Type application/json', () => {
|
|
63
|
+
expect(getResponse().type).to.equal('application/json');
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
it('returns an explicit error message', () => {
|
|
67
|
+
expect(getResponse().body).to.deep.equal({ error: NO_DATASET_ERROR });
|
|
68
|
+
});
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
describe('GET /dataset/latest', () => {
|
|
72
|
+
let response;
|
|
73
|
+
|
|
74
|
+
context('when no dataset has been generated', () => {
|
|
75
|
+
before(async () => {
|
|
76
|
+
await removeDataset();
|
|
77
|
+
response = await request.get(METADATA_URL);
|
|
78
|
+
});
|
|
79
|
+
|
|
80
|
+
itRespondsWithNoDatasetError(() => response);
|
|
81
|
+
});
|
|
82
|
+
|
|
83
|
+
context('when the archive described by the metadata is missing', () => {
|
|
84
|
+
before(async () => {
|
|
85
|
+
await storeDataset();
|
|
86
|
+
await fs.rm(storage.archivePath(ARCHIVE_FILENAME));
|
|
87
|
+
response = await request.get(METADATA_URL);
|
|
88
|
+
});
|
|
89
|
+
|
|
90
|
+
after(removeDataset);
|
|
91
|
+
|
|
92
|
+
itRespondsWithNoDatasetError(() => response);
|
|
93
|
+
});
|
|
94
|
+
|
|
95
|
+
context('when a dataset exists', () => {
|
|
96
|
+
before(async () => {
|
|
97
|
+
await storeDataset();
|
|
98
|
+
response = await request.get(METADATA_URL);
|
|
99
|
+
});
|
|
100
|
+
|
|
101
|
+
after(removeDataset);
|
|
102
|
+
|
|
103
|
+
it('responds with 200 status code', () => {
|
|
104
|
+
expect(response.status).to.equal(200);
|
|
105
|
+
});
|
|
106
|
+
|
|
107
|
+
it('responds with Content-Type application/json', () => {
|
|
108
|
+
expect(response.type).to.equal('application/json');
|
|
109
|
+
});
|
|
110
|
+
|
|
111
|
+
it('returns the dataset metadata along with its download URL', () => {
|
|
112
|
+
const { host } = new URL(response.request.url);
|
|
113
|
+
|
|
114
|
+
expect(response.body).to.deep.equal({
|
|
115
|
+
...METADATA,
|
|
116
|
+
downloadURL: `http://${host}${DOWNLOAD_URL}`,
|
|
117
|
+
});
|
|
118
|
+
});
|
|
119
|
+
});
|
|
120
|
+
|
|
121
|
+
context('behind a reverse proxy', () => {
|
|
122
|
+
before(async () => {
|
|
123
|
+
await storeDataset();
|
|
124
|
+
response = await request
|
|
125
|
+
.get(METADATA_URL)
|
|
126
|
+
.set('X-Forwarded-Proto', 'https')
|
|
127
|
+
.set('X-Forwarded-Host', 'api.example.com');
|
|
128
|
+
});
|
|
129
|
+
|
|
130
|
+
after(removeDataset);
|
|
131
|
+
|
|
132
|
+
it('uses the forwarded protocol and host in the download URL', () => {
|
|
133
|
+
expect(response.body.downloadURL).to.equal(`https://api.example.com${DOWNLOAD_URL}`);
|
|
134
|
+
});
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
context('behind a chain of reverse proxies', () => {
|
|
138
|
+
before(async () => {
|
|
139
|
+
await storeDataset();
|
|
140
|
+
response = await request
|
|
141
|
+
.get(METADATA_URL)
|
|
142
|
+
.set('X-Forwarded-Proto', 'https')
|
|
143
|
+
.set('X-Forwarded-Host', 'api.example.com, edge.internal');
|
|
144
|
+
});
|
|
145
|
+
|
|
146
|
+
after(removeDataset);
|
|
147
|
+
|
|
148
|
+
it('uses the first host in the forwarded list in the download URL', () => {
|
|
149
|
+
expect(response.body.downloadURL).to.equal(`https://api.example.com${DOWNLOAD_URL}`);
|
|
150
|
+
});
|
|
151
|
+
});
|
|
152
|
+
|
|
153
|
+
context('when the stored metadata is corrupted', () => {
|
|
154
|
+
before(async () => {
|
|
155
|
+
await fs.mkdir(storage.path, { recursive: true });
|
|
156
|
+
await fs.writeFile(storage.metadataPath, '{ not json');
|
|
157
|
+
response = await request.get(METADATA_URL);
|
|
158
|
+
});
|
|
159
|
+
|
|
160
|
+
after(removeDataset);
|
|
161
|
+
|
|
162
|
+
it('responds with 500 status code', () => {
|
|
163
|
+
expect(response.status).to.equal(500);
|
|
164
|
+
});
|
|
165
|
+
|
|
166
|
+
it('responds with Content-Type application/json', () => {
|
|
167
|
+
expect(response.type).to.equal('application/json');
|
|
168
|
+
});
|
|
169
|
+
|
|
170
|
+
it('returns a generic error message', () => {
|
|
171
|
+
expect(response.body).to.deep.equal({ error: 'Internal Server Error' });
|
|
172
|
+
});
|
|
173
|
+
});
|
|
174
|
+
});
|
|
175
|
+
|
|
176
|
+
describe('GET /dataset/latest/download', () => {
|
|
177
|
+
let response;
|
|
178
|
+
|
|
179
|
+
context('when no dataset has been generated', () => {
|
|
180
|
+
before(async () => {
|
|
181
|
+
await removeDataset();
|
|
182
|
+
response = await request.get(DOWNLOAD_URL);
|
|
183
|
+
});
|
|
184
|
+
|
|
185
|
+
itRespondsWithNoDatasetError(() => response);
|
|
186
|
+
});
|
|
187
|
+
|
|
188
|
+
context('when the archive described by the metadata is missing', () => {
|
|
189
|
+
before(async () => {
|
|
190
|
+
await storeDataset();
|
|
191
|
+
await fs.rm(storage.archivePath(ARCHIVE_FILENAME));
|
|
192
|
+
response = await request.get(DOWNLOAD_URL);
|
|
193
|
+
});
|
|
194
|
+
|
|
195
|
+
after(removeDataset);
|
|
196
|
+
|
|
197
|
+
itRespondsWithNoDatasetError(() => response);
|
|
198
|
+
});
|
|
199
|
+
|
|
200
|
+
context('when a dataset exists', () => {
|
|
201
|
+
before(storeDataset);
|
|
202
|
+
|
|
203
|
+
after(removeDataset);
|
|
204
|
+
|
|
205
|
+
describe('without conditions', () => {
|
|
206
|
+
before(async () => {
|
|
207
|
+
response = await request.get(DOWNLOAD_URL).buffer(true).parse(binaryParser);
|
|
208
|
+
});
|
|
209
|
+
|
|
210
|
+
it('responds with 200 status code', () => {
|
|
211
|
+
expect(response.status).to.equal(200);
|
|
212
|
+
});
|
|
213
|
+
|
|
214
|
+
it('responds with Content-Type application/zip', () => {
|
|
215
|
+
expect(response.type).to.equal('application/zip');
|
|
216
|
+
});
|
|
217
|
+
|
|
218
|
+
it('exposes the archive as an attachment named after the archive file', () => {
|
|
219
|
+
expect(response.headers['content-disposition']).to.equal(`attachment; filename="${ARCHIVE_FILENAME}"`);
|
|
220
|
+
});
|
|
221
|
+
|
|
222
|
+
it('exposes the archive size as Content-Length', () => {
|
|
223
|
+
expect(response.headers['content-length']).to.equal(String(ARCHIVE_CONTENT.length));
|
|
224
|
+
});
|
|
225
|
+
|
|
226
|
+
it('exposes the archive checksum as a strong ETag', () => {
|
|
227
|
+
expect(response.headers.etag).to.equal(`"${METADATA.sha256}"`);
|
|
228
|
+
});
|
|
229
|
+
|
|
230
|
+
it('exposes the release date as Last-Modified', () => {
|
|
231
|
+
expect(response.headers['last-modified']).to.equal(new Date(METADATA.releaseDate).toUTCString());
|
|
232
|
+
});
|
|
233
|
+
|
|
234
|
+
it('advertises byte range support', () => {
|
|
235
|
+
expect(response.headers['accept-ranges']).to.equal('bytes');
|
|
236
|
+
});
|
|
237
|
+
|
|
238
|
+
it('returns the archive content', () => {
|
|
239
|
+
expect(response.body.equals(ARCHIVE_CONTENT)).to.be.true;
|
|
240
|
+
});
|
|
241
|
+
});
|
|
242
|
+
|
|
243
|
+
describe('with a conditional request', () => {
|
|
244
|
+
it('returns 304 with no body when If-None-Match matches the archive checksum', async () => {
|
|
245
|
+
const conditionalResponse = await request.get(DOWNLOAD_URL).set('If-None-Match', `"${METADATA.sha256}"`);
|
|
246
|
+
|
|
247
|
+
expect(conditionalResponse.status).to.equal(304);
|
|
248
|
+
expect(conditionalResponse.text).to.be.empty;
|
|
249
|
+
});
|
|
250
|
+
|
|
251
|
+
it('returns 200 with the archive when If-None-Match does not match', async () => {
|
|
252
|
+
const conditionalResponse = await request.get(DOWNLOAD_URL).set('If-None-Match', '"another-checksum"');
|
|
253
|
+
|
|
254
|
+
expect(conditionalResponse.status).to.equal(200);
|
|
255
|
+
});
|
|
256
|
+
|
|
257
|
+
it('returns 304 with no body when If-Modified-Since is at or after the release date', async () => {
|
|
258
|
+
const conditionalResponse = await request.get(DOWNLOAD_URL).set('If-Modified-Since', new Date(METADATA.releaseDate).toUTCString());
|
|
259
|
+
|
|
260
|
+
expect(conditionalResponse.status).to.equal(304);
|
|
261
|
+
expect(conditionalResponse.text).to.be.empty;
|
|
262
|
+
});
|
|
263
|
+
|
|
264
|
+
it('returns 200 with the archive when If-Modified-Since is before the release date', async () => {
|
|
265
|
+
const conditionalResponse = await request.get(DOWNLOAD_URL).set('If-Modified-Since', new Date('2025-12-31T00:00:00Z').toUTCString());
|
|
266
|
+
|
|
267
|
+
expect(conditionalResponse.status).to.equal(200);
|
|
268
|
+
});
|
|
269
|
+
|
|
270
|
+
it('returns a JSON error with 412 status code when If-Match does not match the archive checksum', async () => {
|
|
271
|
+
const conditionalResponse = await request.get(DOWNLOAD_URL).set('If-Match', '"another-checksum"');
|
|
272
|
+
|
|
273
|
+
expect(conditionalResponse.status).to.equal(412);
|
|
274
|
+
expect(conditionalResponse.type).to.equal('application/json');
|
|
275
|
+
expect(conditionalResponse.headers).to.not.have.any.keys('content-disposition', 'last-modified');
|
|
276
|
+
expect(conditionalResponse.body).to.deep.equal({ error: 'Precondition Failed' });
|
|
277
|
+
});
|
|
278
|
+
});
|
|
279
|
+
|
|
280
|
+
describe('with a range request', () => {
|
|
281
|
+
it('returns the requested bytes with 206 status code', async () => {
|
|
282
|
+
const rangeResponse = await request.get(DOWNLOAD_URL).set('Range', 'bytes=0-4').buffer(true).parse(binaryParser);
|
|
283
|
+
|
|
284
|
+
expect(rangeResponse.status).to.equal(206);
|
|
285
|
+
expect(rangeResponse.headers['content-range']).to.equal(`bytes 0-4/${ARCHIVE_CONTENT.length}`);
|
|
286
|
+
expect(rangeResponse.body.equals(ARCHIVE_CONTENT.subarray(0, 5))).to.be.true;
|
|
287
|
+
});
|
|
288
|
+
|
|
289
|
+
it('returns a JSON error with 416 status code when the range cannot be satisfied', async () => {
|
|
290
|
+
const rangeResponse = await request.get(DOWNLOAD_URL).set('Range', `bytes=${ARCHIVE_CONTENT.length}-`);
|
|
291
|
+
|
|
292
|
+
expect(rangeResponse.status).to.equal(416);
|
|
293
|
+
expect(rangeResponse.type).to.equal('application/json');
|
|
294
|
+
expect(rangeResponse.headers['content-range']).to.equal(`bytes */${ARCHIVE_CONTENT.length}`);
|
|
295
|
+
expect(rangeResponse.headers).to.not.have.any.keys('content-disposition', 'last-modified');
|
|
296
|
+
expect(rangeResponse.headers.etag).to.not.equal(`"${METADATA.sha256}"`);
|
|
297
|
+
expect(rangeResponse.body).to.deep.equal({ error: 'Range Not Satisfiable' });
|
|
298
|
+
});
|
|
299
|
+
|
|
300
|
+
it('returns 200 with the full archive when If-Range does not match the current archive', async () => {
|
|
301
|
+
const rangeResponse = await request.get(DOWNLOAD_URL).set('If-Range', '"stale-checksum"').set('Range', 'bytes=0-4').buffer(true)
|
|
302
|
+
.parse(binaryParser);
|
|
303
|
+
|
|
304
|
+
expect(rangeResponse.status).to.equal(200);
|
|
305
|
+
expect(rangeResponse.body.equals(ARCHIVE_CONTENT)).to.be.true;
|
|
306
|
+
});
|
|
307
|
+
});
|
|
308
|
+
|
|
309
|
+
describe('with a HEAD request', () => {
|
|
310
|
+
it('returns the archive headers without its content', async () => {
|
|
311
|
+
const headResponse = await request.head(DOWNLOAD_URL);
|
|
312
|
+
|
|
313
|
+
expect(headResponse.status).to.equal(200);
|
|
314
|
+
expect(headResponse.headers['content-length']).to.equal(String(ARCHIVE_CONTENT.length));
|
|
315
|
+
expect(headResponse.headers.etag).to.equal(`"${METADATA.sha256}"`);
|
|
316
|
+
expect(headResponse.text).to.be.oneOf([ undefined, '' ]);
|
|
317
|
+
});
|
|
318
|
+
});
|
|
319
|
+
});
|
|
320
|
+
|
|
321
|
+
context('when the archive file name starts with a dot', () => {
|
|
322
|
+
const DOTFILE_ARCHIVE_FILENAME = '.weekly-2026-01-01.zip';
|
|
323
|
+
|
|
324
|
+
before(async () => {
|
|
325
|
+
await fs.mkdir(storage.path, { recursive: true });
|
|
326
|
+
await fs.writeFile(storage.archivePath(DOTFILE_ARCHIVE_FILENAME), ARCHIVE_CONTENT);
|
|
327
|
+
await storage.save({ ...METADATA, filename: DOTFILE_ARCHIVE_FILENAME });
|
|
328
|
+
response = await request.get(DOWNLOAD_URL).buffer(true).parse(binaryParser);
|
|
329
|
+
});
|
|
330
|
+
|
|
331
|
+
after(removeDataset);
|
|
332
|
+
|
|
333
|
+
it('responds with 200 status code', () => {
|
|
334
|
+
expect(response.status).to.equal(200);
|
|
335
|
+
});
|
|
336
|
+
|
|
337
|
+
it('returns the archive content', () => {
|
|
338
|
+
expect(response.body.equals(ARCHIVE_CONTENT)).to.be.true;
|
|
339
|
+
});
|
|
340
|
+
});
|
|
341
|
+
|
|
342
|
+
context('when the archive cannot be read', () => {
|
|
343
|
+
before(async () => {
|
|
344
|
+
await storeDataset();
|
|
345
|
+
sinon.stub(fsApi, 'createReadStream').returns(new Readable({ read() { this.destroy(new Error('Disk failure')); } })); // `send` opens the archive through the `fs` module at transfer time, once it has already described the archive on the response
|
|
346
|
+
response = await request.get(DOWNLOAD_URL);
|
|
347
|
+
});
|
|
348
|
+
|
|
349
|
+
after(async () => {
|
|
350
|
+
sinon.restore();
|
|
351
|
+
await removeDataset();
|
|
352
|
+
});
|
|
353
|
+
|
|
354
|
+
it('responds with 500 status code', () => {
|
|
355
|
+
expect(response.status).to.equal(500);
|
|
356
|
+
});
|
|
357
|
+
|
|
358
|
+
it('responds with Content-Type application/json', () => {
|
|
359
|
+
expect(response.type).to.equal('application/json');
|
|
360
|
+
});
|
|
361
|
+
|
|
362
|
+
it('returns a generic error message', () => {
|
|
363
|
+
expect(response.body).to.deep.equal({ error: 'Internal Server Error' });
|
|
364
|
+
});
|
|
365
|
+
|
|
366
|
+
it('does not describe the archive', () => {
|
|
367
|
+
expect(response.headers).to.not.have.any.keys('content-disposition', 'last-modified', 'accept-ranges');
|
|
368
|
+
expect(response.headers.etag).to.not.equal(`"${METADATA.sha256}"`);
|
|
369
|
+
});
|
|
370
|
+
});
|
|
371
|
+
|
|
372
|
+
context('when the archive is replaced between the metadata lookup and the transfer', () => {
|
|
373
|
+
before(async () => {
|
|
374
|
+
await storeDataset();
|
|
375
|
+
sinon.stub(DatasetStorage.prototype, 'findLatest').resolves({ ...METADATA, filename: 'sandbox-2026-01-02.zip' }); // The metadata describes an archive that is no longer on disk, as when a generation completes right after the lookup
|
|
376
|
+
response = await request.get(DOWNLOAD_URL);
|
|
377
|
+
});
|
|
378
|
+
|
|
379
|
+
after(async () => {
|
|
380
|
+
sinon.restore();
|
|
381
|
+
await removeDataset();
|
|
382
|
+
});
|
|
383
|
+
|
|
384
|
+
itRespondsWithNoDatasetError(() => response);
|
|
385
|
+
});
|
|
386
|
+
});
|
|
387
|
+
|
|
388
|
+
describe('GET /dataset', () => {
|
|
389
|
+
it('responds with 404 status code', async () => {
|
|
390
|
+
const response = await request.get(`${basePath}/v1/dataset`);
|
|
391
|
+
|
|
392
|
+
expect(response.status).to.equal(404);
|
|
393
|
+
});
|
|
394
|
+
});
|
|
395
|
+
});
|
|
@@ -54,6 +54,14 @@ describe('Docs API', () => {
|
|
|
54
54
|
it('/version/{serviceId}/{termsType}/{date}', () => {
|
|
55
55
|
expect(subject).to.have.property('/version/{serviceId}/{termsType}/{date}');
|
|
56
56
|
});
|
|
57
|
+
|
|
58
|
+
it('/dataset/latest', () => {
|
|
59
|
+
expect(subject).to.have.property('/dataset/latest');
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
it('/dataset/latest/download', () => {
|
|
63
|
+
expect(subject).to.have.property('/dataset/latest/download');
|
|
64
|
+
});
|
|
57
65
|
});
|
|
58
66
|
});
|
|
59
67
|
});
|