@opentermsarchive/engine 15.3.1 → 16.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -11,18 +11,16 @@ import logger from '../src/logger/index.js';
11
11
 
12
12
  program
13
13
  .name('ota dataset')
14
- .description('Export the versions dataset into a ZIP file and optionally publish it to GitHub releases, GitLab releases, or data.gouv.fr')
14
+ .description('Export the versions dataset into a ZIP file stored locally and optionally publish it to GitHub releases, GitLab releases, or data.gouv.fr')
15
15
  .option('-f, --file <filename>', 'file name of the generated dataset')
16
16
  .option('-p, --publish', 'publish dataset. Supports GitHub releases (OTA_ENGINE_GITHUB_TOKEN), GitLab releases (OTA_ENGINE_GITLAB_TOKEN), or data.gouv.fr (OTA_ENGINE_DATAGOUV_API_KEY + config)')
17
- .option('-r, --remove-local-copy', 'remove local copy of dataset after publishing. Works only in combination with --publish option')
18
17
  .option('--schedule', 'schedule automatic dataset generation');
19
18
 
20
- const { schedule, publish, removeLocalCopy, file: fileName } = program.parse().opts();
19
+ const { schedule, publish, file: fileName } = program.parse().opts();
21
20
 
22
21
  const options = {
23
22
  fileName,
24
23
  shouldPublish: publish,
25
- shouldRemoveLocalCopy: removeLocalCopy,
26
24
  };
27
25
 
28
26
  if (!schedule) {
@@ -34,5 +32,5 @@ if (!schedule) {
34
32
  logger.info('The scheduler is running…');
35
33
  logger.info(`Dataset will be published ${humanReadableSchedule.toLowerCase()} in the timezone of this machine`);
36
34
 
37
- new Cron(config.get('@opentermsarchive/engine.dataset.publishingSchedule'), () => release(options)); // eslint-disable-line no-new
35
+ new Cron(trackingSchedule, { catch: error => logger.error(`Dataset release failed: ${error.stack}`) }, () => release(options)); // eslint-disable-line no-new
38
36
  }
package/bin/ota.js CHANGED
@@ -14,6 +14,6 @@ program
14
14
  .command('apply-technical-upgrades', 'Apply technical upgrades by generating new versions from the latest snapshots using updated declarations, engine logic, or dependencies')
15
15
  .command('validate', 'Run a series of tests to check the validity of terms declarations')
16
16
  .command('lint', 'Check format and stylistic errors in declarations and auto fix them')
17
- .command('dataset', 'Export the versions dataset into a ZIP file and optionally publish it to GitHub releases')
17
+ .command('dataset', 'Export the versions dataset into a ZIP file stored locally and optionally publish it to GitHub releases, GitLab releases, or data.gouv.fr')
18
18
  .command('serve', 'Start the collection metadata API server')
19
19
  .parse(process.argv);
@@ -46,7 +46,8 @@
46
46
  "timestampPrefix": true
47
47
  },
48
48
  "dataset": {
49
- "publishingSchedule": "30 8 * * MON"
49
+ "publishingSchedule": "30 8 * * MON",
50
+ "storagePath": "./data/datasets"
50
51
  },
51
52
  "collection-api": {
52
53
  "feed": {
package/config/test.json CHANGED
@@ -43,7 +43,9 @@
43
43
  },
44
44
  "dataset": {
45
45
  "title": "sandbox",
46
- "versionsRepositoryURL": "https://github.com/OpenTermsArchive/sandbox-versions"
46
+ "versionsRepositoryURL": "https://github.com/OpenTermsArchive/sandbox-versions",
47
+ "apiBaseURL": "https://gitlab.example.test/api/v4",
48
+ "storagePath": "./test/data/datasets"
47
49
  },
48
50
  "collection-api": {
49
51
  "port": 3000,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@opentermsarchive/engine",
3
- "version": "15.3.1",
3
+ "version": "16.0.1",
4
4
  "description": "Tracks and makes visible changes to the terms of online services",
5
5
  "homepage": "https://opentermsarchive.org",
6
6
  "bugs": {
@@ -34,7 +34,7 @@
34
34
  "scripts": {
35
35
  "commit-messages:lint": "commitlint --from=main --to=HEAD",
36
36
  "dataset:generate": "node bin/ota.js dataset",
37
- "dataset:release": "node bin/ota.js dataset --publish --remove-local-copy",
37
+ "dataset:release": "node bin/ota.js dataset --publish",
38
38
  "dataset:scheduler": "npm run dataset:release -- --schedule",
39
39
  "declarations:lint": "node bin/ota.js lint",
40
40
  "declarations:validate": "node bin/ota.js validate declarations",
@@ -1,5 +1,7 @@
1
1
  import config from 'config';
2
2
 
3
+ import * as license from '../../../src/dataset/license.js';
4
+
3
5
  const LOCALE = 'en-EN';
4
6
  const DATE_OPTIONS = { year: 'numeric', month: 'long', day: 'numeric' };
5
7
 
@@ -60,6 +62,6 @@ This dataset represents each version of a document as a separate [Markdown](http
60
62
 
61
63
  ### License
62
64
 
63
- This dataset is made available under an [Open Database (OdBL) License](https://opendatacommons.org/licenses/odbl/1.0/) by Open Terms Archive Contributors.
65
+ This dataset is made available under an [${license.DISPLAY_NAME}](${license.INFO_URL}) by Open Terms Archive Contributors.
64
66
  `;
65
67
  }
@@ -1,11 +1,14 @@
1
+ import { createHash } from 'crypto';
1
2
  import fsApi from 'fs';
2
3
  import path from 'path';
4
+ import { pipeline } from 'stream/promises';
3
5
  import { fileURLToPath } from 'url';
4
6
 
5
7
  import archiver from 'archiver';
6
8
  import config from 'config';
7
9
 
8
10
  import RepositoryFactory from '../../../src/archivist/recorder/repositories/factory.js';
11
+ import { TEMPORARY_SUFFIX } from '../../../src/dataset/storage.js';
9
12
  import * as renamer from '../../utils/renamer/index.js';
10
13
  import readme from '../assets/README.template.js';
11
14
  import logger from '../logger/index.js';
@@ -19,83 +22,96 @@ const ARCHIVE_FORMAT = 'zip'; // for supported formats, see https://www.archiver
19
22
  export default async function generate({ archivePath, releaseDate }) {
20
23
  const versionsRepository = await RepositoryFactory.create(config.get('@opentermsarchive/engine.recorder.versions.storage')).initialize();
21
24
 
22
- const archive = await initializeArchive(archivePath);
25
+ const temporaryArchivePath = `${archivePath}${TEMPORARY_SUFFIX}`;
26
+ const archive = await initializeArchive(temporaryArchivePath, path.basename(archivePath, path.extname(archivePath)));
23
27
 
24
- await renamer.loadRules();
28
+ try {
29
+ await renamer.loadRules();
25
30
 
26
- const services = new Set();
27
- let firstVersionDate = new Date();
28
- let lastVersionDate = new Date(0);
31
+ const services = new Set();
32
+ const terms = new Set();
33
+ let firstVersionDate = new Date();
34
+ let lastVersionDate = new Date(0);
29
35
 
30
- let index = 1;
36
+ let versionsCount = 0;
31
37
 
32
- for await (const version of versionsRepository.iterate()) {
33
- const { content, fetchDate } = version;
38
+ for await (const version of versionsRepository.iterate()) {
39
+ const { content, fetchDate } = version;
34
40
 
35
- for (const { serviceId, termsType } of renamer.applyRules(version.serviceId, version.termsType)) {
36
- if (firstVersionDate > fetchDate) {
37
- firstVersionDate = fetchDate;
38
- }
41
+ for (const { serviceId, termsType } of renamer.applyRules(version.serviceId, version.termsType)) {
42
+ if (firstVersionDate > fetchDate) {
43
+ firstVersionDate = fetchDate;
44
+ }
39
45
 
40
- if (fetchDate > lastVersionDate) {
41
- lastVersionDate = fetchDate;
42
- }
46
+ if (fetchDate > lastVersionDate) {
47
+ lastVersionDate = fetchDate;
48
+ }
43
49
 
44
- services.add(serviceId);
50
+ services.add(serviceId);
51
+ terms.add(`${serviceId}/${termsType}`);
52
+ versionsCount++;
45
53
 
46
- const versionPath = generateVersionPath({ serviceId, termsType, fetchDate });
54
+ const versionPath = generateVersionPath({ serviceId, termsType, fetchDate });
47
55
 
48
- logger.info({ message: versionPath, counter: index, hash: version.id });
56
+ logger.info({ message: versionPath, counter: versionsCount, hash: version.id });
49
57
 
50
- archive.stream.append(
51
- content,
52
- { name: `${archive.basename}/${versionPath}` },
53
- );
54
- index++;
58
+ archive.stream.append(
59
+ content,
60
+ { name: `${archive.basename}/${versionPath}` },
61
+ );
62
+ }
55
63
  }
56
- }
57
64
 
58
- archive.stream.append(
59
- readme({
65
+ archive.stream.append(
66
+ readme({
67
+ servicesCount: services.size,
68
+ releaseDate,
69
+ firstVersionDate,
70
+ lastVersionDate,
71
+ }),
72
+ { name: `${archive.basename}/README.md` },
73
+ );
74
+ archive.stream.append(
75
+ fsApi.readFileSync(path.resolve(__dirname, '../assets/LICENSE')),
76
+ { name: `${archive.basename}/LICENSE` },
77
+ );
78
+
79
+ await Promise.all([ archive.stream.finalize(), archive.done ]); // Both promises settle on the same underlying zip module error; awaiting only `done` left `finalize`'s rejection unhandled
80
+ await fs.rename(temporaryArchivePath, archivePath); // The archive appears under its final name only once complete, so a crash never leaves a truncated dataset where the collection API would serve it
81
+
82
+ const { size } = await fs.stat(archivePath);
83
+
84
+ return {
60
85
  servicesCount: services.size,
61
- releaseDate,
86
+ termsCount: terms.size,
87
+ versionsCount,
62
88
  firstVersionDate,
63
89
  lastVersionDate,
64
- }),
65
- { name: `${archive.basename}/README.md` },
66
- );
67
- archive.stream.append(
68
- fsApi.readFileSync(path.resolve(__dirname, '../assets/LICENSE')),
69
- { name: `${archive.basename}/LICENSE` },
70
- );
71
-
72
- archive.stream.finalize();
73
-
74
- await archive.done;
75
- await versionsRepository.finalize();
76
-
77
- return {
78
- servicesCount: services.size,
79
- firstVersionDate,
80
- lastVersionDate,
81
- };
90
+ size,
91
+ sha256: archive.hash.digest('hex'),
92
+ };
93
+ } catch (error) {
94
+ archive.stream.destroy();
95
+ await archive.done.catch(() => {}); // The write stream has to be closed before the file can be removed on Windows
96
+ await fs.rm(temporaryArchivePath, { force: true });
97
+ throw error;
98
+ } finally {
99
+ await versionsRepository.finalize();
100
+ }
82
101
  }
83
102
 
84
- async function initializeArchive(targetPath) {
103
+ async function initializeArchive(targetPath, basename) {
85
104
  await fs.mkdir(path.dirname(targetPath), { recursive: true });
86
105
 
87
- const basename = path.basename(targetPath, path.extname(targetPath));
88
-
89
106
  const output = fsApi.createWriteStream(targetPath);
90
107
  const stream = archiver(ARCHIVE_FORMAT, { zlib: { level: 9 } }); // set compression to max level
108
+ const hash = createHash('sha256');
91
109
 
92
- const done = new Promise(resolve => {
93
- output.on('close', resolve);
94
- });
110
+ stream.on('data', chunk => hash.update(chunk)); // Hashing the bytes on their way to disk avoids reading back an archive that can weigh gigabytes
95
111
 
96
- stream.pipe(output);
112
+ const done = pipeline(stream, output); // Unlike waiting for the close event, the pipeline promise also rejects when either stream fails
97
113
 
98
- return { basename, stream, done };
114
+ return { basename, stream, hash, done };
99
115
  }
100
116
 
101
117
  function generateVersionPath({ serviceId, termsType, fetchDate }) {
@@ -1,20 +1,26 @@
1
+ import { createHash } from 'crypto';
1
2
  import fs from 'fs/promises';
2
3
  import path from 'path';
3
4
  import { fileURLToPath } from 'url';
4
5
 
5
6
  import { expect, use } from 'chai';
7
+ import chaiAsPromised from 'chai-as-promised';
6
8
  import config from 'config';
7
9
  import dircompare from 'dir-compare';
8
10
  import mime from 'mime';
9
11
  import StreamZip from 'node-stream-zip';
12
+ import sinon from 'sinon';
10
13
 
11
14
  import GitRepository from '../../../src/archivist/recorder/repositories/git/index.js';
12
15
  import Version from '../../../src/archivist/recorder/version.js';
16
+ import { TEMPORARY_SUFFIX } from '../../../src/dataset/storage.js';
13
17
 
14
18
  import generateArchive from './index.js';
15
19
 
16
20
  const __dirname = path.dirname(fileURLToPath(import.meta.url));
17
21
 
22
+ use(chaiAsPromised);
23
+
18
24
  const FIRST_SERVICE_PROVIDER_ID = 'ServiceA';
19
25
  const SECOND_SERVICE_PROVIDER_ID = 'ServiceB';
20
26
 
@@ -41,6 +47,7 @@ describe('Export', () => {
41
47
  const EXPECTED_DATASET_PATH = path.resolve(__dirname, './test/fixtures/dataset');
42
48
 
43
49
  let repository;
50
+ let stats;
44
51
  let zip;
45
52
 
46
53
  before(async function () {
@@ -84,7 +91,7 @@ describe('Export', () => {
84
91
  snapshotId: SNAPSHOT_ID,
85
92
  }));
86
93
 
87
- await generateArchive({
94
+ stats = await generateArchive({
88
95
  archivePath: ARCHIVE_PATH,
89
96
  releaseDate: new Date(RELEASE_DATE),
90
97
  });
@@ -108,6 +115,71 @@ describe('Export', () => {
108
115
  it('has the proper contents', () => {
109
116
  expect(`${TMP_PATH}/${ARCHIVE_NAME}`).to.have.sameContentAs(EXPECTED_DATASET_PATH);
110
117
  });
118
+
119
+ describe('returned stats', () => {
120
+ it('counts the services', () => {
121
+ expect(stats.servicesCount).to.equal(2);
122
+ });
123
+
124
+ it('counts the distinct terms', () => {
125
+ expect(stats.termsCount).to.equal(3);
126
+ });
127
+
128
+ it('counts the versions', () => {
129
+ expect(stats.versionsCount).to.equal(4);
130
+ });
131
+
132
+ it('exposes the first version date', () => {
133
+ expect(stats.firstVersionDate).to.deep.equal(new Date(FIRST_FETCH_DATE));
134
+ });
135
+
136
+ it('exposes the last version date', () => {
137
+ expect(stats.lastVersionDate).to.deep.equal(new Date(THIRD_FETCH_DATE));
138
+ });
139
+
140
+ it('exposes the archive size in bytes', async () => {
141
+ expect(stats.size).to.equal((await fs.stat(ARCHIVE_PATH)).size);
142
+ });
143
+
144
+ it('exposes the archive SHA-256 checksum', async () => {
145
+ expect(stats.sha256).to.equal(createHash('sha256').update(await fs.readFile(ARCHIVE_PATH)).digest('hex'));
146
+ });
147
+ });
148
+ });
149
+
150
+ context('when the generation fails', () => {
151
+ const TMP_PATH = path.resolve(__dirname, './tmp');
152
+ const FAILING_ARCHIVE_PATH = path.resolve(TMP_PATH, 'failing-dataset.zip');
153
+
154
+ let error;
155
+
156
+ before(async function () {
157
+ this.timeout(10000);
158
+ sinon.stub(GitRepository.prototype, 'iterate').throws(new Error('Repository failure'));
159
+
160
+ try {
161
+ await generateArchive({ archivePath: FAILING_ARCHIVE_PATH, releaseDate: new Date(RELEASE_DATE) });
162
+ } catch (generationError) {
163
+ error = generationError;
164
+ }
165
+ });
166
+
167
+ after(async () => {
168
+ sinon.restore();
169
+ await fs.rm(TMP_PATH, { recursive: true, force: true });
170
+ });
171
+
172
+ it('rejects with the underlying error', () => {
173
+ expect(error).to.be.an('error').with.property('message', 'Repository failure');
174
+ });
175
+
176
+ it('leaves no archive at the target path', async () => {
177
+ await expect(fs.access(FAILING_ARCHIVE_PATH)).to.be.rejected;
178
+ });
179
+
180
+ it('leaves no temporary file', async () => {
181
+ await expect(fs.access(`${FAILING_ARCHIVE_PATH}${TEMPORARY_SUFFIX}`)).to.be.rejected;
182
+ });
111
183
  });
112
184
  });
113
185
 
@@ -37,4 +37,4 @@ This dataset represents each version of a document as a separate [Markdown](http
37
37
 
38
38
  ### License
39
39
 
40
- This dataset is made available under an [Open Database (OdBL) License](https://opendatacommons.org/licenses/odbl/1.0/) by Open Terms Archive Contributors.
40
+ This dataset is made available under an [Open Database (ODbL) License](https://opendatacommons.org/licenses/odbl/1.0/) by Open Terms Archive Contributors.
@@ -1,21 +1,43 @@
1
- import fs from 'fs';
2
1
  import path from 'path';
3
2
 
4
3
  import config from 'config';
5
4
 
5
+ import { toISODateWithoutMilliseconds } from '../../src/archivist/utils/date.js';
6
+ import * as license from '../../src/dataset/license.js';
7
+ import DatasetStorage from '../../src/dataset/storage.js';
8
+
6
9
  import generateRelease from './export/index.js';
7
10
  import logger from './logger/index.js';
8
11
  import publishRelease from './publish/index.js';
9
12
 
10
- export async function release({ shouldPublish, shouldRemoveLocalCopy, fileName }) {
13
+ export async function release({ shouldPublish, fileName }) {
11
14
  const releaseDate = new Date();
12
- const archiveName = fileName || `${config.get('@opentermsarchive/engine.dataset.title').toLowerCase().replace(/[^a-zA-Z0-9.\-_]/g, '-')}-${releaseDate.toISOString().replace(/T.*/, '')}`;
13
- const archivePath = `${path.basename(archiveName, '.zip')}.zip`; // allow to pass filename or filename.zip as the archive name and have filename.zip as the result name
15
+ const title = config.get('@opentermsarchive/engine.dataset.title');
16
+ const storage = new DatasetStorage(config.get('@opentermsarchive/engine.dataset.storagePath'));
17
+ const archiveName = fileName || `${title.toLowerCase().replace(/[^a-zA-Z0-9.\-_]/g, '-')}-${releaseDate.toISOString().replace(/T.*/, '')}`;
18
+ const filename = `${path.basename(archiveName, '.zip')}.zip`; // allow to pass filename or filename.zip as the archive name and have filename.zip as the result name
19
+ const archivePath = storage.archivePath(filename);
14
20
 
15
21
  logger.info('Start exporting dataset…');
16
22
 
17
23
  const stats = await generateRelease({ archivePath, releaseDate });
18
24
 
25
+ await storage.save({
26
+ filename,
27
+ title,
28
+ license: license.SPDX_ID,
29
+ releaseDate: toISODateWithoutMilliseconds(releaseDate),
30
+ firstVersionDate: toISODateWithoutMilliseconds(stats.firstVersionDate),
31
+ lastVersionDate: toISODateWithoutMilliseconds(stats.lastVersionDate),
32
+ servicesCount: stats.servicesCount,
33
+ termsCount: stats.termsCount,
34
+ versionsCount: stats.versionsCount,
35
+ size: stats.size,
36
+ sha256: stats.sha256,
37
+ });
38
+
39
+ await storage.removePreviousArchives().catch(error => logger.warn(`Failed to remove previous dataset archives: ${error.message}`)); // The new dataset is already saved and servable; do not fail the release over stale files left behind
40
+
19
41
  logger.info(`Dataset exported in ${archivePath}`);
20
42
 
21
43
  if (!shouldPublish) {
@@ -36,12 +58,4 @@ export async function release({ shouldPublish, shouldRemoveLocalCopy, fileName }
36
58
  logger.info(` - ${result.platform}: ${result.url}`);
37
59
  });
38
60
  }
39
-
40
- if (!shouldRemoveLocalCopy) {
41
- return;
42
- }
43
-
44
- fs.unlinkSync(archivePath);
45
-
46
- logger.info(`Removed local copy ${archivePath}`);
47
61
  }
@@ -0,0 +1,195 @@
1
+ import { createHash } from 'crypto';
2
+ import fs from 'fs/promises';
3
+ import path from 'path';
4
+
5
+ import { expect, use } from 'chai';
6
+ import chaiAsPromised from 'chai-as-promised';
7
+ import config from 'config';
8
+ import sinon from 'sinon';
9
+
10
+ import RepositoryFactory from '../../src/archivist/recorder/repositories/factory.js';
11
+ import Version from '../../src/archivist/recorder/version.js';
12
+ import DatasetStorage, { TEMPORARY_SUFFIX } from '../../src/dataset/storage.js';
13
+
14
+ import { release } from './index.js';
15
+
16
+ use(chaiAsPromised);
17
+
18
+ const FIRST_SERVICE_PROVIDER_ID = 'ServiceA';
19
+ const SECOND_SERVICE_PROVIDER_ID = 'ServiceB';
20
+
21
+ const FIRST_TERMS_TYPE = 'Terms of Service';
22
+ const SECOND_TERMS_TYPE = 'Privacy Policy';
23
+
24
+ const FIRST_FETCH_DATE = '2021-01-01T11:27:00.000Z';
25
+ const SECOND_FETCH_DATE = '2021-01-11T11:32:47.000Z';
26
+ const THIRD_FETCH_DATE = '2022-01-06T11:32:47.000Z';
27
+ const FOURTH_FETCH_DATE = '2022-01-01T12:12:24.000Z';
28
+
29
+ const SNAPSHOT_ID = '721ce4a63ad399ecbdb548a66d6d327e7bc97876';
30
+
31
+ const STALE_ARCHIVE_FILENAME = 'stale.zip';
32
+
33
+ describe('Dataset release', () => {
34
+ describe('#release', () => {
35
+ const storage = new DatasetStorage(config.get('@opentermsarchive/engine.dataset.storagePath')); // Instantiated up front, so a `before` hook failing partway through still leaves `after` a valid `storage` to clean up
36
+
37
+ let repository;
38
+ let metadata;
39
+
40
+ before(async function () {
41
+ this.timeout(10000);
42
+ repository = RepositoryFactory.create(config.get('@opentermsarchive/engine.recorder.versions.storage'));
43
+ await repository.initialize();
44
+ await repository.removeAll();
45
+
46
+ const versions = [
47
+ [ FIRST_SERVICE_PROVIDER_ID, FIRST_TERMS_TYPE, FIRST_FETCH_DATE ],
48
+ [ FIRST_SERVICE_PROVIDER_ID, FIRST_TERMS_TYPE, SECOND_FETCH_DATE ],
49
+ [ SECOND_SERVICE_PROVIDER_ID, FIRST_TERMS_TYPE, THIRD_FETCH_DATE ],
50
+ [ SECOND_SERVICE_PROVIDER_ID, SECOND_TERMS_TYPE, FOURTH_FETCH_DATE ],
51
+ ];
52
+
53
+ for (const [ serviceId, termsType, fetchDate ] of versions) {
54
+ await repository.save(new Version({
55
+ serviceId,
56
+ termsType,
57
+ content: `Content of ${serviceId} ${termsType} fetched on ${fetchDate}`,
58
+ fetchDate,
59
+ snapshotId: SNAPSHOT_ID,
60
+ }));
61
+ }
62
+
63
+ await fs.mkdir(storage.path, { recursive: true });
64
+ await fs.writeFile(storage.archivePath(STALE_ARCHIVE_FILENAME), 'stale archive');
65
+
66
+ await release({});
67
+
68
+ metadata = await storage.findLatest();
69
+ });
70
+
71
+ after(async () => {
72
+ await repository.removeAll();
73
+ await fs.rm(storage.path, { recursive: true, force: true });
74
+ });
75
+
76
+ it('names the archive after the dataset title and the release date', () => {
77
+ expect(metadata.filename).to.match(/^sandbox-\d{4}-\d{2}-\d{2}\.zip$/);
78
+ });
79
+
80
+ it('stores the archive in the storage directory', async () => {
81
+ await expect(fs.access(storage.archivePath(metadata.filename))).to.be.fulfilled;
82
+ });
83
+
84
+ it('writes no archive in the current working directory', async () => {
85
+ await expect(fs.access(path.resolve(process.cwd(), metadata.filename))).to.be.rejected;
86
+ });
87
+
88
+ it('removes previous archives', async () => {
89
+ await expect(fs.access(storage.archivePath(STALE_ARCHIVE_FILENAME))).to.be.rejected;
90
+ });
91
+
92
+ it('leaves no temporary file', async () => {
93
+ const entries = await fs.readdir(storage.path);
94
+
95
+ expect(entries.filter(entry => entry.endsWith(TEMPORARY_SUFFIX))).to.be.empty;
96
+ });
97
+
98
+ describe('metadata', () => {
99
+ it('describes the dataset title', () => {
100
+ expect(metadata.title).to.equal('sandbox');
101
+ });
102
+
103
+ it('describes the license as an SPDX identifier', () => {
104
+ expect(metadata.license).to.equal('ODbL-1.0');
105
+ });
106
+
107
+ it('describes the release date as an ISO 8601 date without milliseconds', () => {
108
+ expect(metadata.releaseDate).to.match(/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z$/);
109
+ });
110
+
111
+ it('describes the first version date', () => {
112
+ expect(metadata.firstVersionDate).to.equal('2021-01-01T11:27:00Z');
113
+ });
114
+
115
+ it('describes the last version date', () => {
116
+ expect(metadata.lastVersionDate).to.equal('2022-01-06T11:32:47Z');
117
+ });
118
+
119
+ it('counts the services', () => {
120
+ expect(metadata.servicesCount).to.equal(2);
121
+ });
122
+
123
+ it('counts the distinct terms', () => {
124
+ expect(metadata.termsCount).to.equal(3);
125
+ });
126
+
127
+ it('counts the versions', () => {
128
+ expect(metadata.versionsCount).to.equal(4);
129
+ });
130
+
131
+ it('describes the archive size in bytes', async () => {
132
+ expect(metadata.size).to.equal((await fs.stat(storage.archivePath(metadata.filename))).size);
133
+ });
134
+
135
+ it('describes the archive SHA-256 checksum', async () => {
136
+ expect(metadata.sha256).to.equal(createHash('sha256').update(await fs.readFile(storage.archivePath(metadata.filename))).digest('hex'));
137
+ });
138
+ });
139
+
140
+ context('when removing previous archives fails', () => {
141
+ let error;
142
+
143
+ before(async function () {
144
+ this.timeout(10000);
145
+
146
+ const staleArchivePath = storage.archivePath(metadata.filename); // the archive from the outer release(), now standing in the way of this second one
147
+ const originalRm = fs.rm.bind(fs);
148
+
149
+ sinon.stub(fs, 'rm').callsFake((path, options) => (path === staleArchivePath ? Promise.reject(new Error('Permission denied')) : originalRm(path, options)));
150
+
151
+ try {
152
+ await release({ fileName: 'second-release' });
153
+ } catch (releaseError) {
154
+ error = releaseError;
155
+ }
156
+ });
157
+
158
+ after(() => {
159
+ sinon.restore();
160
+ });
161
+
162
+ it('resolves despite the cleanup failure', () => {
163
+ expect(error).to.be.undefined;
164
+ });
165
+ });
166
+
167
+ context('when a custom file name is given without extension', () => {
168
+ let customMetadata;
169
+
170
+ before(async function () {
171
+ this.timeout(10000);
172
+ await release({ fileName: 'custom' });
173
+ customMetadata = await storage.findLatest();
174
+ });
175
+
176
+ it('names the archive after the given file name', () => {
177
+ expect(customMetadata.filename).to.equal('custom.zip');
178
+ });
179
+ });
180
+
181
+ context('when a custom file name is given with a .zip extension', () => {
182
+ let customMetadata;
183
+
184
+ before(async function () {
185
+ this.timeout(10000);
186
+ await release({ fileName: 'custom.zip' });
187
+ customMetadata = await storage.findLatest();
188
+ });
189
+
190
+ it('names the archive after the given file name', () => {
191
+ expect(customMetadata.filename).to.equal('custom.zip');
192
+ });
193
+ });
194
+ });
195
+ });
@@ -33,6 +33,7 @@ logger.format = combine(
33
33
 
34
34
  export function createModuleLogger(moduleName) {
35
35
  return {
36
+ debug: message => logger.debug(message, { module: moduleName }),
36
37
  info: message => logger.info(message, { module: moduleName }),
37
38
  warn: message => logger.warn(message, { module: moduleName }),
38
39
  error: message => logger.error(message, { module: moduleName }),