@opentermsarchive/engine 11.0.2 → 12.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.eslintrc.yaml CHANGED
@@ -37,6 +37,9 @@ rules:
37
37
  - error
38
38
  - always-multiline
39
39
  consistent-return: 0
40
+ curly:
41
+ - error
42
+ - all
40
43
  function-paren-newline:
41
44
  - error
42
45
  - multiline
package/README.md CHANGED
@@ -2,6 +2,8 @@
2
2
 
3
3
  This codebase is a Node.js module enabling downloading, archiving and publishing versions of terms obtained online. It can be used independently from the Open Terms Archive ecosystem. For a high-level overview of Open Terms Archive’s wider goals and processes, please read its [public homepage](https://opentermsarchive.org).
4
4
 
5
+ [![DPG Badge](https://img.shields.io/badge/Verified-DPG-3333AB?logo=data:image/svg%2bxml;base64,PHN2ZyB3aWR0aD0iMzEiIGhlaWdodD0iMzMiIHZpZXdCb3g9IjAgMCAzMSAzMyIgZmlsbD0ibm9uZSIgeG1sbnM9Imh0dHA6Ly93d3cudzMub3JnLzIwMDAvc3ZnIj4KPHBhdGggZD0iTTE0LjIwMDggMjEuMzY3OEwxMC4xNzM2IDE4LjAxMjRMMTEuNTIxOSAxNi40MDAzTDEzLjk5MjggMTguNDU5TDE5LjYyNjkgMTIuMjExMUwyMS4xOTA5IDEzLjYxNkwxNC4yMDA4IDIxLjM2NzhaTTI0LjYyNDEgOS4zNTEyN0wyNC44MDcxIDMuMDcyOTdMMTguODgxIDUuMTg2NjJMMTUuMzMxNCAtMi4zMzA4MmUtMDVMMTEuNzgyMSA1LjE4NjYyTDUuODU2MDEgMy4wNzI5N0w2LjAzOTA2IDkuMzUxMjdMMCAxMS4xMTc3TDMuODQ1MjEgMTYuMDg5NUwwIDIxLjA2MTJMNi4wMzkwNiAyMi44Mjc3TDUuODU2MDEgMjkuMTA2TDExLjc4MjEgMjYuOTkyM0wxNS4zMzE0IDMyLjE3OUwxOC44ODEgMjYuOTkyM0wyNC44MDcxIDI5LjEwNkwyNC42MjQxIDIyLjgyNzdMMzAuNjYzMSAyMS4wNjEyTDI2LjgxNzYgMTYuMDg5NUwzMC42NjMxIDExLjExNzdMMjQuNjI0MSA5LjM1MTI3WiIgZmlsbD0id2hpdGUiLz4KPC9zdmc+Cg==)](https://digitalpublicgoods.net/r/open-terms-archive)
6
+
5
7
  For documentation, visit [docs.opentermsarchive.org](https://docs.opentermsarchive.org/)
6
8
 
7
9
  ## Testing
@@ -47,6 +47,11 @@
47
47
  },
48
48
  "dataset": {
49
49
  "publishingSchedule": "30 8 * * MON"
50
+ },
51
+ "collection-api": {
52
+ "feed": {
53
+ "limit": 100
54
+ }
50
55
  }
51
56
  }
52
57
  }
package/config/test.json CHANGED
@@ -47,7 +47,10 @@
47
47
  },
48
48
  "collection-api": {
49
49
  "port": 3000,
50
- "basePath": "/collection-api"
50
+ "basePath": "/collection-api",
51
+ "feed": {
52
+ "limit": 3
53
+ }
51
54
  }
52
55
  }
53
56
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@opentermsarchive/engine",
3
- "version": "11.0.2",
3
+ "version": "12.0.1",
4
4
  "description": "Tracks and makes visible changes to the terms of online services",
5
5
  "homepage": "https://opentermsarchive.org",
6
6
  "bugs": {
@@ -100,7 +100,8 @@
100
100
  "swagger-ui-express": "^5.0.1",
101
101
  "turndown": "^7.2.1",
102
102
  "winston": "^3.17.0",
103
- "winston-mail": "^2.0.0"
103
+ "winston-mail": "^2.0.0",
104
+ "xml-js": "^1.6.11"
104
105
  },
105
106
  "devDependencies": {
106
107
  "@commitlint/cli": "^19.8.1",
@@ -39,7 +39,7 @@ async function removeDuplicateIssues() {
39
39
  }
40
40
 
41
41
  for (const [ title, duplicateIssues ] of issuesByTitle) {
42
- if (duplicateIssues.length === 1) continue;
42
+ if (duplicateIssues.length === 1) { continue; }
43
43
 
44
44
  const originalIssue = duplicateIssues.reduce((oldest, current) => (new Date(current.created_at) < new Date(oldest.created_at) ? current : oldest));
45
45
 
@@ -18,7 +18,7 @@ describe('Collection', () => {
18
18
  try {
19
19
  metadataBackup = await fs.readFile(metadataPath, 'utf8');
20
20
  } catch (error) {
21
- if (error.code !== 'ENOENT') throw error;
21
+ if (error.code !== 'ENOENT') { throw error; }
22
22
  }
23
23
  });
24
24
 
@@ -3,6 +3,13 @@
3
3
  * @class Record
4
4
  * @private
5
5
  */
6
+
7
+ export const TITLE_PREFIXES = Object.freeze({
8
+ firstRecord: 'First record of',
9
+ technicalUpgrade: 'Apply technical or declaration upgrade on',
10
+ update: 'Record new changes of',
11
+ });
12
+
6
13
  export default class Record {
7
14
  #content;
8
15
 
@@ -32,6 +39,20 @@ export default class Record {
32
39
  this.#content = content;
33
40
  }
34
41
 
42
+ get displayTitle() {
43
+ let prefix;
44
+
45
+ if (this.isFirstRecord) {
46
+ prefix = TITLE_PREFIXES.firstRecord;
47
+ } else if (this.isTechnicalUpgrade) {
48
+ prefix = TITLE_PREFIXES.technicalUpgrade;
49
+ } else {
50
+ prefix = TITLE_PREFIXES.update;
51
+ }
52
+
53
+ return `${prefix} ${this.serviceId} ${this.termsType}`;
54
+ }
55
+
35
56
  validate() {
36
57
  for (const requiredParam of this.constructor.REQUIRED_PARAMS) {
37
58
  if (requiredParam == 'content') {
@@ -2,18 +2,27 @@ import path from 'path';
2
2
 
3
3
  import mime from 'mime';
4
4
 
5
+ import { TITLE_PREFIXES } from '../../record.js';
5
6
  import Snapshot from '../../snapshot.js';
6
7
  import Version from '../../version.js';
7
8
 
8
- export const COMMIT_MESSAGE_PREFIXES = {
9
- startTracking: 'First record of',
10
- technicalUpgrade: 'Apply technical or declaration upgrade on',
11
- update: 'Record new changes of',
9
+ // Prefixes for commits that represent an actual content change detected at the service source
10
+ const CHANGE_PREFIXES = {
11
+ startTracking: TITLE_PREFIXES.firstRecord,
12
+ update: TITLE_PREFIXES.update,
12
13
  deprecated_startTracking: 'Start tracking',
13
- deprecated_refilter: 'Refilter',
14
14
  deprecated_update: 'Update',
15
15
  };
16
16
 
17
+ // Prefixes for commits that re-render an existing snapshot (e.g. with updated extraction rules) without any change at the service source
18
+ const TECHNICAL_UPGRADE_PREFIXES = {
19
+ technicalUpgrade: TITLE_PREFIXES.technicalUpgrade,
20
+ deprecated_refilter: 'Refilter',
21
+ };
22
+
23
+ export const CHANGE_COMMIT_MESSAGE_PREFIXES = CHANGE_PREFIXES;
24
+ export const COMMIT_MESSAGE_PREFIXES = { ...CHANGE_PREFIXES, ...TECHNICAL_UPGRADE_PREFIXES };
25
+
17
26
  export const TERMS_TYPE_AND_DOCUMENT_ID_SEPARATOR = ' #';
18
27
  export const SNAPSHOT_ID_MARKER = '%SNAPSHOT_ID';
19
28
  const SINGLE_SOURCE_DOCUMENT_PREFIX = 'This version was recorded after extracting from snapshot';
@@ -22,13 +31,9 @@ const MULTIPLE_SOURCE_DOCUMENTS_PREFIX = 'This version was recorded after extrac
22
31
  export const COMMIT_MESSAGE_PREFIXES_REGEXP = new RegExp(`^(${Object.values(COMMIT_MESSAGE_PREFIXES).join('|')})`);
23
32
 
24
33
  export function toPersistence(record, snapshotIdentiferTemplate) {
25
- const { serviceId, termsType, documentId, isTechnicalUpgrade, snapshotIds = [], mimeType, isFirstRecord, metadata } = record;
34
+ const { serviceId, termsType, documentId, snapshotIds = [], mimeType, metadata } = record;
26
35
 
27
- let prefix = isTechnicalUpgrade ? COMMIT_MESSAGE_PREFIXES.technicalUpgrade : COMMIT_MESSAGE_PREFIXES.update;
28
-
29
- prefix = isFirstRecord ? COMMIT_MESSAGE_PREFIXES.startTracking : prefix;
30
-
31
- const subject = `${prefix} ${serviceId} ${termsType}`;
36
+ const subject = record.displayTitle;
32
37
  const documentIdMessage = `${documentId ? `Document ID ${documentId}\n\n` : ''}`;
33
38
  let snapshotIdsMessage;
34
39
 
@@ -91,6 +96,10 @@ function generateFileName(termsType, documentId, extension) {
91
96
  }
92
97
 
93
98
  export function generateFilePath(serviceId, termsType, documentId, mimeType) {
99
+ if (termsType === undefined) {
100
+ return `${serviceId}/*`; // If only serviceId is provided, return a pattern to match all files for that service
101
+ }
102
+
94
103
  const extension = mime.getExtension(mimeType) || '*'; // If mime type is undefined, an asterisk is set as an extension. Used to match all files for the given service ID, terms type and document ID when mime type is unknown
95
104
 
96
105
  return `${serviceId}/${generateFileName(termsType, documentId, extension)}`; // Do not use `path.join` as even for Windows, the path should be with `/` and not `\`
@@ -68,8 +68,20 @@ export default class Git {
68
68
  return this.git.push();
69
69
  }
70
70
 
71
- listCommits(options = []) {
72
- return this.log([ '--reverse', '--no-merges', '--name-only', ...options ]); // Returns all commits in chronological order (`--reverse`), excluding merge commits (`--no-merges`), with modified files names (`--name-only`)
71
+ listCommits(options = [], { reverse = true, skip, maxCount } = {}) {
72
+ const reverseOption = reverse ? ['--reverse'] : [];
73
+ const skipOption = skip !== undefined ? [`--skip=${skip}`] : [];
74
+ const maxCountOption = maxCount !== undefined ? [`--max-count=${maxCount}`] : [];
75
+
76
+ return this.log([
77
+ ...reverseOption, // When `reverse` is true, lists commits oldest-first; otherwise the default newest-first applies
78
+ '--author-date-order', // Best-effort author-date ordering: with --max-count, git applies the cap topologically, so the page can miss strictly-newer commits that #getCommits' JS resort cannot recover
79
+ '--no-merges', // Exclude merge commits — records are stored as regular commits, never as merges
80
+ '--name-only', // Append the modified file names below each commit, used by `toDomain` to extract the record's file path
81
+ ...skipOption, // Optional `--skip=N`: drop the first N matching commits (pagination offset)
82
+ ...maxCountOption, // Optional `--max-count=N`: cap the result to N commits (pagination limit)
83
+ ...options, // Caller-supplied options: typically grep filters on commit messages and a path filter (`-- pathspec`)
84
+ ]);
73
85
  }
74
86
 
75
87
  async getCommit(options) {
@@ -88,16 +88,43 @@ export default class GitRepository extends RepositoryInterface {
88
88
  return this.#toDomain(commit);
89
89
  }
90
90
 
91
- async findAll() {
92
- return Promise.all((await this.#getCommits()).map(commit => this.#toDomain(commit, { deferContentLoading: true })));
91
+ async findAll({ limit, offset, includeTechnicalUpgrades = true } = {}) {
92
+ return Promise.all((await this.#getCommits({ limit, offset, includeTechnicalUpgrades })).map(commit => this.#toDomain(commit, { deferContentLoading: true })));
93
93
  }
94
94
 
95
- async count() {
96
- return (await this.git.log(Object.values(DataMapper.COMMIT_MESSAGE_PREFIXES).map(prefix => `--grep=${prefix}`))).length;
95
+ async findByService(serviceId, { limit, offset, includeTechnicalUpgrades = true } = {}) {
96
+ const pathPattern = DataMapper.generateFilePath(serviceId);
97
+
98
+ return Promise.all((await this.#getCommits({ pathFilter: pathPattern, limit, offset, includeTechnicalUpgrades })).map(commit => this.#toDomain(commit, { deferContentLoading: true })));
99
+ }
100
+
101
+ async findByServiceAndTermsType(serviceId, termsType, { limit, offset, includeTechnicalUpgrades = true } = {}) {
102
+ const pathPattern = DataMapper.generateFilePath(serviceId, termsType);
103
+
104
+ return Promise.all((await this.#getCommits({ pathFilter: pathPattern, limit, offset, includeTechnicalUpgrades })).map(commit => this.#toDomain(commit, { deferContentLoading: true })));
105
+ }
106
+
107
+ async count(serviceId, termsType) {
108
+ const grepOptions = Object.values(DataMapper.COMMIT_MESSAGE_PREFIXES).map(prefix => `--grep=${prefix}`);
109
+ const pathOptions = [];
110
+
111
+ if (serviceId && termsType) {
112
+ const pathPattern = DataMapper.generateFilePath(serviceId, termsType);
113
+
114
+ pathOptions.push('--', pathPattern);
115
+ } else if (serviceId) {
116
+ const pathPattern = DataMapper.generateFilePath(serviceId);
117
+
118
+ pathOptions.push('--', pathPattern);
119
+ } else {
120
+ pathOptions.push('--', '*/*'); // Count all records (exclude root directory files)
121
+ }
122
+
123
+ return (await this.git.log([ ...grepOptions, ...pathOptions ])).length;
97
124
  }
98
125
 
99
126
  async* iterate() {
100
- const commits = await this.#getCommits();
127
+ const commits = await this.#getCommits({ reverse: true });
101
128
 
102
129
  for (const commit of commits) {
103
130
  yield this.#toDomain(commit);
@@ -131,12 +158,41 @@ export default class GitRepository extends RepositoryInterface {
131
158
  record.content = pdfBuffer;
132
159
  }
133
160
 
134
- async #getCommits() {
135
- return (await this.git.listCommits())
136
- .filter(commit => // Skip non-record commits (e.g., README or LICENSE updates)
137
- DataMapper.COMMIT_MESSAGE_PREFIXES_REGEXP.test(commit.message) // Commits generated by the engine have messages that match predefined prefixes
138
- && path.dirname(commit.diff.files[0].file) !== '.') // Assumes one record per commit; records must be in a serviceId folder, not root
139
- .sort((commitA, commitB) => new Date(commitA.date) - new Date(commitB.date)); // Make sure that the commits are sorted in ascending chronological order
161
+ async #getCommits({ pathFilter, reverse = false, limit, offset, includeTechnicalUpgrades = true } = {}) {
162
+ const prefixes = includeTechnicalUpgrades
163
+ ? DataMapper.COMMIT_MESSAGE_PREFIXES
164
+ : DataMapper.CHANGE_COMMIT_MESSAGE_PREFIXES;
165
+ const grepOptions = Object.values(prefixes).flatMap(prefix => [ '--grep', prefix ]);
166
+ const pathOptions = pathFilter
167
+ ? [ '--', pathFilter ]
168
+ : [ '--', '*/*' ]; // Exclude root directory files by only matching files in subdirectories
169
+
170
+ const options = [ ...grepOptions, ...pathOptions ];
171
+
172
+ // Use git-level pagination for performance: `--skip` and `--max-count` count in topological order, not strictly chronological.
173
+ // In records history, the only commits whose author date is out of step with their topological position are technical upgrades.
174
+ // The only caller currently relying on pagination is the feed endpoint, which already filters technical upgrades out via `includeTechnicalUpgrades: false`, so the paginated set has no chronological/topological divergence in practice.
175
+ // If a future caller needs paginated access that includes technical upgrades, switch to the approach proposed in https://github.com/OpenTermsArchive/engine/issues/1243.
176
+ const paginationOptions = {};
177
+
178
+ if (offset !== undefined) {
179
+ paginationOptions.skip = offset;
180
+ }
181
+
182
+ if (limit !== undefined) {
183
+ paginationOptions.maxCount = limit;
184
+ }
185
+
186
+ const commits = await this.git.listCommits(options, { reverse: false, ...paginationOptions }); // Get commits without git's --reverse for better performance, filtered at git level
187
+
188
+ commits.sort((commitA, commitB) => {
189
+ const dateA = new Date(commitA.date);
190
+ const dateB = new Date(commitB.date);
191
+
192
+ return reverse ? dateA - dateB : dateB - dateA;
193
+ });
194
+
195
+ return commits;
140
196
  }
141
197
 
142
198
  static async writeFile({ filePath, content }) {
@@ -540,8 +540,253 @@ describe('GitRepository', () => {
540
540
  }
541
541
  });
542
542
 
543
- it('returns records in ascending order', () => {
544
- expect(records.map(record => record.fetchDate)).to.deep.equal([ FETCH_DATE_EARLIER, FETCH_DATE, FETCH_DATE_LATER ]);
543
+ it('returns records in descending order', () => {
544
+ expect(records.map(record => record.fetchDate)).to.deep.equal([ FETCH_DATE_LATER, FETCH_DATE, FETCH_DATE_EARLIER ]);
545
+ });
546
+
547
+ context('with includeTechnicalUpgrades: false', () => {
548
+ let filteredRecords;
549
+
550
+ before(async () => {
551
+ filteredRecords = await subject.findAll({ includeTechnicalUpgrades: false });
552
+ });
553
+
554
+ it('excludes technical upgrade records', () => {
555
+ expect(filteredRecords.length).to.equal(2);
556
+ });
557
+
558
+ it('only returns records that represent actual content changes', () => {
559
+ for (const record of filteredRecords) {
560
+ expect(record.isTechnicalUpgrade).to.not.be.true;
561
+ }
562
+ });
563
+
564
+ it('returns the expected records in descending order', () => {
565
+ expect(filteredRecords.map(record => record.fetchDate)).to.deep.equal([ FETCH_DATE_LATER, FETCH_DATE ]);
566
+ });
567
+ });
568
+ });
569
+
570
+ describe('#findByServiceAndTermsType', () => {
571
+ const expectedIds = [];
572
+ let records;
573
+
574
+ before(async function () {
575
+ this.timeout(5000);
576
+
577
+ const { id: id1 } = await subject.save(new Version({
578
+ serviceId: SERVICE_PROVIDER_ID,
579
+ termsType: TERMS_TYPE,
580
+ content: CONTENT,
581
+ fetchDate: FETCH_DATE,
582
+ snapshotIds: [SNAPSHOT_ID],
583
+ }));
584
+
585
+ expectedIds.push(id1);
586
+
587
+ const { id: id2 } = await subject.save(new Version({
588
+ serviceId: SERVICE_PROVIDER_ID,
589
+ termsType: TERMS_TYPE,
590
+ content: `${CONTENT} - updated`,
591
+ fetchDate: FETCH_DATE_LATER,
592
+ snapshotIds: [SNAPSHOT_ID],
593
+ }));
594
+
595
+ expectedIds.push(id2);
596
+
597
+ await subject.save(new Version({
598
+ serviceId: 'other_service',
599
+ termsType: 'Privacy Policy',
600
+ content: `${CONTENT} - other`,
601
+ fetchDate: FETCH_DATE,
602
+ snapshotIds: [SNAPSHOT_ID],
603
+ }));
604
+
605
+ (records = await subject.findByServiceAndTermsType(SERVICE_PROVIDER_ID, TERMS_TYPE));
606
+ });
607
+
608
+ after(() => subject.removeAll());
609
+
610
+ it('returns only matching records', () => {
611
+ expect(records.length).to.equal(2);
612
+ });
613
+
614
+ it('returns Version objects', () => {
615
+ for (const record of records) {
616
+ expect(record).to.be.an.instanceof(Version);
617
+ }
618
+ });
619
+
620
+ it('returns records with matching service ID', () => {
621
+ for (const record of records) {
622
+ expect(record.serviceId).to.equal(SERVICE_PROVIDER_ID);
623
+ }
624
+ });
625
+
626
+ it('returns records with matching terms type', () => {
627
+ for (const record of records) {
628
+ expect(record.termsType).to.equal(TERMS_TYPE);
629
+ }
630
+ });
631
+
632
+ it('returns records in descending order', () => {
633
+ expect(records.map(record => record.fetchDate)).to.deep.equal([ FETCH_DATE_LATER, FETCH_DATE ]);
634
+ });
635
+
636
+ it('returns records with correct IDs', () => {
637
+ expect(records.map(record => record.id)).to.have.members(expectedIds);
638
+ });
639
+
640
+ context('when no matching records exist', () => {
641
+ it('returns an empty array', async () => {
642
+ const result = await subject.findByServiceAndTermsType('non_existent_service', 'Non Existent Terms');
643
+
644
+ expect(result).to.be.an('array').that.is.empty;
645
+ });
646
+ });
647
+
648
+ context('with includeTechnicalUpgrades: false', () => {
649
+ let filteredRecords;
650
+ let technicalUpgradeId;
651
+
652
+ before(async () => {
653
+ ({ id: technicalUpgradeId } = await subject.save(new Version({
654
+ serviceId: SERVICE_PROVIDER_ID,
655
+ termsType: TERMS_TYPE,
656
+ content: `${CONTENT} - technical upgrade`,
657
+ fetchDate: FETCH_DATE_EARLIER,
658
+ snapshotIds: [SNAPSHOT_ID],
659
+ isTechnicalUpgrade: true,
660
+ })));
661
+
662
+ filteredRecords = await subject.findByServiceAndTermsType(SERVICE_PROVIDER_ID, TERMS_TYPE, { includeTechnicalUpgrades: false });
663
+ });
664
+
665
+ it('excludes technical upgrade records', () => {
666
+ expect(filteredRecords.map(record => record.id)).to.not.include(technicalUpgradeId);
667
+ });
668
+
669
+ it('only returns records that represent actual content changes', () => {
670
+ for (const record of filteredRecords) {
671
+ expect(record.isTechnicalUpgrade).to.not.be.true;
672
+ }
673
+ });
674
+ });
675
+ });
676
+
677
+ describe('#findByService', () => {
678
+ const OTHER_TERMS_TYPE = 'Privacy Policy';
679
+ const expectedIds = [];
680
+ let records;
681
+
682
+ before(async function () {
683
+ this.timeout(5000);
684
+
685
+ const { id: id1 } = await subject.save(new Version({
686
+ serviceId: SERVICE_PROVIDER_ID,
687
+ termsType: TERMS_TYPE,
688
+ content: CONTENT,
689
+ fetchDate: FETCH_DATE,
690
+ snapshotIds: [SNAPSHOT_ID],
691
+ }));
692
+
693
+ expectedIds.push(id1);
694
+
695
+ const { id: id2 } = await subject.save(new Version({
696
+ serviceId: SERVICE_PROVIDER_ID,
697
+ termsType: TERMS_TYPE,
698
+ content: `${CONTENT} - updated`,
699
+ fetchDate: FETCH_DATE_LATER,
700
+ snapshotIds: [SNAPSHOT_ID],
701
+ }));
702
+
703
+ expectedIds.push(id2);
704
+
705
+ const { id: id3 } = await subject.save(new Version({
706
+ serviceId: SERVICE_PROVIDER_ID,
707
+ termsType: OTHER_TERMS_TYPE,
708
+ content: `${CONTENT} - other terms type`,
709
+ fetchDate: FETCH_DATE_EARLIER,
710
+ snapshotIds: [SNAPSHOT_ID],
711
+ }));
712
+
713
+ expectedIds.push(id3);
714
+
715
+ await subject.save(new Version({
716
+ serviceId: 'other_service',
717
+ termsType: TERMS_TYPE,
718
+ content: `${CONTENT} - other service`,
719
+ fetchDate: FETCH_DATE,
720
+ snapshotIds: [SNAPSHOT_ID],
721
+ }));
722
+
723
+ (records = await subject.findByService(SERVICE_PROVIDER_ID));
724
+ });
725
+
726
+ after(() => subject.removeAll());
727
+
728
+ it('returns only matching records', () => {
729
+ expect(records.length).to.equal(3);
730
+ });
731
+
732
+ it('returns Version objects', () => {
733
+ for (const record of records) {
734
+ expect(record).to.be.an.instanceof(Version);
735
+ }
736
+ });
737
+
738
+ it('returns records with matching service ID', () => {
739
+ for (const record of records) {
740
+ expect(record.serviceId).to.equal(SERVICE_PROVIDER_ID);
741
+ }
742
+ });
743
+
744
+ it('returns records across all terms types of the service', () => {
745
+ expect(new Set(records.map(record => record.termsType))).to.deep.equal(new Set([ TERMS_TYPE, OTHER_TERMS_TYPE ]));
746
+ });
747
+
748
+ it('returns records in descending order', () => {
749
+ expect(records.map(record => record.fetchDate)).to.deep.equal([ FETCH_DATE_LATER, FETCH_DATE, FETCH_DATE_EARLIER ]);
750
+ });
751
+
752
+ it('returns records with correct IDs', () => {
753
+ expect(records.map(record => record.id)).to.have.members(expectedIds);
754
+ });
755
+
756
+ context('when no matching records exist', () => {
757
+ it('returns an empty array', async () => {
758
+ const result = await subject.findByService('non_existent_service');
759
+
760
+ expect(result).to.be.an('array').that.is.empty;
761
+ });
762
+ });
763
+
764
+ context('with includeTechnicalUpgrades: false', () => {
765
+ let filteredRecords;
766
+ let technicalUpgradeId;
767
+
768
+ before(async () => {
769
+ ({ id: technicalUpgradeId } = await subject.save(new Version({
770
+ serviceId: SERVICE_PROVIDER_ID,
771
+ termsType: TERMS_TYPE,
772
+ content: `${CONTENT} - technical upgrade`,
773
+ fetchDate: new Date('2000-01-03T12:00:00.000Z'),
774
+ snapshotIds: [SNAPSHOT_ID],
775
+ isTechnicalUpgrade: true,
776
+ })));
777
+
778
+ filteredRecords = await subject.findByService(SERVICE_PROVIDER_ID, { includeTechnicalUpgrades: false });
779
+ });
780
+
781
+ it('excludes technical upgrade records', () => {
782
+ expect(filteredRecords.map(record => record.id)).to.not.include(technicalUpgradeId);
783
+ });
784
+
785
+ it('only returns records that represent actual content changes', () => {
786
+ for (const record of filteredRecords) {
787
+ expect(record.isTechnicalUpgrade).to.not.be.true;
788
+ }
789
+ });
545
790
  });
546
791
  });
547
792
 
@@ -582,6 +827,37 @@ describe('GitRepository', () => {
582
827
  it('returns the proper count', () => {
583
828
  expect(count).to.equal(3);
584
829
  });
830
+
831
+ context('with serviceId and termsType filters', () => {
832
+ it('returns count for specific service and terms type', async () => {
833
+ const filteredCount = await subject.count(SERVICE_PROVIDER_ID, TERMS_TYPE);
834
+
835
+ expect(filteredCount).to.equal(3);
836
+ });
837
+
838
+ it('returns zero for non-existent service', async () => {
839
+ const filteredCount = await subject.count('non-existent-service', TERMS_TYPE);
840
+
841
+ expect(filteredCount).to.equal(0);
842
+ });
843
+ });
844
+
845
+ context('with only serviceId filter', () => {
846
+ it('returns count for all terms types of a service', async () => {
847
+ // Add a version with different terms type
848
+ await subject.save(new Version({
849
+ serviceId: SERVICE_PROVIDER_ID,
850
+ termsType: 'Different Terms',
851
+ content: CONTENT,
852
+ fetchDate: FETCH_DATE,
853
+ snapshotIds: [SNAPSHOT_ID],
854
+ }));
855
+
856
+ const filteredCount = await subject.count(SERVICE_PROVIDER_ID);
857
+
858
+ expect(filteredCount).to.equal(4); // 3 from TERMS_TYPE + 1 from 'Different Terms'
859
+ });
860
+ });
585
861
  });
586
862
 
587
863
  describe('#findLatest', () => {
@@ -1101,8 +1377,8 @@ describe('GitRepository', () => {
1101
1377
  }
1102
1378
  });
1103
1379
 
1104
- it('returns records in ascending order', () => {
1105
- expect(records.map(record => record.fetchDate)).to.deep.equal([ FETCH_DATE_EARLIER, FETCH_DATE, FETCH_DATE_LATER ]);
1380
+ it('returns records in descending order', () => {
1381
+ expect(records.map(record => record.fetchDate)).to.deep.equal([ FETCH_DATE_LATER, FETCH_DATE, FETCH_DATE_EARLIER ]);
1106
1382
  });
1107
1383
  });
1108
1384
 
@@ -1462,8 +1738,8 @@ describe('GitRepository', () => {
1462
1738
  }
1463
1739
  });
1464
1740
 
1465
- it('returns records in ascending order', () => {
1466
- expect(records.map(record => record.fetchDate)).to.deep.equal(expectedDates);
1741
+ it('returns records in descending order', () => {
1742
+ expect(records.map(record => record.fetchDate)).to.deep.equal([...expectedDates].reverse());
1467
1743
  });
1468
1744
  });
1469
1745