@opentermsarchive/engine 11.0.2 → 12.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.eslintrc.yaml CHANGED
@@ -37,6 +37,9 @@ rules:
37
37
  - error
38
38
  - always-multiline
39
39
  consistent-return: 0
40
+ curly:
41
+ - error
42
+ - all
40
43
  function-paren-newline:
41
44
  - error
42
45
  - multiline
@@ -47,6 +47,11 @@
47
47
  },
48
48
  "dataset": {
49
49
  "publishingSchedule": "30 8 * * MON"
50
+ },
51
+ "collection-api": {
52
+ "feed": {
53
+ "limit": 100
54
+ }
50
55
  }
51
56
  }
52
57
  }
package/config/test.json CHANGED
@@ -47,7 +47,10 @@
47
47
  },
48
48
  "collection-api": {
49
49
  "port": 3000,
50
- "basePath": "/collection-api"
50
+ "basePath": "/collection-api",
51
+ "feed": {
52
+ "limit": 3
53
+ }
51
54
  }
52
55
  }
53
56
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@opentermsarchive/engine",
3
- "version": "11.0.2",
3
+ "version": "12.0.0",
4
4
  "description": "Tracks and makes visible changes to the terms of online services",
5
5
  "homepage": "https://opentermsarchive.org",
6
6
  "bugs": {
@@ -100,7 +100,8 @@
100
100
  "swagger-ui-express": "^5.0.1",
101
101
  "turndown": "^7.2.1",
102
102
  "winston": "^3.17.0",
103
- "winston-mail": "^2.0.0"
103
+ "winston-mail": "^2.0.0",
104
+ "xml-js": "^1.6.11"
104
105
  },
105
106
  "devDependencies": {
106
107
  "@commitlint/cli": "^19.8.1",
@@ -39,7 +39,7 @@ async function removeDuplicateIssues() {
39
39
  }
40
40
 
41
41
  for (const [ title, duplicateIssues ] of issuesByTitle) {
42
- if (duplicateIssues.length === 1) continue;
42
+ if (duplicateIssues.length === 1) { continue; }
43
43
 
44
44
  const originalIssue = duplicateIssues.reduce((oldest, current) => (new Date(current.created_at) < new Date(oldest.created_at) ? current : oldest));
45
45
 
@@ -18,7 +18,7 @@ describe('Collection', () => {
18
18
  try {
19
19
  metadataBackup = await fs.readFile(metadataPath, 'utf8');
20
20
  } catch (error) {
21
- if (error.code !== 'ENOENT') throw error;
21
+ if (error.code !== 'ENOENT') { throw error; }
22
22
  }
23
23
  });
24
24
 
@@ -3,6 +3,13 @@
3
3
  * @class Record
4
4
  * @private
5
5
  */
6
+
7
+ export const TITLE_PREFIXES = Object.freeze({
8
+ firstRecord: 'First record of',
9
+ technicalUpgrade: 'Apply technical or declaration upgrade on',
10
+ update: 'Record new changes of',
11
+ });
12
+
6
13
  export default class Record {
7
14
  #content;
8
15
 
@@ -32,6 +39,20 @@ export default class Record {
32
39
  this.#content = content;
33
40
  }
34
41
 
42
+ get displayTitle() {
43
+ let prefix;
44
+
45
+ if (this.isFirstRecord) {
46
+ prefix = TITLE_PREFIXES.firstRecord;
47
+ } else if (this.isTechnicalUpgrade) {
48
+ prefix = TITLE_PREFIXES.technicalUpgrade;
49
+ } else {
50
+ prefix = TITLE_PREFIXES.update;
51
+ }
52
+
53
+ return `${prefix} ${this.serviceId} ${this.termsType}`;
54
+ }
55
+
35
56
  validate() {
36
57
  for (const requiredParam of this.constructor.REQUIRED_PARAMS) {
37
58
  if (requiredParam == 'content') {
@@ -2,18 +2,27 @@ import path from 'path';
2
2
 
3
3
  import mime from 'mime';
4
4
 
5
+ import { TITLE_PREFIXES } from '../../record.js';
5
6
  import Snapshot from '../../snapshot.js';
6
7
  import Version from '../../version.js';
7
8
 
8
- export const COMMIT_MESSAGE_PREFIXES = {
9
- startTracking: 'First record of',
10
- technicalUpgrade: 'Apply technical or declaration upgrade on',
11
- update: 'Record new changes of',
9
+ // Prefixes for commits that represent an actual content change detected at the service source
10
+ const CHANGE_PREFIXES = {
11
+ startTracking: TITLE_PREFIXES.firstRecord,
12
+ update: TITLE_PREFIXES.update,
12
13
  deprecated_startTracking: 'Start tracking',
13
- deprecated_refilter: 'Refilter',
14
14
  deprecated_update: 'Update',
15
15
  };
16
16
 
17
+ // Prefixes for commits that re-render an existing snapshot (e.g. with updated extraction rules) without any change at the service source
18
+ const TECHNICAL_UPGRADE_PREFIXES = {
19
+ technicalUpgrade: TITLE_PREFIXES.technicalUpgrade,
20
+ deprecated_refilter: 'Refilter',
21
+ };
22
+
23
+ export const CHANGE_COMMIT_MESSAGE_PREFIXES = CHANGE_PREFIXES;
24
+ export const COMMIT_MESSAGE_PREFIXES = { ...CHANGE_PREFIXES, ...TECHNICAL_UPGRADE_PREFIXES };
25
+
17
26
  export const TERMS_TYPE_AND_DOCUMENT_ID_SEPARATOR = ' #';
18
27
  export const SNAPSHOT_ID_MARKER = '%SNAPSHOT_ID';
19
28
  const SINGLE_SOURCE_DOCUMENT_PREFIX = 'This version was recorded after extracting from snapshot';
@@ -22,13 +31,9 @@ const MULTIPLE_SOURCE_DOCUMENTS_PREFIX = 'This version was recorded after extrac
22
31
  export const COMMIT_MESSAGE_PREFIXES_REGEXP = new RegExp(`^(${Object.values(COMMIT_MESSAGE_PREFIXES).join('|')})`);
23
32
 
24
33
  export function toPersistence(record, snapshotIdentiferTemplate) {
25
- const { serviceId, termsType, documentId, isTechnicalUpgrade, snapshotIds = [], mimeType, isFirstRecord, metadata } = record;
34
+ const { serviceId, termsType, documentId, snapshotIds = [], mimeType, metadata } = record;
26
35
 
27
- let prefix = isTechnicalUpgrade ? COMMIT_MESSAGE_PREFIXES.technicalUpgrade : COMMIT_MESSAGE_PREFIXES.update;
28
-
29
- prefix = isFirstRecord ? COMMIT_MESSAGE_PREFIXES.startTracking : prefix;
30
-
31
- const subject = `${prefix} ${serviceId} ${termsType}`;
36
+ const subject = record.displayTitle;
32
37
  const documentIdMessage = `${documentId ? `Document ID ${documentId}\n\n` : ''}`;
33
38
  let snapshotIdsMessage;
34
39
 
@@ -91,6 +96,10 @@ function generateFileName(termsType, documentId, extension) {
91
96
  }
92
97
 
93
98
  export function generateFilePath(serviceId, termsType, documentId, mimeType) {
99
+ if (termsType === undefined) {
100
+ return `${serviceId}/*`; // If only serviceId is provided, return a pattern to match all files for that service
101
+ }
102
+
94
103
  const extension = mime.getExtension(mimeType) || '*'; // If mime type is undefined, an asterisk is set as an extension. Used to match all files for the given service ID, terms type and document ID when mime type is unknown
95
104
 
96
105
  return `${serviceId}/${generateFileName(termsType, documentId, extension)}`; // Do not use `path.join` as even for Windows, the path should be with `/` and not `\`
@@ -68,8 +68,20 @@ export default class Git {
68
68
  return this.git.push();
69
69
  }
70
70
 
71
- listCommits(options = []) {
72
- return this.log([ '--reverse', '--no-merges', '--name-only', ...options ]); // Returns all commits in chronological order (`--reverse`), excluding merge commits (`--no-merges`), with modified files names (`--name-only`)
71
+ listCommits(options = [], { reverse = true, skip, maxCount } = {}) {
72
+ const reverseOption = reverse ? ['--reverse'] : [];
73
+ const skipOption = skip !== undefined ? [`--skip=${skip}`] : [];
74
+ const maxCountOption = maxCount !== undefined ? [`--max-count=${maxCount}`] : [];
75
+
76
+ return this.log([
77
+ ...reverseOption, // When `reverse` is true, lists commits oldest-first; otherwise the default newest-first applies
78
+ '--author-date-order', // Best-effort author-date ordering: with --max-count, git applies the cap topologically, so the page can miss strictly-newer commits that #getCommits' JS resort cannot recover
79
+ '--no-merges', // Exclude merge commits — records are stored as regular commits, never as merges
80
+ '--name-only', // Append the modified file names below each commit, used by `toDomain` to extract the record's file path
81
+ ...skipOption, // Optional `--skip=N`: drop the first N matching commits (pagination offset)
82
+ ...maxCountOption, // Optional `--max-count=N`: cap the result to N commits (pagination limit)
83
+ ...options, // Caller-supplied options: typically grep filters on commit messages and a path filter (`-- pathspec`)
84
+ ]);
73
85
  }
74
86
 
75
87
  async getCommit(options) {
@@ -88,16 +88,43 @@ export default class GitRepository extends RepositoryInterface {
88
88
  return this.#toDomain(commit);
89
89
  }
90
90
 
91
- async findAll() {
92
- return Promise.all((await this.#getCommits()).map(commit => this.#toDomain(commit, { deferContentLoading: true })));
91
+ async findAll({ limit, offset, includeTechnicalUpgrades = true } = {}) {
92
+ return Promise.all((await this.#getCommits({ limit, offset, includeTechnicalUpgrades })).map(commit => this.#toDomain(commit, { deferContentLoading: true })));
93
93
  }
94
94
 
95
- async count() {
96
- return (await this.git.log(Object.values(DataMapper.COMMIT_MESSAGE_PREFIXES).map(prefix => `--grep=${prefix}`))).length;
95
+ async findByService(serviceId, { limit, offset, includeTechnicalUpgrades = true } = {}) {
96
+ const pathPattern = DataMapper.generateFilePath(serviceId);
97
+
98
+ return Promise.all((await this.#getCommits({ pathFilter: pathPattern, limit, offset, includeTechnicalUpgrades })).map(commit => this.#toDomain(commit, { deferContentLoading: true })));
99
+ }
100
+
101
+ async findByServiceAndTermsType(serviceId, termsType, { limit, offset, includeTechnicalUpgrades = true } = {}) {
102
+ const pathPattern = DataMapper.generateFilePath(serviceId, termsType);
103
+
104
+ return Promise.all((await this.#getCommits({ pathFilter: pathPattern, limit, offset, includeTechnicalUpgrades })).map(commit => this.#toDomain(commit, { deferContentLoading: true })));
105
+ }
106
+
107
+ async count(serviceId, termsType) {
108
+ const grepOptions = Object.values(DataMapper.COMMIT_MESSAGE_PREFIXES).map(prefix => `--grep=${prefix}`);
109
+ const pathOptions = [];
110
+
111
+ if (serviceId && termsType) {
112
+ const pathPattern = DataMapper.generateFilePath(serviceId, termsType);
113
+
114
+ pathOptions.push('--', pathPattern);
115
+ } else if (serviceId) {
116
+ const pathPattern = DataMapper.generateFilePath(serviceId);
117
+
118
+ pathOptions.push('--', pathPattern);
119
+ } else {
120
+ pathOptions.push('--', '*/*'); // Count all records (exclude root directory files)
121
+ }
122
+
123
+ return (await this.git.log([ ...grepOptions, ...pathOptions ])).length;
97
124
  }
98
125
 
99
126
  async* iterate() {
100
- const commits = await this.#getCommits();
127
+ const commits = await this.#getCommits({ reverse: true });
101
128
 
102
129
  for (const commit of commits) {
103
130
  yield this.#toDomain(commit);
@@ -131,12 +158,41 @@ export default class GitRepository extends RepositoryInterface {
131
158
  record.content = pdfBuffer;
132
159
  }
133
160
 
134
- async #getCommits() {
135
- return (await this.git.listCommits())
136
- .filter(commit => // Skip non-record commits (e.g., README or LICENSE updates)
137
- DataMapper.COMMIT_MESSAGE_PREFIXES_REGEXP.test(commit.message) // Commits generated by the engine have messages that match predefined prefixes
138
- && path.dirname(commit.diff.files[0].file) !== '.') // Assumes one record per commit; records must be in a serviceId folder, not root
139
- .sort((commitA, commitB) => new Date(commitA.date) - new Date(commitB.date)); // Make sure that the commits are sorted in ascending chronological order
161
+ async #getCommits({ pathFilter, reverse = false, limit, offset, includeTechnicalUpgrades = true } = {}) {
162
+ const prefixes = includeTechnicalUpgrades
163
+ ? DataMapper.COMMIT_MESSAGE_PREFIXES
164
+ : DataMapper.CHANGE_COMMIT_MESSAGE_PREFIXES;
165
+ const grepOptions = Object.values(prefixes).flatMap(prefix => [ '--grep', prefix ]);
166
+ const pathOptions = pathFilter
167
+ ? [ '--', pathFilter ]
168
+ : [ '--', '*/*' ]; // Exclude root directory files by only matching files in subdirectories
169
+
170
+ const options = [ ...grepOptions, ...pathOptions ];
171
+
172
+ // Use git-level pagination for performance: `--skip` and `--max-count` count in topological order, not strictly chronological.
173
+ // In records history, the only commits whose author date is out of step with their topological position are technical upgrades.
174
+ // The only caller currently relying on pagination is the feed endpoint, which already filters technical upgrades out via `includeTechnicalUpgrades: false`, so the paginated set has no chronological/topological divergence in practice.
175
+ // If a future caller needs paginated access that includes technical upgrades, switch to the approach proposed in https://github.com/OpenTermsArchive/engine/issues/1243.
176
+ const paginationOptions = {};
177
+
178
+ if (offset !== undefined) {
179
+ paginationOptions.skip = offset;
180
+ }
181
+
182
+ if (limit !== undefined) {
183
+ paginationOptions.maxCount = limit;
184
+ }
185
+
186
+ const commits = await this.git.listCommits(options, { reverse: false, ...paginationOptions }); // Get commits without git's --reverse for better performance, filtered at git level
187
+
188
+ commits.sort((commitA, commitB) => {
189
+ const dateA = new Date(commitA.date);
190
+ const dateB = new Date(commitB.date);
191
+
192
+ return reverse ? dateA - dateB : dateB - dateA;
193
+ });
194
+
195
+ return commits;
140
196
  }
141
197
 
142
198
  static async writeFile({ filePath, content }) {
@@ -540,8 +540,253 @@ describe('GitRepository', () => {
540
540
  }
541
541
  });
542
542
 
543
- it('returns records in ascending order', () => {
544
- expect(records.map(record => record.fetchDate)).to.deep.equal([ FETCH_DATE_EARLIER, FETCH_DATE, FETCH_DATE_LATER ]);
543
+ it('returns records in descending order', () => {
544
+ expect(records.map(record => record.fetchDate)).to.deep.equal([ FETCH_DATE_LATER, FETCH_DATE, FETCH_DATE_EARLIER ]);
545
+ });
546
+
547
+ context('with includeTechnicalUpgrades: false', () => {
548
+ let filteredRecords;
549
+
550
+ before(async () => {
551
+ filteredRecords = await subject.findAll({ includeTechnicalUpgrades: false });
552
+ });
553
+
554
+ it('excludes technical upgrade records', () => {
555
+ expect(filteredRecords.length).to.equal(2);
556
+ });
557
+
558
+ it('only returns records that represent actual content changes', () => {
559
+ for (const record of filteredRecords) {
560
+ expect(record.isTechnicalUpgrade).to.not.be.true;
561
+ }
562
+ });
563
+
564
+ it('returns the expected records in descending order', () => {
565
+ expect(filteredRecords.map(record => record.fetchDate)).to.deep.equal([ FETCH_DATE_LATER, FETCH_DATE ]);
566
+ });
567
+ });
568
+ });
569
+
570
+ describe('#findByServiceAndTermsType', () => {
571
+ const expectedIds = [];
572
+ let records;
573
+
574
+ before(async function () {
575
+ this.timeout(5000);
576
+
577
+ const { id: id1 } = await subject.save(new Version({
578
+ serviceId: SERVICE_PROVIDER_ID,
579
+ termsType: TERMS_TYPE,
580
+ content: CONTENT,
581
+ fetchDate: FETCH_DATE,
582
+ snapshotIds: [SNAPSHOT_ID],
583
+ }));
584
+
585
+ expectedIds.push(id1);
586
+
587
+ const { id: id2 } = await subject.save(new Version({
588
+ serviceId: SERVICE_PROVIDER_ID,
589
+ termsType: TERMS_TYPE,
590
+ content: `${CONTENT} - updated`,
591
+ fetchDate: FETCH_DATE_LATER,
592
+ snapshotIds: [SNAPSHOT_ID],
593
+ }));
594
+
595
+ expectedIds.push(id2);
596
+
597
+ await subject.save(new Version({
598
+ serviceId: 'other_service',
599
+ termsType: 'Privacy Policy',
600
+ content: `${CONTENT} - other`,
601
+ fetchDate: FETCH_DATE,
602
+ snapshotIds: [SNAPSHOT_ID],
603
+ }));
604
+
605
+ (records = await subject.findByServiceAndTermsType(SERVICE_PROVIDER_ID, TERMS_TYPE));
606
+ });
607
+
608
+ after(() => subject.removeAll());
609
+
610
+ it('returns only matching records', () => {
611
+ expect(records.length).to.equal(2);
612
+ });
613
+
614
+ it('returns Version objects', () => {
615
+ for (const record of records) {
616
+ expect(record).to.be.an.instanceof(Version);
617
+ }
618
+ });
619
+
620
+ it('returns records with matching service ID', () => {
621
+ for (const record of records) {
622
+ expect(record.serviceId).to.equal(SERVICE_PROVIDER_ID);
623
+ }
624
+ });
625
+
626
+ it('returns records with matching terms type', () => {
627
+ for (const record of records) {
628
+ expect(record.termsType).to.equal(TERMS_TYPE);
629
+ }
630
+ });
631
+
632
+ it('returns records in descending order', () => {
633
+ expect(records.map(record => record.fetchDate)).to.deep.equal([ FETCH_DATE_LATER, FETCH_DATE ]);
634
+ });
635
+
636
+ it('returns records with correct IDs', () => {
637
+ expect(records.map(record => record.id)).to.have.members(expectedIds);
638
+ });
639
+
640
+ context('when no matching records exist', () => {
641
+ it('returns an empty array', async () => {
642
+ const result = await subject.findByServiceAndTermsType('non_existent_service', 'Non Existent Terms');
643
+
644
+ expect(result).to.be.an('array').that.is.empty;
645
+ });
646
+ });
647
+
648
+ context('with includeTechnicalUpgrades: false', () => {
649
+ let filteredRecords;
650
+ let technicalUpgradeId;
651
+
652
+ before(async () => {
653
+ ({ id: technicalUpgradeId } = await subject.save(new Version({
654
+ serviceId: SERVICE_PROVIDER_ID,
655
+ termsType: TERMS_TYPE,
656
+ content: `${CONTENT} - technical upgrade`,
657
+ fetchDate: FETCH_DATE_EARLIER,
658
+ snapshotIds: [SNAPSHOT_ID],
659
+ isTechnicalUpgrade: true,
660
+ })));
661
+
662
+ filteredRecords = await subject.findByServiceAndTermsType(SERVICE_PROVIDER_ID, TERMS_TYPE, { includeTechnicalUpgrades: false });
663
+ });
664
+
665
+ it('excludes technical upgrade records', () => {
666
+ expect(filteredRecords.map(record => record.id)).to.not.include(technicalUpgradeId);
667
+ });
668
+
669
+ it('only returns records that represent actual content changes', () => {
670
+ for (const record of filteredRecords) {
671
+ expect(record.isTechnicalUpgrade).to.not.be.true;
672
+ }
673
+ });
674
+ });
675
+ });
676
+
677
+ describe('#findByService', () => {
678
+ const OTHER_TERMS_TYPE = 'Privacy Policy';
679
+ const expectedIds = [];
680
+ let records;
681
+
682
+ before(async function () {
683
+ this.timeout(5000);
684
+
685
+ const { id: id1 } = await subject.save(new Version({
686
+ serviceId: SERVICE_PROVIDER_ID,
687
+ termsType: TERMS_TYPE,
688
+ content: CONTENT,
689
+ fetchDate: FETCH_DATE,
690
+ snapshotIds: [SNAPSHOT_ID],
691
+ }));
692
+
693
+ expectedIds.push(id1);
694
+
695
+ const { id: id2 } = await subject.save(new Version({
696
+ serviceId: SERVICE_PROVIDER_ID,
697
+ termsType: TERMS_TYPE,
698
+ content: `${CONTENT} - updated`,
699
+ fetchDate: FETCH_DATE_LATER,
700
+ snapshotIds: [SNAPSHOT_ID],
701
+ }));
702
+
703
+ expectedIds.push(id2);
704
+
705
+ const { id: id3 } = await subject.save(new Version({
706
+ serviceId: SERVICE_PROVIDER_ID,
707
+ termsType: OTHER_TERMS_TYPE,
708
+ content: `${CONTENT} - other terms type`,
709
+ fetchDate: FETCH_DATE_EARLIER,
710
+ snapshotIds: [SNAPSHOT_ID],
711
+ }));
712
+
713
+ expectedIds.push(id3);
714
+
715
+ await subject.save(new Version({
716
+ serviceId: 'other_service',
717
+ termsType: TERMS_TYPE,
718
+ content: `${CONTENT} - other service`,
719
+ fetchDate: FETCH_DATE,
720
+ snapshotIds: [SNAPSHOT_ID],
721
+ }));
722
+
723
+ (records = await subject.findByService(SERVICE_PROVIDER_ID));
724
+ });
725
+
726
+ after(() => subject.removeAll());
727
+
728
+ it('returns only matching records', () => {
729
+ expect(records.length).to.equal(3);
730
+ });
731
+
732
+ it('returns Version objects', () => {
733
+ for (const record of records) {
734
+ expect(record).to.be.an.instanceof(Version);
735
+ }
736
+ });
737
+
738
+ it('returns records with matching service ID', () => {
739
+ for (const record of records) {
740
+ expect(record.serviceId).to.equal(SERVICE_PROVIDER_ID);
741
+ }
742
+ });
743
+
744
+ it('returns records across all terms types of the service', () => {
745
+ expect(new Set(records.map(record => record.termsType))).to.deep.equal(new Set([ TERMS_TYPE, OTHER_TERMS_TYPE ]));
746
+ });
747
+
748
+ it('returns records in descending order', () => {
749
+ expect(records.map(record => record.fetchDate)).to.deep.equal([ FETCH_DATE_LATER, FETCH_DATE, FETCH_DATE_EARLIER ]);
750
+ });
751
+
752
+ it('returns records with correct IDs', () => {
753
+ expect(records.map(record => record.id)).to.have.members(expectedIds);
754
+ });
755
+
756
+ context('when no matching records exist', () => {
757
+ it('returns an empty array', async () => {
758
+ const result = await subject.findByService('non_existent_service');
759
+
760
+ expect(result).to.be.an('array').that.is.empty;
761
+ });
762
+ });
763
+
764
+ context('with includeTechnicalUpgrades: false', () => {
765
+ let filteredRecords;
766
+ let technicalUpgradeId;
767
+
768
+ before(async () => {
769
+ ({ id: technicalUpgradeId } = await subject.save(new Version({
770
+ serviceId: SERVICE_PROVIDER_ID,
771
+ termsType: TERMS_TYPE,
772
+ content: `${CONTENT} - technical upgrade`,
773
+ fetchDate: new Date('2000-01-03T12:00:00.000Z'),
774
+ snapshotIds: [SNAPSHOT_ID],
775
+ isTechnicalUpgrade: true,
776
+ })));
777
+
778
+ filteredRecords = await subject.findByService(SERVICE_PROVIDER_ID, { includeTechnicalUpgrades: false });
779
+ });
780
+
781
+ it('excludes technical upgrade records', () => {
782
+ expect(filteredRecords.map(record => record.id)).to.not.include(technicalUpgradeId);
783
+ });
784
+
785
+ it('only returns records that represent actual content changes', () => {
786
+ for (const record of filteredRecords) {
787
+ expect(record.isTechnicalUpgrade).to.not.be.true;
788
+ }
789
+ });
545
790
  });
546
791
  });
547
792
 
@@ -582,6 +827,37 @@ describe('GitRepository', () => {
582
827
  it('returns the proper count', () => {
583
828
  expect(count).to.equal(3);
584
829
  });
830
+
831
+ context('with serviceId and termsType filters', () => {
832
+ it('returns count for specific service and terms type', async () => {
833
+ const filteredCount = await subject.count(SERVICE_PROVIDER_ID, TERMS_TYPE);
834
+
835
+ expect(filteredCount).to.equal(3);
836
+ });
837
+
838
+ it('returns zero for non-existent service', async () => {
839
+ const filteredCount = await subject.count('non-existent-service', TERMS_TYPE);
840
+
841
+ expect(filteredCount).to.equal(0);
842
+ });
843
+ });
844
+
845
+ context('with only serviceId filter', () => {
846
+ it('returns count for all terms types of a service', async () => {
847
+ // Add a version with different terms type
848
+ await subject.save(new Version({
849
+ serviceId: SERVICE_PROVIDER_ID,
850
+ termsType: 'Different Terms',
851
+ content: CONTENT,
852
+ fetchDate: FETCH_DATE,
853
+ snapshotIds: [SNAPSHOT_ID],
854
+ }));
855
+
856
+ const filteredCount = await subject.count(SERVICE_PROVIDER_ID);
857
+
858
+ expect(filteredCount).to.equal(4); // 3 from TERMS_TYPE + 1 from 'Different Terms'
859
+ });
860
+ });
585
861
  });
586
862
 
587
863
  describe('#findLatest', () => {
@@ -1101,8 +1377,8 @@ describe('GitRepository', () => {
1101
1377
  }
1102
1378
  });
1103
1379
 
1104
- it('returns records in ascending order', () => {
1105
- expect(records.map(record => record.fetchDate)).to.deep.equal([ FETCH_DATE_EARLIER, FETCH_DATE, FETCH_DATE_LATER ]);
1380
+ it('returns records in descending order', () => {
1381
+ expect(records.map(record => record.fetchDate)).to.deep.equal([ FETCH_DATE_LATER, FETCH_DATE, FETCH_DATE_EARLIER ]);
1106
1382
  });
1107
1383
  });
1108
1384
 
@@ -1462,8 +1738,8 @@ describe('GitRepository', () => {
1462
1738
  }
1463
1739
  });
1464
1740
 
1465
- it('returns records in ascending order', () => {
1466
- expect(records.map(record => record.fetchDate)).to.deep.equal(expectedDates);
1741
+ it('returns records in descending order', () => {
1742
+ expect(records.map(record => record.fetchDate)).to.deep.equal([...expectedDates].reverse());
1467
1743
  });
1468
1744
  });
1469
1745
 
@@ -70,21 +70,59 @@ class RepositoryInterface {
70
70
  }
71
71
 
72
72
  /**
73
- * Find all records
73
+ * Find all records, in descending chronological order (newest first; opposite of #iterate)
74
74
  * For performance reasons, the content of the records will not be loaded by default. Use #loadRecordContent to load the content of individual records
75
- * @see RepositoryInterface#loadRecordContent
76
- * @returns {Promise<Array<Record>>} Promise that will be resolved with an array of all records
75
+ * @see RepositoryInterface#loadRecordContent
76
+ * @see RepositoryInterface#iterate
77
+ * @param {object} [options] - Query options
78
+ * @param {number} [options.limit] - Maximum number of records to return
79
+ * @param {number} [options.offset] - Number of records to skip
80
+ * @param {boolean} [options.includeTechnicalUpgrades] - When false, exclude technical upgrade records (re-renders of existing snapshots) and only return records that represent actual content changes. Default: true
81
+ * @returns {Promise<Array<Record>>} Promise that will be resolved with an array of records in descending chronological order
77
82
  */
78
- async findAll() {
83
+ async findAll(options = {}) {
79
84
  throw new Error(`#findAll method is not implemented in ${this.constructor.name}`);
80
85
  }
81
86
 
87
+ /**
88
+ * Find all records for a specific service, in descending chronological order
89
+ * For performance reasons, the content of the records will not be loaded by default. Use #loadRecordContent to load the content of individual records
90
+ * @see RepositoryInterface#loadRecordContent
91
+ * @param {string} serviceId - Service ID of records to find
92
+ * @param {object} [options] - Query options
93
+ * @param {number} [options.limit] - Maximum number of records to return
94
+ * @param {number} [options.offset] - Number of records to skip
95
+ * @param {boolean} [options.includeTechnicalUpgrades] - When false, exclude technical upgrade records (re-renders of existing snapshots) and only return records that represent actual content changes. Default: true
96
+ * @returns {Promise<Array<Record>>} Promise that will be resolved with an array of matching records in descending chronological order
97
+ */
98
+ async findByService(serviceId, options = {}) {
99
+ throw new Error(`#findByService method is not implemented in ${this.constructor.name}`);
100
+ }
101
+
102
+ /**
103
+ * Find all records for a specific service and terms type, in descending chronological order
104
+ * For performance reasons, the content of the records will not be loaded by default. Use #loadRecordContent to load the content of individual records
105
+ * @see RepositoryInterface#loadRecordContent
106
+ * @param {string} serviceId - Service ID of records to find
107
+ * @param {string} termsType - Terms type of records to find
108
+ * @param {object} [options] - Query options
109
+ * @param {number} [options.limit] - Maximum number of records to return
110
+ * @param {number} [options.offset] - Number of records to skip
111
+ * @param {boolean} [options.includeTechnicalUpgrades] - When false, exclude technical upgrade records (re-renders of existing snapshots) and only return records that represent actual content changes. Default: true
112
+ * @returns {Promise<Array<Record>>} Promise that will be resolved with an array of matching records in descending chronological order
113
+ */
114
+ async findByServiceAndTermsType(serviceId, termsType, options = {}) {
115
+ throw new Error(`#findByServiceAndTermsType method is not implemented in ${this.constructor.name}`);
116
+ }
117
+
82
118
  /**
83
119
  * Count the total number of records in the repository
84
120
  * For performance reasons, use this method rather than counting the number of entries returned by #findAll if you only need the size of a repository
85
- * @returns {Promise<number>} Promise that will be resolved with the total number of records
121
+ * @param {string} [serviceId] - Optional service ID to filter records
122
+ * @param {string} [termsType] - Optional terms type to filter records (requires serviceId)
123
+ * @returns {Promise<number>} Promise that will be resolved with the total number of records
86
124
  */
87
- async count() {
125
+ async count(serviceId, termsType) {
88
126
  throw new Error(`#count method is not implemented in ${this.constructor.name}`);
89
127
  }
90
128