openaleph-client 1.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
openaleph/__init__.py ADDED
File without changes
openaleph/api.py ADDED
@@ -0,0 +1,525 @@
1
+ import importlib.metadata
2
+ import json
3
+ import uuid
4
+ import logging
5
+ from itertools import count
6
+ from pathlib import Path
7
+ from urllib.parse import urlencode, urljoin
8
+ from banal import ensure_dict, ensure_list
9
+ from requests import RequestException, Session
10
+ from requests.exceptions import HTTPError
11
+ from requests_toolbelt import MultipartEncoder # type: ignore
12
+ from typing import Dict, Mapping, Iterable, Iterator, List, Optional, Any
13
+
14
+ from openaleph import settings
15
+ from openaleph.errors import AlephException
16
+ from openaleph.util import backoff, prop_push
17
+
18
+ log = logging.getLogger(__name__)
19
+ MIME = "application/octet-stream"
20
+ VERSION = importlib.metadata.version("openaleph")
21
+
22
+
23
+ class APIResultSet(object):
24
+ def __init__(self, api: "AlephAPI", url: str):
25
+ self.api = api
26
+ self.url = url
27
+ self.current = 0
28
+ self.result = self.api._request("GET", self.url)
29
+
30
+ def __iter__(self):
31
+ return self
32
+
33
+ def __next__(self):
34
+ if self.index >= self.result.get("limit"):
35
+ next_url = self.result.get("next")
36
+ if next_url is None:
37
+ raise StopIteration
38
+ self.result = self.api._request("GET", next_url)
39
+ try:
40
+ item = self.result.get("results", [])[self.index]
41
+ except IndexError:
42
+ raise StopIteration
43
+ self.current += 1
44
+ return self._patch(item)
45
+
46
+ next = __next__
47
+
48
+ def _patch(self, item):
49
+ return item
50
+
51
+ @property
52
+ def index(self):
53
+ return self.current - self.result.get("offset")
54
+
55
+ def __len__(self):
56
+ return self.result.get("total")
57
+
58
+ def __repr__(self):
59
+ return "<APIResultSet(%r, %r)>" % (self.url, len(self))
60
+
61
+
62
+ class EntityResultSet(APIResultSet):
63
+ def __init__(self, api: "AlephAPI", url: str, publisher: bool):
64
+ super(EntityResultSet, self).__init__(api, url)
65
+ self.publisher = publisher
66
+
67
+ def _patch(self, item):
68
+ return self.api._patch_entity(item, self.publisher)
69
+
70
+
71
+ class EntitySetItemsResultSet(APIResultSet):
72
+ def __init__(self, api: "AlephAPI", url: str, publisher: bool):
73
+ super(EntitySetItemsResultSet, self).__init__(api, url)
74
+ self.publisher = publisher
75
+
76
+ def _patch(self, item):
77
+ entity = ensure_dict(item.get("entity"))
78
+ item["entity"] = self.api._patch_entity(entity, self.publisher)
79
+ return item
80
+
81
+
82
+ class AlephAPI(object):
83
+ def __init__(
84
+ self,
85
+ host: Optional[str] = settings.HOST,
86
+ api_key: Optional[str] = settings.API_KEY,
87
+ session_id: Optional[str] = None,
88
+ retries: int = settings.MAX_TRIES,
89
+ ):
90
+
91
+ if not host:
92
+ raise AlephException("No host environment variable found")
93
+ self.base_url = urljoin(host, "/api/2/")
94
+ self.retries = retries
95
+ session_id = session_id or str(uuid.uuid4())
96
+ self.session: Session = Session()
97
+ self.session.headers["X-Aleph-Session"] = session_id
98
+ self.session.headers["User-Agent"] = "openaleph/%s" % VERSION
99
+ if api_key is not None:
100
+ self.session.headers["Authorization"] = "ApiKey %s" % api_key
101
+
102
+ def _make_url(
103
+ self,
104
+ path: str,
105
+ query: Optional[str] = None,
106
+ filters: Optional[List] = None,
107
+ params: Optional[Mapping[str, Any]] = None,
108
+ ):
109
+ """Construct the target url from given args"""
110
+ url = self.base_url + path
111
+ params = params or {}
112
+ params_list = list(params.items())
113
+ if query:
114
+ params_list.append(("q", query))
115
+ if filters:
116
+ for key, val in filters:
117
+ if val is not None:
118
+ params_list.append(("filter:" + key, val))
119
+ if len(params_list):
120
+ params_filter = [(k, v) for k, v in params_list if v is not None]
121
+ url = url + "?" + urlencode(params_filter)
122
+ return url
123
+
124
+ def _patch_entity(
125
+ self, entity: Dict, publisher: bool, collection: Optional[Dict] = None
126
+ ):
127
+ """Add extra properties from context to the given entity."""
128
+ properties: Dict = entity.get("properties", {})
129
+ collection_: Dict = collection or entity.get("collection") or {}
130
+ links: Dict = entity.get("links", {})
131
+ api_url = links.get("self")
132
+ if api_url is None:
133
+ api_url = "entities/%s" % entity.get("id")
134
+ api_url = self._make_url(api_url)
135
+ prop_push(properties, "alephUrl", api_url)
136
+
137
+ if publisher:
138
+ # Context: setting the original publisher or collection
139
+ # label can help make the data more traceable when merging
140
+ # data from multiple sources.
141
+ publisher_label = collection_.get("label")
142
+ publisher_label = collection_.get("publisher", publisher_label)
143
+ prop_push(properties, "publisher", publisher_label)
144
+
145
+ publisher_url = collection_.get("links", {}).get("ui")
146
+ publisher_url = collection_.get("publisher_url", publisher_url)
147
+ prop_push(properties, "publisherUrl", publisher_url)
148
+
149
+ entity["properties"] = properties
150
+ return entity
151
+
152
+ def _request(self, method: str, url: str, **kwargs) -> Dict:
153
+ """A single point to make the http requests.
154
+
155
+ Having a single point to make all requests let's us set headers, manage
156
+ successful and failed responses and possibly manage session etc
157
+ conviniently in a single place.
158
+ """
159
+ try:
160
+ response = self.session.request(method=method, url=url, **kwargs)
161
+ response.raise_for_status()
162
+ except (RequestException, HTTPError) as exc:
163
+ raise AlephException(exc) from exc
164
+
165
+ if len(response.text):
166
+ return response.json()
167
+ return {}
168
+
169
+ def search(
170
+ self,
171
+ query: str,
172
+ schema: Optional[str] = None,
173
+ schemata: Optional[str] = None,
174
+ filters: Optional[List] = None,
175
+ publisher: bool = False,
176
+ params: Optional[Mapping[str, Any]] = None,
177
+ ) -> "EntityResultSet":
178
+ """Conduct a search and return the search results."""
179
+ filters_list: List = ensure_list(filters)
180
+ if schema is not None:
181
+ filters_list.append(("schema", schema))
182
+ if schemata is not None:
183
+ filters_list.append(("schemata", schemata))
184
+ if schema is None and schemata is None:
185
+ filters_list.append(("schemata", "Thing"))
186
+ url = self._make_url(
187
+ "entities", query=query, filters=filters_list, params=params
188
+ )
189
+ return EntityResultSet(self, url, publisher)
190
+
191
+ def get_collection(self, collection_id: str) -> Dict:
192
+ """Get a single collection by ID (not foreign ID!)."""
193
+ url = self._make_url(f"collections/{collection_id}")
194
+ return self._request("GET", url)
195
+
196
+ def reingest_collection(self, collection_id: str, index: bool = False):
197
+ """Re-ingest all documents in a collection."""
198
+ url = self._make_url(
199
+ f"collections/{collection_id}/reingest", params={"index": index}
200
+ )
201
+ return self._request("POST", url)
202
+
203
+ def reindex_collection(
204
+ self, collection_id: str, flush: bool = False, sync: bool = False
205
+ ):
206
+ """Re-index all entities in a collection."""
207
+ params = {"sync": sync, "flush": flush}
208
+ url = self._make_url(f"collections/{collection_id}/reindex", params=params)
209
+ return self._request("POST", url)
210
+
211
+ def delete_collection(self, collection_id: str, sync: bool = False):
212
+ """Delete a collection by ID"""
213
+ params = {"sync": sync}
214
+ url = self._make_url(f"collections/{collection_id}", params=params)
215
+ return self._request("DELETE", url)
216
+
217
+ def flush_collection(self, collection_id: str, sync: bool = False):
218
+ """Empty all contents from a collection by ID"""
219
+ params = {"sync": sync, "keep_metadata": True}
220
+ url = self._make_url(f"collections/{collection_id}", params=params)
221
+ return self._request("DELETE", url)
222
+
223
+ def get_entity(self, entity_id: str, publisher: bool = False) -> Dict:
224
+ """Get a single entity by ID."""
225
+ url = self._make_url(f"entities/{entity_id}")
226
+ entity = self._request("GET", url)
227
+ return self._patch_entity(entity, publisher)
228
+
229
+ def delete_entity(self, entity_id: str) -> Dict:
230
+ """Delete a single entity by ID."""
231
+ url = self._make_url(f"entities/{entity_id}")
232
+ return self._request("DELETE", url)
233
+
234
+ def get_collection_by_foreign_id(self, foreign_id: str) -> Optional[Dict]:
235
+ """Get a dict representing a collection based on its foreign ID."""
236
+ if foreign_id is None:
237
+ return None
238
+ filters = [("foreign_id", foreign_id)]
239
+ for coll in self.filter_collections(filters=filters):
240
+ return coll
241
+ return None
242
+
243
+ def load_collection_by_foreign_id(
244
+ self, foreign_id: str, config: Optional[Dict] = None
245
+ ) -> Dict:
246
+ """Get a collection by its foreign ID, or create one. Setting clear
247
+ will clear any found collection."""
248
+ collection = self.get_collection_by_foreign_id(foreign_id)
249
+ if collection is not None:
250
+ return collection
251
+
252
+ config_: Dict = ensure_dict(config)
253
+ return self.create_collection(
254
+ {
255
+ "foreign_id": foreign_id,
256
+ "label": config_.get("label", foreign_id),
257
+ "casefile": config_.get("casefile", False),
258
+ "category": config_.get("category", "other"),
259
+ "languages": config_.get("languages", []),
260
+ "summary": config_.get("summary", ""),
261
+ }
262
+ )
263
+
264
+ def filter_collections(
265
+ self, query: Optional[str] = None, filters: Optional[List] = None, **kwargs
266
+ ) -> "APIResultSet":
267
+ """Filter collections for the given query and/or filters.
268
+
269
+ params
270
+ ------
271
+ query: query string
272
+ filters: list of key, value pairs to filter collections
273
+ kwargs: extra arguments for api call such as page, limit etc
274
+ """
275
+ if not query and not filters:
276
+ raise ValueError("One of query or filters is required")
277
+
278
+ url = self._make_url("collections", query=query, filters=filters, params=kwargs)
279
+ return APIResultSet(self, url)
280
+
281
+ def create_collection(self, data: Dict) -> Dict:
282
+ """Create a collection from the given data.
283
+
284
+ params
285
+ ------
286
+ data: dict with foreign_id, label, category etc. See `CollectionSchema`
287
+ for more details.
288
+ """
289
+ url = self._make_url("collections")
290
+ return self._request("POST", url, json=data)
291
+
292
+ def update_collection(
293
+ self, collection_id: str, data: Dict, sync: bool = False
294
+ ) -> Dict:
295
+ """Update an existing collection using the given data.
296
+
297
+ params
298
+ ------
299
+ collection_id: id of the collection to update
300
+ data: dict with foreign_id, label, category etc. See `CollectionSchema`
301
+ for more details.
302
+ """
303
+ params = {"sync": sync}
304
+ url = self._make_url(f"collections/{collection_id}", params=params)
305
+ return self._request("PUT", url, json=data)
306
+
307
+ def stream_entities(
308
+ self,
309
+ collection: Optional[Dict] = None,
310
+ include: Optional[List] = None,
311
+ schema: Optional[str] = None,
312
+ publisher: bool = False,
313
+ ) -> Iterator[Dict]:
314
+ """Iterate over all entities in the given collection.
315
+
316
+ params
317
+ ------
318
+ collection_id: id of the collection to stream
319
+ include: an array of fields from the index to include.
320
+ """
321
+ url = self._make_url("entities/_stream")
322
+ if collection is not None:
323
+ collection_id = collection.get("id")
324
+ url = f"collections/{collection_id}/_stream"
325
+ url = self._make_url(url)
326
+ params = {"include": include, "schema": schema}
327
+ try:
328
+ res = self.session.get(url, params=params, stream=True)
329
+ res.raise_for_status()
330
+ for entity in res.iter_lines(chunk_size=None):
331
+ entity = json.loads(entity)
332
+ yield self._patch_entity(
333
+ entity, publisher=publisher, collection=collection
334
+ )
335
+ except (RequestException, HTTPError) as exc:
336
+ raise AlephException(exc) from exc
337
+
338
+ def _bulk_chunk(
339
+ self,
340
+ collection_id: str,
341
+ chunk: List,
342
+ entityset_id: Optional[str] = None,
343
+ force: bool = False,
344
+ unsafe: bool = False,
345
+ cleaned: bool = False,
346
+ ):
347
+ for attempt in count(1):
348
+ url = self._make_url(f"collections/{collection_id}/_bulk")
349
+ params = {"entityset_id": entityset_id}
350
+ if unsafe:
351
+ params["safe"] = "false"
352
+ if cleaned:
353
+ params["clean"] = "false"
354
+ try:
355
+ response = self.session.post(url, json=chunk, params=params)
356
+ response.raise_for_status()
357
+ return
358
+ except (RequestException, HTTPError) as exc:
359
+ ae = AlephException(exc)
360
+ if not ae.transient or attempt > self.retries:
361
+ if not force:
362
+ raise ae from exc
363
+ log.error(ae)
364
+ return
365
+ backoff(ae, attempt)
366
+
367
+ def write_entity(
368
+ self, collection_id: str, entity: Dict, entity_id: Optional[str] = None, **kw
369
+ ) -> Dict:
370
+ """Create a single entity via the API, in the given
371
+ collection.
372
+
373
+ params
374
+ ------
375
+ collection_id: id of the collection to use. This will overwrite any
376
+ existing collection specified in the entity dict
377
+ entity_id: id for the entity to be created. This will overwrite any
378
+ existing entity specified in the entity dict
379
+ entity: A dict object containing the values of the entity
380
+ """
381
+ entity["collection_id"] = collection_id
382
+
383
+ if entity_id is not None:
384
+ entity["id"] = entity_id
385
+
386
+ for attempt in count(1):
387
+ if entity_id is not None:
388
+ url = self._make_url("entities/{}").format(entity_id)
389
+ else:
390
+ url = self._make_url("entities")
391
+ try:
392
+ return self._request("POST", url, json=entity)
393
+ except RequestException as exc:
394
+ ae = AlephException(exc)
395
+ if not ae.transient or attempt > self.retries:
396
+ log.error(ae)
397
+ raise exc
398
+ backoff(ae, attempt)
399
+
400
+ return {}
401
+
402
+ def write_entities(
403
+ self, collection_id: str, entities: Iterable, chunk_size: int = 1000, **kw
404
+ ):
405
+ """Create entities in bulk via the API, in the given
406
+ collection.
407
+
408
+ params
409
+ ------
410
+ collection_id: id of the collection to use
411
+ entities: an iterable of entities to upload
412
+ """
413
+ chunk = []
414
+ for entity in entities:
415
+ if hasattr(entity, "to_dict"):
416
+ entity = entity.to_dict()
417
+ chunk.append(entity)
418
+ if len(chunk) >= chunk_size:
419
+ self._bulk_chunk(collection_id, chunk, **kw)
420
+ chunk = []
421
+ if len(chunk):
422
+ self._bulk_chunk(collection_id, chunk, **kw)
423
+
424
+ def match(
425
+ self,
426
+ entity: Dict,
427
+ collection_ids: Optional[str] = None,
428
+ url: Optional[str] = None,
429
+ publisher: bool = False,
430
+ ) -> Iterator[List]:
431
+ """Find similar entities given a sample entity."""
432
+ params = {"collection_ids": ensure_list(collection_ids)}
433
+ if url is None:
434
+ url = self._make_url("match")
435
+ try:
436
+ response = self.session.post(url, json=entity, params=params) # type: ignore
437
+ response.raise_for_status()
438
+ for result in response.json().get("results", []):
439
+ yield self._patch_entity(result, publisher=publisher)
440
+ except (RequestException, HTTPError) as exc:
441
+ raise AlephException(exc) from exc
442
+
443
+ def entitysets(
444
+ self,
445
+ collection_id: Optional[str] = None,
446
+ set_types: Optional[List] = None,
447
+ prefix: Optional[str] = None,
448
+ ) -> "APIResultSet":
449
+ """Stream EntitySets"""
450
+ filters_collection = [("collection_id", collection_id)]
451
+ filters_type = [("type", t) for t in ensure_list(set_types)]
452
+ filters = [*filters_collection, *filters_type]
453
+ params = {"prefix": prefix}
454
+ url = self._make_url("entitysets", filters=filters, params=params)
455
+ return APIResultSet(self, url)
456
+
457
+ def entitysetitems(
458
+ self, entityset_id: str, publisher: bool = False
459
+ ) -> "APIResultSet":
460
+ url = self._make_url(f"entitysets/{entityset_id}/items")
461
+ return EntitySetItemsResultSet(self, url, publisher=publisher)
462
+
463
+ def ingest_upload(
464
+ self,
465
+ collection_id: str,
466
+ file_path: Optional[Path] = None,
467
+ metadata: Optional[Dict] = None,
468
+ sync: bool = False,
469
+ index: bool = True,
470
+ ) -> Dict:
471
+ """
472
+ Create an empty folder in a collection or upload a document to it
473
+
474
+ params
475
+ ------
476
+ collection_id: id of the collection to upload to
477
+ file_path: path of the file to upload. None while creating folders
478
+ metadata: dict containing metadata for the file or folders. In case of
479
+ files, metadata contains foreign_id of the parent. Metadata for a
480
+ directory contains foreign_id for itself as well as its parent and the
481
+ name of the directory.
482
+ """
483
+ url_path = "collections/{0}/ingest".format(collection_id)
484
+ params = {"sync": sync, "index": index}
485
+ url = self._make_url(url_path, params=params)
486
+ if not file_path or file_path.is_dir():
487
+ data = {"meta": json.dumps(metadata)}
488
+ return self._request("POST", url, data=data)
489
+
490
+ for attempt in count(1):
491
+ try:
492
+ with file_path.open("rb") as fh:
493
+ # use multipart encoder to allow uploading very large files
494
+ m = MultipartEncoder(
495
+ fields={
496
+ "meta": json.dumps(metadata),
497
+ "file": (file_path.name, fh, MIME),
498
+ }
499
+ )
500
+ headers = {"Content-Type": m.content_type}
501
+ return self._request("POST", url, data=m, headers=headers)
502
+ except AlephException as ae:
503
+ if not ae.transient or attempt > self.retries:
504
+ raise ae from ae
505
+ backoff(ae, attempt)
506
+ return {}
507
+
508
+ def create_entityset(
509
+ self, collection_id: str, type: str, label: str, summary: Optional[str]
510
+ ) -> Dict:
511
+ """Create an EntitySet inside a collection"""
512
+ url = self._make_url("entitysets")
513
+ data: Dict = {
514
+ "collection_id": collection_id,
515
+ "type": type,
516
+ "label": label,
517
+ "summary": summary,
518
+ "entities": [],
519
+ }
520
+ return self._request("POST", url, data=data)
521
+
522
+ def delete_entityset(self, entityset_id: str, sync: bool = False):
523
+ """Delete an EntitySet by id"""
524
+ url = self._make_url(f"entitysets/{entityset_id}", params={"sync": sync})
525
+ return self._request("DELETE", url)