openaleph-client 1.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- openaleph/__init__.py +0 -0
- openaleph/api.py +525 -0
- openaleph/cli.py +411 -0
- openaleph/crawldir.py +238 -0
- openaleph/errors.py +22 -0
- openaleph/fetchdir.py +77 -0
- openaleph/settings.py +13 -0
- openaleph/util.py +20 -0
- openaleph_client-1.0.1.dist-info/METADATA +165 -0
- openaleph_client-1.0.1.dist-info/RECORD +14 -0
- openaleph_client-1.0.1.dist-info/WHEEL +5 -0
- openaleph_client-1.0.1.dist-info/entry_points.txt +2 -0
- openaleph_client-1.0.1.dist-info/licenses/LICENSE +22 -0
- openaleph_client-1.0.1.dist-info/top_level.txt +1 -0
openaleph/__init__.py
ADDED
|
File without changes
|
openaleph/api.py
ADDED
|
@@ -0,0 +1,525 @@
|
|
|
1
|
+
import importlib.metadata
|
|
2
|
+
import json
|
|
3
|
+
import uuid
|
|
4
|
+
import logging
|
|
5
|
+
from itertools import count
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from urllib.parse import urlencode, urljoin
|
|
8
|
+
from banal import ensure_dict, ensure_list
|
|
9
|
+
from requests import RequestException, Session
|
|
10
|
+
from requests.exceptions import HTTPError
|
|
11
|
+
from requests_toolbelt import MultipartEncoder # type: ignore
|
|
12
|
+
from typing import Dict, Mapping, Iterable, Iterator, List, Optional, Any
|
|
13
|
+
|
|
14
|
+
from openaleph import settings
|
|
15
|
+
from openaleph.errors import AlephException
|
|
16
|
+
from openaleph.util import backoff, prop_push
|
|
17
|
+
|
|
18
|
+
log = logging.getLogger(__name__)
|
|
19
|
+
MIME = "application/octet-stream"
|
|
20
|
+
VERSION = importlib.metadata.version("openaleph")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class APIResultSet(object):
|
|
24
|
+
def __init__(self, api: "AlephAPI", url: str):
|
|
25
|
+
self.api = api
|
|
26
|
+
self.url = url
|
|
27
|
+
self.current = 0
|
|
28
|
+
self.result = self.api._request("GET", self.url)
|
|
29
|
+
|
|
30
|
+
def __iter__(self):
|
|
31
|
+
return self
|
|
32
|
+
|
|
33
|
+
def __next__(self):
|
|
34
|
+
if self.index >= self.result.get("limit"):
|
|
35
|
+
next_url = self.result.get("next")
|
|
36
|
+
if next_url is None:
|
|
37
|
+
raise StopIteration
|
|
38
|
+
self.result = self.api._request("GET", next_url)
|
|
39
|
+
try:
|
|
40
|
+
item = self.result.get("results", [])[self.index]
|
|
41
|
+
except IndexError:
|
|
42
|
+
raise StopIteration
|
|
43
|
+
self.current += 1
|
|
44
|
+
return self._patch(item)
|
|
45
|
+
|
|
46
|
+
next = __next__
|
|
47
|
+
|
|
48
|
+
def _patch(self, item):
|
|
49
|
+
return item
|
|
50
|
+
|
|
51
|
+
@property
|
|
52
|
+
def index(self):
|
|
53
|
+
return self.current - self.result.get("offset")
|
|
54
|
+
|
|
55
|
+
def __len__(self):
|
|
56
|
+
return self.result.get("total")
|
|
57
|
+
|
|
58
|
+
def __repr__(self):
|
|
59
|
+
return "<APIResultSet(%r, %r)>" % (self.url, len(self))
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class EntityResultSet(APIResultSet):
|
|
63
|
+
def __init__(self, api: "AlephAPI", url: str, publisher: bool):
|
|
64
|
+
super(EntityResultSet, self).__init__(api, url)
|
|
65
|
+
self.publisher = publisher
|
|
66
|
+
|
|
67
|
+
def _patch(self, item):
|
|
68
|
+
return self.api._patch_entity(item, self.publisher)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class EntitySetItemsResultSet(APIResultSet):
|
|
72
|
+
def __init__(self, api: "AlephAPI", url: str, publisher: bool):
|
|
73
|
+
super(EntitySetItemsResultSet, self).__init__(api, url)
|
|
74
|
+
self.publisher = publisher
|
|
75
|
+
|
|
76
|
+
def _patch(self, item):
|
|
77
|
+
entity = ensure_dict(item.get("entity"))
|
|
78
|
+
item["entity"] = self.api._patch_entity(entity, self.publisher)
|
|
79
|
+
return item
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class AlephAPI(object):
|
|
83
|
+
def __init__(
|
|
84
|
+
self,
|
|
85
|
+
host: Optional[str] = settings.HOST,
|
|
86
|
+
api_key: Optional[str] = settings.API_KEY,
|
|
87
|
+
session_id: Optional[str] = None,
|
|
88
|
+
retries: int = settings.MAX_TRIES,
|
|
89
|
+
):
|
|
90
|
+
|
|
91
|
+
if not host:
|
|
92
|
+
raise AlephException("No host environment variable found")
|
|
93
|
+
self.base_url = urljoin(host, "/api/2/")
|
|
94
|
+
self.retries = retries
|
|
95
|
+
session_id = session_id or str(uuid.uuid4())
|
|
96
|
+
self.session: Session = Session()
|
|
97
|
+
self.session.headers["X-Aleph-Session"] = session_id
|
|
98
|
+
self.session.headers["User-Agent"] = "openaleph/%s" % VERSION
|
|
99
|
+
if api_key is not None:
|
|
100
|
+
self.session.headers["Authorization"] = "ApiKey %s" % api_key
|
|
101
|
+
|
|
102
|
+
def _make_url(
|
|
103
|
+
self,
|
|
104
|
+
path: str,
|
|
105
|
+
query: Optional[str] = None,
|
|
106
|
+
filters: Optional[List] = None,
|
|
107
|
+
params: Optional[Mapping[str, Any]] = None,
|
|
108
|
+
):
|
|
109
|
+
"""Construct the target url from given args"""
|
|
110
|
+
url = self.base_url + path
|
|
111
|
+
params = params or {}
|
|
112
|
+
params_list = list(params.items())
|
|
113
|
+
if query:
|
|
114
|
+
params_list.append(("q", query))
|
|
115
|
+
if filters:
|
|
116
|
+
for key, val in filters:
|
|
117
|
+
if val is not None:
|
|
118
|
+
params_list.append(("filter:" + key, val))
|
|
119
|
+
if len(params_list):
|
|
120
|
+
params_filter = [(k, v) for k, v in params_list if v is not None]
|
|
121
|
+
url = url + "?" + urlencode(params_filter)
|
|
122
|
+
return url
|
|
123
|
+
|
|
124
|
+
def _patch_entity(
|
|
125
|
+
self, entity: Dict, publisher: bool, collection: Optional[Dict] = None
|
|
126
|
+
):
|
|
127
|
+
"""Add extra properties from context to the given entity."""
|
|
128
|
+
properties: Dict = entity.get("properties", {})
|
|
129
|
+
collection_: Dict = collection or entity.get("collection") or {}
|
|
130
|
+
links: Dict = entity.get("links", {})
|
|
131
|
+
api_url = links.get("self")
|
|
132
|
+
if api_url is None:
|
|
133
|
+
api_url = "entities/%s" % entity.get("id")
|
|
134
|
+
api_url = self._make_url(api_url)
|
|
135
|
+
prop_push(properties, "alephUrl", api_url)
|
|
136
|
+
|
|
137
|
+
if publisher:
|
|
138
|
+
# Context: setting the original publisher or collection
|
|
139
|
+
# label can help make the data more traceable when merging
|
|
140
|
+
# data from multiple sources.
|
|
141
|
+
publisher_label = collection_.get("label")
|
|
142
|
+
publisher_label = collection_.get("publisher", publisher_label)
|
|
143
|
+
prop_push(properties, "publisher", publisher_label)
|
|
144
|
+
|
|
145
|
+
publisher_url = collection_.get("links", {}).get("ui")
|
|
146
|
+
publisher_url = collection_.get("publisher_url", publisher_url)
|
|
147
|
+
prop_push(properties, "publisherUrl", publisher_url)
|
|
148
|
+
|
|
149
|
+
entity["properties"] = properties
|
|
150
|
+
return entity
|
|
151
|
+
|
|
152
|
+
def _request(self, method: str, url: str, **kwargs) -> Dict:
|
|
153
|
+
"""A single point to make the http requests.
|
|
154
|
+
|
|
155
|
+
Having a single point to make all requests let's us set headers, manage
|
|
156
|
+
successful and failed responses and possibly manage session etc
|
|
157
|
+
conviniently in a single place.
|
|
158
|
+
"""
|
|
159
|
+
try:
|
|
160
|
+
response = self.session.request(method=method, url=url, **kwargs)
|
|
161
|
+
response.raise_for_status()
|
|
162
|
+
except (RequestException, HTTPError) as exc:
|
|
163
|
+
raise AlephException(exc) from exc
|
|
164
|
+
|
|
165
|
+
if len(response.text):
|
|
166
|
+
return response.json()
|
|
167
|
+
return {}
|
|
168
|
+
|
|
169
|
+
def search(
|
|
170
|
+
self,
|
|
171
|
+
query: str,
|
|
172
|
+
schema: Optional[str] = None,
|
|
173
|
+
schemata: Optional[str] = None,
|
|
174
|
+
filters: Optional[List] = None,
|
|
175
|
+
publisher: bool = False,
|
|
176
|
+
params: Optional[Mapping[str, Any]] = None,
|
|
177
|
+
) -> "EntityResultSet":
|
|
178
|
+
"""Conduct a search and return the search results."""
|
|
179
|
+
filters_list: List = ensure_list(filters)
|
|
180
|
+
if schema is not None:
|
|
181
|
+
filters_list.append(("schema", schema))
|
|
182
|
+
if schemata is not None:
|
|
183
|
+
filters_list.append(("schemata", schemata))
|
|
184
|
+
if schema is None and schemata is None:
|
|
185
|
+
filters_list.append(("schemata", "Thing"))
|
|
186
|
+
url = self._make_url(
|
|
187
|
+
"entities", query=query, filters=filters_list, params=params
|
|
188
|
+
)
|
|
189
|
+
return EntityResultSet(self, url, publisher)
|
|
190
|
+
|
|
191
|
+
def get_collection(self, collection_id: str) -> Dict:
|
|
192
|
+
"""Get a single collection by ID (not foreign ID!)."""
|
|
193
|
+
url = self._make_url(f"collections/{collection_id}")
|
|
194
|
+
return self._request("GET", url)
|
|
195
|
+
|
|
196
|
+
def reingest_collection(self, collection_id: str, index: bool = False):
|
|
197
|
+
"""Re-ingest all documents in a collection."""
|
|
198
|
+
url = self._make_url(
|
|
199
|
+
f"collections/{collection_id}/reingest", params={"index": index}
|
|
200
|
+
)
|
|
201
|
+
return self._request("POST", url)
|
|
202
|
+
|
|
203
|
+
def reindex_collection(
|
|
204
|
+
self, collection_id: str, flush: bool = False, sync: bool = False
|
|
205
|
+
):
|
|
206
|
+
"""Re-index all entities in a collection."""
|
|
207
|
+
params = {"sync": sync, "flush": flush}
|
|
208
|
+
url = self._make_url(f"collections/{collection_id}/reindex", params=params)
|
|
209
|
+
return self._request("POST", url)
|
|
210
|
+
|
|
211
|
+
def delete_collection(self, collection_id: str, sync: bool = False):
|
|
212
|
+
"""Delete a collection by ID"""
|
|
213
|
+
params = {"sync": sync}
|
|
214
|
+
url = self._make_url(f"collections/{collection_id}", params=params)
|
|
215
|
+
return self._request("DELETE", url)
|
|
216
|
+
|
|
217
|
+
def flush_collection(self, collection_id: str, sync: bool = False):
|
|
218
|
+
"""Empty all contents from a collection by ID"""
|
|
219
|
+
params = {"sync": sync, "keep_metadata": True}
|
|
220
|
+
url = self._make_url(f"collections/{collection_id}", params=params)
|
|
221
|
+
return self._request("DELETE", url)
|
|
222
|
+
|
|
223
|
+
def get_entity(self, entity_id: str, publisher: bool = False) -> Dict:
|
|
224
|
+
"""Get a single entity by ID."""
|
|
225
|
+
url = self._make_url(f"entities/{entity_id}")
|
|
226
|
+
entity = self._request("GET", url)
|
|
227
|
+
return self._patch_entity(entity, publisher)
|
|
228
|
+
|
|
229
|
+
def delete_entity(self, entity_id: str) -> Dict:
|
|
230
|
+
"""Delete a single entity by ID."""
|
|
231
|
+
url = self._make_url(f"entities/{entity_id}")
|
|
232
|
+
return self._request("DELETE", url)
|
|
233
|
+
|
|
234
|
+
def get_collection_by_foreign_id(self, foreign_id: str) -> Optional[Dict]:
|
|
235
|
+
"""Get a dict representing a collection based on its foreign ID."""
|
|
236
|
+
if foreign_id is None:
|
|
237
|
+
return None
|
|
238
|
+
filters = [("foreign_id", foreign_id)]
|
|
239
|
+
for coll in self.filter_collections(filters=filters):
|
|
240
|
+
return coll
|
|
241
|
+
return None
|
|
242
|
+
|
|
243
|
+
def load_collection_by_foreign_id(
|
|
244
|
+
self, foreign_id: str, config: Optional[Dict] = None
|
|
245
|
+
) -> Dict:
|
|
246
|
+
"""Get a collection by its foreign ID, or create one. Setting clear
|
|
247
|
+
will clear any found collection."""
|
|
248
|
+
collection = self.get_collection_by_foreign_id(foreign_id)
|
|
249
|
+
if collection is not None:
|
|
250
|
+
return collection
|
|
251
|
+
|
|
252
|
+
config_: Dict = ensure_dict(config)
|
|
253
|
+
return self.create_collection(
|
|
254
|
+
{
|
|
255
|
+
"foreign_id": foreign_id,
|
|
256
|
+
"label": config_.get("label", foreign_id),
|
|
257
|
+
"casefile": config_.get("casefile", False),
|
|
258
|
+
"category": config_.get("category", "other"),
|
|
259
|
+
"languages": config_.get("languages", []),
|
|
260
|
+
"summary": config_.get("summary", ""),
|
|
261
|
+
}
|
|
262
|
+
)
|
|
263
|
+
|
|
264
|
+
def filter_collections(
|
|
265
|
+
self, query: Optional[str] = None, filters: Optional[List] = None, **kwargs
|
|
266
|
+
) -> "APIResultSet":
|
|
267
|
+
"""Filter collections for the given query and/or filters.
|
|
268
|
+
|
|
269
|
+
params
|
|
270
|
+
------
|
|
271
|
+
query: query string
|
|
272
|
+
filters: list of key, value pairs to filter collections
|
|
273
|
+
kwargs: extra arguments for api call such as page, limit etc
|
|
274
|
+
"""
|
|
275
|
+
if not query and not filters:
|
|
276
|
+
raise ValueError("One of query or filters is required")
|
|
277
|
+
|
|
278
|
+
url = self._make_url("collections", query=query, filters=filters, params=kwargs)
|
|
279
|
+
return APIResultSet(self, url)
|
|
280
|
+
|
|
281
|
+
def create_collection(self, data: Dict) -> Dict:
|
|
282
|
+
"""Create a collection from the given data.
|
|
283
|
+
|
|
284
|
+
params
|
|
285
|
+
------
|
|
286
|
+
data: dict with foreign_id, label, category etc. See `CollectionSchema`
|
|
287
|
+
for more details.
|
|
288
|
+
"""
|
|
289
|
+
url = self._make_url("collections")
|
|
290
|
+
return self._request("POST", url, json=data)
|
|
291
|
+
|
|
292
|
+
def update_collection(
|
|
293
|
+
self, collection_id: str, data: Dict, sync: bool = False
|
|
294
|
+
) -> Dict:
|
|
295
|
+
"""Update an existing collection using the given data.
|
|
296
|
+
|
|
297
|
+
params
|
|
298
|
+
------
|
|
299
|
+
collection_id: id of the collection to update
|
|
300
|
+
data: dict with foreign_id, label, category etc. See `CollectionSchema`
|
|
301
|
+
for more details.
|
|
302
|
+
"""
|
|
303
|
+
params = {"sync": sync}
|
|
304
|
+
url = self._make_url(f"collections/{collection_id}", params=params)
|
|
305
|
+
return self._request("PUT", url, json=data)
|
|
306
|
+
|
|
307
|
+
def stream_entities(
|
|
308
|
+
self,
|
|
309
|
+
collection: Optional[Dict] = None,
|
|
310
|
+
include: Optional[List] = None,
|
|
311
|
+
schema: Optional[str] = None,
|
|
312
|
+
publisher: bool = False,
|
|
313
|
+
) -> Iterator[Dict]:
|
|
314
|
+
"""Iterate over all entities in the given collection.
|
|
315
|
+
|
|
316
|
+
params
|
|
317
|
+
------
|
|
318
|
+
collection_id: id of the collection to stream
|
|
319
|
+
include: an array of fields from the index to include.
|
|
320
|
+
"""
|
|
321
|
+
url = self._make_url("entities/_stream")
|
|
322
|
+
if collection is not None:
|
|
323
|
+
collection_id = collection.get("id")
|
|
324
|
+
url = f"collections/{collection_id}/_stream"
|
|
325
|
+
url = self._make_url(url)
|
|
326
|
+
params = {"include": include, "schema": schema}
|
|
327
|
+
try:
|
|
328
|
+
res = self.session.get(url, params=params, stream=True)
|
|
329
|
+
res.raise_for_status()
|
|
330
|
+
for entity in res.iter_lines(chunk_size=None):
|
|
331
|
+
entity = json.loads(entity)
|
|
332
|
+
yield self._patch_entity(
|
|
333
|
+
entity, publisher=publisher, collection=collection
|
|
334
|
+
)
|
|
335
|
+
except (RequestException, HTTPError) as exc:
|
|
336
|
+
raise AlephException(exc) from exc
|
|
337
|
+
|
|
338
|
+
def _bulk_chunk(
|
|
339
|
+
self,
|
|
340
|
+
collection_id: str,
|
|
341
|
+
chunk: List,
|
|
342
|
+
entityset_id: Optional[str] = None,
|
|
343
|
+
force: bool = False,
|
|
344
|
+
unsafe: bool = False,
|
|
345
|
+
cleaned: bool = False,
|
|
346
|
+
):
|
|
347
|
+
for attempt in count(1):
|
|
348
|
+
url = self._make_url(f"collections/{collection_id}/_bulk")
|
|
349
|
+
params = {"entityset_id": entityset_id}
|
|
350
|
+
if unsafe:
|
|
351
|
+
params["safe"] = "false"
|
|
352
|
+
if cleaned:
|
|
353
|
+
params["clean"] = "false"
|
|
354
|
+
try:
|
|
355
|
+
response = self.session.post(url, json=chunk, params=params)
|
|
356
|
+
response.raise_for_status()
|
|
357
|
+
return
|
|
358
|
+
except (RequestException, HTTPError) as exc:
|
|
359
|
+
ae = AlephException(exc)
|
|
360
|
+
if not ae.transient or attempt > self.retries:
|
|
361
|
+
if not force:
|
|
362
|
+
raise ae from exc
|
|
363
|
+
log.error(ae)
|
|
364
|
+
return
|
|
365
|
+
backoff(ae, attempt)
|
|
366
|
+
|
|
367
|
+
def write_entity(
|
|
368
|
+
self, collection_id: str, entity: Dict, entity_id: Optional[str] = None, **kw
|
|
369
|
+
) -> Dict:
|
|
370
|
+
"""Create a single entity via the API, in the given
|
|
371
|
+
collection.
|
|
372
|
+
|
|
373
|
+
params
|
|
374
|
+
------
|
|
375
|
+
collection_id: id of the collection to use. This will overwrite any
|
|
376
|
+
existing collection specified in the entity dict
|
|
377
|
+
entity_id: id for the entity to be created. This will overwrite any
|
|
378
|
+
existing entity specified in the entity dict
|
|
379
|
+
entity: A dict object containing the values of the entity
|
|
380
|
+
"""
|
|
381
|
+
entity["collection_id"] = collection_id
|
|
382
|
+
|
|
383
|
+
if entity_id is not None:
|
|
384
|
+
entity["id"] = entity_id
|
|
385
|
+
|
|
386
|
+
for attempt in count(1):
|
|
387
|
+
if entity_id is not None:
|
|
388
|
+
url = self._make_url("entities/{}").format(entity_id)
|
|
389
|
+
else:
|
|
390
|
+
url = self._make_url("entities")
|
|
391
|
+
try:
|
|
392
|
+
return self._request("POST", url, json=entity)
|
|
393
|
+
except RequestException as exc:
|
|
394
|
+
ae = AlephException(exc)
|
|
395
|
+
if not ae.transient or attempt > self.retries:
|
|
396
|
+
log.error(ae)
|
|
397
|
+
raise exc
|
|
398
|
+
backoff(ae, attempt)
|
|
399
|
+
|
|
400
|
+
return {}
|
|
401
|
+
|
|
402
|
+
def write_entities(
|
|
403
|
+
self, collection_id: str, entities: Iterable, chunk_size: int = 1000, **kw
|
|
404
|
+
):
|
|
405
|
+
"""Create entities in bulk via the API, in the given
|
|
406
|
+
collection.
|
|
407
|
+
|
|
408
|
+
params
|
|
409
|
+
------
|
|
410
|
+
collection_id: id of the collection to use
|
|
411
|
+
entities: an iterable of entities to upload
|
|
412
|
+
"""
|
|
413
|
+
chunk = []
|
|
414
|
+
for entity in entities:
|
|
415
|
+
if hasattr(entity, "to_dict"):
|
|
416
|
+
entity = entity.to_dict()
|
|
417
|
+
chunk.append(entity)
|
|
418
|
+
if len(chunk) >= chunk_size:
|
|
419
|
+
self._bulk_chunk(collection_id, chunk, **kw)
|
|
420
|
+
chunk = []
|
|
421
|
+
if len(chunk):
|
|
422
|
+
self._bulk_chunk(collection_id, chunk, **kw)
|
|
423
|
+
|
|
424
|
+
def match(
|
|
425
|
+
self,
|
|
426
|
+
entity: Dict,
|
|
427
|
+
collection_ids: Optional[str] = None,
|
|
428
|
+
url: Optional[str] = None,
|
|
429
|
+
publisher: bool = False,
|
|
430
|
+
) -> Iterator[List]:
|
|
431
|
+
"""Find similar entities given a sample entity."""
|
|
432
|
+
params = {"collection_ids": ensure_list(collection_ids)}
|
|
433
|
+
if url is None:
|
|
434
|
+
url = self._make_url("match")
|
|
435
|
+
try:
|
|
436
|
+
response = self.session.post(url, json=entity, params=params) # type: ignore
|
|
437
|
+
response.raise_for_status()
|
|
438
|
+
for result in response.json().get("results", []):
|
|
439
|
+
yield self._patch_entity(result, publisher=publisher)
|
|
440
|
+
except (RequestException, HTTPError) as exc:
|
|
441
|
+
raise AlephException(exc) from exc
|
|
442
|
+
|
|
443
|
+
def entitysets(
|
|
444
|
+
self,
|
|
445
|
+
collection_id: Optional[str] = None,
|
|
446
|
+
set_types: Optional[List] = None,
|
|
447
|
+
prefix: Optional[str] = None,
|
|
448
|
+
) -> "APIResultSet":
|
|
449
|
+
"""Stream EntitySets"""
|
|
450
|
+
filters_collection = [("collection_id", collection_id)]
|
|
451
|
+
filters_type = [("type", t) for t in ensure_list(set_types)]
|
|
452
|
+
filters = [*filters_collection, *filters_type]
|
|
453
|
+
params = {"prefix": prefix}
|
|
454
|
+
url = self._make_url("entitysets", filters=filters, params=params)
|
|
455
|
+
return APIResultSet(self, url)
|
|
456
|
+
|
|
457
|
+
def entitysetitems(
|
|
458
|
+
self, entityset_id: str, publisher: bool = False
|
|
459
|
+
) -> "APIResultSet":
|
|
460
|
+
url = self._make_url(f"entitysets/{entityset_id}/items")
|
|
461
|
+
return EntitySetItemsResultSet(self, url, publisher=publisher)
|
|
462
|
+
|
|
463
|
+
def ingest_upload(
|
|
464
|
+
self,
|
|
465
|
+
collection_id: str,
|
|
466
|
+
file_path: Optional[Path] = None,
|
|
467
|
+
metadata: Optional[Dict] = None,
|
|
468
|
+
sync: bool = False,
|
|
469
|
+
index: bool = True,
|
|
470
|
+
) -> Dict:
|
|
471
|
+
"""
|
|
472
|
+
Create an empty folder in a collection or upload a document to it
|
|
473
|
+
|
|
474
|
+
params
|
|
475
|
+
------
|
|
476
|
+
collection_id: id of the collection to upload to
|
|
477
|
+
file_path: path of the file to upload. None while creating folders
|
|
478
|
+
metadata: dict containing metadata for the file or folders. In case of
|
|
479
|
+
files, metadata contains foreign_id of the parent. Metadata for a
|
|
480
|
+
directory contains foreign_id for itself as well as its parent and the
|
|
481
|
+
name of the directory.
|
|
482
|
+
"""
|
|
483
|
+
url_path = "collections/{0}/ingest".format(collection_id)
|
|
484
|
+
params = {"sync": sync, "index": index}
|
|
485
|
+
url = self._make_url(url_path, params=params)
|
|
486
|
+
if not file_path or file_path.is_dir():
|
|
487
|
+
data = {"meta": json.dumps(metadata)}
|
|
488
|
+
return self._request("POST", url, data=data)
|
|
489
|
+
|
|
490
|
+
for attempt in count(1):
|
|
491
|
+
try:
|
|
492
|
+
with file_path.open("rb") as fh:
|
|
493
|
+
# use multipart encoder to allow uploading very large files
|
|
494
|
+
m = MultipartEncoder(
|
|
495
|
+
fields={
|
|
496
|
+
"meta": json.dumps(metadata),
|
|
497
|
+
"file": (file_path.name, fh, MIME),
|
|
498
|
+
}
|
|
499
|
+
)
|
|
500
|
+
headers = {"Content-Type": m.content_type}
|
|
501
|
+
return self._request("POST", url, data=m, headers=headers)
|
|
502
|
+
except AlephException as ae:
|
|
503
|
+
if not ae.transient or attempt > self.retries:
|
|
504
|
+
raise ae from ae
|
|
505
|
+
backoff(ae, attempt)
|
|
506
|
+
return {}
|
|
507
|
+
|
|
508
|
+
def create_entityset(
|
|
509
|
+
self, collection_id: str, type: str, label: str, summary: Optional[str]
|
|
510
|
+
) -> Dict:
|
|
511
|
+
"""Create an EntitySet inside a collection"""
|
|
512
|
+
url = self._make_url("entitysets")
|
|
513
|
+
data: Dict = {
|
|
514
|
+
"collection_id": collection_id,
|
|
515
|
+
"type": type,
|
|
516
|
+
"label": label,
|
|
517
|
+
"summary": summary,
|
|
518
|
+
"entities": [],
|
|
519
|
+
}
|
|
520
|
+
return self._request("POST", url, data=data)
|
|
521
|
+
|
|
522
|
+
def delete_entityset(self, entityset_id: str, sync: bool = False):
|
|
523
|
+
"""Delete an EntitySet by id"""
|
|
524
|
+
url = self._make_url(f"entitysets/{entityset_id}", params={"sync": sync})
|
|
525
|
+
return self._request("DELETE", url)
|