deepsights-api 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- deepsights/__init__.py +23 -0
- deepsights/answers/__init__.py +26 -0
- deepsights/answers/answer.py +106 -0
- deepsights/answers/answer_v1.py +55 -0
- deepsights/answers/model.py +107 -0
- deepsights/api/__init__.py +22 -0
- deepsights/api/api.py +231 -0
- deepsights/api/model.py +96 -0
- deepsights/api/quota.py +54 -0
- deepsights/contentstore/__init__.py +24 -0
- deepsights/contentstore/_search.py +229 -0
- deepsights/contentstore/model.py +76 -0
- deepsights/contentstore/news.py +139 -0
- deepsights/contentstore/secondary.py +139 -0
- deepsights/documents/__init__.py +44 -0
- deepsights/documents/_cache.py +39 -0
- deepsights/documents/_segmenter.py +113 -0
- deepsights/documents/delete.py +82 -0
- deepsights/documents/download.py +63 -0
- deepsights/documents/load.py +172 -0
- deepsights/documents/model.py +161 -0
- deepsights/documents/search.py +182 -0
- deepsights/documents/upload.py +130 -0
- deepsights/minions/__init__.py +0 -0
- deepsights/minions/_minions.py +59 -0
- deepsights/reports/__init__.py +24 -0
- deepsights/reports/model.py +141 -0
- deepsights/reports/report.py +95 -0
- deepsights/utils/__init__.py +31 -0
- deepsights/utils/_cache.py +63 -0
- deepsights/utils/_ranking.py +201 -0
- deepsights/utils/_utils.py +45 -0
- deepsights/utils/model.py +91 -0
- deepsights_api-0.2.0.dist-info/LICENSE +201 -0
- deepsights_api-0.2.0.dist-info/METADATA +92 -0
- deepsights_api-0.2.0.dist-info/RECORD +39 -0
- deepsights_api-0.2.0.dist-info/WHEEL +5 -0
- deepsights_api-0.2.0.dist-info/top_level.txt +1 -0
- src/deepsights/__init__.py +23 -0
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# Copyright 2024 Market Logic Software AG. All Rights Reserved.
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""
|
|
16
|
+
This module contains the functions to retrieve documents from the DeepSights API.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from deepsights.documents._cache import (
|
|
20
|
+
set_document,
|
|
21
|
+
has_document,
|
|
22
|
+
get_document,
|
|
23
|
+
remove_document,
|
|
24
|
+
get_document_cache_size,
|
|
25
|
+
set_document_page,
|
|
26
|
+
has_document_page,
|
|
27
|
+
get_document_page,
|
|
28
|
+
remove_document_page,
|
|
29
|
+
get_document_page_cache_size,
|
|
30
|
+
)
|
|
31
|
+
from deepsights.documents.model import (
|
|
32
|
+
Document,
|
|
33
|
+
DocumentPage,
|
|
34
|
+
DocumentPageSearchResult,
|
|
35
|
+
DocumentSearchResult,
|
|
36
|
+
)
|
|
37
|
+
from deepsights.documents.upload import document_upload, document_wait_for_processing
|
|
38
|
+
from deepsights.documents.download import document_download
|
|
39
|
+
from deepsights.documents.delete import documents_delete, document_wait_for_deletion
|
|
40
|
+
from deepsights.documents.load import (
|
|
41
|
+
documents_load,
|
|
42
|
+
document_pages_load,
|
|
43
|
+
)
|
|
44
|
+
from deepsights.documents.search import documents_search, document_pages_search
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# Copyright 2024 Market Logic Software AG. All Rights Reserved.
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""
|
|
16
|
+
This module contains the functions to cache documents and document pages.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from deepsights.utils import create_global_lru_cache
|
|
20
|
+
|
|
21
|
+
#############################################
|
|
22
|
+
# a global static LRU cache for 1k docs
|
|
23
|
+
(
|
|
24
|
+
set_document,
|
|
25
|
+
has_document,
|
|
26
|
+
get_document,
|
|
27
|
+
remove_document,
|
|
28
|
+
get_document_cache_size,
|
|
29
|
+
) = create_global_lru_cache(1000)
|
|
30
|
+
|
|
31
|
+
#############################################
|
|
32
|
+
# a global static LRU cache for 100k pages
|
|
33
|
+
(
|
|
34
|
+
set_document_page,
|
|
35
|
+
has_document_page,
|
|
36
|
+
get_document_page,
|
|
37
|
+
remove_document_page,
|
|
38
|
+
get_document_page_cache_size,
|
|
39
|
+
) = create_global_lru_cache(100000)
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
# Copyright 2024 Market Logic Software AG. All Rights Reserved.
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""
|
|
16
|
+
This module contains the functions to segment the content of a document.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
import re
|
|
20
|
+
from typing import List, Dict
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
#############################################
|
|
24
|
+
def _parse_page(content):
|
|
25
|
+
"""
|
|
26
|
+
Construct segments as semantic units from the given page structure, guessing from font sizes.
|
|
27
|
+
Generates markdown for headings.
|
|
28
|
+
|
|
29
|
+
Args:
|
|
30
|
+
|
|
31
|
+
content (list): The list of tuples representing the page structure, where each tuple contains
|
|
32
|
+
the font size and the corresponding text.
|
|
33
|
+
|
|
34
|
+
Returns:
|
|
35
|
+
|
|
36
|
+
list: The list of segments as semantic units generated from the page structure.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
# no more content?
|
|
40
|
+
if len(content) == 0:
|
|
41
|
+
return []
|
|
42
|
+
|
|
43
|
+
# we want to find the next segment now and then recurse on the tail of the content
|
|
44
|
+
next_segment = ""
|
|
45
|
+
|
|
46
|
+
# track what font size we had last
|
|
47
|
+
last_font_size = None
|
|
48
|
+
|
|
49
|
+
# inspect next item
|
|
50
|
+
while len(content) > 0:
|
|
51
|
+
(next_font_size, next_text) = next_item = content.pop(0)
|
|
52
|
+
|
|
53
|
+
# check font size
|
|
54
|
+
if last_font_size is not None:
|
|
55
|
+
# compare to last
|
|
56
|
+
if next_font_size > 1.4 * last_font_size:
|
|
57
|
+
# new section
|
|
58
|
+
content.insert(0, next_item)
|
|
59
|
+
break
|
|
60
|
+
|
|
61
|
+
last_font_size = next_font_size
|
|
62
|
+
|
|
63
|
+
# if we have sth to append...
|
|
64
|
+
if next_text is not None:
|
|
65
|
+
# append text & remember page
|
|
66
|
+
next_segment += next_text + "\n"
|
|
67
|
+
|
|
68
|
+
# get rid of duplicate whitespace (except LF)
|
|
69
|
+
segment_text = re.sub(r"[^\S\n]+", " ", next_segment)
|
|
70
|
+
|
|
71
|
+
# get rid of 3+ LFs
|
|
72
|
+
segment_text = re.sub(r"\n{3,}", "\n\n", segment_text, re.MULTILINE)
|
|
73
|
+
|
|
74
|
+
# recurse on tail
|
|
75
|
+
segments = [segment_text]
|
|
76
|
+
segments.extend(_parse_page(content))
|
|
77
|
+
|
|
78
|
+
return segments
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
#############################################
|
|
82
|
+
def segment_landscape_page(page_structure: Dict) -> List[Dict]:
|
|
83
|
+
"""Segmentation strategy that follows pages
|
|
84
|
+
|
|
85
|
+
Args:
|
|
86
|
+
|
|
87
|
+
page_structure (Dict): The structure of the page.
|
|
88
|
+
|
|
89
|
+
Returns:
|
|
90
|
+
|
|
91
|
+
List[Dict]: A list of segmented sections.
|
|
92
|
+
|
|
93
|
+
"""
|
|
94
|
+
# collect font / text tuples by page
|
|
95
|
+
content = []
|
|
96
|
+
if page_structure["semantic_sections"] is not None:
|
|
97
|
+
for section in page_structure["semantic_sections"]:
|
|
98
|
+
for element in section["semantic_section_elements"]:
|
|
99
|
+
text = element["text_value"]["text"].strip()
|
|
100
|
+
|
|
101
|
+
font_size = 2 * int(
|
|
102
|
+
500
|
|
103
|
+
* float(element["normalized_layout"]["height"])
|
|
104
|
+
/ len(text.split("\n"))
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
content.append((font_size, text))
|
|
108
|
+
|
|
109
|
+
# collect segments
|
|
110
|
+
segments = list(filter(lambda s: len(s) > 40, _parse_page(content)))
|
|
111
|
+
|
|
112
|
+
# now merge into one
|
|
113
|
+
return "\n\n".join(segments)
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
# Copyright 2024 Market Logic Software AG. All Rights Reserved.
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""
|
|
16
|
+
This module contains the functions to delete documents from the DeepSights API.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
import time
|
|
20
|
+
from typing import List
|
|
21
|
+
import requests
|
|
22
|
+
from deepsights.api import DeepSights
|
|
23
|
+
from deepsights.documents._cache import remove_document
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
#################################################
|
|
27
|
+
def documents_delete(api: DeepSights, document_ids: List):
|
|
28
|
+
"""
|
|
29
|
+
Delete documents from the DeepSights API.
|
|
30
|
+
|
|
31
|
+
Args:
|
|
32
|
+
|
|
33
|
+
api (DeepSights): An instance of the DeepSights API client.
|
|
34
|
+
document_ids (List): A list of document IDs to be deleted.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
# delete documents one by one
|
|
38
|
+
for document_id in document_ids:
|
|
39
|
+
api.delete(
|
|
40
|
+
f"/artifact-service/artifacts/{document_id}",
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
# remove from cache
|
|
44
|
+
remove_document(document_id)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
#################################################
|
|
48
|
+
def document_wait_for_deletion(api: DeepSights, document_id: str, timeout: int = 60):
|
|
49
|
+
"""
|
|
50
|
+
Wait for the document to be deleted.
|
|
51
|
+
|
|
52
|
+
Args:
|
|
53
|
+
|
|
54
|
+
api (DeepSights): An instance of the DeepSights API client.
|
|
55
|
+
document_id (str): The ID of the document to wait for.
|
|
56
|
+
timeout (int, optional): The maximum time to wait for the document to be deleted, in seconds. Defaults to 60.
|
|
57
|
+
|
|
58
|
+
Raises:
|
|
59
|
+
|
|
60
|
+
TimeoutError: If the document fails to delete within the specified timeout.
|
|
61
|
+
ValueError: If the document deletion fails with an error message.
|
|
62
|
+
"""
|
|
63
|
+
# wait for completion
|
|
64
|
+
start = time.time()
|
|
65
|
+
while time.time() - start < timeout:
|
|
66
|
+
try:
|
|
67
|
+
response = api.get(f"/artifact-service/artifacts/{document_id}")
|
|
68
|
+
except requests.exceptions.HTTPError as e:
|
|
69
|
+
if e.response.status_code == 404:
|
|
70
|
+
remove_document(document_id)
|
|
71
|
+
return
|
|
72
|
+
|
|
73
|
+
raise e
|
|
74
|
+
|
|
75
|
+
if response["status"] in ("DELETING", "SCHEDULED_FOR_DELETING"):
|
|
76
|
+
time.sleep(2)
|
|
77
|
+
elif response["status"].startswith("FAILED"):
|
|
78
|
+
raise ValueError(
|
|
79
|
+
f"Document {document_id} failed to delete: {response['error_message']}"
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
raise TimeoutError(f"Document {document_id} failed to delete in {timeout} seconds.")
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
# Copyright 2024 Market Logic Software AG. All Rights Reserved.
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""
|
|
16
|
+
This module contains the functions to download documents to the DeepSights API.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
import os
|
|
20
|
+
import urllib.request
|
|
21
|
+
import tempfile
|
|
22
|
+
from deepsights.api import DeepSights
|
|
23
|
+
from deepsights.documents.load import documents_load
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
#################################################
|
|
27
|
+
def document_download(api: DeepSights, document_id: str, local_path: str):
|
|
28
|
+
"""
|
|
29
|
+
Download a document from the DeepSights API.
|
|
30
|
+
|
|
31
|
+
Args:
|
|
32
|
+
|
|
33
|
+
api (DeepSights): An instance of the DeepSights API client.
|
|
34
|
+
document_id (str): The ID of the document to download.
|
|
35
|
+
local_path (str): The local path to save the downloaded document in.
|
|
36
|
+
|
|
37
|
+
Raises:
|
|
38
|
+
|
|
39
|
+
FileNotFoundError: If the local directory does not exist.
|
|
40
|
+
ValueError: If the document fails to download.
|
|
41
|
+
|
|
42
|
+
Returns:
|
|
43
|
+
|
|
44
|
+
Tuple[str, str]: A tuple containing the file name and the local path of the downloaded document.
|
|
45
|
+
"""
|
|
46
|
+
# check if local path exists
|
|
47
|
+
if not os.path.exists(local_path):
|
|
48
|
+
raise FileNotFoundError(f"Local directory {local_path} does not exist.")
|
|
49
|
+
|
|
50
|
+
# obtain real filename
|
|
51
|
+
document = documents_load(api, [document_id])[0]
|
|
52
|
+
|
|
53
|
+
# obtain download link
|
|
54
|
+
response = api.get(
|
|
55
|
+
f"/artifact-service/artifacts/{document_id}/gcs-object-link",
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
# download document to temp file
|
|
59
|
+
local_filename = tempfile.mktemp(dir=local_path)
|
|
60
|
+
urllib.request.urlretrieve(response["signed_link"], local_filename)
|
|
61
|
+
|
|
62
|
+
# return the filename
|
|
63
|
+
return document.file_name, local_filename
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
# Copyright 2024 Market Logic Software AG. All Rights Reserved.
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""
|
|
16
|
+
This module contains the functions to load documents from the DeepSights API.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from typing import List
|
|
20
|
+
from deepsights.api import DeepSights
|
|
21
|
+
from deepsights.utils import run_in_parallel
|
|
22
|
+
from deepsights.documents._cache import (
|
|
23
|
+
get_document,
|
|
24
|
+
get_document_cache_size,
|
|
25
|
+
has_document,
|
|
26
|
+
set_document,
|
|
27
|
+
get_document_page,
|
|
28
|
+
get_document_page_cache_size,
|
|
29
|
+
has_document_page,
|
|
30
|
+
set_document_page,
|
|
31
|
+
)
|
|
32
|
+
from deepsights.documents.model import Document, DocumentPage
|
|
33
|
+
from deepsights.documents._segmenter import segment_landscape_page
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
#################################################
|
|
37
|
+
def document_pages_load(api: DeepSights, page_ids: List[str]):
|
|
38
|
+
"""
|
|
39
|
+
Load document pages from the cache or fetch them from the API if not cached.
|
|
40
|
+
|
|
41
|
+
Args:
|
|
42
|
+
|
|
43
|
+
api (DeepSights): The DeepSights API object.
|
|
44
|
+
page_ids (List[str]): A list of page IDs to load.
|
|
45
|
+
|
|
46
|
+
Returns:
|
|
47
|
+
|
|
48
|
+
List[DocumentPage]: A list of loaded document pages.
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
assert (
|
|
52
|
+
len(page_ids) < get_document_page_cache_size()
|
|
53
|
+
), "Cannot load more document pages than the cache size."
|
|
54
|
+
|
|
55
|
+
# touch cached document pages
|
|
56
|
+
for page_id in page_ids:
|
|
57
|
+
get_document_page(page_id)
|
|
58
|
+
|
|
59
|
+
# filter uncached document pages
|
|
60
|
+
uncached_document_page_ids = [
|
|
61
|
+
page_id for page_id in page_ids if not has_document_page(page_id)
|
|
62
|
+
]
|
|
63
|
+
|
|
64
|
+
# load uncached document pages
|
|
65
|
+
def _load_document_page(page_id: str):
|
|
66
|
+
result = api.get(f"/artifact-service/pages/{page_id}", timeout=5)
|
|
67
|
+
|
|
68
|
+
# map the document page
|
|
69
|
+
return DocumentPage(
|
|
70
|
+
id=result["id"],
|
|
71
|
+
page_number=result["number"],
|
|
72
|
+
text=segment_landscape_page(result),
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
uncached_document_pages = run_in_parallel(
|
|
76
|
+
_load_document_page, uncached_document_page_ids, max_workers=5
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
# set in cache
|
|
80
|
+
for page in uncached_document_pages:
|
|
81
|
+
set_document_page(page.id, page)
|
|
82
|
+
|
|
83
|
+
# collect results
|
|
84
|
+
return [get_document_page(page_id) for page_id in page_ids]
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
#################################################
|
|
88
|
+
def documents_load(
|
|
89
|
+
api: DeepSights,
|
|
90
|
+
document_ids: List[str],
|
|
91
|
+
force_load: bool = False,
|
|
92
|
+
load_pages: bool = False,
|
|
93
|
+
):
|
|
94
|
+
"""
|
|
95
|
+
Load documents from the DeepSights API.
|
|
96
|
+
|
|
97
|
+
Args:
|
|
98
|
+
|
|
99
|
+
api (DeepSights): The DeepSights API object.
|
|
100
|
+
document_ids (List[str]): A list of document IDs to load.
|
|
101
|
+
force_load (bool, optional): Whether to force load the documents, even if in cache. Defaults to False.
|
|
102
|
+
load_pages (bool, optional): Whether to load the pages of the documents. Defaults to False.
|
|
103
|
+
|
|
104
|
+
Returns:
|
|
105
|
+
|
|
106
|
+
List[Document]: A list of loaded documents.
|
|
107
|
+
"""
|
|
108
|
+
assert (
|
|
109
|
+
len(document_ids) < get_document_cache_size()
|
|
110
|
+
), "Cannot load more documents than the cache size."
|
|
111
|
+
|
|
112
|
+
# touch cached documents
|
|
113
|
+
if not force_load:
|
|
114
|
+
for doc_id in document_ids:
|
|
115
|
+
get_document(doc_id)
|
|
116
|
+
|
|
117
|
+
# filter uncached documents
|
|
118
|
+
uncached_document_ids = [
|
|
119
|
+
doc_id for doc_id in document_ids if force_load or not has_document(doc_id)
|
|
120
|
+
]
|
|
121
|
+
|
|
122
|
+
# load uncached documents
|
|
123
|
+
def _load_document(document_id: str):
|
|
124
|
+
result = api.get(f"/artifact-service/artifacts/{document_id}", timeout=5)
|
|
125
|
+
|
|
126
|
+
# capitalze the first letter of the summary
|
|
127
|
+
result["summary"] = result["summary"][0].upper() + result["summary"][1:]
|
|
128
|
+
|
|
129
|
+
# map the document
|
|
130
|
+
return Document.model_validate(result)
|
|
131
|
+
|
|
132
|
+
uncached_documents = run_in_parallel(
|
|
133
|
+
_load_document, uncached_document_ids, max_workers=5
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
# set in cache
|
|
137
|
+
for doc in uncached_documents:
|
|
138
|
+
set_document(doc.id, doc)
|
|
139
|
+
|
|
140
|
+
# load pages if desired
|
|
141
|
+
if load_pages:
|
|
142
|
+
# collect docs that need page loading
|
|
143
|
+
document_ids_to_load_pages = [
|
|
144
|
+
doc_id for doc_id in document_ids if not get_document(doc_id).page_ids
|
|
145
|
+
]
|
|
146
|
+
|
|
147
|
+
# load pages
|
|
148
|
+
def _load_pages(document_id: str):
|
|
149
|
+
document = get_document(document_id)
|
|
150
|
+
|
|
151
|
+
# load page ids
|
|
152
|
+
result = api.get(
|
|
153
|
+
f"/artifact-service/artifacts/{document_id}/page-ids", timeout=5
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
# set page ids
|
|
157
|
+
document.page_ids = result["ids"]
|
|
158
|
+
|
|
159
|
+
return document.page_ids
|
|
160
|
+
|
|
161
|
+
page_ids = run_in_parallel(
|
|
162
|
+
_load_pages, document_ids_to_load_pages, max_workers=5
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
# flatten page ids
|
|
166
|
+
page_ids = [page_id for page_ids in page_ids for page_id in page_ids]
|
|
167
|
+
|
|
168
|
+
# now load actual pages
|
|
169
|
+
document_pages_load(api, page_ids)
|
|
170
|
+
|
|
171
|
+
# collect results
|
|
172
|
+
return [get_document(doc_id) for doc_id in document_ids]
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
# Copyright 2024 Market Logic Software AG. All Rights Reserved.
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""
|
|
16
|
+
This module contains the model classes for documents.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from typing import List, Optional
|
|
20
|
+
from datetime import datetime
|
|
21
|
+
from pydantic import Field
|
|
22
|
+
from deepsights.utils import DeepSightsIdModel, DeepSightsIdTitleModel
|
|
23
|
+
from deepsights.documents._cache import get_document_page, get_document
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
#################################################
|
|
27
|
+
class DocumentPage(DeepSightsIdModel):
|
|
28
|
+
"""
|
|
29
|
+
Represents a document page.
|
|
30
|
+
|
|
31
|
+
Attributes:
|
|
32
|
+
|
|
33
|
+
page_number (Optional[int], optional): The number of the page.
|
|
34
|
+
text (Optional[str]): The text content of the page (optional).
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
page_number: Optional[int] = Field(
|
|
38
|
+
default=None, description="The number of the page (one-based)."
|
|
39
|
+
)
|
|
40
|
+
text: Optional[str] = Field(
|
|
41
|
+
default=None, description="The text content of the page."
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
#################################################
|
|
46
|
+
class Document(DeepSightsIdTitleModel):
|
|
47
|
+
"""
|
|
48
|
+
Represents a document.
|
|
49
|
+
|
|
50
|
+
Attributes:
|
|
51
|
+
|
|
52
|
+
status (str, optional): The status of the document.
|
|
53
|
+
source (str, optional): The source of the document.
|
|
54
|
+
file_name (str, optional): The name of the file.
|
|
55
|
+
file_size (int, optional): The size of the file.
|
|
56
|
+
description (str, optional): The description of the document.
|
|
57
|
+
timestamp (datetime, optional): The timestamp of the document.
|
|
58
|
+
page_ids (List[DocumentPage], optional): The list of page IDs in the document.
|
|
59
|
+
number_of_pages (int, optional): The total number of pages in the document.
|
|
60
|
+
"""
|
|
61
|
+
|
|
62
|
+
status: Optional[str] = Field(description="The processing status of the document.")
|
|
63
|
+
source: Optional[str] = Field(
|
|
64
|
+
alias="ai_generated_source",
|
|
65
|
+
default="n/a",
|
|
66
|
+
description="The human-readable source of the document.",
|
|
67
|
+
)
|
|
68
|
+
file_name: Optional[str] = Field(default=None, description="The name of the file.")
|
|
69
|
+
file_size: Optional[int] = Field(
|
|
70
|
+
default=None, description="The size of the file in bytes."
|
|
71
|
+
)
|
|
72
|
+
description: Optional[str] = Field(
|
|
73
|
+
alias="summary", description="The human-readable summary of the document."
|
|
74
|
+
)
|
|
75
|
+
timestamp: Optional[datetime] = Field(
|
|
76
|
+
alias="publication_date",
|
|
77
|
+
default=None,
|
|
78
|
+
description="The publication timestamp of the document.",
|
|
79
|
+
)
|
|
80
|
+
page_ids: List[DocumentPage] = Field(
|
|
81
|
+
default=None, description="The list of page IDs in the document."
|
|
82
|
+
)
|
|
83
|
+
number_of_pages: Optional[int] = Field(
|
|
84
|
+
alias="total_pages",
|
|
85
|
+
default=None,
|
|
86
|
+
description="The total number of pages in the document.",
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
@property
|
|
90
|
+
def pages(self) -> List[DocumentPage]:
|
|
91
|
+
return [get_document_page(page_id) for page_id in self.page_ids]
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
#################################################
|
|
95
|
+
class DocumentPageSearchResult(DeepSightsIdModel):
|
|
96
|
+
"""
|
|
97
|
+
Represents a search result for a page.
|
|
98
|
+
|
|
99
|
+
Attributes:
|
|
100
|
+
|
|
101
|
+
document_id (str): The ID of the document.
|
|
102
|
+
score (float): The score of the search result.
|
|
103
|
+
"""
|
|
104
|
+
|
|
105
|
+
document_id: str = Field(
|
|
106
|
+
description="The ID of the document to which the page belongs."
|
|
107
|
+
)
|
|
108
|
+
score: float = Field(description="The score of the search result.")
|
|
109
|
+
|
|
110
|
+
@property
|
|
111
|
+
def page_number(self) -> DocumentPage:
|
|
112
|
+
page = get_document_page(self.id)
|
|
113
|
+
return page.page_number if page else None
|
|
114
|
+
|
|
115
|
+
@property
|
|
116
|
+
def text(self) -> str:
|
|
117
|
+
page = get_document_page(self.id)
|
|
118
|
+
return page.text if page else None
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
#################################################
|
|
122
|
+
class DocumentSearchResult(DeepSightsIdModel):
|
|
123
|
+
"""
|
|
124
|
+
Represents the search result for a document.
|
|
125
|
+
|
|
126
|
+
Attributes:
|
|
127
|
+
|
|
128
|
+
page_matches (List[DocumentPageSearchResult]): The search results for each page of the document.
|
|
129
|
+
score_rank (Optional[int]): The rank of the document based on the score.
|
|
130
|
+
age_rank (Optional[int]): The rank of the document based on the age.
|
|
131
|
+
rank (Optional[int]): The overall rank of the document.
|
|
132
|
+
"""
|
|
133
|
+
|
|
134
|
+
page_matches: List[DocumentPageSearchResult] = Field(
|
|
135
|
+
default=[], description="The matching page search results for the document."
|
|
136
|
+
)
|
|
137
|
+
rank: Optional[int] = Field(
|
|
138
|
+
default=None, description="The final rank of the item in the search results."
|
|
139
|
+
)
|
|
140
|
+
score_rank: Optional[int] = Field(
|
|
141
|
+
default=None, description="The rank of the item based on its score."
|
|
142
|
+
)
|
|
143
|
+
age_rank: Optional[int] = Field(
|
|
144
|
+
default=None, description="The rank of the item based on its age; may be None."
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
@property
|
|
148
|
+
def document(self) -> Document:
|
|
149
|
+
return get_document(self.id)
|
|
150
|
+
|
|
151
|
+
@property
|
|
152
|
+
def timestamp(self) -> datetime:
|
|
153
|
+
document = self.document
|
|
154
|
+
return document.timestamp if document else None
|
|
155
|
+
|
|
156
|
+
#############################################
|
|
157
|
+
def __repr__(self) -> str:
|
|
158
|
+
if self.document is not None:
|
|
159
|
+
return f"{self.__class__.__name__}@{self.id}: {self.document.title}"
|
|
160
|
+
|
|
161
|
+
return super().__repr__()
|