PyPI - deepdoc-lib - Versions diffs - 0.2.0__py3-none-any.whl - Mend

deepdoc-lib 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (78) hide show

deepdoc/README.md +122 -0
deepdoc/README_zh.md +116 -0
deepdoc/__init__.py +43 -0
deepdoc/_version.py +34 -0
deepdoc/common/__init__.py +52 -0
deepdoc/common/config_utils.py +63 -0
deepdoc/common/connection_utils.py +73 -0
deepdoc/common/file_utils.py +19 -0
deepdoc/common/misc_utils.py +44 -0
deepdoc/common/model_store.py +369 -0
deepdoc/common/settings.py +42 -0
deepdoc/common/tiktoken_cache.py +84 -0
deepdoc/common/token_utils.py +96 -0
deepdoc/config.py +149 -0
deepdoc/depend/find_codec.py +42 -0
deepdoc/depend/nltk_manager.py +114 -0
deepdoc/depend/prompts/vision_llm_describe_prompt.md +23 -0
deepdoc/depend/prompts/vision_llm_figure_describe_prompt.md +24 -0
deepdoc/depend/prompts.py +35 -0
deepdoc/depend/rag_tokenizer.py +578 -0
deepdoc/depend/simple_cv_model.py +469 -0
deepdoc/depend/surname.py +91 -0
deepdoc/depend/timeout.py +73 -0
deepdoc/depend/vision_llm_chunk.py +35 -0
deepdoc/dict/README.md +19 -0
deepdoc/dict/huqie.txt +555629 -0
deepdoc/download_models.py +169 -0
deepdoc/llm_adapter/__init__.py +15 -0
deepdoc/llm_adapter/adapter.py +223 -0
deepdoc/llm_adapter/utils.py +104 -0
deepdoc/llm_adapter/vision.py +163 -0
deepdoc/parser/__init__.py +42 -0
deepdoc/parser/docling_parser.py +889 -0
deepdoc/parser/docx_parser.py +150 -0
deepdoc/parser/excel_parser.py +270 -0
deepdoc/parser/figure_parser.py +182 -0
deepdoc/parser/html_parser.py +221 -0
deepdoc/parser/json_parser.py +179 -0
deepdoc/parser/markdown_parser.py +321 -0
deepdoc/parser/mineru_parser.py +646 -0
deepdoc/parser/pdf_parser.py +1591 -0
deepdoc/parser/ppt_parser.py +96 -0
deepdoc/parser/resume/__init__.py +109 -0
deepdoc/parser/resume/entities/__init__.py +15 -0
deepdoc/parser/resume/entities/corporations.py +128 -0
deepdoc/parser/resume/entities/degrees.py +44 -0
deepdoc/parser/resume/entities/industries.py +712 -0
deepdoc/parser/resume/entities/regions.py +789 -0
deepdoc/parser/resume/entities/res/corp.tks.freq.json +65 -0
deepdoc/parser/resume/entities/res/corp_baike_len.csv +31480 -0
deepdoc/parser/resume/entities/res/corp_tag.json +14939 -0
deepdoc/parser/resume/entities/res/good_corp.json +911 -0
deepdoc/parser/resume/entities/res/good_sch.json +595 -0
deepdoc/parser/resume/entities/res/school.rank.csv +1627 -0
deepdoc/parser/resume/entities/res/schools.csv +5713 -0
deepdoc/parser/resume/entities/schools.py +91 -0
deepdoc/parser/resume/step_one.py +189 -0
deepdoc/parser/resume/step_two.py +692 -0
deepdoc/parser/tcadp_parser.py +538 -0
deepdoc/parser/txt_parser.py +64 -0
deepdoc/parser/utils.py +33 -0
deepdoc/vision/__init__.py +90 -0
deepdoc/vision/layout_recognizer.py +481 -0
deepdoc/vision/ocr.py +757 -0
deepdoc/vision/operators.py +733 -0
deepdoc/vision/postprocess.py +370 -0
deepdoc/vision/recognizer.py +451 -0
deepdoc/vision/seeit.py +87 -0
deepdoc/vision/t_ocr.py +101 -0
deepdoc/vision/t_recognizer.py +186 -0
deepdoc/vision/table_structure_recognizer.py +617 -0
deepdoc_lib-0.2.0.dist-info/METADATA +246 -0
deepdoc_lib-0.2.0.dist-info/RECORD +78 -0
deepdoc_lib-0.2.0.dist-info/WHEEL +5 -0
deepdoc_lib-0.2.0.dist-info/entry_points.txt +2 -0
deepdoc_lib-0.2.0.dist-info/licenses/LICENSE +201 -0
deepdoc_lib-0.2.0.dist-info/top_level.txt +2 -0
scripts/download_models.py +10 -0

deepdoc/parser/html_parser.py ADDED Viewed

@@ -0,0 +1,221 @@
+# -*- coding: utf-8 -*-
+#
+#  Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
+#
+#  Licensed under the Apache License, Version 2.0 (the "License");
+#  you may not use this file except in compliance with the License.
+#  You may obtain a copy of the License at
+#
+#      http://www.apache.org/licenses/LICENSE-2.0
+#
+#  Unless required by applicable law or agreed to in writing, software
+#  distributed under the License is distributed on an "AS IS" BASIS,
+#  WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+#  See the License for the specific language governing permissions and
+#  limitations under the License.
+#
+import html
+import uuid
+import chardet
+from bs4 import BeautifulSoup, NavigableString, Tag, Comment
+from ..config import TokenizerConfig
+from ..depend.find_codec import find_codec
+from ..depend.rag_tokenizer import RagTokenizer
+def get_encoding(file):
+    with open(file,'rb') as f:
+        tmp = chardet.detect(f.read())
+        return tmp['encoding']
+BLOCK_TAGS = [
+    "h1", "h2", "h3", "h4", "h5", "h6",
+    "p", "div", "article", "section", "aside",
+    "ul", "ol", "li",
+    "table", "pre", "code", "blockquote",
+    "figure", "figcaption"
+]
+TITLE_TAGS = {"h1": "#", "h2": "##", "h3": "###", "h4": "#####", "h5": "#####", "h6": "######"}
+class RAGFlowHtmlParser:
+    def __init__(self, tokenizer_cfg: TokenizerConfig | None = None):
+        if tokenizer_cfg is None:
+            tokenizer_cfg = TokenizerConfig.from_env()
+        self.tokenizer_cfg = tokenizer_cfg
+        self.tokenizer = RagTokenizer(
+            dict_prefix=tokenizer_cfg.resolve_dict_prefix(),
+            offline=tokenizer_cfg.offline,
+            nltk_data_dir=tokenizer_cfg.nltk_data_dir,
+        )
+    def __call__(self, fnm, binary=None, chunk_token_num=512):
+        if binary:
+            encoding = find_codec(binary)
+            txt = binary.decode(encoding, errors="ignore")
+        else:
+            with open(fnm, "r",encoding=get_encoding(fnm)) as f:
+                txt = f.read()
+        return self.parser_txt(txt, chunk_token_num)
+    def parser_txt(self, txt, chunk_token_num):
+        if not isinstance(txt, str):
+            raise TypeError("txt type should be string!")
+        temp_sections = []
+        soup = BeautifulSoup(txt, "html5lib")
+        # delete <style> tag
+        for style_tag in soup.find_all(["style", "script"]):
+            style_tag.decompose()
+        # delete <script> tag in <div>
+        for div_tag in soup.find_all("div"):
+            for script_tag in div_tag.find_all("script"):
+                script_tag.decompose()
+        # delete inline style
+        for tag in soup.find_all(True):
+            if 'style' in tag.attrs:
+                del tag.attrs['style']
+        # delete HTML comment
+        for comment in soup.find_all(string=lambda text: isinstance(text, Comment)):
+            comment.extract()
+        self.read_text_recursively(soup.body, temp_sections, chunk_token_num=chunk_token_num)
+        block_txt_list, table_list = self.merge_block_text(temp_sections)
+        sections = self.chunk_block(block_txt_list, chunk_token_num=chunk_token_num)
+        for table in table_list:
+            sections.append(table.get("content", ""))
+        return sections
+    def split_table(self, html_table, chunk_token_num=512):
+        soup = BeautifulSoup(html_table, "html.parser")
+        rows = soup.find_all("tr")
+        tables = []
+        current_table = []
+        current_count = 0
+        table_str_list = []
+        for row in rows:
+            tks_str = self.tokenizer.tokenize(str(row))
+            token_count = len(tks_str.split(" ")) if tks_str else 0
+            if current_count + token_count > chunk_token_num:
+                tables.append(current_table)
+                current_table = []
+                current_count = 0
+            current_table.append(row)
+            current_count += token_count
+        if current_table:
+            tables.append(current_table)
+        for table_rows in tables:
+            new_table = soup.new_tag("table")
+            for row in table_rows:
+                new_table.append(row)
+            table_str_list.append(str(new_table))
+        return table_str_list
+    def read_text_recursively(self, element, parser_result, chunk_token_num=512, parent_name=None, block_id=None):
+        if isinstance(element, NavigableString):
+            content = element.strip()
+            def is_valid_html(content):
+                try:
+                    soup = BeautifulSoup(content, "html.parser")
+                    return bool(soup.find())
+                except Exception:
+                    return False
+            return_info = []
+            if content:
+                if is_valid_html(content):
+                    soup = BeautifulSoup(content, "html.parser")
+                    child_info = self.read_text_recursively(soup, parser_result, chunk_token_num, element.name, block_id)
+                    parser_result.extend(child_info)
+                else:
+                    info = {"content": element.strip(), "tag_name": "inner_text", "metadata": {"block_id": block_id}}
+                    if parent_name:
+                        info["tag_name"] = parent_name
+                    return_info.append(info)
+            return return_info
+        elif isinstance(element, Tag):
+            if str.lower(element.name) == "table":
+                table_info_list = []
+                table_id = str(uuid.uuid1())
+                table_list = [html.unescape(str(element))]
+                for t in table_list:
+                    table_info_list.append({"content": t, "tag_name": "table",
+                                            "metadata": {"table_id": table_id, "index": table_list.index(t)}})
+                return table_info_list
+            else:
+                if str.lower(element.name) in BLOCK_TAGS:
+                    block_id = str(uuid.uuid1())
+                for child in element.children:
+                    child_info = self.read_text_recursively(child, parser_result, chunk_token_num, element.name,
+                                                           block_id)
+                    parser_result.extend(child_info)
+        return []
+    def merge_block_text(self, parser_result):
+        block_content = []
+        current_content = ""
+        table_info_list = []
+        last_block_id = None
+        for item in parser_result:
+            content = item.get("content")
+            tag_name = item.get("tag_name")
+            title_flag = tag_name in TITLE_TAGS
+            block_id = item.get("metadata", {}).get("block_id")
+            if block_id:
+                if title_flag:
+                    content = f"{TITLE_TAGS[tag_name]} {content}"
+                if last_block_id != block_id:
+                    if last_block_id is not None:
+                        block_content.append(current_content)
+                    current_content = content
+                    last_block_id = block_id
+                else:
+                    current_content += (" " if current_content else "") + content
+            else:
+                if tag_name == "table":
+                    table_info_list.append(item)
+                else:
+                    current_content += (" " if current_content else "") + content
+        if current_content:
+            block_content.append(current_content)
+        return block_content, table_info_list
+    def chunk_block(self, block_txt_list, chunk_token_num=512):
+        chunks = []
+        current_block = ""
+        current_token_count = 0
+        for block in block_txt_list:
+            tks_str = self.tokenizer.tokenize(block)
+            block_token_count = len(tks_str.split(" ")) if tks_str else 0
+            if block_token_count > chunk_token_num:
+                if current_block:
+                    chunks.append(current_block)
+                start = 0
+                tokens = tks_str.split(" ")
+                while start < len(tokens):
+                    end = start + chunk_token_num
+                    split_tokens = tokens[start:end]
+                    chunks.append(" ".join(split_tokens))
+                    start = end
+                current_block = ""
+                current_token_count = 0
+            else:
+                if current_token_count + block_token_count <= chunk_token_num:
+                    current_block += ("\n" if current_block else "") + block
+                    current_token_count += block_token_count
+                else:
+                    chunks.append(current_block)
+                    current_block = block
+                    current_token_count = block_token_count
+        if current_block:
+            chunks.append(current_block)
+        return chunks

deepdoc/parser/json_parser.py ADDED Viewed

@@ -0,0 +1,179 @@
+# -*- coding: utf-8 -*-
+#
+#  Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
+#
+#  Licensed under the Apache License, Version 2.0 (the "License");
+#  you may not use this file except in compliance with the License.
+#  You may obtain a copy of the License at
+#
+#      http://www.apache.org/licenses/LICENSE-2.0
+#
+#  Unless required by applicable law or agreed to in writing, software
+#  distributed under the License is distributed on an "AS IS" BASIS,
+#  WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+#  See the License for the specific language governing permissions and
+#  limitations under the License.
+#
+# The following documents are mainly referenced, and only adaptation modifications have been made
+# from https://github.com/langchain-ai/langchain/blob/master/libs/text-splitters/langchain_text_splitters/json.py
+import json
+from typing import Any
+from ..depend.find_codec import find_codec
+class RAGFlowJsonParser:
+    def __init__(self, max_chunk_size: int = 2000, min_chunk_size: int | None = None):
+        super().__init__()
+        self.max_chunk_size = max_chunk_size * 2
+        self.min_chunk_size = min_chunk_size if min_chunk_size is not None else max(max_chunk_size - 200, 50)
+    def __call__(self, binary):
+        encoding = find_codec(binary)
+        txt = binary.decode(encoding, errors="ignore")
+        if self.is_jsonl_format(txt):
+            sections = self._parse_jsonl(txt)
+        else:
+            sections = self._parse_json(txt)
+        return sections
+    @staticmethod
+    def _json_size(data: dict) -> int:
+        """Calculate the size of the serialized JSON object."""
+        return len(json.dumps(data, ensure_ascii=False))
+    @staticmethod
+    def _set_nested_dict(d: dict, path: list[str], value: Any) -> None:
+        """Set a value in a nested dictionary based on the given path."""
+        for key in path[:-1]:
+            d = d.setdefault(key, {})
+        d[path[-1]] = value
+    def _list_to_dict_preprocessing(self, data: Any) -> Any:
+        if isinstance(data, dict):
+            # Process each key-value pair in the dictionary
+            return {k: self._list_to_dict_preprocessing(v) for k, v in data.items()}
+        elif isinstance(data, list):
+            # Convert the list to a dictionary with index-based keys
+            return {str(i): self._list_to_dict_preprocessing(item) for i, item in enumerate(data)}
+        else:
+            # Base case: the item is neither a dict nor a list, so return it unchanged
+            return data
+    def _json_split(
+        self,
+        data,
+        current_path: list[str] | None,
+        chunks: list[dict] | None,
+    ) -> list[dict]:
+        """
+        Split json into maximum size dictionaries while preserving structure.
+        """
+        current_path = current_path or []
+        chunks = chunks or [{}]
+        if isinstance(data, dict):
+            for key, value in data.items():
+                new_path = current_path + [key]
+                chunk_size = self._json_size(chunks[-1])
+                size = self._json_size({key: value})
+                remaining = self.max_chunk_size - chunk_size
+                if size < remaining:
+                    # Add item to current chunk
+                    self._set_nested_dict(chunks[-1], new_path, value)
+                else:
+                    if chunk_size >= self.min_chunk_size:
+                        # Chunk is big enough, start a new chunk
+                        chunks.append({})
+                    # Iterate
+                    self._json_split(value, new_path, chunks)
+        else:
+            # handle single item
+            self._set_nested_dict(chunks[-1], current_path, data)
+        return chunks
+    def split_json(
+        self,
+        json_data,
+        convert_lists: bool = False,
+    ) -> list[dict]:
+        """Splits JSON into a list of JSON chunks"""
+        if convert_lists:
+            preprocessed_data = self._list_to_dict_preprocessing(json_data)
+            chunks = self._json_split(preprocessed_data, None, None)
+        else:
+            chunks = self._json_split(json_data, None, None)
+        # Remove the last chunk if it's empty
+        if not chunks[-1]:
+            chunks.pop()
+        return chunks
+    def split_text(
+        self,
+        json_data: dict[str, Any],
+        convert_lists: bool = False,
+        ensure_ascii: bool = True,
+    ) -> list[str]:
+        """Splits JSON into a list of JSON formatted strings"""
+        chunks = self.split_json(json_data=json_data, convert_lists=convert_lists)
+        # Convert to string
+        return [json.dumps(chunk, ensure_ascii=ensure_ascii) for chunk in chunks]
+    def _parse_json(self, content: str) -> list[str]:
+        sections = []
+        try:
+            json_data = json.loads(content)
+            chunks = self.split_json(json_data, True)
+            sections = [json.dumps(line, ensure_ascii=False) for line in chunks if line]
+        except json.JSONDecodeError:
+            pass
+        return sections
+    def _parse_jsonl(self, content: str) -> list[str]:
+        lines = content.strip().splitlines()
+        all_chunks = []
+        for line in lines:
+            if not line.strip():
+                continue
+            try:
+                data = json.loads(line)
+                chunks = self.split_json(data, convert_lists=True)
+                all_chunks.extend(json.dumps(chunk, ensure_ascii=False) for chunk in chunks if chunk)
+            except json.JSONDecodeError:
+                continue
+        return all_chunks
+    def is_jsonl_format(self, txt: str, sample_limit: int = 10, threshold: float = 0.8) -> bool:
+        lines = [line.strip() for line in txt.strip().splitlines() if line.strip()]
+        if not lines:
+            return False
+        try:
+            json.loads(txt)
+            return False
+        except json.JSONDecodeError:
+            pass
+        sample_limit = min(len(lines), sample_limit)
+        sample_lines = lines[:sample_limit]
+        valid_lines = sum(1 for line in sample_lines if self._is_valid_json(line))
+        if not valid_lines:
+            return False
+        return (valid_lines / len(sample_lines)) >= threshold
+    def _is_valid_json(self, line: str) -> bool:
+        try:
+            json.loads(line)
+            return True
+        except json.JSONDecodeError:
+            return False