bibtexparser 1.4.4__tar.gz → 2.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. bibtexparser-2.0.0/LICENSE +21 -0
  2. bibtexparser-2.0.0/MANIFEST.in +4 -0
  3. bibtexparser-2.0.0/PKG-INFO +127 -0
  4. bibtexparser-2.0.0/README.md +96 -0
  5. bibtexparser-2.0.0/bibtexparser/__init__.py +11 -0
  6. bibtexparser-2.0.0/bibtexparser/entrypoint.py +402 -0
  7. bibtexparser-2.0.0/bibtexparser/exceptions.py +61 -0
  8. bibtexparser-2.0.0/bibtexparser/library.py +291 -0
  9. bibtexparser-2.0.0/bibtexparser/middlewares/__init__.py +23 -0
  10. bibtexparser-2.0.0/bibtexparser/middlewares/enclosing.py +297 -0
  11. bibtexparser-2.0.0/bibtexparser/middlewares/fieldkeys.py +51 -0
  12. bibtexparser-2.0.0/bibtexparser/middlewares/interpolate.py +78 -0
  13. bibtexparser-2.0.0/bibtexparser/middlewares/latex_encoding.py +233 -0
  14. bibtexparser-2.0.0/bibtexparser/middlewares/middleware.py +234 -0
  15. bibtexparser-2.0.0/bibtexparser/middlewares/month.py +213 -0
  16. bibtexparser-2.0.0/bibtexparser/middlewares/names.py +659 -0
  17. bibtexparser-2.0.0/bibtexparser/middlewares/parsestack.py +25 -0
  18. bibtexparser-2.0.0/bibtexparser/middlewares/sorting_blocks.py +167 -0
  19. bibtexparser-2.0.0/bibtexparser/middlewares/sorting_entry_fields.py +75 -0
  20. bibtexparser-2.0.0/bibtexparser/model.py +576 -0
  21. bibtexparser-2.0.0/bibtexparser/py.typed +0 -0
  22. bibtexparser-2.0.0/bibtexparser/splitter.py +491 -0
  23. bibtexparser-2.0.0/bibtexparser/writer.py +234 -0
  24. bibtexparser-2.0.0/bibtexparser.egg-info/PKG-INFO +127 -0
  25. bibtexparser-2.0.0/bibtexparser.egg-info/SOURCES.txt +59 -0
  26. bibtexparser-2.0.0/bibtexparser.egg-info/requires.txt +10 -0
  27. bibtexparser-2.0.0/pyproject.toml +48 -0
  28. {bibtexparser-1.4.4 → bibtexparser-2.0.0}/setup.cfg +4 -4
  29. bibtexparser-2.0.0/tests/__init__.py +0 -0
  30. bibtexparser-2.0.0/tests/e2e_example.py +55 -0
  31. bibtexparser-2.0.0/tests/middleware_tests/__init__.py +0 -0
  32. bibtexparser-2.0.0/tests/middleware_tests/middleware_test_util.py +55 -0
  33. bibtexparser-2.0.0/tests/middleware_tests/test_block_middleware.py +147 -0
  34. bibtexparser-2.0.0/tests/middleware_tests/test_custom_middleware_smoke.py +31 -0
  35. bibtexparser-2.0.0/tests/middleware_tests/test_enclosing.py +654 -0
  36. bibtexparser-2.0.0/tests/middleware_tests/test_fieldkeys.py +70 -0
  37. bibtexparser-2.0.0/tests/middleware_tests/test_interpolate.py +68 -0
  38. bibtexparser-2.0.0/tests/middleware_tests/test_latex_encoding.py +255 -0
  39. bibtexparser-2.0.0/tests/middleware_tests/test_month.py +267 -0
  40. bibtexparser-2.0.0/tests/middleware_tests/test_names.py +1179 -0
  41. bibtexparser-2.0.0/tests/middleware_tests/test_sorting_blocks.py +223 -0
  42. bibtexparser-2.0.0/tests/middleware_tests/test_sorting_entry_fields.py +84 -0
  43. bibtexparser-2.0.0/tests/resources/gbk_test.bib +6 -0
  44. bibtexparser-2.0.0/tests/resources.py +50 -0
  45. bibtexparser-2.0.0/tests/splitter_tests/__init__.py +0 -0
  46. bibtexparser-2.0.0/tests/splitter_tests/test_splitter_basic.py +394 -0
  47. bibtexparser-2.0.0/tests/splitter_tests/test_splitter_block_start_detection.py +283 -0
  48. bibtexparser-2.0.0/tests/splitter_tests/test_splitter_entry.py +394 -0
  49. bibtexparser-2.0.0/tests/splitter_tests/test_splitter_explicit_comment.py +26 -0
  50. bibtexparser-2.0.0/tests/splitter_tests/test_splitter_implicit_comments.py +91 -0
  51. bibtexparser-2.0.0/tests/splitter_tests/test_splitter_many_newlines.py +86 -0
  52. bibtexparser-2.0.0/tests/splitter_tests/test_splitter_parenthesis_blocks.py +209 -0
  53. bibtexparser-2.0.0/tests/splitter_tests/test_splitter_parse_failures.py +62 -0
  54. bibtexparser-2.0.0/tests/splitter_tests/test_splitter_preamble.py +25 -0
  55. bibtexparser-2.0.0/tests/splitter_tests/test_splitter_string.py +43 -0
  56. bibtexparser-2.0.0/tests/test_entrypoint.py +615 -0
  57. bibtexparser-2.0.0/tests/test_library.py +254 -0
  58. bibtexparser-2.0.0/tests/test_model.py +547 -0
  59. bibtexparser-2.0.0/tests/test_writer.py +210 -0
  60. bibtexparser-1.4.4/CODE_OF_CONDUCT.md +0 -128
  61. bibtexparser-1.4.4/CONTRIBUTING.md +0 -20
  62. bibtexparser-1.4.4/COPYING +0 -201
  63. bibtexparser-1.4.4/MANIFEST.in +0 -5
  64. bibtexparser-1.4.4/PKG-INFO +0 -66
  65. bibtexparser-1.4.4/README.rst +0 -55
  66. bibtexparser-1.4.4/bibtexparser/__init__.py +0 -108
  67. bibtexparser-1.4.4/bibtexparser/bibdatabase.py +0 -270
  68. bibtexparser-1.4.4/bibtexparser/bibtexexpression.py +0 -286
  69. bibtexparser-1.4.4/bibtexparser/bparser.py +0 -340
  70. bibtexparser-1.4.4/bibtexparser/bwriter.py +0 -229
  71. bibtexparser-1.4.4/bibtexparser/customization.py +0 -657
  72. bibtexparser-1.4.4/bibtexparser/latexenc.py +0 -2693
  73. bibtexparser-1.4.4/bibtexparser.egg-info/PKG-INFO +0 -66
  74. bibtexparser-1.4.4/bibtexparser.egg-info/SOURCES.txt +0 -28
  75. bibtexparser-1.4.4/bibtexparser.egg-info/requires.txt +0 -1
  76. bibtexparser-1.4.4/docs/Makefile +0 -153
  77. bibtexparser-1.4.4/docs/source/bibtex_conv.rst +0 -55
  78. bibtexparser-1.4.4/docs/source/bibtexparser.rst +0 -47
  79. bibtexparser-1.4.4/docs/source/conf.py +0 -247
  80. bibtexparser-1.4.4/docs/source/demo.py +0 -53
  81. bibtexparser-1.4.4/docs/source/index.rst +0 -46
  82. bibtexparser-1.4.4/docs/source/install.rst +0 -78
  83. bibtexparser-1.4.4/docs/source/logging.rst +0 -82
  84. bibtexparser-1.4.4/docs/source/tutorial.rst +0 -390
  85. bibtexparser-1.4.4/requirements.txt +0 -1
  86. bibtexparser-1.4.4/setup.py +0 -33
  87. {bibtexparser-1.4.4 → bibtexparser-2.0.0}/bibtexparser.egg-info/dependency_links.txt +0 -0
  88. {bibtexparser-1.4.4 → bibtexparser-2.0.0}/bibtexparser.egg-info/top_level.txt +0 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2021 Michael Weiss
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,4 @@
1
+ # The sdist ships the full test suite, so that it can be run from source.
2
+ graft tests
3
+
4
+ global-exclude __pycache__ *.py[cod]
@@ -0,0 +1,127 @@
1
+ Metadata-Version: 2.4
2
+ Name: bibtexparser
3
+ Version: 2.0.0
4
+ Summary: Bibtex parser for python 3
5
+ Author-email: Michael Weiss and other contributors <code@mweiss.ch>
6
+ Maintainer-email: Michael Weiss <code@mweiss.ch>
7
+ License-Expression: MIT
8
+ Project-URL: Homepage, https://github.com/sciunto-org/python-bibtexparser
9
+ Project-URL: Repository, https://github.com/sciunto-org/python-bibtexparser
10
+ Classifier: Development Status :: 5 - Production/Stable
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.10
13
+ Classifier: Programming Language :: Python :: 3.11
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Classifier: Programming Language :: Python :: 3.13
16
+ Classifier: Programming Language :: Python :: 3.14
17
+ Classifier: Topic :: Text Processing :: Markup :: LaTeX
18
+ Classifier: Operating System :: OS Independent
19
+ Requires-Python: >=3.10
20
+ Description-Content-Type: text/markdown
21
+ License-File: LICENSE
22
+ Requires-Dist: pylatexenc~=2.10
23
+ Provides-Extra: docs
24
+ Requires-Dist: sphinx; extra == "docs"
25
+ Provides-Extra: test
26
+ Requires-Dist: pytest; extra == "test"
27
+ Requires-Dist: pytest-xdist; extra == "test"
28
+ Requires-Dist: pytest-cov; extra == "test"
29
+ Requires-Dist: jupyter; extra == "test"
30
+ Dynamic: license-file
31
+
32
+ # python-bibtexparser v2
33
+
34
+ Welcome to python-bibtexparser, a parser for `.bib` files with a long history and wide adoption.
35
+
36
+ Bibtexparser is available in two versions: V1 and V2. **V2 is the current, recommended version** and the default you get from PyPI. It provides an overall more robust and faster experience than v1, and is where all development and maintenance effort goes. Install it using pip:
37
+
38
+ ```bash
39
+ pip install bibtexparser
40
+ ```
41
+
42
+ Or you can install the latest development version directly from the main branch:
43
+ ```bash
44
+ pip install --no-cache-dir --force-reinstall git+https://github.com/sciunto-org/python-bibtexparser@main
45
+ ```
46
+
47
+ If instead you still need v1, e.g. for a legacy project which you don't want to migrate right now, pin it explicitly:
48
+
49
+ ```bash
50
+ pip install bibtexparser~=1.0
51
+ ```
52
+
53
+ Note that v2 is a rewrite: a lot has changed since v1, including the primary entrypoints, the data structures and the way parsing and writing are customized.
54
+ Existing v1 code will not run unchanged on v2 - have a look at our [migration guide](https://bibtexparser.readthedocs.io/en/main/migrate.html) when upgrading. While v2 has been thoroughly tested and has been in pre-release for a long time, please don't hesitate to report any issues you may encounter.
55
+
56
+ V1 is in maintenance mode: small PRs are still accepted, but only as long as they are backwards compatible and don't introduce much additional technical debt.
57
+ Development of version one happens on the dedicated [v1 branch](https://github.com/sciunto-org/python-bibtexparser/tree/v1).
58
+
59
+ ## Documentation
60
+ Go check out our documentation on [https://bibtexparser.readthedocs.io/en/main/](https://bibtexparser.readthedocs.io/en/main/).
61
+
62
+ ## Advantages of `v2` compared to `v1`
63
+
64
+ - :rocket: Order of magnitudes faster
65
+ - :wrench: Easily customizable parsing **and** writing
66
+ - :herb: Access to raw, unparsed bibtex.
67
+ - :shield: Fault-Tolerant: Able to parse files with syntax errors
68
+ - :mahjong: Massively simplified, more robust handling of de- and encoding (special chars, ...).
69
+ - :copyright: Permissive MIT license
70
+
71
+ ## TLDR Usage Example
72
+
73
+ ```python
74
+ # Parsing a bibtex string with default values
75
+ bib_database = bibtexparser.parse_string(bibtex_string)
76
+ # Converting it back to a bibtex string, again with default values
77
+ new_bibtex_string = bibtexparser.write_string(bib_database)
78
+ ```
79
+
80
+ Slightly more involved example:
81
+
82
+ ```python
83
+
84
+ # Lets parse some bibtex string.
85
+ bib_database = bibtexparser.parse_string(bibtex_string,
86
+ # Middleware layers to transform parsed entries.
87
+ # Here, we split multiple authors from each other and then extract first name, last name, ... for each
88
+ append_middleware=[SeparateCoAuthors(), SplitNameParts()],
89
+ )
90
+
91
+ # Here you have a `bib_database` with all parsed bibtex blocks.
92
+
93
+ # Let's transform it back to a bibtex_string.
94
+ new_bibtex_string = bibtexparser.write_string(bib_database,
95
+ # Revert above transformation
96
+ prepend_middleware=[MergeNameParts(), MergeCoAuthors()]
97
+ )
98
+ ```
99
+
100
+ These examples really only show the bare minimum.
101
+ Consult the documentation for a list of available middleware, parsing options and write-formatting options.
102
+
103
+ ## Architecture and Terminology
104
+
105
+ ![bibtexparser](https://user-images.githubusercontent.com/4815944/193734283-f19f94e8-7986-4acf-b1a3-1d215e297224.png)
106
+
107
+ The architecture consists of the following components:
108
+
109
+ #### Library
110
+ Reflects the contents of a parsed bibtex files, including all comments, entries, strings, preambles and their metadata (e.g. order).
111
+
112
+ #### A Splitter
113
+ Splits a bibtex string into basic blocks (Entry, String, Preamble, ...), with correspondingly split content (e.g. fields on Entry, key-value on String, ...).
114
+ The splitter aims to be forgiving when facing invalid bibtex: A line starting with a block definition (``@....``) ends the previous block, even if not yet every bracket is closed, failing the parsing of the previous block. Correspondingly, one block type is "ParsingFailedBlock".
115
+
116
+ #### Middleware
117
+ Middleware layers transform a library and its blocks, for example by decoding latex special characters, interpolating string references, resolving crossreferences or re-ordering blocks. Thus, the choice of middleware allows to customize parsing and writing to ones specific usecase. Note: Middlewares, by default, do not mutate their input, but return a modified copy.
118
+
119
+ #### Writer
120
+ Writes the content of a bibtex library to a ``.bib`` file. Optional formatting parameters can be passed using a corresponding dedicated data structure.
121
+
122
+ ## About
123
+
124
+ Since 2022, `bibtexparser` is primarily written and maintained by Michael Weiss ([@MiWeiss](https://github.com/MiWeiss/)), supported by various awesome contributors.
125
+
126
+ Credits and thanks to the many contributors who helped creating this library, including
127
+ François Boulogne ([@sciunto](https://github.com/sciunto/), creator of the first version) and Olivier Mangin ([@omangin](https://github.com/omangin/), long-term contributor).
@@ -0,0 +1,96 @@
1
+ # python-bibtexparser v2
2
+
3
+ Welcome to python-bibtexparser, a parser for `.bib` files with a long history and wide adoption.
4
+
5
+ Bibtexparser is available in two versions: V1 and V2. **V2 is the current, recommended version** and the default you get from PyPI. It provides an overall more robust and faster experience than v1, and is where all development and maintenance effort goes. Install it using pip:
6
+
7
+ ```bash
8
+ pip install bibtexparser
9
+ ```
10
+
11
+ Or you can install the latest development version directly from the main branch:
12
+ ```bash
13
+ pip install --no-cache-dir --force-reinstall git+https://github.com/sciunto-org/python-bibtexparser@main
14
+ ```
15
+
16
+ If instead you still need v1, e.g. for a legacy project which you don't want to migrate right now, pin it explicitly:
17
+
18
+ ```bash
19
+ pip install bibtexparser~=1.0
20
+ ```
21
+
22
+ Note that v2 is a rewrite: a lot has changed since v1, including the primary entrypoints, the data structures and the way parsing and writing are customized.
23
+ Existing v1 code will not run unchanged on v2 - have a look at our [migration guide](https://bibtexparser.readthedocs.io/en/main/migrate.html) when upgrading. While v2 has been thoroughly tested and has been in pre-release for a long time, please don't hesitate to report any issues you may encounter.
24
+
25
+ V1 is in maintenance mode: small PRs are still accepted, but only as long as they are backwards compatible and don't introduce much additional technical debt.
26
+ Development of version one happens on the dedicated [v1 branch](https://github.com/sciunto-org/python-bibtexparser/tree/v1).
27
+
28
+ ## Documentation
29
+ Go check out our documentation on [https://bibtexparser.readthedocs.io/en/main/](https://bibtexparser.readthedocs.io/en/main/).
30
+
31
+ ## Advantages of `v2` compared to `v1`
32
+
33
+ - :rocket: Order of magnitudes faster
34
+ - :wrench: Easily customizable parsing **and** writing
35
+ - :herb: Access to raw, unparsed bibtex.
36
+ - :shield: Fault-Tolerant: Able to parse files with syntax errors
37
+ - :mahjong: Massively simplified, more robust handling of de- and encoding (special chars, ...).
38
+ - :copyright: Permissive MIT license
39
+
40
+ ## TLDR Usage Example
41
+
42
+ ```python
43
+ # Parsing a bibtex string with default values
44
+ bib_database = bibtexparser.parse_string(bibtex_string)
45
+ # Converting it back to a bibtex string, again with default values
46
+ new_bibtex_string = bibtexparser.write_string(bib_database)
47
+ ```
48
+
49
+ Slightly more involved example:
50
+
51
+ ```python
52
+
53
+ # Lets parse some bibtex string.
54
+ bib_database = bibtexparser.parse_string(bibtex_string,
55
+ # Middleware layers to transform parsed entries.
56
+ # Here, we split multiple authors from each other and then extract first name, last name, ... for each
57
+ append_middleware=[SeparateCoAuthors(), SplitNameParts()],
58
+ )
59
+
60
+ # Here you have a `bib_database` with all parsed bibtex blocks.
61
+
62
+ # Let's transform it back to a bibtex_string.
63
+ new_bibtex_string = bibtexparser.write_string(bib_database,
64
+ # Revert above transformation
65
+ prepend_middleware=[MergeNameParts(), MergeCoAuthors()]
66
+ )
67
+ ```
68
+
69
+ These examples really only show the bare minimum.
70
+ Consult the documentation for a list of available middleware, parsing options and write-formatting options.
71
+
72
+ ## Architecture and Terminology
73
+
74
+ ![bibtexparser](https://user-images.githubusercontent.com/4815944/193734283-f19f94e8-7986-4acf-b1a3-1d215e297224.png)
75
+
76
+ The architecture consists of the following components:
77
+
78
+ #### Library
79
+ Reflects the contents of a parsed bibtex files, including all comments, entries, strings, preambles and their metadata (e.g. order).
80
+
81
+ #### A Splitter
82
+ Splits a bibtex string into basic blocks (Entry, String, Preamble, ...), with correspondingly split content (e.g. fields on Entry, key-value on String, ...).
83
+ The splitter aims to be forgiving when facing invalid bibtex: A line starting with a block definition (``@....``) ends the previous block, even if not yet every bracket is closed, failing the parsing of the previous block. Correspondingly, one block type is "ParsingFailedBlock".
84
+
85
+ #### Middleware
86
+ Middleware layers transform a library and its blocks, for example by decoding latex special characters, interpolating string references, resolving crossreferences or re-ordering blocks. Thus, the choice of middleware allows to customize parsing and writing to ones specific usecase. Note: Middlewares, by default, do not mutate their input, but return a modified copy.
87
+
88
+ #### Writer
89
+ Writes the content of a bibtex library to a ``.bib`` file. Optional formatting parameters can be passed using a corresponding dedicated data structure.
90
+
91
+ ## About
92
+
93
+ Since 2022, `bibtexparser` is primarily written and maintained by Michael Weiss ([@MiWeiss](https://github.com/MiWeiss/)), supported by various awesome contributors.
94
+
95
+ Credits and thanks to the many contributors who helped creating this library, including
96
+ François Boulogne ([@sciunto](https://github.com/sciunto/), creator of the first version) and Olivier Mangin ([@omangin](https://github.com/omangin/), long-term contributor).
@@ -0,0 +1,11 @@
1
+ import bibtexparser.exceptions
2
+ import bibtexparser.middlewares
3
+ import bibtexparser.model
4
+ from bibtexparser.entrypoint import parse_file
5
+ from bibtexparser.entrypoint import parse_string
6
+ from bibtexparser.entrypoint import write_file
7
+ from bibtexparser.entrypoint import write_string
8
+ from bibtexparser.library import Library
9
+ from bibtexparser.writer import BibtexFormat
10
+
11
+ __version__ = "2.0.0"
@@ -0,0 +1,402 @@
1
+ import codecs
2
+ import logging
3
+ import warnings
4
+ from collections.abc import Iterable
5
+ from copy import deepcopy
6
+ from typing import Optional
7
+ from typing import TextIO
8
+
9
+ from .library import Library
10
+ from .middlewares.enclosing import REMOVED_ENCLOSING_KEY
11
+ from .middlewares.middleware import Middleware
12
+ from .middlewares.parsestack import default_parse_stack
13
+ from .middlewares.parsestack import default_unparse_stack
14
+ from .model import Block
15
+ from .model import String
16
+ from .splitter import Splitter
17
+ from .writer import BibtexFormat
18
+ from .writer import write
19
+
20
+ logger = logging.getLogger(__name__)
21
+
22
+ #: Marks a seeded copy of a pre-existing `@string`, dropped before merging back.
23
+ _PREEXISTING_STRING_KEY = "bibtexparser_preexisting_string"
24
+
25
+ #: Number of blocks from which on `write_string`/`write_file` warn if the unparse
26
+ #: stack deep-copies blocks. Copying costs roughly 30-60 µs per entry, i.e. it
27
+ #: starts to dominate the write time at this size.
28
+ LARGE_LIBRARY_WARNING_THRESHOLD = 10_000
29
+
30
+
31
+ def _build_parse_stack(
32
+ parse_stack: Iterable[Middleware] | None,
33
+ append_middleware: Iterable[Middleware] | None,
34
+ ) -> list[Middleware]:
35
+ # Materialize upfront: the arguments may be one-shot iterators.
36
+ parse_stack = None if parse_stack is None else list(parse_stack)
37
+ append_middleware = None if append_middleware is None else list(append_middleware)
38
+
39
+ if parse_stack is not None and append_middleware is not None:
40
+ raise ValueError(
41
+ "Provided both parse_stack and append_middleware. "
42
+ "Only one should be provided. "
43
+ "(append_middleware should only be used with the default parse_stack, "
44
+ "i.e., when the passed parse_stack is None.)"
45
+ )
46
+
47
+ if parse_stack is None:
48
+ parse_stack = default_parse_stack(allow_inplace_modification=True)
49
+
50
+ if append_middleware is None:
51
+ return list(parse_stack)
52
+
53
+ parse_stack_types = {type(m) for m in parse_stack}
54
+ append_stack_types = {type(m) for m in append_middleware}
55
+ stack_types_intersect = parse_stack_types.intersection(append_stack_types)
56
+ if len(stack_types_intersect) > 0:
57
+ warnings.warn(
58
+ "Some middleware passed in append_middleware are "
59
+ f"already in the default parse_stack ({stack_types_intersect})."
60
+ )
61
+
62
+ return list(parse_stack) + list(append_middleware)
63
+
64
+
65
+ def _build_unparse_stack(
66
+ unparse_stack: Iterable[Middleware] | None,
67
+ prepend_middleware: Iterable[Middleware] | None,
68
+ ) -> list[Middleware]:
69
+ # Materialize upfront: the arguments may be one-shot iterators.
70
+ unparse_stack = None if unparse_stack is None else list(unparse_stack)
71
+ prepend_middleware = None if prepend_middleware is None else list(prepend_middleware)
72
+
73
+ if unparse_stack is not None and prepend_middleware is not None:
74
+ raise ValueError(
75
+ "Provided both unparse_stack and prepend_middleware. "
76
+ "Only one should be provided. "
77
+ "(prepend_middleware should only be used with the default unparse_stack, "
78
+ "i.e., when the passed unparse_stack is None.)"
79
+ )
80
+
81
+ if unparse_stack is None:
82
+ unparse_stack = default_unparse_stack(allow_inplace_modification=False)
83
+
84
+ if prepend_middleware is None:
85
+ return list(unparse_stack)
86
+
87
+ parse_stack_types = {type(m) for m in unparse_stack}
88
+ append_stack_types = {type(m) for m in prepend_middleware}
89
+ stack_types_intersect = parse_stack_types.intersection(append_stack_types)
90
+ if len(stack_types_intersect) > 0:
91
+ warnings.warn(
92
+ "Some middleware passed in prepend_middleware are "
93
+ f"already in the default unparse_stack ({stack_types_intersect})."
94
+ )
95
+
96
+ return list(prepend_middleware) + list(unparse_stack)
97
+
98
+
99
+ def _warn_if_large_library_is_copied(library: Library, unparse_stack: list[Middleware]) -> None:
100
+ """Warn if writing ``library`` will deep-copy its blocks and that is likely slow.
101
+
102
+ Middlewares with ``allow_inplace_modification=False`` (the default unparse stack
103
+ is built that way) deep-copy every block they transform, which dominates the
104
+ write time of large libraries.
105
+ """
106
+ n_blocks = len(library.blocks)
107
+ if n_blocks < LARGE_LIBRARY_WARNING_THRESHOLD:
108
+ return
109
+ if all(middleware.allow_inplace_modification for middleware in unparse_stack):
110
+ return
111
+ logger.warning(
112
+ f"Writing a library with {n_blocks} blocks: the unparse stack deep-copies blocks "
113
+ "(it contains middlewares with allow_inplace_modification=False), "
114
+ "which is slow for large libraries. "
115
+ "If you do not need the library after writing, pass an unparse stack whose "
116
+ "middlewares all allow in-place modification, e.g. "
117
+ "`unparse_stack=bibtexparser.middlewares.default_unparse_stack("
118
+ "allow_inplace_modification=True)`."
119
+ )
120
+
121
+
122
+ def _handle_deprecated_write_params(
123
+ unparse_stack: Iterable[Middleware] | None,
124
+ prepend_middleware: Iterable[Middleware] | None,
125
+ kwargs: dict,
126
+ function_name: str,
127
+ ) -> tuple[Iterable[Middleware] | None, Iterable[Middleware] | None]:
128
+ """Handle deprecated parameter names for write functions.
129
+
130
+ :param unparse_stack: Current unparse_stack value
131
+ :param prepend_middleware: Current prepend_middleware value
132
+ :param kwargs: Dictionary of keyword arguments to check for deprecated params
133
+ :param function_name: Name of the calling function (for error messages)
134
+ :return: Tuple of (unparse_stack, prepend_middleware) with deprecated values migrated
135
+ """
136
+ if "parse_stack" in kwargs:
137
+ warnings.warn(
138
+ "Parameter 'parse_stack' is deprecated. Use 'unparse_stack' instead.",
139
+ DeprecationWarning,
140
+ stacklevel=3,
141
+ )
142
+ if unparse_stack is not None:
143
+ raise ValueError(
144
+ "Cannot provide both 'parse_stack' (deprecated) and 'unparse_stack'. "
145
+ "Use 'unparse_stack' instead."
146
+ )
147
+ unparse_stack = kwargs.pop("parse_stack")
148
+
149
+ if "append_middleware" in kwargs:
150
+ warnings.warn(
151
+ "Parameter 'append_middleware' is deprecated. Use 'prepend_middleware' instead.",
152
+ DeprecationWarning,
153
+ stacklevel=3,
154
+ )
155
+ if prepend_middleware is not None:
156
+ raise ValueError(
157
+ "Cannot provide both 'append_middleware' (deprecated) and 'prepend_middleware'. "
158
+ "Use 'prepend_middleware' instead."
159
+ )
160
+ prepend_middleware = kwargs.pop("append_middleware")
161
+
162
+ if kwargs:
163
+ raise TypeError(f"{function_name}() got unexpected keyword arguments: {', '.join(kwargs)}")
164
+
165
+ return unparse_stack, prepend_middleware
166
+
167
+
168
+ def parse_string(
169
+ bibtex_str: str,
170
+ parse_stack: Iterable[Middleware] | None = None,
171
+ append_middleware: Iterable[Middleware] | None = None,
172
+ library: Library | None = None,
173
+ ) -> Library:
174
+ """Parse a BibTeX string.
175
+
176
+ :param bibtex_str: BibTeX string to parse
177
+ :param parse_stack:
178
+ List of middleware to apply to the database after splitting.
179
+ If ``None`` (default), a default stack will be used providing simple standard functionality.
180
+
181
+ :param append_middleware:
182
+ List of middleware to append to the default stack
183
+ (ignored if a not-``None`` parse_stack is passed).
184
+
185
+ :param library:
186
+ Library to add the newly parsed blocks to.
187
+ If ``None`` (default), a new library is created and returned.
188
+ If a library is passed, it is returned (mutated) and:
189
+
190
+ - the parse stack is applied **only** to the newly parsed blocks;
191
+ blocks already contained in the passed library are left untouched
192
+ (they were already transformed when they were parsed);
193
+ - ``@string`` blocks already contained in the passed library are visible
194
+ to the parse stack, i.e. string references in ``bibtex_str`` resolve
195
+ against them (unless ``bibtex_str`` redefines the same key);
196
+ - keys defined both in the passed library and in ``bibtex_str``
197
+ do not raise, but yield ``DuplicateBlockKeyBlock`` instances
198
+ (see ``library.failed_blocks``), just like duplicates within a single string.
199
+
200
+ :return: Library: Parsed BibTeX database
201
+ """
202
+ splitter = Splitter(bibstr=bibtex_str)
203
+ parsed = splitter.split()
204
+
205
+ _seed_preexisting_strings(parsed, library)
206
+
207
+ middleware: Middleware
208
+ for middleware in _build_parse_stack(parse_stack, append_middleware):
209
+ parsed = middleware.transform(library=parsed)
210
+
211
+ if library is None:
212
+ return parsed
213
+
214
+ new_blocks = [b for b in parsed.blocks if not _is_seeded_string(b)]
215
+ library.add(new_blocks, fail_on_duplicate_key=False)
216
+ return library
217
+
218
+
219
+ def _is_seeded_string(block: Block) -> bool:
220
+ """True for blocks seeded by `_seed_preexisting_strings` (and their transformations)."""
221
+ return bool(block.get_parser_metadata(_PREEXISTING_STRING_KEY))
222
+
223
+
224
+ def _restore_enclosing(string: String) -> None:
225
+ """Make sure the value of an already-parsed string is enclosed again.
226
+
227
+ The parse stack expects freshly split (i.e. still enclosed) values.
228
+ Feeding it an already-stripped value would make that value be treated
229
+ as an unenclosed literal (a string reference), which does not round-trip
230
+ to valid bibtex.
231
+ """
232
+ enclosing = string.parser_metadata.pop(REMOVED_ENCLOSING_KEY, None)
233
+ if string.enclosing == "no-enclosing" or enclosing == "no-enclosing":
234
+ return
235
+ value = string.value
236
+ if not isinstance(value, str):
237
+ return
238
+ if enclosing is None and (
239
+ (value.startswith("{") and value.endswith("}"))
240
+ or (value.startswith('"') and value.endswith('"'))
241
+ ):
242
+ return
243
+ string.value = f'"{value}"' if enclosing == '"' else f"{{{value}}}"
244
+
245
+
246
+ def _seed_preexisting_strings(parsed: Library, library: Library | None) -> list[String]:
247
+ """Make the ``@string`` blocks of an existing library visible to the parse stack.
248
+
249
+ Copies (never the originals, which must not be transformed again) of the
250
+ strings of ``library`` are added to ``parsed``, unless the newly parsed
251
+ content redefines the same key. The copies are tagged so that they can be
252
+ dropped again before merging the parsed blocks back into ``library``.
253
+
254
+ :param parsed: The freshly split library, modified in place.
255
+ :param library: The pre-existing library, or ``None``.
256
+ :return: The seeded (tagged) string copies.
257
+ """
258
+ if library is None:
259
+ return []
260
+
261
+ # Bibtex string keys are case-insensitive, hence compare in lower case.
262
+ redefined = {key.lower() for key in parsed.strings_dict}
263
+ seeds = []
264
+ for key, string in library.strings_dict.items():
265
+ if key.lower() in redefined:
266
+ continue
267
+ seed = deepcopy(string)
268
+ seed.set_parser_metadata(_PREEXISTING_STRING_KEY, True)
269
+ _restore_enclosing(seed)
270
+ seeds.append(seed)
271
+
272
+ if seeds:
273
+ parsed.add(seeds, fail_on_duplicate_key=False)
274
+ return seeds
275
+
276
+
277
+ def parse_file(
278
+ path: str,
279
+ parse_stack: Iterable[Middleware] | None = None,
280
+ append_middleware: Iterable[Middleware] | None = None,
281
+ encoding: str = "UTF-8",
282
+ ) -> Library:
283
+ """Parse a BibTeX file
284
+
285
+ :param path: Path to BibTeX file
286
+ :param parse_stack:
287
+ List of middleware to apply to the database after splitting.
288
+ If ``None`` (default), a default stack will be used providing simple standard functionality.
289
+
290
+ :param append_middleware:
291
+ List of middleware to append to the default stack
292
+ (ignored if a not-``None`` parse_stack is passed).
293
+
294
+ :param encoding: Encoding of the .bib file. Default encoding is ``"UTF-8"``.
295
+ :return: Library: Parsed BibTeX library
296
+ :raises LookupError: If the specified encoding is not recognized.
297
+ """
298
+ try:
299
+ codecs.lookup(encoding)
300
+ except LookupError:
301
+ raise LookupError(f"Unknown encoding: {encoding!r}")
302
+
303
+ with open(path, encoding=encoding) as f:
304
+ bibtex_str = f.read()
305
+ return parse_string(
306
+ bibtex_str, parse_stack=parse_stack, append_middleware=append_middleware
307
+ )
308
+
309
+
310
+ def write_file(
311
+ file: str | TextIO,
312
+ library: Library,
313
+ unparse_stack: Iterable[Middleware] | None = None,
314
+ prepend_middleware: Iterable[Middleware] | None = None,
315
+ bibtex_format: BibtexFormat | None = None,
316
+ encoding: str = "UTF-8",
317
+ **kwargs,
318
+ ) -> None:
319
+ """Write a BibTeX database to a file.
320
+
321
+ The passed library is never modified, unless *every* middleware in the
322
+ unparse stack allows in-place modification (e.g.
323
+ ``unparse_stack=default_unparse_stack(allow_inplace_modification=True)``).
324
+
325
+ :param file: File to write to. Can be a file name or a file object.
326
+ :param library: BibTeX database to serialize.
327
+ :param unparse_stack: List of middleware to apply to the database before writing.
328
+ If None, a default stack will be used.
329
+ :param prepend_middleware: List of middleware to prepend to the default stack.
330
+ Only applicable if `unparse_stack` is None.
331
+ :param bibtex_format: Customized BibTeX format to use (optional).
332
+ :param encoding: Encoding of the .bib file. Default encoding is ``"UTF-8"``.
333
+ Writing a library with at least ``LARGE_LIBRARY_WARNING_THRESHOLD`` blocks logs a warning
334
+ if the unparse stack deep-copies blocks (middlewares with ``allow_inplace_modification=False``),
335
+ as that is slow; pass an all-in-place stack to avoid it.
336
+
337
+ .. deprecated:: (next version)
338
+ Parameters 'parse_stack' and 'append_middleware' are deprecated, will be deleted soon.
339
+ Use 'unparse_stack' and 'prepend_middleware' instead.
340
+ """
341
+ unparse_stack, prepend_middleware = _handle_deprecated_write_params(
342
+ unparse_stack, prepend_middleware, kwargs, "write_file"
343
+ )
344
+
345
+ bibtex_str = write_string(
346
+ library=library,
347
+ unparse_stack=unparse_stack,
348
+ prepend_middleware=prepend_middleware,
349
+ bibtex_format=bibtex_format,
350
+ )
351
+ if isinstance(file, str):
352
+ with open(file, "w", encoding=encoding) as f:
353
+ f.write(bibtex_str)
354
+ else:
355
+ file.write(bibtex_str)
356
+
357
+
358
+ def write_string(
359
+ library: Library,
360
+ unparse_stack: Iterable[Middleware] | None = None,
361
+ prepend_middleware: Iterable[Middleware] | None = None,
362
+ bibtex_format: Optional["BibtexFormat"] = None,
363
+ **kwargs,
364
+ ) -> str:
365
+ """Serialize a BibTeX database to a string.
366
+
367
+ The passed library is never modified, unless *every* middleware in the
368
+ unparse stack allows in-place modification (e.g.
369
+ ``unparse_stack=default_unparse_stack(allow_inplace_modification=True)``).
370
+
371
+ :param library: BibTeX database to serialize.
372
+ :param unparse_stack: List of middleware to apply to the database before writing.
373
+ If None, a default stack will be used.
374
+ :param prepend_middleware: List of middleware to prepend to the default stack.
375
+ Only applicable if `unparse_stack` is None.
376
+ :param bibtex_format: Customized BibTeX format to use (optional).
377
+ Writing a library with at least ``LARGE_LIBRARY_WARNING_THRESHOLD`` blocks logs a warning
378
+ if the unparse stack deep-copies blocks (middlewares with ``allow_inplace_modification=False``),
379
+ as that is slow; pass an all-in-place stack to avoid it.
380
+
381
+ .. deprecated:: (next version)
382
+ Parameters 'parse_stack' and 'append_middleware' are deprecated.
383
+ Use 'unparse_stack' and 'prepend_middleware' instead.
384
+ """
385
+ unparse_stack, prepend_middleware = _handle_deprecated_write_params(
386
+ unparse_stack, prepend_middleware, kwargs, "write_string"
387
+ )
388
+
389
+ stack = _build_unparse_stack(unparse_stack, prepend_middleware)
390
+ _warn_if_large_library_is_copied(library, stack)
391
+ inplace = [middleware.allow_inplace_modification for middleware in stack]
392
+ if any(inplace) and not all(inplace):
393
+ # Some middleware would mutate the passed library before a copying
394
+ # middleware gets to run; copy once upfront so the caller's library
395
+ # stays untouched (an all-in-place stack is the caller's explicit opt-in).
396
+ library = deepcopy(library)
397
+
398
+ middleware: Middleware
399
+ for middleware in stack:
400
+ library = middleware.transform(library=library)
401
+
402
+ return write(library, bibtex_format=bibtex_format)