bibtexparser 1.4.4__tar.gz → 2.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bibtexparser-2.0.0/LICENSE +21 -0
- bibtexparser-2.0.0/MANIFEST.in +4 -0
- bibtexparser-2.0.0/PKG-INFO +127 -0
- bibtexparser-2.0.0/README.md +96 -0
- bibtexparser-2.0.0/bibtexparser/__init__.py +11 -0
- bibtexparser-2.0.0/bibtexparser/entrypoint.py +402 -0
- bibtexparser-2.0.0/bibtexparser/exceptions.py +61 -0
- bibtexparser-2.0.0/bibtexparser/library.py +291 -0
- bibtexparser-2.0.0/bibtexparser/middlewares/__init__.py +23 -0
- bibtexparser-2.0.0/bibtexparser/middlewares/enclosing.py +297 -0
- bibtexparser-2.0.0/bibtexparser/middlewares/fieldkeys.py +51 -0
- bibtexparser-2.0.0/bibtexparser/middlewares/interpolate.py +78 -0
- bibtexparser-2.0.0/bibtexparser/middlewares/latex_encoding.py +233 -0
- bibtexparser-2.0.0/bibtexparser/middlewares/middleware.py +234 -0
- bibtexparser-2.0.0/bibtexparser/middlewares/month.py +213 -0
- bibtexparser-2.0.0/bibtexparser/middlewares/names.py +659 -0
- bibtexparser-2.0.0/bibtexparser/middlewares/parsestack.py +25 -0
- bibtexparser-2.0.0/bibtexparser/middlewares/sorting_blocks.py +167 -0
- bibtexparser-2.0.0/bibtexparser/middlewares/sorting_entry_fields.py +75 -0
- bibtexparser-2.0.0/bibtexparser/model.py +576 -0
- bibtexparser-2.0.0/bibtexparser/py.typed +0 -0
- bibtexparser-2.0.0/bibtexparser/splitter.py +491 -0
- bibtexparser-2.0.0/bibtexparser/writer.py +234 -0
- bibtexparser-2.0.0/bibtexparser.egg-info/PKG-INFO +127 -0
- bibtexparser-2.0.0/bibtexparser.egg-info/SOURCES.txt +59 -0
- bibtexparser-2.0.0/bibtexparser.egg-info/requires.txt +10 -0
- bibtexparser-2.0.0/pyproject.toml +48 -0
- {bibtexparser-1.4.4 → bibtexparser-2.0.0}/setup.cfg +4 -4
- bibtexparser-2.0.0/tests/__init__.py +0 -0
- bibtexparser-2.0.0/tests/e2e_example.py +55 -0
- bibtexparser-2.0.0/tests/middleware_tests/__init__.py +0 -0
- bibtexparser-2.0.0/tests/middleware_tests/middleware_test_util.py +55 -0
- bibtexparser-2.0.0/tests/middleware_tests/test_block_middleware.py +147 -0
- bibtexparser-2.0.0/tests/middleware_tests/test_custom_middleware_smoke.py +31 -0
- bibtexparser-2.0.0/tests/middleware_tests/test_enclosing.py +654 -0
- bibtexparser-2.0.0/tests/middleware_tests/test_fieldkeys.py +70 -0
- bibtexparser-2.0.0/tests/middleware_tests/test_interpolate.py +68 -0
- bibtexparser-2.0.0/tests/middleware_tests/test_latex_encoding.py +255 -0
- bibtexparser-2.0.0/tests/middleware_tests/test_month.py +267 -0
- bibtexparser-2.0.0/tests/middleware_tests/test_names.py +1179 -0
- bibtexparser-2.0.0/tests/middleware_tests/test_sorting_blocks.py +223 -0
- bibtexparser-2.0.0/tests/middleware_tests/test_sorting_entry_fields.py +84 -0
- bibtexparser-2.0.0/tests/resources/gbk_test.bib +6 -0
- bibtexparser-2.0.0/tests/resources.py +50 -0
- bibtexparser-2.0.0/tests/splitter_tests/__init__.py +0 -0
- bibtexparser-2.0.0/tests/splitter_tests/test_splitter_basic.py +394 -0
- bibtexparser-2.0.0/tests/splitter_tests/test_splitter_block_start_detection.py +283 -0
- bibtexparser-2.0.0/tests/splitter_tests/test_splitter_entry.py +394 -0
- bibtexparser-2.0.0/tests/splitter_tests/test_splitter_explicit_comment.py +26 -0
- bibtexparser-2.0.0/tests/splitter_tests/test_splitter_implicit_comments.py +91 -0
- bibtexparser-2.0.0/tests/splitter_tests/test_splitter_many_newlines.py +86 -0
- bibtexparser-2.0.0/tests/splitter_tests/test_splitter_parenthesis_blocks.py +209 -0
- bibtexparser-2.0.0/tests/splitter_tests/test_splitter_parse_failures.py +62 -0
- bibtexparser-2.0.0/tests/splitter_tests/test_splitter_preamble.py +25 -0
- bibtexparser-2.0.0/tests/splitter_tests/test_splitter_string.py +43 -0
- bibtexparser-2.0.0/tests/test_entrypoint.py +615 -0
- bibtexparser-2.0.0/tests/test_library.py +254 -0
- bibtexparser-2.0.0/tests/test_model.py +547 -0
- bibtexparser-2.0.0/tests/test_writer.py +210 -0
- bibtexparser-1.4.4/CODE_OF_CONDUCT.md +0 -128
- bibtexparser-1.4.4/CONTRIBUTING.md +0 -20
- bibtexparser-1.4.4/COPYING +0 -201
- bibtexparser-1.4.4/MANIFEST.in +0 -5
- bibtexparser-1.4.4/PKG-INFO +0 -66
- bibtexparser-1.4.4/README.rst +0 -55
- bibtexparser-1.4.4/bibtexparser/__init__.py +0 -108
- bibtexparser-1.4.4/bibtexparser/bibdatabase.py +0 -270
- bibtexparser-1.4.4/bibtexparser/bibtexexpression.py +0 -286
- bibtexparser-1.4.4/bibtexparser/bparser.py +0 -340
- bibtexparser-1.4.4/bibtexparser/bwriter.py +0 -229
- bibtexparser-1.4.4/bibtexparser/customization.py +0 -657
- bibtexparser-1.4.4/bibtexparser/latexenc.py +0 -2693
- bibtexparser-1.4.4/bibtexparser.egg-info/PKG-INFO +0 -66
- bibtexparser-1.4.4/bibtexparser.egg-info/SOURCES.txt +0 -28
- bibtexparser-1.4.4/bibtexparser.egg-info/requires.txt +0 -1
- bibtexparser-1.4.4/docs/Makefile +0 -153
- bibtexparser-1.4.4/docs/source/bibtex_conv.rst +0 -55
- bibtexparser-1.4.4/docs/source/bibtexparser.rst +0 -47
- bibtexparser-1.4.4/docs/source/conf.py +0 -247
- bibtexparser-1.4.4/docs/source/demo.py +0 -53
- bibtexparser-1.4.4/docs/source/index.rst +0 -46
- bibtexparser-1.4.4/docs/source/install.rst +0 -78
- bibtexparser-1.4.4/docs/source/logging.rst +0 -82
- bibtexparser-1.4.4/docs/source/tutorial.rst +0 -390
- bibtexparser-1.4.4/requirements.txt +0 -1
- bibtexparser-1.4.4/setup.py +0 -33
- {bibtexparser-1.4.4 → bibtexparser-2.0.0}/bibtexparser.egg-info/dependency_links.txt +0 -0
- {bibtexparser-1.4.4 → bibtexparser-2.0.0}/bibtexparser.egg-info/top_level.txt +0 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2021 Michael Weiss
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: bibtexparser
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: Bibtex parser for python 3
|
|
5
|
+
Author-email: Michael Weiss and other contributors <code@mweiss.ch>
|
|
6
|
+
Maintainer-email: Michael Weiss <code@mweiss.ch>
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
Project-URL: Homepage, https://github.com/sciunto-org/python-bibtexparser
|
|
9
|
+
Project-URL: Repository, https://github.com/sciunto-org/python-bibtexparser
|
|
10
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
17
|
+
Classifier: Topic :: Text Processing :: Markup :: LaTeX
|
|
18
|
+
Classifier: Operating System :: OS Independent
|
|
19
|
+
Requires-Python: >=3.10
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Requires-Dist: pylatexenc~=2.10
|
|
23
|
+
Provides-Extra: docs
|
|
24
|
+
Requires-Dist: sphinx; extra == "docs"
|
|
25
|
+
Provides-Extra: test
|
|
26
|
+
Requires-Dist: pytest; extra == "test"
|
|
27
|
+
Requires-Dist: pytest-xdist; extra == "test"
|
|
28
|
+
Requires-Dist: pytest-cov; extra == "test"
|
|
29
|
+
Requires-Dist: jupyter; extra == "test"
|
|
30
|
+
Dynamic: license-file
|
|
31
|
+
|
|
32
|
+
# python-bibtexparser v2
|
|
33
|
+
|
|
34
|
+
Welcome to python-bibtexparser, a parser for `.bib` files with a long history and wide adoption.
|
|
35
|
+
|
|
36
|
+
Bibtexparser is available in two versions: V1 and V2. **V2 is the current, recommended version** and the default you get from PyPI. It provides an overall more robust and faster experience than v1, and is where all development and maintenance effort goes. Install it using pip:
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
pip install bibtexparser
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Or you can install the latest development version directly from the main branch:
|
|
43
|
+
```bash
|
|
44
|
+
pip install --no-cache-dir --force-reinstall git+https://github.com/sciunto-org/python-bibtexparser@main
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
If instead you still need v1, e.g. for a legacy project which you don't want to migrate right now, pin it explicitly:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
pip install bibtexparser~=1.0
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
Note that v2 is a rewrite: a lot has changed since v1, including the primary entrypoints, the data structures and the way parsing and writing are customized.
|
|
54
|
+
Existing v1 code will not run unchanged on v2 - have a look at our [migration guide](https://bibtexparser.readthedocs.io/en/main/migrate.html) when upgrading. While v2 has been thoroughly tested and has been in pre-release for a long time, please don't hesitate to report any issues you may encounter.
|
|
55
|
+
|
|
56
|
+
V1 is in maintenance mode: small PRs are still accepted, but only as long as they are backwards compatible and don't introduce much additional technical debt.
|
|
57
|
+
Development of version one happens on the dedicated [v1 branch](https://github.com/sciunto-org/python-bibtexparser/tree/v1).
|
|
58
|
+
|
|
59
|
+
## Documentation
|
|
60
|
+
Go check out our documentation on [https://bibtexparser.readthedocs.io/en/main/](https://bibtexparser.readthedocs.io/en/main/).
|
|
61
|
+
|
|
62
|
+
## Advantages of `v2` compared to `v1`
|
|
63
|
+
|
|
64
|
+
- :rocket: Order of magnitudes faster
|
|
65
|
+
- :wrench: Easily customizable parsing **and** writing
|
|
66
|
+
- :herb: Access to raw, unparsed bibtex.
|
|
67
|
+
- :shield: Fault-Tolerant: Able to parse files with syntax errors
|
|
68
|
+
- :mahjong: Massively simplified, more robust handling of de- and encoding (special chars, ...).
|
|
69
|
+
- :copyright: Permissive MIT license
|
|
70
|
+
|
|
71
|
+
## TLDR Usage Example
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
# Parsing a bibtex string with default values
|
|
75
|
+
bib_database = bibtexparser.parse_string(bibtex_string)
|
|
76
|
+
# Converting it back to a bibtex string, again with default values
|
|
77
|
+
new_bibtex_string = bibtexparser.write_string(bib_database)
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
Slightly more involved example:
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
|
|
84
|
+
# Lets parse some bibtex string.
|
|
85
|
+
bib_database = bibtexparser.parse_string(bibtex_string,
|
|
86
|
+
# Middleware layers to transform parsed entries.
|
|
87
|
+
# Here, we split multiple authors from each other and then extract first name, last name, ... for each
|
|
88
|
+
append_middleware=[SeparateCoAuthors(), SplitNameParts()],
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
# Here you have a `bib_database` with all parsed bibtex blocks.
|
|
92
|
+
|
|
93
|
+
# Let's transform it back to a bibtex_string.
|
|
94
|
+
new_bibtex_string = bibtexparser.write_string(bib_database,
|
|
95
|
+
# Revert above transformation
|
|
96
|
+
prepend_middleware=[MergeNameParts(), MergeCoAuthors()]
|
|
97
|
+
)
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
These examples really only show the bare minimum.
|
|
101
|
+
Consult the documentation for a list of available middleware, parsing options and write-formatting options.
|
|
102
|
+
|
|
103
|
+
## Architecture and Terminology
|
|
104
|
+
|
|
105
|
+

|
|
106
|
+
|
|
107
|
+
The architecture consists of the following components:
|
|
108
|
+
|
|
109
|
+
#### Library
|
|
110
|
+
Reflects the contents of a parsed bibtex files, including all comments, entries, strings, preambles and their metadata (e.g. order).
|
|
111
|
+
|
|
112
|
+
#### A Splitter
|
|
113
|
+
Splits a bibtex string into basic blocks (Entry, String, Preamble, ...), with correspondingly split content (e.g. fields on Entry, key-value on String, ...).
|
|
114
|
+
The splitter aims to be forgiving when facing invalid bibtex: A line starting with a block definition (``@....``) ends the previous block, even if not yet every bracket is closed, failing the parsing of the previous block. Correspondingly, one block type is "ParsingFailedBlock".
|
|
115
|
+
|
|
116
|
+
#### Middleware
|
|
117
|
+
Middleware layers transform a library and its blocks, for example by decoding latex special characters, interpolating string references, resolving crossreferences or re-ordering blocks. Thus, the choice of middleware allows to customize parsing and writing to ones specific usecase. Note: Middlewares, by default, do not mutate their input, but return a modified copy.
|
|
118
|
+
|
|
119
|
+
#### Writer
|
|
120
|
+
Writes the content of a bibtex library to a ``.bib`` file. Optional formatting parameters can be passed using a corresponding dedicated data structure.
|
|
121
|
+
|
|
122
|
+
## About
|
|
123
|
+
|
|
124
|
+
Since 2022, `bibtexparser` is primarily written and maintained by Michael Weiss ([@MiWeiss](https://github.com/MiWeiss/)), supported by various awesome contributors.
|
|
125
|
+
|
|
126
|
+
Credits and thanks to the many contributors who helped creating this library, including
|
|
127
|
+
François Boulogne ([@sciunto](https://github.com/sciunto/), creator of the first version) and Olivier Mangin ([@omangin](https://github.com/omangin/), long-term contributor).
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
# python-bibtexparser v2
|
|
2
|
+
|
|
3
|
+
Welcome to python-bibtexparser, a parser for `.bib` files with a long history and wide adoption.
|
|
4
|
+
|
|
5
|
+
Bibtexparser is available in two versions: V1 and V2. **V2 is the current, recommended version** and the default you get from PyPI. It provides an overall more robust and faster experience than v1, and is where all development and maintenance effort goes. Install it using pip:
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
pip install bibtexparser
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
Or you can install the latest development version directly from the main branch:
|
|
12
|
+
```bash
|
|
13
|
+
pip install --no-cache-dir --force-reinstall git+https://github.com/sciunto-org/python-bibtexparser@main
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
If instead you still need v1, e.g. for a legacy project which you don't want to migrate right now, pin it explicitly:
|
|
17
|
+
|
|
18
|
+
```bash
|
|
19
|
+
pip install bibtexparser~=1.0
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
Note that v2 is a rewrite: a lot has changed since v1, including the primary entrypoints, the data structures and the way parsing and writing are customized.
|
|
23
|
+
Existing v1 code will not run unchanged on v2 - have a look at our [migration guide](https://bibtexparser.readthedocs.io/en/main/migrate.html) when upgrading. While v2 has been thoroughly tested and has been in pre-release for a long time, please don't hesitate to report any issues you may encounter.
|
|
24
|
+
|
|
25
|
+
V1 is in maintenance mode: small PRs are still accepted, but only as long as they are backwards compatible and don't introduce much additional technical debt.
|
|
26
|
+
Development of version one happens on the dedicated [v1 branch](https://github.com/sciunto-org/python-bibtexparser/tree/v1).
|
|
27
|
+
|
|
28
|
+
## Documentation
|
|
29
|
+
Go check out our documentation on [https://bibtexparser.readthedocs.io/en/main/](https://bibtexparser.readthedocs.io/en/main/).
|
|
30
|
+
|
|
31
|
+
## Advantages of `v2` compared to `v1`
|
|
32
|
+
|
|
33
|
+
- :rocket: Order of magnitudes faster
|
|
34
|
+
- :wrench: Easily customizable parsing **and** writing
|
|
35
|
+
- :herb: Access to raw, unparsed bibtex.
|
|
36
|
+
- :shield: Fault-Tolerant: Able to parse files with syntax errors
|
|
37
|
+
- :mahjong: Massively simplified, more robust handling of de- and encoding (special chars, ...).
|
|
38
|
+
- :copyright: Permissive MIT license
|
|
39
|
+
|
|
40
|
+
## TLDR Usage Example
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
# Parsing a bibtex string with default values
|
|
44
|
+
bib_database = bibtexparser.parse_string(bibtex_string)
|
|
45
|
+
# Converting it back to a bibtex string, again with default values
|
|
46
|
+
new_bibtex_string = bibtexparser.write_string(bib_database)
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Slightly more involved example:
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
|
|
53
|
+
# Lets parse some bibtex string.
|
|
54
|
+
bib_database = bibtexparser.parse_string(bibtex_string,
|
|
55
|
+
# Middleware layers to transform parsed entries.
|
|
56
|
+
# Here, we split multiple authors from each other and then extract first name, last name, ... for each
|
|
57
|
+
append_middleware=[SeparateCoAuthors(), SplitNameParts()],
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
# Here you have a `bib_database` with all parsed bibtex blocks.
|
|
61
|
+
|
|
62
|
+
# Let's transform it back to a bibtex_string.
|
|
63
|
+
new_bibtex_string = bibtexparser.write_string(bib_database,
|
|
64
|
+
# Revert above transformation
|
|
65
|
+
prepend_middleware=[MergeNameParts(), MergeCoAuthors()]
|
|
66
|
+
)
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
These examples really only show the bare minimum.
|
|
70
|
+
Consult the documentation for a list of available middleware, parsing options and write-formatting options.
|
|
71
|
+
|
|
72
|
+
## Architecture and Terminology
|
|
73
|
+
|
|
74
|
+

|
|
75
|
+
|
|
76
|
+
The architecture consists of the following components:
|
|
77
|
+
|
|
78
|
+
#### Library
|
|
79
|
+
Reflects the contents of a parsed bibtex files, including all comments, entries, strings, preambles and their metadata (e.g. order).
|
|
80
|
+
|
|
81
|
+
#### A Splitter
|
|
82
|
+
Splits a bibtex string into basic blocks (Entry, String, Preamble, ...), with correspondingly split content (e.g. fields on Entry, key-value on String, ...).
|
|
83
|
+
The splitter aims to be forgiving when facing invalid bibtex: A line starting with a block definition (``@....``) ends the previous block, even if not yet every bracket is closed, failing the parsing of the previous block. Correspondingly, one block type is "ParsingFailedBlock".
|
|
84
|
+
|
|
85
|
+
#### Middleware
|
|
86
|
+
Middleware layers transform a library and its blocks, for example by decoding latex special characters, interpolating string references, resolving crossreferences or re-ordering blocks. Thus, the choice of middleware allows to customize parsing and writing to ones specific usecase. Note: Middlewares, by default, do not mutate their input, but return a modified copy.
|
|
87
|
+
|
|
88
|
+
#### Writer
|
|
89
|
+
Writes the content of a bibtex library to a ``.bib`` file. Optional formatting parameters can be passed using a corresponding dedicated data structure.
|
|
90
|
+
|
|
91
|
+
## About
|
|
92
|
+
|
|
93
|
+
Since 2022, `bibtexparser` is primarily written and maintained by Michael Weiss ([@MiWeiss](https://github.com/MiWeiss/)), supported by various awesome contributors.
|
|
94
|
+
|
|
95
|
+
Credits and thanks to the many contributors who helped creating this library, including
|
|
96
|
+
François Boulogne ([@sciunto](https://github.com/sciunto/), creator of the first version) and Olivier Mangin ([@omangin](https://github.com/omangin/), long-term contributor).
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import bibtexparser.exceptions
|
|
2
|
+
import bibtexparser.middlewares
|
|
3
|
+
import bibtexparser.model
|
|
4
|
+
from bibtexparser.entrypoint import parse_file
|
|
5
|
+
from bibtexparser.entrypoint import parse_string
|
|
6
|
+
from bibtexparser.entrypoint import write_file
|
|
7
|
+
from bibtexparser.entrypoint import write_string
|
|
8
|
+
from bibtexparser.library import Library
|
|
9
|
+
from bibtexparser.writer import BibtexFormat
|
|
10
|
+
|
|
11
|
+
__version__ = "2.0.0"
|
|
@@ -0,0 +1,402 @@
|
|
|
1
|
+
import codecs
|
|
2
|
+
import logging
|
|
3
|
+
import warnings
|
|
4
|
+
from collections.abc import Iterable
|
|
5
|
+
from copy import deepcopy
|
|
6
|
+
from typing import Optional
|
|
7
|
+
from typing import TextIO
|
|
8
|
+
|
|
9
|
+
from .library import Library
|
|
10
|
+
from .middlewares.enclosing import REMOVED_ENCLOSING_KEY
|
|
11
|
+
from .middlewares.middleware import Middleware
|
|
12
|
+
from .middlewares.parsestack import default_parse_stack
|
|
13
|
+
from .middlewares.parsestack import default_unparse_stack
|
|
14
|
+
from .model import Block
|
|
15
|
+
from .model import String
|
|
16
|
+
from .splitter import Splitter
|
|
17
|
+
from .writer import BibtexFormat
|
|
18
|
+
from .writer import write
|
|
19
|
+
|
|
20
|
+
logger = logging.getLogger(__name__)
|
|
21
|
+
|
|
22
|
+
#: Marks a seeded copy of a pre-existing `@string`, dropped before merging back.
|
|
23
|
+
_PREEXISTING_STRING_KEY = "bibtexparser_preexisting_string"
|
|
24
|
+
|
|
25
|
+
#: Number of blocks from which on `write_string`/`write_file` warn if the unparse
|
|
26
|
+
#: stack deep-copies blocks. Copying costs roughly 30-60 µs per entry, i.e. it
|
|
27
|
+
#: starts to dominate the write time at this size.
|
|
28
|
+
LARGE_LIBRARY_WARNING_THRESHOLD = 10_000
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _build_parse_stack(
|
|
32
|
+
parse_stack: Iterable[Middleware] | None,
|
|
33
|
+
append_middleware: Iterable[Middleware] | None,
|
|
34
|
+
) -> list[Middleware]:
|
|
35
|
+
# Materialize upfront: the arguments may be one-shot iterators.
|
|
36
|
+
parse_stack = None if parse_stack is None else list(parse_stack)
|
|
37
|
+
append_middleware = None if append_middleware is None else list(append_middleware)
|
|
38
|
+
|
|
39
|
+
if parse_stack is not None and append_middleware is not None:
|
|
40
|
+
raise ValueError(
|
|
41
|
+
"Provided both parse_stack and append_middleware. "
|
|
42
|
+
"Only one should be provided. "
|
|
43
|
+
"(append_middleware should only be used with the default parse_stack, "
|
|
44
|
+
"i.e., when the passed parse_stack is None.)"
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
if parse_stack is None:
|
|
48
|
+
parse_stack = default_parse_stack(allow_inplace_modification=True)
|
|
49
|
+
|
|
50
|
+
if append_middleware is None:
|
|
51
|
+
return list(parse_stack)
|
|
52
|
+
|
|
53
|
+
parse_stack_types = {type(m) for m in parse_stack}
|
|
54
|
+
append_stack_types = {type(m) for m in append_middleware}
|
|
55
|
+
stack_types_intersect = parse_stack_types.intersection(append_stack_types)
|
|
56
|
+
if len(stack_types_intersect) > 0:
|
|
57
|
+
warnings.warn(
|
|
58
|
+
"Some middleware passed in append_middleware are "
|
|
59
|
+
f"already in the default parse_stack ({stack_types_intersect})."
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
return list(parse_stack) + list(append_middleware)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _build_unparse_stack(
|
|
66
|
+
unparse_stack: Iterable[Middleware] | None,
|
|
67
|
+
prepend_middleware: Iterable[Middleware] | None,
|
|
68
|
+
) -> list[Middleware]:
|
|
69
|
+
# Materialize upfront: the arguments may be one-shot iterators.
|
|
70
|
+
unparse_stack = None if unparse_stack is None else list(unparse_stack)
|
|
71
|
+
prepend_middleware = None if prepend_middleware is None else list(prepend_middleware)
|
|
72
|
+
|
|
73
|
+
if unparse_stack is not None and prepend_middleware is not None:
|
|
74
|
+
raise ValueError(
|
|
75
|
+
"Provided both unparse_stack and prepend_middleware. "
|
|
76
|
+
"Only one should be provided. "
|
|
77
|
+
"(prepend_middleware should only be used with the default unparse_stack, "
|
|
78
|
+
"i.e., when the passed unparse_stack is None.)"
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
if unparse_stack is None:
|
|
82
|
+
unparse_stack = default_unparse_stack(allow_inplace_modification=False)
|
|
83
|
+
|
|
84
|
+
if prepend_middleware is None:
|
|
85
|
+
return list(unparse_stack)
|
|
86
|
+
|
|
87
|
+
parse_stack_types = {type(m) for m in unparse_stack}
|
|
88
|
+
append_stack_types = {type(m) for m in prepend_middleware}
|
|
89
|
+
stack_types_intersect = parse_stack_types.intersection(append_stack_types)
|
|
90
|
+
if len(stack_types_intersect) > 0:
|
|
91
|
+
warnings.warn(
|
|
92
|
+
"Some middleware passed in prepend_middleware are "
|
|
93
|
+
f"already in the default unparse_stack ({stack_types_intersect})."
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
return list(prepend_middleware) + list(unparse_stack)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _warn_if_large_library_is_copied(library: Library, unparse_stack: list[Middleware]) -> None:
|
|
100
|
+
"""Warn if writing ``library`` will deep-copy its blocks and that is likely slow.
|
|
101
|
+
|
|
102
|
+
Middlewares with ``allow_inplace_modification=False`` (the default unparse stack
|
|
103
|
+
is built that way) deep-copy every block they transform, which dominates the
|
|
104
|
+
write time of large libraries.
|
|
105
|
+
"""
|
|
106
|
+
n_blocks = len(library.blocks)
|
|
107
|
+
if n_blocks < LARGE_LIBRARY_WARNING_THRESHOLD:
|
|
108
|
+
return
|
|
109
|
+
if all(middleware.allow_inplace_modification for middleware in unparse_stack):
|
|
110
|
+
return
|
|
111
|
+
logger.warning(
|
|
112
|
+
f"Writing a library with {n_blocks} blocks: the unparse stack deep-copies blocks "
|
|
113
|
+
"(it contains middlewares with allow_inplace_modification=False), "
|
|
114
|
+
"which is slow for large libraries. "
|
|
115
|
+
"If you do not need the library after writing, pass an unparse stack whose "
|
|
116
|
+
"middlewares all allow in-place modification, e.g. "
|
|
117
|
+
"`unparse_stack=bibtexparser.middlewares.default_unparse_stack("
|
|
118
|
+
"allow_inplace_modification=True)`."
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _handle_deprecated_write_params(
|
|
123
|
+
unparse_stack: Iterable[Middleware] | None,
|
|
124
|
+
prepend_middleware: Iterable[Middleware] | None,
|
|
125
|
+
kwargs: dict,
|
|
126
|
+
function_name: str,
|
|
127
|
+
) -> tuple[Iterable[Middleware] | None, Iterable[Middleware] | None]:
|
|
128
|
+
"""Handle deprecated parameter names for write functions.
|
|
129
|
+
|
|
130
|
+
:param unparse_stack: Current unparse_stack value
|
|
131
|
+
:param prepend_middleware: Current prepend_middleware value
|
|
132
|
+
:param kwargs: Dictionary of keyword arguments to check for deprecated params
|
|
133
|
+
:param function_name: Name of the calling function (for error messages)
|
|
134
|
+
:return: Tuple of (unparse_stack, prepend_middleware) with deprecated values migrated
|
|
135
|
+
"""
|
|
136
|
+
if "parse_stack" in kwargs:
|
|
137
|
+
warnings.warn(
|
|
138
|
+
"Parameter 'parse_stack' is deprecated. Use 'unparse_stack' instead.",
|
|
139
|
+
DeprecationWarning,
|
|
140
|
+
stacklevel=3,
|
|
141
|
+
)
|
|
142
|
+
if unparse_stack is not None:
|
|
143
|
+
raise ValueError(
|
|
144
|
+
"Cannot provide both 'parse_stack' (deprecated) and 'unparse_stack'. "
|
|
145
|
+
"Use 'unparse_stack' instead."
|
|
146
|
+
)
|
|
147
|
+
unparse_stack = kwargs.pop("parse_stack")
|
|
148
|
+
|
|
149
|
+
if "append_middleware" in kwargs:
|
|
150
|
+
warnings.warn(
|
|
151
|
+
"Parameter 'append_middleware' is deprecated. Use 'prepend_middleware' instead.",
|
|
152
|
+
DeprecationWarning,
|
|
153
|
+
stacklevel=3,
|
|
154
|
+
)
|
|
155
|
+
if prepend_middleware is not None:
|
|
156
|
+
raise ValueError(
|
|
157
|
+
"Cannot provide both 'append_middleware' (deprecated) and 'prepend_middleware'. "
|
|
158
|
+
"Use 'prepend_middleware' instead."
|
|
159
|
+
)
|
|
160
|
+
prepend_middleware = kwargs.pop("append_middleware")
|
|
161
|
+
|
|
162
|
+
if kwargs:
|
|
163
|
+
raise TypeError(f"{function_name}() got unexpected keyword arguments: {', '.join(kwargs)}")
|
|
164
|
+
|
|
165
|
+
return unparse_stack, prepend_middleware
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def parse_string(
|
|
169
|
+
bibtex_str: str,
|
|
170
|
+
parse_stack: Iterable[Middleware] | None = None,
|
|
171
|
+
append_middleware: Iterable[Middleware] | None = None,
|
|
172
|
+
library: Library | None = None,
|
|
173
|
+
) -> Library:
|
|
174
|
+
"""Parse a BibTeX string.
|
|
175
|
+
|
|
176
|
+
:param bibtex_str: BibTeX string to parse
|
|
177
|
+
:param parse_stack:
|
|
178
|
+
List of middleware to apply to the database after splitting.
|
|
179
|
+
If ``None`` (default), a default stack will be used providing simple standard functionality.
|
|
180
|
+
|
|
181
|
+
:param append_middleware:
|
|
182
|
+
List of middleware to append to the default stack
|
|
183
|
+
(ignored if a not-``None`` parse_stack is passed).
|
|
184
|
+
|
|
185
|
+
:param library:
|
|
186
|
+
Library to add the newly parsed blocks to.
|
|
187
|
+
If ``None`` (default), a new library is created and returned.
|
|
188
|
+
If a library is passed, it is returned (mutated) and:
|
|
189
|
+
|
|
190
|
+
- the parse stack is applied **only** to the newly parsed blocks;
|
|
191
|
+
blocks already contained in the passed library are left untouched
|
|
192
|
+
(they were already transformed when they were parsed);
|
|
193
|
+
- ``@string`` blocks already contained in the passed library are visible
|
|
194
|
+
to the parse stack, i.e. string references in ``bibtex_str`` resolve
|
|
195
|
+
against them (unless ``bibtex_str`` redefines the same key);
|
|
196
|
+
- keys defined both in the passed library and in ``bibtex_str``
|
|
197
|
+
do not raise, but yield ``DuplicateBlockKeyBlock`` instances
|
|
198
|
+
(see ``library.failed_blocks``), just like duplicates within a single string.
|
|
199
|
+
|
|
200
|
+
:return: Library: Parsed BibTeX database
|
|
201
|
+
"""
|
|
202
|
+
splitter = Splitter(bibstr=bibtex_str)
|
|
203
|
+
parsed = splitter.split()
|
|
204
|
+
|
|
205
|
+
_seed_preexisting_strings(parsed, library)
|
|
206
|
+
|
|
207
|
+
middleware: Middleware
|
|
208
|
+
for middleware in _build_parse_stack(parse_stack, append_middleware):
|
|
209
|
+
parsed = middleware.transform(library=parsed)
|
|
210
|
+
|
|
211
|
+
if library is None:
|
|
212
|
+
return parsed
|
|
213
|
+
|
|
214
|
+
new_blocks = [b for b in parsed.blocks if not _is_seeded_string(b)]
|
|
215
|
+
library.add(new_blocks, fail_on_duplicate_key=False)
|
|
216
|
+
return library
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _is_seeded_string(block: Block) -> bool:
|
|
220
|
+
"""True for blocks seeded by `_seed_preexisting_strings` (and their transformations)."""
|
|
221
|
+
return bool(block.get_parser_metadata(_PREEXISTING_STRING_KEY))
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _restore_enclosing(string: String) -> None:
|
|
225
|
+
"""Make sure the value of an already-parsed string is enclosed again.
|
|
226
|
+
|
|
227
|
+
The parse stack expects freshly split (i.e. still enclosed) values.
|
|
228
|
+
Feeding it an already-stripped value would make that value be treated
|
|
229
|
+
as an unenclosed literal (a string reference), which does not round-trip
|
|
230
|
+
to valid bibtex.
|
|
231
|
+
"""
|
|
232
|
+
enclosing = string.parser_metadata.pop(REMOVED_ENCLOSING_KEY, None)
|
|
233
|
+
if string.enclosing == "no-enclosing" or enclosing == "no-enclosing":
|
|
234
|
+
return
|
|
235
|
+
value = string.value
|
|
236
|
+
if not isinstance(value, str):
|
|
237
|
+
return
|
|
238
|
+
if enclosing is None and (
|
|
239
|
+
(value.startswith("{") and value.endswith("}"))
|
|
240
|
+
or (value.startswith('"') and value.endswith('"'))
|
|
241
|
+
):
|
|
242
|
+
return
|
|
243
|
+
string.value = f'"{value}"' if enclosing == '"' else f"{{{value}}}"
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _seed_preexisting_strings(parsed: Library, library: Library | None) -> list[String]:
|
|
247
|
+
"""Make the ``@string`` blocks of an existing library visible to the parse stack.
|
|
248
|
+
|
|
249
|
+
Copies (never the originals, which must not be transformed again) of the
|
|
250
|
+
strings of ``library`` are added to ``parsed``, unless the newly parsed
|
|
251
|
+
content redefines the same key. The copies are tagged so that they can be
|
|
252
|
+
dropped again before merging the parsed blocks back into ``library``.
|
|
253
|
+
|
|
254
|
+
:param parsed: The freshly split library, modified in place.
|
|
255
|
+
:param library: The pre-existing library, or ``None``.
|
|
256
|
+
:return: The seeded (tagged) string copies.
|
|
257
|
+
"""
|
|
258
|
+
if library is None:
|
|
259
|
+
return []
|
|
260
|
+
|
|
261
|
+
# Bibtex string keys are case-insensitive, hence compare in lower case.
|
|
262
|
+
redefined = {key.lower() for key in parsed.strings_dict}
|
|
263
|
+
seeds = []
|
|
264
|
+
for key, string in library.strings_dict.items():
|
|
265
|
+
if key.lower() in redefined:
|
|
266
|
+
continue
|
|
267
|
+
seed = deepcopy(string)
|
|
268
|
+
seed.set_parser_metadata(_PREEXISTING_STRING_KEY, True)
|
|
269
|
+
_restore_enclosing(seed)
|
|
270
|
+
seeds.append(seed)
|
|
271
|
+
|
|
272
|
+
if seeds:
|
|
273
|
+
parsed.add(seeds, fail_on_duplicate_key=False)
|
|
274
|
+
return seeds
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def parse_file(
|
|
278
|
+
path: str,
|
|
279
|
+
parse_stack: Iterable[Middleware] | None = None,
|
|
280
|
+
append_middleware: Iterable[Middleware] | None = None,
|
|
281
|
+
encoding: str = "UTF-8",
|
|
282
|
+
) -> Library:
|
|
283
|
+
"""Parse a BibTeX file
|
|
284
|
+
|
|
285
|
+
:param path: Path to BibTeX file
|
|
286
|
+
:param parse_stack:
|
|
287
|
+
List of middleware to apply to the database after splitting.
|
|
288
|
+
If ``None`` (default), a default stack will be used providing simple standard functionality.
|
|
289
|
+
|
|
290
|
+
:param append_middleware:
|
|
291
|
+
List of middleware to append to the default stack
|
|
292
|
+
(ignored if a not-``None`` parse_stack is passed).
|
|
293
|
+
|
|
294
|
+
:param encoding: Encoding of the .bib file. Default encoding is ``"UTF-8"``.
|
|
295
|
+
:return: Library: Parsed BibTeX library
|
|
296
|
+
:raises LookupError: If the specified encoding is not recognized.
|
|
297
|
+
"""
|
|
298
|
+
try:
|
|
299
|
+
codecs.lookup(encoding)
|
|
300
|
+
except LookupError:
|
|
301
|
+
raise LookupError(f"Unknown encoding: {encoding!r}")
|
|
302
|
+
|
|
303
|
+
with open(path, encoding=encoding) as f:
|
|
304
|
+
bibtex_str = f.read()
|
|
305
|
+
return parse_string(
|
|
306
|
+
bibtex_str, parse_stack=parse_stack, append_middleware=append_middleware
|
|
307
|
+
)
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def write_file(
|
|
311
|
+
file: str | TextIO,
|
|
312
|
+
library: Library,
|
|
313
|
+
unparse_stack: Iterable[Middleware] | None = None,
|
|
314
|
+
prepend_middleware: Iterable[Middleware] | None = None,
|
|
315
|
+
bibtex_format: BibtexFormat | None = None,
|
|
316
|
+
encoding: str = "UTF-8",
|
|
317
|
+
**kwargs,
|
|
318
|
+
) -> None:
|
|
319
|
+
"""Write a BibTeX database to a file.
|
|
320
|
+
|
|
321
|
+
The passed library is never modified, unless *every* middleware in the
|
|
322
|
+
unparse stack allows in-place modification (e.g.
|
|
323
|
+
``unparse_stack=default_unparse_stack(allow_inplace_modification=True)``).
|
|
324
|
+
|
|
325
|
+
:param file: File to write to. Can be a file name or a file object.
|
|
326
|
+
:param library: BibTeX database to serialize.
|
|
327
|
+
:param unparse_stack: List of middleware to apply to the database before writing.
|
|
328
|
+
If None, a default stack will be used.
|
|
329
|
+
:param prepend_middleware: List of middleware to prepend to the default stack.
|
|
330
|
+
Only applicable if `unparse_stack` is None.
|
|
331
|
+
:param bibtex_format: Customized BibTeX format to use (optional).
|
|
332
|
+
:param encoding: Encoding of the .bib file. Default encoding is ``"UTF-8"``.
|
|
333
|
+
Writing a library with at least ``LARGE_LIBRARY_WARNING_THRESHOLD`` blocks logs a warning
|
|
334
|
+
if the unparse stack deep-copies blocks (middlewares with ``allow_inplace_modification=False``),
|
|
335
|
+
as that is slow; pass an all-in-place stack to avoid it.
|
|
336
|
+
|
|
337
|
+
.. deprecated:: (next version)
|
|
338
|
+
Parameters 'parse_stack' and 'append_middleware' are deprecated, will be deleted soon.
|
|
339
|
+
Use 'unparse_stack' and 'prepend_middleware' instead.
|
|
340
|
+
"""
|
|
341
|
+
unparse_stack, prepend_middleware = _handle_deprecated_write_params(
|
|
342
|
+
unparse_stack, prepend_middleware, kwargs, "write_file"
|
|
343
|
+
)
|
|
344
|
+
|
|
345
|
+
bibtex_str = write_string(
|
|
346
|
+
library=library,
|
|
347
|
+
unparse_stack=unparse_stack,
|
|
348
|
+
prepend_middleware=prepend_middleware,
|
|
349
|
+
bibtex_format=bibtex_format,
|
|
350
|
+
)
|
|
351
|
+
if isinstance(file, str):
|
|
352
|
+
with open(file, "w", encoding=encoding) as f:
|
|
353
|
+
f.write(bibtex_str)
|
|
354
|
+
else:
|
|
355
|
+
file.write(bibtex_str)
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def write_string(
|
|
359
|
+
library: Library,
|
|
360
|
+
unparse_stack: Iterable[Middleware] | None = None,
|
|
361
|
+
prepend_middleware: Iterable[Middleware] | None = None,
|
|
362
|
+
bibtex_format: Optional["BibtexFormat"] = None,
|
|
363
|
+
**kwargs,
|
|
364
|
+
) -> str:
|
|
365
|
+
"""Serialize a BibTeX database to a string.
|
|
366
|
+
|
|
367
|
+
The passed library is never modified, unless *every* middleware in the
|
|
368
|
+
unparse stack allows in-place modification (e.g.
|
|
369
|
+
``unparse_stack=default_unparse_stack(allow_inplace_modification=True)``).
|
|
370
|
+
|
|
371
|
+
:param library: BibTeX database to serialize.
|
|
372
|
+
:param unparse_stack: List of middleware to apply to the database before writing.
|
|
373
|
+
If None, a default stack will be used.
|
|
374
|
+
:param prepend_middleware: List of middleware to prepend to the default stack.
|
|
375
|
+
Only applicable if `unparse_stack` is None.
|
|
376
|
+
:param bibtex_format: Customized BibTeX format to use (optional).
|
|
377
|
+
Writing a library with at least ``LARGE_LIBRARY_WARNING_THRESHOLD`` blocks logs a warning
|
|
378
|
+
if the unparse stack deep-copies blocks (middlewares with ``allow_inplace_modification=False``),
|
|
379
|
+
as that is slow; pass an all-in-place stack to avoid it.
|
|
380
|
+
|
|
381
|
+
.. deprecated:: (next version)
|
|
382
|
+
Parameters 'parse_stack' and 'append_middleware' are deprecated.
|
|
383
|
+
Use 'unparse_stack' and 'prepend_middleware' instead.
|
|
384
|
+
"""
|
|
385
|
+
unparse_stack, prepend_middleware = _handle_deprecated_write_params(
|
|
386
|
+
unparse_stack, prepend_middleware, kwargs, "write_string"
|
|
387
|
+
)
|
|
388
|
+
|
|
389
|
+
stack = _build_unparse_stack(unparse_stack, prepend_middleware)
|
|
390
|
+
_warn_if_large_library_is_copied(library, stack)
|
|
391
|
+
inplace = [middleware.allow_inplace_modification for middleware in stack]
|
|
392
|
+
if any(inplace) and not all(inplace):
|
|
393
|
+
# Some middleware would mutate the passed library before a copying
|
|
394
|
+
# middleware gets to run; copy once upfront so the caller's library
|
|
395
|
+
# stays untouched (an all-in-place stack is the caller's explicit opt-in).
|
|
396
|
+
library = deepcopy(library)
|
|
397
|
+
|
|
398
|
+
middleware: Middleware
|
|
399
|
+
for middleware in stack:
|
|
400
|
+
library = middleware.transform(library=library)
|
|
401
|
+
|
|
402
|
+
return write(library, bibtex_format=bibtex_format)
|