edgar-itemize 1.0.0rc2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- edgar_itemize-1.0.0rc2/LICENSE +30 -0
- edgar_itemize-1.0.0rc2/LICENSES/PSF-2.0.txt +48 -0
- edgar_itemize-1.0.0rc2/PKG-INFO +281 -0
- edgar_itemize-1.0.0rc2/README.md +254 -0
- edgar_itemize-1.0.0rc2/pyproject.toml +52 -0
- edgar_itemize-1.0.0rc2/pyproject.toml.orig +43 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/__init__.py +30 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/_html_parser.py +559 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/agenda.py +934 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/ars_kind.py +133 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/blocks.py +58 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/candidates.py +1872 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/classify.py +183 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/cli.py +359 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/conformance.py +670 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/control.py +166 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/eval/__init__.py +0 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/eval/gold.py +54 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/eval/judge.py +85 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/eval/metrics.py +171 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/fetch.py +227 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/grammar/__init__.py +2 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/grammar/base.py +31 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/grammar/contract.py +160 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/grammar/form10k.py +316 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/grammar/form10q.py +322 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/headings.py +282 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/llm.py +389 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/manifest.py +290 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/normalize_html.py +593 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/normalize_text.py +172 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/normcache.py +212 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/offsets.py +43 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/pipeline.py +298 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/prepare.py +81 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/schema.py +105 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/select.py +76 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/sequence.py +180 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/sgml.py +234 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/toc.py +373 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/tree.py +964 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/tree_contract.py +757 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/__init__.py +0 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/app.py +743 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/original.py +123 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/runread.py +256 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/sets.py +427 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/static/index.html +1424 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/static/turns.html +501 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/turns.py +60 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/windex.py +194 -0
- edgar_itemize-1.0.0rc2/src/edgar_itemize/writer.py +18 -0
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Malcolm Wardlaw
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
22
|
+
|
|
23
|
+
---
|
|
24
|
+
|
|
25
|
+
Third-party code
|
|
26
|
+
|
|
27
|
+
src/edgar_itemize/_html_parser.py is a verbatim copy of CPython 3.11.15's
|
|
28
|
+
Lib/html/parser.py (the html.parser module), Copyright (c) 2001-2023 Python
|
|
29
|
+
Software Foundation; All Rights Reserved, and is distributed under the Python
|
|
30
|
+
Software Foundation License Version 2, reproduced in LICENSES/PSF-2.0.txt.
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
PYTHON SOFTWARE FOUNDATION LICENSE VERSION 2
|
|
2
|
+
--------------------------------------------
|
|
3
|
+
|
|
4
|
+
1. This LICENSE AGREEMENT is between the Python Software Foundation
|
|
5
|
+
("PSF"), and the Individual or Organization ("Licensee") accessing and
|
|
6
|
+
otherwise using this software ("Python") in source or binary form and
|
|
7
|
+
its associated documentation.
|
|
8
|
+
|
|
9
|
+
2. Subject to the terms and conditions of this License Agreement, PSF hereby
|
|
10
|
+
grants Licensee a nonexclusive, royalty-free, world-wide license to reproduce,
|
|
11
|
+
analyze, test, perform and/or display publicly, prepare derivative works,
|
|
12
|
+
distribute, and otherwise use Python alone or in any derivative version,
|
|
13
|
+
provided, however, that PSF's License Agreement and PSF's notice of copyright,
|
|
14
|
+
i.e., "Copyright (c) 2001, 2002, 2003, 2004, 2005, 2006, 2007, 2008, 2009, 2010,
|
|
15
|
+
2011, 2012, 2013, 2014, 2015, 2016, 2017, 2018, 2019, 2020, 2021, 2022, 2023 Python Software Foundation;
|
|
16
|
+
All Rights Reserved" are retained in Python alone or in any derivative version
|
|
17
|
+
prepared by Licensee.
|
|
18
|
+
|
|
19
|
+
3. In the event Licensee prepares a derivative work that is based on
|
|
20
|
+
or incorporates Python or any part thereof, and wants to make
|
|
21
|
+
the derivative work available to others as provided herein, then
|
|
22
|
+
Licensee hereby agrees to include in any such work a brief summary of
|
|
23
|
+
the changes made to Python.
|
|
24
|
+
|
|
25
|
+
4. PSF is making Python available to Licensee on an "AS IS"
|
|
26
|
+
basis. PSF MAKES NO REPRESENTATIONS OR WARRANTIES, EXPRESS OR
|
|
27
|
+
IMPLIED. BY WAY OF EXAMPLE, BUT NOT LIMITATION, PSF MAKES NO AND
|
|
28
|
+
DISCLAIMS ANY REPRESENTATION OR WARRANTY OF MERCHANTABILITY OR FITNESS
|
|
29
|
+
FOR ANY PARTICULAR PURPOSE OR THAT THE USE OF PYTHON WILL NOT
|
|
30
|
+
INFRINGE ANY THIRD PARTY RIGHTS.
|
|
31
|
+
|
|
32
|
+
5. PSF SHALL NOT BE LIABLE TO LICENSEE OR ANY OTHER USERS OF PYTHON
|
|
33
|
+
FOR ANY INCIDENTAL, SPECIAL, OR CONSEQUENTIAL DAMAGES OR LOSS AS
|
|
34
|
+
A RESULT OF MODIFYING, DISTRIBUTING, OR OTHERWISE USING PYTHON,
|
|
35
|
+
OR ANY DERIVATIVE THEREOF, EVEN IF ADVISED OF THE POSSIBILITY THEREOF.
|
|
36
|
+
|
|
37
|
+
6. This License Agreement will automatically terminate upon a material
|
|
38
|
+
breach of its terms and conditions.
|
|
39
|
+
|
|
40
|
+
7. Nothing in this License Agreement shall be deemed to create any
|
|
41
|
+
relationship of agency, partnership, or joint venture between PSF and
|
|
42
|
+
Licensee. This License Agreement does not grant permission to use PSF
|
|
43
|
+
trademarks or trade name in a trademark sense to endorse or promote
|
|
44
|
+
products or services of Licensee, or any third party.
|
|
45
|
+
|
|
46
|
+
8. By copying, installing or otherwise using Python, Licensee
|
|
47
|
+
agrees to be bound by the terms and conditions of this License
|
|
48
|
+
Agreement.
|
|
@@ -0,0 +1,281 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: edgar-itemize
|
|
3
|
+
Version: 1.0.0rc2
|
|
4
|
+
Summary: Deterministic agenda-structure (Part/Item/Section) parser for SEC EDGAR filings
|
|
5
|
+
Author: Malcolm Wardlaw
|
|
6
|
+
Author-email: Malcolm Wardlaw <malcolm.wardlaw@uga.edu>
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
License-File: LICENSES/PSF-2.0.txt
|
|
10
|
+
Requires-Dist: regex>=2024.0
|
|
11
|
+
Requires-Dist: pyarrow>=14
|
|
12
|
+
Requires-Dist: pyyaml>=6
|
|
13
|
+
Requires-Dist: pytest>=8 ; extra == 'dev'
|
|
14
|
+
Requires-Dist: polars>=1.0 ; extra == 'dev'
|
|
15
|
+
Requires-Dist: mkdocs-material>=9.5 ; extra == 'docs'
|
|
16
|
+
Requires-Dist: httpx>=0.28.1 ; extra == 'judge'
|
|
17
|
+
Requires-Dist: polars>=1.0 ; extra == 'polars'
|
|
18
|
+
Requires-Dist: fastapi>=0.141.1 ; extra == 'viewer'
|
|
19
|
+
Requires-Dist: uvicorn[standard]>=0.52.4 ; extra == 'viewer'
|
|
20
|
+
Requires-Python: >=3.11
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Provides-Extra: docs
|
|
23
|
+
Provides-Extra: judge
|
|
24
|
+
Provides-Extra: polars
|
|
25
|
+
Provides-Extra: viewer
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
|
|
28
|
+
# edgar-itemize
|
|
29
|
+
|
|
30
|
+
edgar-itemize turns the raw filings in the SEC's EDGAR archive into a table of their
|
|
31
|
+
headings. For each filing it recovers the agenda the document is written to: Part and Item
|
|
32
|
+
in a 10-K or 10-Q, sub-headings inside an Item, and Article, Section and the lettered and
|
|
33
|
+
numbered clauses of a credit agreement filed as an EX-10 exhibit. Each heading is one row
|
|
34
|
+
that records where its section starts and ends as byte offsets into the submission file
|
|
35
|
+
exactly as the SEC serves it, so the text of any section, from Item 1A down to a single
|
|
36
|
+
covenant clause, is one slice of a file you already have. It reads the whole EDGAR era,
|
|
37
|
+
from 1993 plain text through publisher HTML to inline XBRL.
|
|
38
|
+
|
|
39
|
+
* **Manual** <https://malcolmwardlaw.github.io/edgar-itemize/>
|
|
40
|
+
* **Output Demo** <https://malcolmwardlaw.github.io/edgar-itemize/demo/
|
|
41
|
+
|
|
42
|
+
## Why this exists
|
|
43
|
+
|
|
44
|
+
* **The whole agenda, down to the clauses.** Most tools pull out a few top-level sections.
|
|
45
|
+
10-Ks, and the contracts attached as EX-10 exhibits even more, have deep agenda
|
|
46
|
+
structures: a study of covenants needs the affirmative and negative covenant Articles and
|
|
47
|
+
each clause one or two levels beneath them. The tree goes to that depth.
|
|
48
|
+
* **Your corpus, downloaded once, parsed locally.** You pull as much or as little of EDGAR
|
|
49
|
+
as you want, up to the whole archive, and parse it on your own machine. Rate-limited SEC
|
|
50
|
+
downloads are slow but happen once; storage is cheap; re-parsing with different choices
|
|
51
|
+
costs nothing but compute.
|
|
52
|
+
* **Deterministic.** The parser is a fixed set of written rules, not a language model
|
|
53
|
+
deciding over a corpus. The same release on the same file gives the same rows every time,
|
|
54
|
+
every heading names the rules that produced it, and every candidate heading that was
|
|
55
|
+
dropped is written out with the reason.
|
|
56
|
+
* **Replicable by hash, with the data untouched.** Each tagged release gives exactly the
|
|
57
|
+
same results on exactly the same EDGAR files, checked by SHA-256 of the input and the
|
|
58
|
+
output. The filings themselves are never modified or redistributed, so nobody has to
|
|
59
|
+
archive gigabytes of derived text as validation, and every value traces directly to a
|
|
60
|
+
byte range in a public file.
|
|
61
|
+
* **Improves by frozen tagged releases.** The output is evaluated by human review and by
|
|
62
|
+
language-model judges, and the evaluation feeds hand-written rule changes. Current results
|
|
63
|
+
are good, with room left (see [Measured accuracy](#measured-accuracy)). Each improvement
|
|
64
|
+
ships as a new tagged release that keeps the two guarantees above; earlier releases stay
|
|
65
|
+
as they were.
|
|
66
|
+
* **Public and updatable.** The author keeps working on it. Anyone who finds a set of
|
|
67
|
+
errors can package them and send them in, and they are folded into the next release.
|
|
68
|
+
|
|
69
|
+
## What to do with it
|
|
70
|
+
|
|
71
|
+
**Start small.** `examples/` holds manifests and expected hashes for the nine filings of the
|
|
72
|
+
demo and for ten credit agreements; the
|
|
73
|
+
[quickstart](https://malcolmwardlaw.github.io/edgar-itemize/quickstart/) fetches, parses,
|
|
74
|
+
checks and opens the nine in four commands, the first of which fetches them:
|
|
75
|
+
|
|
76
|
+
```
|
|
77
|
+
for k in 10k ex10 ex13; do edgar-itemize fetch --manifest examples/sample_manifest_$k.parquet; done
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
**Get the sections.** Build a manifest from the SEC's index, fetch the files, parse, then
|
|
81
|
+
pull the sections you want into one parquet with their text (the `examples/` scripts are in
|
|
82
|
+
the repository):
|
|
83
|
+
|
|
84
|
+
```
|
|
85
|
+
edgar-itemize manifest --years 2019-2020 --no-only-present --out manifest.parquet
|
|
86
|
+
edgar-itemize fetch --manifest manifest.parquet
|
|
87
|
+
edgar-itemize parse --manifest manifest.parquet --out run_10k --kind 10k --no-text --partition-by year
|
|
88
|
+
python examples/pull_items.py --run run_10k --items "ITEM 1A,ITEM 7" --out items.parquet
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
For credit agreements, parse EX-10 exhibits with `--kind ex10` (the manifest then needs a
|
|
92
|
+
`sequence` column naming the exhibit's document) and select the Articles whose
|
|
93
|
+
title names the covenants from the `nodes` table (their numbers vary by agreement); each
|
|
94
|
+
Article's span includes all of its clauses. The details are under
|
|
95
|
+
[From nothing to a parsed corpus](#from-nothing-to-a-parsed-corpus) below.
|
|
96
|
+
|
|
97
|
+
**Check a result or a paper's claim.** Confirm that your install reproduces the release, then
|
|
98
|
+
compare a full run of your own with the per-corpus hash parquet published with the release:
|
|
99
|
+
|
|
100
|
+
```
|
|
101
|
+
edgar-itemize verify --data-root /path/to/edgar
|
|
102
|
+
python examples/check_hashes.py --run run_10k --expected hashes_1.0.0_10k.parquet
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
**Help.** Report a wrong parse as an issue, following
|
|
106
|
+
[CONTRIBUTING.md](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/CONTRIBUTING.md):
|
|
107
|
+
the accession number, the document sequence, the edgar-itemize version, and the byte offsets
|
|
108
|
+
of what the parser produced and what you expected. A set of verdicts from your own review
|
|
109
|
+
is welcome the same way, as an issue or a pull request carrying those four fields per row; a
|
|
110
|
+
file format for exchanging verdicts is planned for release 1.0.1.
|
|
111
|
+
|
|
112
|
+
A live demo is at <https://malcolmwardlaw.github.io/edgar-itemize/demo/>, and the manual,
|
|
113
|
+
the long form of everything below, at <https://malcolmwardlaw.github.io/edgar-itemize/>.
|
|
114
|
+
|
|
115
|
+
## Install
|
|
116
|
+
|
|
117
|
+
From PyPI (Python 3.11 or later):
|
|
118
|
+
|
|
119
|
+
```
|
|
120
|
+
pip install edgar-itemize
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
As a container, the frozen reference artifact (digest-pinned base image, `uv` and every
|
|
124
|
+
dependency from the lockfile). Pull it by the digest given in the release notes of the
|
|
125
|
+
version you cite:
|
|
126
|
+
|
|
127
|
+
```
|
|
128
|
+
docker pull ghcr.io/malcolmwardlaw/edgar-itemize:<version>
|
|
129
|
+
docker run --rm -v /path/to/edgar:/data:ro ghcr.io/malcolmwardlaw/edgar-itemize@sha256:<digest> verify --data-root /data
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
From a clone, with [uv](https://docs.astral.sh/uv/) (prefix the commands below with
|
|
133
|
+
`uv run`):
|
|
134
|
+
|
|
135
|
+
```
|
|
136
|
+
git clone https://github.com/MalcolmWardlaw/edgar-itemize
|
|
137
|
+
cd edgar-itemize
|
|
138
|
+
uv sync --all-extras
|
|
139
|
+
uv run pytest -q
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
The parser needs only `regex`, `pyarrow` and `pyyaml`; HTML is read with a vendored copy of
|
|
143
|
+
the standard library's `html.parser`, which reports exact source offsets. The browser viewer
|
|
144
|
+
and the LLM judge are optional extras (`pip install 'edgar-itemize[viewer,judge]'`).
|
|
145
|
+
|
|
146
|
+
## From nothing to a parsed corpus
|
|
147
|
+
|
|
148
|
+
The parser reads raw full-submission files from a mirror laid out as the SEC serves them.
|
|
149
|
+
Three commands build that mirror from the SEC's public archive and parse it:
|
|
150
|
+
|
|
151
|
+
```
|
|
152
|
+
export EDGAR_ITEMIZE_DATA_ROOT=/data/edgar # the mirror root; no default
|
|
153
|
+
export EDGAR_ITEMIZE_USER_AGENT="Jane Doe jane@example.org" # required by the SEC
|
|
154
|
+
|
|
155
|
+
# 1. a manifest from the public quarterly index (master.idx, cached under the data root)
|
|
156
|
+
edgar-itemize manifest --years 2019-2020 --no-only-present --out manifest_10k_2019_2020.parquet
|
|
157
|
+
|
|
158
|
+
# 2. the submission files it names (only those you do not already have)
|
|
159
|
+
edgar-itemize fetch --manifest manifest_10k_2019_2020.parquet --dry-run # counts, no network
|
|
160
|
+
edgar-itemize fetch --manifest manifest_10k_2019_2020.parquet
|
|
161
|
+
|
|
162
|
+
# 3. parse
|
|
163
|
+
edgar-itemize parse --manifest manifest_10k_2019_2020.parquet --out run_10k --kind 10k \
|
|
164
|
+
--workers 8 --no-text --partition-by year --progress
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
* `manifest` keeps the 10-K family by default (`--forms` for others; `10-Q-family` expands
|
|
168
|
+
to the 10-Q forms), drops `/A` amendments unless `--include-amendments`, and takes
|
|
169
|
+
*filing* years. `--no-only-present` is for a manifest that feeds `fetch`; by default only
|
|
170
|
+
rows whose file is already in the mirror are kept.
|
|
171
|
+
* `fetch` skips files already present, writes each file atomically, and logs the SHA-256 of
|
|
172
|
+
every file it saved to `$EDGAR_ITEMIZE_DATA_ROOT/fetch_log.jsonl`. Both commands follow the
|
|
173
|
+
SEC's fair-access rules (a declared `User-Agent`, at most ten requests per second, backoff
|
|
174
|
+
on HTTP 429 and 503). The full 10-K corpus is about 230,000 files and several hundred GB;
|
|
175
|
+
if you already hold a mirror, skip `fetch`.
|
|
176
|
+
* `parse --kind` is `10k` (the primary document of a 10-K or 10-Q submission), `ex10`,
|
|
177
|
+
`ex13` or `text` (a bare text file with no SGML envelope). The output does not depend on
|
|
178
|
+
`--workers` or on partitioning, and a partitioned run is resumable.
|
|
179
|
+
|
|
180
|
+
Mirror layout:
|
|
181
|
+
|
|
182
|
+
```
|
|
183
|
+
$EDGAR_ITEMIZE_DATA_ROOT/
|
|
184
|
+
archives/edgar/data/<CIK>/<accession>.txt the full-submission files, read as latin-1
|
|
185
|
+
full-index/<year>/QTR<n>/master.idx the cached quarterly indexes (manifest)
|
|
186
|
+
fetch_log.jsonl what fetch saved, with SHA-256
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
One filing at a time, with the scored candidates (`-c`) and the rejected ones with their
|
|
190
|
+
reasons (`-r`); a file path needs no data root:
|
|
191
|
+
|
|
192
|
+
```
|
|
193
|
+
edgar-itemize show /data/edgar/archives/edgar/data/320193/0000320193-20-000096.txt -c -r
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
## Output
|
|
197
|
+
|
|
198
|
+
A run writes three parquet tables per partition: `nodes` (one row per heading in the tree:
|
|
199
|
+
label, title, byte span, ordinal path, confidence, rule ids), `documents` (one row per
|
|
200
|
+
parsed document: era and publisher profile, counts, items found, the input file's SHA-256)
|
|
201
|
+
and `rejected` (every heading candidate that was dropped, with its reason). Files are read
|
|
202
|
+
as latin-1 so that one character is one byte; slicing the raw file at a node's
|
|
203
|
+
`raw_start:raw_end` gives exactly its span. Every column, every closed vocabulary and what
|
|
204
|
+
the version promises about each is in the
|
|
205
|
+
[output contract](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/docs/OUTPUT_CONTRACT.md).
|
|
206
|
+
|
|
207
|
+
## Reproducibility
|
|
208
|
+
|
|
209
|
+
> Two runs of the same release of edgar-itemize on input files with the same `input_sha256`
|
|
210
|
+
> produce the same `nodes` and `rejected` rows and the same `documents` rows, regardless of
|
|
211
|
+
> machine, worker count or partitioning.
|
|
212
|
+
|
|
213
|
+
Each release ships a conformance set of about 2,000 documents with their input and output
|
|
214
|
+
hashes (the inputs are never redistributed; the manifest resolves against your mirror).
|
|
215
|
+
Check an install with:
|
|
216
|
+
|
|
217
|
+
```
|
|
218
|
+
edgar-itemize verify --data-root /path/to/edgar
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
It prints one line with the input and output hash counts and exits 0 only when every
|
|
222
|
+
present input matches and every output hash equals the shipped one; quote that line in a
|
|
223
|
+
replication package. A tagged release is never changed; a wrong output is recorded in
|
|
224
|
+
[ERRATA.md](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/ERRATA.md) against the
|
|
225
|
+
version and fixed in the next minor. The policy is
|
|
226
|
+
[docs/VERSIONING.md](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/docs/VERSIONING.md).
|
|
227
|
+
|
|
228
|
+
## Measured accuracy
|
|
229
|
+
|
|
230
|
+
At the Turn 13 close, whose tree behaviour 1.0.0 carries, on the fixed 1,000-window audit
|
|
231
|
+
bank (`runs/judge/turn13-audit-rescore.txt`):
|
|
232
|
+
|
|
233
|
+
| measure | value |
|
|
234
|
+
|---|---|
|
|
235
|
+
| accepted precision | 94.4% |
|
|
236
|
+
| top-level precision | 98.2% |
|
|
237
|
+
| recall gap | 11.0% |
|
|
238
|
+
|
|
239
|
+
Agreement on 10-K Item boundaries with other open tools
|
|
240
|
+
(`runs/judge/turn13-external-before-after.txt`): datamule 91.6%; EDGAR-CORPUS 90.2%
|
|
241
|
+
(fiscal years 1993-2000) and 90.3% (2001-2020).
|
|
242
|
+
|
|
243
|
+
The `runs/...` names are the lab's identifiers for the stored artifacts the numbers were
|
|
244
|
+
read from; the full evaluation record behind them ships with release 1.0.1. How the
|
|
245
|
+
numbers were measured is in the manual's
|
|
246
|
+
[evaluation](https://malcolmwardlaw.github.io/edgar-itemize/evaluation/) page.
|
|
247
|
+
|
|
248
|
+
## Out of scope
|
|
249
|
+
|
|
250
|
+
* XBRL financial data and tables: the parser recovers headings, not figures.
|
|
251
|
+
* Text cleaning: the output is spans into the raw file. `examples/pull_items.py` slices
|
|
252
|
+
them as filed, HTML tags and all; stripping markup and cleaning the text is left to the
|
|
253
|
+
user.
|
|
254
|
+
* Machine learning: none inside the parser, and no randomness.
|
|
255
|
+
|
|
256
|
+
## Viewer and judge
|
|
257
|
+
|
|
258
|
+
`edgar-itemize serve` (needs the `viewer` extra) shows a filing's text with the agenda
|
|
259
|
+
structure as colour bands, with the candidates and rejections; it reads its bank and run
|
|
260
|
+
registries from the directory named by `EDGAR_ITEMIZE_VIEWER_CONFIG`, and has none by default.
|
|
261
|
+
`edgar-itemize judge` (the `judge` extra) asks a language model about a filing's heading
|
|
262
|
+
candidates; it is evaluation tooling only and never part of the parser.
|
|
263
|
+
|
|
264
|
+
## Citing
|
|
265
|
+
|
|
266
|
+
Name the version you ran. The citation metadata is in
|
|
267
|
+
[CITATION.cff](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/CITATION.cff);
|
|
268
|
+
the Zenodo DOI is `<Zenodo concept DOI, added at the 1.0.0 release>`. See the manual's
|
|
269
|
+
[citing](https://malcolmwardlaw.github.io/edgar-itemize/citing/) page.
|
|
270
|
+
|
|
271
|
+
## License
|
|
272
|
+
|
|
273
|
+
MIT.
|
|
274
|
+
|
|
275
|
+
### Third-party code
|
|
276
|
+
|
|
277
|
+
`src/edgar_itemize/_html_parser.py` is a verbatim copy of CPython 3.11.15's
|
|
278
|
+
`Lib/html/parser.py` (`html.parser`), copyright the Python Software Foundation and
|
|
279
|
+
licensed under the Python Software Foundation License Version 2 (`LICENSES/PSF-2.0.txt`).
|
|
280
|
+
It is vendored so that the tokenisation of malformed HTML does not change with the
|
|
281
|
+
interpreter's patch level (`docs/VERSIONING.md` section 5).
|
|
@@ -0,0 +1,254 @@
|
|
|
1
|
+
# edgar-itemize
|
|
2
|
+
|
|
3
|
+
edgar-itemize turns the raw filings in the SEC's EDGAR archive into a table of their
|
|
4
|
+
headings. For each filing it recovers the agenda the document is written to: Part and Item
|
|
5
|
+
in a 10-K or 10-Q, sub-headings inside an Item, and Article, Section and the lettered and
|
|
6
|
+
numbered clauses of a credit agreement filed as an EX-10 exhibit. Each heading is one row
|
|
7
|
+
that records where its section starts and ends as byte offsets into the submission file
|
|
8
|
+
exactly as the SEC serves it, so the text of any section, from Item 1A down to a single
|
|
9
|
+
covenant clause, is one slice of a file you already have. It reads the whole EDGAR era,
|
|
10
|
+
from 1993 plain text through publisher HTML to inline XBRL.
|
|
11
|
+
|
|
12
|
+
* **Manual** <https://malcolmwardlaw.github.io/edgar-itemize/>
|
|
13
|
+
* **Output Demo** <https://malcolmwardlaw.github.io/edgar-itemize/demo/
|
|
14
|
+
|
|
15
|
+
## Why this exists
|
|
16
|
+
|
|
17
|
+
* **The whole agenda, down to the clauses.** Most tools pull out a few top-level sections.
|
|
18
|
+
10-Ks, and the contracts attached as EX-10 exhibits even more, have deep agenda
|
|
19
|
+
structures: a study of covenants needs the affirmative and negative covenant Articles and
|
|
20
|
+
each clause one or two levels beneath them. The tree goes to that depth.
|
|
21
|
+
* **Your corpus, downloaded once, parsed locally.** You pull as much or as little of EDGAR
|
|
22
|
+
as you want, up to the whole archive, and parse it on your own machine. Rate-limited SEC
|
|
23
|
+
downloads are slow but happen once; storage is cheap; re-parsing with different choices
|
|
24
|
+
costs nothing but compute.
|
|
25
|
+
* **Deterministic.** The parser is a fixed set of written rules, not a language model
|
|
26
|
+
deciding over a corpus. The same release on the same file gives the same rows every time,
|
|
27
|
+
every heading names the rules that produced it, and every candidate heading that was
|
|
28
|
+
dropped is written out with the reason.
|
|
29
|
+
* **Replicable by hash, with the data untouched.** Each tagged release gives exactly the
|
|
30
|
+
same results on exactly the same EDGAR files, checked by SHA-256 of the input and the
|
|
31
|
+
output. The filings themselves are never modified or redistributed, so nobody has to
|
|
32
|
+
archive gigabytes of derived text as validation, and every value traces directly to a
|
|
33
|
+
byte range in a public file.
|
|
34
|
+
* **Improves by frozen tagged releases.** The output is evaluated by human review and by
|
|
35
|
+
language-model judges, and the evaluation feeds hand-written rule changes. Current results
|
|
36
|
+
are good, with room left (see [Measured accuracy](#measured-accuracy)). Each improvement
|
|
37
|
+
ships as a new tagged release that keeps the two guarantees above; earlier releases stay
|
|
38
|
+
as they were.
|
|
39
|
+
* **Public and updatable.** The author keeps working on it. Anyone who finds a set of
|
|
40
|
+
errors can package them and send them in, and they are folded into the next release.
|
|
41
|
+
|
|
42
|
+
## What to do with it
|
|
43
|
+
|
|
44
|
+
**Start small.** `examples/` holds manifests and expected hashes for the nine filings of the
|
|
45
|
+
demo and for ten credit agreements; the
|
|
46
|
+
[quickstart](https://malcolmwardlaw.github.io/edgar-itemize/quickstart/) fetches, parses,
|
|
47
|
+
checks and opens the nine in four commands, the first of which fetches them:
|
|
48
|
+
|
|
49
|
+
```
|
|
50
|
+
for k in 10k ex10 ex13; do edgar-itemize fetch --manifest examples/sample_manifest_$k.parquet; done
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
**Get the sections.** Build a manifest from the SEC's index, fetch the files, parse, then
|
|
54
|
+
pull the sections you want into one parquet with their text (the `examples/` scripts are in
|
|
55
|
+
the repository):
|
|
56
|
+
|
|
57
|
+
```
|
|
58
|
+
edgar-itemize manifest --years 2019-2020 --no-only-present --out manifest.parquet
|
|
59
|
+
edgar-itemize fetch --manifest manifest.parquet
|
|
60
|
+
edgar-itemize parse --manifest manifest.parquet --out run_10k --kind 10k --no-text --partition-by year
|
|
61
|
+
python examples/pull_items.py --run run_10k --items "ITEM 1A,ITEM 7" --out items.parquet
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
For credit agreements, parse EX-10 exhibits with `--kind ex10` (the manifest then needs a
|
|
65
|
+
`sequence` column naming the exhibit's document) and select the Articles whose
|
|
66
|
+
title names the covenants from the `nodes` table (their numbers vary by agreement); each
|
|
67
|
+
Article's span includes all of its clauses. The details are under
|
|
68
|
+
[From nothing to a parsed corpus](#from-nothing-to-a-parsed-corpus) below.
|
|
69
|
+
|
|
70
|
+
**Check a result or a paper's claim.** Confirm that your install reproduces the release, then
|
|
71
|
+
compare a full run of your own with the per-corpus hash parquet published with the release:
|
|
72
|
+
|
|
73
|
+
```
|
|
74
|
+
edgar-itemize verify --data-root /path/to/edgar
|
|
75
|
+
python examples/check_hashes.py --run run_10k --expected hashes_1.0.0_10k.parquet
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
**Help.** Report a wrong parse as an issue, following
|
|
79
|
+
[CONTRIBUTING.md](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/CONTRIBUTING.md):
|
|
80
|
+
the accession number, the document sequence, the edgar-itemize version, and the byte offsets
|
|
81
|
+
of what the parser produced and what you expected. A set of verdicts from your own review
|
|
82
|
+
is welcome the same way, as an issue or a pull request carrying those four fields per row; a
|
|
83
|
+
file format for exchanging verdicts is planned for release 1.0.1.
|
|
84
|
+
|
|
85
|
+
A live demo is at <https://malcolmwardlaw.github.io/edgar-itemize/demo/>, and the manual,
|
|
86
|
+
the long form of everything below, at <https://malcolmwardlaw.github.io/edgar-itemize/>.
|
|
87
|
+
|
|
88
|
+
## Install
|
|
89
|
+
|
|
90
|
+
From PyPI (Python 3.11 or later):
|
|
91
|
+
|
|
92
|
+
```
|
|
93
|
+
pip install edgar-itemize
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
As a container, the frozen reference artifact (digest-pinned base image, `uv` and every
|
|
97
|
+
dependency from the lockfile). Pull it by the digest given in the release notes of the
|
|
98
|
+
version you cite:
|
|
99
|
+
|
|
100
|
+
```
|
|
101
|
+
docker pull ghcr.io/malcolmwardlaw/edgar-itemize:<version>
|
|
102
|
+
docker run --rm -v /path/to/edgar:/data:ro ghcr.io/malcolmwardlaw/edgar-itemize@sha256:<digest> verify --data-root /data
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
From a clone, with [uv](https://docs.astral.sh/uv/) (prefix the commands below with
|
|
106
|
+
`uv run`):
|
|
107
|
+
|
|
108
|
+
```
|
|
109
|
+
git clone https://github.com/MalcolmWardlaw/edgar-itemize
|
|
110
|
+
cd edgar-itemize
|
|
111
|
+
uv sync --all-extras
|
|
112
|
+
uv run pytest -q
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
The parser needs only `regex`, `pyarrow` and `pyyaml`; HTML is read with a vendored copy of
|
|
116
|
+
the standard library's `html.parser`, which reports exact source offsets. The browser viewer
|
|
117
|
+
and the LLM judge are optional extras (`pip install 'edgar-itemize[viewer,judge]'`).
|
|
118
|
+
|
|
119
|
+
## From nothing to a parsed corpus
|
|
120
|
+
|
|
121
|
+
The parser reads raw full-submission files from a mirror laid out as the SEC serves them.
|
|
122
|
+
Three commands build that mirror from the SEC's public archive and parse it:
|
|
123
|
+
|
|
124
|
+
```
|
|
125
|
+
export EDGAR_ITEMIZE_DATA_ROOT=/data/edgar # the mirror root; no default
|
|
126
|
+
export EDGAR_ITEMIZE_USER_AGENT="Jane Doe jane@example.org" # required by the SEC
|
|
127
|
+
|
|
128
|
+
# 1. a manifest from the public quarterly index (master.idx, cached under the data root)
|
|
129
|
+
edgar-itemize manifest --years 2019-2020 --no-only-present --out manifest_10k_2019_2020.parquet
|
|
130
|
+
|
|
131
|
+
# 2. the submission files it names (only those you do not already have)
|
|
132
|
+
edgar-itemize fetch --manifest manifest_10k_2019_2020.parquet --dry-run # counts, no network
|
|
133
|
+
edgar-itemize fetch --manifest manifest_10k_2019_2020.parquet
|
|
134
|
+
|
|
135
|
+
# 3. parse
|
|
136
|
+
edgar-itemize parse --manifest manifest_10k_2019_2020.parquet --out run_10k --kind 10k \
|
|
137
|
+
--workers 8 --no-text --partition-by year --progress
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
* `manifest` keeps the 10-K family by default (`--forms` for others; `10-Q-family` expands
|
|
141
|
+
to the 10-Q forms), drops `/A` amendments unless `--include-amendments`, and takes
|
|
142
|
+
*filing* years. `--no-only-present` is for a manifest that feeds `fetch`; by default only
|
|
143
|
+
rows whose file is already in the mirror are kept.
|
|
144
|
+
* `fetch` skips files already present, writes each file atomically, and logs the SHA-256 of
|
|
145
|
+
every file it saved to `$EDGAR_ITEMIZE_DATA_ROOT/fetch_log.jsonl`. Both commands follow the
|
|
146
|
+
SEC's fair-access rules (a declared `User-Agent`, at most ten requests per second, backoff
|
|
147
|
+
on HTTP 429 and 503). The full 10-K corpus is about 230,000 files and several hundred GB;
|
|
148
|
+
if you already hold a mirror, skip `fetch`.
|
|
149
|
+
* `parse --kind` is `10k` (the primary document of a 10-K or 10-Q submission), `ex10`,
|
|
150
|
+
`ex13` or `text` (a bare text file with no SGML envelope). The output does not depend on
|
|
151
|
+
`--workers` or on partitioning, and a partitioned run is resumable.
|
|
152
|
+
|
|
153
|
+
Mirror layout:
|
|
154
|
+
|
|
155
|
+
```
|
|
156
|
+
$EDGAR_ITEMIZE_DATA_ROOT/
|
|
157
|
+
archives/edgar/data/<CIK>/<accession>.txt the full-submission files, read as latin-1
|
|
158
|
+
full-index/<year>/QTR<n>/master.idx the cached quarterly indexes (manifest)
|
|
159
|
+
fetch_log.jsonl what fetch saved, with SHA-256
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
One filing at a time, with the scored candidates (`-c`) and the rejected ones with their
|
|
163
|
+
reasons (`-r`); a file path needs no data root:
|
|
164
|
+
|
|
165
|
+
```
|
|
166
|
+
edgar-itemize show /data/edgar/archives/edgar/data/320193/0000320193-20-000096.txt -c -r
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
## Output
|
|
170
|
+
|
|
171
|
+
A run writes three parquet tables per partition: `nodes` (one row per heading in the tree:
|
|
172
|
+
label, title, byte span, ordinal path, confidence, rule ids), `documents` (one row per
|
|
173
|
+
parsed document: era and publisher profile, counts, items found, the input file's SHA-256)
|
|
174
|
+
and `rejected` (every heading candidate that was dropped, with its reason). Files are read
|
|
175
|
+
as latin-1 so that one character is one byte; slicing the raw file at a node's
|
|
176
|
+
`raw_start:raw_end` gives exactly its span. Every column, every closed vocabulary and what
|
|
177
|
+
the version promises about each is in the
|
|
178
|
+
[output contract](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/docs/OUTPUT_CONTRACT.md).
|
|
179
|
+
|
|
180
|
+
## Reproducibility
|
|
181
|
+
|
|
182
|
+
> Two runs of the same release of edgar-itemize on input files with the same `input_sha256`
|
|
183
|
+
> produce the same `nodes` and `rejected` rows and the same `documents` rows, regardless of
|
|
184
|
+
> machine, worker count or partitioning.
|
|
185
|
+
|
|
186
|
+
Each release ships a conformance set of about 2,000 documents with their input and output
|
|
187
|
+
hashes (the inputs are never redistributed; the manifest resolves against your mirror).
|
|
188
|
+
Check an install with:
|
|
189
|
+
|
|
190
|
+
```
|
|
191
|
+
edgar-itemize verify --data-root /path/to/edgar
|
|
192
|
+
```
|
|
193
|
+
|
|
194
|
+
It prints one line with the input and output hash counts and exits 0 only when every
|
|
195
|
+
present input matches and every output hash equals the shipped one; quote that line in a
|
|
196
|
+
replication package. A tagged release is never changed; a wrong output is recorded in
|
|
197
|
+
[ERRATA.md](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/ERRATA.md) against the
|
|
198
|
+
version and fixed in the next minor. The policy is
|
|
199
|
+
[docs/VERSIONING.md](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/docs/VERSIONING.md).
|
|
200
|
+
|
|
201
|
+
## Measured accuracy
|
|
202
|
+
|
|
203
|
+
At the Turn 13 close, whose tree behaviour 1.0.0 carries, on the fixed 1,000-window audit
|
|
204
|
+
bank (`runs/judge/turn13-audit-rescore.txt`):
|
|
205
|
+
|
|
206
|
+
| measure | value |
|
|
207
|
+
|---|---|
|
|
208
|
+
| accepted precision | 94.4% |
|
|
209
|
+
| top-level precision | 98.2% |
|
|
210
|
+
| recall gap | 11.0% |
|
|
211
|
+
|
|
212
|
+
Agreement on 10-K Item boundaries with other open tools
|
|
213
|
+
(`runs/judge/turn13-external-before-after.txt`): datamule 91.6%; EDGAR-CORPUS 90.2%
|
|
214
|
+
(fiscal years 1993-2000) and 90.3% (2001-2020).
|
|
215
|
+
|
|
216
|
+
The `runs/...` names are the lab's identifiers for the stored artifacts the numbers were
|
|
217
|
+
read from; the full evaluation record behind them ships with release 1.0.1. How the
|
|
218
|
+
numbers were measured is in the manual's
|
|
219
|
+
[evaluation](https://malcolmwardlaw.github.io/edgar-itemize/evaluation/) page.
|
|
220
|
+
|
|
221
|
+
## Out of scope
|
|
222
|
+
|
|
223
|
+
* XBRL financial data and tables: the parser recovers headings, not figures.
|
|
224
|
+
* Text cleaning: the output is spans into the raw file. `examples/pull_items.py` slices
|
|
225
|
+
them as filed, HTML tags and all; stripping markup and cleaning the text is left to the
|
|
226
|
+
user.
|
|
227
|
+
* Machine learning: none inside the parser, and no randomness.
|
|
228
|
+
|
|
229
|
+
## Viewer and judge
|
|
230
|
+
|
|
231
|
+
`edgar-itemize serve` (needs the `viewer` extra) shows a filing's text with the agenda
|
|
232
|
+
structure as colour bands, with the candidates and rejections; it reads its bank and run
|
|
233
|
+
registries from the directory named by `EDGAR_ITEMIZE_VIEWER_CONFIG`, and has none by default.
|
|
234
|
+
`edgar-itemize judge` (the `judge` extra) asks a language model about a filing's heading
|
|
235
|
+
candidates; it is evaluation tooling only and never part of the parser.
|
|
236
|
+
|
|
237
|
+
## Citing
|
|
238
|
+
|
|
239
|
+
Name the version you ran. The citation metadata is in
|
|
240
|
+
[CITATION.cff](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/CITATION.cff);
|
|
241
|
+
the Zenodo DOI is `<Zenodo concept DOI, added at the 1.0.0 release>`. See the manual's
|
|
242
|
+
[citing](https://malcolmwardlaw.github.io/edgar-itemize/citing/) page.
|
|
243
|
+
|
|
244
|
+
## License
|
|
245
|
+
|
|
246
|
+
MIT.
|
|
247
|
+
|
|
248
|
+
### Third-party code
|
|
249
|
+
|
|
250
|
+
`src/edgar_itemize/_html_parser.py` is a verbatim copy of CPython 3.11.15's
|
|
251
|
+
`Lib/html/parser.py` (`html.parser`), copyright the Python Software Foundation and
|
|
252
|
+
licensed under the Python Software Foundation License Version 2 (`LICENSES/PSF-2.0.txt`).
|
|
253
|
+
It is vendored so that the tokenisation of malformed HTML does not change with the
|
|
254
|
+
interpreter's patch level (`docs/VERSIONING.md` section 5).
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "edgar-itemize"
|
|
3
|
+
version = "1.0.0rc2"
|
|
4
|
+
description = "Deterministic agenda-structure (Part/Item/Section) parser for SEC EDGAR filings"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = "MIT"
|
|
7
|
+
license-files = [
|
|
8
|
+
"LICENSE",
|
|
9
|
+
"LICENSES/PSF-2.0.txt",
|
|
10
|
+
]
|
|
11
|
+
requires-python = ">=3.11"
|
|
12
|
+
dependencies = [
|
|
13
|
+
"regex>=2024.0",
|
|
14
|
+
"pyarrow>=14",
|
|
15
|
+
"pyyaml>=6",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
[[project.authors]]
|
|
19
|
+
name = "Malcolm Wardlaw"
|
|
20
|
+
email = "malcolm.wardlaw@uga.edu"
|
|
21
|
+
|
|
22
|
+
[project.optional-dependencies]
|
|
23
|
+
viewer = [
|
|
24
|
+
"fastapi>=0.141.1",
|
|
25
|
+
"uvicorn[standard]>=0.52.4",
|
|
26
|
+
]
|
|
27
|
+
judge = ["httpx>=0.28.1"]
|
|
28
|
+
polars = ["polars>=1.0"]
|
|
29
|
+
dev = [
|
|
30
|
+
"pytest>=8",
|
|
31
|
+
"polars>=1.0",
|
|
32
|
+
]
|
|
33
|
+
docs = ["mkdocs-material>=9.5"]
|
|
34
|
+
|
|
35
|
+
[project.scripts]
|
|
36
|
+
edgar-itemize = "edgar_itemize.cli:main"
|
|
37
|
+
|
|
38
|
+
[dependency-groups]
|
|
39
|
+
dev = [
|
|
40
|
+
"polars>=1.0",
|
|
41
|
+
"pytest>=8",
|
|
42
|
+
]
|
|
43
|
+
|
|
44
|
+
[build-system]
|
|
45
|
+
requires = ["uv_build>=0.10.8,<0.11.0"]
|
|
46
|
+
build-backend = "uv_build"
|
|
47
|
+
|
|
48
|
+
[tool.pytest.ini_options]
|
|
49
|
+
testpaths = [
|
|
50
|
+
"tests",
|
|
51
|
+
"tests_lab",
|
|
52
|
+
]
|