edgar-itemize 1.0.0rc2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. edgar_itemize-1.0.0rc2/LICENSE +30 -0
  2. edgar_itemize-1.0.0rc2/LICENSES/PSF-2.0.txt +48 -0
  3. edgar_itemize-1.0.0rc2/PKG-INFO +281 -0
  4. edgar_itemize-1.0.0rc2/README.md +254 -0
  5. edgar_itemize-1.0.0rc2/pyproject.toml +52 -0
  6. edgar_itemize-1.0.0rc2/pyproject.toml.orig +43 -0
  7. edgar_itemize-1.0.0rc2/src/edgar_itemize/__init__.py +30 -0
  8. edgar_itemize-1.0.0rc2/src/edgar_itemize/_html_parser.py +559 -0
  9. edgar_itemize-1.0.0rc2/src/edgar_itemize/agenda.py +934 -0
  10. edgar_itemize-1.0.0rc2/src/edgar_itemize/ars_kind.py +133 -0
  11. edgar_itemize-1.0.0rc2/src/edgar_itemize/blocks.py +58 -0
  12. edgar_itemize-1.0.0rc2/src/edgar_itemize/candidates.py +1872 -0
  13. edgar_itemize-1.0.0rc2/src/edgar_itemize/classify.py +183 -0
  14. edgar_itemize-1.0.0rc2/src/edgar_itemize/cli.py +359 -0
  15. edgar_itemize-1.0.0rc2/src/edgar_itemize/conformance.py +670 -0
  16. edgar_itemize-1.0.0rc2/src/edgar_itemize/control.py +166 -0
  17. edgar_itemize-1.0.0rc2/src/edgar_itemize/eval/__init__.py +0 -0
  18. edgar_itemize-1.0.0rc2/src/edgar_itemize/eval/gold.py +54 -0
  19. edgar_itemize-1.0.0rc2/src/edgar_itemize/eval/judge.py +85 -0
  20. edgar_itemize-1.0.0rc2/src/edgar_itemize/eval/metrics.py +171 -0
  21. edgar_itemize-1.0.0rc2/src/edgar_itemize/fetch.py +227 -0
  22. edgar_itemize-1.0.0rc2/src/edgar_itemize/grammar/__init__.py +2 -0
  23. edgar_itemize-1.0.0rc2/src/edgar_itemize/grammar/base.py +31 -0
  24. edgar_itemize-1.0.0rc2/src/edgar_itemize/grammar/contract.py +160 -0
  25. edgar_itemize-1.0.0rc2/src/edgar_itemize/grammar/form10k.py +316 -0
  26. edgar_itemize-1.0.0rc2/src/edgar_itemize/grammar/form10q.py +322 -0
  27. edgar_itemize-1.0.0rc2/src/edgar_itemize/headings.py +282 -0
  28. edgar_itemize-1.0.0rc2/src/edgar_itemize/llm.py +389 -0
  29. edgar_itemize-1.0.0rc2/src/edgar_itemize/manifest.py +290 -0
  30. edgar_itemize-1.0.0rc2/src/edgar_itemize/normalize_html.py +593 -0
  31. edgar_itemize-1.0.0rc2/src/edgar_itemize/normalize_text.py +172 -0
  32. edgar_itemize-1.0.0rc2/src/edgar_itemize/normcache.py +212 -0
  33. edgar_itemize-1.0.0rc2/src/edgar_itemize/offsets.py +43 -0
  34. edgar_itemize-1.0.0rc2/src/edgar_itemize/pipeline.py +298 -0
  35. edgar_itemize-1.0.0rc2/src/edgar_itemize/prepare.py +81 -0
  36. edgar_itemize-1.0.0rc2/src/edgar_itemize/schema.py +105 -0
  37. edgar_itemize-1.0.0rc2/src/edgar_itemize/select.py +76 -0
  38. edgar_itemize-1.0.0rc2/src/edgar_itemize/sequence.py +180 -0
  39. edgar_itemize-1.0.0rc2/src/edgar_itemize/sgml.py +234 -0
  40. edgar_itemize-1.0.0rc2/src/edgar_itemize/toc.py +373 -0
  41. edgar_itemize-1.0.0rc2/src/edgar_itemize/tree.py +964 -0
  42. edgar_itemize-1.0.0rc2/src/edgar_itemize/tree_contract.py +757 -0
  43. edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/__init__.py +0 -0
  44. edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/app.py +743 -0
  45. edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/original.py +123 -0
  46. edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/runread.py +256 -0
  47. edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/sets.py +427 -0
  48. edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/static/index.html +1424 -0
  49. edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/static/turns.html +501 -0
  50. edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/turns.py +60 -0
  51. edgar_itemize-1.0.0rc2/src/edgar_itemize/viewer/windex.py +194 -0
  52. edgar_itemize-1.0.0rc2/src/edgar_itemize/writer.py +18 -0
@@ -0,0 +1,30 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Malcolm Wardlaw
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
22
+
23
+ ---
24
+
25
+ Third-party code
26
+
27
+ src/edgar_itemize/_html_parser.py is a verbatim copy of CPython 3.11.15's
28
+ Lib/html/parser.py (the html.parser module), Copyright (c) 2001-2023 Python
29
+ Software Foundation; All Rights Reserved, and is distributed under the Python
30
+ Software Foundation License Version 2, reproduced in LICENSES/PSF-2.0.txt.
@@ -0,0 +1,48 @@
1
+ PYTHON SOFTWARE FOUNDATION LICENSE VERSION 2
2
+ --------------------------------------------
3
+
4
+ 1. This LICENSE AGREEMENT is between the Python Software Foundation
5
+ ("PSF"), and the Individual or Organization ("Licensee") accessing and
6
+ otherwise using this software ("Python") in source or binary form and
7
+ its associated documentation.
8
+
9
+ 2. Subject to the terms and conditions of this License Agreement, PSF hereby
10
+ grants Licensee a nonexclusive, royalty-free, world-wide license to reproduce,
11
+ analyze, test, perform and/or display publicly, prepare derivative works,
12
+ distribute, and otherwise use Python alone or in any derivative version,
13
+ provided, however, that PSF's License Agreement and PSF's notice of copyright,
14
+ i.e., "Copyright (c) 2001, 2002, 2003, 2004, 2005, 2006, 2007, 2008, 2009, 2010,
15
+ 2011, 2012, 2013, 2014, 2015, 2016, 2017, 2018, 2019, 2020, 2021, 2022, 2023 Python Software Foundation;
16
+ All Rights Reserved" are retained in Python alone or in any derivative version
17
+ prepared by Licensee.
18
+
19
+ 3. In the event Licensee prepares a derivative work that is based on
20
+ or incorporates Python or any part thereof, and wants to make
21
+ the derivative work available to others as provided herein, then
22
+ Licensee hereby agrees to include in any such work a brief summary of
23
+ the changes made to Python.
24
+
25
+ 4. PSF is making Python available to Licensee on an "AS IS"
26
+ basis. PSF MAKES NO REPRESENTATIONS OR WARRANTIES, EXPRESS OR
27
+ IMPLIED. BY WAY OF EXAMPLE, BUT NOT LIMITATION, PSF MAKES NO AND
28
+ DISCLAIMS ANY REPRESENTATION OR WARRANTY OF MERCHANTABILITY OR FITNESS
29
+ FOR ANY PARTICULAR PURPOSE OR THAT THE USE OF PYTHON WILL NOT
30
+ INFRINGE ANY THIRD PARTY RIGHTS.
31
+
32
+ 5. PSF SHALL NOT BE LIABLE TO LICENSEE OR ANY OTHER USERS OF PYTHON
33
+ FOR ANY INCIDENTAL, SPECIAL, OR CONSEQUENTIAL DAMAGES OR LOSS AS
34
+ A RESULT OF MODIFYING, DISTRIBUTING, OR OTHERWISE USING PYTHON,
35
+ OR ANY DERIVATIVE THEREOF, EVEN IF ADVISED OF THE POSSIBILITY THEREOF.
36
+
37
+ 6. This License Agreement will automatically terminate upon a material
38
+ breach of its terms and conditions.
39
+
40
+ 7. Nothing in this License Agreement shall be deemed to create any
41
+ relationship of agency, partnership, or joint venture between PSF and
42
+ Licensee. This License Agreement does not grant permission to use PSF
43
+ trademarks or trade name in a trademark sense to endorse or promote
44
+ products or services of Licensee, or any third party.
45
+
46
+ 8. By copying, installing or otherwise using Python, Licensee
47
+ agrees to be bound by the terms and conditions of this License
48
+ Agreement.
@@ -0,0 +1,281 @@
1
+ Metadata-Version: 2.4
2
+ Name: edgar-itemize
3
+ Version: 1.0.0rc2
4
+ Summary: Deterministic agenda-structure (Part/Item/Section) parser for SEC EDGAR filings
5
+ Author: Malcolm Wardlaw
6
+ Author-email: Malcolm Wardlaw <malcolm.wardlaw@uga.edu>
7
+ License-Expression: MIT
8
+ License-File: LICENSE
9
+ License-File: LICENSES/PSF-2.0.txt
10
+ Requires-Dist: regex>=2024.0
11
+ Requires-Dist: pyarrow>=14
12
+ Requires-Dist: pyyaml>=6
13
+ Requires-Dist: pytest>=8 ; extra == 'dev'
14
+ Requires-Dist: polars>=1.0 ; extra == 'dev'
15
+ Requires-Dist: mkdocs-material>=9.5 ; extra == 'docs'
16
+ Requires-Dist: httpx>=0.28.1 ; extra == 'judge'
17
+ Requires-Dist: polars>=1.0 ; extra == 'polars'
18
+ Requires-Dist: fastapi>=0.141.1 ; extra == 'viewer'
19
+ Requires-Dist: uvicorn[standard]>=0.52.4 ; extra == 'viewer'
20
+ Requires-Python: >=3.11
21
+ Provides-Extra: dev
22
+ Provides-Extra: docs
23
+ Provides-Extra: judge
24
+ Provides-Extra: polars
25
+ Provides-Extra: viewer
26
+ Description-Content-Type: text/markdown
27
+
28
+ # edgar-itemize
29
+
30
+ edgar-itemize turns the raw filings in the SEC's EDGAR archive into a table of their
31
+ headings. For each filing it recovers the agenda the document is written to: Part and Item
32
+ in a 10-K or 10-Q, sub-headings inside an Item, and Article, Section and the lettered and
33
+ numbered clauses of a credit agreement filed as an EX-10 exhibit. Each heading is one row
34
+ that records where its section starts and ends as byte offsets into the submission file
35
+ exactly as the SEC serves it, so the text of any section, from Item 1A down to a single
36
+ covenant clause, is one slice of a file you already have. It reads the whole EDGAR era,
37
+ from 1993 plain text through publisher HTML to inline XBRL.
38
+
39
+ * **Manual** <https://malcolmwardlaw.github.io/edgar-itemize/>
40
+ * **Output Demo** <https://malcolmwardlaw.github.io/edgar-itemize/demo/
41
+
42
+ ## Why this exists
43
+
44
+ * **The whole agenda, down to the clauses.** Most tools pull out a few top-level sections.
45
+ 10-Ks, and the contracts attached as EX-10 exhibits even more, have deep agenda
46
+ structures: a study of covenants needs the affirmative and negative covenant Articles and
47
+ each clause one or two levels beneath them. The tree goes to that depth.
48
+ * **Your corpus, downloaded once, parsed locally.** You pull as much or as little of EDGAR
49
+ as you want, up to the whole archive, and parse it on your own machine. Rate-limited SEC
50
+ downloads are slow but happen once; storage is cheap; re-parsing with different choices
51
+ costs nothing but compute.
52
+ * **Deterministic.** The parser is a fixed set of written rules, not a language model
53
+ deciding over a corpus. The same release on the same file gives the same rows every time,
54
+ every heading names the rules that produced it, and every candidate heading that was
55
+ dropped is written out with the reason.
56
+ * **Replicable by hash, with the data untouched.** Each tagged release gives exactly the
57
+ same results on exactly the same EDGAR files, checked by SHA-256 of the input and the
58
+ output. The filings themselves are never modified or redistributed, so nobody has to
59
+ archive gigabytes of derived text as validation, and every value traces directly to a
60
+ byte range in a public file.
61
+ * **Improves by frozen tagged releases.** The output is evaluated by human review and by
62
+ language-model judges, and the evaluation feeds hand-written rule changes. Current results
63
+ are good, with room left (see [Measured accuracy](#measured-accuracy)). Each improvement
64
+ ships as a new tagged release that keeps the two guarantees above; earlier releases stay
65
+ as they were.
66
+ * **Public and updatable.** The author keeps working on it. Anyone who finds a set of
67
+ errors can package them and send them in, and they are folded into the next release.
68
+
69
+ ## What to do with it
70
+
71
+ **Start small.** `examples/` holds manifests and expected hashes for the nine filings of the
72
+ demo and for ten credit agreements; the
73
+ [quickstart](https://malcolmwardlaw.github.io/edgar-itemize/quickstart/) fetches, parses,
74
+ checks and opens the nine in four commands, the first of which fetches them:
75
+
76
+ ```
77
+ for k in 10k ex10 ex13; do edgar-itemize fetch --manifest examples/sample_manifest_$k.parquet; done
78
+ ```
79
+
80
+ **Get the sections.** Build a manifest from the SEC's index, fetch the files, parse, then
81
+ pull the sections you want into one parquet with their text (the `examples/` scripts are in
82
+ the repository):
83
+
84
+ ```
85
+ edgar-itemize manifest --years 2019-2020 --no-only-present --out manifest.parquet
86
+ edgar-itemize fetch --manifest manifest.parquet
87
+ edgar-itemize parse --manifest manifest.parquet --out run_10k --kind 10k --no-text --partition-by year
88
+ python examples/pull_items.py --run run_10k --items "ITEM 1A,ITEM 7" --out items.parquet
89
+ ```
90
+
91
+ For credit agreements, parse EX-10 exhibits with `--kind ex10` (the manifest then needs a
92
+ `sequence` column naming the exhibit's document) and select the Articles whose
93
+ title names the covenants from the `nodes` table (their numbers vary by agreement); each
94
+ Article's span includes all of its clauses. The details are under
95
+ [From nothing to a parsed corpus](#from-nothing-to-a-parsed-corpus) below.
96
+
97
+ **Check a result or a paper's claim.** Confirm that your install reproduces the release, then
98
+ compare a full run of your own with the per-corpus hash parquet published with the release:
99
+
100
+ ```
101
+ edgar-itemize verify --data-root /path/to/edgar
102
+ python examples/check_hashes.py --run run_10k --expected hashes_1.0.0_10k.parquet
103
+ ```
104
+
105
+ **Help.** Report a wrong parse as an issue, following
106
+ [CONTRIBUTING.md](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/CONTRIBUTING.md):
107
+ the accession number, the document sequence, the edgar-itemize version, and the byte offsets
108
+ of what the parser produced and what you expected. A set of verdicts from your own review
109
+ is welcome the same way, as an issue or a pull request carrying those four fields per row; a
110
+ file format for exchanging verdicts is planned for release 1.0.1.
111
+
112
+ A live demo is at <https://malcolmwardlaw.github.io/edgar-itemize/demo/>, and the manual,
113
+ the long form of everything below, at <https://malcolmwardlaw.github.io/edgar-itemize/>.
114
+
115
+ ## Install
116
+
117
+ From PyPI (Python 3.11 or later):
118
+
119
+ ```
120
+ pip install edgar-itemize
121
+ ```
122
+
123
+ As a container, the frozen reference artifact (digest-pinned base image, `uv` and every
124
+ dependency from the lockfile). Pull it by the digest given in the release notes of the
125
+ version you cite:
126
+
127
+ ```
128
+ docker pull ghcr.io/malcolmwardlaw/edgar-itemize:<version>
129
+ docker run --rm -v /path/to/edgar:/data:ro ghcr.io/malcolmwardlaw/edgar-itemize@sha256:<digest> verify --data-root /data
130
+ ```
131
+
132
+ From a clone, with [uv](https://docs.astral.sh/uv/) (prefix the commands below with
133
+ `uv run`):
134
+
135
+ ```
136
+ git clone https://github.com/MalcolmWardlaw/edgar-itemize
137
+ cd edgar-itemize
138
+ uv sync --all-extras
139
+ uv run pytest -q
140
+ ```
141
+
142
+ The parser needs only `regex`, `pyarrow` and `pyyaml`; HTML is read with a vendored copy of
143
+ the standard library's `html.parser`, which reports exact source offsets. The browser viewer
144
+ and the LLM judge are optional extras (`pip install 'edgar-itemize[viewer,judge]'`).
145
+
146
+ ## From nothing to a parsed corpus
147
+
148
+ The parser reads raw full-submission files from a mirror laid out as the SEC serves them.
149
+ Three commands build that mirror from the SEC's public archive and parse it:
150
+
151
+ ```
152
+ export EDGAR_ITEMIZE_DATA_ROOT=/data/edgar # the mirror root; no default
153
+ export EDGAR_ITEMIZE_USER_AGENT="Jane Doe jane@example.org" # required by the SEC
154
+
155
+ # 1. a manifest from the public quarterly index (master.idx, cached under the data root)
156
+ edgar-itemize manifest --years 2019-2020 --no-only-present --out manifest_10k_2019_2020.parquet
157
+
158
+ # 2. the submission files it names (only those you do not already have)
159
+ edgar-itemize fetch --manifest manifest_10k_2019_2020.parquet --dry-run # counts, no network
160
+ edgar-itemize fetch --manifest manifest_10k_2019_2020.parquet
161
+
162
+ # 3. parse
163
+ edgar-itemize parse --manifest manifest_10k_2019_2020.parquet --out run_10k --kind 10k \
164
+ --workers 8 --no-text --partition-by year --progress
165
+ ```
166
+
167
+ * `manifest` keeps the 10-K family by default (`--forms` for others; `10-Q-family` expands
168
+ to the 10-Q forms), drops `/A` amendments unless `--include-amendments`, and takes
169
+ *filing* years. `--no-only-present` is for a manifest that feeds `fetch`; by default only
170
+ rows whose file is already in the mirror are kept.
171
+ * `fetch` skips files already present, writes each file atomically, and logs the SHA-256 of
172
+ every file it saved to `$EDGAR_ITEMIZE_DATA_ROOT/fetch_log.jsonl`. Both commands follow the
173
+ SEC's fair-access rules (a declared `User-Agent`, at most ten requests per second, backoff
174
+ on HTTP 429 and 503). The full 10-K corpus is about 230,000 files and several hundred GB;
175
+ if you already hold a mirror, skip `fetch`.
176
+ * `parse --kind` is `10k` (the primary document of a 10-K or 10-Q submission), `ex10`,
177
+ `ex13` or `text` (a bare text file with no SGML envelope). The output does not depend on
178
+ `--workers` or on partitioning, and a partitioned run is resumable.
179
+
180
+ Mirror layout:
181
+
182
+ ```
183
+ $EDGAR_ITEMIZE_DATA_ROOT/
184
+ archives/edgar/data/<CIK>/<accession>.txt the full-submission files, read as latin-1
185
+ full-index/<year>/QTR<n>/master.idx the cached quarterly indexes (manifest)
186
+ fetch_log.jsonl what fetch saved, with SHA-256
187
+ ```
188
+
189
+ One filing at a time, with the scored candidates (`-c`) and the rejected ones with their
190
+ reasons (`-r`); a file path needs no data root:
191
+
192
+ ```
193
+ edgar-itemize show /data/edgar/archives/edgar/data/320193/0000320193-20-000096.txt -c -r
194
+ ```
195
+
196
+ ## Output
197
+
198
+ A run writes three parquet tables per partition: `nodes` (one row per heading in the tree:
199
+ label, title, byte span, ordinal path, confidence, rule ids), `documents` (one row per
200
+ parsed document: era and publisher profile, counts, items found, the input file's SHA-256)
201
+ and `rejected` (every heading candidate that was dropped, with its reason). Files are read
202
+ as latin-1 so that one character is one byte; slicing the raw file at a node's
203
+ `raw_start:raw_end` gives exactly its span. Every column, every closed vocabulary and what
204
+ the version promises about each is in the
205
+ [output contract](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/docs/OUTPUT_CONTRACT.md).
206
+
207
+ ## Reproducibility
208
+
209
+ > Two runs of the same release of edgar-itemize on input files with the same `input_sha256`
210
+ > produce the same `nodes` and `rejected` rows and the same `documents` rows, regardless of
211
+ > machine, worker count or partitioning.
212
+
213
+ Each release ships a conformance set of about 2,000 documents with their input and output
214
+ hashes (the inputs are never redistributed; the manifest resolves against your mirror).
215
+ Check an install with:
216
+
217
+ ```
218
+ edgar-itemize verify --data-root /path/to/edgar
219
+ ```
220
+
221
+ It prints one line with the input and output hash counts and exits 0 only when every
222
+ present input matches and every output hash equals the shipped one; quote that line in a
223
+ replication package. A tagged release is never changed; a wrong output is recorded in
224
+ [ERRATA.md](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/ERRATA.md) against the
225
+ version and fixed in the next minor. The policy is
226
+ [docs/VERSIONING.md](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/docs/VERSIONING.md).
227
+
228
+ ## Measured accuracy
229
+
230
+ At the Turn 13 close, whose tree behaviour 1.0.0 carries, on the fixed 1,000-window audit
231
+ bank (`runs/judge/turn13-audit-rescore.txt`):
232
+
233
+ | measure | value |
234
+ |---|---|
235
+ | accepted precision | 94.4% |
236
+ | top-level precision | 98.2% |
237
+ | recall gap | 11.0% |
238
+
239
+ Agreement on 10-K Item boundaries with other open tools
240
+ (`runs/judge/turn13-external-before-after.txt`): datamule 91.6%; EDGAR-CORPUS 90.2%
241
+ (fiscal years 1993-2000) and 90.3% (2001-2020).
242
+
243
+ The `runs/...` names are the lab's identifiers for the stored artifacts the numbers were
244
+ read from; the full evaluation record behind them ships with release 1.0.1. How the
245
+ numbers were measured is in the manual's
246
+ [evaluation](https://malcolmwardlaw.github.io/edgar-itemize/evaluation/) page.
247
+
248
+ ## Out of scope
249
+
250
+ * XBRL financial data and tables: the parser recovers headings, not figures.
251
+ * Text cleaning: the output is spans into the raw file. `examples/pull_items.py` slices
252
+ them as filed, HTML tags and all; stripping markup and cleaning the text is left to the
253
+ user.
254
+ * Machine learning: none inside the parser, and no randomness.
255
+
256
+ ## Viewer and judge
257
+
258
+ `edgar-itemize serve` (needs the `viewer` extra) shows a filing's text with the agenda
259
+ structure as colour bands, with the candidates and rejections; it reads its bank and run
260
+ registries from the directory named by `EDGAR_ITEMIZE_VIEWER_CONFIG`, and has none by default.
261
+ `edgar-itemize judge` (the `judge` extra) asks a language model about a filing's heading
262
+ candidates; it is evaluation tooling only and never part of the parser.
263
+
264
+ ## Citing
265
+
266
+ Name the version you ran. The citation metadata is in
267
+ [CITATION.cff](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/CITATION.cff);
268
+ the Zenodo DOI is `<Zenodo concept DOI, added at the 1.0.0 release>`. See the manual's
269
+ [citing](https://malcolmwardlaw.github.io/edgar-itemize/citing/) page.
270
+
271
+ ## License
272
+
273
+ MIT.
274
+
275
+ ### Third-party code
276
+
277
+ `src/edgar_itemize/_html_parser.py` is a verbatim copy of CPython 3.11.15's
278
+ `Lib/html/parser.py` (`html.parser`), copyright the Python Software Foundation and
279
+ licensed under the Python Software Foundation License Version 2 (`LICENSES/PSF-2.0.txt`).
280
+ It is vendored so that the tokenisation of malformed HTML does not change with the
281
+ interpreter's patch level (`docs/VERSIONING.md` section 5).
@@ -0,0 +1,254 @@
1
+ # edgar-itemize
2
+
3
+ edgar-itemize turns the raw filings in the SEC's EDGAR archive into a table of their
4
+ headings. For each filing it recovers the agenda the document is written to: Part and Item
5
+ in a 10-K or 10-Q, sub-headings inside an Item, and Article, Section and the lettered and
6
+ numbered clauses of a credit agreement filed as an EX-10 exhibit. Each heading is one row
7
+ that records where its section starts and ends as byte offsets into the submission file
8
+ exactly as the SEC serves it, so the text of any section, from Item 1A down to a single
9
+ covenant clause, is one slice of a file you already have. It reads the whole EDGAR era,
10
+ from 1993 plain text through publisher HTML to inline XBRL.
11
+
12
+ * **Manual** <https://malcolmwardlaw.github.io/edgar-itemize/>
13
+ * **Output Demo** <https://malcolmwardlaw.github.io/edgar-itemize/demo/
14
+
15
+ ## Why this exists
16
+
17
+ * **The whole agenda, down to the clauses.** Most tools pull out a few top-level sections.
18
+ 10-Ks, and the contracts attached as EX-10 exhibits even more, have deep agenda
19
+ structures: a study of covenants needs the affirmative and negative covenant Articles and
20
+ each clause one or two levels beneath them. The tree goes to that depth.
21
+ * **Your corpus, downloaded once, parsed locally.** You pull as much or as little of EDGAR
22
+ as you want, up to the whole archive, and parse it on your own machine. Rate-limited SEC
23
+ downloads are slow but happen once; storage is cheap; re-parsing with different choices
24
+ costs nothing but compute.
25
+ * **Deterministic.** The parser is a fixed set of written rules, not a language model
26
+ deciding over a corpus. The same release on the same file gives the same rows every time,
27
+ every heading names the rules that produced it, and every candidate heading that was
28
+ dropped is written out with the reason.
29
+ * **Replicable by hash, with the data untouched.** Each tagged release gives exactly the
30
+ same results on exactly the same EDGAR files, checked by SHA-256 of the input and the
31
+ output. The filings themselves are never modified or redistributed, so nobody has to
32
+ archive gigabytes of derived text as validation, and every value traces directly to a
33
+ byte range in a public file.
34
+ * **Improves by frozen tagged releases.** The output is evaluated by human review and by
35
+ language-model judges, and the evaluation feeds hand-written rule changes. Current results
36
+ are good, with room left (see [Measured accuracy](#measured-accuracy)). Each improvement
37
+ ships as a new tagged release that keeps the two guarantees above; earlier releases stay
38
+ as they were.
39
+ * **Public and updatable.** The author keeps working on it. Anyone who finds a set of
40
+ errors can package them and send them in, and they are folded into the next release.
41
+
42
+ ## What to do with it
43
+
44
+ **Start small.** `examples/` holds manifests and expected hashes for the nine filings of the
45
+ demo and for ten credit agreements; the
46
+ [quickstart](https://malcolmwardlaw.github.io/edgar-itemize/quickstart/) fetches, parses,
47
+ checks and opens the nine in four commands, the first of which fetches them:
48
+
49
+ ```
50
+ for k in 10k ex10 ex13; do edgar-itemize fetch --manifest examples/sample_manifest_$k.parquet; done
51
+ ```
52
+
53
+ **Get the sections.** Build a manifest from the SEC's index, fetch the files, parse, then
54
+ pull the sections you want into one parquet with their text (the `examples/` scripts are in
55
+ the repository):
56
+
57
+ ```
58
+ edgar-itemize manifest --years 2019-2020 --no-only-present --out manifest.parquet
59
+ edgar-itemize fetch --manifest manifest.parquet
60
+ edgar-itemize parse --manifest manifest.parquet --out run_10k --kind 10k --no-text --partition-by year
61
+ python examples/pull_items.py --run run_10k --items "ITEM 1A,ITEM 7" --out items.parquet
62
+ ```
63
+
64
+ For credit agreements, parse EX-10 exhibits with `--kind ex10` (the manifest then needs a
65
+ `sequence` column naming the exhibit's document) and select the Articles whose
66
+ title names the covenants from the `nodes` table (their numbers vary by agreement); each
67
+ Article's span includes all of its clauses. The details are under
68
+ [From nothing to a parsed corpus](#from-nothing-to-a-parsed-corpus) below.
69
+
70
+ **Check a result or a paper's claim.** Confirm that your install reproduces the release, then
71
+ compare a full run of your own with the per-corpus hash parquet published with the release:
72
+
73
+ ```
74
+ edgar-itemize verify --data-root /path/to/edgar
75
+ python examples/check_hashes.py --run run_10k --expected hashes_1.0.0_10k.parquet
76
+ ```
77
+
78
+ **Help.** Report a wrong parse as an issue, following
79
+ [CONTRIBUTING.md](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/CONTRIBUTING.md):
80
+ the accession number, the document sequence, the edgar-itemize version, and the byte offsets
81
+ of what the parser produced and what you expected. A set of verdicts from your own review
82
+ is welcome the same way, as an issue or a pull request carrying those four fields per row; a
83
+ file format for exchanging verdicts is planned for release 1.0.1.
84
+
85
+ A live demo is at <https://malcolmwardlaw.github.io/edgar-itemize/demo/>, and the manual,
86
+ the long form of everything below, at <https://malcolmwardlaw.github.io/edgar-itemize/>.
87
+
88
+ ## Install
89
+
90
+ From PyPI (Python 3.11 or later):
91
+
92
+ ```
93
+ pip install edgar-itemize
94
+ ```
95
+
96
+ As a container, the frozen reference artifact (digest-pinned base image, `uv` and every
97
+ dependency from the lockfile). Pull it by the digest given in the release notes of the
98
+ version you cite:
99
+
100
+ ```
101
+ docker pull ghcr.io/malcolmwardlaw/edgar-itemize:<version>
102
+ docker run --rm -v /path/to/edgar:/data:ro ghcr.io/malcolmwardlaw/edgar-itemize@sha256:<digest> verify --data-root /data
103
+ ```
104
+
105
+ From a clone, with [uv](https://docs.astral.sh/uv/) (prefix the commands below with
106
+ `uv run`):
107
+
108
+ ```
109
+ git clone https://github.com/MalcolmWardlaw/edgar-itemize
110
+ cd edgar-itemize
111
+ uv sync --all-extras
112
+ uv run pytest -q
113
+ ```
114
+
115
+ The parser needs only `regex`, `pyarrow` and `pyyaml`; HTML is read with a vendored copy of
116
+ the standard library's `html.parser`, which reports exact source offsets. The browser viewer
117
+ and the LLM judge are optional extras (`pip install 'edgar-itemize[viewer,judge]'`).
118
+
119
+ ## From nothing to a parsed corpus
120
+
121
+ The parser reads raw full-submission files from a mirror laid out as the SEC serves them.
122
+ Three commands build that mirror from the SEC's public archive and parse it:
123
+
124
+ ```
125
+ export EDGAR_ITEMIZE_DATA_ROOT=/data/edgar # the mirror root; no default
126
+ export EDGAR_ITEMIZE_USER_AGENT="Jane Doe jane@example.org" # required by the SEC
127
+
128
+ # 1. a manifest from the public quarterly index (master.idx, cached under the data root)
129
+ edgar-itemize manifest --years 2019-2020 --no-only-present --out manifest_10k_2019_2020.parquet
130
+
131
+ # 2. the submission files it names (only those you do not already have)
132
+ edgar-itemize fetch --manifest manifest_10k_2019_2020.parquet --dry-run # counts, no network
133
+ edgar-itemize fetch --manifest manifest_10k_2019_2020.parquet
134
+
135
+ # 3. parse
136
+ edgar-itemize parse --manifest manifest_10k_2019_2020.parquet --out run_10k --kind 10k \
137
+ --workers 8 --no-text --partition-by year --progress
138
+ ```
139
+
140
+ * `manifest` keeps the 10-K family by default (`--forms` for others; `10-Q-family` expands
141
+ to the 10-Q forms), drops `/A` amendments unless `--include-amendments`, and takes
142
+ *filing* years. `--no-only-present` is for a manifest that feeds `fetch`; by default only
143
+ rows whose file is already in the mirror are kept.
144
+ * `fetch` skips files already present, writes each file atomically, and logs the SHA-256 of
145
+ every file it saved to `$EDGAR_ITEMIZE_DATA_ROOT/fetch_log.jsonl`. Both commands follow the
146
+ SEC's fair-access rules (a declared `User-Agent`, at most ten requests per second, backoff
147
+ on HTTP 429 and 503). The full 10-K corpus is about 230,000 files and several hundred GB;
148
+ if you already hold a mirror, skip `fetch`.
149
+ * `parse --kind` is `10k` (the primary document of a 10-K or 10-Q submission), `ex10`,
150
+ `ex13` or `text` (a bare text file with no SGML envelope). The output does not depend on
151
+ `--workers` or on partitioning, and a partitioned run is resumable.
152
+
153
+ Mirror layout:
154
+
155
+ ```
156
+ $EDGAR_ITEMIZE_DATA_ROOT/
157
+ archives/edgar/data/<CIK>/<accession>.txt the full-submission files, read as latin-1
158
+ full-index/<year>/QTR<n>/master.idx the cached quarterly indexes (manifest)
159
+ fetch_log.jsonl what fetch saved, with SHA-256
160
+ ```
161
+
162
+ One filing at a time, with the scored candidates (`-c`) and the rejected ones with their
163
+ reasons (`-r`); a file path needs no data root:
164
+
165
+ ```
166
+ edgar-itemize show /data/edgar/archives/edgar/data/320193/0000320193-20-000096.txt -c -r
167
+ ```
168
+
169
+ ## Output
170
+
171
+ A run writes three parquet tables per partition: `nodes` (one row per heading in the tree:
172
+ label, title, byte span, ordinal path, confidence, rule ids), `documents` (one row per
173
+ parsed document: era and publisher profile, counts, items found, the input file's SHA-256)
174
+ and `rejected` (every heading candidate that was dropped, with its reason). Files are read
175
+ as latin-1 so that one character is one byte; slicing the raw file at a node's
176
+ `raw_start:raw_end` gives exactly its span. Every column, every closed vocabulary and what
177
+ the version promises about each is in the
178
+ [output contract](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/docs/OUTPUT_CONTRACT.md).
179
+
180
+ ## Reproducibility
181
+
182
+ > Two runs of the same release of edgar-itemize on input files with the same `input_sha256`
183
+ > produce the same `nodes` and `rejected` rows and the same `documents` rows, regardless of
184
+ > machine, worker count or partitioning.
185
+
186
+ Each release ships a conformance set of about 2,000 documents with their input and output
187
+ hashes (the inputs are never redistributed; the manifest resolves against your mirror).
188
+ Check an install with:
189
+
190
+ ```
191
+ edgar-itemize verify --data-root /path/to/edgar
192
+ ```
193
+
194
+ It prints one line with the input and output hash counts and exits 0 only when every
195
+ present input matches and every output hash equals the shipped one; quote that line in a
196
+ replication package. A tagged release is never changed; a wrong output is recorded in
197
+ [ERRATA.md](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/ERRATA.md) against the
198
+ version and fixed in the next minor. The policy is
199
+ [docs/VERSIONING.md](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/docs/VERSIONING.md).
200
+
201
+ ## Measured accuracy
202
+
203
+ At the Turn 13 close, whose tree behaviour 1.0.0 carries, on the fixed 1,000-window audit
204
+ bank (`runs/judge/turn13-audit-rescore.txt`):
205
+
206
+ | measure | value |
207
+ |---|---|
208
+ | accepted precision | 94.4% |
209
+ | top-level precision | 98.2% |
210
+ | recall gap | 11.0% |
211
+
212
+ Agreement on 10-K Item boundaries with other open tools
213
+ (`runs/judge/turn13-external-before-after.txt`): datamule 91.6%; EDGAR-CORPUS 90.2%
214
+ (fiscal years 1993-2000) and 90.3% (2001-2020).
215
+
216
+ The `runs/...` names are the lab's identifiers for the stored artifacts the numbers were
217
+ read from; the full evaluation record behind them ships with release 1.0.1. How the
218
+ numbers were measured is in the manual's
219
+ [evaluation](https://malcolmwardlaw.github.io/edgar-itemize/evaluation/) page.
220
+
221
+ ## Out of scope
222
+
223
+ * XBRL financial data and tables: the parser recovers headings, not figures.
224
+ * Text cleaning: the output is spans into the raw file. `examples/pull_items.py` slices
225
+ them as filed, HTML tags and all; stripping markup and cleaning the text is left to the
226
+ user.
227
+ * Machine learning: none inside the parser, and no randomness.
228
+
229
+ ## Viewer and judge
230
+
231
+ `edgar-itemize serve` (needs the `viewer` extra) shows a filing's text with the agenda
232
+ structure as colour bands, with the candidates and rejections; it reads its bank and run
233
+ registries from the directory named by `EDGAR_ITEMIZE_VIEWER_CONFIG`, and has none by default.
234
+ `edgar-itemize judge` (the `judge` extra) asks a language model about a filing's heading
235
+ candidates; it is evaluation tooling only and never part of the parser.
236
+
237
+ ## Citing
238
+
239
+ Name the version you ran. The citation metadata is in
240
+ [CITATION.cff](https://github.com/MalcolmWardlaw/edgar-itemize/blob/main/CITATION.cff);
241
+ the Zenodo DOI is `<Zenodo concept DOI, added at the 1.0.0 release>`. See the manual's
242
+ [citing](https://malcolmwardlaw.github.io/edgar-itemize/citing/) page.
243
+
244
+ ## License
245
+
246
+ MIT.
247
+
248
+ ### Third-party code
249
+
250
+ `src/edgar_itemize/_html_parser.py` is a verbatim copy of CPython 3.11.15's
251
+ `Lib/html/parser.py` (`html.parser`), copyright the Python Software Foundation and
252
+ licensed under the Python Software Foundation License Version 2 (`LICENSES/PSF-2.0.txt`).
253
+ It is vendored so that the tokenisation of malformed HTML does not change with the
254
+ interpreter's patch level (`docs/VERSIONING.md` section 5).
@@ -0,0 +1,52 @@
1
+ [project]
2
+ name = "edgar-itemize"
3
+ version = "1.0.0rc2"
4
+ description = "Deterministic agenda-structure (Part/Item/Section) parser for SEC EDGAR filings"
5
+ readme = "README.md"
6
+ license = "MIT"
7
+ license-files = [
8
+ "LICENSE",
9
+ "LICENSES/PSF-2.0.txt",
10
+ ]
11
+ requires-python = ">=3.11"
12
+ dependencies = [
13
+ "regex>=2024.0",
14
+ "pyarrow>=14",
15
+ "pyyaml>=6",
16
+ ]
17
+
18
+ [[project.authors]]
19
+ name = "Malcolm Wardlaw"
20
+ email = "malcolm.wardlaw@uga.edu"
21
+
22
+ [project.optional-dependencies]
23
+ viewer = [
24
+ "fastapi>=0.141.1",
25
+ "uvicorn[standard]>=0.52.4",
26
+ ]
27
+ judge = ["httpx>=0.28.1"]
28
+ polars = ["polars>=1.0"]
29
+ dev = [
30
+ "pytest>=8",
31
+ "polars>=1.0",
32
+ ]
33
+ docs = ["mkdocs-material>=9.5"]
34
+
35
+ [project.scripts]
36
+ edgar-itemize = "edgar_itemize.cli:main"
37
+
38
+ [dependency-groups]
39
+ dev = [
40
+ "polars>=1.0",
41
+ "pytest>=8",
42
+ ]
43
+
44
+ [build-system]
45
+ requires = ["uv_build>=0.10.8,<0.11.0"]
46
+ build-backend = "uv_build"
47
+
48
+ [tool.pytest.ini_options]
49
+ testpaths = [
50
+ "tests",
51
+ "tests_lab",
52
+ ]