dcmspec 0.2.2__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dcmspec-0.2.2 → dcmspec-0.3.0}/CHANGELOG.md +29 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/PKG-INFO +5 -3
- {dcmspec-0.2.2 → dcmspec-0.3.0}/README.md +2 -2
- {dcmspec-0.2.2 → dcmspec-0.3.0}/pyproject.toml +3 -1
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/.DS_Store +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/iodattributes.py +28 -5
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/modattributes.py +36 -13
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/csv_table_spec_parser.py +16 -7
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/doc_handler.py +15 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/dom_table_spec_parser.py +67 -34
- dcmspec-0.3.0/src/dcmspec/iod_spec_printer.py +254 -0
- dcmspec-0.3.0/src/dcmspec/spec_printer.py +414 -0
- dcmspec-0.2.2/src/dcmspec/iod_spec_printer.py +0 -59
- dcmspec-0.2.2/src/dcmspec/spec_printer.py +0 -131
- {dcmspec-0.2.2 → dcmspec-0.3.0}/LICENSE +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/__init__.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/.DS_Store +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/__init__.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/__init__.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/dataelements.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/iodmodules.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/tdwiicontent.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/uidvalues.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/upsdimseattributes.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/upsioddimseattributes.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/README.md +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/__init__.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/config/README.md +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/config/iod_explorer_config.json +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/config/iod_explorer_config_debug.json +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/config/iod_explorer_config_example.json +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/config/iod_explorer_config_minimal_logging.json +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/iod_explorer.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/config.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/dom_utils.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/iod_spec_builder.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/json_spec_store.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/module_registry.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/pdf_doc_handler.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/progress.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/service_attribute_defaults.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/service_attribute_model.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/spec_factory.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/spec_merger.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/spec_model.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/spec_parser.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/spec_store.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/ups_xhtml_doc_handler.py +0 -0
- {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/xhtml_doc_handler.py +0 -0
|
@@ -2,6 +2,35 @@
|
|
|
2
2
|
|
|
3
3
|
These release notes summarize key changes, improvements, and breaking updates for each version of **dcmspec**.
|
|
4
4
|
|
|
5
|
+
## [0.3.0] - 2025-11-27
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
|
|
9
|
+
- CSV output mode in `SpecPrinter` via `print_csv` method
|
|
10
|
+
- Excel OOXML output mode in `SpecPrinter` via `print_xlsx` method
|
|
11
|
+
- Optional `output` parameter in `SpecPrinter` for writing to files
|
|
12
|
+
- Optional `column_width` parameter in `print_table` and `print_xlsx`
|
|
13
|
+
- `--print-mode` option in `modattributes` CLI for CSV and OOXML output
|
|
14
|
+
- `--output` option in `modattributes` CLI for directing output to files
|
|
15
|
+
|
|
16
|
+
### Changed
|
|
17
|
+
|
|
18
|
+
- Switch PyPI version badge in `README` from `badge.fury.io` to `shields.io` for faster updates
|
|
19
|
+
- Update modattributes CLI example for CSV and XLSX output
|
|
20
|
+
- Improved nesting level color schemes for consistency and better contrast and readability.
|
|
21
|
+
|
|
22
|
+
### Fixed
|
|
23
|
+
|
|
24
|
+
- Correct handling of `include_table` names in `DOMTableSpecParser`
|
|
25
|
+
- Correct handling of empty node attributes in `SpecPrinter`
|
|
26
|
+
|
|
27
|
+
## [0.2.3] - 2025-09-29
|
|
28
|
+
|
|
29
|
+
### Fixed
|
|
30
|
+
|
|
31
|
+
- Hotfix: Force UTF-8 decoding for DICOM standard XHTML downloads to prevent mojibake when server omits charset ([#85](https://github.com/dwikler/dcmspec/issues/85)).
|
|
32
|
+
- Hotfix: Add missing `progress_observer` argument to `CSVTableSpecParser.parse` for interface compatibility and to prevent `TypeError` when used with `SpecFactory` ([#86](https://github.com/dwikler/dcmspec/issues/86)).
|
|
33
|
+
|
|
5
34
|
## [0.2.2] - 2025-09-25
|
|
6
35
|
|
|
7
36
|
### Fixed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dcmspec
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Toolkit for extracting, parsing, and processing DICOM specifications.
|
|
5
5
|
License: Apache-2.0
|
|
6
6
|
License-File: LICENSE
|
|
@@ -25,8 +25,10 @@ Provides-Extra: pdf
|
|
|
25
25
|
Requires-Dist: anytree (>=2.13.0,<3.0.0)
|
|
26
26
|
Requires-Dist: bs4 (>=0.0.2,<0.0.3)
|
|
27
27
|
Requires-Dist: camelot-py (>=1.0.0,<2.0.0) ; extra == "pdf"
|
|
28
|
+
Requires-Dist: html2text (>=2025.4.15,<2026.0.0)
|
|
28
29
|
Requires-Dist: lxml (>=5.4.0,<6.0.0)
|
|
29
30
|
Requires-Dist: opencv-python-headless (>=4.12.0.88,<5.0.0) ; extra == "pdf"
|
|
31
|
+
Requires-Dist: openpyxl (>=3.1.5,<4.0.0)
|
|
30
32
|
Requires-Dist: openpyxl (>=3.1.5,<4.0.0) ; extra == "pdf"
|
|
31
33
|
Requires-Dist: pandas (>=2.3.1,<3.0.0) ; extra == "pdf"
|
|
32
34
|
Requires-Dist: pdfplumber (>=0.11.7,<0.12.0) ; extra == "pdf"
|
|
@@ -43,9 +45,9 @@ Project-URL: Source, https://github.com/dwikler/dcmspec
|
|
|
43
45
|
Description-Content-Type: text/markdown
|
|
44
46
|
|
|
45
47
|
[](https://github.com/dwikler/dcmspec/actions/workflows/test.yml)
|
|
46
|
-
[](https://pypi.org/project/dcmspec/)
|
|
47
49
|
[](https://pypi.org/project/dcmspec/)
|
|
48
|
-
[](https://doi.org/10.5281/zenodo.17742287)
|
|
49
51
|
|
|
50
52
|
# dcmspec
|
|
51
53
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
[](https://github.com/dwikler/dcmspec/actions/workflows/test.yml)
|
|
2
|
-
[](https://pypi.org/project/dcmspec/)
|
|
3
3
|
[](https://pypi.org/project/dcmspec/)
|
|
4
|
-
[](https://doi.org/10.5281/zenodo.17742287)
|
|
5
5
|
|
|
6
6
|
# dcmspec
|
|
7
7
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "dcmspec"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.3.0"
|
|
4
4
|
description = "Toolkit for extracting, parsing, and processing DICOM specifications."
|
|
5
5
|
authors = [{ name = "David Wikler", email = "david.wikler@ulb.be" }]
|
|
6
6
|
license = { text = "Apache-2.0" }
|
|
@@ -29,6 +29,8 @@ dependencies = [
|
|
|
29
29
|
"requests (>=2.32.3,<3.0.0)",
|
|
30
30
|
"lxml (>=5.4.0,<6.0.0)",
|
|
31
31
|
"rich (>=14.0.0,<15.0.0)",
|
|
32
|
+
"html2text (>=2025.4.15,<2026.0.0)",
|
|
33
|
+
"openpyxl (>=3.1.5,<4.0.0)",
|
|
32
34
|
]
|
|
33
35
|
|
|
34
36
|
[project.urls]
|
|
Binary file
|
|
@@ -58,12 +58,25 @@ def main():
|
|
|
58
58
|
parser.add_argument("--config", help="Path to the configuration file")
|
|
59
59
|
parser.add_argument(
|
|
60
60
|
"--print-mode",
|
|
61
|
-
choices=["table", "tree", "none"],
|
|
61
|
+
choices=["table", "tree", "csv", "xlsx", "none"],
|
|
62
62
|
default="table",
|
|
63
|
-
help="Print as 'table' (default), 'tree', or 'none' to skip printing"
|
|
63
|
+
help="Print as 'table' (default), 'tree', 'csv', 'xlsx' (requires --output), or 'none' to skip printing"
|
|
64
|
+
)
|
|
65
|
+
parser.add_argument(
|
|
66
|
+
"--no-color",
|
|
67
|
+
action="store_true",
|
|
68
|
+
help="Disable colorized output (default: color enabled for terminal, disabled for file output)"
|
|
69
|
+
)
|
|
70
|
+
parser.add_argument(
|
|
71
|
+
"--output",
|
|
72
|
+
type=str,
|
|
73
|
+
help="Path to output file. If not specified (and not xlsx), prints to stdout."
|
|
64
74
|
)
|
|
65
75
|
args = parser.parse_args()
|
|
66
76
|
|
|
77
|
+
if args.print_mode == "xlsx" and not args.output:
|
|
78
|
+
parser.error("--output is required when --print-mode xlsx")
|
|
79
|
+
|
|
67
80
|
cache_file_name = "Part3.xhtml"
|
|
68
81
|
model_file_name = f"Part3_{args.table}_expanded.json"
|
|
69
82
|
table_id = args.table
|
|
@@ -114,11 +127,21 @@ def main():
|
|
|
114
127
|
)
|
|
115
128
|
|
|
116
129
|
# Print the model
|
|
117
|
-
printer = IODSpecPrinter(model)
|
|
130
|
+
printer = IODSpecPrinter(model, output=getattr(args, "output", None))
|
|
131
|
+
|
|
132
|
+
# Only disable color if --no-color is set; SpecPrinter disables color for file outputs automatically
|
|
133
|
+
use_color = not args.no_color
|
|
134
|
+
|
|
118
135
|
if args.print_mode == "tree":
|
|
119
|
-
printer.print_tree(
|
|
136
|
+
printer.print_tree(
|
|
137
|
+
attr_names=["elem_tag", "elem_type", "elem_name"], attr_widths=[11, 2, 100], colorize=use_color
|
|
138
|
+
)
|
|
120
139
|
elif args.print_mode == "table":
|
|
121
|
-
printer.print_table(colorize=
|
|
140
|
+
printer.print_table(colorize=use_color, column_widths=[30, 11, 4, 60])
|
|
141
|
+
elif args.print_mode == "csv":
|
|
142
|
+
printer.print_csv(colorize=use_color)
|
|
143
|
+
elif args.print_mode == "xlsx":
|
|
144
|
+
printer.print_xlsx(column_widths=[60, 16, 10, 100], colorize=use_color)
|
|
122
145
|
# else: do not print anything if print_mode == "none"
|
|
123
146
|
|
|
124
147
|
if __name__ == "__main__":
|
|
@@ -4,7 +4,7 @@ Features:
|
|
|
4
4
|
- Download and parse DICOM Module Attributes tables from Part 3 of the DICOM standard.
|
|
5
5
|
- Optionally merge additional information (VR, VM, Keyword, Status) from Part 6.
|
|
6
6
|
- Cache the resulting model as a JSON file for future runs and as a structured representation of the standard.
|
|
7
|
-
- Print the resulting model as a table or
|
|
7
|
+
- Print the resulting model as a table, tree, csv, or xlsx.
|
|
8
8
|
- Supports caching, configuration files, and command-line options for flexible workflows.
|
|
9
9
|
|
|
10
10
|
Usage:
|
|
@@ -113,12 +113,12 @@ def main():
|
|
|
113
113
|
|
|
114
114
|
The tool parses the specified Module Attributes table to extract all attributes, tags, types,
|
|
115
115
|
and descriptions for the module. Optionally, it can merge in VR, VM, Keyword, or Status information
|
|
116
|
-
from Part 6. The output can be printed as a table or
|
|
116
|
+
from Part 6. The output can be printed as a table, tree, csv, or xlsx (xlsx requires --output).
|
|
117
117
|
|
|
118
118
|
The resulting model is cached as a JSON file. The primary purpose of this cache file is to provide
|
|
119
|
-
a structured, machine-readable representation of the module's attributes, which can be used for
|
|
120
|
-
or integration in other tools. As a secondary benefit, the cache file is also used
|
|
121
|
-
of the CLI scripts.
|
|
119
|
+
a structured, machine-readable representation of the module's attributes, which can be used for
|
|
120
|
+
further processing or integration in other tools. As a secondary benefit, the cache file is also used
|
|
121
|
+
to speed up subsequent runs of the CLI scripts.
|
|
122
122
|
|
|
123
123
|
Usage:
|
|
124
124
|
poetry run python -m src.dcmspec.apps.cli.modattributes <table_id> [options]
|
|
@@ -129,14 +129,17 @@ def main():
|
|
|
129
129
|
--include-depth (int): Depth to which included tables should be parsed (default: unlimited).
|
|
130
130
|
--force-parse: Force reparsing of the DOM and regeneration of the JSON model.
|
|
131
131
|
--force-download: Force download of the input file and regeneration of the model.
|
|
132
|
-
--print-mode (str): Print as 'table' (default), 'tree', or 'none' to skip printing.
|
|
132
|
+
--print-mode (str): Print as 'table' (default), 'tree', 'csv', 'xlsx', or 'none' to skip printing.
|
|
133
|
+
'xlsx' requires --output.
|
|
134
|
+
--output (str): Path to output file. If not specified (and not xlsx), prints to stdout.
|
|
133
135
|
--add-part6 (list): Specification(s) to merge from Part 6 (e.g., --add-part6 VR VM).
|
|
134
136
|
--force-update: Force update of the specifications merged from part 6, even if cached.
|
|
135
137
|
-d, --debug: Enable debug logging to the console.
|
|
136
138
|
-v, --verbose: Enable verbose (info-level) logging to the console.
|
|
137
139
|
|
|
138
|
-
|
|
140
|
+
Examples:
|
|
139
141
|
poetry run python -m src.dcmspec.apps.cli.modattributes table_C.7-1 --add-part6 VR VM
|
|
142
|
+
poetry run python -m src.dcmspec.apps.cli.modattributes table_C.7-1 --print-mode xlsx --output module.xlsx
|
|
140
143
|
|
|
141
144
|
"""
|
|
142
145
|
# Parse command-line arguments
|
|
@@ -164,9 +167,19 @@ def main():
|
|
|
164
167
|
)
|
|
165
168
|
parser.add_argument(
|
|
166
169
|
"--print-mode",
|
|
167
|
-
choices=["table", "tree", "none"],
|
|
170
|
+
choices=["table", "tree", "csv", "xlsx", "none"],
|
|
168
171
|
default="table",
|
|
169
|
-
help="Print as 'table' (default), 'tree', or 'none' to skip printing"
|
|
172
|
+
help="Print as 'table' (default), 'tree', 'csv', 'xlsx' (requires --output), or 'none' to skip printing"
|
|
173
|
+
)
|
|
174
|
+
parser.add_argument(
|
|
175
|
+
"--no-color",
|
|
176
|
+
action="store_true",
|
|
177
|
+
help="Disable colorized output (default: color enabled for terminal, disabled for file output)"
|
|
178
|
+
)
|
|
179
|
+
parser.add_argument(
|
|
180
|
+
"--output",
|
|
181
|
+
type=str,
|
|
182
|
+
help="Path to output file. If not specified (and not xlsx), prints to stdout."
|
|
170
183
|
)
|
|
171
184
|
parser.add_argument(
|
|
172
185
|
"--add-part6",
|
|
@@ -189,9 +202,12 @@ def main():
|
|
|
189
202
|
action="store_true",
|
|
190
203
|
help="Enable verbose (info-level) logging to the console"
|
|
191
204
|
)
|
|
192
|
-
|
|
205
|
+
|
|
193
206
|
args = parser.parse_args()
|
|
194
207
|
|
|
208
|
+
if args.print_mode == "xlsx" and not args.output:
|
|
209
|
+
parser.error("--output is required when --print-mode xlsx")
|
|
210
|
+
|
|
195
211
|
# Set up logger
|
|
196
212
|
logger = logging.getLogger("modattributes")
|
|
197
213
|
handler = logging.StreamHandler()
|
|
@@ -254,11 +270,18 @@ def main():
|
|
|
254
270
|
model = module_model
|
|
255
271
|
|
|
256
272
|
logger.debug("Model ready for printing/output")
|
|
257
|
-
printer = SpecPrinter(model)
|
|
273
|
+
printer = SpecPrinter(model, output=args.output)
|
|
274
|
+
|
|
275
|
+
# Only disable color if --no-color is set; SpecPrinter disables color for file outputs automatically
|
|
276
|
+
use_color = not args.no_color
|
|
258
277
|
if args.print_mode == "tree":
|
|
259
|
-
printer.print_tree(colorize=
|
|
278
|
+
printer.print_tree(attr_names=["elem_tag", "elem_type", "elem_name"], attr_widths=[11, 2, 100], colorize=use_color)
|
|
260
279
|
elif args.print_mode == "table":
|
|
261
|
-
printer.print_table(colorize=
|
|
280
|
+
printer.print_table(colorize=use_color, column_widths=[30, 11, 4, 60])
|
|
281
|
+
elif args.print_mode == "csv":
|
|
282
|
+
printer.print_csv(colorize=use_color)
|
|
283
|
+
elif args.print_mode == "xlsx":
|
|
284
|
+
printer.print_xlsx(column_widths=[60, 16, 10, 100], colorize=use_color)
|
|
262
285
|
# else: do not print anything if print_mode == "none"
|
|
263
286
|
|
|
264
287
|
if __name__ == "__main__":
|
|
@@ -3,10 +3,11 @@
|
|
|
3
3
|
Provides the CSVTableSpecParser class for parsing DICOM specification tables in CSV format,
|
|
4
4
|
converting them into structured in-memory representations using anytree.
|
|
5
5
|
"""
|
|
6
|
-
from typing import Tuple
|
|
6
|
+
from typing import Dict, List, Tuple, Optional
|
|
7
7
|
from anytree import Node
|
|
8
8
|
|
|
9
9
|
from dcmspec.spec_parser import SpecParser
|
|
10
|
+
from dcmspec.progress import ProgressObserver
|
|
10
11
|
|
|
11
12
|
class CSVTableSpecParser(SpecParser):
|
|
12
13
|
"""Base parser for DICOM Specification IHE tables in CSV-like format."""
|
|
@@ -14,10 +15,11 @@ class CSVTableSpecParser(SpecParser):
|
|
|
14
15
|
def parse(
|
|
15
16
|
self,
|
|
16
17
|
table: dict,
|
|
17
|
-
column_to_attr,
|
|
18
|
-
name_attr="elem_name",
|
|
19
|
-
table_id=None,
|
|
20
|
-
include_depth=None,
|
|
18
|
+
column_to_attr: dict,
|
|
19
|
+
name_attr: str = "elem_name",
|
|
20
|
+
table_id: Optional[str] = None,
|
|
21
|
+
include_depth: Optional[int] = None,
|
|
22
|
+
progress_observer: Optional[ProgressObserver] = None,
|
|
21
23
|
) -> Tuple[Node, Node]:
|
|
22
24
|
"""Parse specification metadata and content from a single table dict.
|
|
23
25
|
|
|
@@ -27,11 +29,18 @@ class CSVTableSpecParser(SpecParser):
|
|
|
27
29
|
name_attr (str): The attribute to use for node names.
|
|
28
30
|
table_id (str, optional): Table identifier for model parsing.
|
|
29
31
|
include_depth (int, optional): The depth to which included tables should be parsed.
|
|
32
|
+
progress_observer (Optional[ProgressObserver]):
|
|
33
|
+
Accepted for interface compatibility, but ignored in this parser.
|
|
34
|
+
Included so that this method can be called with the same arguments as other table parsers.
|
|
30
35
|
|
|
31
36
|
Returns:
|
|
32
37
|
tuple: (metadata_node, content_node)
|
|
33
38
|
|
|
34
39
|
"""
|
|
40
|
+
if progress_observer is not None and hasattr(self, "logger"):
|
|
41
|
+
self.logger.debug(
|
|
42
|
+
"Progress reporting is not supported yet for CSV parsing and will be ignored."
|
|
43
|
+
)
|
|
35
44
|
# Use the header and data from the grouped table dict
|
|
36
45
|
header = table.get("header", [])
|
|
37
46
|
data = table.get("data", [])
|
|
@@ -47,8 +56,8 @@ class CSVTableSpecParser(SpecParser):
|
|
|
47
56
|
|
|
48
57
|
def parse_table(
|
|
49
58
|
self,
|
|
50
|
-
tables:
|
|
51
|
-
column_to_attr:
|
|
59
|
+
tables: List[List[List[str]]], # List of tables, each a list of rows (list of str)
|
|
60
|
+
column_to_attr: Dict[int, str],
|
|
52
61
|
name_attr: str = "elem_name",
|
|
53
62
|
) -> Node:
|
|
54
63
|
"""Build a tree from tables using column mapping and '>' nesting logic.
|
|
@@ -94,6 +94,8 @@ class DocHandler:
|
|
|
94
94
|
try:
|
|
95
95
|
with requests.get(url, timeout=30, stream=True, headers={"Accept-Encoding": "identity"}) as response:
|
|
96
96
|
response.raise_for_status()
|
|
97
|
+
self._set_response_encoding(response)
|
|
98
|
+
|
|
97
99
|
total = int(response.headers.get('content-length', 0))
|
|
98
100
|
chunk_size = 8192
|
|
99
101
|
if binary:
|
|
@@ -109,6 +111,19 @@ class DocHandler:
|
|
|
109
111
|
self.logger.error(f"Failed to save file {file_path}: {e}")
|
|
110
112
|
raise RuntimeError(f"Failed to save file {file_path}: {e}") from e
|
|
111
113
|
|
|
114
|
+
def _set_response_encoding(self, response):
|
|
115
|
+
"""Set response.encoding to UTF-8 only if the Content-Type header does not specify a charset.
|
|
116
|
+
|
|
117
|
+
Force utf-8 decoding if the web server does not specify the charset in the HTTP response
|
|
118
|
+
Content-Type header (DICOM standard XHTML files are always UTF-8 encoded).
|
|
119
|
+
"""
|
|
120
|
+
content_type = response.headers.get("Content-Type", "")
|
|
121
|
+
if "charset=" not in content_type.lower():
|
|
122
|
+
response.encoding = "utf-8"
|
|
123
|
+
self.logger.debug("No charset in Content-Type header; forcing UTF-8 decoding.")
|
|
124
|
+
else:
|
|
125
|
+
self.logger.debug(f"Using server-specified encoding from Content-Type: {content_type}")
|
|
126
|
+
|
|
112
127
|
def _report_progress(self, downloaded, total, progress_observer, last_percent):
|
|
113
128
|
"""Report progress if percent changed.
|
|
114
129
|
|
|
@@ -3,17 +3,20 @@
|
|
|
3
3
|
Provides the DOMSpecParser class for parsing DICOM specification tables from XHTML documents,
|
|
4
4
|
converting them into structured in-memory representations using anytree.
|
|
5
5
|
"""
|
|
6
|
+
|
|
6
7
|
from contextlib import contextmanager
|
|
7
8
|
import re
|
|
8
9
|
import unicodedata
|
|
10
|
+
from typing import Any, Dict, Optional, Union
|
|
11
|
+
|
|
9
12
|
from unidecode import unidecode
|
|
10
13
|
from anytree import Node
|
|
11
14
|
from bs4 import BeautifulSoup, Tag
|
|
12
|
-
|
|
13
|
-
from dcmspec.spec_parser import SpecParser
|
|
15
|
+
import html2text
|
|
14
16
|
|
|
17
|
+
from dcmspec.spec_parser import SpecParser
|
|
15
18
|
from dcmspec.dom_utils import DOMUtils
|
|
16
|
-
from dcmspec.progress import Progress, ProgressStatus, calculate_percent
|
|
19
|
+
from dcmspec.progress import Progress, ProgressObserver, ProgressStatus, calculate_percent
|
|
17
20
|
|
|
18
21
|
class DOMTableSpecParser(SpecParser):
|
|
19
22
|
"""Parser for DICOM specification tables in XHTML DOM format.
|
|
@@ -43,7 +46,7 @@ class DOMTableSpecParser(SpecParser):
|
|
|
43
46
|
column_to_attr: Dict[int, str],
|
|
44
47
|
name_attr: str,
|
|
45
48
|
include_depth: Optional[int] = None, # None means unlimited
|
|
46
|
-
progress_observer: Optional[
|
|
49
|
+
progress_observer: Optional[ProgressObserver] = None,
|
|
47
50
|
skip_columns: Optional[list[int]] = None,
|
|
48
51
|
unformatted: Optional[Union[bool, Dict[int, bool]]] = True,
|
|
49
52
|
) -> tuple[Node, Node]:
|
|
@@ -59,7 +62,7 @@ class DOMTableSpecParser(SpecParser):
|
|
|
59
62
|
name_attr (str): The attribute name to use for building node names.
|
|
60
63
|
include_depth (Optional[int], optional): The depth to which included tables should be parsed.
|
|
61
64
|
None means unlimited.
|
|
62
|
-
progress_observer (Optional[
|
|
65
|
+
progress_observer (Optional[ProgressObserver]): Optional observer to report parsing progress.
|
|
63
66
|
skip_columns (Optional[list[int]]): List of column indices to skip if the row is missing a column.
|
|
64
67
|
This argument is typically set via `parser_kwargs` when using SpecFactory.
|
|
65
68
|
unformatted (Optional[Union[bool, Dict[int, bool]]]):
|
|
@@ -132,7 +135,7 @@ class DOMTableSpecParser(SpecParser):
|
|
|
132
135
|
name_attr: str,
|
|
133
136
|
table_nesting_level: int = 0,
|
|
134
137
|
include_depth: Optional[int] = None, # None means unlimited
|
|
135
|
-
progress_observer: Optional[
|
|
138
|
+
progress_observer: Optional[ProgressObserver] = None,
|
|
136
139
|
skip_columns: Optional[list[int]] = None,
|
|
137
140
|
visited_tables: Optional[set] = None,
|
|
138
141
|
unformatted_list: Optional[list[bool]] = None,
|
|
@@ -151,7 +154,7 @@ class DOMTableSpecParser(SpecParser):
|
|
|
151
154
|
name_attr: tree node attribute name to use to build node name
|
|
152
155
|
table_nesting_level: The nesting level of the table (used for recursion call only).
|
|
153
156
|
include_depth: The depth to which included tables should be parsed.
|
|
154
|
-
progress_observer (Optional[
|
|
157
|
+
progress_observer (Optional[ProgressObserver]): Optional observer to report parsing progress.
|
|
155
158
|
skip_columns (Optional[list[int]]): List of column indices to skip if the row is missing a column.
|
|
156
159
|
visited_tables (Optional[set]): Set of table IDs that have been visited to prevent infinite recursion.
|
|
157
160
|
unformatted_list (Optional[list[bool]]): List of booleans indicating whether to extract each column as
|
|
@@ -287,7 +290,7 @@ class DOMTableSpecParser(SpecParser):
|
|
|
287
290
|
unformatted_list: list[bool],
|
|
288
291
|
level_nodes: Dict[int, Node],
|
|
289
292
|
root: Node,
|
|
290
|
-
progress_observer: Optional[
|
|
293
|
+
progress_observer: Optional[ProgressObserver] = None
|
|
291
294
|
) -> None:
|
|
292
295
|
"""Process all rows in the table, handling recursion, nesting, and node creation."""
|
|
293
296
|
rows = table.find_all("tr")[1:]
|
|
@@ -506,6 +509,7 @@ class DOMTableSpecParser(SpecParser):
|
|
|
506
509
|
|
|
507
510
|
return logical_cells, logical_col_idx, physical_col_idx
|
|
508
511
|
|
|
512
|
+
|
|
509
513
|
def _extract_cell_value(
|
|
510
514
|
self,
|
|
511
515
|
cell: Tag,
|
|
@@ -518,11 +522,27 @@ class DOMTableSpecParser(SpecParser):
|
|
|
518
522
|
if unformatted_list and logical_col_idx < len(unformatted_list)
|
|
519
523
|
else True
|
|
520
524
|
)
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
525
|
+
|
|
526
|
+
# Guard clause: if formatted HTML is required, return content as-is
|
|
527
|
+
if not use_unformatted:
|
|
528
|
+
# Keep original HTML content
|
|
524
529
|
return self._clean_extracted_text(cell.decode_contents())
|
|
525
530
|
|
|
531
|
+
# Use html2text for better readability of unformatted text extraction
|
|
532
|
+
converter = self._create_html2text_converter()
|
|
533
|
+
raw_text = converter.handle(str(cell))
|
|
534
|
+
return self._clean_extracted_text(raw_text)
|
|
535
|
+
|
|
536
|
+
|
|
537
|
+
def _create_html2text_converter(self) -> html2text.HTML2Text:
|
|
538
|
+
"""Create and configure an html2text converter for consistent text extraction."""
|
|
539
|
+
converter = html2text.HTML2Text()
|
|
540
|
+
converter.ignore_links = True # Remove URLs
|
|
541
|
+
converter.ignore_images = True # Remove image references
|
|
542
|
+
converter.ignore_emphasis = True # Remove Markdown emphasis
|
|
543
|
+
converter.body_width = 0 # Disable word wrapping
|
|
544
|
+
return converter
|
|
545
|
+
|
|
526
546
|
def _update_rowspan_trackers(
|
|
527
547
|
self,
|
|
528
548
|
physical_col_idx: int,
|
|
@@ -781,15 +801,7 @@ class DOMTableSpecParser(SpecParser):
|
|
|
781
801
|
return header
|
|
782
802
|
|
|
783
803
|
def _clean_extracted_text(self, text: str) -> str:
|
|
784
|
-
"""Clean extracted text using Unicode normalization and regex.
|
|
785
|
-
|
|
786
|
-
Args:
|
|
787
|
-
text (str): The text to be cleaned.
|
|
788
|
-
|
|
789
|
-
Returns:
|
|
790
|
-
str: The cleaned text.
|
|
791
|
-
|
|
792
|
-
"""
|
|
804
|
+
"""Clean extracted text using Unicode normalization and regex."""
|
|
793
805
|
# Normalize unicode characters to compatibility form
|
|
794
806
|
cleaned = unicodedata.normalize('NFKC', text)
|
|
795
807
|
|
|
@@ -805,27 +817,48 @@ class DOMTableSpecParser(SpecParser):
|
|
|
805
817
|
# Remove stray  character
|
|
806
818
|
cleaned = cleaned.replace('\u00c2', '')
|
|
807
819
|
|
|
820
|
+
# Collapse multiple newlines (including those separated by spaces/tabs) into a single newline
|
|
821
|
+
cleaned = re.sub(r'(\n\s*){2,}', '\n', cleaned)
|
|
822
|
+
|
|
808
823
|
return cleaned.strip()
|
|
809
824
|
|
|
810
|
-
|
|
811
|
-
|
|
825
|
+
@staticmethod
|
|
826
|
+
def _sanitize_string(input_string: str) -> str:
|
|
827
|
+
r"""Sanitize a string to make it safe for use as a node attribute name.
|
|
812
828
|
|
|
813
|
-
|
|
814
|
-
-
|
|
815
|
-
-
|
|
829
|
+
Transformations applied:
|
|
830
|
+
- Convert to lowercase.
|
|
831
|
+
- Transliterate non-ASCII characters to ASCII.
|
|
832
|
+
- Replace spaces, slashes, newlines, and dots with underscores ("_").
|
|
833
|
+
- Replace parentheses with dashes ("-").
|
|
834
|
+
- Remove all characters except letters, digits, underscores, and dashes.
|
|
835
|
+
- Collapse multiple consecutive underscores into a single underscore.
|
|
836
|
+
- Remove leading and trailing underscores for cleanliness.
|
|
837
|
+
- Return a default name if the result is empty after sanitization.
|
|
816
838
|
|
|
817
839
|
Args:
|
|
818
|
-
input_string (str): The string to
|
|
840
|
+
input_string (str): The original string to sanitize.
|
|
819
841
|
|
|
820
842
|
Returns:
|
|
821
|
-
str:
|
|
843
|
+
str: A sanitized version of the input string, suitable for use as an identifier.
|
|
844
|
+
or "unnamed_node" if sanitization results in an empty string.
|
|
845
|
+
|
|
846
|
+
Example:
|
|
847
|
+
>>> DOMTableSpecParser._sanitize_string(
|
|
848
|
+
'>>Include\\nTable C.36.2.2.19-1 "RT Beam Limiting Device Definition Macro Attributes"\\n.'
|
|
849
|
+
)
|
|
850
|
+
'include_table_c_36_2_2_19-1_rt_beam_limiting_device_definition_macro_attributes'
|
|
851
|
+
>>> DOMTableSpecParser._sanitize_string('...')
|
|
852
|
+
'unnamed_node'
|
|
822
853
|
|
|
823
854
|
"""
|
|
824
|
-
# Normalize the string to NFC form and transliterate to ASCII
|
|
825
855
|
normalized_str = unidecode(input_string.lower())
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
856
|
+
sanitized = re.sub(r"[ /\n\\.]", "_", normalized_str) # spaces, slashes, newlines, dots → _
|
|
857
|
+
sanitized = re.sub(r"[()]", "-", sanitized) # parentheses → -
|
|
858
|
+
sanitized = re.sub(r"[^a-z0-9_-]", "", sanitized) # remove other chars
|
|
859
|
+
sanitized = re.sub(r"_+", "_", sanitized) # collapse multiple underscores
|
|
860
|
+
sanitized = sanitized.strip("_") # remove leading/trailing underscores
|
|
861
|
+
|
|
862
|
+
# Fallback to default name if sanitization resulted in empty string
|
|
863
|
+
return sanitized or "unnamed_node"
|
|
864
|
+
|