dcmspec 0.2.2__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. {dcmspec-0.2.2 → dcmspec-0.3.0}/CHANGELOG.md +29 -0
  2. {dcmspec-0.2.2 → dcmspec-0.3.0}/PKG-INFO +5 -3
  3. {dcmspec-0.2.2 → dcmspec-0.3.0}/README.md +2 -2
  4. {dcmspec-0.2.2 → dcmspec-0.3.0}/pyproject.toml +3 -1
  5. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/.DS_Store +0 -0
  6. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/iodattributes.py +28 -5
  7. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/modattributes.py +36 -13
  8. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/csv_table_spec_parser.py +16 -7
  9. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/doc_handler.py +15 -0
  10. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/dom_table_spec_parser.py +67 -34
  11. dcmspec-0.3.0/src/dcmspec/iod_spec_printer.py +254 -0
  12. dcmspec-0.3.0/src/dcmspec/spec_printer.py +414 -0
  13. dcmspec-0.2.2/src/dcmspec/iod_spec_printer.py +0 -59
  14. dcmspec-0.2.2/src/dcmspec/spec_printer.py +0 -131
  15. {dcmspec-0.2.2 → dcmspec-0.3.0}/LICENSE +0 -0
  16. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/__init__.py +0 -0
  17. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/.DS_Store +0 -0
  18. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/__init__.py +0 -0
  19. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/__init__.py +0 -0
  20. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/dataelements.py +0 -0
  21. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/iodmodules.py +0 -0
  22. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/tdwiicontent.py +0 -0
  23. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/uidvalues.py +0 -0
  24. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/upsdimseattributes.py +0 -0
  25. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/cli/upsioddimseattributes.py +0 -0
  26. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/README.md +0 -0
  27. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/__init__.py +0 -0
  28. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/config/README.md +0 -0
  29. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/config/iod_explorer_config.json +0 -0
  30. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/config/iod_explorer_config_debug.json +0 -0
  31. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/config/iod_explorer_config_example.json +0 -0
  32. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/config/iod_explorer_config_minimal_logging.json +0 -0
  33. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/apps/ui/iod_explorer/iod_explorer.py +0 -0
  34. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/config.py +0 -0
  35. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/dom_utils.py +0 -0
  36. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/iod_spec_builder.py +0 -0
  37. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/json_spec_store.py +0 -0
  38. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/module_registry.py +0 -0
  39. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/pdf_doc_handler.py +0 -0
  40. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/progress.py +0 -0
  41. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/service_attribute_defaults.py +0 -0
  42. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/service_attribute_model.py +0 -0
  43. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/spec_factory.py +0 -0
  44. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/spec_merger.py +0 -0
  45. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/spec_model.py +0 -0
  46. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/spec_parser.py +0 -0
  47. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/spec_store.py +0 -0
  48. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/ups_xhtml_doc_handler.py +0 -0
  49. {dcmspec-0.2.2 → dcmspec-0.3.0}/src/dcmspec/xhtml_doc_handler.py +0 -0
@@ -2,6 +2,35 @@
2
2
 
3
3
  These release notes summarize key changes, improvements, and breaking updates for each version of **dcmspec**.
4
4
 
5
+ ## [0.3.0] - 2025-11-27
6
+
7
+ ### Added
8
+
9
+ - CSV output mode in `SpecPrinter` via `print_csv` method
10
+ - Excel OOXML output mode in `SpecPrinter` via `print_xlsx` method
11
+ - Optional `output` parameter in `SpecPrinter` for writing to files
12
+ - Optional `column_width` parameter in `print_table` and `print_xlsx`
13
+ - `--print-mode` option in `modattributes` CLI for CSV and OOXML output
14
+ - `--output` option in `modattributes` CLI for directing output to files
15
+
16
+ ### Changed
17
+
18
+ - Switch PyPI version badge in `README` from `badge.fury.io` to `shields.io` for faster updates
19
+ - Update modattributes CLI example for CSV and XLSX output
20
+ - Improved nesting level color schemes for consistency and better contrast and readability.
21
+
22
+ ### Fixed
23
+
24
+ - Correct handling of `include_table` names in `DOMTableSpecParser`
25
+ - Correct handling of empty node attributes in `SpecPrinter`
26
+
27
+ ## [0.2.3] - 2025-09-29
28
+
29
+ ### Fixed
30
+
31
+ - Hotfix: Force UTF-8 decoding for DICOM standard XHTML downloads to prevent mojibake when server omits charset ([#85](https://github.com/dwikler/dcmspec/issues/85)).
32
+ - Hotfix: Add missing `progress_observer` argument to `CSVTableSpecParser.parse` for interface compatibility and to prevent `TypeError` when used with `SpecFactory` ([#86](https://github.com/dwikler/dcmspec/issues/86)).
33
+
5
34
  ## [0.2.2] - 2025-09-25
6
35
 
7
36
  ### Fixed
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dcmspec
3
- Version: 0.2.2
3
+ Version: 0.3.0
4
4
  Summary: Toolkit for extracting, parsing, and processing DICOM specifications.
5
5
  License: Apache-2.0
6
6
  License-File: LICENSE
@@ -25,8 +25,10 @@ Provides-Extra: pdf
25
25
  Requires-Dist: anytree (>=2.13.0,<3.0.0)
26
26
  Requires-Dist: bs4 (>=0.0.2,<0.0.3)
27
27
  Requires-Dist: camelot-py (>=1.0.0,<2.0.0) ; extra == "pdf"
28
+ Requires-Dist: html2text (>=2025.4.15,<2026.0.0)
28
29
  Requires-Dist: lxml (>=5.4.0,<6.0.0)
29
30
  Requires-Dist: opencv-python-headless (>=4.12.0.88,<5.0.0) ; extra == "pdf"
31
+ Requires-Dist: openpyxl (>=3.1.5,<4.0.0)
30
32
  Requires-Dist: openpyxl (>=3.1.5,<4.0.0) ; extra == "pdf"
31
33
  Requires-Dist: pandas (>=2.3.1,<3.0.0) ; extra == "pdf"
32
34
  Requires-Dist: pdfplumber (>=0.11.7,<0.12.0) ; extra == "pdf"
@@ -43,9 +45,9 @@ Project-URL: Source, https://github.com/dwikler/dcmspec
43
45
  Description-Content-Type: text/markdown
44
46
 
45
47
  [![tests](https://github.com/dwikler/dcmspec/actions/workflows/test.yml/badge.svg)](https://github.com/dwikler/dcmspec/actions/workflows/test.yml)
46
- [![PyPI version](https://badge.fury.io/py/dcmspec.svg)](https://badge.fury.io/py/dcmspec)
48
+ [![PyPI version](https://img.shields.io/pypi/v/dcmspec.svg)](https://pypi.org/project/dcmspec/)
47
49
  [![Python versions](https://img.shields.io/pypi/pyversions/dcmspec.svg)](https://pypi.org/project/dcmspec/)
48
- [![DOI](https://zenodo.org/badge/DOI/10.5281/zenodo.17206999.svg)](https://doi.org/10.5281/zenodo.17206999)
50
+ [![DOI](https://zenodo.org/badge/DOI/10.5281/zenodo.17742287.svg)](https://doi.org/10.5281/zenodo.17742287)
49
51
 
50
52
  # dcmspec
51
53
 
@@ -1,7 +1,7 @@
1
1
  [![tests](https://github.com/dwikler/dcmspec/actions/workflows/test.yml/badge.svg)](https://github.com/dwikler/dcmspec/actions/workflows/test.yml)
2
- [![PyPI version](https://badge.fury.io/py/dcmspec.svg)](https://badge.fury.io/py/dcmspec)
2
+ [![PyPI version](https://img.shields.io/pypi/v/dcmspec.svg)](https://pypi.org/project/dcmspec/)
3
3
  [![Python versions](https://img.shields.io/pypi/pyversions/dcmspec.svg)](https://pypi.org/project/dcmspec/)
4
- [![DOI](https://zenodo.org/badge/DOI/10.5281/zenodo.17206999.svg)](https://doi.org/10.5281/zenodo.17206999)
4
+ [![DOI](https://zenodo.org/badge/DOI/10.5281/zenodo.17742287.svg)](https://doi.org/10.5281/zenodo.17742287)
5
5
 
6
6
  # dcmspec
7
7
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "dcmspec"
3
- version = "0.2.2"
3
+ version = "0.3.0"
4
4
  description = "Toolkit for extracting, parsing, and processing DICOM specifications."
5
5
  authors = [{ name = "David Wikler", email = "david.wikler@ulb.be" }]
6
6
  license = { text = "Apache-2.0" }
@@ -29,6 +29,8 @@ dependencies = [
29
29
  "requests (>=2.32.3,<3.0.0)",
30
30
  "lxml (>=5.4.0,<6.0.0)",
31
31
  "rich (>=14.0.0,<15.0.0)",
32
+ "html2text (>=2025.4.15,<2026.0.0)",
33
+ "openpyxl (>=3.1.5,<4.0.0)",
32
34
  ]
33
35
 
34
36
  [project.urls]
@@ -58,12 +58,25 @@ def main():
58
58
  parser.add_argument("--config", help="Path to the configuration file")
59
59
  parser.add_argument(
60
60
  "--print-mode",
61
- choices=["table", "tree", "none"],
61
+ choices=["table", "tree", "csv", "xlsx", "none"],
62
62
  default="table",
63
- help="Print as 'table' (default), 'tree', or 'none' to skip printing"
63
+ help="Print as 'table' (default), 'tree', 'csv', 'xlsx' (requires --output), or 'none' to skip printing"
64
+ )
65
+ parser.add_argument(
66
+ "--no-color",
67
+ action="store_true",
68
+ help="Disable colorized output (default: color enabled for terminal, disabled for file output)"
69
+ )
70
+ parser.add_argument(
71
+ "--output",
72
+ type=str,
73
+ help="Path to output file. If not specified (and not xlsx), prints to stdout."
64
74
  )
65
75
  args = parser.parse_args()
66
76
 
77
+ if args.print_mode == "xlsx" and not args.output:
78
+ parser.error("--output is required when --print-mode xlsx")
79
+
67
80
  cache_file_name = "Part3.xhtml"
68
81
  model_file_name = f"Part3_{args.table}_expanded.json"
69
82
  table_id = args.table
@@ -114,11 +127,21 @@ def main():
114
127
  )
115
128
 
116
129
  # Print the model
117
- printer = IODSpecPrinter(model)
130
+ printer = IODSpecPrinter(model, output=getattr(args, "output", None))
131
+
132
+ # Only disable color if --no-color is set; SpecPrinter disables color for file outputs automatically
133
+ use_color = not args.no_color
134
+
118
135
  if args.print_mode == "tree":
119
- printer.print_tree(colorize=True)
136
+ printer.print_tree(
137
+ attr_names=["elem_tag", "elem_type", "elem_name"], attr_widths=[11, 2, 100], colorize=use_color
138
+ )
120
139
  elif args.print_mode == "table":
121
- printer.print_table(colorize=True)
140
+ printer.print_table(colorize=use_color, column_widths=[30, 11, 4, 60])
141
+ elif args.print_mode == "csv":
142
+ printer.print_csv(colorize=use_color)
143
+ elif args.print_mode == "xlsx":
144
+ printer.print_xlsx(column_widths=[60, 16, 10, 100], colorize=use_color)
122
145
  # else: do not print anything if print_mode == "none"
123
146
 
124
147
  if __name__ == "__main__":
@@ -4,7 +4,7 @@ Features:
4
4
  - Download and parse DICOM Module Attributes tables from Part 3 of the DICOM standard.
5
5
  - Optionally merge additional information (VR, VM, Keyword, Status) from Part 6.
6
6
  - Cache the resulting model as a JSON file for future runs and as a structured representation of the standard.
7
- - Print the resulting model as a table or tree.
7
+ - Print the resulting model as a table, tree, csv, or xlsx.
8
8
  - Supports caching, configuration files, and command-line options for flexible workflows.
9
9
 
10
10
  Usage:
@@ -113,12 +113,12 @@ def main():
113
113
 
114
114
  The tool parses the specified Module Attributes table to extract all attributes, tags, types,
115
115
  and descriptions for the module. Optionally, it can merge in VR, VM, Keyword, or Status information
116
- from Part 6. The output can be printed as a table or tree.
116
+ from Part 6. The output can be printed as a table, tree, csv, or xlsx (xlsx requires --output).
117
117
 
118
118
  The resulting model is cached as a JSON file. The primary purpose of this cache file is to provide
119
- a structured, machine-readable representation of the module's attributes, which can be used for further processing
120
- or integration in other tools. As a secondary benefit, the cache file is also used to speed up subsequent runs
121
- of the CLI scripts.
119
+ a structured, machine-readable representation of the module's attributes, which can be used for
120
+ further processing or integration in other tools. As a secondary benefit, the cache file is also used
121
+ to speed up subsequent runs of the CLI scripts.
122
122
 
123
123
  Usage:
124
124
  poetry run python -m src.dcmspec.apps.cli.modattributes <table_id> [options]
@@ -129,14 +129,17 @@ def main():
129
129
  --include-depth (int): Depth to which included tables should be parsed (default: unlimited).
130
130
  --force-parse: Force reparsing of the DOM and regeneration of the JSON model.
131
131
  --force-download: Force download of the input file and regeneration of the model.
132
- --print-mode (str): Print as 'table' (default), 'tree', or 'none' to skip printing.
132
+ --print-mode (str): Print as 'table' (default), 'tree', 'csv', 'xlsx', or 'none' to skip printing.
133
+ 'xlsx' requires --output.
134
+ --output (str): Path to output file. If not specified (and not xlsx), prints to stdout.
133
135
  --add-part6 (list): Specification(s) to merge from Part 6 (e.g., --add-part6 VR VM).
134
136
  --force-update: Force update of the specifications merged from part 6, even if cached.
135
137
  -d, --debug: Enable debug logging to the console.
136
138
  -v, --verbose: Enable verbose (info-level) logging to the console.
137
139
 
138
- Example:
140
+ Examples:
139
141
  poetry run python -m src.dcmspec.apps.cli.modattributes table_C.7-1 --add-part6 VR VM
142
+ poetry run python -m src.dcmspec.apps.cli.modattributes table_C.7-1 --print-mode xlsx --output module.xlsx
140
143
 
141
144
  """
142
145
  # Parse command-line arguments
@@ -164,9 +167,19 @@ def main():
164
167
  )
165
168
  parser.add_argument(
166
169
  "--print-mode",
167
- choices=["table", "tree", "none"],
170
+ choices=["table", "tree", "csv", "xlsx", "none"],
168
171
  default="table",
169
- help="Print as 'table' (default), 'tree', or 'none' to skip printing"
172
+ help="Print as 'table' (default), 'tree', 'csv', 'xlsx' (requires --output), or 'none' to skip printing"
173
+ )
174
+ parser.add_argument(
175
+ "--no-color",
176
+ action="store_true",
177
+ help="Disable colorized output (default: color enabled for terminal, disabled for file output)"
178
+ )
179
+ parser.add_argument(
180
+ "--output",
181
+ type=str,
182
+ help="Path to output file. If not specified (and not xlsx), prints to stdout."
170
183
  )
171
184
  parser.add_argument(
172
185
  "--add-part6",
@@ -189,9 +202,12 @@ def main():
189
202
  action="store_true",
190
203
  help="Enable verbose (info-level) logging to the console"
191
204
  )
192
-
205
+
193
206
  args = parser.parse_args()
194
207
 
208
+ if args.print_mode == "xlsx" and not args.output:
209
+ parser.error("--output is required when --print-mode xlsx")
210
+
195
211
  # Set up logger
196
212
  logger = logging.getLogger("modattributes")
197
213
  handler = logging.StreamHandler()
@@ -254,11 +270,18 @@ def main():
254
270
  model = module_model
255
271
 
256
272
  logger.debug("Model ready for printing/output")
257
- printer = SpecPrinter(model)
273
+ printer = SpecPrinter(model, output=args.output)
274
+
275
+ # Only disable color if --no-color is set; SpecPrinter disables color for file outputs automatically
276
+ use_color = not args.no_color
258
277
  if args.print_mode == "tree":
259
- printer.print_tree(colorize=True)
278
+ printer.print_tree(attr_names=["elem_tag", "elem_type", "elem_name"], attr_widths=[11, 2, 100], colorize=use_color)
260
279
  elif args.print_mode == "table":
261
- printer.print_table(colorize=True)
280
+ printer.print_table(colorize=use_color, column_widths=[30, 11, 4, 60])
281
+ elif args.print_mode == "csv":
282
+ printer.print_csv(colorize=use_color)
283
+ elif args.print_mode == "xlsx":
284
+ printer.print_xlsx(column_widths=[60, 16, 10, 100], colorize=use_color)
262
285
  # else: do not print anything if print_mode == "none"
263
286
 
264
287
  if __name__ == "__main__":
@@ -3,10 +3,11 @@
3
3
  Provides the CSVTableSpecParser class for parsing DICOM specification tables in CSV format,
4
4
  converting them into structured in-memory representations using anytree.
5
5
  """
6
- from typing import Tuple
6
+ from typing import Dict, List, Tuple, Optional
7
7
  from anytree import Node
8
8
 
9
9
  from dcmspec.spec_parser import SpecParser
10
+ from dcmspec.progress import ProgressObserver
10
11
 
11
12
  class CSVTableSpecParser(SpecParser):
12
13
  """Base parser for DICOM Specification IHE tables in CSV-like format."""
@@ -14,10 +15,11 @@ class CSVTableSpecParser(SpecParser):
14
15
  def parse(
15
16
  self,
16
17
  table: dict,
17
- column_to_attr,
18
- name_attr="elem_name",
19
- table_id=None,
20
- include_depth=None,
18
+ column_to_attr: dict,
19
+ name_attr: str = "elem_name",
20
+ table_id: Optional[str] = None,
21
+ include_depth: Optional[int] = None,
22
+ progress_observer: Optional[ProgressObserver] = None,
21
23
  ) -> Tuple[Node, Node]:
22
24
  """Parse specification metadata and content from a single table dict.
23
25
 
@@ -27,11 +29,18 @@ class CSVTableSpecParser(SpecParser):
27
29
  name_attr (str): The attribute to use for node names.
28
30
  table_id (str, optional): Table identifier for model parsing.
29
31
  include_depth (int, optional): The depth to which included tables should be parsed.
32
+ progress_observer (Optional[ProgressObserver]):
33
+ Accepted for interface compatibility, but ignored in this parser.
34
+ Included so that this method can be called with the same arguments as other table parsers.
30
35
 
31
36
  Returns:
32
37
  tuple: (metadata_node, content_node)
33
38
 
34
39
  """
40
+ if progress_observer is not None and hasattr(self, "logger"):
41
+ self.logger.debug(
42
+ "Progress reporting is not supported yet for CSV parsing and will be ignored."
43
+ )
35
44
  # Use the header and data from the grouped table dict
36
45
  header = table.get("header", [])
37
46
  data = table.get("data", [])
@@ -47,8 +56,8 @@ class CSVTableSpecParser(SpecParser):
47
56
 
48
57
  def parse_table(
49
58
  self,
50
- tables: list, # List of tables, each a list of rows (list of str)
51
- column_to_attr: dict,
59
+ tables: List[List[List[str]]], # List of tables, each a list of rows (list of str)
60
+ column_to_attr: Dict[int, str],
52
61
  name_attr: str = "elem_name",
53
62
  ) -> Node:
54
63
  """Build a tree from tables using column mapping and '>' nesting logic.
@@ -94,6 +94,8 @@ class DocHandler:
94
94
  try:
95
95
  with requests.get(url, timeout=30, stream=True, headers={"Accept-Encoding": "identity"}) as response:
96
96
  response.raise_for_status()
97
+ self._set_response_encoding(response)
98
+
97
99
  total = int(response.headers.get('content-length', 0))
98
100
  chunk_size = 8192
99
101
  if binary:
@@ -109,6 +111,19 @@ class DocHandler:
109
111
  self.logger.error(f"Failed to save file {file_path}: {e}")
110
112
  raise RuntimeError(f"Failed to save file {file_path}: {e}") from e
111
113
 
114
+ def _set_response_encoding(self, response):
115
+ """Set response.encoding to UTF-8 only if the Content-Type header does not specify a charset.
116
+
117
+ Force utf-8 decoding if the web server does not specify the charset in the HTTP response
118
+ Content-Type header (DICOM standard XHTML files are always UTF-8 encoded).
119
+ """
120
+ content_type = response.headers.get("Content-Type", "")
121
+ if "charset=" not in content_type.lower():
122
+ response.encoding = "utf-8"
123
+ self.logger.debug("No charset in Content-Type header; forcing UTF-8 decoding.")
124
+ else:
125
+ self.logger.debug(f"Using server-specified encoding from Content-Type: {content_type}")
126
+
112
127
  def _report_progress(self, downloaded, total, progress_observer, last_percent):
113
128
  """Report progress if percent changed.
114
129
 
@@ -3,17 +3,20 @@
3
3
  Provides the DOMSpecParser class for parsing DICOM specification tables from XHTML documents,
4
4
  converting them into structured in-memory representations using anytree.
5
5
  """
6
+
6
7
  from contextlib import contextmanager
7
8
  import re
8
9
  import unicodedata
10
+ from typing import Any, Dict, Optional, Union
11
+
9
12
  from unidecode import unidecode
10
13
  from anytree import Node
11
14
  from bs4 import BeautifulSoup, Tag
12
- from typing import Any, Dict, Optional, Union
13
- from dcmspec.spec_parser import SpecParser
15
+ import html2text
14
16
 
17
+ from dcmspec.spec_parser import SpecParser
15
18
  from dcmspec.dom_utils import DOMUtils
16
- from dcmspec.progress import Progress, ProgressStatus, calculate_percent
19
+ from dcmspec.progress import Progress, ProgressObserver, ProgressStatus, calculate_percent
17
20
 
18
21
  class DOMTableSpecParser(SpecParser):
19
22
  """Parser for DICOM specification tables in XHTML DOM format.
@@ -43,7 +46,7 @@ class DOMTableSpecParser(SpecParser):
43
46
  column_to_attr: Dict[int, str],
44
47
  name_attr: str,
45
48
  include_depth: Optional[int] = None, # None means unlimited
46
- progress_observer: Optional[Any] = None,
49
+ progress_observer: Optional[ProgressObserver] = None,
47
50
  skip_columns: Optional[list[int]] = None,
48
51
  unformatted: Optional[Union[bool, Dict[int, bool]]] = True,
49
52
  ) -> tuple[Node, Node]:
@@ -59,7 +62,7 @@ class DOMTableSpecParser(SpecParser):
59
62
  name_attr (str): The attribute name to use for building node names.
60
63
  include_depth (Optional[int], optional): The depth to which included tables should be parsed.
61
64
  None means unlimited.
62
- progress_observer (Optional[object], optional): Optional observer to report download progress.
65
+ progress_observer (Optional[ProgressObserver]): Optional observer to report parsing progress.
63
66
  skip_columns (Optional[list[int]]): List of column indices to skip if the row is missing a column.
64
67
  This argument is typically set via `parser_kwargs` when using SpecFactory.
65
68
  unformatted (Optional[Union[bool, Dict[int, bool]]]):
@@ -132,7 +135,7 @@ class DOMTableSpecParser(SpecParser):
132
135
  name_attr: str,
133
136
  table_nesting_level: int = 0,
134
137
  include_depth: Optional[int] = None, # None means unlimited
135
- progress_observer: Optional[Any] = None,
138
+ progress_observer: Optional[ProgressObserver] = None,
136
139
  skip_columns: Optional[list[int]] = None,
137
140
  visited_tables: Optional[set] = None,
138
141
  unformatted_list: Optional[list[bool]] = None,
@@ -151,7 +154,7 @@ class DOMTableSpecParser(SpecParser):
151
154
  name_attr: tree node attribute name to use to build node name
152
155
  table_nesting_level: The nesting level of the table (used for recursion call only).
153
156
  include_depth: The depth to which included tables should be parsed.
154
- progress_observer (Optional[object], optional): Optional observer to report download progress.
157
+ progress_observer (Optional[ProgressObserver]): Optional observer to report parsing progress.
155
158
  skip_columns (Optional[list[int]]): List of column indices to skip if the row is missing a column.
156
159
  visited_tables (Optional[set]): Set of table IDs that have been visited to prevent infinite recursion.
157
160
  unformatted_list (Optional[list[bool]]): List of booleans indicating whether to extract each column as
@@ -287,7 +290,7 @@ class DOMTableSpecParser(SpecParser):
287
290
  unformatted_list: list[bool],
288
291
  level_nodes: Dict[int, Node],
289
292
  root: Node,
290
- progress_observer: Optional[Any] = None
293
+ progress_observer: Optional[ProgressObserver] = None
291
294
  ) -> None:
292
295
  """Process all rows in the table, handling recursion, nesting, and node creation."""
293
296
  rows = table.find_all("tr")[1:]
@@ -506,6 +509,7 @@ class DOMTableSpecParser(SpecParser):
506
509
 
507
510
  return logical_cells, logical_col_idx, physical_col_idx
508
511
 
512
+
509
513
  def _extract_cell_value(
510
514
  self,
511
515
  cell: Tag,
@@ -518,11 +522,27 @@ class DOMTableSpecParser(SpecParser):
518
522
  if unformatted_list and logical_col_idx < len(unformatted_list)
519
523
  else True
520
524
  )
521
- if use_unformatted:
522
- return self._clean_extracted_text(cell.get_text(separator="\n", strip=True))
523
- else:
525
+
526
+ # Guard clause: if formatted HTML is required, return content as-is
527
+ if not use_unformatted:
528
+ # Keep original HTML content
524
529
  return self._clean_extracted_text(cell.decode_contents())
525
530
 
531
+ # Use html2text for better readability of unformatted text extraction
532
+ converter = self._create_html2text_converter()
533
+ raw_text = converter.handle(str(cell))
534
+ return self._clean_extracted_text(raw_text)
535
+
536
+
537
+ def _create_html2text_converter(self) -> html2text.HTML2Text:
538
+ """Create and configure an html2text converter for consistent text extraction."""
539
+ converter = html2text.HTML2Text()
540
+ converter.ignore_links = True # Remove URLs
541
+ converter.ignore_images = True # Remove image references
542
+ converter.ignore_emphasis = True # Remove Markdown emphasis
543
+ converter.body_width = 0 # Disable word wrapping
544
+ return converter
545
+
526
546
  def _update_rowspan_trackers(
527
547
  self,
528
548
  physical_col_idx: int,
@@ -781,15 +801,7 @@ class DOMTableSpecParser(SpecParser):
781
801
  return header
782
802
 
783
803
  def _clean_extracted_text(self, text: str) -> str:
784
- """Clean extracted text using Unicode normalization and regex.
785
-
786
- Args:
787
- text (str): The text to be cleaned.
788
-
789
- Returns:
790
- str: The cleaned text.
791
-
792
- """
804
+ """Clean extracted text using Unicode normalization and regex."""
793
805
  # Normalize unicode characters to compatibility form
794
806
  cleaned = unicodedata.normalize('NFKC', text)
795
807
 
@@ -805,27 +817,48 @@ class DOMTableSpecParser(SpecParser):
805
817
  # Remove stray  character
806
818
  cleaned = cleaned.replace('\u00c2', '')
807
819
 
820
+ # Collapse multiple newlines (including those separated by spaces/tabs) into a single newline
821
+ cleaned = re.sub(r'(\n\s*){2,}', '\n', cleaned)
822
+
808
823
  return cleaned.strip()
809
824
 
810
- def _sanitize_string(self, input_string: str) -> str:
811
- """Sanitize string to use it as a node attribute name.
825
+ @staticmethod
826
+ def _sanitize_string(input_string: str) -> str:
827
+ r"""Sanitize a string to make it safe for use as a node attribute name.
812
828
 
813
- - Convert non-ASCII characters to closest ASCII equivalents
814
- - Replace space characters and slashes with underscores
815
- - Replace parentheses characters with dashes
829
+ Transformations applied:
830
+ - Convert to lowercase.
831
+ - Transliterate non-ASCII characters to ASCII.
832
+ - Replace spaces, slashes, newlines, and dots with underscores ("_").
833
+ - Replace parentheses with dashes ("-").
834
+ - Remove all characters except letters, digits, underscores, and dashes.
835
+ - Collapse multiple consecutive underscores into a single underscore.
836
+ - Remove leading and trailing underscores for cleanliness.
837
+ - Return a default name if the result is empty after sanitization.
816
838
 
817
839
  Args:
818
- input_string (str): The string to be sanitized.
840
+ input_string (str): The original string to sanitize.
819
841
 
820
842
  Returns:
821
- str: The sanitized string.
843
+ str: A sanitized version of the input string, suitable for use as an identifier.
844
+ or "unnamed_node" if sanitization results in an empty string.
845
+
846
+ Example:
847
+ >>> DOMTableSpecParser._sanitize_string(
848
+ '>>Include\\nTable C.36.2.2.19-1 "RT Beam Limiting Device Definition Macro Attributes"\\n.'
849
+ )
850
+ 'include_table_c_36_2_2_19-1_rt_beam_limiting_device_definition_macro_attributes'
851
+ >>> DOMTableSpecParser._sanitize_string('...')
852
+ 'unnamed_node'
822
853
 
823
854
  """
824
- # Normalize the string to NFC form and transliterate to ASCII
825
855
  normalized_str = unidecode(input_string.lower())
826
- # Replace spaces and slashes with underscores, parentheses with dashes, and single quotes with underscores
827
- return re.sub(
828
- r"[ /\-()']",
829
- lambda match: "-" if match.group(0) in "()" else "_",
830
- normalized_str,
831
- )
856
+ sanitized = re.sub(r"[ /\n\\.]", "_", normalized_str) # spaces, slashes, newlines, dots → _
857
+ sanitized = re.sub(r"[()]", "-", sanitized) # parentheses → -
858
+ sanitized = re.sub(r"[^a-z0-9_-]", "", sanitized) # remove other chars
859
+ sanitized = re.sub(r"_+", "_", sanitized) # collapse multiple underscores
860
+ sanitized = sanitized.strip("_") # remove leading/trailing underscores
861
+
862
+ # Fallback to default name if sanitization resulted in empty string
863
+ return sanitized or "unnamed_node"
864
+