ctrl-kd 4.5.0__tar.gz → 4.5.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. {ctrl_kd-4.5.0/src/ctrl_kd.egg-info → ctrl_kd-4.5.1}/PKG-INFO +8 -2
  2. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/README.md +7 -1
  3. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/pyproject.toml +8 -3
  4. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1/src/ctrl_kd.egg-info}/PKG-INFO +8 -2
  5. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrl_kd.egg-info/SOURCES.txt +4 -0
  6. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/__init__.py +1 -1
  7. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/core.py +63 -2
  8. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/emit.py +61 -7
  9. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/pdf.py +801 -243
  10. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_ctrlkd.py +595 -117
  11. ctrl_kd-4.5.1/tests/test_fidelity_gate.py +642 -0
  12. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_lj6dtp_colour_restore.py +3 -3
  13. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_lj6dtp_heading_face.py +3 -2
  14. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_lj6dtp_pcl_rectangles.py +3 -2
  15. ctrl_kd-4.5.1/tests/test_load_plugins.py +107 -0
  16. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_modern_lint.py +10 -8
  17. ctrl_kd-4.5.1/tests/test_paper_verdicts.py +90 -0
  18. ctrl_kd-4.5.1/tests/test_pcl_fidelity.py +102 -0
  19. ctrl_kd-4.5.1/tests/test_pcl_tolerance.py +567 -0
  20. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_pix.py +6 -7
  21. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_polarity_gate.py +1 -1
  22. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_printed_fidelity.py +96 -23
  23. ctrl_kd-4.5.1/tests/test_samples.py +140 -0
  24. ctrl_kd-4.5.1/tests/test_sawyer_corpus.py +208 -0
  25. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_screenplay_detection.py +2 -2
  26. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_style_leading.py +96 -56
  27. ctrl_kd-4.5.0/tests/test_fidelity_gate.py +0 -231
  28. ctrl_kd-4.5.0/tests/test_samples.py +0 -117
  29. ctrl_kd-4.5.0/tests/test_sawyer_corpus.py +0 -209
  30. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/LICENSE +0 -0
  31. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/setup.cfg +0 -0
  32. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrl_kd.egg-info/dependency_links.txt +0 -0
  33. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrl_kd.egg-info/entry_points.txt +0 -0
  34. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrl_kd.egg-info/top_level.txt +0 -0
  35. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/afm.py +0 -0
  36. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/cli.py +0 -0
  37. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/convert.py +0 -0
  38. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/fontmap.py +0 -0
  39. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/info.py +0 -0
  40. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/layout.py +0 -0
  41. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/pictures.py +0 -0
  42. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/pix.py +0 -0
  43. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/piximg.py +0 -0
  44. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/samples/LYING.WS +0 -0
  45. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/samples/OCAPTAIN.WS +0 -0
  46. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/samples/TWAINLET.WS +0 -0
  47. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/samples/WARPRAYR.WS +0 -0
  48. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/symbolmap.py +0 -0
  49. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/typestyles.py +0 -0
  50. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/writer.py +0 -0
  51. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/src/ctrlkd/wschange.py +0 -0
  52. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_endnote_leading_gap.py +0 -0
  53. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_flags_toc_inline.py +0 -0
  54. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_glyph_aspect.py +0 -0
  55. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_layout_marks.py +0 -0
  56. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_lj6dtp_bullet_glyph.py +0 -0
  57. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_lj6dtp_char_substitution.py +0 -0
  58. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_lj6dtp_hp_patterns.py +0 -0
  59. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_lj6dtp_legend_line_spacing.py +0 -0
  60. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_lj6dtp_shading_table_rules.py +0 -0
  61. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_lj6dtp_table_rule_weight.py +0 -0
  62. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_modern_box_regions.py +0 -0
  63. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_modern_line_spacing.py +0 -0
  64. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_note_rulings_20260824.py +0 -0
  65. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_pictures.py +0 -0
  66. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_screenplay_pdf.py +0 -0
  67. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_screenplay_rendering.py +0 -0
  68. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_sentence_spacing_n9.py +0 -0
  69. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_verse_quote_couplet.py +0 -0
  70. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_verse_spacing.py +0 -0
  71. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_writer.py +0 -0
  72. {ctrl_kd-4.5.0 → ctrl_kd-4.5.1}/tests/test_wschange.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ctrl-kd
3
- Version: 4.5.0
3
+ Version: 4.5.1
4
4
  Summary: Convert WordStar 4-7 documents and print-to-disk files to text, Markdown, HTML, RTF, or PDF. ^KD: save and done.
5
5
  Author: Jon Michaels
6
6
  License: MIT
@@ -66,6 +66,7 @@ More information: **[FAQ.md](FAQ.md)**, and **[ERAS.md](ERAS.md)**.
66
66
  $ brew install jonmichaels/tap/ctrl-kd # macOS / Linuxbrew
67
67
  $ pipx install ctrl-kd # or: pip install ctrl-kd
68
68
  ```
69
+ Download Windows x86_64: [Latest Version](https://github.com/jonmichaels/ctrl-kd/releases/latest/download/ctrl-kd-windows-x86_64.zip)
69
70
 
70
71
  Python ≥ 3.9, no dependencies. Library API: `ctrlkd.convert(data, to='html')`.
71
72
 
@@ -97,7 +98,12 @@ My own test files are personal and are not distributed — this repo's tests use
97
98
  synthetic fixtures and some public domain docs I retyped in WordStar 4 and
98
99
  WordStar 7 in DOSBox-X, plus Robert J. Sawyer's public WS7 archive (opt-in,
99
100
  `pytest -m sawyer`; see `tests/SAWYER-CORPUS.md`) you can run your own tests
100
- against that if you have a copy.
101
+ against that if you have a copy: download the archive, point
102
+ `CTRLKD_SAWYER_ARCHIVE` at its top-level directory (the one holding
103
+ `CONVERT.WS`, `INSET/`, `ARTICLES/`), and verify the path before arming —
104
+ `tests/SAWYER-CORPUS.md` has a one-line smoke check. A `CTRLKD_PRIVATE_CORPUS`
105
+ variable also exists, for my own private regression fixtures — it has no
106
+ effect unless you're me.
101
107
 
102
108
  ## Credits
103
109
 
@@ -47,6 +47,7 @@ More information: **[FAQ.md](FAQ.md)**, and **[ERAS.md](ERAS.md)**.
47
47
  $ brew install jonmichaels/tap/ctrl-kd # macOS / Linuxbrew
48
48
  $ pipx install ctrl-kd # or: pip install ctrl-kd
49
49
  ```
50
+ Download Windows x86_64: [Latest Version](https://github.com/jonmichaels/ctrl-kd/releases/latest/download/ctrl-kd-windows-x86_64.zip)
50
51
 
51
52
  Python ≥ 3.9, no dependencies. Library API: `ctrlkd.convert(data, to='html')`.
52
53
 
@@ -78,7 +79,12 @@ My own test files are personal and are not distributed — this repo's tests use
78
79
  synthetic fixtures and some public domain docs I retyped in WordStar 4 and
79
80
  WordStar 7 in DOSBox-X, plus Robert J. Sawyer's public WS7 archive (opt-in,
80
81
  `pytest -m sawyer`; see `tests/SAWYER-CORPUS.md`) you can run your own tests
81
- against that if you have a copy.
82
+ against that if you have a copy: download the archive, point
83
+ `CTRLKD_SAWYER_ARCHIVE` at its top-level directory (the one holding
84
+ `CONVERT.WS`, `INSET/`, `ARTICLES/`), and verify the path before arming —
85
+ `tests/SAWYER-CORPUS.md` has a one-line smoke check. A `CTRLKD_PRIVATE_CORPUS`
86
+ variable also exists, for my own private regression fixtures — it has no
87
+ effect unless you're me.
82
88
 
83
89
  ## Credits
84
90
 
@@ -48,8 +48,11 @@ ctrlkd = ["samples/*.WS"]
48
48
  # DESELECTED here, not skipped: it never even collects into the report, so
49
49
  # a bare run's summary line is an honest denominator.
50
50
  #
51
- # Arming (see tests/sawyer_fixture.py, conftest.py):
52
- # sawyer CTRLKD_SAWYER_ARCHIVE=/path (legacy alias: CTRLKD_CORPUS_SOURCE)
51
+ # Arming (see tests/sawyer_fixture.py, conftest.py, tools/pcl_tolerance.py):
52
+ # sawyer CTRLKD_SAWYER_ARCHIVE=/path
53
+ # pcl CTRLKD_PRIVATE_CORPUS=/path (2026-09-05, planning #197)
54
+ # paper CTRLKD_PRIVATE_CORPUS=/path (2026-09-06, planning #200; verdicts
55
+ # file path itself overridable with CTRLKD_PAPER_VERDICTS)
53
56
  #
54
57
  # tools/run-full-suite.sh overrides this filter (`-o addopts=""`) for an
55
58
  # armed run, where an unarmed gate FAILS instead of silently not running --
@@ -57,5 +60,7 @@ ctrlkd = ["samples/*.WS"]
57
60
  testpaths = ["tests"]
58
61
  markers = [
59
62
  "sawyer: opt-in tier 2 -- an explicit, committed list of documents from Robert J. Sawyer's public WordStar 7 archive (arm with CTRLKD_SAWYER_ARCHIVE)",
63
+ "pcl: opt-in -- coordinate-level fidelity gate against real WordStar 7 LaserJet PCL captures (arm with CTRLKD_PRIVATE_CORPUS); see tools/pcl_tolerance.py and tests/test_pcl_fidelity.py",
64
+ "paper: opt-in -- machine-readable verdicts for the 69 M479fdw paper-scan pages (arm with CTRLKD_PRIVATE_CORPUS); see tools/PAPER-VERDICTS.md and tests/test_paper_verdicts.py",
60
65
  ]
61
- addopts = "-m \"not sawyer\""
66
+ addopts = "-m \"not sawyer and not pcl and not paper\""
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ctrl-kd
3
- Version: 4.5.0
3
+ Version: 4.5.1
4
4
  Summary: Convert WordStar 4-7 documents and print-to-disk files to text, Markdown, HTML, RTF, or PDF. ^KD: save and done.
5
5
  Author: Jon Michaels
6
6
  License: MIT
@@ -66,6 +66,7 @@ More information: **[FAQ.md](FAQ.md)**, and **[ERAS.md](ERAS.md)**.
66
66
  $ brew install jonmichaels/tap/ctrl-kd # macOS / Linuxbrew
67
67
  $ pipx install ctrl-kd # or: pip install ctrl-kd
68
68
  ```
69
+ Download Windows x86_64: [Latest Version](https://github.com/jonmichaels/ctrl-kd/releases/latest/download/ctrl-kd-windows-x86_64.zip)
69
70
 
70
71
  Python ≥ 3.9, no dependencies. Library API: `ctrlkd.convert(data, to='html')`.
71
72
 
@@ -97,7 +98,12 @@ My own test files are personal and are not distributed — this repo's tests use
97
98
  synthetic fixtures and some public domain docs I retyped in WordStar 4 and
98
99
  WordStar 7 in DOSBox-X, plus Robert J. Sawyer's public WS7 archive (opt-in,
99
100
  `pytest -m sawyer`; see `tests/SAWYER-CORPUS.md`) you can run your own tests
100
- against that if you have a copy.
101
+ against that if you have a copy: download the archive, point
102
+ `CTRLKD_SAWYER_ARCHIVE` at its top-level directory (the one holding
103
+ `CONVERT.WS`, `INSET/`, `ARTICLES/`), and verify the path before arming —
104
+ `tests/SAWYER-CORPUS.md` has a one-line smoke check. A `CTRLKD_PRIVATE_CORPUS`
105
+ variable also exists, for my own private regression fixtures — it has no
106
+ effect unless you're me.
101
107
 
102
108
  ## Credits
103
109
 
@@ -42,10 +42,14 @@ tests/test_lj6dtp_legend_line_spacing.py
42
42
  tests/test_lj6dtp_pcl_rectangles.py
43
43
  tests/test_lj6dtp_shading_table_rules.py
44
44
  tests/test_lj6dtp_table_rule_weight.py
45
+ tests/test_load_plugins.py
45
46
  tests/test_modern_box_regions.py
46
47
  tests/test_modern_line_spacing.py
47
48
  tests/test_modern_lint.py
48
49
  tests/test_note_rulings_20260824.py
50
+ tests/test_paper_verdicts.py
51
+ tests/test_pcl_fidelity.py
52
+ tests/test_pcl_tolerance.py
49
53
  tests/test_pictures.py
50
54
  tests/test_pix.py
51
55
  tests/test_polarity_gate.py
@@ -8,4 +8,4 @@ from .pdf import emit_pdf # registers the 'pdf' format
8
8
  from .convert import convert, select_notes, DEFAULT_NOTE_KINDS, ALL_NOTE_KINDS
9
9
  from .info import document_info
10
10
 
11
- __version__ = '4.5.0'
11
+ __version__ = '4.5.1'
@@ -1260,6 +1260,37 @@ class Document:
1260
1260
  # FONT (not just its text) mid-file keeps only the last one seen.
1261
1261
  header_fonts: dict = field(default_factory=dict)
1262
1262
  footer_fonts: dict = field(default_factory=dict)
1263
+ # A `.h#`/`.f#` line's own RIGHT/CENTER/DECIMAL-align tab (WSFORMAT type-9
1264
+ # symmetric sequence), when its typed text carries one: (char_idx, cols,
1265
+ # abs_hmi) -- char_idx is where the tab's own BAKED padding run (already
1266
+ # expanded to `cols` space characters when the file was last saved,
1267
+ # exactly like a body span's tab, see `_symmetric_blocks`' `_tab_columns`)
1268
+ # sits within the FINAL decoded `headers[line]`/`footers[line]` string;
1269
+ # `abs_hmi` is the tab's own absolute target column (content[2:4], same
1270
+ # field a body span's `tabN` mark carries), from the document's own left
1271
+ # reference, in HMI (1/1800in).
1272
+ #
1273
+ # WHY THIS EXISTS (planning #202, -README's own running head): a body
1274
+ # span's tab gets this same absolute target as a MARK a Printed-mode
1275
+ # renderer can consult live (`_decode_spans`' `tab_target_at`) -- but
1276
+ # `.h#`/`.f#` text is stored as a flat, already-`#`-unaware STRING, so the
1277
+ # `cols` count baked into it was fixed FOREVER at whatever the page
1278
+ # number's width was assumed to be WHEN THE FILE WAS SAVED (always 1
1279
+ # digit -- WordStar's own screen shows the LITERAL '#' token, one column,
1280
+ # never the eventual printed number). Real WS7 re-evaluates the tab at
1281
+ # PRINT TIME, using THAT page's own actual page-number width, so a
1282
+ # right-aligned "WordStar 7.0 Archive / #" header shifts one column
1283
+ # LEFT the moment the page number grows a second digit (measured,
1284
+ # -README.WS pages 9->10: WS7's own "WordStar" moves from x=345.6pt to
1285
+ # x=338.4pt, exactly 7.2pt = one Courier column, while the text's own
1286
+ # RIGHT edge -- "9" ends at 518.4pt, "10" also ends at 518.4pt -- stays
1287
+ # utterly fixed). Carrying `(char_idx, cols, abs_hmi)` lets the Printed
1288
+ # emitter redo that same per-page arithmetic instead of trusting the
1289
+ # baked, page-1-shaped `cols`. None when this line's text has no tab of
1290
+ # its own (the common case, and the only case before this existed) --
1291
+ # `_hf_line_ops` falls back to the baked text unchanged, byte-identical.
1292
+ header_tabs: dict = field(default_factory=dict)
1293
+ footer_tabs: dict = field(default_factory=dict)
1263
1294
  # Every .he/.h1-.h5/.fo/.f1-.f5 IN DOCUMENT ORDER, with the block it
1264
1295
  # precedes: ('H'|'F', line 1-5, text, block_index). WordStar applies a
1265
1296
  # running head from the page where it is defined -- on that page itself
@@ -2538,7 +2569,8 @@ def _parse_collect_dot(cmd: bytes, doc, encoding: str, block_index: int):
2538
2569
  doc.meta['line_numbering'] = value if value > 0 else None
2539
2570
 
2540
2571
 
2541
- def _parse_head_foot(cmd: bytes, doc, encoding: str, anchor=None, font_idx=None):
2572
+ def _parse_head_foot(cmd: bytes, doc, encoding: str, anchor=None, font_idx=None,
2573
+ tab_mark=None):
2542
2574
  """Record `.he`/`.h1`-`.h5` and `.fo`/`.f1`-`.f5` text on the Document.
2543
2575
 
2544
2576
  `.HE` and `.FO` are line 1; the numbered forms select their own line, so a
@@ -2554,6 +2586,12 @@ def _parse_head_foot(cmd: bytes, doc, encoding: str, anchor=None, font_idx=None)
2554
2586
  Font block exactly like body text can, and that block contributes no
2555
2587
  bytes of its own to the cleaned stream, so nothing but the caller's own
2556
2588
  line_marks lookup can recover it once we are down here working on bytes.
2589
+
2590
+ `tab_mark` (planning #202, -README's own running head) is the caller's
2591
+ own `line_marks` 'tab' entry for this line, if any: `(rel, abs_hmi,
2592
+ leader, cols)` -- `rel` a BYTE offset from the start of `cmd` (the SAME
2593
+ coordinate space `_symmetric_blocks` recorded it in), the rest exactly
2594
+ what a body span's own tab mark carries. See `Document.header_tabs`.
2557
2595
  """
2558
2596
  m = _HEAD_FOOT_RE.match(cmd)
2559
2597
  if not m:
@@ -2565,22 +2603,38 @@ def _parse_head_foot(cmd: bytes, doc, encoding: str, anchor=None, font_idx=None)
2565
2603
  # body uses -- control-range middles are chart glyphs, the rest are the
2566
2604
  # byte's own cp437 character.
2567
2605
  raw_txt = m.group(2)
2606
+ tab_byte_idx = tab_mark[0] - m.start(2) if tab_mark is not None else None
2607
+ tab_char_idx = None
2568
2608
  parts, pos = [], 0
2569
2609
  for t in re.finditer(rb'\x1b(.)\x1c', raw_txt, re.S):
2610
+ if (tab_char_idx is None and tab_byte_idx is not None
2611
+ and tab_byte_idx <= t.start()):
2612
+ tab_char_idx = sum(len(p) for p in parts) + (tab_byte_idx - pos)
2570
2613
  parts.append(raw_txt[pos:t.start()].decode(encoding, 'replace'))
2571
2614
  x = t.group(1)[0]
2572
2615
  parts.append(CP437_GRAPHICS[x] if x < 0x20 or x == 0x7F
2573
2616
  else bytes([x]).decode(encoding, 'replace'))
2574
2617
  pos = t.end()
2618
+ if tab_char_idx is None and tab_byte_idx is not None:
2619
+ tab_char_idx = sum(len(p) for p in parts) + (tab_byte_idx - pos)
2575
2620
  parts.append(raw_txt[pos:].decode(encoding, 'replace'))
2576
2621
  text = ''.join(parts).rstrip()
2577
2622
  kind = 'H' if tag.startswith(b'H') else 'F'
2578
2623
  which = doc.headers if kind == 'H' else doc.footers
2579
2624
  which_fonts = doc.header_fonts if kind == 'H' else doc.footer_fonts
2625
+ which_tabs = doc.header_tabs if kind == 'H' else doc.footer_tabs
2580
2626
  second = tag[1:2]
2581
2627
  line = 1 if second in (b'E', b'O') else int(second)
2582
2628
  which[line] = text
2583
2629
  which_fonts[line] = font_idx
2630
+ # A tab whose own byte offset landed at or past the '#'-bearing text (or
2631
+ # whose target char offset ended up past what rstrip() kept) has nothing
2632
+ # left to reposition -- None, same as a line with no tab at all.
2633
+ if (tab_mark is not None and tab_char_idx is not None
2634
+ and 0 <= tab_char_idx <= len(text)):
2635
+ which_tabs[line] = (tab_char_idx, tab_mark[3], tab_mark[1])
2636
+ else:
2637
+ which_tabs[line] = None
2584
2638
  if anchor is not None:
2585
2639
  doc.hf_events.append((kind, line, text, anchor))
2586
2640
 
@@ -4251,11 +4305,18 @@ def parse_ws(data: bytes, encoding: str = 'cp437') -> Document:
4251
4305
  # dot-command check) is the only place left to find it.
4252
4306
  hf_font_idx = next((m[1] for _rel, m in line_marks if m[0] == 'font'),
4253
4307
  None)
4308
+ # planning #202: a right/center/decimal-align tab typed into a
4309
+ # `.h#`/`.f#` argument (-README's own running head) is the SAME
4310
+ # 'tab' mark a body span reads via tab_target_at -- found here the
4311
+ # same way hf_font_idx is, since _parse_head_foot only ever sees
4312
+ # bytes, never this line's own marks.
4313
+ hf_tab_mark = next(((rel, m[1], m[2], m[3]) for rel, m in line_marks
4314
+ if m[0] == 'tab'), None)
4254
4315
  _parse_head_foot(cmd if strip_hibit else raw.rstrip(), doc,
4255
4316
  encoding,
4256
4317
  anchor=len(doc.blocks) + (1 if cur.lines or
4257
4318
  cur_line.spans else 0),
4258
- font_idx=hf_font_idx)
4319
+ font_idx=hf_font_idx, tab_mark=hf_tab_mark)
4259
4320
  # The index of the block this entry POINTS AT -- the one that follows it,
4260
4321
  # which is the block still open (if it has content) or the next to open.
4261
4322
  # "This heading is in the table of contents" refers forward, not back.
@@ -72,7 +72,17 @@ def formats():
72
72
  def load_plugins():
73
73
  """Discover third-party emitters via the 'ctrlkd.emitters' entry-point group."""
74
74
  from importlib.metadata import entry_points
75
- for ep in entry_points(group='ctrlkd.emitters'):
75
+ eps = entry_points()
76
+ # Python 3.10+: entry_points() returns a SelectableGroups object with a
77
+ # .select(group=...) method (and accepts group= directly). Python 3.9:
78
+ # it returns a plain dict keyed by group name, with no select/group=
79
+ # support at all -- pyproject's requires-python is >=3.9, so both paths
80
+ # must work (planning #203; CI run 34011057595 failed 19 tests on 3.9).
81
+ if hasattr(eps, 'select'):
82
+ group = eps.select(group='ctrlkd.emitters')
83
+ else:
84
+ group = eps.get('ctrlkd.emitters', [])
85
+ for ep in group:
76
86
  if ep.name not in _REGISTRY:
77
87
  fn = ep.load()
78
88
  _REGISTRY[ep.name] = {'fn': fn, 'ext': getattr(fn, 'ext', '.' + ep.name)}
@@ -1000,7 +1010,8 @@ def _html_img(r, pictures, image_links, idx):
1000
1010
 
1001
1011
 
1002
1012
  def _html_line(spans, refs, keep, keep_ws=False, shown_map=None, inline_styling=True,
1003
- pix_map=None, pictures='off', image_links=None, sentence_spacing=False):
1013
+ pix_map=None, pictures='off', image_links=None, sentence_spacing=False,
1014
+ merge_tab_position_tags=False):
1004
1015
  """Render one already-decided list of Spans (a logical line, or one
1005
1016
  Line's worth of a paragraph unit -- callers choose). Coalesces adjacent
1006
1017
  identically-styled spans unconditionally: cheap, idempotent for a
@@ -1016,9 +1027,27 @@ def _html_line(spans, refs, keep, keep_ws=False, shown_map=None, inline_styling=
1016
1027
  raw incoming spans -- before this is the SINGLE choke point every
1017
1028
  HTML render path (Printed physical lines, Modern paragraph units,
1018
1029
  _html_slice's structure-row slices alike) funnels through, so one
1019
- application here covers all of them."""
1030
+ application here covers all of them.
1031
+
1032
+ `merge_tab_position_tags` (issue #204, centred structure rows only --
1033
+ see `_html_centered_row`'s own docstring for why it is the ONE caller
1034
+ that passes it): strips `tabhmi<N>`/`tableader<N>` tags (Printed-only
1035
+ positioning data; no HTML `_TAG`/class exists for either) from a
1036
+ COPY of each span's style set before coalescing, purely so an
1037
+ adjacent same-styled run merges across a tab-run instead of being
1038
+ fenced off by it. Defaults False -- every other caller (ordinary
1039
+ Modern paragraphs, bullet/def-list slices) keeps the existing
1040
+ behaviour, where a tab-run stays its own span."""
1020
1041
  if sentence_spacing:
1021
1042
  spans = sentence_spacing_spans(spans)
1043
+ if merge_tab_position_tags:
1044
+ spans = [Span(s.text, frozenset(t for t in s.styles
1045
+ if not (t.startswith('tabhmi')
1046
+ or t.startswith('tableader'))))
1047
+ if any(t.startswith('tabhmi') or t.startswith('tableader')
1048
+ for t in s.styles)
1049
+ else s
1050
+ for s in spans]
1022
1051
  out = []
1023
1052
  for s in split_graphic_spans(coalesce_spans(spans)):
1024
1053
  pctl = next((t for t in s.styles if t.startswith('pctl')), None)
@@ -1447,7 +1476,7 @@ def _classify_modern_blocks(doc):
1447
1476
 
1448
1477
  def _html_slice(line, start, end, refs, keep, shown_map, inline_styling=True,
1449
1478
  pix_map=None, pictures='off', image_links=None,
1450
- sentence_spacing=False):
1479
+ sentence_spacing=False, merge_tab_position_tags=False):
1451
1480
  # Round 13 (main-merge reconciliation): `_html_line` now takes a bare
1452
1481
  # spans LIST directly (the b23 overhaul's own signature -- it runs
1453
1482
  # `coalesce_spans`/`split_graphic_spans` over its argument), not a
@@ -1463,7 +1492,8 @@ def _html_slice(line, start, end, refs, keep, shown_map, inline_styling=True,
1463
1492
  return _html_line(_slice_spans(line.spans, start, end),
1464
1493
  refs, keep, shown_map=shown_map, inline_styling=inline_styling,
1465
1494
  pix_map=pix_map, pictures=pictures, image_links=image_links,
1466
- sentence_spacing=sentence_spacing)
1495
+ sentence_spacing=sentence_spacing,
1496
+ merge_tab_position_tags=merge_tab_position_tags)
1467
1497
 
1468
1498
 
1469
1499
  def _html_centered_row(line, s, refs, keep, shown_map, inline_styling=True,
@@ -1472,12 +1502,36 @@ def _html_centered_row(line, s, refs, keep, shown_map, inline_styling=True,
1472
1502
  """The centred line's own text with its alignment padding sliced off
1473
1503
  (both mechanisms: a real align=center tag already had the M3 strip
1474
1504
  upstream, so lead/trail are 0 and this is a no-op; spaces-only
1475
- centering strips the padding here for the first time)."""
1505
+ centering strips the padding here for the first time).
1506
+
1507
+ Issue #204: `merge_tab_position_tags=True` here, and ONLY here -- a
1508
+ centred structure row is exactly the shape WSFORMAT.WS's own
1509
+ Symmetric-Sequences table row is (a short "label <tab-run> value"
1510
+ line the spaces-centering heuristic pulled out of the flow, see this
1511
+ module's own `classify_rows` caller): a WordStar "tabs and dot-leader"
1512
+ symmetrical sequence expands to literal leader characters already,
1513
+ but the Printed-only `tabhmi`/`tableader` tags riding on that Span
1514
+ fence it into its own `<span>` -- all-spaces text that then trips
1515
+ `_html_span`'s "typescript indent" heuristic into a run of `&nbsp;`
1516
+ entities, which a browser renders as a wide, uncollapsed gap. sr's
1517
+ own Modern HTML instead merges the tab-run into ONE span with the
1518
+ surrounding text, literal spaces intact (`<span
1519
+ ...>4 Endnote</span>`) -- a browser's own default
1520
+ whitespace collapsing then renders that as a single visible space,
1521
+ matching real WS7's own plain-gap look (paper-scan-verified,
1522
+ planning #204's own review pack). Scoped to centred rows only:
1523
+ trying the SAME merge in `_html_line` generally (bullet/def-list
1524
+ slices, ordinary paragraphs) additionally collapsed unrelated
1525
+ label/long-prose rows sr does NOT collapse (e.g. WSFORMAT.WS's own
1526
+ "00h ^@ <tab> Fix the print position...") -- reverted; this is not a
1527
+ general "tab-runs never get their own span" rule, only a centred
1528
+ structure row's own."""
1476
1529
  raw = ''.join(sp.text for sp in line.spans)
1477
1530
  lead = len(raw) - len(raw.lstrip(' '))
1478
1531
  trail = len(raw) - len(raw.rstrip(' '))
1479
1532
  return _html_slice(line, lead, len(raw) - trail, refs, keep, shown_map, inline_styling,
1480
- pix_map, pictures, image_links, sentence_spacing)
1533
+ pix_map, pictures, image_links, sentence_spacing,
1534
+ merge_tab_position_tags=True)
1481
1535
 
1482
1536
 
1483
1537
  def _html_toc_index(doc):