ctrl-kd 4.8.0__tar.gz → 4.8.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. {ctrl_kd-4.8.0/src/ctrl_kd.egg-info → ctrl_kd-4.8.1}/PKG-INFO +1 -1
  2. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1/src/ctrl_kd.egg-info}/PKG-INFO +1 -1
  3. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrl_kd.egg-info/SOURCES.txt +2 -0
  4. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/__init__.py +1 -1
  5. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/core.py +51 -2
  6. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/emit.py +46 -0
  7. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/symbolmap.py +40 -3
  8. ctrl_kd-4.8.1/tests/test_style_strikeout_runs_until_cleared.py +297 -0
  9. ctrl_kd-4.8.1/tests/test_style_symbol_map_resolved_face.py +173 -0
  10. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_symbol_span_graphic_chars.py +30 -12
  11. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/LICENSE +0 -0
  12. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/README.md +0 -0
  13. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/pyproject.toml +0 -0
  14. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/setup.cfg +0 -0
  15. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrl_kd.egg-info/dependency_links.txt +0 -0
  16. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrl_kd.egg-info/entry_points.txt +0 -0
  17. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrl_kd.egg-info/top_level.txt +0 -0
  18. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/afm.py +0 -0
  19. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/cli.py +0 -0
  20. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/convert.py +0 -0
  21. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/fontmap.py +0 -0
  22. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/info.py +0 -0
  23. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/layout.py +0 -0
  24. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/pdf.py +0 -0
  25. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/pictures.py +0 -0
  26. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/pix.py +0 -0
  27. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/piximg.py +0 -0
  28. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/samples/LYING.WS +0 -0
  29. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/samples/OCAPTAIN.WS +0 -0
  30. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/samples/TWAINLET.WS +0 -0
  31. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/samples/WARPRAYR.WS +0 -0
  32. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/typestyles.py +0 -0
  33. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/writer.py +0 -0
  34. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/wschange.py +0 -0
  35. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_bare_tab_modulus8.py +0 -0
  36. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_ctrlkd.py +0 -0
  37. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_dot_comment_prints_nothing.py +0 -0
  38. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_driver_substitutions_exports.py +0 -0
  39. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_endnote_leading_gap.py +0 -0
  40. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_fidelity_gate.py +0 -0
  41. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_flags_toc_inline.py +0 -0
  42. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_form_feed_line_marks.py +0 -0
  43. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_glyph_aspect.py +0 -0
  44. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_graphic_cell_ops.py +0 -0
  45. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_graphic_cells.py +0 -0
  46. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_head_foot_placement_m16.py +0 -0
  47. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_head_foot_timing.py +0 -0
  48. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_html_print_stylesheet.py +0 -0
  49. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_justification.py +0 -0
  50. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_layout_marks.py +0 -0
  51. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_bullet_glyph.py +0 -0
  52. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_char_substitution.py +0 -0
  53. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_colour_restore.py +0 -0
  54. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_heading_face.py +0 -0
  55. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_hp_patterns.py +0 -0
  56. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_legend_line_spacing.py +0 -0
  57. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_pcl_rectangles.py +0 -0
  58. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_shading_table_rules.py +0 -0
  59. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_table_rule_weight.py +0 -0
  60. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_load_plugins.py +0 -0
  61. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_mailmerge_data_files.py +0 -0
  62. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_merge_page_number_variable.py +0 -0
  63. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_blank_after_tightened_line.py +0 -0
  64. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_box_regions.py +0 -0
  65. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_centring_clip.py +0 -0
  66. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_columns.py +0 -0
  67. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_line_spacing.py +0 -0
  68. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_lint.py +0 -0
  69. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_page_baseline.py +0 -0
  70. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_page_number.py +0 -0
  71. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_rtf_structure_rows.py +0 -0
  72. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_sheet_height.py +0 -0
  73. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_verse_defrow.py +0 -0
  74. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_note_rulings_20260824.py +0 -0
  75. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_page_parity.py +0 -0
  76. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_paper_verdicts.py +0 -0
  77. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_fidelity.py +0 -0
  78. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_page_membership.py +0 -0
  79. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_page_size_eject.py +0 -0
  80. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_tolerance.py +0 -0
  81. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_transparent_print.py +0 -0
  82. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_v4_column_geometry.py +0 -0
  83. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_v4_untriaged_fixes.py +0 -0
  84. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_vertical_wraparound.py +0 -0
  85. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_peseta_euro_driver.py +0 -0
  86. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pf_print_reformat.py +0 -0
  87. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pictures.py +0 -0
  88. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pix.py +0 -0
  89. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pn_page_timing.py +0 -0
  90. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_polarity_gate.py +0 -0
  91. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_print_control_labels_invisible.py +0 -0
  92. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_printed_fidelity.py +0 -0
  93. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_rtf_facing_pages.py +0 -0
  94. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_rtf_head_style_attrs.py +0 -0
  95. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_rtf_keep_with_next.py +0 -0
  96. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_rtf_page_numbers.py +0 -0
  97. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_rtf_section_spine.py +0 -0
  98. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_samples.py +0 -0
  99. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_sawyer_corpus.py +0 -0
  100. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_screenplay_detection.py +0 -0
  101. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_screenplay_pdf.py +0 -0
  102. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_screenplay_rendering.py +0 -0
  103. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_sentence_spacing_n9.py +0 -0
  104. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_structural_checks.py +0 -0
  105. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_style_leading.py +0 -0
  106. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_trailing_pa_break.py +0 -0
  107. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_verse_quote_couplet.py +0 -0
  108. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_verse_spacing.py +0 -0
  109. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_width_and_cp1252_memo.py +0 -0
  110. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_writer.py +0 -0
  111. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_wschange.py +0 -0
  112. {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_zero_width_tab.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ctrl-kd
3
- Version: 4.8.0
3
+ Version: 4.8.1
4
4
  Summary: Convert WordStar 4-7 documents and print-to-disk files to text, Markdown, HTML, RTF, or PDF. ^KD: save and done.
5
5
  Author: Jon Michaels
6
6
  License: MIT
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ctrl-kd
3
- Version: 4.8.0
3
+ Version: 4.8.1
4
4
  Summary: Convert WordStar 4-7 documents and print-to-disk files to text, Markdown, HTML, RTF, or PDF. ^KD: save and done.
5
5
  Author: Jon Michaels
6
6
  License: MIT
@@ -98,6 +98,8 @@ tests/test_screenplay_rendering.py
98
98
  tests/test_sentence_spacing_n9.py
99
99
  tests/test_structural_checks.py
100
100
  tests/test_style_leading.py
101
+ tests/test_style_strikeout_runs_until_cleared.py
102
+ tests/test_style_symbol_map_resolved_face.py
101
103
  tests/test_symbol_span_graphic_chars.py
102
104
  tests/test_trailing_pa_break.py
103
105
  tests/test_verse_quote_couplet.py
@@ -8,4 +8,4 @@ from .pdf import emit_pdf # registers the 'pdf' format
8
8
  from .convert import convert, select_notes, DEFAULT_NOTE_KINDS, ALL_NOTE_KINDS
9
9
  from .info import document_info
10
10
 
11
- __version__ = '4.8.0'
11
+ __version__ = '4.8.1'
@@ -5226,6 +5226,26 @@ def _symmetric_blocks(data: bytes, encoding: str, raw_out=None):
5226
5226
  includes, driver[0], sorted(shift_runs), marks, header,
5227
5227
  pcl_programs)
5228
5228
 
5229
+ # The style attributes that KEEP RUNNING past the paragraph that turned them
5230
+ # on, until a later style's own attrs_OFF word clears them (ruling 2026-09-16,
5231
+ # WordStar-Feature-Decision-Register, "Style-library strikeout runs until a
5232
+ # style clears it").
5233
+ #
5234
+ # WSFORMAT.TXT states the inherit rule for the attribute words as a whole --
5235
+ # "if both corresponding bits are off, then the attribute is inherited from
5236
+ # the current state" -- but the only attribute the WS7 LaserJet captures can
5237
+ # actually DEMONSTRATE it for is strikeout: in every corpus document that
5238
+ # turns bold, underline or italic on in a style, the next style used sets
5239
+ # that same bit in its own attrs_off word, so an inheriting model and a
5240
+ # reset-per-paragraph model produce identical pages and the captures cannot
5241
+ # choose between them. Strikeout is the one bit no corpus style ever clears,
5242
+ # which is exactly why its run is visible (NOVEL.WS: 954 overstrike dash rows
5243
+ # across all 52 captured pages). So the rule is applied to strikeout only,
5244
+ # on measured evidence, and the wider question is left open rather than
5245
+ # guessed. Add a bit here only with a capture that shows it.
5246
+ STICKY_STYLE_ATTRS = ((0x01, 'strike'),)
5247
+
5248
+
5229
5249
  def _parse_style_library(raw: bytes, base: int, encoding: str = 'cp437'):
5230
5250
  """The paragraph style library at file-absolute offset `base`.
5231
5251
 
@@ -6501,8 +6521,34 @@ def parse_ws(data: bytes, encoding: str = 'cp437') -> Document:
6501
6521
  # value forward, not reset to nothing -- style_fmt.clear()
6502
6522
  # below must not lose it.
6503
6523
  _prev_vmi = style_fmt.get('line_height_vmi')
6524
+ # Ruling 2026-09-16 ("Style-library strikeout runs until a
6525
+ # style clears it"): an attribute a style turns ON stays on
6526
+ # past that paragraph. WSFORMAT.TXT's own rule for the two
6527
+ # attribute words is "if both corresponding bits are off,
6528
+ # then the attribute is inherited from the current state" --
6529
+ # so only a later style that sets the bit in its attrs_OFF
6530
+ # word ends the run. Real WS7 agrees: NOVEL.WS's H3 sets
6531
+ # strikeout on the title page and nothing in that file's
6532
+ # library ever clears it, and the LaserJet capture prints
6533
+ # the overstrike dash row on all 52 pages.
6534
+ _prev_sticky = style_fmt.get('sticky_attrs', frozenset())
6504
6535
  style_fmt.clear()
6505
6536
  style_fmt['style_id'] = slot
6537
+ _sticky = set(_prev_sticky)
6538
+ if entry is not None:
6539
+ _on, _off = entry.get('attrs_on') or 0, entry.get('attrs_off') or 0
6540
+ for _bit, _tag in STICKY_STYLE_ATTRS:
6541
+ if _on & _bit:
6542
+ _sticky.add(_tag)
6543
+ elif _off & _bit:
6544
+ _sticky.discard(_tag)
6545
+ style_fmt['sticky_attrs'] = frozenset(_sticky)
6546
+ if _sticky:
6547
+ # Also for an UNRESOLVABLE handle (a 0x03xx editing-temp
6548
+ # pool entry, or a slot this file's library doesn't
6549
+ # carry): it declares neither word, so every attribute
6550
+ # inherits, including the running one.
6551
+ style_fmt['attrs'] = frozenset(_sticky)
6506
6552
  if entry is not None:
6507
6553
  style_fmt['style_name'] = entry['name']
6508
6554
  style_fmt['heading'] = _style_heading_level(entry['name'])
@@ -6522,8 +6568,11 @@ def parse_ws(data: bytes, encoding: str = 'cp437') -> Document:
6522
6568
  # HMI 1/1800in -> print columns at 10 CPI,
6523
6569
  # the unit .lm/.rm already use (180 = 1 col)
6524
6570
  style_fmt[dst_k] = round(hmi / 180)
6525
- if entry.get('attrs'):
6526
- style_fmt['attrs'] = entry['attrs']
6571
+ if entry.get('attrs') or _sticky:
6572
+ # The style's own ON bits, plus every sticky
6573
+ # attribute still running from an earlier style.
6574
+ style_fmt['attrs'] = frozenset(
6575
+ entry.get('attrs') or ()) | style_fmt['sticky_attrs']
6527
6576
  # Register C5: the style's own declared colour index
6528
6577
  # (0-15, WSFORMAT's fixed CGA/EGA palette -- same
6529
6578
  # space as an inline type-1 colour change). `is not
@@ -1467,6 +1467,36 @@ def _html_colour_used(doc):
1467
1467
  return sorted(used)
1468
1468
 
1469
1469
 
1470
+ # A style attribute a block is RUNNING UNDER without its own style record
1471
+ # declaring it -- `core.STICKY_STYLE_ATTRS`, ruling 2026-09-16 ("Style-library
1472
+ # strikeout runs until a style clears it"). HTML is the one emitter whose
1473
+ # paragraph-level attributes ride a CSS class keyed to the STYLE SLOT rather
1474
+ # than to the block, so an attribute inherited from an EARLIER style has
1475
+ # nowhere to land on this slot's own rule; it gets a class of its own instead.
1476
+ # Every other emitter merges `block.style_attrs` into its runs
1477
+ # (`core.effective_span_styles`) and picks the same attribute up for free.
1478
+ # The property text is the same table `_style_css` writes for a style's own
1479
+ # declared attributes, so the two paths cannot drift.
1480
+ _INHERITED_ATTR_CSS = {
1481
+ 'b': 'font-weight:bold',
1482
+ 'i': 'font-style:italic',
1483
+ 'u': 'text-decoration:underline',
1484
+ 'strike': 'text-decoration:line-through',
1485
+ 'sub': 'vertical-align:sub;font-size:smaller',
1486
+ 'sup': 'vertical-align:super;font-size:smaller',
1487
+ }
1488
+
1489
+
1490
+ def _style_own_attrs(doc):
1491
+ """slot -> the attributes that slot's OWN record turns on."""
1492
+ return {e['slot']: e.get('attrs') or frozenset() for e in doc.styles}
1493
+
1494
+
1495
+ def _inherited_attrs(b, own):
1496
+ """The attributes on this block that its own style record does not declare."""
1497
+ return b.style_attrs - own.get(b.style_id, frozenset())
1498
+
1499
+
1470
1500
  def _style_css(doc, printed=True, inline_styling=True):
1471
1501
  """CSS rules derived from the style records themselves -- a PASS-THROUGH
1472
1502
  of the file's own data (Jon, 2026-08-04: never hardwire a style name to
@@ -1570,6 +1600,14 @@ def _style_css(doc, printed=True, inline_styling=True):
1570
1600
  for n in _html_colour_used(doc):
1571
1601
  rules.append('.ws-colour-%d { color:#%02x%02x%02x }'
1572
1602
  % ((n,) + _CGA_PALETTE[n % 16]))
1603
+ # One rule per attribute some block actually inherited -- never a fixed
1604
+ # six, so a document with no running attribute gets no extra CSS at all.
1605
+ own = _style_own_attrs(doc)
1606
+ inherited = set()
1607
+ for b in doc.blocks:
1608
+ inherited |= _inherited_attrs(b, own)
1609
+ for tag in sorted(inherited & set(_INHERITED_ATTR_CSS)):
1610
+ rules.append('.ws-inherit-%s { %s }' % (tag, _INHERITED_ATTR_CSS[tag]))
1573
1611
  return '\n'.join(rules)
1574
1612
 
1575
1613
 
@@ -1865,6 +1903,7 @@ def emit_html(doc, mode='printed', title='', notes=DEFAULT_NOTE_KINDS,
1865
1903
  pix_map = {r.index: r for r in (pix_results or [])}
1866
1904
  keep = frozenset(notes)
1867
1905
  style_class = {}
1906
+ style_own = _style_own_attrs(doc)
1868
1907
  if styles:
1869
1908
  style_class = {s['slot']: ' class="%s"' % _style_slug(s)
1870
1909
  for s in doc.styles}
@@ -1993,6 +2032,13 @@ def emit_html(doc, mode='printed', title='', notes=DEFAULT_NOTE_KINDS,
1993
2032
  parts.append('<hr class="pb">')
1994
2033
  continue
1995
2034
  cls = style_class.get(b.style_id, '')
2035
+ if styles:
2036
+ # An attribute still running from an EARLIER style (ruling
2037
+ # 2026-09-16): the slot's own rule cannot carry it. See
2038
+ # `_INHERITED_ATTR_CSS`.
2039
+ for tag in sorted(_inherited_attrs(b, style_own)
2040
+ & set(_INHERITED_ATTR_CSS)):
2041
+ cls = _add_html_class(cls, 'ws-inherit-' + tag)
1996
2042
  keep_para, keepn_para = keep_plan.get(bi, (False, False))
1997
2043
  if keep_para:
1998
2044
  cls = _add_html_class(cls, 'ws-keep')
@@ -134,19 +134,56 @@ def transliterate(text, kind):
134
134
  return ''.join(_dingbat(c) for c in text)
135
135
  return text
136
136
 
137
+ # Typestyle numbers whose NAME announces a non-text face: the table's own
138
+ # word for the glyph repertoire, not a typeface a sentence can be set in.
139
+ # 82 (ZapfDingbats) and 192 (Symbol) are matched by name above and never
140
+ # reach this set. Numbers, not substrings: 'Pica' contains 'pi' and
141
+ # 'Presentations' contains 'ps' -- a substring test on this table is a trap.
142
+ NON_TEXT_TYPESTYLES = frozenset({
143
+ 33, # Borders
144
+ 68, # LucidaMath
145
+ 143, # Math-7 (HPLJ)
146
+ 144, # Math-8 (HPLJ)
147
+ 166, # PI
148
+ 188, # Math
149
+ 234, # TD Logos
150
+ 244, # Greek (PS (Universal Greek))
151
+ })
152
+
153
+
137
154
  def font_translit_kind(font_entry):
138
- """The transliteration a font run needs, from the block's own symbol-map
139
- bits first, typestyle name as fallback. None = ordinary text font."""
155
+ """The transliteration a font run needs, from the typestyle NAME first and
156
+ the block's own symbol-map bits only as the fallback for a face the name
157
+ table cannot resolve. None = ordinary text font."""
140
158
  if not font_entry:
141
159
  return None
142
160
  # NAME first: it is the specific signal. The coarse symbol-map bits can
143
161
  # say 'math' for both faces (PS.TST's Dingbats row transliterated to
144
- # Greek until this ordering); bits remain the fallback for unnamed fonts.
162
+ # Greek until this ordering).
145
163
  name = (font_entry.get('typestyle_name') or '').lower()
146
164
  if name.startswith('symbol'):
147
165
  return 'math'
148
166
  if 'dingbat' in name:
149
167
  return 'symbols'
168
+ # A RESOLVED ORDINARY FACE WINS over the character-set bits (ruling
169
+ # 2026-09-16). WordStar's symbol_map bits say which upper-128 (0x80-0xFF)
170
+ # table the EXTENDED characters of a run use; they were never a
171
+ # "replace the alphabet" switch. NOVEL.WS's own 'MS Front Pages' /
172
+ # 'Font: Normal' styles carry typestyle 3 (Courier) with the math bits
173
+ # set, and ten pages of ordinary English prose were being redirected
174
+ # through Adobe Symbol's encoding, so every Latin letter came out as
175
+ # the Greek letter sitting at its keyboard position. Real WS7's
176
+ # LaserJet output prints those pages as plain readable Courier.
177
+ # Characters that genuinely have no home in the body face are still
178
+ # picked up one at a time by the per-character
179
+ # fallback (`char_translit_kind`, pdf.py's `_symbol_fallback_split`),
180
+ # which is what that mechanism was built for.
181
+ #
182
+ # The bits stay the fallback for the case the docstring always claimed:
183
+ # a face the 245-entry name table cannot resolve, and the handful of
184
+ # named faces that are themselves non-text repertoires.
185
+ if name and font_entry.get('typestyle_number') not in NON_TEXT_TYPESTYLES:
186
+ return None
150
187
  sm = font_entry.get('symbol_map')
151
188
  if sm in ('math', 'symbols'):
152
189
  return sm
@@ -0,0 +1,297 @@
1
+ """A STYLE'S STRIKEOUT RUNS UNTIL A STYLE CLEARS IT (ruling 2026-09-16).
2
+
3
+ A WS7 paragraph style record carries TWO attribute words: `attrs_on` at byte
4
+ 91 and `attrs_off` at byte 93. WSFORMAT.TXT states the model exactly:
5
+
6
+ Bits set in the first word indicate those attributes are explicitly set
7
+ to the on state. Bits set in the second word indicate those attributes
8
+ are explicitly set off. If both corresponding bits are off, then the
9
+ attribute is inherited from the current state.
10
+
11
+ So a style that turns strikeout on (bit 0x01) starts a run that only ENDS at
12
+ a later style whose own `attrs_off` sets the same bit. Both engines used to
13
+ reset every attribute at each style switch, which ended the run with the
14
+ styled paragraph itself.
15
+
16
+ Real WS7 settles it. A manuscript in the Sawyer archive selects a heading
17
+ style carrying `attrs_on = 0x41` (strikeout | bold) on its title page, and no
18
+ other style in that file's library ever sets 0x01 in its `attrs_off` -- not
19
+ even the style the manuscript body itself is set in. Its captured LaserJet
20
+ print strikes through all 52 pages: WS7 renders strikeout by reprinting a row
21
+ of hyphens over the same baseline (a daisy-wheel overstrike the LaserJet
22
+ driver kept), and that dash row appears on every page of the capture.
23
+
24
+ SCOPE, and the question deliberately left open: this is applied to strikeout
25
+ ONLY (`core.STICKY_STYLE_ATTRS`). WSFORMAT's inherit sentence is written for
26
+ the attribute words as a whole, but in every corpus document that turns bold,
27
+ underline or italic on in a style, the next style used sets that same bit in
28
+ its own `attrs_off` word -- so an inheriting model and a reset-per-paragraph
29
+ model print identical pages and the captures cannot choose between them.
30
+ Strikeout is the one bit no corpus style ever clears, which is exactly why
31
+ its run is visible. The last test here pins that narrow scope so widening it
32
+ is a deliberate act with new evidence behind it, not a drift.
33
+
34
+ Synthetic fixtures only -- style libraries are built byte by byte below.
35
+ """
36
+ import re
37
+ import zlib
38
+
39
+ import pytest
40
+
41
+ from ctrlkd import core, emit, emit_layout, pdf
42
+
43
+ HARD = b'\x0d\x0a'
44
+
45
+ # WSFORMAT's own attribute bits (core.py's `entry['attrs']` decode).
46
+ STRIKE, DOUBLE, UNDERLINE, SUB, SUPER, BOLD, ITALIC = (
47
+ 0x01, 0x02, 0x08, 0x10, 0x20, 0x40, 0x80)
48
+ # The real corpus shape: an `attrs_off` word that clears everything EXCEPT
49
+ # strikeout (and the spec's own unlabelled 0x04 bit).
50
+ CLEARS_ALL_BUT_STRIKE = 0xFA
51
+ CLEARS_EVERYTHING = 0xFB
52
+
53
+
54
+ def ws7_block(cmd, content=b''):
55
+ count = (len(content) + 4).to_bytes(2, 'little')
56
+ return b'\x1d' + count + bytes([cmd]) + content + count + b'\x1d'
57
+
58
+
59
+ def style_record(attrs_on=0, attrs_off=0):
60
+ """A 102-byte style record with every field at its inherit sentinel except
61
+ the two attribute words. Same construction as test_modern_lint.py's own
62
+ `_style_record`, extended with `attrs_off` (bytes 93-94), which is the
63
+ field this behaviour turns on."""
64
+ rec = bytearray(102)
65
+ rec[0:2] = (0xFFFF).to_bytes(2, 'little') # font word0: inherited
66
+ rec[10:12] = (1800).to_bytes(2, 'little') # left margin HMI
67
+ rec[12:14] = (0xFFFE).to_bytes(2, 'little') # right margin: inherited
68
+ rec[14:16] = (0xFFFE).to_bytes(2, 'little') # para margin: inherited
69
+ rec[18] = rec[19] = 0xFF # tab counts: inherited
70
+ for k in range(32):
71
+ rec[20 + 2 * k:22 + 2 * k] = (0xBEEF).to_bytes(2, 'little')
72
+ rec[86] = 0 # justification: left
73
+ rec[87] = 1 # word wrap on
74
+ rec[88:90] = (0xFFFF).to_bytes(2, 'little') # line height: inherited
75
+ rec[90] = 0xFF # line spacing: inherited
76
+ rec[91:93] = attrs_on.to_bytes(2, 'little')
77
+ rec[93:95] = attrs_off.to_bytes(2, 'little')
78
+ rec[95] = 0xFF # colour: inherited
79
+ return bytes(rec)
80
+
81
+
82
+ def style_library(entries):
83
+ """`entries` = [(name, record_or_None)]."""
84
+ n = len(entries)
85
+ items, records = b'', b''
86
+ rec_base = 13 + 5 + 33 * n
87
+ for name, rec in entries:
88
+ nm = name.encode('cp437').ljust(24)
89
+ if rec is None:
90
+ items += nm + b'\x00' + bytes(4) + bytes(4)
91
+ else:
92
+ items += nm + b'\x02' + bytes(4) + (rec_base + len(records)).to_bytes(4, 'little')
93
+ records += rec
94
+ head = (b'\x1a\x55' + (1).to_bytes(2, 'little') + b'\x01'
95
+ + n.to_bytes(2, 'little') + (102).to_bytes(2, 'little')
96
+ + (13).to_bytes(4, 'little'))
97
+ return head + bytes([n]) + bytes(4) + items + records
98
+
99
+
100
+ def select(slot):
101
+ """One 0x11 paragraph-style-select block for `slot` of this file's pool."""
102
+ return ws7_block(0x11, (0x0200 | slot).to_bytes(2, 'little')
103
+ + (0x0201).to_bytes(2, 'little')
104
+ + (0x0300).to_bytes(2, 'little')
105
+ + (0x0201).to_bytes(2, 'little'))
106
+
107
+
108
+ def build(entries, body):
109
+ header = ws7_block(0x00, bytes([0x70]) + bytes(11) + bytes(4))
110
+ doc_body = header + body
111
+ base = ((len(doc_body) + 127) // 128) * 128
112
+ data = bytearray(doc_body.ljust(base, b'\x1a')) + style_library(entries)
113
+ data[4 + 12:4 + 16] = base.to_bytes(4, 'little')
114
+ return core.parse_ws(bytes(data))
115
+
116
+
117
+ # ------------------------------------------------------------- the fixtures
118
+
119
+ PLAIN_TEXT = b'Plain paragraph before any styled one at all.'
120
+ STRUCK_TEXT = b'The styled heading that turns strikeout on.'
121
+ AFTER_TEXT = b'The following paragraph, whose own style never clears it.'
122
+ CLEARED_TEXT = b'The paragraph whose style does clear it again.'
123
+
124
+ LIBRARY = [
125
+ ('Plain', style_record(attrs_on=0, attrs_off=CLEARS_ALL_BUT_STRIKE)),
126
+ ('Struck', style_record(attrs_on=STRIKE | BOLD,
127
+ attrs_off=CLEARS_ALL_BUT_STRIKE & ~BOLD)),
128
+ ('Clears', style_record(attrs_on=0, attrs_off=CLEARS_EVERYTHING)),
129
+ ]
130
+ SLOT_PLAIN, SLOT_STRUCK, SLOT_CLEARS = 0, 1, 2
131
+
132
+
133
+ def run_doc():
134
+ """Plain -> Struck -> Plain (inherits) -> Clears (ends the run)."""
135
+ return build(LIBRARY,
136
+ select(SLOT_PLAIN) + PLAIN_TEXT + HARD + HARD +
137
+ select(SLOT_STRUCK) + STRUCK_TEXT + HARD + HARD +
138
+ select(SLOT_PLAIN) + AFTER_TEXT + HARD + HARD +
139
+ select(SLOT_CLEARS) + CLEARED_TEXT + HARD)
140
+
141
+
142
+ def block_for(doc, text):
143
+ wanted = text.decode('cp437')
144
+ for b in doc.blocks:
145
+ joined = ''.join(sp.text for ln in b.lines for sp in ln.spans)
146
+ if wanted[:20] in joined:
147
+ return b
148
+ raise AssertionError(f'no block carrying {wanted[:20]!r}')
149
+
150
+
151
+ def decoded_streams(pdf_bytes):
152
+ out = []
153
+ for m in re.finditer(rb'stream\r?\n(.*?)\r?\nendstream', pdf_bytes, re.S):
154
+ try:
155
+ out.append(zlib.decompress(m[1]))
156
+ except zlib.error:
157
+ out.append(m[1])
158
+ return b'\n'.join(out)
159
+
160
+
161
+ # --------------------------------------------------------------- the model
162
+
163
+ def test_the_run_starts_continues_and_ends_on_the_blocks_themselves():
164
+ doc = run_doc()
165
+ assert 'strike' not in block_for(doc, PLAIN_TEXT).style_attrs, \
166
+ 'strikeout before the style that turns it on'
167
+ assert 'strike' in block_for(doc, STRUCK_TEXT).style_attrs, \
168
+ 'the style that turns strikeout on did not'
169
+ assert 'strike' in block_for(doc, AFTER_TEXT).style_attrs, \
170
+ 'the run stopped at the styled paragraph instead of continuing'
171
+ assert 'strike' not in block_for(doc, CLEARED_TEXT).style_attrs, \
172
+ 'a style setting 0x01 in attrs_off did not end the run'
173
+
174
+
175
+ def test_an_unresolvable_style_handle_inherits_rather_than_resets():
176
+ """A 0x03xx editing-temp handle declares neither attribute word, so every
177
+ attribute inherits -- the same sentence of the spec, applied to the case
178
+ where there is no record to read."""
179
+ body = (select(SLOT_STRUCK) + STRUCK_TEXT + HARD + HARD +
180
+ ws7_block(0x11, (0x0301).to_bytes(2, 'little') * 4) +
181
+ AFTER_TEXT + HARD)
182
+ doc = build(LIBRARY, body)
183
+ assert 'strike' in block_for(doc, AFTER_TEXT).style_attrs
184
+
185
+
186
+ # ------------------------------------------- one rule, every output format
187
+
188
+ def fmt_outputs(doc, mode):
189
+ return {
190
+ 'text': emit.emit_text(doc, mode),
191
+ 'markdown': emit.emit_markdown(doc, mode),
192
+ 'html': emit.emit_html(doc, mode),
193
+ 'rtf': emit.emit_rtf(doc, mode),
194
+ 'layout': emit_layout(doc, mode),
195
+ }
196
+
197
+
198
+ def rtf_group_for(body, needle):
199
+ """The `{...}` character-run group that carries `needle` -- RTF's own
200
+ scope for a run's attributes, so the assertion reads exactly one run."""
201
+ end = body.index(needle)
202
+ start = body.rindex('{', 0, end)
203
+ return body[start:body.index('}', end) + 1]
204
+
205
+
206
+ @pytest.mark.parametrize('mode', ['printed', 'modern'])
207
+ def test_rtf_marks_the_inherited_paragraph_struck(mode):
208
+ out = fmt_outputs(run_doc(), mode)['rtf']
209
+ body = out.decode('cp1252', 'replace') if isinstance(out, bytes) else out
210
+ assert '\\strike' in rtf_group_for(body, 'never clears'), \
211
+ f'{mode}: the inherited paragraph carries no \\strike'
212
+ assert '\\strike' not in rtf_group_for(body, 'does clear it again'), \
213
+ f'{mode}: the run did not end at the clearing style'
214
+
215
+
216
+ def html_tag_for(body, needle):
217
+ """The opening tag of the element carrying `needle`."""
218
+ end = body.index(needle)
219
+ start = body.rindex('<', 0, end)
220
+ return body[start:body.index('>', start) + 1]
221
+
222
+
223
+ @pytest.mark.parametrize('mode', ['printed', 'modern'])
224
+ def test_html_marks_the_inherited_paragraph_struck(mode):
225
+ out = fmt_outputs(run_doc(), mode)['html']
226
+ body = out.decode('utf-8') if isinstance(out, bytes) else out
227
+ # HTML's paragraph attributes ride a CSS class keyed to the STYLE SLOT, so
228
+ # an attribute inherited from an EARLIER style needs its own class
229
+ # (`emit._INHERITED_ATTR_CSS`) -- the slot's own rule cannot carry it.
230
+ assert '.ws-inherit-strike { text-decoration:line-through }' in body, \
231
+ f'{mode}: no CSS rule for the inherited run'
232
+ assert 'ws-inherit-strike' in html_tag_for(body, 'never clears'), \
233
+ f'{mode}: the inherited paragraph is not marked struck'
234
+ assert 'ws-inherit-strike' not in html_tag_for(body, 'does clear it again'), \
235
+ f'{mode}: the run did not end at the clearing style'
236
+
237
+
238
+ @pytest.mark.parametrize('mode', ['printed', 'modern'])
239
+ def test_layout_json_carries_the_inherited_run(mode):
240
+ out = fmt_outputs(run_doc(), mode)['layout']
241
+ body = out.decode('utf-8') if isinstance(out, bytes) else out
242
+ assert body.count('"strike"') >= 2, \
243
+ f'{mode}: fewer struck runs than the styled paragraph plus the one it runs into'
244
+
245
+
246
+ def test_markdown_modern_marks_the_inherited_paragraph_struck():
247
+ """Modern Markdown carries character attributes; PRINTED Markdown is a
248
+ verbatim monospace page inside a fence and has no inline markers at all,
249
+ in either behaviour -- there is nothing for this rule to change there."""
250
+ out = fmt_outputs(run_doc(), 'modern')['markdown']
251
+ body = out.decode('utf-8') if isinstance(out, bytes) else out
252
+ line = next(l for l in body.splitlines() if 'never clears' in l)
253
+ assert line.strip().startswith('~~'), f'no strikethrough markers: {line!r}'
254
+
255
+
256
+ @pytest.mark.parametrize('mode', ['printed', 'modern'])
257
+ def test_pdf_draws_a_rule_over_the_inherited_paragraph(mode):
258
+ """Strikeout in PDF is a stroked rule over the run (`pdf.py`'s own
259
+ `... m ... l S`). More rules are drawn when the run continues than when
260
+ the very next style ends it."""
261
+ running = decoded_streams(pdf.emit_pdf(run_doc(), mode=mode))
262
+ ended = build(LIBRARY,
263
+ select(SLOT_STRUCK) + STRUCK_TEXT + HARD + HARD +
264
+ select(SLOT_CLEARS) + AFTER_TEXT + HARD)
265
+ ended_ops = decoded_streams(pdf.emit_pdf(ended, mode=mode))
266
+ assert running.count(b' l S') > ended_ops.count(b' l S'), \
267
+ f'{mode}: the continuing run drew no more rules than the cleared one'
268
+
269
+
270
+ def test_text_output_is_unaffected_and_still_carries_the_words():
271
+ """Plain text has no representation for strikeout in either behaviour --
272
+ named here so the format is not silently missing from the list above."""
273
+ out = fmt_outputs(run_doc(), 'printed')['text']
274
+ body = out.decode('utf-8') if isinstance(out, bytes) else out
275
+ for sample in (PLAIN_TEXT, STRUCK_TEXT, AFTER_TEXT, CLEARED_TEXT):
276
+ assert sample.decode('cp437')[:20] in body
277
+
278
+
279
+ # ------------------------------------------------------------ the scope pin
280
+
281
+ def test_only_strikeout_is_sticky():
282
+ """The narrow scope, pinned. Bold here is turned on by one style and never
283
+ cleared by the next one's `attrs_off`, and it still must NOT carry: no WS7
284
+ capture in the corpus can show whether real WordStar would carry it, so
285
+ the rule is not extended to bold, underline or italic on a guess. Widening
286
+ `core.STICKY_STYLE_ATTRS` means finding a capture that settles it."""
287
+ assert core.STICKY_STYLE_ATTRS == ((0x01, 'strike'),)
288
+ library = [
289
+ ('Bold', style_record(attrs_on=BOLD, attrs_off=0)),
290
+ ('Next', style_record(attrs_on=0, attrs_off=0)),
291
+ ]
292
+ doc = build(library,
293
+ select(0) + STRUCK_TEXT + HARD + HARD +
294
+ select(1) + AFTER_TEXT + HARD)
295
+ assert 'b' in block_for(doc, STRUCK_TEXT).style_attrs
296
+ assert 'b' not in block_for(doc, AFTER_TEXT).style_attrs, \
297
+ 'bold carried past its own paragraph -- the scope was widened silently'
@@ -0,0 +1,173 @@
1
+ """A RESOLVED ORDINARY TYPEFACE BEATS THE CHARACTER-SET BITS (ruling 2026-09-16).
2
+
3
+ A WS5+ font record carries a typestyle word whose low nine bits name a
4
+ typeface out of WSFORMAT's own 245-entry table and whose bits 12-13 pick one
5
+ of four upper-128 character sets (cp437, cp850, math, symbols). The two are
6
+ independent fields. `font_translit_kind` used to let the character-set bits
7
+ overrule ANY name that did not itself say "symbol" or "dingbat", so a font
8
+ record reading "Courier, math character set" redirected its whole run through
9
+ Adobe Symbol's encoding -- and every ordinary Latin letter came out as the
10
+ Greek letter sitting at its keyboard position.
11
+
12
+ That is a real corpus record, not a hypothetical: the front-matter paragraph
13
+ styles of a WS7 manuscript in the Sawyer archive carry typestyle word 0x6603
14
+ (number 3 = Courier, letter-quality, symbol_map = math) and ten pages of
15
+ ordinary English prose rendered as Greek in PDF, RTF, Markdown and the layout
16
+ JSON alike. Real WS7's own LaserJet output prints those pages as plain
17
+ readable Courier.
18
+
19
+ THE RULE: the character-set bits are the fallback for a face the name table
20
+ cannot resolve (and for the handful of named faces that are themselves
21
+ non-text repertoires -- Symbol, ZapfDingbats, Math, PI, Greek...). A resolved
22
+ ordinary face wins. Characters that genuinely have no home in the body face
23
+ are still rescued one at a time by the per-character fallback
24
+ (`char_translit_kind`), which is what that mechanism was built for.
25
+
26
+ Synthetic fixtures only -- the typestyle words below are transcribed into
27
+ constructed bytes, no corpus file is read.
28
+ """
29
+ import re
30
+ import zlib
31
+
32
+ import pytest
33
+
34
+ from ctrlkd import core, emit, emit_layout, pdf
35
+ from ctrlkd.symbolmap import font_translit_kind
36
+
37
+ HARD = b'\x0d\x0a'
38
+ PROSE = b'If you choose to depict the physicist on the cover'
39
+ # Long enough that the WS5+ text body is unambiguous -- a two-word document
40
+ # does not exercise the parse path this test is about.
41
+ GREEKABLE = b'alpha beta gamma alpha beta gamma alpha beta gamma'
42
+
43
+ # typestyle words, built field by field from core._font_entry's own bit table:
44
+ # bits 0-8 typeface number (WSFORMAT's name table)
45
+ # bits 12-13 character set: 0 cp437, 1 cp850, 2 math, 3 symbols
46
+ # bit 14 letter quality
47
+ # bit 15 proportional
48
+ MATH_BITS = 2 << 12
49
+ COURIER_MATH = 3 | MATH_BITS | 0x4000 # 0x6603, the real corpus word
50
+ SYMBOL_MATH = 192 | MATH_BITS | 0x4000 # a genuine Symbol face
51
+ DINGBATS_MATH = 82 | MATH_BITS | 0x4000 # ZapfDingbats, bits say 'math'
52
+ UNNAMED_MATH = 300 | MATH_BITS | 0x4000 # past the 245-entry name table
53
+
54
+
55
+ def ws7_block(cmd, content=b''):
56
+ count = (len(content) + 4).to_bytes(2, 'little')
57
+ return b'\x1d' + count + bytes([cmd]) + content + count + b'\x1d'
58
+
59
+
60
+ def font_block(style, w=180, h=240):
61
+ """One type-2 Font symmetric block: width (HMI), height (VMI), typestyle,
62
+ then three zeroed 'previous' words."""
63
+ content = (w.to_bytes(2, 'little') + h.to_bytes(2, 'little') +
64
+ style.to_bytes(2, 'little') + b'\x00' * 6)
65
+ return ws7_block(0x02, content)
66
+
67
+
68
+ def doc_in_face(style, text=PROSE):
69
+ return core.parse_ws(ws7_block(0x00) + font_block(style) + text + HARD)
70
+
71
+
72
+ def decoded_streams(pdf_bytes):
73
+ out = []
74
+ for m in re.finditer(rb'stream\r?\n(.*?)\r?\nendstream', pdf_bytes, re.S):
75
+ try:
76
+ out.append(zlib.decompress(m[1]))
77
+ except zlib.error:
78
+ out.append(m[1])
79
+ return b'\n'.join(out)
80
+
81
+
82
+ def symbol_resource_names(pdf_bytes):
83
+ obj_font = {int(m[1]): m[2] for m in
84
+ re.finditer(rb'(\d+) 0 obj\s*<< /Type /Font /Subtype /Type1 '
85
+ rb'/BaseFont /(\S+)', pdf_bytes)}
86
+ return {m[1] for m in re.finditer(rb'/(F\d+) (\d+) 0 R', pdf_bytes)
87
+ if obj_font.get(int(m[2])) in (b'Symbol', b'ZapfDingbats')}
88
+
89
+
90
+ # --------------------------------------------------------------- the verdict
91
+
92
+ def test_resolved_ordinary_face_ignores_the_character_set_bits():
93
+ entry = {'typestyle_name': 'Courier', 'typestyle_number': 3,
94
+ 'symbol_map': 'math'}
95
+ assert font_translit_kind(entry) is None
96
+
97
+
98
+ def test_a_named_symbol_face_still_transliterates():
99
+ """The name is still the specific signal -- unchanged, and the reason the
100
+ name is read BEFORE the bits (a Dingbats row whose coarse bits read 'math'
101
+ must transliterate as Dingbats, not as Greek)."""
102
+ assert font_translit_kind({'typestyle_name': 'Symbol',
103
+ 'typestyle_number': 192,
104
+ 'symbol_map': 'math'}) == 'math'
105
+ assert font_translit_kind({'typestyle_name': 'ZapfDingbats',
106
+ 'typestyle_number': 82,
107
+ 'symbol_map': 'math'}) == 'symbols'
108
+
109
+
110
+ def test_an_unresolved_face_still_falls_back_to_the_bits():
111
+ """The fallback the function's docstring always claimed: a typeface number
112
+ the 245-entry table does not carry has no name to prefer, so the
113
+ character-set bits are all there is."""
114
+ assert font_translit_kind({'typestyle_name': None,
115
+ 'typestyle_number': 300,
116
+ 'symbol_map': 'math'}) == 'math'
117
+
118
+
119
+ def test_a_named_non_text_repertoire_still_falls_back_to_the_bits():
120
+ """'Math', 'PI' and 'Greek' are names for a glyph repertoire, not for a
121
+ typeface a sentence can be set in -- they are not 'ordinary faces' and do
122
+ not win over the bits."""
123
+ for number, name in ((188, 'Math'), (166, 'PI'),
124
+ (244, 'Greek (PS (Universal Greek))')):
125
+ assert font_translit_kind({'typestyle_name': name,
126
+ 'typestyle_number': number,
127
+ 'symbol_map': 'math'}) == 'math', name
128
+
129
+
130
+ # ------------------------------------------------- one rule, every emitter
131
+
132
+ @pytest.mark.parametrize('mode', ['printed', 'modern'])
133
+ def test_courier_math_prose_stays_latin_in_every_text_format(mode):
134
+ """Every emitter reads the SAME verdict, so one rule fixes all of them at
135
+ once. Before the ruling each of these carried the Greek transliteration."""
136
+ doc = doc_in_face(COURIER_MATH)
137
+ for name, out in (
138
+ ('text', emit.emit_text(doc, mode)),
139
+ ('markdown', emit.emit_markdown(doc, mode)),
140
+ ('html', emit.emit_html(doc, mode)),
141
+ ('rtf', emit.emit_rtf(doc, mode)),
142
+ ('layout', emit_layout(doc, mode))):
143
+ text = out.decode('utf-8', 'replace') if isinstance(out, bytes) else out
144
+ assert 'If you choose' in text, f'{name}/{mode}: prose is not Latin'
145
+ assert not any('\u0370' <= c <= '\u03ff' for c in text), \
146
+ f'{name}/{mode}: a Greek code point reached the output'
147
+
148
+
149
+ @pytest.mark.parametrize('mode', ['printed', 'modern'])
150
+ def test_courier_math_prose_uses_no_symbol_font_in_the_pdf(mode):
151
+ doc = doc_in_face(COURIER_MATH)
152
+ pdf_bytes = pdf.emit_pdf(doc, mode=mode)
153
+ assert not symbol_resource_names(pdf_bytes), \
154
+ f'{mode}: a Symbol/ZapfDingbats font resource was registered'
155
+ # Printed sets the whole physical line in one Tj, Modern one word per
156
+ # Tj -- the word is what both share.
157
+ assert b'choose' in decoded_streams(pdf_bytes)
158
+
159
+
160
+ @pytest.mark.parametrize('mode', ['printed', 'modern'])
161
+ def test_a_genuine_symbol_face_is_untouched_by_the_ruling(mode):
162
+ """The guard on the other side: a run really set in Symbol still resolves
163
+ to the Symbol resource and still writes that face's own byte codes."""
164
+ doc = doc_in_face(SYMBOL_MATH, text=GREEKABLE)
165
+ pdf_bytes = pdf.emit_pdf(doc, mode=mode)
166
+ assert symbol_resource_names(pdf_bytes), \
167
+ f'{mode}: a real Symbol face lost its Symbol resource'
168
+
169
+
170
+ def test_an_unresolved_face_is_untouched_by_the_ruling():
171
+ doc = doc_in_face(UNNAMED_MATH, text=GREEKABLE)
172
+ kinds = [font_translit_kind(f) for f in doc.fonts]
173
+ assert 'math' in kinds, 'the bits fallback stopped working for unnamed faces'
@@ -1,6 +1,16 @@
1
1
  """cp437 block/shade/box glyphs degrading to '?' in a Symbol-mapped span
2
2
  (Printed and Modern PDF alike).
3
3
 
4
+ PARTLY SUPERSEDED, 2026-09-16: the premise below -- "a real WS7 quirk this
5
+ project has to live with", a font block named Brush Script whose character-set
6
+ bits read 'math' resolving to the Adobe Symbol face -- is exactly the misread
7
+ the 2026-09-16 ruling ends ("a resolved ordinary typeface beats the
8
+ character-set bits"; class tests in
9
+ tests/test_style_symbol_map_resolved_face.py). That run now stays on its own
10
+ ordinary face. The GEOMETRY guarantee this file was written for is unchanged
11
+ and still the point of every test here: box/shade glyphs draw as vector fills
12
+ and a middle dot stays a real byte, in every font state a span can be in.
13
+
4
14
  FOUND against the real corpus: sawyer/REF/-LASERJE.FNT line 9 -- twelve
5
15
  cp437 glyphs ('░▒▓│┤╡╢╖╕╣║╗') typed under a font block whose typestyle is
6
16
  Brush Script, then a real WS7 quirk this project has to live with: that
@@ -127,24 +137,32 @@ def test_no_font_ordinary_font_and_symbol_mapped_font_all_draw_vectors_not_quest
127
137
  # a decoded-Python-string comparison.
128
138
 
129
139
 
130
- def test_symbol_mapped_span_middle_dot_and_box_run_still_select_the_symbol_font():
131
- """The fix must not stop the Symbol-mapped span's OWN font resolution:
132
- its Tj text (middle dot, byte 0xB7 -- Adobe Symbol's periodcentered)
133
- still runs under a /BaseFont /Symbol resource, and the box run right
134
- before it still draws as geometry (no font operator needed for a fill)."""
140
+ def test_symbol_mapped_span_no_longer_selects_the_symbol_font_at_all():
141
+ """SUPERSEDED BY THE 2026-09-16 RULING ("a resolved ordinary typeface beats
142
+ the character-set bits"). This case -- typestyle 54, 'Brush Script', with
143
+ the coarse symbol-map bits reading 'math' -- is the very misread that
144
+ ruling ends: Brush Script is a perfectly ordinary named face, so its run
145
+ now stays on the ordinary text font and the character-set bits govern only
146
+ the extended characters, as WordStar intended.
147
+
148
+ What this file's own bug fix guaranteed is unchanged and still checked
149
+ here: the twelve box/shade glyphs draw as vector geometry and the middle
150
+ dot stays a real 0xB7 text byte, never the '?' degradation. The byte is
151
+ the same either way -- cp1252 and Adobe Symbol both carry periodcentered
152
+ at 0xB7 -- so the visible page does not move; only the font resource does.
153
+ See tests/test_style_symbol_map_resolved_face.py for the ruling's own
154
+ class tests."""
135
155
  doc = _build_doc()
136
156
  pdf_bytes = pdf.emit_pdf(doc, mode='printed')
137
- symbol_names = _symbol_font_names(pdf_bytes)
138
- assert symbol_names, 'no /BaseFont /Symbol resource registered at all'
157
+ assert not _symbol_font_names(pdf_bytes), (
158
+ 'a /BaseFont /Symbol resource was registered for a run whose typeface '
159
+ 'name resolves to an ordinary face')
139
160
  content = _decoded_content(pdf_bytes)
140
161
  # The LAST Tj in the stream is the third span's middle dot -- one bare
141
- # 0xB7 byte, under one of the Symbol resource names, never '?' (0x3f).
162
+ # 0xB7 byte, never '?' (0x3f).
142
163
  last_tj = list(re.finditer(rb'/(F\d+) \d+ Tf[^()]*\(([^()]*)\) Tj', content))[-1]
143
- assert last_tj[1] in symbol_names, (
144
- f'last Tj ({last_tj[2]!r}) is not under a Symbol font resource')
145
164
  assert last_tj[2] == b'\xb7', (
146
- f'Symbol-mapped middle dot did not encode as periodcentered (0xb7): '
147
- f'got {last_tj[2]!r}')
165
+ f'the middle dot did not survive as a real byte: got {last_tj[2]!r}')
148
166
 
149
167
 
150
168
  def test_fontless_and_ordinary_font_spans_are_unaffected_baseline():
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes