ctrl-kd 4.8.0__tar.gz → 4.8.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ctrl_kd-4.8.0/src/ctrl_kd.egg-info → ctrl_kd-4.8.1}/PKG-INFO +1 -1
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1/src/ctrl_kd.egg-info}/PKG-INFO +1 -1
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrl_kd.egg-info/SOURCES.txt +2 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/__init__.py +1 -1
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/core.py +51 -2
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/emit.py +46 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/symbolmap.py +40 -3
- ctrl_kd-4.8.1/tests/test_style_strikeout_runs_until_cleared.py +297 -0
- ctrl_kd-4.8.1/tests/test_style_symbol_map_resolved_face.py +173 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_symbol_span_graphic_chars.py +30 -12
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/LICENSE +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/README.md +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/pyproject.toml +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/setup.cfg +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrl_kd.egg-info/dependency_links.txt +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrl_kd.egg-info/entry_points.txt +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrl_kd.egg-info/top_level.txt +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/afm.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/cli.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/convert.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/fontmap.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/info.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/layout.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/pdf.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/pictures.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/pix.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/piximg.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/samples/LYING.WS +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/samples/OCAPTAIN.WS +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/samples/TWAINLET.WS +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/samples/WARPRAYR.WS +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/typestyles.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/writer.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/src/ctrlkd/wschange.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_bare_tab_modulus8.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_ctrlkd.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_dot_comment_prints_nothing.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_driver_substitutions_exports.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_endnote_leading_gap.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_fidelity_gate.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_flags_toc_inline.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_form_feed_line_marks.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_glyph_aspect.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_graphic_cell_ops.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_graphic_cells.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_head_foot_placement_m16.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_head_foot_timing.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_html_print_stylesheet.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_justification.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_layout_marks.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_bullet_glyph.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_char_substitution.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_colour_restore.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_heading_face.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_hp_patterns.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_legend_line_spacing.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_pcl_rectangles.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_shading_table_rules.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_lj6dtp_table_rule_weight.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_load_plugins.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_mailmerge_data_files.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_merge_page_number_variable.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_blank_after_tightened_line.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_box_regions.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_centring_clip.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_columns.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_line_spacing.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_lint.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_page_baseline.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_page_number.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_rtf_structure_rows.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_sheet_height.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_modern_verse_defrow.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_note_rulings_20260824.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_page_parity.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_paper_verdicts.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_fidelity.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_page_membership.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_page_size_eject.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_tolerance.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_transparent_print.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_v4_column_geometry.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_v4_untriaged_fixes.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pcl_vertical_wraparound.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_peseta_euro_driver.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pf_print_reformat.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pictures.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pix.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_pn_page_timing.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_polarity_gate.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_print_control_labels_invisible.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_printed_fidelity.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_rtf_facing_pages.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_rtf_head_style_attrs.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_rtf_keep_with_next.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_rtf_page_numbers.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_rtf_section_spine.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_samples.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_sawyer_corpus.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_screenplay_detection.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_screenplay_pdf.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_screenplay_rendering.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_sentence_spacing_n9.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_structural_checks.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_style_leading.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_trailing_pa_break.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_verse_quote_couplet.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_verse_spacing.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_width_and_cp1252_memo.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_writer.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_wschange.py +0 -0
- {ctrl_kd-4.8.0 → ctrl_kd-4.8.1}/tests/test_zero_width_tab.py +0 -0
|
@@ -98,6 +98,8 @@ tests/test_screenplay_rendering.py
|
|
|
98
98
|
tests/test_sentence_spacing_n9.py
|
|
99
99
|
tests/test_structural_checks.py
|
|
100
100
|
tests/test_style_leading.py
|
|
101
|
+
tests/test_style_strikeout_runs_until_cleared.py
|
|
102
|
+
tests/test_style_symbol_map_resolved_face.py
|
|
101
103
|
tests/test_symbol_span_graphic_chars.py
|
|
102
104
|
tests/test_trailing_pa_break.py
|
|
103
105
|
tests/test_verse_quote_couplet.py
|
|
@@ -5226,6 +5226,26 @@ def _symmetric_blocks(data: bytes, encoding: str, raw_out=None):
|
|
|
5226
5226
|
includes, driver[0], sorted(shift_runs), marks, header,
|
|
5227
5227
|
pcl_programs)
|
|
5228
5228
|
|
|
5229
|
+
# The style attributes that KEEP RUNNING past the paragraph that turned them
|
|
5230
|
+
# on, until a later style's own attrs_OFF word clears them (ruling 2026-09-16,
|
|
5231
|
+
# WordStar-Feature-Decision-Register, "Style-library strikeout runs until a
|
|
5232
|
+
# style clears it").
|
|
5233
|
+
#
|
|
5234
|
+
# WSFORMAT.TXT states the inherit rule for the attribute words as a whole --
|
|
5235
|
+
# "if both corresponding bits are off, then the attribute is inherited from
|
|
5236
|
+
# the current state" -- but the only attribute the WS7 LaserJet captures can
|
|
5237
|
+
# actually DEMONSTRATE it for is strikeout: in every corpus document that
|
|
5238
|
+
# turns bold, underline or italic on in a style, the next style used sets
|
|
5239
|
+
# that same bit in its own attrs_off word, so an inheriting model and a
|
|
5240
|
+
# reset-per-paragraph model produce identical pages and the captures cannot
|
|
5241
|
+
# choose between them. Strikeout is the one bit no corpus style ever clears,
|
|
5242
|
+
# which is exactly why its run is visible (NOVEL.WS: 954 overstrike dash rows
|
|
5243
|
+
# across all 52 captured pages). So the rule is applied to strikeout only,
|
|
5244
|
+
# on measured evidence, and the wider question is left open rather than
|
|
5245
|
+
# guessed. Add a bit here only with a capture that shows it.
|
|
5246
|
+
STICKY_STYLE_ATTRS = ((0x01, 'strike'),)
|
|
5247
|
+
|
|
5248
|
+
|
|
5229
5249
|
def _parse_style_library(raw: bytes, base: int, encoding: str = 'cp437'):
|
|
5230
5250
|
"""The paragraph style library at file-absolute offset `base`.
|
|
5231
5251
|
|
|
@@ -6501,8 +6521,34 @@ def parse_ws(data: bytes, encoding: str = 'cp437') -> Document:
|
|
|
6501
6521
|
# value forward, not reset to nothing -- style_fmt.clear()
|
|
6502
6522
|
# below must not lose it.
|
|
6503
6523
|
_prev_vmi = style_fmt.get('line_height_vmi')
|
|
6524
|
+
# Ruling 2026-09-16 ("Style-library strikeout runs until a
|
|
6525
|
+
# style clears it"): an attribute a style turns ON stays on
|
|
6526
|
+
# past that paragraph. WSFORMAT.TXT's own rule for the two
|
|
6527
|
+
# attribute words is "if both corresponding bits are off,
|
|
6528
|
+
# then the attribute is inherited from the current state" --
|
|
6529
|
+
# so only a later style that sets the bit in its attrs_OFF
|
|
6530
|
+
# word ends the run. Real WS7 agrees: NOVEL.WS's H3 sets
|
|
6531
|
+
# strikeout on the title page and nothing in that file's
|
|
6532
|
+
# library ever clears it, and the LaserJet capture prints
|
|
6533
|
+
# the overstrike dash row on all 52 pages.
|
|
6534
|
+
_prev_sticky = style_fmt.get('sticky_attrs', frozenset())
|
|
6504
6535
|
style_fmt.clear()
|
|
6505
6536
|
style_fmt['style_id'] = slot
|
|
6537
|
+
_sticky = set(_prev_sticky)
|
|
6538
|
+
if entry is not None:
|
|
6539
|
+
_on, _off = entry.get('attrs_on') or 0, entry.get('attrs_off') or 0
|
|
6540
|
+
for _bit, _tag in STICKY_STYLE_ATTRS:
|
|
6541
|
+
if _on & _bit:
|
|
6542
|
+
_sticky.add(_tag)
|
|
6543
|
+
elif _off & _bit:
|
|
6544
|
+
_sticky.discard(_tag)
|
|
6545
|
+
style_fmt['sticky_attrs'] = frozenset(_sticky)
|
|
6546
|
+
if _sticky:
|
|
6547
|
+
# Also for an UNRESOLVABLE handle (a 0x03xx editing-temp
|
|
6548
|
+
# pool entry, or a slot this file's library doesn't
|
|
6549
|
+
# carry): it declares neither word, so every attribute
|
|
6550
|
+
# inherits, including the running one.
|
|
6551
|
+
style_fmt['attrs'] = frozenset(_sticky)
|
|
6506
6552
|
if entry is not None:
|
|
6507
6553
|
style_fmt['style_name'] = entry['name']
|
|
6508
6554
|
style_fmt['heading'] = _style_heading_level(entry['name'])
|
|
@@ -6522,8 +6568,11 @@ def parse_ws(data: bytes, encoding: str = 'cp437') -> Document:
|
|
|
6522
6568
|
# HMI 1/1800in -> print columns at 10 CPI,
|
|
6523
6569
|
# the unit .lm/.rm already use (180 = 1 col)
|
|
6524
6570
|
style_fmt[dst_k] = round(hmi / 180)
|
|
6525
|
-
if entry.get('attrs'):
|
|
6526
|
-
|
|
6571
|
+
if entry.get('attrs') or _sticky:
|
|
6572
|
+
# The style's own ON bits, plus every sticky
|
|
6573
|
+
# attribute still running from an earlier style.
|
|
6574
|
+
style_fmt['attrs'] = frozenset(
|
|
6575
|
+
entry.get('attrs') or ()) | style_fmt['sticky_attrs']
|
|
6527
6576
|
# Register C5: the style's own declared colour index
|
|
6528
6577
|
# (0-15, WSFORMAT's fixed CGA/EGA palette -- same
|
|
6529
6578
|
# space as an inline type-1 colour change). `is not
|
|
@@ -1467,6 +1467,36 @@ def _html_colour_used(doc):
|
|
|
1467
1467
|
return sorted(used)
|
|
1468
1468
|
|
|
1469
1469
|
|
|
1470
|
+
# A style attribute a block is RUNNING UNDER without its own style record
|
|
1471
|
+
# declaring it -- `core.STICKY_STYLE_ATTRS`, ruling 2026-09-16 ("Style-library
|
|
1472
|
+
# strikeout runs until a style clears it"). HTML is the one emitter whose
|
|
1473
|
+
# paragraph-level attributes ride a CSS class keyed to the STYLE SLOT rather
|
|
1474
|
+
# than to the block, so an attribute inherited from an EARLIER style has
|
|
1475
|
+
# nowhere to land on this slot's own rule; it gets a class of its own instead.
|
|
1476
|
+
# Every other emitter merges `block.style_attrs` into its runs
|
|
1477
|
+
# (`core.effective_span_styles`) and picks the same attribute up for free.
|
|
1478
|
+
# The property text is the same table `_style_css` writes for a style's own
|
|
1479
|
+
# declared attributes, so the two paths cannot drift.
|
|
1480
|
+
_INHERITED_ATTR_CSS = {
|
|
1481
|
+
'b': 'font-weight:bold',
|
|
1482
|
+
'i': 'font-style:italic',
|
|
1483
|
+
'u': 'text-decoration:underline',
|
|
1484
|
+
'strike': 'text-decoration:line-through',
|
|
1485
|
+
'sub': 'vertical-align:sub;font-size:smaller',
|
|
1486
|
+
'sup': 'vertical-align:super;font-size:smaller',
|
|
1487
|
+
}
|
|
1488
|
+
|
|
1489
|
+
|
|
1490
|
+
def _style_own_attrs(doc):
|
|
1491
|
+
"""slot -> the attributes that slot's OWN record turns on."""
|
|
1492
|
+
return {e['slot']: e.get('attrs') or frozenset() for e in doc.styles}
|
|
1493
|
+
|
|
1494
|
+
|
|
1495
|
+
def _inherited_attrs(b, own):
|
|
1496
|
+
"""The attributes on this block that its own style record does not declare."""
|
|
1497
|
+
return b.style_attrs - own.get(b.style_id, frozenset())
|
|
1498
|
+
|
|
1499
|
+
|
|
1470
1500
|
def _style_css(doc, printed=True, inline_styling=True):
|
|
1471
1501
|
"""CSS rules derived from the style records themselves -- a PASS-THROUGH
|
|
1472
1502
|
of the file's own data (Jon, 2026-08-04: never hardwire a style name to
|
|
@@ -1570,6 +1600,14 @@ def _style_css(doc, printed=True, inline_styling=True):
|
|
|
1570
1600
|
for n in _html_colour_used(doc):
|
|
1571
1601
|
rules.append('.ws-colour-%d { color:#%02x%02x%02x }'
|
|
1572
1602
|
% ((n,) + _CGA_PALETTE[n % 16]))
|
|
1603
|
+
# One rule per attribute some block actually inherited -- never a fixed
|
|
1604
|
+
# six, so a document with no running attribute gets no extra CSS at all.
|
|
1605
|
+
own = _style_own_attrs(doc)
|
|
1606
|
+
inherited = set()
|
|
1607
|
+
for b in doc.blocks:
|
|
1608
|
+
inherited |= _inherited_attrs(b, own)
|
|
1609
|
+
for tag in sorted(inherited & set(_INHERITED_ATTR_CSS)):
|
|
1610
|
+
rules.append('.ws-inherit-%s { %s }' % (tag, _INHERITED_ATTR_CSS[tag]))
|
|
1573
1611
|
return '\n'.join(rules)
|
|
1574
1612
|
|
|
1575
1613
|
|
|
@@ -1865,6 +1903,7 @@ def emit_html(doc, mode='printed', title='', notes=DEFAULT_NOTE_KINDS,
|
|
|
1865
1903
|
pix_map = {r.index: r for r in (pix_results or [])}
|
|
1866
1904
|
keep = frozenset(notes)
|
|
1867
1905
|
style_class = {}
|
|
1906
|
+
style_own = _style_own_attrs(doc)
|
|
1868
1907
|
if styles:
|
|
1869
1908
|
style_class = {s['slot']: ' class="%s"' % _style_slug(s)
|
|
1870
1909
|
for s in doc.styles}
|
|
@@ -1993,6 +2032,13 @@ def emit_html(doc, mode='printed', title='', notes=DEFAULT_NOTE_KINDS,
|
|
|
1993
2032
|
parts.append('<hr class="pb">')
|
|
1994
2033
|
continue
|
|
1995
2034
|
cls = style_class.get(b.style_id, '')
|
|
2035
|
+
if styles:
|
|
2036
|
+
# An attribute still running from an EARLIER style (ruling
|
|
2037
|
+
# 2026-09-16): the slot's own rule cannot carry it. See
|
|
2038
|
+
# `_INHERITED_ATTR_CSS`.
|
|
2039
|
+
for tag in sorted(_inherited_attrs(b, style_own)
|
|
2040
|
+
& set(_INHERITED_ATTR_CSS)):
|
|
2041
|
+
cls = _add_html_class(cls, 'ws-inherit-' + tag)
|
|
1996
2042
|
keep_para, keepn_para = keep_plan.get(bi, (False, False))
|
|
1997
2043
|
if keep_para:
|
|
1998
2044
|
cls = _add_html_class(cls, 'ws-keep')
|
|
@@ -134,19 +134,56 @@ def transliterate(text, kind):
|
|
|
134
134
|
return ''.join(_dingbat(c) for c in text)
|
|
135
135
|
return text
|
|
136
136
|
|
|
137
|
+
# Typestyle numbers whose NAME announces a non-text face: the table's own
|
|
138
|
+
# word for the glyph repertoire, not a typeface a sentence can be set in.
|
|
139
|
+
# 82 (ZapfDingbats) and 192 (Symbol) are matched by name above and never
|
|
140
|
+
# reach this set. Numbers, not substrings: 'Pica' contains 'pi' and
|
|
141
|
+
# 'Presentations' contains 'ps' -- a substring test on this table is a trap.
|
|
142
|
+
NON_TEXT_TYPESTYLES = frozenset({
|
|
143
|
+
33, # Borders
|
|
144
|
+
68, # LucidaMath
|
|
145
|
+
143, # Math-7 (HPLJ)
|
|
146
|
+
144, # Math-8 (HPLJ)
|
|
147
|
+
166, # PI
|
|
148
|
+
188, # Math
|
|
149
|
+
234, # TD Logos
|
|
150
|
+
244, # Greek (PS (Universal Greek))
|
|
151
|
+
})
|
|
152
|
+
|
|
153
|
+
|
|
137
154
|
def font_translit_kind(font_entry):
|
|
138
|
-
"""The transliteration a font run needs, from the
|
|
139
|
-
|
|
155
|
+
"""The transliteration a font run needs, from the typestyle NAME first and
|
|
156
|
+
the block's own symbol-map bits only as the fallback for a face the name
|
|
157
|
+
table cannot resolve. None = ordinary text font."""
|
|
140
158
|
if not font_entry:
|
|
141
159
|
return None
|
|
142
160
|
# NAME first: it is the specific signal. The coarse symbol-map bits can
|
|
143
161
|
# say 'math' for both faces (PS.TST's Dingbats row transliterated to
|
|
144
|
-
# Greek until this ordering)
|
|
162
|
+
# Greek until this ordering).
|
|
145
163
|
name = (font_entry.get('typestyle_name') or '').lower()
|
|
146
164
|
if name.startswith('symbol'):
|
|
147
165
|
return 'math'
|
|
148
166
|
if 'dingbat' in name:
|
|
149
167
|
return 'symbols'
|
|
168
|
+
# A RESOLVED ORDINARY FACE WINS over the character-set bits (ruling
|
|
169
|
+
# 2026-09-16). WordStar's symbol_map bits say which upper-128 (0x80-0xFF)
|
|
170
|
+
# table the EXTENDED characters of a run use; they were never a
|
|
171
|
+
# "replace the alphabet" switch. NOVEL.WS's own 'MS Front Pages' /
|
|
172
|
+
# 'Font: Normal' styles carry typestyle 3 (Courier) with the math bits
|
|
173
|
+
# set, and ten pages of ordinary English prose were being redirected
|
|
174
|
+
# through Adobe Symbol's encoding, so every Latin letter came out as
|
|
175
|
+
# the Greek letter sitting at its keyboard position. Real WS7's
|
|
176
|
+
# LaserJet output prints those pages as plain readable Courier.
|
|
177
|
+
# Characters that genuinely have no home in the body face are still
|
|
178
|
+
# picked up one at a time by the per-character
|
|
179
|
+
# fallback (`char_translit_kind`, pdf.py's `_symbol_fallback_split`),
|
|
180
|
+
# which is what that mechanism was built for.
|
|
181
|
+
#
|
|
182
|
+
# The bits stay the fallback for the case the docstring always claimed:
|
|
183
|
+
# a face the 245-entry name table cannot resolve, and the handful of
|
|
184
|
+
# named faces that are themselves non-text repertoires.
|
|
185
|
+
if name and font_entry.get('typestyle_number') not in NON_TEXT_TYPESTYLES:
|
|
186
|
+
return None
|
|
150
187
|
sm = font_entry.get('symbol_map')
|
|
151
188
|
if sm in ('math', 'symbols'):
|
|
152
189
|
return sm
|
|
@@ -0,0 +1,297 @@
|
|
|
1
|
+
"""A STYLE'S STRIKEOUT RUNS UNTIL A STYLE CLEARS IT (ruling 2026-09-16).
|
|
2
|
+
|
|
3
|
+
A WS7 paragraph style record carries TWO attribute words: `attrs_on` at byte
|
|
4
|
+
91 and `attrs_off` at byte 93. WSFORMAT.TXT states the model exactly:
|
|
5
|
+
|
|
6
|
+
Bits set in the first word indicate those attributes are explicitly set
|
|
7
|
+
to the on state. Bits set in the second word indicate those attributes
|
|
8
|
+
are explicitly set off. If both corresponding bits are off, then the
|
|
9
|
+
attribute is inherited from the current state.
|
|
10
|
+
|
|
11
|
+
So a style that turns strikeout on (bit 0x01) starts a run that only ENDS at
|
|
12
|
+
a later style whose own `attrs_off` sets the same bit. Both engines used to
|
|
13
|
+
reset every attribute at each style switch, which ended the run with the
|
|
14
|
+
styled paragraph itself.
|
|
15
|
+
|
|
16
|
+
Real WS7 settles it. A manuscript in the Sawyer archive selects a heading
|
|
17
|
+
style carrying `attrs_on = 0x41` (strikeout | bold) on its title page, and no
|
|
18
|
+
other style in that file's library ever sets 0x01 in its `attrs_off` -- not
|
|
19
|
+
even the style the manuscript body itself is set in. Its captured LaserJet
|
|
20
|
+
print strikes through all 52 pages: WS7 renders strikeout by reprinting a row
|
|
21
|
+
of hyphens over the same baseline (a daisy-wheel overstrike the LaserJet
|
|
22
|
+
driver kept), and that dash row appears on every page of the capture.
|
|
23
|
+
|
|
24
|
+
SCOPE, and the question deliberately left open: this is applied to strikeout
|
|
25
|
+
ONLY (`core.STICKY_STYLE_ATTRS`). WSFORMAT's inherit sentence is written for
|
|
26
|
+
the attribute words as a whole, but in every corpus document that turns bold,
|
|
27
|
+
underline or italic on in a style, the next style used sets that same bit in
|
|
28
|
+
its own `attrs_off` word -- so an inheriting model and a reset-per-paragraph
|
|
29
|
+
model print identical pages and the captures cannot choose between them.
|
|
30
|
+
Strikeout is the one bit no corpus style ever clears, which is exactly why
|
|
31
|
+
its run is visible. The last test here pins that narrow scope so widening it
|
|
32
|
+
is a deliberate act with new evidence behind it, not a drift.
|
|
33
|
+
|
|
34
|
+
Synthetic fixtures only -- style libraries are built byte by byte below.
|
|
35
|
+
"""
|
|
36
|
+
import re
|
|
37
|
+
import zlib
|
|
38
|
+
|
|
39
|
+
import pytest
|
|
40
|
+
|
|
41
|
+
from ctrlkd import core, emit, emit_layout, pdf
|
|
42
|
+
|
|
43
|
+
HARD = b'\x0d\x0a'
|
|
44
|
+
|
|
45
|
+
# WSFORMAT's own attribute bits (core.py's `entry['attrs']` decode).
|
|
46
|
+
STRIKE, DOUBLE, UNDERLINE, SUB, SUPER, BOLD, ITALIC = (
|
|
47
|
+
0x01, 0x02, 0x08, 0x10, 0x20, 0x40, 0x80)
|
|
48
|
+
# The real corpus shape: an `attrs_off` word that clears everything EXCEPT
|
|
49
|
+
# strikeout (and the spec's own unlabelled 0x04 bit).
|
|
50
|
+
CLEARS_ALL_BUT_STRIKE = 0xFA
|
|
51
|
+
CLEARS_EVERYTHING = 0xFB
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def ws7_block(cmd, content=b''):
|
|
55
|
+
count = (len(content) + 4).to_bytes(2, 'little')
|
|
56
|
+
return b'\x1d' + count + bytes([cmd]) + content + count + b'\x1d'
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def style_record(attrs_on=0, attrs_off=0):
|
|
60
|
+
"""A 102-byte style record with every field at its inherit sentinel except
|
|
61
|
+
the two attribute words. Same construction as test_modern_lint.py's own
|
|
62
|
+
`_style_record`, extended with `attrs_off` (bytes 93-94), which is the
|
|
63
|
+
field this behaviour turns on."""
|
|
64
|
+
rec = bytearray(102)
|
|
65
|
+
rec[0:2] = (0xFFFF).to_bytes(2, 'little') # font word0: inherited
|
|
66
|
+
rec[10:12] = (1800).to_bytes(2, 'little') # left margin HMI
|
|
67
|
+
rec[12:14] = (0xFFFE).to_bytes(2, 'little') # right margin: inherited
|
|
68
|
+
rec[14:16] = (0xFFFE).to_bytes(2, 'little') # para margin: inherited
|
|
69
|
+
rec[18] = rec[19] = 0xFF # tab counts: inherited
|
|
70
|
+
for k in range(32):
|
|
71
|
+
rec[20 + 2 * k:22 + 2 * k] = (0xBEEF).to_bytes(2, 'little')
|
|
72
|
+
rec[86] = 0 # justification: left
|
|
73
|
+
rec[87] = 1 # word wrap on
|
|
74
|
+
rec[88:90] = (0xFFFF).to_bytes(2, 'little') # line height: inherited
|
|
75
|
+
rec[90] = 0xFF # line spacing: inherited
|
|
76
|
+
rec[91:93] = attrs_on.to_bytes(2, 'little')
|
|
77
|
+
rec[93:95] = attrs_off.to_bytes(2, 'little')
|
|
78
|
+
rec[95] = 0xFF # colour: inherited
|
|
79
|
+
return bytes(rec)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def style_library(entries):
|
|
83
|
+
"""`entries` = [(name, record_or_None)]."""
|
|
84
|
+
n = len(entries)
|
|
85
|
+
items, records = b'', b''
|
|
86
|
+
rec_base = 13 + 5 + 33 * n
|
|
87
|
+
for name, rec in entries:
|
|
88
|
+
nm = name.encode('cp437').ljust(24)
|
|
89
|
+
if rec is None:
|
|
90
|
+
items += nm + b'\x00' + bytes(4) + bytes(4)
|
|
91
|
+
else:
|
|
92
|
+
items += nm + b'\x02' + bytes(4) + (rec_base + len(records)).to_bytes(4, 'little')
|
|
93
|
+
records += rec
|
|
94
|
+
head = (b'\x1a\x55' + (1).to_bytes(2, 'little') + b'\x01'
|
|
95
|
+
+ n.to_bytes(2, 'little') + (102).to_bytes(2, 'little')
|
|
96
|
+
+ (13).to_bytes(4, 'little'))
|
|
97
|
+
return head + bytes([n]) + bytes(4) + items + records
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def select(slot):
|
|
101
|
+
"""One 0x11 paragraph-style-select block for `slot` of this file's pool."""
|
|
102
|
+
return ws7_block(0x11, (0x0200 | slot).to_bytes(2, 'little')
|
|
103
|
+
+ (0x0201).to_bytes(2, 'little')
|
|
104
|
+
+ (0x0300).to_bytes(2, 'little')
|
|
105
|
+
+ (0x0201).to_bytes(2, 'little'))
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def build(entries, body):
|
|
109
|
+
header = ws7_block(0x00, bytes([0x70]) + bytes(11) + bytes(4))
|
|
110
|
+
doc_body = header + body
|
|
111
|
+
base = ((len(doc_body) + 127) // 128) * 128
|
|
112
|
+
data = bytearray(doc_body.ljust(base, b'\x1a')) + style_library(entries)
|
|
113
|
+
data[4 + 12:4 + 16] = base.to_bytes(4, 'little')
|
|
114
|
+
return core.parse_ws(bytes(data))
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
# ------------------------------------------------------------- the fixtures
|
|
118
|
+
|
|
119
|
+
PLAIN_TEXT = b'Plain paragraph before any styled one at all.'
|
|
120
|
+
STRUCK_TEXT = b'The styled heading that turns strikeout on.'
|
|
121
|
+
AFTER_TEXT = b'The following paragraph, whose own style never clears it.'
|
|
122
|
+
CLEARED_TEXT = b'The paragraph whose style does clear it again.'
|
|
123
|
+
|
|
124
|
+
LIBRARY = [
|
|
125
|
+
('Plain', style_record(attrs_on=0, attrs_off=CLEARS_ALL_BUT_STRIKE)),
|
|
126
|
+
('Struck', style_record(attrs_on=STRIKE | BOLD,
|
|
127
|
+
attrs_off=CLEARS_ALL_BUT_STRIKE & ~BOLD)),
|
|
128
|
+
('Clears', style_record(attrs_on=0, attrs_off=CLEARS_EVERYTHING)),
|
|
129
|
+
]
|
|
130
|
+
SLOT_PLAIN, SLOT_STRUCK, SLOT_CLEARS = 0, 1, 2
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def run_doc():
|
|
134
|
+
"""Plain -> Struck -> Plain (inherits) -> Clears (ends the run)."""
|
|
135
|
+
return build(LIBRARY,
|
|
136
|
+
select(SLOT_PLAIN) + PLAIN_TEXT + HARD + HARD +
|
|
137
|
+
select(SLOT_STRUCK) + STRUCK_TEXT + HARD + HARD +
|
|
138
|
+
select(SLOT_PLAIN) + AFTER_TEXT + HARD + HARD +
|
|
139
|
+
select(SLOT_CLEARS) + CLEARED_TEXT + HARD)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def block_for(doc, text):
|
|
143
|
+
wanted = text.decode('cp437')
|
|
144
|
+
for b in doc.blocks:
|
|
145
|
+
joined = ''.join(sp.text for ln in b.lines for sp in ln.spans)
|
|
146
|
+
if wanted[:20] in joined:
|
|
147
|
+
return b
|
|
148
|
+
raise AssertionError(f'no block carrying {wanted[:20]!r}')
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def decoded_streams(pdf_bytes):
|
|
152
|
+
out = []
|
|
153
|
+
for m in re.finditer(rb'stream\r?\n(.*?)\r?\nendstream', pdf_bytes, re.S):
|
|
154
|
+
try:
|
|
155
|
+
out.append(zlib.decompress(m[1]))
|
|
156
|
+
except zlib.error:
|
|
157
|
+
out.append(m[1])
|
|
158
|
+
return b'\n'.join(out)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
# --------------------------------------------------------------- the model
|
|
162
|
+
|
|
163
|
+
def test_the_run_starts_continues_and_ends_on_the_blocks_themselves():
|
|
164
|
+
doc = run_doc()
|
|
165
|
+
assert 'strike' not in block_for(doc, PLAIN_TEXT).style_attrs, \
|
|
166
|
+
'strikeout before the style that turns it on'
|
|
167
|
+
assert 'strike' in block_for(doc, STRUCK_TEXT).style_attrs, \
|
|
168
|
+
'the style that turns strikeout on did not'
|
|
169
|
+
assert 'strike' in block_for(doc, AFTER_TEXT).style_attrs, \
|
|
170
|
+
'the run stopped at the styled paragraph instead of continuing'
|
|
171
|
+
assert 'strike' not in block_for(doc, CLEARED_TEXT).style_attrs, \
|
|
172
|
+
'a style setting 0x01 in attrs_off did not end the run'
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def test_an_unresolvable_style_handle_inherits_rather_than_resets():
|
|
176
|
+
"""A 0x03xx editing-temp handle declares neither attribute word, so every
|
|
177
|
+
attribute inherits -- the same sentence of the spec, applied to the case
|
|
178
|
+
where there is no record to read."""
|
|
179
|
+
body = (select(SLOT_STRUCK) + STRUCK_TEXT + HARD + HARD +
|
|
180
|
+
ws7_block(0x11, (0x0301).to_bytes(2, 'little') * 4) +
|
|
181
|
+
AFTER_TEXT + HARD)
|
|
182
|
+
doc = build(LIBRARY, body)
|
|
183
|
+
assert 'strike' in block_for(doc, AFTER_TEXT).style_attrs
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
# ------------------------------------------- one rule, every output format
|
|
187
|
+
|
|
188
|
+
def fmt_outputs(doc, mode):
|
|
189
|
+
return {
|
|
190
|
+
'text': emit.emit_text(doc, mode),
|
|
191
|
+
'markdown': emit.emit_markdown(doc, mode),
|
|
192
|
+
'html': emit.emit_html(doc, mode),
|
|
193
|
+
'rtf': emit.emit_rtf(doc, mode),
|
|
194
|
+
'layout': emit_layout(doc, mode),
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def rtf_group_for(body, needle):
|
|
199
|
+
"""The `{...}` character-run group that carries `needle` -- RTF's own
|
|
200
|
+
scope for a run's attributes, so the assertion reads exactly one run."""
|
|
201
|
+
end = body.index(needle)
|
|
202
|
+
start = body.rindex('{', 0, end)
|
|
203
|
+
return body[start:body.index('}', end) + 1]
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
@pytest.mark.parametrize('mode', ['printed', 'modern'])
|
|
207
|
+
def test_rtf_marks_the_inherited_paragraph_struck(mode):
|
|
208
|
+
out = fmt_outputs(run_doc(), mode)['rtf']
|
|
209
|
+
body = out.decode('cp1252', 'replace') if isinstance(out, bytes) else out
|
|
210
|
+
assert '\\strike' in rtf_group_for(body, 'never clears'), \
|
|
211
|
+
f'{mode}: the inherited paragraph carries no \\strike'
|
|
212
|
+
assert '\\strike' not in rtf_group_for(body, 'does clear it again'), \
|
|
213
|
+
f'{mode}: the run did not end at the clearing style'
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def html_tag_for(body, needle):
|
|
217
|
+
"""The opening tag of the element carrying `needle`."""
|
|
218
|
+
end = body.index(needle)
|
|
219
|
+
start = body.rindex('<', 0, end)
|
|
220
|
+
return body[start:body.index('>', start) + 1]
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
@pytest.mark.parametrize('mode', ['printed', 'modern'])
|
|
224
|
+
def test_html_marks_the_inherited_paragraph_struck(mode):
|
|
225
|
+
out = fmt_outputs(run_doc(), mode)['html']
|
|
226
|
+
body = out.decode('utf-8') if isinstance(out, bytes) else out
|
|
227
|
+
# HTML's paragraph attributes ride a CSS class keyed to the STYLE SLOT, so
|
|
228
|
+
# an attribute inherited from an EARLIER style needs its own class
|
|
229
|
+
# (`emit._INHERITED_ATTR_CSS`) -- the slot's own rule cannot carry it.
|
|
230
|
+
assert '.ws-inherit-strike { text-decoration:line-through }' in body, \
|
|
231
|
+
f'{mode}: no CSS rule for the inherited run'
|
|
232
|
+
assert 'ws-inherit-strike' in html_tag_for(body, 'never clears'), \
|
|
233
|
+
f'{mode}: the inherited paragraph is not marked struck'
|
|
234
|
+
assert 'ws-inherit-strike' not in html_tag_for(body, 'does clear it again'), \
|
|
235
|
+
f'{mode}: the run did not end at the clearing style'
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
@pytest.mark.parametrize('mode', ['printed', 'modern'])
|
|
239
|
+
def test_layout_json_carries_the_inherited_run(mode):
|
|
240
|
+
out = fmt_outputs(run_doc(), mode)['layout']
|
|
241
|
+
body = out.decode('utf-8') if isinstance(out, bytes) else out
|
|
242
|
+
assert body.count('"strike"') >= 2, \
|
|
243
|
+
f'{mode}: fewer struck runs than the styled paragraph plus the one it runs into'
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def test_markdown_modern_marks_the_inherited_paragraph_struck():
|
|
247
|
+
"""Modern Markdown carries character attributes; PRINTED Markdown is a
|
|
248
|
+
verbatim monospace page inside a fence and has no inline markers at all,
|
|
249
|
+
in either behaviour -- there is nothing for this rule to change there."""
|
|
250
|
+
out = fmt_outputs(run_doc(), 'modern')['markdown']
|
|
251
|
+
body = out.decode('utf-8') if isinstance(out, bytes) else out
|
|
252
|
+
line = next(l for l in body.splitlines() if 'never clears' in l)
|
|
253
|
+
assert line.strip().startswith('~~'), f'no strikethrough markers: {line!r}'
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
@pytest.mark.parametrize('mode', ['printed', 'modern'])
|
|
257
|
+
def test_pdf_draws_a_rule_over_the_inherited_paragraph(mode):
|
|
258
|
+
"""Strikeout in PDF is a stroked rule over the run (`pdf.py`'s own
|
|
259
|
+
`... m ... l S`). More rules are drawn when the run continues than when
|
|
260
|
+
the very next style ends it."""
|
|
261
|
+
running = decoded_streams(pdf.emit_pdf(run_doc(), mode=mode))
|
|
262
|
+
ended = build(LIBRARY,
|
|
263
|
+
select(SLOT_STRUCK) + STRUCK_TEXT + HARD + HARD +
|
|
264
|
+
select(SLOT_CLEARS) + AFTER_TEXT + HARD)
|
|
265
|
+
ended_ops = decoded_streams(pdf.emit_pdf(ended, mode=mode))
|
|
266
|
+
assert running.count(b' l S') > ended_ops.count(b' l S'), \
|
|
267
|
+
f'{mode}: the continuing run drew no more rules than the cleared one'
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def test_text_output_is_unaffected_and_still_carries_the_words():
|
|
271
|
+
"""Plain text has no representation for strikeout in either behaviour --
|
|
272
|
+
named here so the format is not silently missing from the list above."""
|
|
273
|
+
out = fmt_outputs(run_doc(), 'printed')['text']
|
|
274
|
+
body = out.decode('utf-8') if isinstance(out, bytes) else out
|
|
275
|
+
for sample in (PLAIN_TEXT, STRUCK_TEXT, AFTER_TEXT, CLEARED_TEXT):
|
|
276
|
+
assert sample.decode('cp437')[:20] in body
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
# ------------------------------------------------------------ the scope pin
|
|
280
|
+
|
|
281
|
+
def test_only_strikeout_is_sticky():
|
|
282
|
+
"""The narrow scope, pinned. Bold here is turned on by one style and never
|
|
283
|
+
cleared by the next one's `attrs_off`, and it still must NOT carry: no WS7
|
|
284
|
+
capture in the corpus can show whether real WordStar would carry it, so
|
|
285
|
+
the rule is not extended to bold, underline or italic on a guess. Widening
|
|
286
|
+
`core.STICKY_STYLE_ATTRS` means finding a capture that settles it."""
|
|
287
|
+
assert core.STICKY_STYLE_ATTRS == ((0x01, 'strike'),)
|
|
288
|
+
library = [
|
|
289
|
+
('Bold', style_record(attrs_on=BOLD, attrs_off=0)),
|
|
290
|
+
('Next', style_record(attrs_on=0, attrs_off=0)),
|
|
291
|
+
]
|
|
292
|
+
doc = build(library,
|
|
293
|
+
select(0) + STRUCK_TEXT + HARD + HARD +
|
|
294
|
+
select(1) + AFTER_TEXT + HARD)
|
|
295
|
+
assert 'b' in block_for(doc, STRUCK_TEXT).style_attrs
|
|
296
|
+
assert 'b' not in block_for(doc, AFTER_TEXT).style_attrs, \
|
|
297
|
+
'bold carried past its own paragraph -- the scope was widened silently'
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
"""A RESOLVED ORDINARY TYPEFACE BEATS THE CHARACTER-SET BITS (ruling 2026-09-16).
|
|
2
|
+
|
|
3
|
+
A WS5+ font record carries a typestyle word whose low nine bits name a
|
|
4
|
+
typeface out of WSFORMAT's own 245-entry table and whose bits 12-13 pick one
|
|
5
|
+
of four upper-128 character sets (cp437, cp850, math, symbols). The two are
|
|
6
|
+
independent fields. `font_translit_kind` used to let the character-set bits
|
|
7
|
+
overrule ANY name that did not itself say "symbol" or "dingbat", so a font
|
|
8
|
+
record reading "Courier, math character set" redirected its whole run through
|
|
9
|
+
Adobe Symbol's encoding -- and every ordinary Latin letter came out as the
|
|
10
|
+
Greek letter sitting at its keyboard position.
|
|
11
|
+
|
|
12
|
+
That is a real corpus record, not a hypothetical: the front-matter paragraph
|
|
13
|
+
styles of a WS7 manuscript in the Sawyer archive carry typestyle word 0x6603
|
|
14
|
+
(number 3 = Courier, letter-quality, symbol_map = math) and ten pages of
|
|
15
|
+
ordinary English prose rendered as Greek in PDF, RTF, Markdown and the layout
|
|
16
|
+
JSON alike. Real WS7's own LaserJet output prints those pages as plain
|
|
17
|
+
readable Courier.
|
|
18
|
+
|
|
19
|
+
THE RULE: the character-set bits are the fallback for a face the name table
|
|
20
|
+
cannot resolve (and for the handful of named faces that are themselves
|
|
21
|
+
non-text repertoires -- Symbol, ZapfDingbats, Math, PI, Greek...). A resolved
|
|
22
|
+
ordinary face wins. Characters that genuinely have no home in the body face
|
|
23
|
+
are still rescued one at a time by the per-character fallback
|
|
24
|
+
(`char_translit_kind`), which is what that mechanism was built for.
|
|
25
|
+
|
|
26
|
+
Synthetic fixtures only -- the typestyle words below are transcribed into
|
|
27
|
+
constructed bytes, no corpus file is read.
|
|
28
|
+
"""
|
|
29
|
+
import re
|
|
30
|
+
import zlib
|
|
31
|
+
|
|
32
|
+
import pytest
|
|
33
|
+
|
|
34
|
+
from ctrlkd import core, emit, emit_layout, pdf
|
|
35
|
+
from ctrlkd.symbolmap import font_translit_kind
|
|
36
|
+
|
|
37
|
+
HARD = b'\x0d\x0a'
|
|
38
|
+
PROSE = b'If you choose to depict the physicist on the cover'
|
|
39
|
+
# Long enough that the WS5+ text body is unambiguous -- a two-word document
|
|
40
|
+
# does not exercise the parse path this test is about.
|
|
41
|
+
GREEKABLE = b'alpha beta gamma alpha beta gamma alpha beta gamma'
|
|
42
|
+
|
|
43
|
+
# typestyle words, built field by field from core._font_entry's own bit table:
|
|
44
|
+
# bits 0-8 typeface number (WSFORMAT's name table)
|
|
45
|
+
# bits 12-13 character set: 0 cp437, 1 cp850, 2 math, 3 symbols
|
|
46
|
+
# bit 14 letter quality
|
|
47
|
+
# bit 15 proportional
|
|
48
|
+
MATH_BITS = 2 << 12
|
|
49
|
+
COURIER_MATH = 3 | MATH_BITS | 0x4000 # 0x6603, the real corpus word
|
|
50
|
+
SYMBOL_MATH = 192 | MATH_BITS | 0x4000 # a genuine Symbol face
|
|
51
|
+
DINGBATS_MATH = 82 | MATH_BITS | 0x4000 # ZapfDingbats, bits say 'math'
|
|
52
|
+
UNNAMED_MATH = 300 | MATH_BITS | 0x4000 # past the 245-entry name table
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def ws7_block(cmd, content=b''):
|
|
56
|
+
count = (len(content) + 4).to_bytes(2, 'little')
|
|
57
|
+
return b'\x1d' + count + bytes([cmd]) + content + count + b'\x1d'
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def font_block(style, w=180, h=240):
|
|
61
|
+
"""One type-2 Font symmetric block: width (HMI), height (VMI), typestyle,
|
|
62
|
+
then three zeroed 'previous' words."""
|
|
63
|
+
content = (w.to_bytes(2, 'little') + h.to_bytes(2, 'little') +
|
|
64
|
+
style.to_bytes(2, 'little') + b'\x00' * 6)
|
|
65
|
+
return ws7_block(0x02, content)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def doc_in_face(style, text=PROSE):
|
|
69
|
+
return core.parse_ws(ws7_block(0x00) + font_block(style) + text + HARD)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def decoded_streams(pdf_bytes):
|
|
73
|
+
out = []
|
|
74
|
+
for m in re.finditer(rb'stream\r?\n(.*?)\r?\nendstream', pdf_bytes, re.S):
|
|
75
|
+
try:
|
|
76
|
+
out.append(zlib.decompress(m[1]))
|
|
77
|
+
except zlib.error:
|
|
78
|
+
out.append(m[1])
|
|
79
|
+
return b'\n'.join(out)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def symbol_resource_names(pdf_bytes):
|
|
83
|
+
obj_font = {int(m[1]): m[2] for m in
|
|
84
|
+
re.finditer(rb'(\d+) 0 obj\s*<< /Type /Font /Subtype /Type1 '
|
|
85
|
+
rb'/BaseFont /(\S+)', pdf_bytes)}
|
|
86
|
+
return {m[1] for m in re.finditer(rb'/(F\d+) (\d+) 0 R', pdf_bytes)
|
|
87
|
+
if obj_font.get(int(m[2])) in (b'Symbol', b'ZapfDingbats')}
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
# --------------------------------------------------------------- the verdict
|
|
91
|
+
|
|
92
|
+
def test_resolved_ordinary_face_ignores_the_character_set_bits():
|
|
93
|
+
entry = {'typestyle_name': 'Courier', 'typestyle_number': 3,
|
|
94
|
+
'symbol_map': 'math'}
|
|
95
|
+
assert font_translit_kind(entry) is None
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def test_a_named_symbol_face_still_transliterates():
|
|
99
|
+
"""The name is still the specific signal -- unchanged, and the reason the
|
|
100
|
+
name is read BEFORE the bits (a Dingbats row whose coarse bits read 'math'
|
|
101
|
+
must transliterate as Dingbats, not as Greek)."""
|
|
102
|
+
assert font_translit_kind({'typestyle_name': 'Symbol',
|
|
103
|
+
'typestyle_number': 192,
|
|
104
|
+
'symbol_map': 'math'}) == 'math'
|
|
105
|
+
assert font_translit_kind({'typestyle_name': 'ZapfDingbats',
|
|
106
|
+
'typestyle_number': 82,
|
|
107
|
+
'symbol_map': 'math'}) == 'symbols'
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def test_an_unresolved_face_still_falls_back_to_the_bits():
|
|
111
|
+
"""The fallback the function's docstring always claimed: a typeface number
|
|
112
|
+
the 245-entry table does not carry has no name to prefer, so the
|
|
113
|
+
character-set bits are all there is."""
|
|
114
|
+
assert font_translit_kind({'typestyle_name': None,
|
|
115
|
+
'typestyle_number': 300,
|
|
116
|
+
'symbol_map': 'math'}) == 'math'
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def test_a_named_non_text_repertoire_still_falls_back_to_the_bits():
|
|
120
|
+
"""'Math', 'PI' and 'Greek' are names for a glyph repertoire, not for a
|
|
121
|
+
typeface a sentence can be set in -- they are not 'ordinary faces' and do
|
|
122
|
+
not win over the bits."""
|
|
123
|
+
for number, name in ((188, 'Math'), (166, 'PI'),
|
|
124
|
+
(244, 'Greek (PS (Universal Greek))')):
|
|
125
|
+
assert font_translit_kind({'typestyle_name': name,
|
|
126
|
+
'typestyle_number': number,
|
|
127
|
+
'symbol_map': 'math'}) == 'math', name
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
# ------------------------------------------------- one rule, every emitter
|
|
131
|
+
|
|
132
|
+
@pytest.mark.parametrize('mode', ['printed', 'modern'])
|
|
133
|
+
def test_courier_math_prose_stays_latin_in_every_text_format(mode):
|
|
134
|
+
"""Every emitter reads the SAME verdict, so one rule fixes all of them at
|
|
135
|
+
once. Before the ruling each of these carried the Greek transliteration."""
|
|
136
|
+
doc = doc_in_face(COURIER_MATH)
|
|
137
|
+
for name, out in (
|
|
138
|
+
('text', emit.emit_text(doc, mode)),
|
|
139
|
+
('markdown', emit.emit_markdown(doc, mode)),
|
|
140
|
+
('html', emit.emit_html(doc, mode)),
|
|
141
|
+
('rtf', emit.emit_rtf(doc, mode)),
|
|
142
|
+
('layout', emit_layout(doc, mode))):
|
|
143
|
+
text = out.decode('utf-8', 'replace') if isinstance(out, bytes) else out
|
|
144
|
+
assert 'If you choose' in text, f'{name}/{mode}: prose is not Latin'
|
|
145
|
+
assert not any('\u0370' <= c <= '\u03ff' for c in text), \
|
|
146
|
+
f'{name}/{mode}: a Greek code point reached the output'
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
@pytest.mark.parametrize('mode', ['printed', 'modern'])
|
|
150
|
+
def test_courier_math_prose_uses_no_symbol_font_in_the_pdf(mode):
|
|
151
|
+
doc = doc_in_face(COURIER_MATH)
|
|
152
|
+
pdf_bytes = pdf.emit_pdf(doc, mode=mode)
|
|
153
|
+
assert not symbol_resource_names(pdf_bytes), \
|
|
154
|
+
f'{mode}: a Symbol/ZapfDingbats font resource was registered'
|
|
155
|
+
# Printed sets the whole physical line in one Tj, Modern one word per
|
|
156
|
+
# Tj -- the word is what both share.
|
|
157
|
+
assert b'choose' in decoded_streams(pdf_bytes)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
@pytest.mark.parametrize('mode', ['printed', 'modern'])
|
|
161
|
+
def test_a_genuine_symbol_face_is_untouched_by_the_ruling(mode):
|
|
162
|
+
"""The guard on the other side: a run really set in Symbol still resolves
|
|
163
|
+
to the Symbol resource and still writes that face's own byte codes."""
|
|
164
|
+
doc = doc_in_face(SYMBOL_MATH, text=GREEKABLE)
|
|
165
|
+
pdf_bytes = pdf.emit_pdf(doc, mode=mode)
|
|
166
|
+
assert symbol_resource_names(pdf_bytes), \
|
|
167
|
+
f'{mode}: a real Symbol face lost its Symbol resource'
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def test_an_unresolved_face_is_untouched_by_the_ruling():
|
|
171
|
+
doc = doc_in_face(UNNAMED_MATH, text=GREEKABLE)
|
|
172
|
+
kinds = [font_translit_kind(f) for f in doc.fonts]
|
|
173
|
+
assert 'math' in kinds, 'the bits fallback stopped working for unnamed faces'
|
|
@@ -1,6 +1,16 @@
|
|
|
1
1
|
"""cp437 block/shade/box glyphs degrading to '?' in a Symbol-mapped span
|
|
2
2
|
(Printed and Modern PDF alike).
|
|
3
3
|
|
|
4
|
+
PARTLY SUPERSEDED, 2026-09-16: the premise below -- "a real WS7 quirk this
|
|
5
|
+
project has to live with", a font block named Brush Script whose character-set
|
|
6
|
+
bits read 'math' resolving to the Adobe Symbol face -- is exactly the misread
|
|
7
|
+
the 2026-09-16 ruling ends ("a resolved ordinary typeface beats the
|
|
8
|
+
character-set bits"; class tests in
|
|
9
|
+
tests/test_style_symbol_map_resolved_face.py). That run now stays on its own
|
|
10
|
+
ordinary face. The GEOMETRY guarantee this file was written for is unchanged
|
|
11
|
+
and still the point of every test here: box/shade glyphs draw as vector fills
|
|
12
|
+
and a middle dot stays a real byte, in every font state a span can be in.
|
|
13
|
+
|
|
4
14
|
FOUND against the real corpus: sawyer/REF/-LASERJE.FNT line 9 -- twelve
|
|
5
15
|
cp437 glyphs ('░▒▓│┤╡╢╖╕╣║╗') typed under a font block whose typestyle is
|
|
6
16
|
Brush Script, then a real WS7 quirk this project has to live with: that
|
|
@@ -127,24 +137,32 @@ def test_no_font_ordinary_font_and_symbol_mapped_font_all_draw_vectors_not_quest
|
|
|
127
137
|
# a decoded-Python-string comparison.
|
|
128
138
|
|
|
129
139
|
|
|
130
|
-
def
|
|
131
|
-
"""
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
140
|
+
def test_symbol_mapped_span_no_longer_selects_the_symbol_font_at_all():
|
|
141
|
+
"""SUPERSEDED BY THE 2026-09-16 RULING ("a resolved ordinary typeface beats
|
|
142
|
+
the character-set bits"). This case -- typestyle 54, 'Brush Script', with
|
|
143
|
+
the coarse symbol-map bits reading 'math' -- is the very misread that
|
|
144
|
+
ruling ends: Brush Script is a perfectly ordinary named face, so its run
|
|
145
|
+
now stays on the ordinary text font and the character-set bits govern only
|
|
146
|
+
the extended characters, as WordStar intended.
|
|
147
|
+
|
|
148
|
+
What this file's own bug fix guaranteed is unchanged and still checked
|
|
149
|
+
here: the twelve box/shade glyphs draw as vector geometry and the middle
|
|
150
|
+
dot stays a real 0xB7 text byte, never the '?' degradation. The byte is
|
|
151
|
+
the same either way -- cp1252 and Adobe Symbol both carry periodcentered
|
|
152
|
+
at 0xB7 -- so the visible page does not move; only the font resource does.
|
|
153
|
+
See tests/test_style_symbol_map_resolved_face.py for the ruling's own
|
|
154
|
+
class tests."""
|
|
135
155
|
doc = _build_doc()
|
|
136
156
|
pdf_bytes = pdf.emit_pdf(doc, mode='printed')
|
|
137
|
-
|
|
138
|
-
|
|
157
|
+
assert not _symbol_font_names(pdf_bytes), (
|
|
158
|
+
'a /BaseFont /Symbol resource was registered for a run whose typeface '
|
|
159
|
+
'name resolves to an ordinary face')
|
|
139
160
|
content = _decoded_content(pdf_bytes)
|
|
140
161
|
# The LAST Tj in the stream is the third span's middle dot -- one bare
|
|
141
|
-
# 0xB7 byte,
|
|
162
|
+
# 0xB7 byte, never '?' (0x3f).
|
|
142
163
|
last_tj = list(re.finditer(rb'/(F\d+) \d+ Tf[^()]*\(([^()]*)\) Tj', content))[-1]
|
|
143
|
-
assert last_tj[1] in symbol_names, (
|
|
144
|
-
f'last Tj ({last_tj[2]!r}) is not under a Symbol font resource')
|
|
145
164
|
assert last_tj[2] == b'\xb7', (
|
|
146
|
-
f'
|
|
147
|
-
f'got {last_tj[2]!r}')
|
|
165
|
+
f'the middle dot did not survive as a real byte: got {last_tj[2]!r}')
|
|
148
166
|
|
|
149
167
|
|
|
150
168
|
def test_fontless_and_ordinary_font_spans_are_unaffected_baseline():
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|