wcwidth 0.8.4__tar.gz → 0.8.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (118) hide show
  1. {wcwidth-0.8.4 → wcwidth-0.8.5}/PKG-INFO +4 -4
  2. {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/api_c.rst +22 -23
  3. {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/history.rst +5 -0
  4. {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/intro.rst +3 -3
  5. {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/libwcwidth.rst +29 -28
  6. {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/specs.rst +2 -2
  7. {wcwidth-0.8.4 → wcwidth-0.8.5}/pyproject.toml +1 -1
  8. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_textwrap.py +10 -0
  9. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/__init__.py +10 -15
  10. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/_clip.py +5 -5
  11. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/_wcswidth.py +4 -4
  12. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/_width.py +4 -5
  13. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/align.py +6 -6
  14. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/escape_sequences.py +2 -2
  15. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/grapheme.py +1 -1
  16. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/hyperlink.py +1 -1
  17. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/textwrap.py +18 -20
  18. {wcwidth-0.8.4 → wcwidth-0.8.5}/.gitignore +0 -0
  19. {wcwidth-0.8.4 → wcwidth-0.8.5}/.pylintrc +0 -0
  20. {wcwidth-0.8.4 → wcwidth-0.8.5}/LICENSE +0 -0
  21. {wcwidth-0.8.4 → wcwidth-0.8.5}/README.rst +0 -0
  22. {wcwidth-0.8.4 → wcwidth-0.8.5}/bin/show-sequences +0 -0
  23. {wcwidth-0.8.4 → wcwidth-0.8.5}/bin/strip-sequences +0 -0
  24. {wcwidth-0.8.4 → wcwidth-0.8.5}/bin/update-docs.py +0 -0
  25. {wcwidth-0.8.4 → wcwidth-0.8.5}/bin/update-tables.py +0 -0
  26. {wcwidth-0.8.4 → wcwidth-0.8.5}/bin/verify-table-integrity.py +0 -0
  27. {wcwidth-0.8.4 → wcwidth-0.8.5}/bin/wcwidth-browser.py +0 -0
  28. {wcwidth-0.8.4 → wcwidth-0.8.5}/bin/wcwidth-libc-comparator.py +0 -0
  29. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/api_c.rst.j2 +0 -0
  30. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/c_gcb_class.c.j2 +0 -0
  31. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/c_table.c.j2 +0 -0
  32. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/c_term_programs.c.j2 +0 -0
  33. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/c_terminal_overrides.c.j2 +0 -0
  34. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/grapheme_override_per_terminal.py.j2 +0 -0
  35. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/grapheme_override_table.py.j2 +0 -0
  36. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/grapheme_registry.py.j2 +0 -0
  37. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/grapheme_table.c.j2 +0 -0
  38. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/grapheme_table.py.j2 +0 -0
  39. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/python_table.py.j2 +0 -0
  40. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/table_overrides.py.j2 +0 -0
  41. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/tables.h.j2 +0 -0
  42. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/term_programs.py.j2 +0 -0
  43. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/unicode_version.rst.j2 +0 -0
  44. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/unicode_versions.py.j2 +0 -0
  45. {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/wcwidth_config.h.j2 +0 -0
  46. {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/api.rst +0 -0
  47. {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/conf.py +0 -0
  48. {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/developing.rst +0 -0
  49. {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/index.rst +0 -0
  50. {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/related.rst +0 -0
  51. {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/requirements.txt +0 -0
  52. {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/unicode_version.rst +0 -0
  53. {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-develop.txt +0 -0
  54. {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-docs.in +0 -0
  55. {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-tests38.in +0 -0
  56. {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-tests38.txt +0 -0
  57. {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-tests39.in +0 -0
  58. {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-tests39.txt +0 -0
  59. {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-update.in +0 -0
  60. {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-update.txt +0 -0
  61. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/__init__.py +0 -0
  62. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/conftest.py +0 -0
  63. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_ambiguous.py +0 -0
  64. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_benchmarks.py +0 -0
  65. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_clip.py +0 -0
  66. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_clip_cjk_emoji.py +0 -0
  67. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_clip_overtyping.py +0 -0
  68. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_core.py +0 -0
  69. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_emojis.py +0 -0
  70. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_grapheme.py +0 -0
  71. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_hyperlink.py +0 -0
  72. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_justify.py +0 -0
  73. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_sgr_state.py +0 -0
  74. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_term_overrides.py +0 -0
  75. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_text_sizing.py +0 -0
  76. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_ucslevel.py +0 -0
  77. {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_width.py +0 -0
  78. {wcwidth-0.8.4 → wcwidth-0.8.5}/tox.ini +0 -0
  79. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/_constants.py +0 -0
  80. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/_wcwidth.py +0 -0
  81. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/bisearch.py +0 -0
  82. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/control_codes.py +0 -0
  83. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/py.typed +0 -0
  84. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/sgr_state.py +0 -0
  85. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_ambiguous.py +0 -0
  86. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme.py +0 -0
  87. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/__init__.py +0 -0
  88. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_058c7585.py +0 -0
  89. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_2148ff29.py +0 -0
  90. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_244dc88b.py +0 -0
  91. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_336a71a1.py +0 -0
  92. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_33ac73f2.py +0 -0
  93. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_3a1895d5.py +0 -0
  94. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_59180a0d.py +0 -0
  95. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_65c6beb5.py +0 -0
  96. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_78f828e8.py +0 -0
  97. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_8c2ae14d.py +0 -0
  98. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_900356bf.py +0 -0
  99. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_97904ece.py +0 -0
  100. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_a5880eeb.py +0 -0
  101. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_b6d33cf4.py +0 -0
  102. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_c57c295b.py +0 -0
  103. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_cf02297d.py +0 -0
  104. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_e18ac6a7.py +0 -0
  105. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_e2b75ee3.py +0 -0
  106. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_fb5fd79b.py +0 -0
  107. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_fdc99132.py +0 -0
  108. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_registry.py +0 -0
  109. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_mc.py +0 -0
  110. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_overrides.py +0 -0
  111. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_term_programs.py +0 -0
  112. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_vs15.py +0 -0
  113. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_vs16.py +0 -0
  114. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_wide.py +0 -0
  115. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_zero.py +0 -0
  116. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/text_sizing.py +0 -0
  117. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/unicode_versions.py +0 -0
  118. {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/wcwidth.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: wcwidth
3
- Version: 0.8.4
3
+ Version: 0.8.5
4
4
  Summary: Measures the displayed width of unicode strings in a terminal
5
5
  Project-URL: Homepage, https://github.com/jquast/wcwidth
6
6
  Author-email: Jeff Quast <contact@jeffquast.com>
@@ -61,9 +61,9 @@ Some examples of **incorrect results**:
61
61
  >>> 'コンニチハ'.rjust(11, 'X')
62
62
  'XXXXXXコンニチハ'
63
63
 
64
- >>> # result consumes 5 total cells, 6 expected,
65
- >>> 'café'.center(6, 'X')
66
- 'caféX'
64
+ >>> # combining acute accent: result consumes 5 total cells, 6 expected,
65
+ >>> 'cafe\u0301'.center(6, 'X')
66
+ 'caféX'
67
67
 
68
68
  Solution
69
69
  --------
@@ -20,9 +20,9 @@ Unicode character display width: wcwidth, wcswidth, wcstwidth.
20
20
  Return the display width of a single Unicode codepoint.
21
21
 
22
22
  Returns:
23
- 1 or 2 -- display cells occupied
24
- 0 -- zero-width codepoint (combining marks, ZWJ, etc.)
25
- -1 -- non-printable control character
23
+ 1 or 2: display cells occupied
24
+ 0: zero-width codepoint (combining marks, ZWJ, etc.)
25
+ -1: non-printable control character
26
26
 
27
27
  :param ambiguous_width: width for East Asian Ambiguous (A) characters. 1 = narrow (default), 2 = wide (CJK context).
28
28
 
@@ -47,7 +47,7 @@ Unicode character display width: wcwidth, wcswidth, wcstwidth.
47
47
 
48
48
  Terminal-aware variant of wcswidth_u32().
49
49
 
50
- :param term_program: canonical terminal name for override tables (e.g. "kitty", "xterm", "ghostty"). Use NULL for no terminal overrides.
50
+ :param term_program: terminal name for override tables (e.g. "kitty", "xterm", "ghostty"). Use NULL for no terminal overrides.
51
51
 
52
52
  .. c:function:: int wcstwidth_u8(const char *utf8, size_t n, int ambiguous_width, const char *term_program)
53
53
 
@@ -83,9 +83,9 @@ Main entry-points for string display width: wcwidth_width_u32 / wcwidth_width_u8
83
83
  wcwidth_width_u8(), and of the string transforms in clip.h and align.h.
84
84
  Distinct codes let callers distinguish the failure cause.
85
85
 
86
- Every out-param is int rather than wcwidth_error_t: the underlying type of
87
- an enum is implementation-defined, so int keeps the ABI stable across
88
- compilers. Compare against these constants directly.
86
+ Every out-param is int, whose size is fixed. The underlying type of an enum
87
+ is implementation-defined, so int keeps the ABI stable across compilers.
88
+ Compare against these constants directly.
89
89
 
90
90
  .. c:enumerator:: WCWIDTH_ERROR_NONE
91
91
 
@@ -144,7 +144,7 @@ Main entry-points for string display width: wcwidth_width_u32 / wcwidth_width_u8
144
144
 
145
145
  Measure the visible width of text, including terminal control sequences such
146
146
  as colors, bold, tabstops, cursor movement, and OSC 66 Text Sizing.
147
- wcwidth_width_u32() encodes its codepoints to UTF-8 and measures as wcwidth_width_u8().
147
+ wcwidth_width_u32() measures the codepoints directly.
148
148
 
149
149
  Returns the width in display cells, or -1 on error.
150
150
 
@@ -404,17 +404,17 @@ Clip text to a visible column range [v_start, v_end).
404
404
 
405
405
  Clip text to the visible column range [opts->v_start, opts->v_end).
406
406
 
407
- Returns a malloc'd string on success, NULL on error. When NULL is
408
- returned, \*error (from width.h) is WCWIDTH_ERROR_UNSUPPORTED for a
409
- terminal sequence this function does not support, another nonzero
410
- wcwidth_error_t for a WCWIDTH_STRICT violation, and WCWIDTH_ERROR_NONE
411
- for an allocation failure. \*error is always written on return.
407
+ Returns a malloc'd string on success, NULL on error. On success, \*out_len
408
+ receives the byte length of the result (excluding the NUL terminator, which
409
+ is always present), and the caller must free the returned pointer with a
410
+ single free() call. When NULL is returned, \*error (from width.h) is
411
+ WCWIDTH_ERROR_UNSUPPORTED for a terminal sequence this function does not
412
+ support, another nonzero wcwidth_error_t for a WCWIDTH_STRICT violation, and
413
+ WCWIDTH_ERROR_NONE for an allocation failure. \*error is always written on
414
+ return.
412
415
 
413
416
  Unsupported: horizontal cursor movement (BS, CR, CUF, CUB, HPA), OSC 8
414
417
  hyperlinks and OSC 66 text sizing.
415
- On success, \*out_len receives the byte length of the result
416
- (excluding NUL terminator, which is always present).
417
- The caller must free the returned pointer with a single free() call.
418
418
 
419
419
  :param text: UTF-8 encoded input string, NOT NUL-terminated.
420
420
  :param text_len: length of text in bytes.
@@ -596,9 +596,8 @@ UTF-8 decoding and encoding.
596
596
  is undefined behavior. Sets \*count\* and returns NULL on allocation
597
597
  failure.
598
598
 
599
- The result is not const, so the `if (p != stack) free(p);` release is a
600
- plain free() -- matching wcwidth_encode_u32() below. Treat the contents as
601
- read-only; the pointer is non-const only so ownership can be released.
599
+ The result type allows a plain free() release, matching wcwidth_encode_u32()
600
+ below. Treat the contents as read-only.
602
601
 
603
602
 
604
603
  .. c:function:: uint32_t *wcwidth_decode_u32_heap(const char *utf8, size_t n, size_t *count)
@@ -663,8 +662,8 @@ Grapheme cluster segmentation for UTF-8 text.
663
662
 
664
663
  .. c:function:: wcwidth_grapheme_iter_t *wcwidth_grapheme_iter_new_u32(const uint32_t *codepoints, size_t n)
665
664
 
666
- Iterate \*codepoints\*, which is borrowed, not copied, and must outlive the
667
- iterator. Returns NULL if allocation fails.
665
+ Iterate \*codepoints\*, which the iterator borrows; the caller must keep it
666
+ alive for the iterator's lifetime. Returns NULL if allocation fails.
668
667
 
669
668
 
670
669
  .. c:function:: const uint32_t *wcwidth_grapheme_next_u32(wcwidth_grapheme_iter_t *iter, size_t *out_len)
@@ -936,8 +935,8 @@ table_types.h
936
935
  Table data model: the interval type, the terminal-override record layouts,
937
936
  and the binary search over them.
938
937
 
939
- Hand-written. The tables themselves -- every WCWIDTH_* interval array, the
940
- terminal override and alias arrays, and their entry counts -- are declared
938
+ Hand-written. The tables themselves (every WCWIDTH_* interval array, the
939
+ terminal override and alias arrays, and their entry counts) are declared
941
940
  in tables.h, which update-tables.py generates.
942
941
 
943
942
  .. c:struct:: wcwidth_interval_t
@@ -1,6 +1,10 @@
1
1
  =======
2
2
  History
3
3
  =======
4
+
5
+ 0.8.5 *2026-09-23*
6
+ * **Bugfix** CJK hyphen break positions for `wrap()`_, `PR #262`_.
7
+
4
8
  0.8.4 *2026-09-17*
5
9
  * **Bugfix** `clip()`_ hangs with OSC 8 hyperlinks in some conditions, `PR #252`_.
6
10
  * **Bugfix** width of ITU T.416 colon-format SGR color parameters, `PR #243`_.
@@ -288,6 +292,7 @@ https://www.cl.cam.ac.uk/~mgk25/ucs/wcwidth.c::
288
292
  .. _`PR #252`: https://github.com/jquast/wcwidth/pull/252
289
293
  .. _`PR #253`: https://github.com/jquast/wcwidth/pull/253
290
294
  .. _`PR #259`: https://github.com/jquast/wcwidth/pull/259
295
+ .. _`PR #262`: https://github.com/jquast/wcwidth/pull/262
291
296
  .. _`Issue #101`: https://github.com/jquast/wcwidth/issues/101
292
297
  .. _`Issue #155`: https://github.com/jquast/wcwidth/issues/155
293
298
  .. _`Issue #211`: https://github.com/jquast/wcwidth/issues/211
@@ -30,9 +30,9 @@ Some examples of **incorrect results**:
30
30
  >>> 'コンニチハ'.rjust(11, 'X')
31
31
  'XXXXXXコンニチハ'
32
32
 
33
- >>> # result consumes 5 total cells, 6 expected,
34
- >>> 'café'.center(6, 'X')
35
- 'caféX'
33
+ >>> # combining acute accent: result consumes 5 total cells, 6 expected,
34
+ >>> 'cafe\u0301'.center(6, 'X')
35
+ 'caféX'
36
36
 
37
37
  Solution
38
38
  --------
@@ -6,8 +6,8 @@ A portable C11 library, mainly for CLI/TUI programs that carefully produce outpu
6
6
 
7
7
  This project is derived from the Python `wcwidth`_ project.
8
8
 
9
- The Python documentation_ closely matches this C library, except that the C API provides UTF-8 and
10
- codepoint array interfaces.
9
+ The Python documentation_ closely matches this C library. The C API adds UTF-8 and codepoint
10
+ array interfaces.
11
11
 
12
12
  The lowest-level functions are derived from POSIX.1-2001 and POSIX.1-2008 `wcwidth(3)`_ and
13
13
  `wcswidth(3)`_, which this library implements as `wcwidth_u32()`_ and `wcswidth_u32()`_. These
@@ -54,7 +54,7 @@ Example Programs
54
54
 
55
55
  Three small CLI utilities demonstrate use of this library.
56
56
 
57
- **textwrap** -- Unicode, CJK, emoji, and terminal sequence-aware text wrapping::
57
+ **textwrap**: Unicode, CJK, emoji, and terminal sequence-aware text wrapping::
58
58
 
59
59
  $ textwrap 42 README.rst
60
60
  ==========
@@ -70,7 +70,7 @@ Three small CLI utilities demonstrate use of this library.
70
70
  Uses environment value, ``$COLUMNS``, if no width argument is given. Use ``-v`` to append a red
71
71
  carriage-return marker.
72
72
 
73
- **width** -- report the display width of each line::
73
+ **width**: report the display width of each line::
74
74
 
75
75
  $ width README.rst
76
76
  10
@@ -86,7 +86,7 @@ carriage-return marker.
86
86
  $ width -v <<< "café résumé"
87
87
  11:café résumé
88
88
 
89
- **align** -- demonstrate left, right, and center alignment::
89
+ **align**: demonstrate left, right, and center alignment::
90
90
 
91
91
  $ echo "hello" | align 20
92
92
  hello hello hello
@@ -101,8 +101,8 @@ corrections, and grapheme clustering are discussed in the Python documentation_.
101
101
  Memory ownership
102
102
  ~~~~~~~~~~~~~~~~
103
103
 
104
- The text transforms allocate their result and return ``NULL`` on failure, be certain to ``free()``
105
- on success:
104
+ The text transforms allocate their result and return ``NULL`` on failure. The caller must
105
+ ``free()`` a successful result:
106
106
 
107
107
  .. code-block:: c
108
108
 
@@ -117,8 +117,8 @@ on success:
117
117
  free(out);
118
118
 
119
119
  `wcwidth_encode_u32()`_ and `wcwidth_decode_u32()`_ instead return the caller's scratch buffer
120
- if the result fits, allocating only when it does not. ``free()`` these with the condition that
121
- the result is at a new address:
120
+ if the result fits, allocating only when it does not. ``free()`` the result when it is at a new
121
+ address:
122
122
 
123
123
  .. code-block:: c
124
124
 
@@ -141,16 +141,16 @@ Every string function takes an explicit length and reads exactly that many units
141
141
  count bytes.
142
142
 
143
143
  There is no NUL-terminated sentinel form; pass ``strlen(text)`` when the text is a C string. The
144
- length is authoritative, so a NUL is an ordinary zero-width character rather than a terminator: it
145
- may appear anywhere, and survives into transform output, whose ``*out_len`` is the true length.
144
+ length is authoritative: a NUL is an ordinary zero-width character that may appear anywhere, and
145
+ it survives into transform output, whose ``*out_len`` is the true length.
146
146
 
147
147
  Alternate encodings
148
148
  ~~~~~~~~~~~~~~~~~~~
149
149
 
150
150
  Use ``_u8`` when your text is UTF-8 and ``_u32`` when you hold decoded codepoints; the two
151
- families mirror each other. Auxiliary strings are UTF-8 in *both* families -- the ``fillchar``
152
- padding argument and the ``initial_indent``/``subsequent_indent``/``placeholder`` wrap options --
153
- since they are short constants, not the text being processed.
151
+ families mirror each other. Auxiliary strings are UTF-8 in *both* families (the ``fillchar``
152
+ padding argument and the ``initial_indent``/``subsequent_indent``/``placeholder`` wrap options),
153
+ because they are short constants.
154
154
 
155
155
  Other encodings (Latin-1, CP437, Shift-JIS, ...) are transcoded by the caller; the library carries
156
156
  no encoding tables. Either transcode to UTF-8 once with iconv(3) or ICU and use the ``_u8`` forms
@@ -185,7 +185,7 @@ throughout, or use the ``_u32`` forms and re-encode the result. `wcwidth_encode
185
185
  free(out);
186
186
 
187
187
  Re-encoding to a legacy charset is the caller's iconv(3) or ICU (``ucnv_*``) call; a byte cast
188
- works only when every codepoint fits the target, where iconv reports ``EILSEQ`` instead.
188
+ works when every codepoint fits the target, and iconv reports ``EILSEQ`` when one does not.
189
189
 
190
190
  wcwidth_u32()
191
191
  ~~~~~~~~~~~~~
@@ -234,8 +234,8 @@ wcwidth_width_u32() and wcwidth_width_u8()
234
234
  ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
235
235
 
236
236
  Measure the visible width of text including terminal control sequences: colors, bold, tabstops,
237
- horizontal cursor movement, and OSC 66 Text Sizing. `wcwidth_width_u32()`_ encodes
238
- its codepoints to UTF-8 and measures as `wcwidth_width_u8()`_:
237
+ horizontal cursor movement, and OSC 66 Text Sizing. `wcwidth_width_u32()`_ measures the
238
+ codepoints directly:
239
239
 
240
240
  .. code-block:: c
241
241
 
@@ -313,7 +313,7 @@ graphemes with a fill string. Returns a ``malloc``\ 'd NUL-terminated string th
313
313
  free(out);
314
314
 
315
315
  Leave ``opts.v_end`` at its ``SIZE_MAX`` default to clip from ``v_start`` through the final column
316
- of *text*, without measuring it first -- the counterpart of the ``-1`` default of Python's `clip()`_:
316
+ of *text*, without measuring it first, matching the ``-1`` default of Python's `clip()`_:
317
317
 
318
318
  .. code-block:: c
319
319
 
@@ -326,9 +326,9 @@ of *text*, without measuring it first -- the counterpart of the ``-1`` default o
326
326
  Some sequences are unsupported, and `wcwidth_clip_u8()`_ returns ``NULL`` with ``*error`` set to
327
327
  ``WCWIDTH_ERROR_UNSUPPORTED``, these sequences are only supported in Python's `clip()`_:
328
328
 
329
- * **Horizontal cursor movement** -- BS, CR, and CSI ending in ``C`` (CUF), ``D`` (CUB) or ``G``
330
- (HPA). There is no counterpart to the ``overtyping`` option of Python's `clip()`_.
331
- * **OSC 8 hyperlinks** and **OSC 66 text sizing** are not supported.
329
+ * **Horizontal cursor movement**: BS, CR, and CSI ending in ``C`` (CUF), ``D`` (CUB) or ``G``
330
+ (HPA). Python's `clip()`_ additionally offers ``overtyping``.
331
+ * **OSC 8 hyperlinks** and **OSC 66 text sizing**: Python's `clip()`_ covers these.
332
332
 
333
333
  .. code-block:: c
334
334
 
@@ -366,7 +366,7 @@ breaks. Both emit a single ``malloc``\ 'd buffer of newline-separated lines:
366
366
  free(out);
367
367
 
368
368
  When the placeholder does not fit within the given width (``max_lines`` truncation),
369
- `wcwidth_wrap_u8()`_ returns ``-2`` rather than ``-1``, so callers can raise a tailored error.
369
+ `wcwidth_wrap_u8()`_ returns ``-2``, so callers can raise a tailored error.
370
370
  `wcwidth_wrap_lines_u8()`_ additionally reports each line's start offset in the output buffer, which
371
371
  matters when a line contains ``'\n'`` from the placeholder itself:
372
372
 
@@ -408,8 +408,8 @@ Differences from the Python package
408
408
  `wcwidth_width_u32()`_ and `wcwidth_width_u8()`_ parse only the sequences that move the cursor
409
409
  within a line or change how much room text occupies: SGR, horizontal cursor movement (CUF, CUB,
410
410
  HPA), and OSC 66 text sizing. Every other recognized sequence is zero-width. Screen clears,
411
- scrolls and vertical movement are indeterminate -- their column effect depends on terminal state
412
- the text does not carry -- so ``WCWIDTH_STRICT`` reports them as an error and the other modes
411
+ scrolls and vertical movement are indeterminate (their column effect depends on terminal state
412
+ the text does not carry), so ``WCWIDTH_STRICT`` reports them as an error and the other modes
413
413
  count them as zero-width.
414
414
 
415
415
  ``wcswidth_*()`` and ``wcstwidth_*()`` take no ``wcwidth_control_mode_t`` and return -1 for any
@@ -419,8 +419,8 @@ and `wcwidth_center_u8()`_ match the Python functions exactly.
419
419
  The text transforms are simpler:
420
420
 
421
421
  * `wcwidth_clip_u8()`_ rejects the unsupported sequences described above; every other sequence but
422
- SGR is zero-width, preserved where it appeared rather than clipped as a unit. It otherwise
423
- matches Python's `clip()`_, SGR included.
422
+ SGR is zero-width and preserved at its original position. It otherwise matches Python's
423
+ `clip()`_, SGR included.
424
424
  * `wcwidth_wrap_u8()`_ treats an OSC 8 hyperlink as an ordinary zero-width OSC, so the link is not
425
425
  re-opened on each line; callers must re-emit the opener and terminator themselves.
426
426
  * `wcwidth_wrap_u8()`_ and `wcwidth_wrap_u8_text()`_ split words on the ASCII space alone, where
@@ -432,7 +432,7 @@ Supported Terminals
432
432
  -------------------
433
433
 
434
434
  The ``term_program`` argument selects per-terminal corrections from generated override tables.
435
- The following canonical names are recognized; common ``TERM``/``TERM_PROGRAM`` aliases such as
435
+ The following terminal names are recognized; common ``TERM``/``TERM_PROGRAM`` aliases such as
436
436
  ``vscode`` and ``xterm-kitty`` resolve to them:
437
437
 
438
438
  .. BEGIN_LIST_TERM_PROGRAMS
@@ -445,7 +445,8 @@ The following canonical names are recognized; common ``TERM``/``TERM_PROGRAM`` a
445
445
  .. END_LIST_TERM_PROGRAMS
446
446
 
447
447
  For the most accurate corrections, query the terminal's software version via XTVERSION_
448
- (``CSI > q``) and pass the canonical name. See the Python Corrections_ documentation for details.
448
+ (``CSI > q``) and pass the name from the list above. See the Python Corrections_ documentation
449
+ for details.
449
450
 
450
451
  Unicode Version
451
452
  ---------------
@@ -136,8 +136,8 @@ described in the Virama section header).
136
136
 
137
137
  - A ``Virama`` contributes 0 width.
138
138
  - Most viramas have category ``Mn``, but six have category ``Mc``
139
- (`Spacing Combining Mark`_): these are recognised as viramas first,
140
- not as ``Mc``, so they begin a conjunct rather than capping the cluster.
139
+ (`Spacing Combining Mark`_): these are recognised as viramas, so they begin
140
+ a conjunct.
141
141
  - A ``Consonant`` immediately following a ``Virama`` adds its width to the
142
142
  current grapheme cluster.
143
143
  - The cluster total is capped at 2 cells since 0.8.0, `PR #224`_.
@@ -4,7 +4,7 @@ requires = [ "hatchling" ]
4
4
 
5
5
  [project]
6
6
  name = "wcwidth"
7
- version = "0.8.4" # stamped into __init__.py and wcwidth_config.h by bin/update-tables.py
7
+ version = "0.8.5" # stamped into __init__.py and wcwidth_config.h by bin/update-tables.py
8
8
  description = "Measures the displayed width of unicode strings in a terminal"
9
9
  readme = "README.rst"
10
10
  keywords = [
@@ -118,6 +118,16 @@ def test_wrap_long_words(text, w, break_long, expected):
118
118
  ('a---b', 2, True, True, ['a-', '--', 'b']),
119
119
  ('a-\x1b[31mb', 2, True, True, ['a-\x1b[31m\x1b[0m', '\x1b[31mb\x1b[0m']),
120
120
  ('a-\x1b[31mb', 2, True, False, ['a-\x1b[31m', 'b']),
121
+ ('古古-1abcdef', 3, True, True, ['古', '古-', '1ab', 'cde', 'f']),
122
+ ('古古-1abcdef', 4, True, True, ['古古', '-1ab', 'cdef']),
123
+ ('古古-1abcdef', 5, True, True, ['古古-', '1abcd', 'ef']),
124
+ ('e\u0301e\u0301-1abcdef', 4, True, True, ['e\u0301e\u0301-', '1abc', 'def']),
125
+ ('\U0001f469\U0001f469-1abc', 4, True, True, ['\U0001f469\U0001f469', '-1ab', 'c']),
126
+ ('\x1b[31m漢漢漢-é', 3, True, False, ['\x1b[31m漢', '漢', '漢-', 'é']),
127
+ ('\x1b[31m漢漢漢-é', 3, True, True,
128
+ ['\x1b[31m漢\x1b[0m', '\x1b[31m漢\x1b[0m', '\x1b[31m漢-\x1b[0m', '\x1b[31mé\x1b[0m']),
129
+ ('a-古古古', 6, True, False, ['a-', '古古古']),
130
+ ('a-古古古', 5, True, False, ['a-', '古古', '古']),
121
131
  ])
122
132
  def test_wrap_hyphen_long_words(text, w, break_hyphens, propagate, expected):
123
133
  assert wrap(text, w, break_on_hyphens=break_hyphens, propagate_sgr=propagate) == expected
@@ -47,22 +47,17 @@ from .table_ambiguous import AMBIGUOUS_EASTASIAN
47
47
  from .escape_sequences import iter_sequences, strip_sequences
48
48
  from .unicode_versions import list_versions
49
49
 
50
- # NOTE: this sort order is important for legacy import API compatibility before release 0.7.0
50
+ # Import order matters for legacy API compatibility (releases before 0.7.0).
51
51
  #
52
- # On Python < 3.15 the legacy submodule is eagerly pre-imported for backward compatibility
53
- # (populates sys.modules['wcwidth.wcwidth']). On 3.15+ __lazy_modules__ handles all submodules; the
54
- # legacy shim loads on-demand via file discovery when ``from wcwidth.wcwidth import ...`` is used.
52
+ # The first release put every function in a single 'wcwidth.py' file, and while the top-level
53
+ # 'from wcwidth import wcswidth' was always preferred, the deeper
54
+ # 'from wcwidth.wcwidth import wcswidth' form was always possible too. Both keep working.
55
+ #
56
+ # Below 3.15 the legacy submodule is pre-imported so sys.modules['wcwidth.wcwidth'] is populated
57
+ # during package initialization; a later ``import wcwidth.wcwidth`` would otherwise trigger on-disk
58
+ # file discovery and rebind that name from the function to the module object. On 3.15+
59
+ # __lazy_modules__ covers every submodule and the shim loads on demand.
55
60
  if __import__('sys').version_info < (3, 15):
56
- # Pre-import the legacy submodule so that sys.modules['wcwidth.wcwidth'] is populated during
57
- # package initialization. Without this, a later downstream dependent ``import wcwidth.wcwidth``
58
- # triggers on-disk file discovery which rebinds wcwidth.wcwidth from the function to the module
59
- # object.
60
- #
61
- # this is just a lot of carefulness for the original release that contained all functions in a
62
- # single 'wcwidth.py' file. Even though we always exposed our API at the top-level the preferred
63
- # 'from wcwidth import wcswidth', it was always possible to import them more directly,
64
- # 'from wcwidth.wcwidth import wcswidth'
65
- # -- and we make a lot of effort to allow any such import statements to continue to function.
66
61
  from . import wcwidth as _wcwidth_module # isort:skip
67
62
  from ._wcwidth import wcwidth, _wcmatch_version, _wcversion_value # isort:skip # pylint: disable=wrong-import-position
68
63
 
@@ -76,4 +71,4 @@ __all__ = ('wcwidth', 'wcswidth', 'wcstwidth', 'width', 'iter_sequences', 'iter_
76
71
  'Hyperlink', 'HyperlinkParams', 'TextSizing', 'TextSizingParams')
77
72
 
78
73
  # Version is stamped by code generation (bin/update-tables.py) from pyproject.toml.
79
- __version__ = '0.8.4'
74
+ __version__ = '0.8.5'
@@ -507,9 +507,9 @@ def _clip_painter(
507
507
  col = 0
508
508
  idx = 0
509
509
  # captured_style is a frozen snapshot of current_style taken at the first
510
- # visible character emitted within the clip window (start, end). It stays
511
- # None until that point. current_style, by contrast, is continuously
512
- # updated by SGR sequences throughout the scan.
510
+ # visible character emitted within the clip window (start, end); it stays
511
+ # None until that point. current_style is continuously updated by SGR
512
+ # sequences throughout the scan.
513
513
  #
514
514
  # When propagate_sgr is False, current_style (and therefore captured_style)
515
515
  # remain None, and SGR sequences pass through as literal text.
@@ -757,8 +757,8 @@ def clip(
757
757
  r"""
758
758
  Clip text to display columns (start, end) while preserving all terminal sequences.
759
759
 
760
- This function extracts a substring based on visible column positions rather than character
761
- indices. Terminal escape sequences are preserved in output. If a wide character (width of 2) is
760
+ This function extracts a substring using visible column positions. Terminal escape sequences
761
+ are preserved in output. If a wide character (width of 2) is
762
762
  split at a boundary, it is replaced with ``fillchar``.
763
763
 
764
764
  TAB characters (``\t``) are expanded to spaces up to the next tab stop, controlled by the
@@ -46,8 +46,8 @@ def _scan_zwj_cluster_end(text: str, start: int, end: int) -> int:
46
46
  idx += 1
47
47
  # GB11: \p{ExtPict} Extend* ZWJ × \p{ExtPict}
48
48
  # Extend modifiers (VS16, Fitzpatrick skin tones, etc.) attach to
49
- # the ExtPict *before* the ZWJ, not after it. After ZWJ the next
50
- # codepoint is always an ExtPict directly, no Extend skip needed.
49
+ # the ExtPict *before* the ZWJ. After ZWJ the next codepoint is
50
+ # always an ExtPict directly, no Extend skip needed.
51
51
  if idx < end and ord(text[idx]) in _EMOJI_ZWJ_SET:
52
52
  idx += 1
53
53
  # Skip trailing Extend (VS16, etc.) after ExtPict before next ZWJ
@@ -292,8 +292,8 @@ def wcstwidth(
292
292
  ucs = ord(char)
293
293
 
294
294
  #
295
- # Much of the logic below matches the logic in width(), but is repeated for improved
296
- # performance, they are given matching index reference numbers (starting at #5).
295
+ # Much of the logic below matches width(); it is repeated here for performance, with
296
+ # matching index reference numbers (starting at #5).
297
297
  #
298
298
  # 5. ZWJ (U+200D): consumed without contributing width.
299
299
  # Virama codepoints are treated as zero-width combining marks (Mn). When a
@@ -150,9 +150,8 @@ def width(
150
150
  1
151
151
  """
152
152
  # pylint: disable=too-complex,too-many-branches,too-many-statements,too-many-locals,redefined-variable-type,too-many-nested-blocks
153
- # This could be broken into sub-functions (#1, #3, and #6 especially), but for reduced overhead
154
- # in consideration of this function a likely "hot path", they are inline, breaking many pylint
155
- # complexity rules.
153
+ # This could be split into sub-functions (#1, #3 and #6 especially), but this function is a
154
+ # hot path, so the steps stay inline and the pylint complexity rules are disabled.
156
155
 
157
156
  # Fast path for ASCII printable (no tabs, escapes, or control chars)
158
157
  if text.isascii() and text.isprintable():
@@ -272,7 +271,7 @@ def width(
272
271
  current_col -= n_backward
273
272
  if current_col < 0:
274
273
  current_col = 0
275
- # 2d. OSC 66 Text Sizing — has positive display width
274
+ # 2d. OSC 66 Text Sizing: positive display width
276
275
  elif (ts_meta := m.group('ts_meta')) is not None:
277
276
  ts_text = m.group('ts_text') or ''
278
277
  ts_term = m.group('ts_term')
@@ -281,7 +280,7 @@ def width(
281
280
  TextSizingParams.from_params(ts_meta, control_codes=control_codes),
282
281
  ts_text, ts_term)
283
282
  current_col += text_size.display_width(ambiguous_width)
284
- # 2e. SGR and other zero-width sequences -- no column advance
283
+ # 2e. SGR and other zero-width sequences: no column advance
285
284
  idx = m.end()
286
285
  # Escape sequences break VS16 adjacency: reset last-measured state
287
286
  last_measured_idx = -2
@@ -22,8 +22,8 @@ def ljust(
22
22
  :param text: String to justify, may contain terminal sequences.
23
23
  :param dest_width: Total display width of result in terminal cells.
24
24
  :param fillchar: Single character for padding (default space). Must have
25
- display width of 1 (not wide, not zero-width, not combining). Unicode
26
- characters like ``'·'`` are acceptable. The width is not validated.
25
+ display width of 1. Unicode characters like ``'·'`` are acceptable.
26
+ The width is not validated.
27
27
  :param control_codes: How to handle control sequences when measuring.
28
28
  Passed to :func:`width` for measurement.
29
29
  :param ambiguous_width: Width to use for East Asian Ambiguous (A)
@@ -72,8 +72,8 @@ def rjust(
72
72
  :param text: String to justify, may contain terminal sequences.
73
73
  :param dest_width: Total display width of result in terminal cells.
74
74
  :param fillchar: Single character for padding (default space). Must have
75
- display width of 1 (not wide, not zero-width, not combining). Unicode
76
- characters like ``'·'`` are acceptable. The width is not validated.
75
+ display width of 1. Unicode characters like ``'·'`` are acceptable.
76
+ The width is not validated.
77
77
  :param control_codes: How to handle control sequences when measuring.
78
78
  Passed to :func:`width` for measurement.
79
79
  :param ambiguous_width: Width to use for East Asian Ambiguous (A)
@@ -122,8 +122,8 @@ def center(
122
122
  :param text: String to center, may contain terminal sequences.
123
123
  :param dest_width: Total display width of result in terminal cells.
124
124
  :param fillchar: Single character for padding (default space). Must have
125
- display width of 1 (not wide, not zero-width, not combining). Unicode
126
- characters like ``'·'`` are acceptable. The width is not validated.
125
+ display width of 1. Unicode characters like ``'·'`` are acceptable.
126
+ The width is not validated.
127
127
  :param control_codes: How to handle control sequences when measuring.
128
128
  Passed to :func:`width` for measurement.
129
129
  :param ambiguous_width: Width to use for East Asian Ambiguous (A)
@@ -24,8 +24,8 @@ TEXT_SIZING_PATTERN = re.compile(
24
24
  ZERO_WIDTH_PATTERN = re.compile(
25
25
  # CSI sequences
26
26
  r'\x1b\[[\x30-\x3f]*[\x20-\x2f]*[\x40-\x7e]|'
27
- # OSC sequences, note that text sizing protocol (OSC 66) is special case in width() and clip(),
28
- # and contrary to the variable name, it is positive width.
27
+ # OSC sequences; OSC 66 text sizing gets special handling in width() and clip() and has positive
28
+ # width, despite the ZERO_WIDTH_PATTERN name.
29
29
  r'\x1b\][^\x07\x1b]*(?:\x07|\x1b\\)|'
30
30
  # APC sequences
31
31
  r'\x1b_[^\x1b\x07]*(?:\x07|\x1b\\)|'
@@ -66,7 +66,7 @@ class GCB(IntEnum):
66
66
 
67
67
  # All lru_cache sizes in this file use maxsize=1024, chosen by benchmarking UDHR data (500+
68
68
  # languages) and considering typical process-long sessions: western scripts need ~64 unique
69
- # codepoints, but CJK could reach ~2000 -- but likely not.
69
+ # codepoints, and CJK can reach ~2000.
70
70
  @lru_cache(maxsize=1024)
71
71
  def _grapheme_cluster_break(ucs: int) -> GCB:
72
72
  # pylint: disable=too-many-branches,too-complex
@@ -79,7 +79,7 @@ class Hyperlink(typing.NamedTuple):
79
79
  close_end)`` or ``(-1, -1)`` if not found.
80
80
 
81
81
  Per the OSC 8 specification, terminal emulators treat hyperlinks as a
82
- state attribute, not as nested HTML anchors. A close sequence closes
82
+ state attribute. A close sequence closes
83
83
  the current hyperlink regardless of how many open sequences preceded it.
84
84
  """
85
85
  m = HYPERLINK_CLOSE_RE.search(text, open_end)
@@ -119,8 +119,7 @@ class SequenceTextWrapper(textwrap.TextWrapper):
119
119
  # Build a mapping from stripped text positions to original text positions.
120
120
  #
121
121
  # Track where each character ENDS so that sequences between characters
122
- # attach to the following text (not preceding text). This ensures sequences
123
- # aren't lost when whitespace is dropped.
122
+ # attach to the following text, keeping them when whitespace is dropped.
124
123
  #
125
124
  # char_end[i] = position in original text right after the i-th stripped char
126
125
  char_end: list[int] = []
@@ -197,15 +196,15 @@ class SequenceTextWrapper(textwrap.TextWrapper):
197
196
  """
198
197
  Wrap chunks into lines using sequence-aware width.
199
198
 
200
- Override TextWrapper._wrap_chunks to use _width instead of len. Follows stdlib's algorithm:
199
+ Override TextWrapper._wrap_chunks to measure with _width. Follows stdlib's algorithm:
201
200
  greedily fill lines, handle long words. Also handle OSC hyperlink processing. When
202
201
  hyperlinks span multiple lines, each line gets complete open/close sequences with matching
203
202
  id parameters for hover underlining continuity per OSC 8 spec.
204
203
  """
205
204
  # pylint: disable=too-many-branches,too-many-statements,too-complex,too-many-locals
206
205
  # pylint: disable=too-many-nested-blocks
207
- # the hyperlink code in particular really pushes the complexity rating of this method.
208
- # preferring to keep it "all in one method" because of so much local state and manipulation.
206
+ # The hyperlink code pushes the complexity rating of this method. It stays in one method
207
+ # because of the shared local state.
209
208
  if self.width <= 0:
210
209
  raise ValueError('invalid width %r (must be > 0)' % self.width)
211
210
  if not chunks:
@@ -250,7 +249,7 @@ class SequenceTextWrapper(textwrap.TextWrapper):
250
249
 
251
250
  # Drop leading whitespace (except at very start)
252
251
  # When dropping, transfer any sequences to the next chunk.
253
- # Only drop if there's actual whitespace text, not if it's only sequences.
252
+ # Only drop when actual whitespace text is present.
254
253
  stripped = self._strip_sequences(chunks[-1])
255
254
  if self.drop_whitespace and lines and stripped and not stripped.strip():
256
255
  sequences = self._extract_sequences(chunks[-1])
@@ -281,7 +280,7 @@ class SequenceTextWrapper(textwrap.TextWrapper):
281
280
 
282
281
  # Drop trailing whitespace
283
282
  # When dropping, transfer any sequences to the previous chunk.
284
- # Only drop if there's actual whitespace text, not if it's only sequences.
283
+ # Only drop when actual whitespace text is present.
285
284
  stripped_last = self._strip_sequences(current_line[-1]) if current_line else ''
286
285
  if (self.drop_whitespace and current_line and
287
286
  stripped_last and not stripped_last.strip()):
@@ -424,24 +423,23 @@ class SequenceTextWrapper(textwrap.TextWrapper):
424
423
  if self.break_long_words:
425
424
  break_at_hyphen = False
426
425
  hyphen_end = 0
426
+ # End of the prefix that fits within space_left by display width.
427
+ prefix_end = self._find_break_position(chunk, space_left)
427
428
 
428
- # Handle break_on_hyphens: find last hyphen within space_left
429
+ # Handle break_on_hyphens: find last hyphen in the portion that fits.
429
430
  if self.break_on_hyphens:
430
- # Strip sequences to find hyphen in logical text
431
- stripped = self._strip_sequences(chunk)
432
- if len(stripped) > space_left:
433
- # Find last hyphen in the portion that fits
434
- hyphen_pos = stripped.rfind('-', 0, space_left)
435
- if hyphen_pos > 0 and any(c != '-' for c in stripped[:hyphen_pos]):
436
- # Map back to original position including sequences
437
- hyphen_end = self._map_stripped_pos_to_original(chunk, hyphen_pos + 1)
438
- break_at_hyphen = True
431
+ stripped = self._strip_sequences(chunk[:prefix_end])
432
+ hyphen_pos = stripped.rfind('-')
433
+ if hyphen_pos > 0 and any(c != '-' for c in stripped[:hyphen_pos]):
434
+ # Map back to original position including sequences
435
+ hyphen_end = self._map_stripped_pos_to_original(chunk, hyphen_pos + 1)
436
+ break_at_hyphen = True
439
437
 
440
438
  # Break at grapheme boundaries to avoid splitting multi-codepoint characters
441
439
  if break_at_hyphen:
442
440
  actual_end = hyphen_end
443
441
  else:
444
- actual_end = self._find_break_position(chunk, space_left)
442
+ actual_end = prefix_end
445
443
  # Include first visible unit when break would take only leading sequences.
446
444
  if not cur_line and (
447
445
  actual_end == 0
@@ -551,8 +549,8 @@ def wrap(text: str, width: int = 70, *,
551
549
  r"""
552
550
  Wrap text to fit within given width, returning a list of wrapped lines.
553
551
 
554
- Like :func:`textwrap.wrap`, but measures width in display cells rather than
555
- characters, correctly handling wide characters, combining marks, and terminal
552
+ Like :func:`textwrap.wrap`, but measures width in display cells, correctly
553
+ handling wide characters, combining marks, and terminal
556
554
  escape sequences.
557
555
 
558
556
  :param text: Text to wrap, may contain terminal sequences.
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes