wcwidth 0.8.4__tar.gz → 0.8.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {wcwidth-0.8.4 → wcwidth-0.8.5}/PKG-INFO +4 -4
- {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/api_c.rst +22 -23
- {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/history.rst +5 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/intro.rst +3 -3
- {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/libwcwidth.rst +29 -28
- {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/specs.rst +2 -2
- {wcwidth-0.8.4 → wcwidth-0.8.5}/pyproject.toml +1 -1
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_textwrap.py +10 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/__init__.py +10 -15
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/_clip.py +5 -5
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/_wcswidth.py +4 -4
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/_width.py +4 -5
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/align.py +6 -6
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/escape_sequences.py +2 -2
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/grapheme.py +1 -1
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/hyperlink.py +1 -1
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/textwrap.py +18 -20
- {wcwidth-0.8.4 → wcwidth-0.8.5}/.gitignore +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/.pylintrc +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/LICENSE +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/README.rst +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/bin/show-sequences +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/bin/strip-sequences +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/bin/update-docs.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/bin/update-tables.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/bin/verify-table-integrity.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/bin/wcwidth-browser.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/bin/wcwidth-libc-comparator.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/api_c.rst.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/c_gcb_class.c.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/c_table.c.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/c_term_programs.c.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/c_terminal_overrides.c.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/grapheme_override_per_terminal.py.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/grapheme_override_table.py.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/grapheme_registry.py.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/grapheme_table.c.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/grapheme_table.py.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/python_table.py.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/table_overrides.py.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/tables.h.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/term_programs.py.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/unicode_version.rst.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/unicode_versions.py.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/code_templates/wcwidth_config.h.j2 +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/api.rst +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/conf.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/developing.rst +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/index.rst +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/related.rst +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/requirements.txt +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/docs/unicode_version.rst +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-develop.txt +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-docs.in +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-tests38.in +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-tests38.txt +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-tests39.in +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-tests39.txt +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-update.in +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/requirements-update.txt +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/__init__.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/conftest.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_ambiguous.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_benchmarks.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_clip.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_clip_cjk_emoji.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_clip_overtyping.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_core.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_emojis.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_grapheme.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_hyperlink.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_justify.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_sgr_state.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_term_overrides.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_text_sizing.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_ucslevel.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tests/test_width.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/tox.ini +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/_constants.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/_wcwidth.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/bisearch.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/control_codes.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/py.typed +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/sgr_state.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_ambiguous.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/__init__.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_058c7585.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_2148ff29.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_244dc88b.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_336a71a1.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_33ac73f2.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_3a1895d5.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_59180a0d.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_65c6beb5.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_78f828e8.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_8c2ae14d.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_900356bf.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_97904ece.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_a5880eeb.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_b6d33cf4.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_c57c295b.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_cf02297d.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_e18ac6a7.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_e2b75ee3.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_fb5fd79b.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_known_fdc99132.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_grapheme_overrides/_registry.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_mc.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_overrides.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_term_programs.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_vs15.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_vs16.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_wide.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/table_zero.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/text_sizing.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/unicode_versions.py +0 -0
- {wcwidth-0.8.4 → wcwidth-0.8.5}/wcwidth/wcwidth.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: wcwidth
|
|
3
|
-
Version: 0.8.
|
|
3
|
+
Version: 0.8.5
|
|
4
4
|
Summary: Measures the displayed width of unicode strings in a terminal
|
|
5
5
|
Project-URL: Homepage, https://github.com/jquast/wcwidth
|
|
6
6
|
Author-email: Jeff Quast <contact@jeffquast.com>
|
|
@@ -61,9 +61,9 @@ Some examples of **incorrect results**:
|
|
|
61
61
|
>>> 'コンニチハ'.rjust(11, 'X')
|
|
62
62
|
'XXXXXXコンニチハ'
|
|
63
63
|
|
|
64
|
-
>>> # result consumes 5 total cells, 6 expected,
|
|
65
|
-
>>> '
|
|
66
|
-
'
|
|
64
|
+
>>> # combining acute accent: result consumes 5 total cells, 6 expected,
|
|
65
|
+
>>> 'cafe\u0301'.center(6, 'X')
|
|
66
|
+
'caféX'
|
|
67
67
|
|
|
68
68
|
Solution
|
|
69
69
|
--------
|
|
@@ -20,9 +20,9 @@ Unicode character display width: wcwidth, wcswidth, wcstwidth.
|
|
|
20
20
|
Return the display width of a single Unicode codepoint.
|
|
21
21
|
|
|
22
22
|
Returns:
|
|
23
|
-
1 or 2
|
|
24
|
-
0
|
|
25
|
-
-1
|
|
23
|
+
1 or 2: display cells occupied
|
|
24
|
+
0: zero-width codepoint (combining marks, ZWJ, etc.)
|
|
25
|
+
-1: non-printable control character
|
|
26
26
|
|
|
27
27
|
:param ambiguous_width: width for East Asian Ambiguous (A) characters. 1 = narrow (default), 2 = wide (CJK context).
|
|
28
28
|
|
|
@@ -47,7 +47,7 @@ Unicode character display width: wcwidth, wcswidth, wcstwidth.
|
|
|
47
47
|
|
|
48
48
|
Terminal-aware variant of wcswidth_u32().
|
|
49
49
|
|
|
50
|
-
:param term_program:
|
|
50
|
+
:param term_program: terminal name for override tables (e.g. "kitty", "xterm", "ghostty"). Use NULL for no terminal overrides.
|
|
51
51
|
|
|
52
52
|
.. c:function:: int wcstwidth_u8(const char *utf8, size_t n, int ambiguous_width, const char *term_program)
|
|
53
53
|
|
|
@@ -83,9 +83,9 @@ Main entry-points for string display width: wcwidth_width_u32 / wcwidth_width_u8
|
|
|
83
83
|
wcwidth_width_u8(), and of the string transforms in clip.h and align.h.
|
|
84
84
|
Distinct codes let callers distinguish the failure cause.
|
|
85
85
|
|
|
86
|
-
Every out-param is int
|
|
87
|
-
|
|
88
|
-
|
|
86
|
+
Every out-param is int, whose size is fixed. The underlying type of an enum
|
|
87
|
+
is implementation-defined, so int keeps the ABI stable across compilers.
|
|
88
|
+
Compare against these constants directly.
|
|
89
89
|
|
|
90
90
|
.. c:enumerator:: WCWIDTH_ERROR_NONE
|
|
91
91
|
|
|
@@ -144,7 +144,7 @@ Main entry-points for string display width: wcwidth_width_u32 / wcwidth_width_u8
|
|
|
144
144
|
|
|
145
145
|
Measure the visible width of text, including terminal control sequences such
|
|
146
146
|
as colors, bold, tabstops, cursor movement, and OSC 66 Text Sizing.
|
|
147
|
-
wcwidth_width_u32()
|
|
147
|
+
wcwidth_width_u32() measures the codepoints directly.
|
|
148
148
|
|
|
149
149
|
Returns the width in display cells, or -1 on error.
|
|
150
150
|
|
|
@@ -404,17 +404,17 @@ Clip text to a visible column range [v_start, v_end).
|
|
|
404
404
|
|
|
405
405
|
Clip text to the visible column range [opts->v_start, opts->v_end).
|
|
406
406
|
|
|
407
|
-
Returns a malloc'd string on success, NULL on error.
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
for
|
|
407
|
+
Returns a malloc'd string on success, NULL on error. On success, \*out_len
|
|
408
|
+
receives the byte length of the result (excluding the NUL terminator, which
|
|
409
|
+
is always present), and the caller must free the returned pointer with a
|
|
410
|
+
single free() call. When NULL is returned, \*error (from width.h) is
|
|
411
|
+
WCWIDTH_ERROR_UNSUPPORTED for a terminal sequence this function does not
|
|
412
|
+
support, another nonzero wcwidth_error_t for a WCWIDTH_STRICT violation, and
|
|
413
|
+
WCWIDTH_ERROR_NONE for an allocation failure. \*error is always written on
|
|
414
|
+
return.
|
|
412
415
|
|
|
413
416
|
Unsupported: horizontal cursor movement (BS, CR, CUF, CUB, HPA), OSC 8
|
|
414
417
|
hyperlinks and OSC 66 text sizing.
|
|
415
|
-
On success, \*out_len receives the byte length of the result
|
|
416
|
-
(excluding NUL terminator, which is always present).
|
|
417
|
-
The caller must free the returned pointer with a single free() call.
|
|
418
418
|
|
|
419
419
|
:param text: UTF-8 encoded input string, NOT NUL-terminated.
|
|
420
420
|
:param text_len: length of text in bytes.
|
|
@@ -596,9 +596,8 @@ UTF-8 decoding and encoding.
|
|
|
596
596
|
is undefined behavior. Sets \*count\* and returns NULL on allocation
|
|
597
597
|
failure.
|
|
598
598
|
|
|
599
|
-
The result
|
|
600
|
-
|
|
601
|
-
read-only; the pointer is non-const only so ownership can be released.
|
|
599
|
+
The result type allows a plain free() release, matching wcwidth_encode_u32()
|
|
600
|
+
below. Treat the contents as read-only.
|
|
602
601
|
|
|
603
602
|
|
|
604
603
|
.. c:function:: uint32_t *wcwidth_decode_u32_heap(const char *utf8, size_t n, size_t *count)
|
|
@@ -663,8 +662,8 @@ Grapheme cluster segmentation for UTF-8 text.
|
|
|
663
662
|
|
|
664
663
|
.. c:function:: wcwidth_grapheme_iter_t *wcwidth_grapheme_iter_new_u32(const uint32_t *codepoints, size_t n)
|
|
665
664
|
|
|
666
|
-
Iterate \*codepoints\*, which
|
|
667
|
-
iterator. Returns NULL if allocation fails.
|
|
665
|
+
Iterate \*codepoints\*, which the iterator borrows; the caller must keep it
|
|
666
|
+
alive for the iterator's lifetime. Returns NULL if allocation fails.
|
|
668
667
|
|
|
669
668
|
|
|
670
669
|
.. c:function:: const uint32_t *wcwidth_grapheme_next_u32(wcwidth_grapheme_iter_t *iter, size_t *out_len)
|
|
@@ -936,8 +935,8 @@ table_types.h
|
|
|
936
935
|
Table data model: the interval type, the terminal-override record layouts,
|
|
937
936
|
and the binary search over them.
|
|
938
937
|
|
|
939
|
-
Hand-written. The tables themselves
|
|
940
|
-
terminal override and alias arrays, and their entry counts
|
|
938
|
+
Hand-written. The tables themselves (every WCWIDTH_* interval array, the
|
|
939
|
+
terminal override and alias arrays, and their entry counts) are declared
|
|
941
940
|
in tables.h, which update-tables.py generates.
|
|
942
941
|
|
|
943
942
|
.. c:struct:: wcwidth_interval_t
|
|
@@ -1,6 +1,10 @@
|
|
|
1
1
|
=======
|
|
2
2
|
History
|
|
3
3
|
=======
|
|
4
|
+
|
|
5
|
+
0.8.5 *2026-09-23*
|
|
6
|
+
* **Bugfix** CJK hyphen break positions for `wrap()`_, `PR #262`_.
|
|
7
|
+
|
|
4
8
|
0.8.4 *2026-09-17*
|
|
5
9
|
* **Bugfix** `clip()`_ hangs with OSC 8 hyperlinks in some conditions, `PR #252`_.
|
|
6
10
|
* **Bugfix** width of ITU T.416 colon-format SGR color parameters, `PR #243`_.
|
|
@@ -288,6 +292,7 @@ https://www.cl.cam.ac.uk/~mgk25/ucs/wcwidth.c::
|
|
|
288
292
|
.. _`PR #252`: https://github.com/jquast/wcwidth/pull/252
|
|
289
293
|
.. _`PR #253`: https://github.com/jquast/wcwidth/pull/253
|
|
290
294
|
.. _`PR #259`: https://github.com/jquast/wcwidth/pull/259
|
|
295
|
+
.. _`PR #262`: https://github.com/jquast/wcwidth/pull/262
|
|
291
296
|
.. _`Issue #101`: https://github.com/jquast/wcwidth/issues/101
|
|
292
297
|
.. _`Issue #155`: https://github.com/jquast/wcwidth/issues/155
|
|
293
298
|
.. _`Issue #211`: https://github.com/jquast/wcwidth/issues/211
|
|
@@ -30,9 +30,9 @@ Some examples of **incorrect results**:
|
|
|
30
30
|
>>> 'コンニチハ'.rjust(11, 'X')
|
|
31
31
|
'XXXXXXコンニチハ'
|
|
32
32
|
|
|
33
|
-
>>> # result consumes 5 total cells, 6 expected,
|
|
34
|
-
>>> '
|
|
35
|
-
'
|
|
33
|
+
>>> # combining acute accent: result consumes 5 total cells, 6 expected,
|
|
34
|
+
>>> 'cafe\u0301'.center(6, 'X')
|
|
35
|
+
'caféX'
|
|
36
36
|
|
|
37
37
|
Solution
|
|
38
38
|
--------
|
|
@@ -6,8 +6,8 @@ A portable C11 library, mainly for CLI/TUI programs that carefully produce outpu
|
|
|
6
6
|
|
|
7
7
|
This project is derived from the Python `wcwidth`_ project.
|
|
8
8
|
|
|
9
|
-
The Python documentation_ closely matches this C library
|
|
10
|
-
|
|
9
|
+
The Python documentation_ closely matches this C library. The C API adds UTF-8 and codepoint
|
|
10
|
+
array interfaces.
|
|
11
11
|
|
|
12
12
|
The lowest-level functions are derived from POSIX.1-2001 and POSIX.1-2008 `wcwidth(3)`_ and
|
|
13
13
|
`wcswidth(3)`_, which this library implements as `wcwidth_u32()`_ and `wcswidth_u32()`_. These
|
|
@@ -54,7 +54,7 @@ Example Programs
|
|
|
54
54
|
|
|
55
55
|
Three small CLI utilities demonstrate use of this library.
|
|
56
56
|
|
|
57
|
-
**textwrap
|
|
57
|
+
**textwrap**: Unicode, CJK, emoji, and terminal sequence-aware text wrapping::
|
|
58
58
|
|
|
59
59
|
$ textwrap 42 README.rst
|
|
60
60
|
==========
|
|
@@ -70,7 +70,7 @@ Three small CLI utilities demonstrate use of this library.
|
|
|
70
70
|
Uses environment value, ``$COLUMNS``, if no width argument is given. Use ``-v`` to append a red
|
|
71
71
|
carriage-return marker.
|
|
72
72
|
|
|
73
|
-
**width
|
|
73
|
+
**width**: report the display width of each line::
|
|
74
74
|
|
|
75
75
|
$ width README.rst
|
|
76
76
|
10
|
|
@@ -86,7 +86,7 @@ carriage-return marker.
|
|
|
86
86
|
$ width -v <<< "café résumé"
|
|
87
87
|
11:café résumé
|
|
88
88
|
|
|
89
|
-
**align
|
|
89
|
+
**align**: demonstrate left, right, and center alignment::
|
|
90
90
|
|
|
91
91
|
$ echo "hello" | align 20
|
|
92
92
|
hello hello hello
|
|
@@ -101,8 +101,8 @@ corrections, and grapheme clustering are discussed in the Python documentation_.
|
|
|
101
101
|
Memory ownership
|
|
102
102
|
~~~~~~~~~~~~~~~~
|
|
103
103
|
|
|
104
|
-
The text transforms allocate their result and return ``NULL`` on failure
|
|
105
|
-
|
|
104
|
+
The text transforms allocate their result and return ``NULL`` on failure. The caller must
|
|
105
|
+
``free()`` a successful result:
|
|
106
106
|
|
|
107
107
|
.. code-block:: c
|
|
108
108
|
|
|
@@ -117,8 +117,8 @@ on success:
|
|
|
117
117
|
free(out);
|
|
118
118
|
|
|
119
119
|
`wcwidth_encode_u32()`_ and `wcwidth_decode_u32()`_ instead return the caller's scratch buffer
|
|
120
|
-
if the result fits, allocating only when it does not. ``free()``
|
|
121
|
-
|
|
120
|
+
if the result fits, allocating only when it does not. ``free()`` the result when it is at a new
|
|
121
|
+
address:
|
|
122
122
|
|
|
123
123
|
.. code-block:: c
|
|
124
124
|
|
|
@@ -141,16 +141,16 @@ Every string function takes an explicit length and reads exactly that many units
|
|
|
141
141
|
count bytes.
|
|
142
142
|
|
|
143
143
|
There is no NUL-terminated sentinel form; pass ``strlen(text)`` when the text is a C string. The
|
|
144
|
-
length is authoritative
|
|
145
|
-
|
|
144
|
+
length is authoritative: a NUL is an ordinary zero-width character that may appear anywhere, and
|
|
145
|
+
it survives into transform output, whose ``*out_len`` is the true length.
|
|
146
146
|
|
|
147
147
|
Alternate encodings
|
|
148
148
|
~~~~~~~~~~~~~~~~~~~
|
|
149
149
|
|
|
150
150
|
Use ``_u8`` when your text is UTF-8 and ``_u32`` when you hold decoded codepoints; the two
|
|
151
|
-
families mirror each other. Auxiliary strings are UTF-8 in *both* families
|
|
152
|
-
padding argument and the ``initial_indent``/``subsequent_indent``/``placeholder`` wrap options
|
|
153
|
-
|
|
151
|
+
families mirror each other. Auxiliary strings are UTF-8 in *both* families (the ``fillchar``
|
|
152
|
+
padding argument and the ``initial_indent``/``subsequent_indent``/``placeholder`` wrap options),
|
|
153
|
+
because they are short constants.
|
|
154
154
|
|
|
155
155
|
Other encodings (Latin-1, CP437, Shift-JIS, ...) are transcoded by the caller; the library carries
|
|
156
156
|
no encoding tables. Either transcode to UTF-8 once with iconv(3) or ICU and use the ``_u8`` forms
|
|
@@ -185,7 +185,7 @@ throughout, or use the ``_u32`` forms and re-encode the result. `wcwidth_encode
|
|
|
185
185
|
free(out);
|
|
186
186
|
|
|
187
187
|
Re-encoding to a legacy charset is the caller's iconv(3) or ICU (``ucnv_*``) call; a byte cast
|
|
188
|
-
works
|
|
188
|
+
works when every codepoint fits the target, and iconv reports ``EILSEQ`` when one does not.
|
|
189
189
|
|
|
190
190
|
wcwidth_u32()
|
|
191
191
|
~~~~~~~~~~~~~
|
|
@@ -234,8 +234,8 @@ wcwidth_width_u32() and wcwidth_width_u8()
|
|
|
234
234
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
235
235
|
|
|
236
236
|
Measure the visible width of text including terminal control sequences: colors, bold, tabstops,
|
|
237
|
-
horizontal cursor movement, and OSC 66 Text Sizing. `wcwidth_width_u32()`_
|
|
238
|
-
|
|
237
|
+
horizontal cursor movement, and OSC 66 Text Sizing. `wcwidth_width_u32()`_ measures the
|
|
238
|
+
codepoints directly:
|
|
239
239
|
|
|
240
240
|
.. code-block:: c
|
|
241
241
|
|
|
@@ -313,7 +313,7 @@ graphemes with a fill string. Returns a ``malloc``\ 'd NUL-terminated string th
|
|
|
313
313
|
free(out);
|
|
314
314
|
|
|
315
315
|
Leave ``opts.v_end`` at its ``SIZE_MAX`` default to clip from ``v_start`` through the final column
|
|
316
|
-
of *text*, without measuring it first
|
|
316
|
+
of *text*, without measuring it first, matching the ``-1`` default of Python's `clip()`_:
|
|
317
317
|
|
|
318
318
|
.. code-block:: c
|
|
319
319
|
|
|
@@ -326,9 +326,9 @@ of *text*, without measuring it first -- the counterpart of the ``-1`` default o
|
|
|
326
326
|
Some sequences are unsupported, and `wcwidth_clip_u8()`_ returns ``NULL`` with ``*error`` set to
|
|
327
327
|
``WCWIDTH_ERROR_UNSUPPORTED``, these sequences are only supported in Python's `clip()`_:
|
|
328
328
|
|
|
329
|
-
* **Horizontal cursor movement
|
|
330
|
-
(HPA).
|
|
331
|
-
* **OSC 8 hyperlinks** and **OSC 66 text sizing
|
|
329
|
+
* **Horizontal cursor movement**: BS, CR, and CSI ending in ``C`` (CUF), ``D`` (CUB) or ``G``
|
|
330
|
+
(HPA). Python's `clip()`_ additionally offers ``overtyping``.
|
|
331
|
+
* **OSC 8 hyperlinks** and **OSC 66 text sizing**: Python's `clip()`_ covers these.
|
|
332
332
|
|
|
333
333
|
.. code-block:: c
|
|
334
334
|
|
|
@@ -366,7 +366,7 @@ breaks. Both emit a single ``malloc``\ 'd buffer of newline-separated lines:
|
|
|
366
366
|
free(out);
|
|
367
367
|
|
|
368
368
|
When the placeholder does not fit within the given width (``max_lines`` truncation),
|
|
369
|
-
`wcwidth_wrap_u8()`_ returns ``-2
|
|
369
|
+
`wcwidth_wrap_u8()`_ returns ``-2``, so callers can raise a tailored error.
|
|
370
370
|
`wcwidth_wrap_lines_u8()`_ additionally reports each line's start offset in the output buffer, which
|
|
371
371
|
matters when a line contains ``'\n'`` from the placeholder itself:
|
|
372
372
|
|
|
@@ -408,8 +408,8 @@ Differences from the Python package
|
|
|
408
408
|
`wcwidth_width_u32()`_ and `wcwidth_width_u8()`_ parse only the sequences that move the cursor
|
|
409
409
|
within a line or change how much room text occupies: SGR, horizontal cursor movement (CUF, CUB,
|
|
410
410
|
HPA), and OSC 66 text sizing. Every other recognized sequence is zero-width. Screen clears,
|
|
411
|
-
scrolls and vertical movement are indeterminate
|
|
412
|
-
the text does not carry
|
|
411
|
+
scrolls and vertical movement are indeterminate (their column effect depends on terminal state
|
|
412
|
+
the text does not carry), so ``WCWIDTH_STRICT`` reports them as an error and the other modes
|
|
413
413
|
count them as zero-width.
|
|
414
414
|
|
|
415
415
|
``wcswidth_*()`` and ``wcstwidth_*()`` take no ``wcwidth_control_mode_t`` and return -1 for any
|
|
@@ -419,8 +419,8 @@ and `wcwidth_center_u8()`_ match the Python functions exactly.
|
|
|
419
419
|
The text transforms are simpler:
|
|
420
420
|
|
|
421
421
|
* `wcwidth_clip_u8()`_ rejects the unsupported sequences described above; every other sequence but
|
|
422
|
-
SGR is zero-width
|
|
423
|
-
|
|
422
|
+
SGR is zero-width and preserved at its original position. It otherwise matches Python's
|
|
423
|
+
`clip()`_, SGR included.
|
|
424
424
|
* `wcwidth_wrap_u8()`_ treats an OSC 8 hyperlink as an ordinary zero-width OSC, so the link is not
|
|
425
425
|
re-opened on each line; callers must re-emit the opener and terminator themselves.
|
|
426
426
|
* `wcwidth_wrap_u8()`_ and `wcwidth_wrap_u8_text()`_ split words on the ASCII space alone, where
|
|
@@ -432,7 +432,7 @@ Supported Terminals
|
|
|
432
432
|
-------------------
|
|
433
433
|
|
|
434
434
|
The ``term_program`` argument selects per-terminal corrections from generated override tables.
|
|
435
|
-
The following
|
|
435
|
+
The following terminal names are recognized; common ``TERM``/``TERM_PROGRAM`` aliases such as
|
|
436
436
|
``vscode`` and ``xterm-kitty`` resolve to them:
|
|
437
437
|
|
|
438
438
|
.. BEGIN_LIST_TERM_PROGRAMS
|
|
@@ -445,7 +445,8 @@ The following canonical names are recognized; common ``TERM``/``TERM_PROGRAM`` a
|
|
|
445
445
|
.. END_LIST_TERM_PROGRAMS
|
|
446
446
|
|
|
447
447
|
For the most accurate corrections, query the terminal's software version via XTVERSION_
|
|
448
|
-
(``CSI > q``) and pass the
|
|
448
|
+
(``CSI > q``) and pass the name from the list above. See the Python Corrections_ documentation
|
|
449
|
+
for details.
|
|
449
450
|
|
|
450
451
|
Unicode Version
|
|
451
452
|
---------------
|
|
@@ -136,8 +136,8 @@ described in the Virama section header).
|
|
|
136
136
|
|
|
137
137
|
- A ``Virama`` contributes 0 width.
|
|
138
138
|
- Most viramas have category ``Mn``, but six have category ``Mc``
|
|
139
|
-
(`Spacing Combining Mark`_): these are recognised as viramas
|
|
140
|
-
|
|
139
|
+
(`Spacing Combining Mark`_): these are recognised as viramas, so they begin
|
|
140
|
+
a conjunct.
|
|
141
141
|
- A ``Consonant`` immediately following a ``Virama`` adds its width to the
|
|
142
142
|
current grapheme cluster.
|
|
143
143
|
- The cluster total is capped at 2 cells since 0.8.0, `PR #224`_.
|
|
@@ -4,7 +4,7 @@ requires = [ "hatchling" ]
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "wcwidth"
|
|
7
|
-
version = "0.8.
|
|
7
|
+
version = "0.8.5" # stamped into __init__.py and wcwidth_config.h by bin/update-tables.py
|
|
8
8
|
description = "Measures the displayed width of unicode strings in a terminal"
|
|
9
9
|
readme = "README.rst"
|
|
10
10
|
keywords = [
|
|
@@ -118,6 +118,16 @@ def test_wrap_long_words(text, w, break_long, expected):
|
|
|
118
118
|
('a---b', 2, True, True, ['a-', '--', 'b']),
|
|
119
119
|
('a-\x1b[31mb', 2, True, True, ['a-\x1b[31m\x1b[0m', '\x1b[31mb\x1b[0m']),
|
|
120
120
|
('a-\x1b[31mb', 2, True, False, ['a-\x1b[31m', 'b']),
|
|
121
|
+
('古古-1abcdef', 3, True, True, ['古', '古-', '1ab', 'cde', 'f']),
|
|
122
|
+
('古古-1abcdef', 4, True, True, ['古古', '-1ab', 'cdef']),
|
|
123
|
+
('古古-1abcdef', 5, True, True, ['古古-', '1abcd', 'ef']),
|
|
124
|
+
('e\u0301e\u0301-1abcdef', 4, True, True, ['e\u0301e\u0301-', '1abc', 'def']),
|
|
125
|
+
('\U0001f469\U0001f469-1abc', 4, True, True, ['\U0001f469\U0001f469', '-1ab', 'c']),
|
|
126
|
+
('\x1b[31m漢漢漢-é', 3, True, False, ['\x1b[31m漢', '漢', '漢-', 'é']),
|
|
127
|
+
('\x1b[31m漢漢漢-é', 3, True, True,
|
|
128
|
+
['\x1b[31m漢\x1b[0m', '\x1b[31m漢\x1b[0m', '\x1b[31m漢-\x1b[0m', '\x1b[31mé\x1b[0m']),
|
|
129
|
+
('a-古古古', 6, True, False, ['a-', '古古古']),
|
|
130
|
+
('a-古古古', 5, True, False, ['a-', '古古', '古']),
|
|
121
131
|
])
|
|
122
132
|
def test_wrap_hyphen_long_words(text, w, break_hyphens, propagate, expected):
|
|
123
133
|
assert wrap(text, w, break_on_hyphens=break_hyphens, propagate_sgr=propagate) == expected
|
|
@@ -47,22 +47,17 @@ from .table_ambiguous import AMBIGUOUS_EASTASIAN
|
|
|
47
47
|
from .escape_sequences import iter_sequences, strip_sequences
|
|
48
48
|
from .unicode_versions import list_versions
|
|
49
49
|
|
|
50
|
-
#
|
|
50
|
+
# Import order matters for legacy API compatibility (releases before 0.7.0).
|
|
51
51
|
#
|
|
52
|
-
#
|
|
53
|
-
#
|
|
54
|
-
#
|
|
52
|
+
# The first release put every function in a single 'wcwidth.py' file, and while the top-level
|
|
53
|
+
# 'from wcwidth import wcswidth' was always preferred, the deeper
|
|
54
|
+
# 'from wcwidth.wcwidth import wcswidth' form was always possible too. Both keep working.
|
|
55
|
+
#
|
|
56
|
+
# Below 3.15 the legacy submodule is pre-imported so sys.modules['wcwidth.wcwidth'] is populated
|
|
57
|
+
# during package initialization; a later ``import wcwidth.wcwidth`` would otherwise trigger on-disk
|
|
58
|
+
# file discovery and rebind that name from the function to the module object. On 3.15+
|
|
59
|
+
# __lazy_modules__ covers every submodule and the shim loads on demand.
|
|
55
60
|
if __import__('sys').version_info < (3, 15):
|
|
56
|
-
# Pre-import the legacy submodule so that sys.modules['wcwidth.wcwidth'] is populated during
|
|
57
|
-
# package initialization. Without this, a later downstream dependent ``import wcwidth.wcwidth``
|
|
58
|
-
# triggers on-disk file discovery which rebinds wcwidth.wcwidth from the function to the module
|
|
59
|
-
# object.
|
|
60
|
-
#
|
|
61
|
-
# this is just a lot of carefulness for the original release that contained all functions in a
|
|
62
|
-
# single 'wcwidth.py' file. Even though we always exposed our API at the top-level the preferred
|
|
63
|
-
# 'from wcwidth import wcswidth', it was always possible to import them more directly,
|
|
64
|
-
# 'from wcwidth.wcwidth import wcswidth'
|
|
65
|
-
# -- and we make a lot of effort to allow any such import statements to continue to function.
|
|
66
61
|
from . import wcwidth as _wcwidth_module # isort:skip
|
|
67
62
|
from ._wcwidth import wcwidth, _wcmatch_version, _wcversion_value # isort:skip # pylint: disable=wrong-import-position
|
|
68
63
|
|
|
@@ -76,4 +71,4 @@ __all__ = ('wcwidth', 'wcswidth', 'wcstwidth', 'width', 'iter_sequences', 'iter_
|
|
|
76
71
|
'Hyperlink', 'HyperlinkParams', 'TextSizing', 'TextSizingParams')
|
|
77
72
|
|
|
78
73
|
# Version is stamped by code generation (bin/update-tables.py) from pyproject.toml.
|
|
79
|
-
__version__ = '0.8.
|
|
74
|
+
__version__ = '0.8.5'
|
|
@@ -507,9 +507,9 @@ def _clip_painter(
|
|
|
507
507
|
col = 0
|
|
508
508
|
idx = 0
|
|
509
509
|
# captured_style is a frozen snapshot of current_style taken at the first
|
|
510
|
-
# visible character emitted within the clip window (start, end)
|
|
511
|
-
# None until that point. current_style
|
|
512
|
-
#
|
|
510
|
+
# visible character emitted within the clip window (start, end); it stays
|
|
511
|
+
# None until that point. current_style is continuously updated by SGR
|
|
512
|
+
# sequences throughout the scan.
|
|
513
513
|
#
|
|
514
514
|
# When propagate_sgr is False, current_style (and therefore captured_style)
|
|
515
515
|
# remain None, and SGR sequences pass through as literal text.
|
|
@@ -757,8 +757,8 @@ def clip(
|
|
|
757
757
|
r"""
|
|
758
758
|
Clip text to display columns (start, end) while preserving all terminal sequences.
|
|
759
759
|
|
|
760
|
-
This function extracts a substring
|
|
761
|
-
|
|
760
|
+
This function extracts a substring using visible column positions. Terminal escape sequences
|
|
761
|
+
are preserved in output. If a wide character (width of 2) is
|
|
762
762
|
split at a boundary, it is replaced with ``fillchar``.
|
|
763
763
|
|
|
764
764
|
TAB characters (``\t``) are expanded to spaces up to the next tab stop, controlled by the
|
|
@@ -46,8 +46,8 @@ def _scan_zwj_cluster_end(text: str, start: int, end: int) -> int:
|
|
|
46
46
|
idx += 1
|
|
47
47
|
# GB11: \p{ExtPict} Extend* ZWJ × \p{ExtPict}
|
|
48
48
|
# Extend modifiers (VS16, Fitzpatrick skin tones, etc.) attach to
|
|
49
|
-
# the ExtPict *before* the ZWJ
|
|
50
|
-
#
|
|
49
|
+
# the ExtPict *before* the ZWJ. After ZWJ the next codepoint is
|
|
50
|
+
# always an ExtPict directly, no Extend skip needed.
|
|
51
51
|
if idx < end and ord(text[idx]) in _EMOJI_ZWJ_SET:
|
|
52
52
|
idx += 1
|
|
53
53
|
# Skip trailing Extend (VS16, etc.) after ExtPict before next ZWJ
|
|
@@ -292,8 +292,8 @@ def wcstwidth(
|
|
|
292
292
|
ucs = ord(char)
|
|
293
293
|
|
|
294
294
|
#
|
|
295
|
-
# Much of the logic below matches
|
|
296
|
-
#
|
|
295
|
+
# Much of the logic below matches width(); it is repeated here for performance, with
|
|
296
|
+
# matching index reference numbers (starting at #5).
|
|
297
297
|
#
|
|
298
298
|
# 5. ZWJ (U+200D): consumed without contributing width.
|
|
299
299
|
# Virama codepoints are treated as zero-width combining marks (Mn). When a
|
|
@@ -150,9 +150,8 @@ def width(
|
|
|
150
150
|
1
|
|
151
151
|
"""
|
|
152
152
|
# pylint: disable=too-complex,too-many-branches,too-many-statements,too-many-locals,redefined-variable-type,too-many-nested-blocks
|
|
153
|
-
# This could be
|
|
154
|
-
#
|
|
155
|
-
# complexity rules.
|
|
153
|
+
# This could be split into sub-functions (#1, #3 and #6 especially), but this function is a
|
|
154
|
+
# hot path, so the steps stay inline and the pylint complexity rules are disabled.
|
|
156
155
|
|
|
157
156
|
# Fast path for ASCII printable (no tabs, escapes, or control chars)
|
|
158
157
|
if text.isascii() and text.isprintable():
|
|
@@ -272,7 +271,7 @@ def width(
|
|
|
272
271
|
current_col -= n_backward
|
|
273
272
|
if current_col < 0:
|
|
274
273
|
current_col = 0
|
|
275
|
-
# 2d. OSC 66 Text Sizing
|
|
274
|
+
# 2d. OSC 66 Text Sizing: positive display width
|
|
276
275
|
elif (ts_meta := m.group('ts_meta')) is not None:
|
|
277
276
|
ts_text = m.group('ts_text') or ''
|
|
278
277
|
ts_term = m.group('ts_term')
|
|
@@ -281,7 +280,7 @@ def width(
|
|
|
281
280
|
TextSizingParams.from_params(ts_meta, control_codes=control_codes),
|
|
282
281
|
ts_text, ts_term)
|
|
283
282
|
current_col += text_size.display_width(ambiguous_width)
|
|
284
|
-
# 2e. SGR and other zero-width sequences
|
|
283
|
+
# 2e. SGR and other zero-width sequences: no column advance
|
|
285
284
|
idx = m.end()
|
|
286
285
|
# Escape sequences break VS16 adjacency: reset last-measured state
|
|
287
286
|
last_measured_idx = -2
|
|
@@ -22,8 +22,8 @@ def ljust(
|
|
|
22
22
|
:param text: String to justify, may contain terminal sequences.
|
|
23
23
|
:param dest_width: Total display width of result in terminal cells.
|
|
24
24
|
:param fillchar: Single character for padding (default space). Must have
|
|
25
|
-
display width of 1
|
|
26
|
-
|
|
25
|
+
display width of 1. Unicode characters like ``'·'`` are acceptable.
|
|
26
|
+
The width is not validated.
|
|
27
27
|
:param control_codes: How to handle control sequences when measuring.
|
|
28
28
|
Passed to :func:`width` for measurement.
|
|
29
29
|
:param ambiguous_width: Width to use for East Asian Ambiguous (A)
|
|
@@ -72,8 +72,8 @@ def rjust(
|
|
|
72
72
|
:param text: String to justify, may contain terminal sequences.
|
|
73
73
|
:param dest_width: Total display width of result in terminal cells.
|
|
74
74
|
:param fillchar: Single character for padding (default space). Must have
|
|
75
|
-
display width of 1
|
|
76
|
-
|
|
75
|
+
display width of 1. Unicode characters like ``'·'`` are acceptable.
|
|
76
|
+
The width is not validated.
|
|
77
77
|
:param control_codes: How to handle control sequences when measuring.
|
|
78
78
|
Passed to :func:`width` for measurement.
|
|
79
79
|
:param ambiguous_width: Width to use for East Asian Ambiguous (A)
|
|
@@ -122,8 +122,8 @@ def center(
|
|
|
122
122
|
:param text: String to center, may contain terminal sequences.
|
|
123
123
|
:param dest_width: Total display width of result in terminal cells.
|
|
124
124
|
:param fillchar: Single character for padding (default space). Must have
|
|
125
|
-
display width of 1
|
|
126
|
-
|
|
125
|
+
display width of 1. Unicode characters like ``'·'`` are acceptable.
|
|
126
|
+
The width is not validated.
|
|
127
127
|
:param control_codes: How to handle control sequences when measuring.
|
|
128
128
|
Passed to :func:`width` for measurement.
|
|
129
129
|
:param ambiguous_width: Width to use for East Asian Ambiguous (A)
|
|
@@ -24,8 +24,8 @@ TEXT_SIZING_PATTERN = re.compile(
|
|
|
24
24
|
ZERO_WIDTH_PATTERN = re.compile(
|
|
25
25
|
# CSI sequences
|
|
26
26
|
r'\x1b\[[\x30-\x3f]*[\x20-\x2f]*[\x40-\x7e]|'
|
|
27
|
-
# OSC sequences
|
|
28
|
-
#
|
|
27
|
+
# OSC sequences; OSC 66 text sizing gets special handling in width() and clip() and has positive
|
|
28
|
+
# width, despite the ZERO_WIDTH_PATTERN name.
|
|
29
29
|
r'\x1b\][^\x07\x1b]*(?:\x07|\x1b\\)|'
|
|
30
30
|
# APC sequences
|
|
31
31
|
r'\x1b_[^\x1b\x07]*(?:\x07|\x1b\\)|'
|
|
@@ -66,7 +66,7 @@ class GCB(IntEnum):
|
|
|
66
66
|
|
|
67
67
|
# All lru_cache sizes in this file use maxsize=1024, chosen by benchmarking UDHR data (500+
|
|
68
68
|
# languages) and considering typical process-long sessions: western scripts need ~64 unique
|
|
69
|
-
# codepoints,
|
|
69
|
+
# codepoints, and CJK can reach ~2000.
|
|
70
70
|
@lru_cache(maxsize=1024)
|
|
71
71
|
def _grapheme_cluster_break(ucs: int) -> GCB:
|
|
72
72
|
# pylint: disable=too-many-branches,too-complex
|
|
@@ -79,7 +79,7 @@ class Hyperlink(typing.NamedTuple):
|
|
|
79
79
|
close_end)`` or ``(-1, -1)`` if not found.
|
|
80
80
|
|
|
81
81
|
Per the OSC 8 specification, terminal emulators treat hyperlinks as a
|
|
82
|
-
state attribute
|
|
82
|
+
state attribute. A close sequence closes
|
|
83
83
|
the current hyperlink regardless of how many open sequences preceded it.
|
|
84
84
|
"""
|
|
85
85
|
m = HYPERLINK_CLOSE_RE.search(text, open_end)
|
|
@@ -119,8 +119,7 @@ class SequenceTextWrapper(textwrap.TextWrapper):
|
|
|
119
119
|
# Build a mapping from stripped text positions to original text positions.
|
|
120
120
|
#
|
|
121
121
|
# Track where each character ENDS so that sequences between characters
|
|
122
|
-
# attach to the following text
|
|
123
|
-
# aren't lost when whitespace is dropped.
|
|
122
|
+
# attach to the following text, keeping them when whitespace is dropped.
|
|
124
123
|
#
|
|
125
124
|
# char_end[i] = position in original text right after the i-th stripped char
|
|
126
125
|
char_end: list[int] = []
|
|
@@ -197,15 +196,15 @@ class SequenceTextWrapper(textwrap.TextWrapper):
|
|
|
197
196
|
"""
|
|
198
197
|
Wrap chunks into lines using sequence-aware width.
|
|
199
198
|
|
|
200
|
-
Override TextWrapper._wrap_chunks to
|
|
199
|
+
Override TextWrapper._wrap_chunks to measure with _width. Follows stdlib's algorithm:
|
|
201
200
|
greedily fill lines, handle long words. Also handle OSC hyperlink processing. When
|
|
202
201
|
hyperlinks span multiple lines, each line gets complete open/close sequences with matching
|
|
203
202
|
id parameters for hover underlining continuity per OSC 8 spec.
|
|
204
203
|
"""
|
|
205
204
|
# pylint: disable=too-many-branches,too-many-statements,too-complex,too-many-locals
|
|
206
205
|
# pylint: disable=too-many-nested-blocks
|
|
207
|
-
#
|
|
208
|
-
#
|
|
206
|
+
# The hyperlink code pushes the complexity rating of this method. It stays in one method
|
|
207
|
+
# because of the shared local state.
|
|
209
208
|
if self.width <= 0:
|
|
210
209
|
raise ValueError('invalid width %r (must be > 0)' % self.width)
|
|
211
210
|
if not chunks:
|
|
@@ -250,7 +249,7 @@ class SequenceTextWrapper(textwrap.TextWrapper):
|
|
|
250
249
|
|
|
251
250
|
# Drop leading whitespace (except at very start)
|
|
252
251
|
# When dropping, transfer any sequences to the next chunk.
|
|
253
|
-
# Only drop
|
|
252
|
+
# Only drop when actual whitespace text is present.
|
|
254
253
|
stripped = self._strip_sequences(chunks[-1])
|
|
255
254
|
if self.drop_whitespace and lines and stripped and not stripped.strip():
|
|
256
255
|
sequences = self._extract_sequences(chunks[-1])
|
|
@@ -281,7 +280,7 @@ class SequenceTextWrapper(textwrap.TextWrapper):
|
|
|
281
280
|
|
|
282
281
|
# Drop trailing whitespace
|
|
283
282
|
# When dropping, transfer any sequences to the previous chunk.
|
|
284
|
-
# Only drop
|
|
283
|
+
# Only drop when actual whitespace text is present.
|
|
285
284
|
stripped_last = self._strip_sequences(current_line[-1]) if current_line else ''
|
|
286
285
|
if (self.drop_whitespace and current_line and
|
|
287
286
|
stripped_last and not stripped_last.strip()):
|
|
@@ -424,24 +423,23 @@ class SequenceTextWrapper(textwrap.TextWrapper):
|
|
|
424
423
|
if self.break_long_words:
|
|
425
424
|
break_at_hyphen = False
|
|
426
425
|
hyphen_end = 0
|
|
426
|
+
# End of the prefix that fits within space_left by display width.
|
|
427
|
+
prefix_end = self._find_break_position(chunk, space_left)
|
|
427
428
|
|
|
428
|
-
# Handle break_on_hyphens: find last hyphen
|
|
429
|
+
# Handle break_on_hyphens: find last hyphen in the portion that fits.
|
|
429
430
|
if self.break_on_hyphens:
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
if
|
|
433
|
-
#
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
# Map back to original position including sequences
|
|
437
|
-
hyphen_end = self._map_stripped_pos_to_original(chunk, hyphen_pos + 1)
|
|
438
|
-
break_at_hyphen = True
|
|
431
|
+
stripped = self._strip_sequences(chunk[:prefix_end])
|
|
432
|
+
hyphen_pos = stripped.rfind('-')
|
|
433
|
+
if hyphen_pos > 0 and any(c != '-' for c in stripped[:hyphen_pos]):
|
|
434
|
+
# Map back to original position including sequences
|
|
435
|
+
hyphen_end = self._map_stripped_pos_to_original(chunk, hyphen_pos + 1)
|
|
436
|
+
break_at_hyphen = True
|
|
439
437
|
|
|
440
438
|
# Break at grapheme boundaries to avoid splitting multi-codepoint characters
|
|
441
439
|
if break_at_hyphen:
|
|
442
440
|
actual_end = hyphen_end
|
|
443
441
|
else:
|
|
444
|
-
actual_end =
|
|
442
|
+
actual_end = prefix_end
|
|
445
443
|
# Include first visible unit when break would take only leading sequences.
|
|
446
444
|
if not cur_line and (
|
|
447
445
|
actual_end == 0
|
|
@@ -551,8 +549,8 @@ def wrap(text: str, width: int = 70, *,
|
|
|
551
549
|
r"""
|
|
552
550
|
Wrap text to fit within given width, returning a list of wrapped lines.
|
|
553
551
|
|
|
554
|
-
Like :func:`textwrap.wrap`, but measures width in display cells
|
|
555
|
-
|
|
552
|
+
Like :func:`textwrap.wrap`, but measures width in display cells, correctly
|
|
553
|
+
handling wide characters, combining marks, and terminal
|
|
556
554
|
escape sequences.
|
|
557
555
|
|
|
558
556
|
:param text: Text to wrap, may contain terminal sequences.
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|