md2linkedin 0.2.2__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: md2linkedin
3
- Version: 0.2.2
3
+ Version: 0.3.0
4
4
  Summary: Convert Markdown to LinkedIn-friendly Unicode text
5
5
  Keywords: markdown,linkedin,unicode,formatting,conversion
6
6
  Author: Indrajeet Patil
@@ -21,7 +21,7 @@ Classifier: Programming Language :: Python :: 3 :: Only
21
21
  Classifier: Topic :: Software Development :: Libraries :: Python Modules
22
22
  Classifier: Topic :: Text Processing :: Markup
23
23
  Classifier: Topic :: Utilities
24
- Requires-Dist: click>=8.3.3
24
+ Requires-Dist: click>=8.5.0
25
25
  Requires-Python: >=3.10
26
26
  Project-URL: Homepage, https://github.com/IndrajeetPatil/md2linkedin
27
27
  Project-URL: Documentation, https://www.indrapatil.com/md2linkedin/
@@ -53,6 +53,9 @@ natively.
53
53
  | pip | `pip install md2linkedin` |
54
54
  | uv | `uv add md2linkedin` |
55
55
 
56
+ > **Tip:** Run the CLI without installing it with
57
+ > `uvx md2linkedin post.md`.
58
+
56
59
  ## Usage
57
60
 
58
61
  ### Python API
@@ -85,6 +88,7 @@ print(convert(md))
85
88
  I'm thrilled to share that 𝘄𝗲 𝗷𝘂𝘀𝘁 𝗹𝗮𝘂𝗻𝗰𝗵𝗲𝗱 a new product!
86
89
 
87
90
  Key highlights:
91
+
88
92
  • 𝗣𝗲𝗿𝗳𝗼𝗿𝗺𝗮𝗻𝗰𝗲: 3𝘹 𝘧𝘢𝘴𝘵𝘦𝘳 than the previous version
89
93
  • 𝗥𝗲𝗹𝗶𝗮𝗯𝗶𝗹𝗶𝘁𝘆: 𝙯𝙚𝙧𝙤 𝙙𝙤𝙬𝙣𝙩𝙞𝙢𝙚 deployments
90
94
  • 𝗗𝗲𝘃𝗲𝗹𝗼𝗽𝗲𝗿 𝗨𝗫: clean, intuitive API
@@ -158,6 +162,10 @@ notable limitations:
158
162
  For more examples, check out the package documentation at:
159
163
  <https://www.indrapatil.com/md2linkedin/>
160
164
 
165
+ ## See Also
166
+
167
+ - [md2linkedin Web App](https://019d695e-8448-ee38-5db3-7ca4acf2ce2c.share.connect.posit.cloud/) — an interactive web interface for `md2linkedin`, built by [Yann Cohen](https://github.com/iamYannC).
168
+
161
169
  ## License
162
170
 
163
171
  This project is licensed under the MIT License.
@@ -21,6 +21,9 @@ natively.
21
21
  | pip | `pip install md2linkedin` |
22
22
  | uv | `uv add md2linkedin` |
23
23
 
24
+ > **Tip:** Run the CLI without installing it with
25
+ > `uvx md2linkedin post.md`.
26
+
24
27
  ## Usage
25
28
 
26
29
  ### Python API
@@ -53,6 +56,7 @@ print(convert(md))
53
56
  I'm thrilled to share that 𝘄𝗲 𝗷𝘂𝘀𝘁 𝗹𝗮𝘂𝗻𝗰𝗵𝗲𝗱 a new product!
54
57
 
55
58
  Key highlights:
59
+
56
60
  • 𝗣𝗲𝗿𝗳𝗼𝗿𝗺𝗮𝗻𝗰𝗲: 3𝘹 𝘧𝘢𝘴𝘵𝘦𝘳 than the previous version
57
61
  • 𝗥𝗲𝗹𝗶𝗮𝗯𝗶𝗹𝗶𝘁𝘆: 𝙯𝙚𝙧𝙤 𝙙𝙤𝙬𝙣𝙩𝙞𝙢𝙚 deployments
58
62
  • 𝗗𝗲𝘃𝗲𝗹𝗼𝗽𝗲𝗿 𝗨𝗫: clean, intuitive API
@@ -126,6 +130,10 @@ notable limitations:
126
130
  For more examples, check out the package documentation at:
127
131
  <https://www.indrapatil.com/md2linkedin/>
128
132
 
133
+ ## See Also
134
+
135
+ - [md2linkedin Web App](https://019d695e-8448-ee38-5db3-7ca4acf2ce2c.share.connect.posit.cloud/) — an interactive web interface for `md2linkedin`, built by [Yann Cohen](https://github.com/iamYannC).
136
+
129
137
  ## License
130
138
 
131
139
  This project is licensed under the MIT License.
@@ -0,0 +1,157 @@
1
+ [project]
2
+ name = "md2linkedin"
3
+ version = "0.3.0"
4
+ description = "Convert Markdown to LinkedIn-friendly Unicode text"
5
+ readme = "README.md"
6
+ requires-python = ">=3.10"
7
+ license = "MIT"
8
+ license-files = ["LICENSE.md"]
9
+ keywords = [
10
+ "markdown",
11
+ "linkedin",
12
+ "unicode",
13
+ "formatting",
14
+ "conversion",
15
+ ]
16
+ classifiers = [
17
+ "Development Status :: 3 - Alpha",
18
+ "Intended Audience :: Developers",
19
+ "Intended Audience :: End Users/Desktop",
20
+ "Operating System :: OS Independent",
21
+ "Programming Language :: Python",
22
+ "Programming Language :: Python :: 3.10",
23
+ "Programming Language :: Python :: 3.11",
24
+ "Programming Language :: Python :: 3.12",
25
+ "Programming Language :: Python :: 3.13",
26
+ "Programming Language :: Python :: 3.14",
27
+ "Programming Language :: Python :: 3 :: Only",
28
+ "Topic :: Software Development :: Libraries :: Python Modules",
29
+ "Topic :: Text Processing :: Markup",
30
+ "Topic :: Utilities",
31
+ ]
32
+ dependencies = ["click>=8.5.0"]
33
+
34
+ [[project.authors]]
35
+ name = "Indrajeet Patil"
36
+ email = "patilindrajeet.science@gmail.com"
37
+
38
+ [project.scripts]
39
+ md2linkedin = "md2linkedin._cli:main"
40
+
41
+ [project.urls]
42
+ Homepage = "https://github.com/IndrajeetPatil/md2linkedin"
43
+ Documentation = "https://www.indrapatil.com/md2linkedin/"
44
+ Repository = "https://github.com/IndrajeetPatil/md2linkedin"
45
+ Issues = "https://github.com/IndrajeetPatil/md2linkedin/issues"
46
+ Changelog = "https://github.com/IndrajeetPatil/md2linkedin/blob/main/CHANGELOG.md"
47
+
48
+ [dependency-groups]
49
+ dev = [
50
+ "add-trailing-comma>=4.0.0",
51
+ "coverage>=7.16.1",
52
+ "jupyter>=1.1.1",
53
+ "mkdocstrings-python>=2.0.8",
54
+ "mutmut>=3.8.0",
55
+ "prek>=0.5.3",
56
+ "pytest>=9.1.1",
57
+ "pytest-codspeed==5.0.3",
58
+ "pytest-cov>=7.1.0",
59
+ "pytest-random-order>=1.2.0",
60
+ "pyyaml>=6.0.3",
61
+ "ruff>=0.16.8",
62
+ "ty>=0.0.82",
63
+ "zensical>=0.0.62",
64
+ ]
65
+
66
+ [build-system]
67
+ requires = ["uv_build>=0.12.0,<0.13"]
68
+ build-backend = "uv_build"
69
+
70
+ [tool.ruff]
71
+ fix = true
72
+ preview = true
73
+ unsafe-fixes = true
74
+
75
+ [tool.ruff.lint]
76
+ select = ["ALL"]
77
+ ignore = [
78
+ "missing-trailing-comma",
79
+ "incorrect-blank-line-before-class",
80
+ "multi-line-summary-second-line",
81
+ "CPY",
82
+ "ambiguous-unicode-character-docstring",
83
+ "ambiguous-unicode-character-comment",
84
+ ]
85
+
86
+ [tool.ruff.lint.per-file-ignores]
87
+ "tests/*" = [
88
+ "D",
89
+ "assert",
90
+ "import-private-name",
91
+ "no-self-use",
92
+ "ambiguous-unicode-character-string",
93
+ "magic-value-comparison",
94
+ "too-many-public-methods",
95
+ ]
96
+
97
+ [tool.ruff.format]
98
+ preview = true
99
+ docstring-code-format = true
100
+
101
+ [tool.pytest.ini_options]
102
+ addopts = [
103
+ "--strict-config",
104
+ "--strict-markers",
105
+ "-ra",
106
+ "--verbose",
107
+ "--random-order",
108
+ ]
109
+ testpaths = ["tests"]
110
+ filterwarnings = ["error"]
111
+ xfail_strict = true
112
+ python_files = [
113
+ "test_*.py",
114
+ "test-*.py",
115
+ "tests.py",
116
+ "test.py",
117
+ ]
118
+
119
+ [tool.coverage.run]
120
+ branch = true
121
+ source = ["md2linkedin"]
122
+
123
+ [tool.coverage.report]
124
+ fail_under = 100
125
+ format = "markdown"
126
+ sort = "-Cover"
127
+ show_missing = true
128
+ skip_empty = true
129
+
130
+ [tool.mutmut]
131
+ source_paths = ["src/md2linkedin/"]
132
+ also_copy = ["scripts/"]
133
+ pytest_add_cli_args_test_selection = ["tests/"]
134
+ mutate_only_covered_lines = true
135
+ do_not_mutate_patterns = ['closing_fence = rest\.rfind\(fence\)']
136
+
137
+ [tool.ty.src]
138
+ exclude = [
139
+ ".venv",
140
+ "build",
141
+ "dist",
142
+ "docs",
143
+ ]
144
+
145
+ [tool.ty.environment]
146
+ python-version = "3.10"
147
+
148
+ [tool.ty.rules]
149
+ all = "error"
150
+
151
+ [tool.ty.analysis]
152
+ respect-type-ignore-comments = false
153
+ strict-equality-semantics = true
154
+ strict-generic-narrowing = true
155
+
156
+ [tool.uv]
157
+ required-version = ">=0.12.0"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "md2linkedin"
3
- version = "0.2.2"
3
+ version = "0.3.0"
4
4
  description = "Convert Markdown to LinkedIn-friendly Unicode text"
5
5
  readme = "README.md"
6
6
  authors = [
@@ -26,7 +26,9 @@ classifiers = [
26
26
  "Topic :: Text Processing :: Markup",
27
27
  "Topic :: Utilities",
28
28
  ]
29
- dependencies = ["click>=8.3.3"]
29
+ dependencies = [
30
+ "click>=8.5.0",
31
+ ]
30
32
 
31
33
  [project.scripts]
32
34
  md2linkedin = "md2linkedin._cli:main"
@@ -40,21 +42,24 @@ Changelog = "https://github.com/IndrajeetPatil/md2linkedin/blob/main/CHANGELOG.m
40
42
 
41
43
  [dependency-groups]
42
44
  dev = [
43
- "coverage>=7.13.5",
45
+ "add-trailing-comma>=4.0.0",
46
+ "coverage>=7.16.1",
44
47
  "jupyter>=1.1.1",
45
- "zensical>=0.0.36",
46
- "mkdocstrings-python>=2.0.3",
47
- "ty>=0.0.32",
48
- "prek>=0.3.10",
49
- "pytest>=9.0.3",
48
+ "mkdocstrings-python>=2.0.8",
49
+ "mutmut>=3.8.0",
50
+ "prek>=0.5.3",
51
+ "pytest>=9.1.1",
52
+ "pytest-codspeed==5.0.3",
50
53
  "pytest-cov>=7.1.0",
51
54
  "pytest-random-order>=1.2.0",
52
- "ruff>=0.15.11",
53
- "add-trailing-comma>=4.0.0",
55
+ "pyyaml>=6.0.3",
56
+ "ruff>=0.16.8",
57
+ "ty>=0.0.82",
58
+ "zensical>=0.0.62",
54
59
  ]
55
60
 
56
61
  [build-system]
57
- requires = ["uv_build>=0.11.2,<0.12"]
62
+ requires = ["uv_build>=0.12.0,<0.13"]
58
63
  build-backend = "uv_build"
59
64
 
60
65
  [tool.ruff]
@@ -65,39 +70,35 @@ unsafe-fixes = true
65
70
  [tool.ruff.lint]
66
71
  select = ["ALL"]
67
72
  ignore = [
68
- "COM812",
73
+ # Required for compatibility with Ruff's formatter.
74
+ "missing-trailing-comma",
75
+ # Mutually exclusive with D211 and D212, respectively.
76
+ "incorrect-blank-line-before-class",
77
+ "multi-line-summary-second-line",
78
+ # This project uses a repository-level LICENSE instead of per-file notices.
69
79
  "CPY",
70
- # Docstring content rules — we intentionally omit Returns/Raises sections
71
- # in internal helpers, consistent with ignoring D (pydocstyle) above.
72
- "D",
73
- "DOC",
74
80
  # This package's output IS Unicode Mathematical characters; having them in
75
81
  # docstring examples and block-comment separators is intentional.
76
- "RUF002",
77
- "RUF003",
82
+ "ambiguous-unicode-character-docstring",
83
+ "ambiguous-unicode-character-comment",
78
84
  ]
79
85
 
80
86
  [tool.ruff.lint.per-file-ignores]
81
87
  "tests/*" = [
82
88
  # docstring requirements
83
89
  "D",
84
- "DOC",
85
90
  # tests can use asserts
86
- "S101",
87
- # pytest fixtures (e.g. tmp_path) don't need type annotations
88
- "ANN001",
91
+ "assert",
89
92
  # tests legitimately import private internals to test them directly
90
- "PLC2701",
93
+ "import-private-name",
91
94
  # test methods inside classes don't need @staticmethod — class grouping is intentional
92
- "PLR6301",
95
+ "no-self-use",
93
96
  # test assertions for Unicode output use the literal Unicode chars (intentional)
94
- "RUF001",
95
- # `== ""` vs `not x` — both forms are acceptable in assertions
96
- "PLC1901",
97
+ "ambiguous-unicode-character-string",
97
98
  # magic numbers in assertions are fine (e.g. expected count == 3)
98
- "PLR2004",
99
+ "magic-value-comparison",
99
100
  # large test classes are fine when grouping related cases
100
- "PLR0904",
101
+ "too-many-public-methods",
101
102
  ]
102
103
 
103
104
  [tool.ruff.format]
@@ -133,11 +134,48 @@ sort = "-Cover"
133
134
  show_missing = true
134
135
  skip_empty = true
135
136
 
137
+ [tool.mutmut]
138
+ source_paths = ["src/md2linkedin/"]
139
+ # The regression-gate tests import the comparator from this support package.
140
+ also_copy = ["scripts/"]
141
+ pytest_add_cli_args_test_selection = ["tests/"]
142
+ # Only mutate lines exercised by the test suite so mutmut concentrates on
143
+ # meaningful mutations that the tests should actually catch.
144
+ mutate_only_covered_lines = true
145
+ # Skip mutations that cannot be detected by the tests in practice. The
146
+ # pattern is hoisted onto its own source line so the line-granularity
147
+ # exclusion below suppresses only the equivalence mutation described and
148
+ # not any other operations on the same statement.
149
+ # * ``closing_fence = rest.rfind(fence)`` on a fenced code block is
150
+ # equivalent to ``rest.find(fence)`` because the non-greedy regex that
151
+ # captured the block already trims to the shortest matching span, so
152
+ # no additional ``fence`` marker can appear inside ``rest``.
153
+ #
154
+ # ``encoding=_ENCODING`` mutations (``encoding=None`` / arg-removal) are
155
+ # NOT excluded here — they are covered by focused tests that assert the
156
+ # UTF-8 argument is passed explicitly, so surviving mutants would show up
157
+ # as real test failures rather than being silently skipped.
158
+ do_not_mutate_patterns = [
159
+ 'closing_fence = rest\.rfind\(fence\)',
160
+ ]
161
+
136
162
  [tool.ty.src]
137
163
  exclude = [".venv", "build", "dist", "docs"]
138
164
 
139
165
  [tool.ty.environment]
140
166
  python-version = "3.10"
141
167
 
168
+ [tool.ty.rules]
169
+ # Enable every current diagnostic, including soundness and suppression checks that
170
+ # ty leaves disabled by default. Pinning ty in uv.lock keeps new rules from arriving
171
+ # silently between dependency updates.
172
+ all = "error"
173
+
174
+ [tool.ty.analysis]
175
+ # Prefer sound narrowing over convenience assumptions at external-data boundaries.
176
+ respect-type-ignore-comments = false
177
+ strict-equality-semantics = true
178
+ strict-generic-narrowing = true
179
+
142
180
  [tool.uv]
143
- required-version = ">=0.11.2"
181
+ required-version = ">=0.12.0"
@@ -1,4 +1,4 @@
1
- """Command-line interface for md2linkedin."""
1
+ """Command-line interface for converting Markdown with md2linkedin."""
2
2
 
3
3
  from __future__ import annotations
4
4
 
@@ -54,7 +54,7 @@ def main(
54
54
  preserve_links: bool,
55
55
  no_monospace_code: bool,
56
56
  ) -> None:
57
- """Convert Markdown to LinkedIn-friendly Unicode text.
57
+ r"""Convert Markdown to LinkedIn-friendly Unicode text.
58
58
 
59
59
  Reads from INPUT_FILE (or stdin when INPUT_FILE is omitted) and writes
60
60
  LinkedIn-compatible plain text in which bold and italic formatting is
@@ -80,6 +80,10 @@ def main(
80
80
 
81
81
  # Disable monospace code rendering
82
82
  md2linkedin README.md --no-monospace-code
83
+
84
+ Raises:
85
+ click.UsageError: If neither an input file nor piped stdin is provided.
86
+
83
87
  """
84
88
  monospace_code = not no_monospace_code
85
89
  if input_file is not None:
@@ -112,5 +116,9 @@ def _stdin_is_tty() -> bool:
112
116
 
113
117
  Extracted into its own function so tests can mock it cleanly without
114
118
  fighting Click's own stdin-swapping inside ``CliRunner.invoke``.
119
+
120
+ Returns:
121
+ Whether stdin is connected to an interactive terminal.
122
+
115
123
  """
116
- return sys.stdin.isatty()
124
+ return bool(sys.stdin.isatty())
@@ -8,28 +8,64 @@ regex conflicts (e.g. bold-italic must be processed before bold or italic).
8
8
  from __future__ import annotations
9
9
 
10
10
  import re
11
- import uuid
12
11
  from pathlib import Path
12
+ from typing import TYPE_CHECKING
13
13
 
14
14
  from ._unicode import to_monospace, to_sans_bold, to_sans_bold_italic, to_sans_italic
15
15
 
16
+ if TYPE_CHECKING:
17
+ from collections.abc import Callable
18
+
16
19
  __all__ = ["convert", "convert_file"]
17
20
 
18
21
  _NESTED_BULLET_MIN_INDENT = 2 # spaces of indentation that triggers a nested bullet (‣)
22
+ _ENCODING = "utf-8"
23
+
24
+ _HTML_ENTITIES = {
25
+ "&gt;": ">",
26
+ "&lt;": "<",
27
+ "&amp;": "&",
28
+ "&nbsp;": " ",
29
+ "&quot;": '"',
30
+ "&apos;": "'",
31
+ }
19
32
 
20
33
  # ── Low-level pipeline steps ───────────────────────────────────────────────────
21
34
 
22
35
 
23
36
  def _normalize_line_endings(text: str) -> str:
24
- """Normalize Windows (\\r\\n) and classic Mac (\\r) line endings to \\n."""
37
+ r"""Normalize Windows (\\r\\n) and classic Mac (\\r) line endings to \\n.
38
+
39
+ Returns:
40
+ Text containing only Unix-style line endings.
41
+
42
+ """
25
43
  return text.replace("\r\n", "\n").replace("\r", "\n")
26
44
 
27
45
 
46
+ # Placeholder bodies are drawn from the Private Use Area (U+E000+). A
47
+ # placeholder has to survive every step between _protect_code and
48
+ # _restore_code untouched, and those steps rewrite ASCII alphanumerics
49
+ # (to_sans_*) or apply .upper(). PUA codepoints are immune to both, whereas an
50
+ # ASCII body gets mangled and the placeholder then leaks into the output (#63).
51
+ _PLACEHOLDER_BASE = 0xE000
52
+ # Matches any key built above. Restoring via one regex pass beats a str.replace
53
+ # per key: the keys are only three characters, and CPython's substring search
54
+ # skips much less on a short needle, so repeated scans of a long document
55
+ # dominated the conversion.
56
+ _PLACEHOLDER_RE = re.compile(r"\x00.\x00", re.DOTALL)
57
+
58
+
28
59
  def _protect_code(text: str) -> tuple[str, dict[str, str]]:
29
- """Replace code spans and fenced blocks with unique placeholders.
60
+ r"""Replace code spans and fenced blocks with unique placeholders.
30
61
 
31
62
  Code content must never be transformed by the Unicode mapping steps.
32
- Placeholders are UUID-based so they cannot accidentally match user text.
63
+
64
+ Each placeholder is a Private Use Area character delimited by ``\x00``,
65
+ chosen so that no later pipeline step can alter it; see
66
+ :data:`_PLACEHOLDER_BASE`. Any ``\x00`` already in *text* is dropped,
67
+ which is what keeps the sequentially numbered keys from colliding with
68
+ content that happens to contain Private Use Area characters.
33
69
 
34
70
  Args:
35
71
  text: Markdown text.
@@ -37,11 +73,13 @@ def _protect_code(text: str) -> tuple[str, dict[str, str]]:
37
73
  Returns:
38
74
  A ``(modified_text, placeholder_map)`` tuple where *placeholder_map*
39
75
  maps each placeholder back to its original code string.
76
+
40
77
  """
78
+ text = text.replace("\x00", "")
41
79
  placeholders: dict[str, str] = {}
42
80
 
43
81
  def _replace(match: re.Match[str]) -> str:
44
- key = f"\x00CODE{uuid.uuid4().hex}\x00"
82
+ key = f"\x00{chr(_PLACEHOLDER_BASE + len(placeholders))}\x00"
45
83
  placeholders[key] = match.group(0)
46
84
  return key
47
85
 
@@ -69,6 +107,12 @@ def _restore_code(
69
107
  mapped; all other characters (including Markdown syntax) pass through
70
108
  unchanged, so no nested processing is needed.
71
109
 
110
+ All placeholders are expanded in a single pass, repeated until the text
111
+ stops changing. :func:`_protect_code` matches fenced blocks before inline
112
+ spans, so an inline span that wraps a fenced run swallows the fenced
113
+ placeholder into its own stored value; expanding the outer span
114
+ reintroduces the inner key, which the next pass resolves.
115
+
72
116
  Args:
73
117
  text: Text containing placeholders.
74
118
  placeholders: Map of placeholder → original code string.
@@ -76,28 +120,37 @@ def _restore_code(
76
120
 
77
121
  Returns:
78
122
  Text with all placeholders replaced by their original code content.
123
+
79
124
  """
80
- for key, original in placeholders.items():
125
+
126
+ def _expand(match: re.Match[str]) -> str:
127
+ original = placeholders[match.group(0)]
81
128
  if original.startswith(("```", "~~~")):
82
129
  if monospace:
83
130
  # Strip fences and optional language tag, convert content
84
131
  fence = original[:3]
85
132
  rest = original[3:]
86
133
  # Remove closing fence
87
- body = rest[: rest.rfind(fence)]
88
- # Strip optional language tag (first line of body)
89
- first_nl = body.find("\n")
90
- content = body[first_nl + 1 :] if first_nl != -1 else ""
91
- text = text.replace(key, to_monospace(content))
92
- else:
93
- # Keep fenced blocks as-is (no backtick stripping)
94
- text = text.replace(key, original)
95
- elif monospace:
134
+ closing_fence = rest.rfind(fence)
135
+ body = rest[:closing_fence]
136
+ # Strip the optional language tag, which is only a language tag
137
+ # when a newline terminates it. A single-line run (```hi```) has
138
+ # no tag, so find returns -1 and the slice keeps the whole body.
139
+ content = body[body.find("\n") + 1 :]
140
+ return to_monospace(content)
141
+ # Keep fenced blocks as-is (no backtick stripping)
142
+ return original
143
+ if monospace:
96
144
  # Strip backticks and apply monospace for inline code
97
- text = text.replace(key, to_monospace(original[1:-1]))
98
- else:
99
- # Strip the surrounding backticks for inline code
100
- text = text.replace(key, original[1:-1])
145
+ return to_monospace(original[1:-1])
146
+ # Strip the surrounding backticks for inline code
147
+ return original[1:-1]
148
+
149
+ while placeholders:
150
+ new_text = _PLACEHOLDER_RE.sub(_expand, text)
151
+ if new_text == text:
152
+ break
153
+ text = new_text
101
154
  return text
102
155
 
103
156
 
@@ -112,16 +165,56 @@ def _strip_html_spans(text: str) -> str:
112
165
 
113
166
  Returns:
114
167
  Text with all span elements removed and their inner content preserved.
168
+
115
169
  """
116
- prev = None
117
- while prev != text:
118
- prev = text
119
- text = re.sub(r"<span[^>]*>(.*?)</span>", r"\1", text, flags=re.DOTALL)
170
+ while True:
171
+ new_text = re.sub(r"<span[^>]*>(.*?)</span>", r"\1", text, flags=re.DOTALL)
172
+ if new_text == text:
173
+ return text
174
+ text = new_text
175
+
176
+
177
+ # Emphasis markers. Each style has an asterisk variant and an underscore
178
+ # variant, applied in that order; ``(?<!\\)`` skips backslash-escaped markers.
179
+ _BOLD_ITALIC_PATTERNS = (
180
+ r"(?<!\\)\*{3}(.+?)(?<!\\)\*{3}",
181
+ r"(?<!\\)_{3}(.+?)(?<!\\)_{3}",
182
+ )
183
+ _BOLD_PATTERNS = (
184
+ r"(?<!\\)\*{2}(.+?)(?<!\\)\*{2}",
185
+ r"(?<!\\)__(.+?)(?<!\\)__",
186
+ )
187
+ # Italic also needs negative look-around, so that residual ** markers are never
188
+ # matched, and word-boundary anchors, so that inside_words is left alone.
189
+ _ITALIC_PATTERNS = (
190
+ r"(?<!\\)(?<!\*)\*(?!\*)(.+?)(?<!\\)(?<!\*)\*(?!\*)",
191
+ r"(?<!\w)(?<!\\)_(?!_)(.+?)(?<!\\)(?<!_)_(?!\w)",
192
+ )
193
+
194
+
195
+ def _style_markers(
196
+ text: str,
197
+ patterns: tuple[str, ...],
198
+ style: Callable[[str], str],
199
+ ) -> str:
200
+ """Apply *style* to the text captured by each emphasis pattern in turn.
201
+
202
+ Args:
203
+ text: Input text.
204
+ patterns: Regexes whose first capture group is the text to style.
205
+ style: Unicode mapping function applied to each captured group.
206
+
207
+ Returns:
208
+ Text with every matched marker pair replaced by styled Unicode.
209
+
210
+ """
211
+ for pattern in patterns:
212
+ text = re.sub(pattern, lambda m: style(m.group(1)), text)
120
213
  return text
121
214
 
122
215
 
123
216
  def _convert_bold_italic(text: str) -> str:
124
- """Replace ``***text***`` (or ``___text___``) with bold-italic Unicode.
217
+ r"""Replace ``***text***`` (or ``___text___``) with bold-italic Unicode.
125
218
 
126
219
  Must run before :func:`_convert_bold` and :func:`_convert_italic` to
127
220
  prevent the triple markers from being consumed piecemeal.
@@ -133,21 +226,13 @@ def _convert_bold_italic(text: str) -> str:
133
226
 
134
227
  Returns:
135
228
  Text with bold-italic markers replaced.
229
+
136
230
  """
137
- text = re.sub(
138
- r"(?<!\\)\*{3}(.+?)(?<!\\)\*{3}",
139
- lambda m: to_sans_bold_italic(m.group(1)),
140
- text,
141
- )
142
- return re.sub(
143
- r"(?<!\\)_{3}(.+?)(?<!\\)_{3}",
144
- lambda m: to_sans_bold_italic(m.group(1)),
145
- text,
146
- )
231
+ return _style_markers(text, _BOLD_ITALIC_PATTERNS, to_sans_bold_italic)
147
232
 
148
233
 
149
234
  def _convert_bold(text: str) -> str:
150
- """Replace ``**text**`` (or ``__text__``) with bold Unicode.
235
+ r"""Replace ``**text**`` (or ``__text__``) with bold Unicode.
151
236
 
152
237
  Backslash-escaped markers (``\\**``) are not matched.
153
238
 
@@ -156,21 +241,13 @@ def _convert_bold(text: str) -> str:
156
241
 
157
242
  Returns:
158
243
  Text with bold markers replaced.
244
+
159
245
  """
160
- text = re.sub(
161
- r"(?<!\\)\*{2}(.+?)(?<!\\)\*{2}",
162
- lambda m: to_sans_bold(m.group(1)),
163
- text,
164
- )
165
- return re.sub(
166
- r"(?<!\\)__(.+?)(?<!\\)__",
167
- lambda m: to_sans_bold(m.group(1)),
168
- text,
169
- )
246
+ return _style_markers(text, _BOLD_PATTERNS, to_sans_bold)
170
247
 
171
248
 
172
249
  def _convert_italic(text: str) -> str:
173
- """Replace ``*text*`` or ``_text_`` with italic Unicode.
250
+ r"""Replace ``*text*`` or ``_text_`` with italic Unicode.
174
251
 
175
252
  Uses negative look-around to avoid matching asterisks that are part of
176
253
  bold (``**``) or bold-italic (``***``) markers already consumed by
@@ -182,19 +259,9 @@ def _convert_italic(text: str) -> str:
182
259
 
183
260
  Returns:
184
261
  Text with italic markers replaced.
262
+
185
263
  """
186
- # *text* — negative look-around prevents matching residual ** markers or \* escapes
187
- text = re.sub(
188
- r"(?<!\\)(?<!\*)\*(?!\*)(.+?)(?<!\\)(?<!\*)\*(?!\*)",
189
- lambda m: to_sans_italic(m.group(1)),
190
- text,
191
- )
192
- # _text_ — word-boundary anchors prevent matching inside_words; skip \_ escapes
193
- return re.sub(
194
- r"(?<!\w)(?<!\\)_(?!_)(.+?)(?<!\\)(?<!_)_(?!\w)",
195
- lambda m: to_sans_italic(m.group(1)),
196
- text,
197
- )
264
+ return _style_markers(text, _ITALIC_PATTERNS, to_sans_italic)
198
265
 
199
266
 
200
267
  def _convert_headers(text: str) -> str:
@@ -208,6 +275,7 @@ def _convert_headers(text: str) -> str:
208
275
 
209
276
  Returns:
210
277
  Text with headers replaced by styled plain text.
278
+
211
279
  """
212
280
  separator = "━" * 40
213
281
 
@@ -227,7 +295,7 @@ def _convert_headers(text: str) -> str:
227
295
  atx = re.match(r"^(#{1,6})\s+(.*)", line)
228
296
  if atx:
229
297
  level = len(atx.group(1))
230
- title = atx.group(2).rstrip()
298
+ title = atx.group(2)
231
299
  if level == 1:
232
300
  out.append(_fmt_h1(title))
233
301
  else:
@@ -269,6 +337,7 @@ def _strip_links(text: str, *, preserve: bool = False) -> str:
269
337
 
270
338
  Returns:
271
339
  Text with links handled according to *preserve*.
340
+
272
341
  """
273
342
  if preserve:
274
343
  return text
@@ -290,6 +359,7 @@ def _strip_images(text: str) -> str:
290
359
 
291
360
  Returns:
292
361
  Text with image syntax replaced by alt text.
362
+
293
363
  """
294
364
  # ![alt](url) → alt (empty alt → removed)
295
365
  return re.sub(
@@ -311,6 +381,7 @@ def _convert_bullets(text: str) -> str:
311
381
 
312
382
  Returns:
313
383
  Text with list markers replaced.
384
+
314
385
  """
315
386
 
316
387
  def _bullet(m: re.Match[str]) -> str:
@@ -328,6 +399,7 @@ def _strip_blockquotes(text: str) -> str:
328
399
 
329
400
  Returns:
330
401
  Text with blockquote markers stripped from line beginnings.
402
+
331
403
  """
332
404
  return re.sub(r"^> ?", "", text, flags=re.MULTILINE)
333
405
 
@@ -341,28 +413,22 @@ def _clean_entities(text: str) -> str:
341
413
  Returns:
342
414
  Text with ``&gt;``, ``&lt;``, ``&amp;``, ``&nbsp;``, ``&quot;``
343
415
  replaced by their literal equivalents.
416
+
344
417
  """
345
- replacements = {
346
- "&gt;": ">",
347
- "&lt;": "<",
348
- "&amp;": "&",
349
- "&nbsp;": " ",
350
- "&quot;": '"',
351
- "&apos;": "'",
352
- }
353
- for entity, char in replacements.items():
418
+ for entity, char in _HTML_ENTITIES.items():
354
419
  text = text.replace(entity, char)
355
420
  return text
356
421
 
357
422
 
358
423
  def _clean_escaped_chars(text: str) -> str:
359
- """Remove Markdown backslash escapes (e.g. ``\\*`` → ``*``).
424
+ r"""Remove Markdown backslash escapes (e.g. ``\\*`` → ``*``).
360
425
 
361
426
  Args:
362
427
  text: Input text.
363
428
 
364
429
  Returns:
365
430
  Text with backslash escapes resolved.
431
+
366
432
  """
367
433
  return re.sub(r"\\([\\`*_{}\[\]()#+\-.!])", r"\1", text)
368
434
 
@@ -378,6 +444,7 @@ def _normalize_whitespace(text: str) -> str:
378
444
 
379
445
  Returns:
380
446
  Normalized text with a single trailing newline.
447
+
381
448
  """
382
449
  text = re.sub(r"\n{3,}", "\n\n", text)
383
450
  return text.strip() + "\n"
@@ -392,7 +459,7 @@ def convert(
392
459
  preserve_links: bool = False,
393
460
  monospace_code: bool = True,
394
461
  ) -> str:
395
- """Convert Markdown text to LinkedIn-compatible Unicode plain text.
462
+ r"""Convert Markdown text to LinkedIn-compatible Unicode plain text.
396
463
 
397
464
  Bold (``**text**`` / ``__text__``), italic (``*text*`` / ``_text_``), and
398
465
  bold-italic (``***text***`` / ``___text___``) markers are replaced with
@@ -433,6 +500,7 @@ def convert(
433
500
 
434
501
  >>> convert("")
435
502
  ''
503
+
436
504
  """
437
505
  if not text or not text.strip():
438
506
  return ""
@@ -461,7 +529,7 @@ def convert_file(
461
529
  preserve_links: bool = False,
462
530
  monospace_code: bool = True,
463
531
  ) -> Path:
464
- """Convert a Markdown file and write the result to a ``.txt`` file.
532
+ r"""Convert a Markdown file and write the result to a ``.txt`` file.
465
533
 
466
534
  Args:
467
535
  input_path: Path to the Markdown source file (``.md`` or any text
@@ -489,6 +557,7 @@ def convert_file(
489
557
  '𝗯𝗼𝗹𝗱 and 𝘪𝘵𝘢𝘭𝘪𝘤\\n'
490
558
  >>> os.unlink(tmp)
491
559
  ... os.unlink(str(out))
560
+
492
561
  """
493
562
  input_path = Path(input_path)
494
563
  if not input_path.exists():
@@ -499,11 +568,11 @@ def convert_file(
499
568
  output_path = input_path.with_suffix("").with_suffix(".linkedin.txt")
500
569
  output_path = Path(output_path)
501
570
 
502
- md_text = input_path.read_text(encoding="utf-8")
571
+ md_text = input_path.read_text(encoding=_ENCODING)
503
572
  result = convert(
504
573
  md_text,
505
574
  preserve_links=preserve_links,
506
575
  monospace_code=monospace_code,
507
576
  )
508
- output_path.write_text(result, encoding="utf-8")
577
+ output_path.write_text(result, encoding=_ENCODING)
509
578
  return output_path
@@ -7,6 +7,7 @@ styling in plain-text environments like LinkedIn.
7
7
 
8
8
  from __future__ import annotations
9
9
 
10
+ import string
10
11
  from typing import Literal
11
12
 
12
13
  __all__ = [
@@ -35,6 +36,42 @@ _MONOSPACE_LOWER = 0x1D68A # 𝚊
35
36
  _MONOSPACE_DIGIT = 0x1D7F6 # 𝟶
36
37
 
37
38
 
39
+ # ── Translation tables ─────────────────────────────────────────────────────────
40
+
41
+
42
+ def _build_table(upper: int, lower: int, digit: int | None = None) -> dict[int, int]:
43
+ """Build a :meth:`str.translate` table for one Unicode style block.
44
+
45
+ Args:
46
+ upper: Codepoint of the styled ``A``.
47
+ lower: Codepoint of the styled ``a``.
48
+ digit: Codepoint of the styled ``0``, or ``None`` for styles whose
49
+ Unicode block has no digits (the italic blocks).
50
+
51
+ Returns:
52
+ A mapping from ASCII codepoint to styled codepoint. Characters absent
53
+ from the mapping are left unchanged by :meth:`str.translate`.
54
+
55
+ """
56
+ blocks = [(string.ascii_uppercase, upper), (string.ascii_lowercase, lower)]
57
+ if digit is not None:
58
+ blocks.append((string.digits, digit))
59
+ return {
60
+ ord(char): base + offset
61
+ for chars, base in blocks
62
+ for offset, char in enumerate(chars)
63
+ }
64
+
65
+
66
+ _SANS_BOLD_TABLE = _build_table(_SANS_BOLD_UPPER, _SANS_BOLD_LOWER, _SANS_BOLD_DIGIT)
67
+ _SANS_ITALIC_TABLE = _build_table(_SANS_ITALIC_UPPER, _SANS_ITALIC_LOWER)
68
+ _SANS_BOLD_ITALIC_TABLE = _build_table(
69
+ _SANS_BOLD_ITALIC_UPPER,
70
+ _SANS_BOLD_ITALIC_LOWER,
71
+ )
72
+ _MONOSPACE_TABLE = _build_table(_MONOSPACE_UPPER, _MONOSPACE_LOWER, _MONOSPACE_DIGIT)
73
+
74
+
38
75
  # ── Public mapping functions ───────────────────────────────────────────────────
39
76
 
40
77
 
@@ -61,18 +98,9 @@ def to_sans_bold(text: str) -> str:
61
98
 
62
99
  >>> to_sans_bold("")
63
100
  ''
101
+
64
102
  """
65
- out: list[str] = []
66
- for c in text:
67
- if "A" <= c <= "Z":
68
- out.append(chr(_SANS_BOLD_UPPER + ord(c) - ord("A")))
69
- elif "a" <= c <= "z":
70
- out.append(chr(_SANS_BOLD_LOWER + ord(c) - ord("a")))
71
- elif "0" <= c <= "9":
72
- out.append(chr(_SANS_BOLD_DIGIT + ord(c) - ord("0")))
73
- else:
74
- out.append(c)
75
- return "".join(out)
103
+ return text.translate(_SANS_BOLD_TABLE)
76
104
 
77
105
 
78
106
  def to_sans_italic(text: str) -> str:
@@ -98,16 +126,9 @@ def to_sans_italic(text: str) -> str:
98
126
 
99
127
  >>> to_sans_italic("")
100
128
  ''
129
+
101
130
  """
102
- out: list[str] = []
103
- for c in text:
104
- if "A" <= c <= "Z":
105
- out.append(chr(_SANS_ITALIC_UPPER + ord(c) - ord("A")))
106
- elif "a" <= c <= "z":
107
- out.append(chr(_SANS_ITALIC_LOWER + ord(c) - ord("a")))
108
- else:
109
- out.append(c)
110
- return "".join(out)
131
+ return text.translate(_SANS_ITALIC_TABLE)
111
132
 
112
133
 
113
134
  def to_sans_bold_italic(text: str) -> str:
@@ -130,16 +151,9 @@ def to_sans_bold_italic(text: str) -> str:
130
151
 
131
152
  >>> to_sans_bold_italic("")
132
153
  ''
154
+
133
155
  """
134
- out: list[str] = []
135
- for c in text:
136
- if "A" <= c <= "Z":
137
- out.append(chr(_SANS_BOLD_ITALIC_UPPER + ord(c) - ord("A")))
138
- elif "a" <= c <= "z":
139
- out.append(chr(_SANS_BOLD_ITALIC_LOWER + ord(c) - ord("a")))
140
- else:
141
- out.append(c)
142
- return "".join(out)
156
+ return text.translate(_SANS_BOLD_ITALIC_TABLE)
143
157
 
144
158
 
145
159
  def to_monospace(text: str) -> str:
@@ -165,18 +179,9 @@ def to_monospace(text: str) -> str:
165
179
 
166
180
  >>> to_monospace("")
167
181
  ''
182
+
168
183
  """
169
- out: list[str] = []
170
- for c in text:
171
- if "A" <= c <= "Z":
172
- out.append(chr(_MONOSPACE_UPPER + ord(c) - ord("A")))
173
- elif "a" <= c <= "z":
174
- out.append(chr(_MONOSPACE_LOWER + ord(c) - ord("a")))
175
- elif "0" <= c <= "9":
176
- out.append(chr(_MONOSPACE_DIGIT + ord(c) - ord("0")))
177
- else:
178
- out.append(c)
179
- return "".join(out)
184
+ return text.translate(_MONOSPACE_TABLE)
180
185
 
181
186
 
182
187
  def apply_style(text: str, style: Literal["bold", "italic", "bold_italic"]) -> str:
@@ -205,6 +210,7 @@ def apply_style(text: str, style: Literal["bold", "italic", "bold_italic"]) -> s
205
210
 
206
211
  >>> apply_style("hello", "bold_italic")
207
212
  '𝙝𝙚𝙡𝙡𝙤'
213
+
208
214
  """
209
215
  if style == "bold":
210
216
  return to_sans_bold(text)
File without changes