flowmark 0.3.2__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {flowmark-0.3.2 → flowmark-0.4.0}/PKG-INFO +14 -9
- {flowmark-0.3.2 → flowmark-0.4.0}/README.md +13 -8
- {flowmark-0.3.2 → flowmark-0.4.0}/src/flowmark/__init__.py +4 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/src/flowmark/markdown_filling.py +55 -6
- {flowmark-0.3.2 → flowmark-0.4.0}/src/flowmark/sentence_split_regex.py +43 -18
- flowmark-0.4.0/tests/test_ref_docs.py +44 -0
- flowmark-0.4.0/tests/test_sentences.py +30 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/tests/testdocs/testdoc.orig.md +33 -4
- {flowmark-0.3.2 → flowmark-0.4.0}/tests/testdocs/testdoc.out.plain.md +36 -10
- {flowmark-0.3.2 → flowmark-0.4.0}/tests/testdocs/testdoc.out.semantic.md +36 -10
- flowmark-0.3.2/tests/test_ref_docs.py +0 -47
- {flowmark-0.3.2 → flowmark-0.4.0}/.copier-answers.yml +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/.github/workflows/ci.yml +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/.github/workflows/publish.yml +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/.gitignore +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/LICENSE +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/Makefile +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/development.md +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/devtools/lint.py +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/poetry.lock +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/publishing.md +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/pyproject.toml +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/src/flowmark/cli.py +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/src/flowmark/frontmatter.py +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/src/flowmark/line_wrappers.py +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/src/flowmark/text_filling.py +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/src/flowmark/text_wrapping.py +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/tests/test_filling.py +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/tests/test_frontmatter.py +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/tests/test_wrapping.py +0 -0
- {flowmark-0.3.2 → flowmark-0.4.0}/uv.lock +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: flowmark
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.0
|
|
4
4
|
Summary: Better line wrapping and formatting for plaintext and Markdown
|
|
5
5
|
Project-URL: Repository, https://github.com/jlevy/flowmark
|
|
6
6
|
Author-email: Joshua Levy <joshua@cal.berkeley.edu>
|
|
@@ -18,10 +18,9 @@ Description-Content-Type: text/markdown
|
|
|
18
18
|
Flowmark is a new Python implementation of **text and Markdown line wrapping and
|
|
19
19
|
filling**, with an emphasis on making **git diffs** and **LLM edits** to text documents
|
|
20
20
|
easier to diff and review.
|
|
21
|
-
after updating
|
|
22
21
|
|
|
23
|
-
In addition, it
|
|
24
|
-
|
|
22
|
+
In addition, it offers **Markdown auto-formatting and normalization** as a library or
|
|
23
|
+
from the command line.
|
|
25
24
|
This is much like [markdownfmt](https://github.com/shurcooL/markdownfmt) or
|
|
26
25
|
[prettier's Markdown support](https://prettier.io/blog/2017/11/07/1.8.0) but is pure
|
|
27
26
|
Python and has (in my humble opinion) better options and defaults.
|
|
@@ -73,8 +72,7 @@ The `--auto` option is just the same as `--inplace --nobackup --semantic`.
|
|
|
73
72
|
|
|
74
73
|
## Use Cases
|
|
75
74
|
|
|
76
|
-
|
|
77
|
-
command.
|
|
75
|
+
The main ways to use Flowmark are:
|
|
78
76
|
|
|
79
77
|
- To **autoformat Markdown on save in VSCode/Cursor** or any other editor that supports
|
|
80
78
|
running a command on save.
|
|
@@ -83,6 +81,9 @@ The `--auto` option is just the same as `--inplace --nobackup --semantic`.
|
|
|
83
81
|
formatting styles). This can be especially useful for documentation and editing
|
|
84
82
|
workflows where clean diffs and minimal merge conflicts on GitHub are important.
|
|
85
83
|
|
|
84
|
+
- As a **command line formatter** to format text or Markdown files using the `flowmark`
|
|
85
|
+
command.
|
|
86
|
+
|
|
86
87
|
- As a **library to autoformat Markdown**. For example, it is great to normalize the
|
|
87
88
|
outputs from LLMs to be consistent, or to run on the inputs and outputs of LLM
|
|
88
89
|
transformations that edit text, so that the resulting diffs are clean.
|
|
@@ -95,6 +96,8 @@ The `--auto` option is just the same as `--inplace --nobackup --semantic`.
|
|
|
95
96
|
subsequent indentation** and **when to split words and lines**, e.g. using a word
|
|
96
97
|
splitter that won't break lines within HTML tags.
|
|
97
98
|
|
|
99
|
+
Other features:
|
|
100
|
+
|
|
98
101
|
- Flowmark has the option to to use **semantic line breaks** (using a heuristic to break
|
|
99
102
|
lines on sentences sentences when that is reasonable), which is an underrated feature
|
|
100
103
|
that can **make diffs on GitHub much more readable**. The the change may seem subtle
|
|
@@ -104,8 +107,11 @@ The `--auto` option is just the same as `--inplace --nobackup --semantic`.
|
|
|
104
107
|
[Markdown source](https://github.com/jlevy/flowmark/blob/main/README.md?plain=1) of
|
|
105
108
|
this readme file.)
|
|
106
109
|
|
|
107
|
-
- Very
|
|
108
|
-
|
|
110
|
+
- Very simple and fast **regex-based sentence splitting**. It's just based on letters
|
|
111
|
+
and punctuation so isn't perfect but works well for these purposes (and is much faster
|
|
112
|
+
and simpler than a proper sentence parser like SpaCy).
|
|
113
|
+
It should work fine for English and many other latin/Cyrillic languages but hasn't
|
|
114
|
+
been tested on CJK.
|
|
109
115
|
|
|
110
116
|
It aims to be small and simple and have only a few dependencies, currently only
|
|
111
117
|
[`marko`](https://github.com/frostming/marko),
|
|
@@ -163,7 +169,6 @@ Markdown content. It can:
|
|
|
163
169
|
|
|
164
170
|
- Optionally break lines at sentence boundaries for better diff readability
|
|
165
171
|
|
|
166
|
-
|
|
167
172
|
- Process plaintext with HTML-aware word splitting
|
|
168
173
|
|
|
169
174
|
It is both a library and a command-line tool.
|
|
@@ -3,10 +3,9 @@
|
|
|
3
3
|
Flowmark is a new Python implementation of **text and Markdown line wrapping and
|
|
4
4
|
filling**, with an emphasis on making **git diffs** and **LLM edits** to text documents
|
|
5
5
|
easier to diff and review.
|
|
6
|
-
after updating
|
|
7
6
|
|
|
8
|
-
In addition, it
|
|
9
|
-
|
|
7
|
+
In addition, it offers **Markdown auto-formatting and normalization** as a library or
|
|
8
|
+
from the command line.
|
|
10
9
|
This is much like [markdownfmt](https://github.com/shurcooL/markdownfmt) or
|
|
11
10
|
[prettier's Markdown support](https://prettier.io/blog/2017/11/07/1.8.0) but is pure
|
|
12
11
|
Python and has (in my humble opinion) better options and defaults.
|
|
@@ -58,8 +57,7 @@ The `--auto` option is just the same as `--inplace --nobackup --semantic`.
|
|
|
58
57
|
|
|
59
58
|
## Use Cases
|
|
60
59
|
|
|
61
|
-
|
|
62
|
-
command.
|
|
60
|
+
The main ways to use Flowmark are:
|
|
63
61
|
|
|
64
62
|
- To **autoformat Markdown on save in VSCode/Cursor** or any other editor that supports
|
|
65
63
|
running a command on save.
|
|
@@ -68,6 +66,9 @@ The `--auto` option is just the same as `--inplace --nobackup --semantic`.
|
|
|
68
66
|
formatting styles). This can be especially useful for documentation and editing
|
|
69
67
|
workflows where clean diffs and minimal merge conflicts on GitHub are important.
|
|
70
68
|
|
|
69
|
+
- As a **command line formatter** to format text or Markdown files using the `flowmark`
|
|
70
|
+
command.
|
|
71
|
+
|
|
71
72
|
- As a **library to autoformat Markdown**. For example, it is great to normalize the
|
|
72
73
|
outputs from LLMs to be consistent, or to run on the inputs and outputs of LLM
|
|
73
74
|
transformations that edit text, so that the resulting diffs are clean.
|
|
@@ -80,6 +81,8 @@ The `--auto` option is just the same as `--inplace --nobackup --semantic`.
|
|
|
80
81
|
subsequent indentation** and **when to split words and lines**, e.g. using a word
|
|
81
82
|
splitter that won't break lines within HTML tags.
|
|
82
83
|
|
|
84
|
+
Other features:
|
|
85
|
+
|
|
83
86
|
- Flowmark has the option to to use **semantic line breaks** (using a heuristic to break
|
|
84
87
|
lines on sentences sentences when that is reasonable), which is an underrated feature
|
|
85
88
|
that can **make diffs on GitHub much more readable**. The the change may seem subtle
|
|
@@ -89,8 +92,11 @@ The `--auto` option is just the same as `--inplace --nobackup --semantic`.
|
|
|
89
92
|
[Markdown source](https://github.com/jlevy/flowmark/blob/main/README.md?plain=1) of
|
|
90
93
|
this readme file.)
|
|
91
94
|
|
|
92
|
-
- Very
|
|
93
|
-
|
|
95
|
+
- Very simple and fast **regex-based sentence splitting**. It's just based on letters
|
|
96
|
+
and punctuation so isn't perfect but works well for these purposes (and is much faster
|
|
97
|
+
and simpler than a proper sentence parser like SpaCy).
|
|
98
|
+
It should work fine for English and many other latin/Cyrillic languages but hasn't
|
|
99
|
+
been tested on CJK.
|
|
94
100
|
|
|
95
101
|
It aims to be small and simple and have only a few dependencies, currently only
|
|
96
102
|
[`marko`](https://github.com/frostming/marko),
|
|
@@ -148,7 +154,6 @@ Markdown content. It can:
|
|
|
148
154
|
|
|
149
155
|
- Optionally break lines at sentence boundaries for better diff readability
|
|
150
156
|
|
|
151
|
-
|
|
152
157
|
- Process plaintext with HTML-aware word splitting
|
|
153
158
|
|
|
154
159
|
It is both a library and a command-line tool.
|
|
@@ -1,9 +1,12 @@
|
|
|
1
1
|
__all__ = (
|
|
2
2
|
"fill_text",
|
|
3
3
|
"fill_markdown",
|
|
4
|
+
"first_sentence",
|
|
5
|
+
"first_sentences",
|
|
4
6
|
"html_md_word_splitter",
|
|
5
7
|
"line_wrap_by_sentence",
|
|
6
8
|
"line_wrap_to_width",
|
|
9
|
+
"split_sentences_regex",
|
|
7
10
|
"wrap_paragraph",
|
|
8
11
|
"wrap_paragraph_lines",
|
|
9
12
|
"Wrap",
|
|
@@ -11,5 +14,6 @@ __all__ = (
|
|
|
11
14
|
|
|
12
15
|
from .line_wrappers import line_wrap_by_sentence, line_wrap_to_width
|
|
13
16
|
from .markdown_filling import fill_markdown
|
|
17
|
+
from .sentence_split_regex import first_sentence, first_sentences, split_sentences_regex
|
|
14
18
|
from .text_filling import Wrap, fill_text
|
|
15
19
|
from .text_wrapping import html_md_word_splitter, wrap_paragraph, wrap_paragraph_lines
|
|
@@ -17,10 +17,11 @@ from contextlib import contextmanager
|
|
|
17
17
|
from textwrap import dedent
|
|
18
18
|
from typing import Any, cast
|
|
19
19
|
|
|
20
|
-
from marko import block, inline
|
|
20
|
+
from marko import Renderer, block, inline
|
|
21
21
|
from marko.block import HTMLBlock
|
|
22
|
+
from marko.ext.gfm import GFM
|
|
23
|
+
from marko.ext.gfm import elements as gfm_elements
|
|
22
24
|
from marko.parser import Parser
|
|
23
|
-
from marko.renderer import Renderer
|
|
24
25
|
from marko.source import Source
|
|
25
26
|
from typing_extensions import override
|
|
26
27
|
|
|
@@ -96,9 +97,14 @@ class CustomParser(Parser):
|
|
|
96
97
|
|
|
97
98
|
class _MarkdownNormalizer(Renderer):
|
|
98
99
|
"""
|
|
99
|
-
Render Markdown in normalized form. This is the internal implementation
|
|
100
|
+
Render Markdown in normalized form. This is the internal implementation
|
|
101
|
+
which overrides most of `MarkdownRenderer`.
|
|
102
|
+
|
|
100
103
|
You likely want to use `normalize_markdown()` instead.
|
|
101
|
-
|
|
104
|
+
|
|
105
|
+
Based on:
|
|
106
|
+
https://github.com/frostming/marko/blob/master/marko/md_renderer.py
|
|
107
|
+
https://github.com/frostming/marko/blob/master/marko/ext/gfm/renderer.py
|
|
102
108
|
"""
|
|
103
109
|
|
|
104
110
|
def __init__(self, line_wrapper: LineWrapper) -> None:
|
|
@@ -126,7 +132,14 @@ class _MarkdownNormalizer(Renderer):
|
|
|
126
132
|
# Suppress item breaks on list items following a top-level paragraph.
|
|
127
133
|
if not self._prefix:
|
|
128
134
|
self._suppress_item_break = True
|
|
135
|
+
|
|
129
136
|
children: Any = self.render_children(element)
|
|
137
|
+
|
|
138
|
+
# GFM checkbox support.
|
|
139
|
+
if hasattr(element, "checked"):
|
|
140
|
+
children = f"[{'x' if element.checked else ' '}] {children}" # pyright: ignore
|
|
141
|
+
|
|
142
|
+
# Wrap the text.
|
|
130
143
|
wrapped_text = self._line_wrapper(
|
|
131
144
|
children,
|
|
132
145
|
self._prefix,
|
|
@@ -288,6 +301,33 @@ class _MarkdownNormalizer(Renderer):
|
|
|
288
301
|
return f"`` {text} ``"
|
|
289
302
|
return f"`{element.children}`"
|
|
290
303
|
|
|
304
|
+
# --- GFM Renderer Methods ---
|
|
305
|
+
|
|
306
|
+
def render_strikethrough(self, element: gfm_elements.Strikethrough) -> str:
|
|
307
|
+
return f"~~{self.render_children(element)}~~"
|
|
308
|
+
|
|
309
|
+
def render_table(self, element: gfm_elements.Table) -> str:
|
|
310
|
+
"""Render a GFM table."""
|
|
311
|
+
lines: list[str] = []
|
|
312
|
+
head, *body = element.children
|
|
313
|
+
lines.append(self.render(head))
|
|
314
|
+
lines.append(f"| {' | '.join(element.delimiters)} |\n")
|
|
315
|
+
for row in body:
|
|
316
|
+
lines.append(self.render(row))
|
|
317
|
+
return "".join(lines)
|
|
318
|
+
|
|
319
|
+
def render_table_row(self, element: gfm_elements.TableRow) -> str:
|
|
320
|
+
"""Render a row within a GFM table."""
|
|
321
|
+
return f"| {' | '.join(self.render(cell) for cell in element.children)} |\n"
|
|
322
|
+
|
|
323
|
+
def render_table_cell(self, element: gfm_elements.TableCell) -> str:
|
|
324
|
+
"""Render a cell within a GFM table row."""
|
|
325
|
+
return self.render_children(element).replace("|", "\\|")
|
|
326
|
+
|
|
327
|
+
def render_url(self, element: gfm_elements.Url) -> str:
|
|
328
|
+
"""For GFM autolink URLs, just output the URL directly."""
|
|
329
|
+
return element.dest
|
|
330
|
+
|
|
291
331
|
|
|
292
332
|
def split_sentences_no_min_length(text: str) -> list[str]:
|
|
293
333
|
return split_sentences_regex(text, min_length=0)
|
|
@@ -338,10 +378,19 @@ def fill_markdown(
|
|
|
338
378
|
# If we want to normalize HTML blocks or comments.
|
|
339
379
|
markdown_text = _normalize_html_comments(markdown_text)
|
|
340
380
|
|
|
341
|
-
#
|
|
381
|
+
# Set up our custom parser, and mix in GFM elements.
|
|
382
|
+
# Using Marko's full extension system is tricky with our customizations so simpler
|
|
383
|
+
# to do this manually.
|
|
342
384
|
parser = CustomParser()
|
|
385
|
+
for e in GFM.elements:
|
|
386
|
+
assert e not in parser.block_elements and e not in parser.inline_elements
|
|
387
|
+
parser.add_element(e)
|
|
388
|
+
|
|
389
|
+
renderer = _MarkdownNormalizer(line_wrapper)
|
|
390
|
+
|
|
391
|
+
# Parse and render.
|
|
343
392
|
parsed = parser.parse(markdown_text)
|
|
344
|
-
result =
|
|
393
|
+
result = renderer.render(parsed)
|
|
345
394
|
|
|
346
395
|
# Reattach frontmatter if it was present
|
|
347
396
|
if frontmatter:
|
|
@@ -2,43 +2,43 @@ from collections.abc import Callable
|
|
|
2
2
|
|
|
3
3
|
import regex
|
|
4
4
|
|
|
5
|
-
# These heuristics are from Flowmark:
|
|
6
|
-
# https://github.com/jlevy/atom-flowmark/blob/master/lib/remark-smart-word-wrap.js#L17-L33
|
|
7
|
-
|
|
8
|
-
# They work pretty well when used for formatting and editing documents in English.
|
|
9
|
-
# Note this is smarter than Python textwrap's simple heuristic:
|
|
10
|
-
# https://github.com/python/cpython/blob/main/Lib/textwrap.py#L105-L110
|
|
11
|
-
|
|
12
|
-
# Heuristic: End of sentence must be two letters or more, with the last letter lowercase,
|
|
13
|
-
# followed by a period, exclamation point, question mark. A final or preceding parenthesis
|
|
14
|
-
# or quote is allowed.
|
|
15
|
-
#
|
|
16
|
-
# Does not break on colon or semicolon currently as that seems to have false positives too
|
|
17
|
-
# often with code or other syntax.
|
|
18
|
-
#
|
|
19
5
|
# XXX: Could also handle rare cases with both quotes and parentheses at sentence end
|
|
20
6
|
# but may not be worth it. Also does not detect sentences ending in numerals, which
|
|
21
7
|
# tends to cause too many false positives. Should be OK for most Latin languages but
|
|
22
8
|
# may need to rethink the 2-letter restriction for some languages.
|
|
23
|
-
|
|
9
|
+
# See also:
|
|
10
|
+
# https://github.com/jlevy/atom-flowmark/blob/master/lib/remark-smart-word-wrap.js#L17-L33
|
|
11
|
+
SENTENCE_END_RE = regex.compile(r"(\b\p{L}+[\p{Ll}])([.?!]['\"’”)]?|['\"’”)][.?!]) *$")
|
|
24
12
|
|
|
25
13
|
# Second heuristic: Very short sentences often not so useful.
|
|
26
14
|
SENTENCE_MIN_LENGTH = 15
|
|
27
15
|
|
|
28
16
|
|
|
29
17
|
def heuristic_end_of_sentence(word: str) -> bool:
|
|
30
|
-
return bool(
|
|
18
|
+
return bool(SENTENCE_END_RE.search(word))
|
|
31
19
|
|
|
32
20
|
|
|
33
21
|
def split_sentences_regex(
|
|
34
22
|
text: str,
|
|
35
|
-
heuristic: Callable[[str], bool] = heuristic_end_of_sentence,
|
|
36
23
|
min_length: int = SENTENCE_MIN_LENGTH,
|
|
24
|
+
heuristic: Callable[[str], bool] = heuristic_end_of_sentence,
|
|
37
25
|
) -> list[str]:
|
|
38
26
|
"""
|
|
39
|
-
Split text into sentences using an approximate, fast regex heuristic.
|
|
27
|
+
Split text into sentences using an approximate, fast regex heuristic.
|
|
28
|
+
|
|
40
29
|
Goal is to be conservative, not perfect, avoiding excessive breaks.
|
|
41
30
|
|
|
31
|
+
The default heuristic: End of sentence must be two letters or more,
|
|
32
|
+
with the last letter lowercase, followed by a period, exclamation point,
|
|
33
|
+
question mark. A final or preceding parenthesis or quote is allowed.
|
|
34
|
+
Does not break on colon or semicolon as that seems to have false
|
|
35
|
+
positives too often with code or other syntax.
|
|
36
|
+
|
|
37
|
+
They work pretty well when used for formatting and editing documents
|
|
38
|
+
in English. It should be reasonable for most Latin languages.
|
|
39
|
+
Note this is smarter than Python textwrap's simpler heuristic:
|
|
40
|
+
https://github.com/python/cpython/blob/main/Lib/textwrap.py#L105-L110
|
|
41
|
+
|
|
42
42
|
:param text: The text to split into sentences.
|
|
43
43
|
:param heuristic: A callable that returns True if text ends at the end of a sentence.
|
|
44
44
|
:param min_length: The minimum length of a sentence in characters.
|
|
@@ -59,3 +59,28 @@ def split_sentences_regex(
|
|
|
59
59
|
if sentence:
|
|
60
60
|
sentences.append(" ".join(sentence))
|
|
61
61
|
return sentences
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def first_sentences(
|
|
65
|
+
text: str,
|
|
66
|
+
n: int,
|
|
67
|
+
min_length: int = SENTENCE_MIN_LENGTH,
|
|
68
|
+
heuristic: Callable[[str], bool] = heuristic_end_of_sentence,
|
|
69
|
+
) -> list[str]:
|
|
70
|
+
"""
|
|
71
|
+
Return the first n sentences from the text.
|
|
72
|
+
"""
|
|
73
|
+
return split_sentences_regex(text, min_length=min_length, heuristic=heuristic)[:n]
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def first_sentence(
|
|
77
|
+
text: str,
|
|
78
|
+
min_length: int = SENTENCE_MIN_LENGTH,
|
|
79
|
+
heuristic: Callable[[str], bool] = heuristic_end_of_sentence,
|
|
80
|
+
) -> str:
|
|
81
|
+
"""
|
|
82
|
+
Return the first sentence from the text. Returns input text unchanged if no
|
|
83
|
+
sentences are found.
|
|
84
|
+
"""
|
|
85
|
+
sentences = split_sentences_regex(text, min_length=min_length, heuristic=heuristic)
|
|
86
|
+
return sentences[0] if sentences else text
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
from dataclasses import dataclass
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
from flowmark.markdown_filling import fill_markdown
|
|
5
|
+
|
|
6
|
+
testdoc_dir = Path("tests/testdocs")
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def test_reference_doc_formats():
|
|
10
|
+
"""
|
|
11
|
+
Test that the reference document is formatted correctly with both plain and semantic formats.
|
|
12
|
+
"""
|
|
13
|
+
orig_path = testdoc_dir / "testdoc.orig.md"
|
|
14
|
+
|
|
15
|
+
# Check that original file exists
|
|
16
|
+
assert orig_path.exists(), f"Original test document not found at {orig_path}"
|
|
17
|
+
|
|
18
|
+
# Read the original content
|
|
19
|
+
with open(orig_path) as f:
|
|
20
|
+
orig_content = f.read()
|
|
21
|
+
|
|
22
|
+
@dataclass(frozen=True)
|
|
23
|
+
class TestCase:
|
|
24
|
+
name: str
|
|
25
|
+
filename: str
|
|
26
|
+
by_sentence: bool
|
|
27
|
+
|
|
28
|
+
test_cases: list[TestCase] = [
|
|
29
|
+
TestCase(name="plain", filename="testdoc.out.plain.md", by_sentence=False),
|
|
30
|
+
TestCase(name="semantic", filename="testdoc.out.semantic.md", by_sentence=True),
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
for case in test_cases:
|
|
34
|
+
test_doc = testdoc_dir / case.filename
|
|
35
|
+
expected = test_doc.read_text()
|
|
36
|
+
|
|
37
|
+
actual = fill_markdown(orig_content, semantic=case.by_sentence)
|
|
38
|
+
if actual != expected:
|
|
39
|
+
actual_path = testdoc_dir / f"testdoc.actual.{case.name}.md"
|
|
40
|
+
print(f"actual was different from expected for {case.name}!")
|
|
41
|
+
print(f"Saving actual to: {actual_path}")
|
|
42
|
+
actual_path.write_text(actual)
|
|
43
|
+
|
|
44
|
+
assert expected == actual
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
from flowmark import first_sentence, split_sentences_regex
|
|
2
|
+
|
|
3
|
+
LONG_TEXT = """
|
|
4
|
+
End of sentence must be two letters or more,
|
|
5
|
+
with the last letter lowercase, followed by a period, exclamation point,
|
|
6
|
+
question mark. A final or preceding parenthesis or quote is allowed.
|
|
7
|
+
Does not break on colon or semicolon as that seems to have false
|
|
8
|
+
positives too often with code or other syntax.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
FIRST_SENTENCE = "End of sentence must be two letters or more, with the last letter lowercase, followed by a period, exclamation point, question mark."
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def test_split_sentences():
|
|
15
|
+
assert split_sentences_regex("test!") == ["test!"]
|
|
16
|
+
assert split_sentences_regex("test! random words") == ["test! random words"]
|
|
17
|
+
|
|
18
|
+
split_sentences = split_sentences_regex(LONG_TEXT)
|
|
19
|
+
print(split_sentences)
|
|
20
|
+
assert len(split_sentences) == 3
|
|
21
|
+
assert split_sentences[0] == FIRST_SENTENCE
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def test_first_sentence():
|
|
25
|
+
assert first_sentence(LONG_TEXT) == FIRST_SENTENCE
|
|
26
|
+
|
|
27
|
+
assert first_sentence("") == ""
|
|
28
|
+
assert first_sentence(" ") == " "
|
|
29
|
+
assert first_sentence("hello") == "hello"
|
|
30
|
+
assert first_sentence(" hello\n") == "hello"
|
|
@@ -78,15 +78,15 @@ If the noteholders had converted their $420K at the 20% discount, they would be
|
|
|
78
78
|
## Legalese
|
|
79
79
|
|
|
80
80
|
> ⚖️
|
|
81
|
-
>
|
|
81
|
+
>
|
|
82
82
|
> ↔️ CONVERTIBLE PROMISSORY NOTE
|
|
83
|
-
>
|
|
83
|
+
>
|
|
84
84
|
> Note Series: `___________________________`
|
|
85
85
|
>
|
|
86
86
|
> e. **Amendment and Waiver.**
|
|
87
87
|
> Any term of this Note may be amended or waived with the written consent of Company and
|
|
88
88
|
> the Majority Holders.
|
|
89
|
-
>
|
|
89
|
+
>
|
|
90
90
|
> f. **Governing Law; Venue.** This Note shall be governed by and construed under the laws of the State of `________`, as
|
|
91
91
|
> applied to agreements among `_______` residents, made and to be performed entirely within the State of `______`, without giving effect to conflicts of laws principles. The venue for any dispute arising out of or related to this Note will lie exclusively in the state or federal courts located in King County, Washington, and the parties to this Note irrevocably waive any right to raise forum non conveniens or any other argument that King County, Washington is not the proper venue. The parties to this Note irrevocably consent to personal jurisdiction in the state and federal courts of the state of Washington.
|
|
92
92
|
|
|
@@ -94,7 +94,7 @@ If the noteholders had converted their $420K at the 20% discount, they would be
|
|
|
94
94
|
> Without in any way limiting the representations set forth above, the Holder further
|
|
95
95
|
> agrees not to make any disposition of all or any portion of the Securities unless and
|
|
96
96
|
> until:
|
|
97
|
-
>
|
|
97
|
+
>
|
|
98
98
|
> 1\. There is then in effect a registration statement under the Act covering such proposed
|
|
99
99
|
> disposition and such disposition is made in accordance with such registration statement;
|
|
100
100
|
> or
|
|
@@ -118,6 +118,7 @@ the Center for American Entrepreneurship.)
|
|
|
118
118
|
|
|
119
119
|
A good heuristic is to assume your readers will be **100% intelligent and 100% ignorant**. Of course, in reality, varied experience exists within an individual reader. Most people may already know *something*, and some people are quicker learners than others. Even world class experts most likely only really know parts of a subject, and contributors may have practical experience in one role--perhaps they have been an entrepreneur, for example, but not an investor. So writing with the assumption that each reader could be both has a variety of advantages:
|
|
120
120
|
|
|
121
|
+
|
|
121
122
|
- Assuming 100% ignorance will help people who think they know a lot about a subject understand **what they didn't know they didn't know**, and fill in the gaps in their knowledge. It encourages contribution from experts and people with practical experience.
|
|
122
123
|
- Assuming 100% intelligence makes readers feel respected, and feel proud to be associated with the material. It will incline beginners to contribute their feedback.
|
|
123
124
|
- Ensures that each reader feels capable of tackling a problem, specific or general.
|
|
@@ -281,6 +282,34 @@ of General Social Survey data).
|
|
|
281
282
|
- **Visionary projects.**
|
|
282
283
|
Projects that were in a completely brand-new space without a precedent, going from 0 to 1.
|
|
283
284
|
|
|
285
|
+
Recommendations:
|
|
286
|
+
|
|
287
|
+
- Use `z` (zoxide) instead of `cd`.
|
|
288
|
+
|
|
289
|
+
```shell
|
|
290
|
+
# Use z in place of cd: switch directories (first time):
|
|
291
|
+
z ~/some/long/path/to/foo
|
|
292
|
+
# Thereafter it's faster:
|
|
293
|
+
z foo
|
|
294
|
+
```
|
|
295
|
+
- Use `eza` instead of `ls`. It has color support, support for Nerd Font icons, and
|
|
296
|
+
other improvements.
|
|
297
|
+
|
|
298
|
+
### **3.6 Comparative Features Matrix**
|
|
299
|
+
|
|
300
|
+
The following table summarizes the native capabilities of the primary platforms evaluated against the core requirements:
|
|
301
|
+
|
|
302
|
+
| **Feature** | **Vercel** | **Netlify** | **Cloudflare Pages** | **AWS (S3/CloudFront)** |
|
|
303
|
+
| --- | --- | --- | --- | --- |
|
|
304
|
+
| **CLI Tool Availability** | Yes (vercel) 1 | Yes (netlify) 2 | Yes (wrangler) 3 | Yes (aws) 4 |
|
|
305
|
+
| **Primary CLI Auth Method** | User Access Token 9 | Personal Access Token (PAT) 2 | API Token 10 | IAM Credentials / STS Token 11 |
|
|
306
|
+
| **Native Token/Key Scoping** | User/Team Level 24 | User/Site Level 30 | Account Level (for Pages Edit) 3 | Path/Prefix Level (via IAM) 4 |
|
|
307
|
+
| **Path-Based Deploy Permissions** | No Native Support 14 | No Native Support 2 | No Native Support 3 | Yes (IAM Policies) 4 |
|
|
308
|
+
| **Primary Native Isolation Method** | Project | Site | Project (but weak permissions) | Path Prefix (with IAM) 4 |
|
|
309
|
+
| **Programmatic File Delete/Update** | Deployment API 14 | Deployment API 28 | Deployment API (via Wrangler) | Direct S3 Object API 12-55 |
|
|
310
|
+
| **Programmatic Invalidation API** | Automatic / Limited | Automatic / Limited | Yes (Cloudflare API) | Yes (CloudFront API) 15-16 |
|
|
311
|
+
| **Basic Cost Model** | Per User / Usage 5 | Per User / Usage 5 | Generous Free Tier / Usage | Usage-Based Components 20-64 |
|
|
312
|
+
|
|
284
313
|
### Boldface, italics, and links
|
|
285
314
|
|
|
286
315
|
X [New York City](https://en.wikipedia.org/wiki/New_York_City).
|
|
@@ -433,6 +433,35 @@ Social Survey data).
|
|
|
433
433
|
- **Visionary projects.** Projects that were in a completely brand-new space without a
|
|
434
434
|
precedent, going from 0 to 1.
|
|
435
435
|
|
|
436
|
+
Recommendations:
|
|
437
|
+
|
|
438
|
+
- Use `z` (zoxide) instead of `cd`.
|
|
439
|
+
|
|
440
|
+
```shell
|
|
441
|
+
# Use z in place of cd: switch directories (first time):
|
|
442
|
+
z ~/some/long/path/to/foo
|
|
443
|
+
# Thereafter it's faster:
|
|
444
|
+
z foo
|
|
445
|
+
```
|
|
446
|
+
- Use `eza` instead of `ls`. It has color support, support for Nerd Font icons, and
|
|
447
|
+
other improvements.
|
|
448
|
+
|
|
449
|
+
### **3.6 Comparative Features Matrix**
|
|
450
|
+
|
|
451
|
+
The following table summarizes the native capabilities of the primary platforms
|
|
452
|
+
evaluated against the core requirements:
|
|
453
|
+
|
|
454
|
+
| **Feature** | **Vercel** | **Netlify** | **Cloudflare Pages** | **AWS (S3/CloudFront)** |
|
|
455
|
+
| --- | --- | --- | --- | --- |
|
|
456
|
+
| **CLI Tool Availability** | Yes (vercel) 1 | Yes (netlify) 2 | Yes (wrangler) 3 | Yes (aws) 4 |
|
|
457
|
+
| **Primary CLI Auth Method** | User Access Token 9 | Personal Access Token (PAT) 2 | API Token 10 | IAM Credentials / STS Token 11 |
|
|
458
|
+
| **Native Token/Key Scoping** | User/Team Level 24 | User/Site Level 30 | Account Level (for Pages Edit) 3 | Path/Prefix Level (via IAM) 4 |
|
|
459
|
+
| **Path-Based Deploy Permissions** | No Native Support 14 | No Native Support 2 | No Native Support 3 | Yes (IAM Policies) 4 |
|
|
460
|
+
| **Primary Native Isolation Method** | Project | Site | Project (but weak permissions) | Path Prefix (with IAM) 4 |
|
|
461
|
+
| **Programmatic File Delete/Update** | Deployment API 14 | Deployment API 28 | Deployment API (via Wrangler) | Direct S3 Object API 12-55 |
|
|
462
|
+
| **Programmatic Invalidation API** | Automatic / Limited | Automatic / Limited | Yes (Cloudflare API) | Yes (CloudFront API) 15-16 |
|
|
463
|
+
| **Basic Cost Model** | Per User / Usage 5 | Per User / Usage 5 | Generous Free Tier / Usage | Usage-Based Components 20-64 |
|
|
464
|
+
|
|
436
465
|
### Boldface, italics, and links
|
|
437
466
|
|
|
438
467
|
X [New York City](https://en.wikipedia.org/wiki/New_York_City). XX
|
|
@@ -862,16 +891,13 @@ valuation, shares, fundraising, and dilution (<a
|
|
|
862
891
|
href="http://ownyourventure.com/equitySim.html">source</a>) <br> </div>
|
|
863
892
|
|
|
864
893
|
| Specific AWS Services | Basics | Tips | Gotchas |
|
|
865
|
-
|
|
866
|
-
| [Security and IAM](#security-and-iam) | [📗](#security-and-iam-basics) |
|
|
867
|
-
[📘](#
|
|
868
|
-
[
|
|
869
|
-
[
|
|
870
|
-
[
|
|
871
|
-
[
|
|
872
|
-
[📘](#ami-tips) | [📙](#ami-gotchas-and-limitations) | | [Auto Scaling](#auto-scaling) |
|
|
873
|
-
[📗](#auto-scaling-basics) | [📘](#auto-scaling-tips) |
|
|
874
|
-
[📙](#auto-scaling-gotchas-and-limitations) |
|
|
894
|
+
| --------------------------------------- | -------------------------------- | ------------------------------- | ------------------------------------------------ |
|
|
895
|
+
| [Security and IAM](#security-and-iam) | [📗](#security-and-iam-basics) | [📘](#security-and-iam-tips) | [📙](#security-and-iam-gotchas-and-limitations) |
|
|
896
|
+
| [S3](#s3) | [📗](#s3-basics) | [📘](#s3-tips) | [📙](#s3-gotchas-and-limitations) |
|
|
897
|
+
| [EC2](#ec2) | [📗](#ec2-basics) | [📘](#ec2-tips) | [📙](#ec2-gotchas-and-limitations) |
|
|
898
|
+
| [CloudWatch](#cloudwatch) | [📗](#cloudwatch-basics) | [📘](#cloudwatch-tips) | [📙](#cloudwatch-gotchas-and-limitations) |
|
|
899
|
+
| [AMIs](#amis) | [📗](#ami-basics) | [📘](#ami-tips) | [📙](#ami-gotchas-and-limitations) |
|
|
900
|
+
| [Auto Scaling](#auto-scaling) | [📗](#auto-scaling-basics) | [📘](#auto-scaling-tips) | [📙](#auto-scaling-gotchas-and-limitations) |
|
|
875
901
|
|
|
876
902
|
- 📒 [FAQ](https://aws.amazon.com/cloudwatch/faqs/) ∙
|
|
877
903
|
[Pricing](https://aws.amazon.com/cloudwatch/pricing/) - 🔹Blahxxx - ❗Blahxxx
|
|
@@ -456,6 +456,35 @@ Social Survey data).
|
|
|
456
456
|
- **Visionary projects.** Projects that were in a completely brand-new space without a
|
|
457
457
|
precedent, going from 0 to 1.
|
|
458
458
|
|
|
459
|
+
Recommendations:
|
|
460
|
+
|
|
461
|
+
- Use `z` (zoxide) instead of `cd`.
|
|
462
|
+
|
|
463
|
+
```shell
|
|
464
|
+
# Use z in place of cd: switch directories (first time):
|
|
465
|
+
z ~/some/long/path/to/foo
|
|
466
|
+
# Thereafter it's faster:
|
|
467
|
+
z foo
|
|
468
|
+
```
|
|
469
|
+
- Use `eza` instead of `ls`. It has color support, support for Nerd Font icons, and
|
|
470
|
+
other improvements.
|
|
471
|
+
|
|
472
|
+
### **3.6 Comparative Features Matrix**
|
|
473
|
+
|
|
474
|
+
The following table summarizes the native capabilities of the primary platforms
|
|
475
|
+
evaluated against the core requirements:
|
|
476
|
+
|
|
477
|
+
| **Feature** | **Vercel** | **Netlify** | **Cloudflare Pages** | **AWS (S3/CloudFront)** |
|
|
478
|
+
| --- | --- | --- | --- | --- |
|
|
479
|
+
| **CLI Tool Availability** | Yes (vercel) 1 | Yes (netlify) 2 | Yes (wrangler) 3 | Yes (aws) 4 |
|
|
480
|
+
| **Primary CLI Auth Method** | User Access Token 9 | Personal Access Token (PAT) 2 | API Token 10 | IAM Credentials / STS Token 11 |
|
|
481
|
+
| **Native Token/Key Scoping** | User/Team Level 24 | User/Site Level 30 | Account Level (for Pages Edit) 3 | Path/Prefix Level (via IAM) 4 |
|
|
482
|
+
| **Path-Based Deploy Permissions** | No Native Support 14 | No Native Support 2 | No Native Support 3 | Yes (IAM Policies) 4 |
|
|
483
|
+
| **Primary Native Isolation Method** | Project | Site | Project (but weak permissions) | Path Prefix (with IAM) 4 |
|
|
484
|
+
| **Programmatic File Delete/Update** | Deployment API 14 | Deployment API 28 | Deployment API (via Wrangler) | Direct S3 Object API 12-55 |
|
|
485
|
+
| **Programmatic Invalidation API** | Automatic / Limited | Automatic / Limited | Yes (Cloudflare API) | Yes (CloudFront API) 15-16 |
|
|
486
|
+
| **Basic Cost Model** | Per User / Usage 5 | Per User / Usage 5 | Generous Free Tier / Usage | Usage-Based Components 20-64 |
|
|
487
|
+
|
|
459
488
|
### Boldface, italics, and links
|
|
460
489
|
|
|
461
490
|
X [New York City](https://en.wikipedia.org/wiki/New_York_City). XX
|
|
@@ -903,16 +932,13 @@ valuation, shares, fundraising, and dilution (<a
|
|
|
903
932
|
href="http://ownyourventure.com/equitySim.html">source</a>) <br> </div>
|
|
904
933
|
|
|
905
934
|
| Specific AWS Services | Basics | Tips | Gotchas |
|
|
906
|
-
|
|
907
|
-
| [Security and IAM](#security-and-iam) | [📗](#security-and-iam-basics) |
|
|
908
|
-
[📘](#
|
|
909
|
-
[
|
|
910
|
-
[
|
|
911
|
-
[
|
|
912
|
-
[
|
|
913
|
-
[📘](#ami-tips) | [📙](#ami-gotchas-and-limitations) | | [Auto Scaling](#auto-scaling) |
|
|
914
|
-
[📗](#auto-scaling-basics) | [📘](#auto-scaling-tips) |
|
|
915
|
-
[📙](#auto-scaling-gotchas-and-limitations) |
|
|
935
|
+
| --------------------------------------- | -------------------------------- | ------------------------------- | ------------------------------------------------ |
|
|
936
|
+
| [Security and IAM](#security-and-iam) | [📗](#security-and-iam-basics) | [📘](#security-and-iam-tips) | [📙](#security-and-iam-gotchas-and-limitations) |
|
|
937
|
+
| [S3](#s3) | [📗](#s3-basics) | [📘](#s3-tips) | [📙](#s3-gotchas-and-limitations) |
|
|
938
|
+
| [EC2](#ec2) | [📗](#ec2-basics) | [📘](#ec2-tips) | [📙](#ec2-gotchas-and-limitations) |
|
|
939
|
+
| [CloudWatch](#cloudwatch) | [📗](#cloudwatch-basics) | [📘](#cloudwatch-tips) | [📙](#cloudwatch-gotchas-and-limitations) |
|
|
940
|
+
| [AMIs](#amis) | [📗](#ami-basics) | [📘](#ami-tips) | [📙](#ami-gotchas-and-limitations) |
|
|
941
|
+
| [Auto Scaling](#auto-scaling) | [📗](#auto-scaling-basics) | [📘](#auto-scaling-tips) | [📙](#auto-scaling-gotchas-and-limitations) |
|
|
916
942
|
|
|
917
943
|
- 📒 [FAQ](https://aws.amazon.com/cloudwatch/faqs/) ∙
|
|
918
944
|
[Pricing](https://aws.amazon.com/cloudwatch/pricing/) - 🔹Blahxxx - ❗Blahxxx
|
|
@@ -1,47 +0,0 @@
|
|
|
1
|
-
from pathlib import Path
|
|
2
|
-
from typing import TypedDict
|
|
3
|
-
|
|
4
|
-
from flowmark.markdown_filling import fill_markdown
|
|
5
|
-
|
|
6
|
-
testdoc_dir = Path("tests/testdocs")
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
def test_reference_doc_formats():
|
|
10
|
-
"""Test that the reference document is formatted correctly with both plain and semantic formats."""
|
|
11
|
-
orig_path = testdoc_dir / "testdoc.orig.md"
|
|
12
|
-
|
|
13
|
-
# Check that original file exists
|
|
14
|
-
assert orig_path.exists(), f"Original test document not found at {orig_path}"
|
|
15
|
-
|
|
16
|
-
# Read the original content
|
|
17
|
-
with open(orig_path) as f:
|
|
18
|
-
orig_content = f.read()
|
|
19
|
-
|
|
20
|
-
class TestCase(TypedDict):
|
|
21
|
-
name: str
|
|
22
|
-
filename: str
|
|
23
|
-
by_sentence: bool
|
|
24
|
-
|
|
25
|
-
# Test configurations
|
|
26
|
-
test_cases: list[TestCase] = [
|
|
27
|
-
{"name": "plain", "filename": "testdoc.out.plain.md", "by_sentence": False},
|
|
28
|
-
{"name": "semantic", "filename": "testdoc.out.semantic.md", "by_sentence": True},
|
|
29
|
-
]
|
|
30
|
-
|
|
31
|
-
for case in test_cases:
|
|
32
|
-
output_path = testdoc_dir / case["filename"]
|
|
33
|
-
assert output_path.exists(), (
|
|
34
|
-
f"{case['name'].capitalize()}-processed document not found at {output_path}"
|
|
35
|
-
)
|
|
36
|
-
|
|
37
|
-
# Read the processed content
|
|
38
|
-
with open(output_path) as f:
|
|
39
|
-
processed_content = f.read()
|
|
40
|
-
|
|
41
|
-
# Process the original file using fill_markdown with appropriate option
|
|
42
|
-
expected_content = fill_markdown(orig_content, semantic=case["by_sentence"])
|
|
43
|
-
|
|
44
|
-
# Compare the processed content with the expected output
|
|
45
|
-
assert expected_content == processed_content, (
|
|
46
|
-
f"{case['name'].capitalize()} flowmark processing doesn't match expected output"
|
|
47
|
-
)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|