PyPI - sonatoki - Versions diffs - 0.1.3__tar.gz → 0.1.4__tar.gz - Mend

sonatoki 0.1.3tar.gz → 0.1.4tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (28) hide show

{sonatoki-0.1.3 → sonatoki-0.1.4}/PKG-INFO RENAMED Viewed

@@ -1,6 +1,6 @@
 Metadata-Version: 2.1
 Name: sonatoki
-Version: 0.1.3
+Version: 0.1.4
 Summary: ilo li moku e toki li pana e sona ni: ni li toki ala toki pona?
 Author-Email: "jan Kekan San (@gregdan3)" <gregory.danielson3@gmail.com>
 License: AGPL-3.0-or-later

{sonatoki-0.1.3 → sonatoki-0.1.4}/pyproject.toml RENAMED Viewed

@@ -1,6 +1,6 @@
 [project]
 name = "sonatoki"
-version = "0.1.3"
+version = "0.1.4"
 description = "ilo li moku e toki li pana e sona ni: ni li toki ala toki pona?"
 authors = [
     { name = "jan Kekan San (@gregdan3)", email = "gregory.danielson3@gmail.com" },

{sonatoki-0.1.3 → sonatoki-0.1.4}/src/sonatoki/Configs.py RENAMED Viewed

@@ -9,15 +9,15 @@ from typing_extensions import NotRequired
 from sonatoki.Filters import (
     Filter,
     NimiPu,
-    Numerics,
+    Numeric,
     Syllabic,
     NimiLinku,
     NimiPuAle,
     Alphabetic,
     ProperName,
     Phonotactic,
+    Punctuation,
     NimiLinkuAle,
-    Punctuations,
 )
 from sonatoki.Scorers import Number, Scorer, PassFail, SoftScaling, SoftPassFail
 from sonatoki.Cleaners import Cleaner, ConsecutiveDuplicates
@@ -45,7 +45,7 @@ class IloConfig(TypedDict):
 BaseConfig: IloConfig = {
     "preprocessors": [URLs],
     "cleaners": [ConsecutiveDuplicates],
-    "ignoring_filters": [Numerics, Punctuations],
+    "ignoring_filters": [Numeric, Punctuation],
     "scoring_filters": [],
     "scorer": PassFail,
     "passing_score": 0.8,

{sonatoki-0.1.3 → sonatoki-0.1.4}/src/sonatoki/Filters.py RENAMED Viewed

@@ -131,7 +131,7 @@ class Alphabetic(Filter):
         return set(token.lower()).issubset(ALPHABET_SET)
-class Numerics(Filter):
+class Numeric(Filter):
     """Determine if a given token is entirely numeric.
     Covers all numeric symbols in Unicode.
@@ -147,7 +147,7 @@ class Numerics(Filter):
         return msg.isnumeric()
-class Punctuations(RegexFilter):
+class Punctuation(RegexFilter):
     pattern = re.compile(r"[\p{Punctuation}\p{posix_punct}]+")
@@ -159,6 +159,6 @@ __all__ = [
     "Syllabic",
     "Alphabetic",
     "ProperName",
-    "Punctuations",
-    "Numerics",
+    "Punctuation",
+    "Numeric",
 ]

{sonatoki-0.1.3 → sonatoki-0.1.4}/src/sonatoki/Preprocessors.py RENAMED Viewed

@@ -62,6 +62,13 @@ class URLs(RegexPreprocessor):
     pattern = re.compile(r"https?:\/\/\S+")
+class Reference(RegexPreprocessor):
+    """Remove text contained in double brackets.
+    Often used to fetch articles on Wikipedia, or Magic the Gathering cards."""
+    pattern = re.compile(r"\[\[.+\]\]")
 class DiscordEmotes(RegexPreprocessor):
     """Remove text-formatted Discord emotes `<flags:name:id>`"""
@@ -80,6 +87,13 @@ class DiscordSpecial(RegexPreprocessor):
     pattern = re.compile(r"<id:[a-zA-Z0-9_]{4,}>")
+class AngleBracketObject(RegexPreprocessor):
+    """A generalized version of the Discord-specific angle bracket objects.
+    Removes any contiguous (not broken by whitespace) text in angle brackets."""
+    pattern = re.compile(r"<[^<>\s]+>")
 """
 The following classes are Containers.
@@ -92,23 +106,23 @@ would likely be using a language other than Toki Pona.
 class SingleQuotes(RegexPreprocessor):
-    pattern = re.compile(r"'[^']+'", flags=re.S)  # . matches newline
+    pattern = re.compile(r"'[^']+'", flags=re.DOTALL)
 class DoubleQuotes(RegexPreprocessor):
-    pattern = re.compile(r'"[^"]+"', flags=re.S)
+    pattern = re.compile(r'"[^"]+"', flags=re.DOTALL)
 class Backticks(RegexPreprocessor):
     """Remove paired backticks and their contents `like this`"""
-    pattern = re.compile(r"`[^`]+`", flags=re.S)
+    pattern = re.compile(r"`[^`]+`", flags=re.DOTALL)
 class Spoilers(RegexPreprocessor):
     """Remove paired double bars and their contents `||like this||`"""
-    pattern = re.compile(r"\|\|(?:(?!\|\|).)+\|\|", flags=re.S)
+    pattern = re.compile(r"\|\|(?:(?!\|\|).)+\|\|", flags=re.DOTALL)
 class ArrowQuote(RegexPreprocessor):
@@ -117,7 +131,22 @@ class ArrowQuote(RegexPreprocessor):
     pattern = re.compile(r"^>\ .+$", re.MULTILINE)
+class AllQuotes(RegexPreprocessor):
+    pattern = re.compile(
+        "|".join(
+            [
+                SingleQuotes.pattern.pattern,
+                DoubleQuotes.pattern.pattern,
+                Backticks.pattern.pattern,
+                ArrowQuote.pattern.pattern,
+            ]
+        ),
+        flags=re.MULTILINE | re.DOTALL,
+    )
 __all__ = [
+    "AngleBracketObject",
     "DiscordChannels",
     "DiscordMentions",
     "DiscordSpecial",
@@ -125,7 +154,9 @@ __all__ = [
     "SingleQuotes",
     "DoubleQuotes",
     "ArrowQuote",
+    "AllQuotes",
     "Backticks",
+    "Reference",
     "Spoilers",
     "URLs",
 ]

{sonatoki-0.1.3 → sonatoki-0.1.4}/tests/test_filters.py RENAMED Viewed

@@ -9,13 +9,13 @@ from hypothesis import HealthCheck, given, assume, example, settings
 # LOCAL
 from sonatoki.Filters import (
     NimiPu,
-    Numerics,
+    Numeric,
     Syllabic,
     NimiLinku,
     Alphabetic,
     ProperName,
     Phonotactic,
-    Punctuations,
+    Punctuation,
 )
 from sonatoki.Cleaners import ConsecutiveDuplicates
 from sonatoki.constants import NIMI_PU, NIMI_LINKU
@@ -90,9 +90,9 @@ def test_ProperName(s: str):
 @example("「　」")
 @example(string.punctuation)
 @settings(suppress_health_check=[HealthCheck.filter_too_much])  # FIXME
-def test_Punctuations(s: str):
-    _ = assume(re.fullmatch(Punctuations.pattern.pattern, s))
-    res = Punctuations.filter(s)
+def test_Punctuation(s: str):
+    _ = assume(re.fullmatch(Punctuation.pattern.pattern, s))
+    res = Punctuation.filter(s)
     assert res, repr(s)
@@ -100,5 +100,5 @@ def test_Punctuations(s: str):
 @example("124125")
 @example("99990000")
 def test_Numeric(s: str):
-    res = Numerics.filter(s)
+    res = Numeric.filter(s)
     assert res, repr(s)

{sonatoki-0.1.3 → sonatoki-0.1.4}/tests/test_preprocessors.py RENAMED Viewed

@@ -6,7 +6,9 @@ from hypothesis import given, example
 from sonatoki.Preprocessors import (
     URLs,
     Spoilers,
+    AllQuotes,
     Backticks,
+    Reference,
     ArrowQuote,
     DoubleQuotes,
     SingleQuotes,
@@ -14,6 +16,7 @@ from sonatoki.Preprocessors import (
     DiscordSpecial,
     DiscordChannels,
     DiscordMentions,
+    AngleBracketObject,
 )
@@ -101,3 +104,40 @@ def test_DiscordChannels(s: str):
 def test_DiscordSpecial(s: str):
     res = DiscordSpecial.process(s).strip()
     assert res == "", (repr(s), repr(res))
+@given(
+    st.from_regex(DiscordEmotes.pattern.pattern, fullmatch=True)
+    | st.from_regex(DiscordMentions.pattern.pattern, fullmatch=True)
+    | st.from_regex(DiscordChannels.pattern.pattern, fullmatch=True)
+    | st.from_regex(DiscordSpecial.pattern.pattern, fullmatch=True)
+    | st.from_regex(AngleBracketObject.pattern.pattern, fullmatch=True)
+)
+@example("<https://example.com>")
+@example("<#123124125125>")
+def test_AngleBracketObject(s: str):
+    res = AngleBracketObject.process(s).strip()
+    assert res == "", (repr(s), repr(res))
+@given(
+    st.from_regex(SingleQuotes.pattern.pattern, fullmatch=True)
+    | st.from_regex(DoubleQuotes.pattern.pattern, fullmatch=True)
+    | st.from_regex(Backticks.pattern.pattern, fullmatch=True)
+    | st.from_regex(ArrowQuote.pattern.pattern, fullmatch=True)
+    | st.from_regex(AllQuotes.pattern.pattern, fullmatch=True)
+)
+@example("> bruh")
+@example("`bruh`")
+def test_AllQuotes(s: str):
+    res = AllQuotes.process(s).strip()
+    assert res == "", (repr(s), repr(res))
+@given(st.from_regex(Reference.pattern.pattern, fullmatch=True))
+@example("[[Brainstorm]]")
+@example("[[Phatic Phrases]]")
+@example("[[Yahoo!]]")
+def test_Reference(s: str):
+    res = Reference.process(s).strip()
+    assert res == "", (repr(s), repr(res))

{sonatoki-0.1.3 → sonatoki-0.1.4}/tests/test_scorers.py RENAMED Viewed

@@ -4,38 +4,39 @@ from typing import List, Type
 # PDM
 import pytest
 import hypothesis.strategies as st
-from hypothesis import given
+from hypothesis import given, example
 # LOCAL
 from sonatoki.Filters import (
     Filter,
     NimiPu,
-    Numerics,
+    Numeric,
     Syllabic,
     NimiLinku,
     Alphabetic,
     ProperName,
     Phonotactic,
-    Punctuations,
+    Punctuation,
 )
-from sonatoki.Scorers import Scorer, Scaling, PassFail, SoftScaling
+from sonatoki.Scorers import Scorer, Scaling, PassFail, SoftScaling, SoftPassFail
 # FILESYSTEM
 from .test_utils import token_strategy
 FILTERS = [
     NimiPu,
-    Numerics,
+    Numeric,
     Syllabic,
     NimiLinku,
     Alphabetic,
     ProperName,
     Phonotactic,
-    Punctuations,
+    Punctuation,
 ]
 SCORERS = [
     PassFail,
+    SoftPassFail,
     Scaling,
     SoftScaling,
 ]
@@ -46,6 +47,7 @@ SCORERS = [
     st.lists(st.sampled_from(FILTERS), min_size=1, unique=True),
     st.lists(token_strategy, min_size=0, max_size=10),
 )
+@example(st.sampled_from(FILTERS), [])
 def test_score_bounds(scorer: Scorer, filters: List[Type[Filter]], text: List[str]):
     score = scorer.score(text, filters)
     assert 0 <= score <= 1, (score, filters, text)