PyPI - sonatoki - Versions diffs - 0.1.4__py3-none-any.whl → 0.1.5__py3-none-any.whl - Mend

sonatoki 0.1.4py3-none-any.whl → 0.1.5py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (11) hide show

sonatoki/Filters.py +16 -3
sonatoki/Preprocessors.py +13 -2
sonatoki/Scorers.py +2 -14
sonatoki/Tokenizers.py +22 -7
sonatoki/constants.py +16 -0
sonatoki/ilo.py +0 -12
{sonatoki-0.1.4.dist-info → sonatoki-0.1.5.dist-info}/METADATA +1 -1
sonatoki-0.1.5.dist-info/RECORD +16 -0
sonatoki-0.1.4.dist-info/RECORD +0 -16
{sonatoki-0.1.4.dist-info → sonatoki-0.1.5.dist-info}/WHEEL +0 -0
{sonatoki-0.1.4.dist-info → sonatoki-0.1.5.dist-info}/licenses/LICENSE +0 -0

sonatoki/Filters.py CHANGED Viewed

@@ -1,10 +1,11 @@
 # STL
+import re
 from abc import ABC, abstractmethod
 from typing import Set
 from functools import lru_cache as cache  # cache comes in 3.9
 # PDM
-import regex as re
+import regex
 from typing_extensions import override
 # LOCAL
@@ -13,14 +14,16 @@ from sonatoki.constants import (
     CONSONANTS,
     NIMI_PU_SET,
     ALPHABET_SET,
+    UNICODE_PUNCT,
     ALLOWABLES_SET,
     NIMI_LINKU_SET,
     NIMI_PU_ALE_SET,
     NIMI_LINKU_ALE_SET,
+    PRUNED_POSIX_PUNCT,
     NIMI_LINKU_SANDBOX_SET,
 )
-re.DEFAULT_VERSION = re.VERSION1
+regex.DEFAULT_VERSION = regex.VERSION1
 class Filter(ABC):
@@ -41,6 +44,16 @@ class RegexFilter(Filter):
         return not not re.fullmatch(cls.pattern, token)
+class Regex1Filter(Filter):
+    pattern: "regex.Pattern[str]"
+    @classmethod
+    @override
+    @cache(maxsize=None)
+    def filter(cls, token: str) -> bool:
+        return not not regex.fullmatch(cls.pattern, token)
 class SetFilter(Filter):
     tokens: Set[str]
@@ -148,7 +161,7 @@ class Numeric(Filter):
 class Punctuation(RegexFilter):
-    pattern = re.compile(r"[\p{Punctuation}\p{posix_punct}]+")
+    pattern = re.compile(rf"[{PRUNED_POSIX_PUNCT}{UNICODE_PUNCT}]+")
 __all__ = [

sonatoki/Preprocessors.py CHANGED Viewed

@@ -17,13 +17,14 @@ It is up to the user to order them appropriately.
 """
 # STL
+import re
 from abc import ABC, abstractmethod
 # PDM
-import regex as re
+import regex
 from typing_extensions import override
-re.DEFAULT_VERSION = re.VERSION1
+regex.DEFAULT_VERSION = regex.VERSION1
 class Preprocessor(ABC):
@@ -43,6 +44,16 @@ class RegexPreprocessor(Preprocessor):
         return re.sub(cls.pattern, cls.replace, msg)
+class Regex1Preprocessor(Preprocessor):
+    pattern: "regex.Pattern[str]"
+    replace: str = " "
+    @classmethod
+    @override
+    def process(cls, msg: str) -> str:
+        return regex.sub(cls.pattern, cls.replace, msg)
 """
 The following classes are Ignorables.

sonatoki/Scorers.py CHANGED Viewed

@@ -10,8 +10,6 @@ from typing_extensions import override
 # LOCAL
 from sonatoki.Filters import Filter
-LOG = logging.getLogger(__name__)
 Number = Union[int, float]
 Weights = Dict[str, Number]
@@ -37,12 +35,7 @@ class PassFail(Scorer):
     def score_token(cls, token: str, filters: List[Type[Filter]]) -> Number:
         for f in filters:
             if f.filter(token):
-                score = 1
-                LOG.debug(
-                    "%12s.%s('%s') = %.2f", cls.__name__, f.__name__, token, score
-                )
-                return score
-        LOG.debug("%12s('%s') = 0.00", cls.__name__, token)
+                return 1
         return 0
     @classmethod
@@ -86,12 +79,7 @@ class Scaling(Scorer):
     def score_token(cls, token: str, filters: List[Type[Filter]], scale: int):
         for i, f in enumerate(filters):
             if f.filter(token):
-                score = scale - i
-                LOG.debug(
-                    "%12s.%s('%s') = %.2f", cls.__name__, f.__name__, token, score
-                )
-                return score
-        LOG.debug("%12s('%s') = 0.00", cls.__name__, token)
+                return scale - i
         return 0
     @classmethod

sonatoki/Tokenizers.py CHANGED Viewed

@@ -1,11 +1,15 @@
 # STL
+import re
 from abc import ABC, abstractmethod
 from typing import List
 # PDM
-import regex as re
+import regex
 from typing_extensions import override
+# LOCAL
+from sonatoki.constants import UNICODE_PUNCT, PRUNED_POSIX_PUNCT
 try:
     # PDM
     import nltk
@@ -15,7 +19,7 @@ except ImportError as e:
     nltk = e
-LANGUAGE = "english"  # for NLTK
+regex.DEFAULT_VERSION = regex.VERSION1
 class Tokenizer(ABC):
@@ -42,15 +46,26 @@ class RegexTokenizer(Tokenizer):
         return [clean for word in re.split(cls.pattern, s) if (clean := word.strip())]
+class Regex1Tokenizer(Tokenizer):
+    pattern: "regex.Pattern[str]"
+    @classmethod
+    @override
+    def tokenize(cls, s: str) -> List[str]:
+        return [
+            clean for word in regex.split(cls.pattern, s) if (clean := word.strip())
+        ]
 class WordTokenizerTok(RegexTokenizer):
-    pattern = re.compile(r"""([\p{Punctuation}\p{posix_punct}]+|\s+)""")
-    # TODO: are <> or {} that common as *sentence* delims? [] are already a stretch
-    # TODO: do the typography characters matter?
-    # NOTE: | / and , are *not* sentence delimiters for my purpose
+    pattern = re.compile(rf"""([{PRUNED_POSIX_PUNCT}{UNICODE_PUNCT}]+|\s+)""")
 class SentTokenizerTok(RegexTokenizer):
-    pattern = re.compile(r"""(?<=[.?!:;·…“”"'()\[\]\-]|$)""")
+    pattern = re.compile(r"""(?<=[.?!:;·…“”"'()\[\]\-])|$""", flags=re.MULTILINE)
+    # TODO: are <> or {} that common as *sentence* delims? [] are already a stretch
+    # TODO: do the typography characters matter?
+    # NOTE: | / and , are *not* sentence delimiters for my purpose
 class WordTokenizerRe(RegexTokenizer):

sonatoki/constants.py CHANGED Viewed

@@ -11,6 +11,22 @@ CONSONANTS = "jklmnpstw"
 ALPHABET = VOWELS + CONSONANTS
 ALPHABET_SET = set(ALPHABET)
+LANGUAGE = "english"  # for NLTK
+# `\p{posix_punct}` character class
+POSIX_PUNCT = r"""-!"#$%&'()*+,./:;<=>?@[\]^_`{|}~"""
+PRUNED_POSIX_PUNCT = r"""$+<=>^`|~"""  # only those that are not in UNICODE_PUNCT
+# `\p{Punctuation}` character class
+UNICODE_PUNCT = r"""!"#%&'()*,-./:;?@\[\\\]_{}¡§«¶·»¿;·՚՛՜՝՞՟։֊־׀׃׆׳״؉؊،؍؛؝؞؟٪٫٬٭۔܀܁܂܃܄܅܆܇܈܉܊܋܌܍߷߸߹࠰࠱࠲࠳࠴࠵࠶࠷࠸࠹࠺࠻࠼࠽࠾࡞।॥॰৽੶૰౷಄෴๏๚๛༄༅༆༇༈༉༊་༌།༎༏༐༑༒༔༺༻༼༽྅࿐࿑࿒࿓࿔࿙࿚၊။၌၍၎၏჻፠፡።፣፤፥፦፧፨᐀᙮᚛᚜᛫᛬᛭᜵᜶។៕៖៘៙៚᠀᠁᠂᠃᠄᠅᠆᠇᠈᠉᠊᥄᥅᨞᨟᪠᪡᪢᪣᪤᪥᪦᪨᪩᪪᪫᪬᪭᭚᭛᭜᭝᭞᭟᭠᭽᭾᯼᯽᯾᯿᰻᰼᰽᰾᰿᱾᱿᳀᳁᳂᳃᳄᳅᳆᳇᳓‐‑‒–—―‖‗‘’‚‛“”„‟†‡•‣․‥…‧‰‱′″‴‵‶‷‸‹›※‼‽‾‿⁀⁁⁂⁃⁅⁆⁇⁈⁉⁊⁋⁌⁍⁎⁏⁐⁑⁓⁔⁕⁖⁗⁘⁙⁚⁛⁜⁝⁞⁽⁾₍₎⌈⌉⌊⌋〈〉❨❩❪❫❬❭❮❯❰❱❲❳❴❵⟅⟆⟦⟧⟨⟩⟪⟫⟬⟭⟮⟯⦃⦄⦅⦆⦇⦈⦉⦊⦋⦌⦍⦎⦏⦐⦑⦒⦓⦔⦕⦖⦗⦘⧘⧙⧚⧛⧼⧽⳹⳺⳻⳼⳾⳿⵰⸀⸁⸂⸃⸄⸅⸆⸇⸈⸉⸊⸋⸌⸍⸎⸏⸐⸑⸒⸓⸔⸕⸖⸗⸘⸙⸚⸛⸜⸝⸞⸟⸠⸡⸢⸣⸤⸥⸦⸧⸨⸩⸪⸫⸬⸭⸮⸰⸱⸲⸳⸴⸵⸶⸷⸸⸹⸺⸻⸼⸽⸾⸿⹀⹁⹂⹃⹄⹅⹆⹇⹈⹉⹊⹋⹌⹍⹎⹏⹒⹓⹔⹕⹖⹗⹘⹙⹚⹛⹜⹝、。〃〈〉《》「」『』【】〔〕〖〗〘〙〚〛〜〝〞〟〰〽゠・꓾꓿꘍꘎꘏꙳꙾꛲꛳꛴꛵꛶꛷꡴꡵꡶꡷꣎꣏꣸꣹꣺꣼꤮꤯꥟꧁꧂꧃꧄꧅꧆꧇꧈꧉꧊꧋꧌꧍꧞꧟꩜꩝꩞꩟꫞꫟꫰꫱꯫﴾﴿︐︑︒︓︔︕︖︗︘︙︰︱︲︳︴︵︶︷︸︹︺︻︼︽︾︿﹀﹁﹂﹃﹄﹅﹆﹇﹈﹉﹊﹋﹌﹍﹎﹏﹐﹑﹒﹔﹕﹖﹗﹘﹙﹚﹛﹜﹝﹞﹟﹠﹡﹣﹨﹪﹫！＂＃％＆＇（）＊，－．／：；？＠［＼］＿｛｝｟｠｡｢｣､･𐄀𐄁𐄂𐎟𐏐𐕯𐡗𐤟𐤿𐩐𐩑𐩒𐩓𐩔𐩕𐩖𐩗𐩘𐩿𐫰𐫱𐫲𐫳𐫴𐫵𐫶𐬹𐬺𐬻𐬼𐬽𐬾𐬿𐮙𐮚𐮛𐮜𐺭𐽕𐽖𐽗𐽘𐽙𐾆𐾇𐾈𐾉𑁇𑁈𑁉𑁊𑁋𑁌𑁍𑂻𑂼𑂾𑂿𑃀𑃁𑅀𑅁𑅂𑅃𑅴𑅵𑇅𑇆𑇇𑇈𑇍𑇛𑇝𑇞𑇟𑈸𑈹𑈺𑈻𑈼𑈽𑊩𑑋𑑌𑑍𑑎𑑏𑑚𑑛𑑝𑓆𑗁𑗂𑗃𑗄𑗅𑗆𑗇𑗈𑗉𑗊𑗋𑗌𑗍𑗎𑗏𑗐𑗑𑗒𑗓𑗔𑗕𑗖𑗗𑙁𑙂𑙃𑙠𑙡𑙢𑙣𑙤𑙥𑙦𑙧𑙨𑙩𑙪𑙫𑙬𑚹𑜼𑜽𑜾𑠻𑥄𑥅𑥆𑧢𑨿𑩀𑩁𑩂𑩃𑩄𑩅𑩆𑪚𑪛𑪜𑪞𑪟𑪠𑪡𑪢𑬀𑬁𑬂𑬃𑬄𑬅𑬆𑬇𑬈𑬉𑱁𑱂𑱃𑱄𑱅𑱰𑱱𑻷𑻸𑽃𑽄𑽅𑽆𑽇𑽈𑽉𑽊𑽋𑽌𑽍𑽎𑽏𑿿𒑰𒑱𒑲𒑳𒑴𒿱𒿲𖩮𖩯𖫵𖬷𖬸𖬹𖬺𖬻𖭄𖺗𖺘𖺙𖺚𖿢𛲟𝪇𝪈𝪉𝪊𝪋𞥞𞥟"""
+# NOTE: This list diverges slightly from the raw list, since []\ must be escaped
+# The [] need to be escaped to avoid prematurely closing the regex character class
+# The \ needs to be escaped to be considered as a raw \
+# https://www.compart.com/en/unicode/category
+# https://unicode.org/Public/UNIDATA/UnicodeData.txt
 """Commonly occurring strings which are some kind of valid Toki Pona or external token"""
 ALLOWABLES = {
     "cw",  # Content Warning

sonatoki/ilo.py CHANGED Viewed

@@ -1,5 +1,4 @@
 # STL
-import logging
 from typing import List, Type, Tuple
 # LOCAL
@@ -9,8 +8,6 @@ from sonatoki.Cleaners import Cleaner
 from sonatoki.Tokenizers import Tokenizer
 from sonatoki.Preprocessors import Preprocessor
-LOG = logging.getLogger(__name__)
 class Ilo:
     __preprocessors: List[Type[Preprocessor]]
@@ -20,7 +17,6 @@ class Ilo:
     __scoring_filters: List[Type[Filter]]
     __scorer: Type[Scorer]
     __passing_score: Number
-    logging_threshold: Number = -1
     def __init__(
         self,
@@ -104,14 +100,6 @@ class Ilo:
         score = self.score_tokens(cleaned)
         result = score >= self.__passing_score
-        if score <= self.logging_threshold:
-            LOG.debug("msg: %.2f  %s", score, repr(message))
-            LOG.debug("preproc:   %s", repr(preprocessed))
-            LOG.debug("tokenized: %s", tokenized)
-            LOG.debug("filtered:  %s", filtered)
-            LOG.debug("cleaned:   %s", cleaned)
-        # TODO: Move to each function? Loses ability to control when logging occurs by threshold
         return preprocessed, tokenized, filtered, cleaned, score, result
     def is_toki_pona(self, message: str) -> bool:

{sonatoki-0.1.4.dist-info → sonatoki-0.1.5.dist-info}/METADATA RENAMED Viewed

@@ -1,6 +1,6 @@
 Metadata-Version: 2.1
 Name: sonatoki
-Version: 0.1.4
+Version: 0.1.5
 Summary: ilo li moku e toki li pana e sona ni: ni li toki ala toki pona?
 Author-Email: "jan Kekan San (@gregdan3)" <gregory.danielson3@gmail.com>
 License: AGPL-3.0-or-later

sonatoki-0.1.5.dist-info/RECORD ADDED Viewed

@@ -0,0 +1,16 @@
+sonatoki-0.1.5.dist-info/METADATA,sha256=wJBa9CKSni9dcfQGQZp_-FSfANJHmTfZdYSCMH0Wolg,5225
+sonatoki-0.1.5.dist-info/WHEEL,sha256=vnE8JVcI2Wz7GRKorsPArnBdnW2SWKWGow5gu5tHlRU,90
+sonatoki-0.1.5.dist-info/licenses/LICENSE,sha256=DZak_2itbUtvHzD3E7GNUYSRK6jdOJ-GqncQ2weavLA,34523
+sonatoki/Cleaners.py,sha256=gTZ9dSsnvKVUtxM_ECSZ-_2heh--nD5A9dCQR1ATb1c,1160
+sonatoki/Configs.py,sha256=iY6Lyn1rMi7iF0M62yx0ET4pEb35-QAd1FS0tkyUfSc,1935
+sonatoki/Filters.py,sha256=xanTOKxasW_2OpQXsnk5pzbM1FmG8y46pakuTfMz9Hw,4470
+sonatoki/Preprocessors.py,sha256=FqBcHirsXV_91mj99ju9AnbsHaCFctSnuz_vca9ckSY,4441
+sonatoki/Scorers.py,sha256=w5p4qPzpEhR-xHaOXzqulN01OKtLcSPsTrgKyMhfAaQ,3658
+sonatoki/Tokenizers.py,sha256=Wqyf36d1OST7sVrANlLixDIOYWSctD5rVg1_MlaPrCw,2310
+sonatoki/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
+sonatoki/__main__.py,sha256=6xc-wIrrFo9wTyn4zRQNAmqwmJBtVvCMwV-CrM-hueA,82
+sonatoki/constants.py,sha256=LZVI0322kTsqwrGQIJfgSl6IIZS-ste1rd9isRzcNqM,4883
+sonatoki/ilo.py,sha256=yyLgNPI0Hmb4f1BzX6IRHr11FPChfL2xDR_9odlr8_8,3849
+sonatoki/linku.json,sha256=B5KNdhyM5UEfMciROgh1ECHr3i-ASBeMvwrkzNJX47c,271013
+sonatoki/sandbox.json,sha256=hx6LRsfvmmTtqXcXIyCsfSaGK3DZ-GCdbM8xhZQBHoA,77650
+sonatoki-0.1.5.dist-info/RECORD,,

sonatoki-0.1.4.dist-info/RECORD DELETED Viewed

@@ -1,16 +0,0 @@
-sonatoki-0.1.4.dist-info/METADATA,sha256=cK_EyYXPeY4rm9Plcre-i_DbPJZD06572cYQEIUQ804,5225
-sonatoki-0.1.4.dist-info/WHEEL,sha256=vnE8JVcI2Wz7GRKorsPArnBdnW2SWKWGow5gu5tHlRU,90
-sonatoki-0.1.4.dist-info/licenses/LICENSE,sha256=DZak_2itbUtvHzD3E7GNUYSRK6jdOJ-GqncQ2weavLA,34523
-sonatoki/Cleaners.py,sha256=gTZ9dSsnvKVUtxM_ECSZ-_2heh--nD5A9dCQR1ATb1c,1160
-sonatoki/Configs.py,sha256=iY6Lyn1rMi7iF0M62yx0ET4pEb35-QAd1FS0tkyUfSc,1935
-sonatoki/Filters.py,sha256=dL3XgH62OrVVvc8b6dtR5-JZmErVF4bl7ultAoHHqpo,4190
-sonatoki/Preprocessors.py,sha256=h2sX6nJIIOPotwHL0476VQe4KxERlD_F6nrvxDyuaTs,4205
-sonatoki/Scorers.py,sha256=V293DBiupBiujzuc4yMrKOAiuNTLltIsiCzIAlLeokA,4129
-sonatoki/Tokenizers.py,sha256=fvqxpubs2F63va2RzZKZQhZbFnVaC_9haXIA9Mqznis,1942
-sonatoki/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
-sonatoki/__main__.py,sha256=6xc-wIrrFo9wTyn4zRQNAmqwmJBtVvCMwV-CrM-hueA,82
-sonatoki/constants.py,sha256=m0Z4At6MfbqZRio2glT3J3zT9x_itcWZBT_G82mpaVc,1647
-sonatoki/ilo.py,sha256=oN14iYFKxgjFjjOslgqBrMaIgpnvS5gO6MscbS0JS5A,4343
-sonatoki/linku.json,sha256=B5KNdhyM5UEfMciROgh1ECHr3i-ASBeMvwrkzNJX47c,271013
-sonatoki/sandbox.json,sha256=hx6LRsfvmmTtqXcXIyCsfSaGK3DZ-GCdbM8xhZQBHoA,77650
-sonatoki-0.1.4.dist-info/RECORD,,

{sonatoki-0.1.4.dist-info → sonatoki-0.1.5.dist-info}/WHEEL RENAMED Viewed

File without changes

{sonatoki-0.1.4.dist-info → sonatoki-0.1.5.dist-info}/licenses/LICENSE RENAMED Viewed

File without changes

sonatoki 0.1.4__py3-none-any.whl → 0.1.5__py3-none-any.whl

sonatoki 0.1.4py3-none-any.whl → 0.1.5py3-none-any.whl