avenqor-ai 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- avenqor_ai-0.1.0/LICENSE +21 -0
- avenqor_ai-0.1.0/PKG-INFO +60 -0
- avenqor_ai-0.1.0/README.md +43 -0
- avenqor_ai-0.1.0/pyproject.toml +28 -0
- avenqor_ai-0.1.0/setup.cfg +4 -0
- avenqor_ai-0.1.0/src/avenqor_ai/__init__.py +19 -0
- avenqor_ai-0.1.0/src/avenqor_ai/core.py +89 -0
- avenqor_ai-0.1.0/src/avenqor_ai.egg-info/PKG-INFO +60 -0
- avenqor_ai-0.1.0/src/avenqor_ai.egg-info/SOURCES.txt +10 -0
- avenqor_ai-0.1.0/src/avenqor_ai.egg-info/dependency_links.txt +1 -0
- avenqor_ai-0.1.0/src/avenqor_ai.egg-info/top_level.txt +1 -0
- avenqor_ai-0.1.0/tests/test_core.py +20 -0
avenqor_ai-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Your Name
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: avenqor-ai
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A simple utility package for text and AI-related helper functions
|
|
5
|
+
Author-email: shubham kumar <shubhamk97251@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/yourusername/avenqor-ai
|
|
8
|
+
Project-URL: Repository, https://github.com/yourusername/avenqor-ai
|
|
9
|
+
Keywords: ai,nlp,text,utility
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Requires-Python: >=3.8
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
License-File: LICENSE
|
|
16
|
+
Dynamic: license-file
|
|
17
|
+
|
|
18
|
+
# avenqor-ai
|
|
19
|
+
|
|
20
|
+
A simple Python utility package for text processing and AI-related helper functions.
|
|
21
|
+
|
|
22
|
+
## Installation
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
pip install avenqor-ai
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## Usage
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
from avenqor_ai import clean_text, count_tokens_approx, chunk_text, similarity_score
|
|
32
|
+
|
|
33
|
+
# Clean messy text
|
|
34
|
+
clean_text("Hello World!!\n\n")
|
|
35
|
+
# -> "Hello World!!"
|
|
36
|
+
|
|
37
|
+
# Estimate token count
|
|
38
|
+
count_tokens_approx("Hello world")
|
|
39
|
+
# -> 2
|
|
40
|
+
|
|
41
|
+
# Split long text into chunks (for LLM context windows)
|
|
42
|
+
chunks = chunk_text("your very long document here...", chunk_size=500, overlap=50)
|
|
43
|
+
|
|
44
|
+
# Compare similarity between two texts
|
|
45
|
+
similarity_score("hello world", "hello there")
|
|
46
|
+
# -> 0.33
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
## Functions
|
|
50
|
+
|
|
51
|
+
| Function | Description |
|
|
52
|
+
|---|---|
|
|
53
|
+
| `clean_text(text)` | Normalizes whitespace and strips text |
|
|
54
|
+
| `count_tokens_approx(text)` | Rough token count estimate |
|
|
55
|
+
| `chunk_text(text, chunk_size, overlap)` | Splits text into overlapping chunks |
|
|
56
|
+
| `similarity_score(text1, text2)` | Jaccard word-overlap similarity (0-1) |
|
|
57
|
+
|
|
58
|
+
## License
|
|
59
|
+
|
|
60
|
+
MIT
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# avenqor-ai
|
|
2
|
+
|
|
3
|
+
A simple Python utility package for text processing and AI-related helper functions.
|
|
4
|
+
|
|
5
|
+
## Installation
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
pip install avenqor-ai
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
## Usage
|
|
12
|
+
|
|
13
|
+
```python
|
|
14
|
+
from avenqor_ai import clean_text, count_tokens_approx, chunk_text, similarity_score
|
|
15
|
+
|
|
16
|
+
# Clean messy text
|
|
17
|
+
clean_text("Hello World!!\n\n")
|
|
18
|
+
# -> "Hello World!!"
|
|
19
|
+
|
|
20
|
+
# Estimate token count
|
|
21
|
+
count_tokens_approx("Hello world")
|
|
22
|
+
# -> 2
|
|
23
|
+
|
|
24
|
+
# Split long text into chunks (for LLM context windows)
|
|
25
|
+
chunks = chunk_text("your very long document here...", chunk_size=500, overlap=50)
|
|
26
|
+
|
|
27
|
+
# Compare similarity between two texts
|
|
28
|
+
similarity_score("hello world", "hello there")
|
|
29
|
+
# -> 0.33
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
## Functions
|
|
33
|
+
|
|
34
|
+
| Function | Description |
|
|
35
|
+
|---|---|
|
|
36
|
+
| `clean_text(text)` | Normalizes whitespace and strips text |
|
|
37
|
+
| `count_tokens_approx(text)` | Rough token count estimate |
|
|
38
|
+
| `chunk_text(text, chunk_size, overlap)` | Splits text into overlapping chunks |
|
|
39
|
+
| `similarity_score(text1, text2)` | Jaccard word-overlap similarity (0-1) |
|
|
40
|
+
|
|
41
|
+
## License
|
|
42
|
+
|
|
43
|
+
MIT
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68.0"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "avenqor-ai"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "A simple utility package for text and AI-related helper functions"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.8"
|
|
11
|
+
license = {text = "MIT"}
|
|
12
|
+
authors = [
|
|
13
|
+
{name = "shubham kumar", email = "shubhamk97251@gmail.com"}
|
|
14
|
+
]
|
|
15
|
+
keywords = ["ai", "nlp", "text", "utility"]
|
|
16
|
+
classifiers = [
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"License :: OSI Approved :: MIT License",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
]
|
|
21
|
+
dependencies = []
|
|
22
|
+
|
|
23
|
+
[project.urls]
|
|
24
|
+
Homepage = "https://github.com/yourusername/avenqor-ai"
|
|
25
|
+
Repository = "https://github.com/yourusername/avenqor-ai"
|
|
26
|
+
|
|
27
|
+
[tool.setuptools.packages.find]
|
|
28
|
+
where = ["src"]
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""
|
|
2
|
+
avenqor-ai
|
|
3
|
+
A simple utility package for text and AI-related helper functions.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from .core import (
|
|
7
|
+
clean_text,
|
|
8
|
+
count_tokens_approx,
|
|
9
|
+
chunk_text,
|
|
10
|
+
similarity_score,
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
__version__ = "0.1.0"
|
|
14
|
+
__all__ = [
|
|
15
|
+
"clean_text",
|
|
16
|
+
"count_tokens_approx",
|
|
17
|
+
"chunk_text",
|
|
18
|
+
"similarity_score",
|
|
19
|
+
]
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""Core functions for avenqor_ai."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def clean_text(text: str) -> str:
|
|
7
|
+
"""
|
|
8
|
+
Clean text by removing extra whitespace, special characters noise,
|
|
9
|
+
and normalizing spacing. Useful before feeding text to an LLM.
|
|
10
|
+
|
|
11
|
+
Example:
|
|
12
|
+
>>> clean_text("Hello World!!\\n\\n")
|
|
13
|
+
'Hello World!!'
|
|
14
|
+
"""
|
|
15
|
+
if not isinstance(text, str):
|
|
16
|
+
raise TypeError("clean_text expects a string")
|
|
17
|
+
|
|
18
|
+
text = text.strip()
|
|
19
|
+
text = re.sub(r"\s+", " ", text)
|
|
20
|
+
return text
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def count_tokens_approx(text: str) -> int:
|
|
24
|
+
"""
|
|
25
|
+
Roughly estimate token count (approx 1 token ≈ 4 characters,
|
|
26
|
+
a common rule of thumb for English text with GPT/Claude-like models).
|
|
27
|
+
|
|
28
|
+
Example:
|
|
29
|
+
>>> count_tokens_approx("Hello world")
|
|
30
|
+
3
|
|
31
|
+
"""
|
|
32
|
+
if not isinstance(text, str):
|
|
33
|
+
raise TypeError("count_tokens_approx expects a string")
|
|
34
|
+
|
|
35
|
+
return max(1, len(text) // 4)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def chunk_text(text: str, chunk_size: int = 500, overlap: int = 50) -> list[str]:
|
|
39
|
+
"""
|
|
40
|
+
Split long text into overlapping chunks — useful for feeding
|
|
41
|
+
long documents into an LLM with a limited context window.
|
|
42
|
+
|
|
43
|
+
Args:
|
|
44
|
+
text: input text
|
|
45
|
+
chunk_size: max characters per chunk
|
|
46
|
+
overlap: number of overlapping characters between chunks
|
|
47
|
+
|
|
48
|
+
Example:
|
|
49
|
+
>>> chunks = chunk_text("a" * 1000, chunk_size=400, overlap=50)
|
|
50
|
+
>>> len(chunks)
|
|
51
|
+
3
|
|
52
|
+
"""
|
|
53
|
+
if not isinstance(text, str):
|
|
54
|
+
raise TypeError("chunk_text expects a string")
|
|
55
|
+
if chunk_size <= 0:
|
|
56
|
+
raise ValueError("chunk_size must be positive")
|
|
57
|
+
if overlap >= chunk_size:
|
|
58
|
+
raise ValueError("overlap must be smaller than chunk_size")
|
|
59
|
+
|
|
60
|
+
chunks = []
|
|
61
|
+
start = 0
|
|
62
|
+
while start < len(text):
|
|
63
|
+
end = start + chunk_size
|
|
64
|
+
chunks.append(text[start:end])
|
|
65
|
+
start += chunk_size - overlap
|
|
66
|
+
return chunks
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def similarity_score(text1: str, text2: str) -> float:
|
|
70
|
+
"""
|
|
71
|
+
Compute a simple word-overlap similarity score (Jaccard similarity)
|
|
72
|
+
between two texts. Returns a float between 0 and 1.
|
|
73
|
+
|
|
74
|
+
Example:
|
|
75
|
+
>>> similarity_score("hello world", "hello there")
|
|
76
|
+
0.33
|
|
77
|
+
"""
|
|
78
|
+
if not isinstance(text1, str) or not isinstance(text2, str):
|
|
79
|
+
raise TypeError("similarity_score expects two strings")
|
|
80
|
+
|
|
81
|
+
set1 = set(text1.lower().split())
|
|
82
|
+
set2 = set(text2.lower().split())
|
|
83
|
+
|
|
84
|
+
if not set1 or not set2:
|
|
85
|
+
return 0.0
|
|
86
|
+
|
|
87
|
+
intersection = set1 & set2
|
|
88
|
+
union = set1 | set2
|
|
89
|
+
return round(len(intersection) / len(union), 2)
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: avenqor-ai
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A simple utility package for text and AI-related helper functions
|
|
5
|
+
Author-email: shubham kumar <shubhamk97251@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/yourusername/avenqor-ai
|
|
8
|
+
Project-URL: Repository, https://github.com/yourusername/avenqor-ai
|
|
9
|
+
Keywords: ai,nlp,text,utility
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Requires-Python: >=3.8
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
License-File: LICENSE
|
|
16
|
+
Dynamic: license-file
|
|
17
|
+
|
|
18
|
+
# avenqor-ai
|
|
19
|
+
|
|
20
|
+
A simple Python utility package for text processing and AI-related helper functions.
|
|
21
|
+
|
|
22
|
+
## Installation
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
pip install avenqor-ai
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## Usage
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
from avenqor_ai import clean_text, count_tokens_approx, chunk_text, similarity_score
|
|
32
|
+
|
|
33
|
+
# Clean messy text
|
|
34
|
+
clean_text("Hello World!!\n\n")
|
|
35
|
+
# -> "Hello World!!"
|
|
36
|
+
|
|
37
|
+
# Estimate token count
|
|
38
|
+
count_tokens_approx("Hello world")
|
|
39
|
+
# -> 2
|
|
40
|
+
|
|
41
|
+
# Split long text into chunks (for LLM context windows)
|
|
42
|
+
chunks = chunk_text("your very long document here...", chunk_size=500, overlap=50)
|
|
43
|
+
|
|
44
|
+
# Compare similarity between two texts
|
|
45
|
+
similarity_score("hello world", "hello there")
|
|
46
|
+
# -> 0.33
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
## Functions
|
|
50
|
+
|
|
51
|
+
| Function | Description |
|
|
52
|
+
|---|---|
|
|
53
|
+
| `clean_text(text)` | Normalizes whitespace and strips text |
|
|
54
|
+
| `count_tokens_approx(text)` | Rough token count estimate |
|
|
55
|
+
| `chunk_text(text, chunk_size, overlap)` | Splits text into overlapping chunks |
|
|
56
|
+
| `similarity_score(text1, text2)` | Jaccard word-overlap similarity (0-1) |
|
|
57
|
+
|
|
58
|
+
## License
|
|
59
|
+
|
|
60
|
+
MIT
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
src/avenqor_ai/__init__.py
|
|
5
|
+
src/avenqor_ai/core.py
|
|
6
|
+
src/avenqor_ai.egg-info/PKG-INFO
|
|
7
|
+
src/avenqor_ai.egg-info/SOURCES.txt
|
|
8
|
+
src/avenqor_ai.egg-info/dependency_links.txt
|
|
9
|
+
src/avenqor_ai.egg-info/top_level.txt
|
|
10
|
+
tests/test_core.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
avenqor_ai
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
from avenqor_ai import clean_text, count_tokens_approx, chunk_text, similarity_score
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
def test_clean_text():
|
|
5
|
+
assert clean_text("Hello World!!\n\n") == "Hello World!!"
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def test_count_tokens_approx():
|
|
9
|
+
assert count_tokens_approx("Hello world") == 2
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def test_chunk_text():
|
|
13
|
+
chunks = chunk_text("a" * 1000, chunk_size=400, overlap=50)
|
|
14
|
+
assert len(chunks) == 3
|
|
15
|
+
assert all(len(c) <= 400 for c in chunks)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def test_similarity_score():
|
|
19
|
+
score = similarity_score("hello world", "hello there")
|
|
20
|
+
assert 0 <= score <= 1
|