acuity-framework 3.0.0__tar.gz → 3.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/PKG-INFO +1 -1
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/utils.py +48 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/verification/bplo.py +2 -2
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity_framework.egg-info/PKG-INFO +1 -1
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/pyproject.toml +1 -1
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/setup.py +1 -1
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/tests/test_verification.py +28 -1
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/LICENSE +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/README.md +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/__init__.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/config.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/__init__.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/interfaces.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/ner_crf.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/ner_transformer.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/pipeline.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/postprocessing.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/preprocessing.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/rules.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/recommendation/__init__.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/recommendation/engine.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/recommendation/interfaces.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/recommendation/proximity.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/recommendation/ranker.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/recommendation/similarity.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/recommendation/vectorizer.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/scraper/__init__.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/scraper/interfaces.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/scraper/scraper.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/scraper/utils.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/verification/__init__.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity_framework.egg-info/SOURCES.txt +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity_framework.egg-info/dependency_links.txt +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity_framework.egg-info/requires.txt +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity_framework.egg-info/top_level.txt +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/setup.cfg +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/tests/test_extraction.py +0 -0
- {acuity_framework-3.0.0 → acuity_framework-3.1.0}/tests/test_recommendation.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: acuity-framework
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.1.0
|
|
4
4
|
Summary: ACUITY — A machine learning framework for extracting, verifying, and recommending local micro-enterprise profiles from unstructured community posts.
|
|
5
5
|
Author: ACUITY Research Team
|
|
6
6
|
License: MIT
|
|
@@ -82,3 +82,51 @@ def levenshtein_details(s1: str, s2: str) -> dict:
|
|
|
82
82
|
return {"score": 1.0, "edits": 0, "max_len": 0}
|
|
83
83
|
edits = distance[len(s1)][len(s2)]
|
|
84
84
|
return {"score": 1.0 - (edits / max_len), "edits": edits, "max_len": max_len}
|
|
85
|
+
|
|
86
|
+
import re
|
|
87
|
+
|
|
88
|
+
def _tokenize(s: str) -> list[str]:
|
|
89
|
+
return re.findall(r'\w+', str(s).lower())
|
|
90
|
+
|
|
91
|
+
def token_sort_ratio(s1: str, s2: str) -> float:
|
|
92
|
+
t1 = _tokenize(s1)
|
|
93
|
+
t2 = _tokenize(s2)
|
|
94
|
+
t1.sort()
|
|
95
|
+
t2.sort()
|
|
96
|
+
return levenshtein_ratio(' '.join(t1), ' '.join(t2))
|
|
97
|
+
|
|
98
|
+
def token_set_ratio(s1: str, s2: str) -> float:
|
|
99
|
+
t1 = set(_tokenize(s1))
|
|
100
|
+
t2 = set(_tokenize(s2))
|
|
101
|
+
|
|
102
|
+
intersection = sorted(list(t1.intersection(t2)))
|
|
103
|
+
diff1 = sorted(list(t1.difference(t2)))
|
|
104
|
+
diff2 = sorted(list(t2.difference(t1)))
|
|
105
|
+
|
|
106
|
+
str_intersection = ' '.join(intersection)
|
|
107
|
+
str1 = ' '.join(intersection + diff1).strip()
|
|
108
|
+
str2 = ' '.join(intersection + diff2).strip()
|
|
109
|
+
|
|
110
|
+
score1 = levenshtein_ratio(str1, str2)
|
|
111
|
+
score2 = levenshtein_ratio(str_intersection, str1) if str_intersection else 0.0
|
|
112
|
+
score3 = levenshtein_ratio(str_intersection, str2) if str_intersection else 0.0
|
|
113
|
+
|
|
114
|
+
return max(score1, score2, score3)
|
|
115
|
+
|
|
116
|
+
def hybrid_fuzzy_match(s1: str, s2: str) -> float:
|
|
117
|
+
if not s1 or not s2:
|
|
118
|
+
return 0.0
|
|
119
|
+
|
|
120
|
+
plain_score = levenshtein_ratio(s1.lower(), s2.lower())
|
|
121
|
+
sort_score = token_sort_ratio(s1, s2)
|
|
122
|
+
set_score = token_set_ratio(s1, s2)
|
|
123
|
+
|
|
124
|
+
# Apply penalty to Token-Set if length disparity is massive (to prevent short acronym false positives)
|
|
125
|
+
len1, len2 = len(s1), len(s2)
|
|
126
|
+
if len1 > 0 and len2 > 0:
|
|
127
|
+
ratio = min(len1, len2) / max(len1, len2)
|
|
128
|
+
if ratio < 0.35:
|
|
129
|
+
# Heavily penalize the Token-Set score
|
|
130
|
+
set_score = set_score * ratio
|
|
131
|
+
|
|
132
|
+
return max(plain_score, sort_score, set_score)
|
|
@@ -14,7 +14,7 @@ import csv
|
|
|
14
14
|
from typing import Any
|
|
15
15
|
|
|
16
16
|
from ..config import AcuityConfig
|
|
17
|
-
from ..utils import levenshtein_ratio
|
|
17
|
+
from ..utils import levenshtein_ratio, hybrid_fuzzy_match
|
|
18
18
|
|
|
19
19
|
|
|
20
20
|
class BPLOVerifier:
|
|
@@ -80,7 +80,7 @@ class BPLOVerifier:
|
|
|
80
80
|
if not bplo_name:
|
|
81
81
|
continue
|
|
82
82
|
|
|
83
|
-
score =
|
|
83
|
+
score = hybrid_fuzzy_match(name_lower, bplo_name)
|
|
84
84
|
if score > best_score:
|
|
85
85
|
best_score = score
|
|
86
86
|
best_match = entry
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: acuity-framework
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.1.0
|
|
4
4
|
Summary: ACUITY — A machine learning framework for extracting, verifying, and recommending local micro-enterprise profiles from unstructured community posts.
|
|
5
5
|
Author: ACUITY Research Team
|
|
6
6
|
License: MIT
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "acuity-framework"
|
|
7
|
-
version = "3.
|
|
7
|
+
version = "3.1.0"
|
|
8
8
|
description = "ACUITY — A machine learning framework for extracting, verifying, and recommending local micro-enterprise profiles from unstructured community posts."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = {text = "MIT"}
|
|
@@ -84,7 +84,7 @@ class AcuityBuild(build_py):
|
|
|
84
84
|
def run(self):
|
|
85
85
|
if _is_fun_enabled():
|
|
86
86
|
print(_color(ANSI_BOLD + BANNER, ANSI_CYAN))
|
|
87
|
-
print(_color(" ACUITY installer
|
|
87
|
+
print(_color(" ACUITY installer v3.1.0 — extracting local visibility...", ANSI_YELLOW))
|
|
88
88
|
print()
|
|
89
89
|
_animate()
|
|
90
90
|
print(_color(" Installing ACUITY Framework...", ANSI_YELLOW))
|
|
@@ -4,7 +4,7 @@ Tests for the ACUITY Verification module.
|
|
|
4
4
|
import pytest
|
|
5
5
|
|
|
6
6
|
from acuity.verification import BPLOVerifier
|
|
7
|
-
from acuity.utils import levenshtein_ratio, levenshtein_details
|
|
7
|
+
from acuity.utils import levenshtein_ratio, levenshtein_details, token_sort_ratio, token_set_ratio, hybrid_fuzzy_match
|
|
8
8
|
from acuity.config import AcuityConfig
|
|
9
9
|
|
|
10
10
|
|
|
@@ -104,3 +104,30 @@ class TestBPLOVerifier:
|
|
|
104
104
|
])
|
|
105
105
|
result = verifier.verify("Test Biz")
|
|
106
106
|
assert result["status"] == "Verified"
|
|
107
|
+
|
|
108
|
+
# ── Hybrid Match Tests ─────────────────────────────────────────────────────
|
|
109
|
+
|
|
110
|
+
class TestHybridFuzzyMatch:
|
|
111
|
+
def test_token_sort_ratio(self):
|
|
112
|
+
# Order shouldn't matter
|
|
113
|
+
assert token_sort_ratio("bakeshop juan", "juan bakeshop") == 1.0
|
|
114
|
+
assert token_sort_ratio("juan bakeshop", "bakeshop juan") == 1.0
|
|
115
|
+
|
|
116
|
+
def test_token_set_ratio(self):
|
|
117
|
+
# Extra words shouldn't ruin the score completely
|
|
118
|
+
assert token_set_ratio("juan bakeshop in mamatid", "juan bakeshop") == 1.0
|
|
119
|
+
|
|
120
|
+
def test_hybrid_match_takes_max(self):
|
|
121
|
+
plain = levenshtein_ratio("bakeshop juan", "juan bakeshop") # Will be low
|
|
122
|
+
sort = token_sort_ratio("bakeshop juan", "juan bakeshop") # Will be 1.0
|
|
123
|
+
|
|
124
|
+
hybrid = hybrid_fuzzy_match("bakeshop juan", "juan bakeshop")
|
|
125
|
+
assert hybrid == 1.0
|
|
126
|
+
assert hybrid > plain
|
|
127
|
+
|
|
128
|
+
def test_hybrid_match_penalty(self):
|
|
129
|
+
# "jb" is an acronym for "juan bakeshop". The length ratio is 2 / 13 = 0.15 (which is < 0.35).
|
|
130
|
+
# Token set ratio might normally score it too high if it thinks they share tokens,
|
|
131
|
+
# but with penalty, it should be lowered to avoid false positives.
|
|
132
|
+
score = hybrid_fuzzy_match("jb", "juan bakeshop in the city")
|
|
133
|
+
assert score < 0.5
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity_framework.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|