acuity-framework 3.0.0__tar.gz → 3.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/PKG-INFO +1 -1
  2. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/utils.py +48 -0
  3. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/verification/bplo.py +2 -2
  4. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity_framework.egg-info/PKG-INFO +1 -1
  5. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/pyproject.toml +1 -1
  6. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/setup.py +1 -1
  7. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/tests/test_verification.py +28 -1
  8. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/LICENSE +0 -0
  9. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/README.md +0 -0
  10. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/__init__.py +0 -0
  11. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/config.py +0 -0
  12. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/__init__.py +0 -0
  13. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/interfaces.py +0 -0
  14. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/ner_crf.py +0 -0
  15. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/ner_transformer.py +0 -0
  16. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/pipeline.py +0 -0
  17. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/postprocessing.py +0 -0
  18. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/preprocessing.py +0 -0
  19. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/extraction/rules.py +0 -0
  20. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/recommendation/__init__.py +0 -0
  21. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/recommendation/engine.py +0 -0
  22. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/recommendation/interfaces.py +0 -0
  23. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/recommendation/proximity.py +0 -0
  24. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/recommendation/ranker.py +0 -0
  25. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/recommendation/similarity.py +0 -0
  26. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/recommendation/vectorizer.py +0 -0
  27. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/scraper/__init__.py +0 -0
  28. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/scraper/interfaces.py +0 -0
  29. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/scraper/scraper.py +0 -0
  30. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/scraper/utils.py +0 -0
  31. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity/verification/__init__.py +0 -0
  32. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity_framework.egg-info/SOURCES.txt +0 -0
  33. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity_framework.egg-info/dependency_links.txt +0 -0
  34. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity_framework.egg-info/requires.txt +0 -0
  35. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/acuity_framework.egg-info/top_level.txt +0 -0
  36. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/setup.cfg +0 -0
  37. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/tests/test_extraction.py +0 -0
  38. {acuity_framework-3.0.0 → acuity_framework-3.1.0}/tests/test_recommendation.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: acuity-framework
3
- Version: 3.0.0
3
+ Version: 3.1.0
4
4
  Summary: ACUITY — A machine learning framework for extracting, verifying, and recommending local micro-enterprise profiles from unstructured community posts.
5
5
  Author: ACUITY Research Team
6
6
  License: MIT
@@ -82,3 +82,51 @@ def levenshtein_details(s1: str, s2: str) -> dict:
82
82
  return {"score": 1.0, "edits": 0, "max_len": 0}
83
83
  edits = distance[len(s1)][len(s2)]
84
84
  return {"score": 1.0 - (edits / max_len), "edits": edits, "max_len": max_len}
85
+
86
+ import re
87
+
88
+ def _tokenize(s: str) -> list[str]:
89
+ return re.findall(r'\w+', str(s).lower())
90
+
91
+ def token_sort_ratio(s1: str, s2: str) -> float:
92
+ t1 = _tokenize(s1)
93
+ t2 = _tokenize(s2)
94
+ t1.sort()
95
+ t2.sort()
96
+ return levenshtein_ratio(' '.join(t1), ' '.join(t2))
97
+
98
+ def token_set_ratio(s1: str, s2: str) -> float:
99
+ t1 = set(_tokenize(s1))
100
+ t2 = set(_tokenize(s2))
101
+
102
+ intersection = sorted(list(t1.intersection(t2)))
103
+ diff1 = sorted(list(t1.difference(t2)))
104
+ diff2 = sorted(list(t2.difference(t1)))
105
+
106
+ str_intersection = ' '.join(intersection)
107
+ str1 = ' '.join(intersection + diff1).strip()
108
+ str2 = ' '.join(intersection + diff2).strip()
109
+
110
+ score1 = levenshtein_ratio(str1, str2)
111
+ score2 = levenshtein_ratio(str_intersection, str1) if str_intersection else 0.0
112
+ score3 = levenshtein_ratio(str_intersection, str2) if str_intersection else 0.0
113
+
114
+ return max(score1, score2, score3)
115
+
116
+ def hybrid_fuzzy_match(s1: str, s2: str) -> float:
117
+ if not s1 or not s2:
118
+ return 0.0
119
+
120
+ plain_score = levenshtein_ratio(s1.lower(), s2.lower())
121
+ sort_score = token_sort_ratio(s1, s2)
122
+ set_score = token_set_ratio(s1, s2)
123
+
124
+ # Apply penalty to Token-Set if length disparity is massive (to prevent short acronym false positives)
125
+ len1, len2 = len(s1), len(s2)
126
+ if len1 > 0 and len2 > 0:
127
+ ratio = min(len1, len2) / max(len1, len2)
128
+ if ratio < 0.35:
129
+ # Heavily penalize the Token-Set score
130
+ set_score = set_score * ratio
131
+
132
+ return max(plain_score, sort_score, set_score)
@@ -14,7 +14,7 @@ import csv
14
14
  from typing import Any
15
15
 
16
16
  from ..config import AcuityConfig
17
- from ..utils import levenshtein_ratio
17
+ from ..utils import levenshtein_ratio, hybrid_fuzzy_match
18
18
 
19
19
 
20
20
  class BPLOVerifier:
@@ -80,7 +80,7 @@ class BPLOVerifier:
80
80
  if not bplo_name:
81
81
  continue
82
82
 
83
- score = levenshtein_ratio(name_lower, bplo_name)
83
+ score = hybrid_fuzzy_match(name_lower, bplo_name)
84
84
  if score > best_score:
85
85
  best_score = score
86
86
  best_match = entry
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: acuity-framework
3
- Version: 3.0.0
3
+ Version: 3.1.0
4
4
  Summary: ACUITY — A machine learning framework for extracting, verifying, and recommending local micro-enterprise profiles from unstructured community posts.
5
5
  Author: ACUITY Research Team
6
6
  License: MIT
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "acuity-framework"
7
- version = "3.0.0"
7
+ version = "3.1.0"
8
8
  description = "ACUITY — A machine learning framework for extracting, verifying, and recommending local micro-enterprise profiles from unstructured community posts."
9
9
  readme = "README.md"
10
10
  license = {text = "MIT"}
@@ -84,7 +84,7 @@ class AcuityBuild(build_py):
84
84
  def run(self):
85
85
  if _is_fun_enabled():
86
86
  print(_color(ANSI_BOLD + BANNER, ANSI_CYAN))
87
- print(_color(" ACUITY installer v1.0.0 — extracting local visibility...", ANSI_YELLOW))
87
+ print(_color(" ACUITY installer v3.1.0 — extracting local visibility...", ANSI_YELLOW))
88
88
  print()
89
89
  _animate()
90
90
  print(_color(" Installing ACUITY Framework...", ANSI_YELLOW))
@@ -4,7 +4,7 @@ Tests for the ACUITY Verification module.
4
4
  import pytest
5
5
 
6
6
  from acuity.verification import BPLOVerifier
7
- from acuity.utils import levenshtein_ratio, levenshtein_details
7
+ from acuity.utils import levenshtein_ratio, levenshtein_details, token_sort_ratio, token_set_ratio, hybrid_fuzzy_match
8
8
  from acuity.config import AcuityConfig
9
9
 
10
10
 
@@ -104,3 +104,30 @@ class TestBPLOVerifier:
104
104
  ])
105
105
  result = verifier.verify("Test Biz")
106
106
  assert result["status"] == "Verified"
107
+
108
+ # ── Hybrid Match Tests ─────────────────────────────────────────────────────
109
+
110
+ class TestHybridFuzzyMatch:
111
+ def test_token_sort_ratio(self):
112
+ # Order shouldn't matter
113
+ assert token_sort_ratio("bakeshop juan", "juan bakeshop") == 1.0
114
+ assert token_sort_ratio("juan bakeshop", "bakeshop juan") == 1.0
115
+
116
+ def test_token_set_ratio(self):
117
+ # Extra words shouldn't ruin the score completely
118
+ assert token_set_ratio("juan bakeshop in mamatid", "juan bakeshop") == 1.0
119
+
120
+ def test_hybrid_match_takes_max(self):
121
+ plain = levenshtein_ratio("bakeshop juan", "juan bakeshop") # Will be low
122
+ sort = token_sort_ratio("bakeshop juan", "juan bakeshop") # Will be 1.0
123
+
124
+ hybrid = hybrid_fuzzy_match("bakeshop juan", "juan bakeshop")
125
+ assert hybrid == 1.0
126
+ assert hybrid > plain
127
+
128
+ def test_hybrid_match_penalty(self):
129
+ # "jb" is an acronym for "juan bakeshop". The length ratio is 2 / 13 = 0.15 (which is < 0.35).
130
+ # Token set ratio might normally score it too high if it thinks they share tokens,
131
+ # but with penalty, it should be lowered to avoid false positives.
132
+ score = hybrid_fuzzy_match("jb", "juan bakeshop in the city")
133
+ assert score < 0.5