acuity-framework 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- acuity/__init__.py +27 -0
- acuity/config.py +47 -0
- acuity/extraction/__init__.py +22 -0
- acuity/extraction/ner_crf.py +164 -0
- acuity/extraction/ner_transformer.py +76 -0
- acuity/extraction/pipeline.py +171 -0
- acuity/extraction/postprocessing.py +74 -0
- acuity/extraction/preprocessing.py +57 -0
- acuity/extraction/rules.py +49 -0
- acuity/recommendation/__init__.py +16 -0
- acuity/recommendation/engine.py +138 -0
- acuity/recommendation/proximity.py +35 -0
- acuity/recommendation/ranker.py +61 -0
- acuity/recommendation/similarity.py +48 -0
- acuity/recommendation/vectorizer.py +174 -0
- acuity/scraper/__init__.py +16 -0
- acuity/scraper/scraper.py +376 -0
- acuity/scraper/utils.py +62 -0
- acuity/utils.py +84 -0
- acuity/verification/__init__.py +16 -0
- acuity/verification/bplo.py +134 -0
- acuity_framework-1.0.0.dist-info/METADATA +255 -0
- acuity_framework-1.0.0.dist-info/RECORD +26 -0
- acuity_framework-1.0.0.dist-info/WHEEL +5 -0
- acuity_framework-1.0.0.dist-info/licenses/LICENSE +21 -0
- acuity_framework-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""
|
|
2
|
+
ACUITY Framework — Recommender Module
|
|
3
|
+
|
|
4
|
+
Recommendation engine using TF-IDF, cosine similarity, and Haversine
|
|
5
|
+
proximity ranking to match user queries with local business profiles.
|
|
6
|
+
|
|
7
|
+
Quick Start:
|
|
8
|
+
>>> from acuity.recommendation import RecommendationEngine
|
|
9
|
+
>>> engine = RecommendationEngine()
|
|
10
|
+
>>> engine.set_profiles([{"name": "Juan's Bakery", "description": "Fresh bread daily"}])
|
|
11
|
+
>>> results = engine.recommend("bakery")
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from .engine import RecommendationEngine
|
|
15
|
+
|
|
16
|
+
__all__ = ["RecommendationEngine"]
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
"""
|
|
2
|
+
ACUITY Framework — Recommendation Engine
|
|
3
|
+
|
|
4
|
+
Main orchestrator that combines textual relevance (TF-IDF + cosine)
|
|
5
|
+
with geographic proximity (Haversine) to produce ranked results.
|
|
6
|
+
|
|
7
|
+
Usage:
|
|
8
|
+
>>> from acuity.recommendation import RecommendationEngine
|
|
9
|
+
>>> engine = RecommendationEngine()
|
|
10
|
+
>>> engine.set_profiles([
|
|
11
|
+
... {"name": "Juan's Bakery", "description": "Fresh bread and pastries daily"},
|
|
12
|
+
... {"name": "JC Auto Repair", "description": "Vulcanizing and oil change"},
|
|
13
|
+
... ])
|
|
14
|
+
>>> results = engine.recommend(query="bakery", user_lat=14.25, user_lon=121.10)
|
|
15
|
+
"""
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import json
|
|
19
|
+
|
|
20
|
+
from .vectorizer import build_tfidf_matrix, transform_query
|
|
21
|
+
from .similarity import compute_cosine_scores
|
|
22
|
+
from .proximity import haversine_distance
|
|
23
|
+
from .ranker import rank_results
|
|
24
|
+
from ..config import AcuityConfig
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class RecommendationEngine:
|
|
28
|
+
"""End-to-end recommendation engine for ACUITY.
|
|
29
|
+
|
|
30
|
+
Combines TF-IDF text similarity with Haversine geographic proximity
|
|
31
|
+
to rank business profiles against user queries.
|
|
32
|
+
|
|
33
|
+
Args:
|
|
34
|
+
config: An ``AcuityConfig`` instance. If ``None``, uses defaults.
|
|
35
|
+
relevance_weight: Override the config's relevance weight.
|
|
36
|
+
proximity_weight: Override the config's proximity weight.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
def __init__(
|
|
40
|
+
self,
|
|
41
|
+
config: AcuityConfig | None = None,
|
|
42
|
+
relevance_weight: float | None = None,
|
|
43
|
+
proximity_weight: float | None = None,
|
|
44
|
+
):
|
|
45
|
+
self.config = config or AcuityConfig()
|
|
46
|
+
self.profiles: list[dict] = []
|
|
47
|
+
self.relevance_weight = relevance_weight if relevance_weight is not None else self.config.relevance_weight
|
|
48
|
+
self.proximity_weight = proximity_weight if proximity_weight is not None else self.config.proximity_weight
|
|
49
|
+
self._tfidf_matrix = None
|
|
50
|
+
self._vectorizer = None
|
|
51
|
+
|
|
52
|
+
def set_profiles(self, profiles: list[dict]) -> None:
|
|
53
|
+
"""Load business profiles from an in-memory list.
|
|
54
|
+
|
|
55
|
+
Accepts profiles with any combination of the following keys:
|
|
56
|
+
``name``, ``business_name``, ``description``, ``categories``, ``services``.
|
|
57
|
+
|
|
58
|
+
Args:
|
|
59
|
+
profiles: List of business profile dictionaries.
|
|
60
|
+
"""
|
|
61
|
+
self.profiles = profiles
|
|
62
|
+
# Build TF-IDF matrix from profile text fields
|
|
63
|
+
texts = []
|
|
64
|
+
for p in self.profiles:
|
|
65
|
+
name = p.get("name") or p.get("business_name") or ""
|
|
66
|
+
desc = p.get("description") or ""
|
|
67
|
+
cats = " ".join(p.get("categories") or [])
|
|
68
|
+
srvs = " ".join(p.get("services") or [])
|
|
69
|
+
|
|
70
|
+
texts.append(f"{name} {desc} {cats} {srvs}")
|
|
71
|
+
self._vectorizer, self._tfidf_matrix = build_tfidf_matrix(texts)
|
|
72
|
+
print(f"Loaded {len(self.profiles)} profiles. TF-IDF matrix built.")
|
|
73
|
+
|
|
74
|
+
def load_profiles(self, path: str) -> None:
|
|
75
|
+
"""Load business profiles from a JSON file.
|
|
76
|
+
|
|
77
|
+
Args:
|
|
78
|
+
path: Path to a JSON file containing a list of profile dicts.
|
|
79
|
+
"""
|
|
80
|
+
with open(path, "r", encoding="utf-8") as f:
|
|
81
|
+
data = json.load(f)
|
|
82
|
+
self.set_profiles(data)
|
|
83
|
+
|
|
84
|
+
def recommend(
|
|
85
|
+
self,
|
|
86
|
+
query: str,
|
|
87
|
+
user_lat: float | None = None,
|
|
88
|
+
user_lon: float | None = None,
|
|
89
|
+
top_k: int | None = None,
|
|
90
|
+
) -> list[dict]:
|
|
91
|
+
"""Return top-k ranked business profiles for *query*.
|
|
92
|
+
|
|
93
|
+
Args:
|
|
94
|
+
query: User's search query text.
|
|
95
|
+
user_lat: User's latitude (optional, for proximity scoring).
|
|
96
|
+
user_lon: User's longitude (optional, for proximity scoring).
|
|
97
|
+
top_k: Number of results to return. Defaults to ``config.default_top_k``.
|
|
98
|
+
|
|
99
|
+
Returns:
|
|
100
|
+
List of profile dicts augmented with ``relevance_score``,
|
|
101
|
+
``distance_km``, ``proximity_score``, and ``final_score``.
|
|
102
|
+
"""
|
|
103
|
+
if top_k is None:
|
|
104
|
+
top_k = self.config.default_top_k
|
|
105
|
+
|
|
106
|
+
if not self.profiles or self._tfidf_matrix is None or self._vectorizer is None:
|
|
107
|
+
return []
|
|
108
|
+
|
|
109
|
+
# Textual relevance
|
|
110
|
+
query_vec = transform_query(self._vectorizer, query)
|
|
111
|
+
cosine_scores = compute_cosine_scores(self._tfidf_matrix, query_vec)
|
|
112
|
+
|
|
113
|
+
# Proximity (if user location provided)
|
|
114
|
+
distances: list[float | None] = []
|
|
115
|
+
if user_lat is not None and user_lon is not None:
|
|
116
|
+
for profile in self.profiles:
|
|
117
|
+
biz_lat = profile.get("latitude")
|
|
118
|
+
biz_lon = profile.get("longitude")
|
|
119
|
+
if biz_lat is not None and biz_lon is not None:
|
|
120
|
+
distances.append(
|
|
121
|
+
haversine_distance(user_lat, user_lon, biz_lat, biz_lon)
|
|
122
|
+
)
|
|
123
|
+
else:
|
|
124
|
+
distances.append(None)
|
|
125
|
+
else:
|
|
126
|
+
distances = [None] * len(self.profiles)
|
|
127
|
+
|
|
128
|
+
# Combine & rank
|
|
129
|
+
results = rank_results(
|
|
130
|
+
profiles=self.profiles,
|
|
131
|
+
cosine_scores=cosine_scores,
|
|
132
|
+
distances=distances,
|
|
133
|
+
relevance_weight=self.relevance_weight,
|
|
134
|
+
proximity_weight=self.proximity_weight,
|
|
135
|
+
top_k=top_k,
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
return results
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""
|
|
2
|
+
ACUITY Framework — Haversine Proximity
|
|
3
|
+
|
|
4
|
+
Calculates geographic distance between two coordinate pairs using the
|
|
5
|
+
Haversine formula for great-circle distance on a sphere.
|
|
6
|
+
Used for proximity-based ranking of businesses relative to the user.
|
|
7
|
+
"""
|
|
8
|
+
import math
|
|
9
|
+
|
|
10
|
+
# Earth's mean radius in kilometres
|
|
11
|
+
EARTH_RADIUS_KM = 6371.0
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def haversine_distance(
|
|
15
|
+
lat1: float, lon1: float,
|
|
16
|
+
lat2: float, lon2: float,
|
|
17
|
+
) -> float:
|
|
18
|
+
"""Return the great-circle distance (km) between two lat/lon points.
|
|
19
|
+
|
|
20
|
+
Args:
|
|
21
|
+
lat1, lon1: User's coordinates (decimal degrees).
|
|
22
|
+
lat2, lon2: Business's coordinates (decimal degrees).
|
|
23
|
+
|
|
24
|
+
Returns:
|
|
25
|
+
Distance in kilometres.
|
|
26
|
+
"""
|
|
27
|
+
lat1, lon1, lat2, lon2 = map(math.radians, [lat1, lon1, lat2, lon2])
|
|
28
|
+
|
|
29
|
+
dlat = lat2 - lat1
|
|
30
|
+
dlon = lon2 - lon1
|
|
31
|
+
|
|
32
|
+
a = math.sin(dlat / 2) ** 2 + math.cos(lat1) * math.cos(lat2) * math.sin(dlon / 2) ** 2
|
|
33
|
+
c = 2 * math.asin(math.sqrt(a))
|
|
34
|
+
|
|
35
|
+
return EARTH_RADIUS_KM * c
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""
|
|
2
|
+
ACUITY Framework — Combined Ranker
|
|
3
|
+
|
|
4
|
+
Merges textual relevance scores and proximity distances into a single
|
|
5
|
+
ranking, using a configurable weighting scheme.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def rank_results(
|
|
11
|
+
profiles: list[dict],
|
|
12
|
+
cosine_scores: list[float],
|
|
13
|
+
distances: list[float | None],
|
|
14
|
+
relevance_weight: float = 0.6,
|
|
15
|
+
proximity_weight: float = 0.4,
|
|
16
|
+
top_k: int = 10,
|
|
17
|
+
) -> list[dict]:
|
|
18
|
+
"""Produce a ranked list of business profiles.
|
|
19
|
+
|
|
20
|
+
The final score is a weighted combination of:
|
|
21
|
+
- Textual relevance (cosine similarity, higher = better)
|
|
22
|
+
- Proximity score (inverse distance, closer = better)
|
|
23
|
+
|
|
24
|
+
Args:
|
|
25
|
+
profiles: List of business profile dicts.
|
|
26
|
+
cosine_scores: List of cosine similarity values per profile.
|
|
27
|
+
distances: List of distances in km (``None`` if unavailable).
|
|
28
|
+
relevance_weight: Weight for textual relevance [0, 1].
|
|
29
|
+
proximity_weight: Weight for proximity [0, 1].
|
|
30
|
+
top_k: How many results to return.
|
|
31
|
+
|
|
32
|
+
Returns:
|
|
33
|
+
Sorted list of profile dicts with added scoring fields:
|
|
34
|
+
``relevance_score``, ``distance_km``, ``proximity_score``, ``final_score``.
|
|
35
|
+
"""
|
|
36
|
+
scored: list[dict] = []
|
|
37
|
+
|
|
38
|
+
for i, profile in enumerate(profiles):
|
|
39
|
+
relevance = float(cosine_scores[i])
|
|
40
|
+
|
|
41
|
+
dist = distances[i]
|
|
42
|
+
if dist is not None:
|
|
43
|
+
# Inverse distance proximity score (1.0 = 0 km away)
|
|
44
|
+
proximity_score = 1.0 / (1.0 + dist)
|
|
45
|
+
else:
|
|
46
|
+
proximity_score = 0.0
|
|
47
|
+
|
|
48
|
+
final_score = (relevance_weight * relevance) + (proximity_weight * proximity_score)
|
|
49
|
+
|
|
50
|
+
scored.append({
|
|
51
|
+
**profile,
|
|
52
|
+
"relevance_score": round(float(relevance), 4),
|
|
53
|
+
"distance_km": round(float(dist), 2) if dist is not None else None,
|
|
54
|
+
"proximity_score": round(float(proximity_score), 4),
|
|
55
|
+
"final_score": round(float(final_score), 4),
|
|
56
|
+
})
|
|
57
|
+
|
|
58
|
+
# Sort descending by final score
|
|
59
|
+
scored.sort(key=lambda x: x["final_score"], reverse=True)
|
|
60
|
+
|
|
61
|
+
return list(scored[:top_k])
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
"""
|
|
2
|
+
ACUITY Framework — Cosine Similarity
|
|
3
|
+
|
|
4
|
+
Computes relevance scores between a user query vector and the
|
|
5
|
+
TF-IDF business profile vectors using pure Python.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import math
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def compute_cosine_scores(
|
|
13
|
+
tfidf_matrix: list[dict[str, float]],
|
|
14
|
+
query_vec: list[dict[str, float]],
|
|
15
|
+
) -> list[float]:
|
|
16
|
+
"""Return a list of cosine similarity scores for every profile.
|
|
17
|
+
|
|
18
|
+
Args:
|
|
19
|
+
tfidf_matrix: List of sparse dictionary vectors representing profiles.
|
|
20
|
+
query_vec: List containing one sparse dictionary vector for the user query.
|
|
21
|
+
|
|
22
|
+
Returns:
|
|
23
|
+
List of similarity scores in [0, 1].
|
|
24
|
+
"""
|
|
25
|
+
if not query_vec:
|
|
26
|
+
return [0.0] * len(tfidf_matrix)
|
|
27
|
+
|
|
28
|
+
query = query_vec[0]
|
|
29
|
+
query_mag = math.sqrt(sum(val ** 2 for val in query.values()))
|
|
30
|
+
|
|
31
|
+
if query_mag == 0:
|
|
32
|
+
return [0.0] * len(tfidf_matrix)
|
|
33
|
+
|
|
34
|
+
scores = []
|
|
35
|
+
for vector in tfidf_matrix:
|
|
36
|
+
dot_product = 0.0
|
|
37
|
+
for term, weight in query.items():
|
|
38
|
+
if term in vector:
|
|
39
|
+
dot_product += weight * vector[term]
|
|
40
|
+
|
|
41
|
+
doc_mag = math.sqrt(sum(val ** 2 for val in vector.values()))
|
|
42
|
+
|
|
43
|
+
if doc_mag == 0:
|
|
44
|
+
scores.append(0.0)
|
|
45
|
+
else:
|
|
46
|
+
scores.append(dot_product / (query_mag * doc_mag))
|
|
47
|
+
|
|
48
|
+
return scores
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
"""
|
|
2
|
+
ACUITY Framework — TF-IDF Vectorizer
|
|
3
|
+
|
|
4
|
+
Converts business profile descriptions (and user queries) into
|
|
5
|
+
TF-IDF feature vectors for similarity computation using pure Python.
|
|
6
|
+
No external dependencies (no scikit-learn, no numpy).
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import math
|
|
11
|
+
import re
|
|
12
|
+
from collections import defaultdict
|
|
13
|
+
|
|
14
|
+
# Standard English stop words
|
|
15
|
+
STOP_WORDS = set([
|
|
16
|
+
"i", "me", "my", "myself", "we", "our", "ours", "ourselves", "you", "your",
|
|
17
|
+
"yours", "yourself", "yourselves", "he", "him", "his", "himself", "she",
|
|
18
|
+
"her", "hers", "herself", "it", "its", "itself", "they", "them", "their",
|
|
19
|
+
"theirs", "themselves", "what", "which", "who", "whom", "this", "that",
|
|
20
|
+
"these", "those", "am", "is", "are", "was", "were", "be", "been", "being",
|
|
21
|
+
"have", "has", "had", "having", "do", "does", "did", "doing", "a", "an",
|
|
22
|
+
"the", "and", "but", "if", "or", "because", "as", "until", "while", "of",
|
|
23
|
+
"at", "by", "for", "with", "about", "against", "between", "into", "through",
|
|
24
|
+
"during", "before", "after", "above", "below", "to", "from", "up", "down",
|
|
25
|
+
"in", "out", "on", "off", "over", "under", "again", "further", "then",
|
|
26
|
+
"once", "here", "there", "when", "where", "why", "how", "all", "any",
|
|
27
|
+
"both", "each", "few", "more", "most", "other", "some", "such", "no",
|
|
28
|
+
"nor", "not", "only", "own", "same", "so", "than", "too", "very", "s",
|
|
29
|
+
"t", "can", "will", "just", "don", "should", "now"
|
|
30
|
+
])
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class CustomTfidfVectorizer:
|
|
34
|
+
"""Pure-Python TF-IDF vectorizer with n-gram support.
|
|
35
|
+
|
|
36
|
+
This implementation requires no external dependencies and produces
|
|
37
|
+
sparse dictionary-based vectors for memory efficiency.
|
|
38
|
+
|
|
39
|
+
Args:
|
|
40
|
+
ngram_range: Tuple of (min_n, max_n) for n-gram extraction.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
def __init__(self, ngram_range: tuple[int, int] = (1, 2)):
|
|
44
|
+
self.ngram_range = ngram_range
|
|
45
|
+
self.idf_weights: dict[str, float] = {}
|
|
46
|
+
self.vocabulary: set[str] = set()
|
|
47
|
+
self.n_documents: int = 0
|
|
48
|
+
|
|
49
|
+
def _tokenize_and_ngrams(self, text: str) -> list[str]:
|
|
50
|
+
"""Tokenize text and generate n-grams."""
|
|
51
|
+
text = text.lower()
|
|
52
|
+
tokens = re.findall(r'\b[a-z0-9]+\b', text)
|
|
53
|
+
|
|
54
|
+
# Remove stop words
|
|
55
|
+
tokens = [t for t in tokens if t not in STOP_WORDS]
|
|
56
|
+
|
|
57
|
+
ngrams = []
|
|
58
|
+
min_n, max_n = self.ngram_range
|
|
59
|
+
|
|
60
|
+
# Unigrams
|
|
61
|
+
if min_n <= 1:
|
|
62
|
+
ngrams.extend(tokens)
|
|
63
|
+
|
|
64
|
+
# Bigrams, Trigrams, etc.
|
|
65
|
+
for n in range(max(2, min_n), max_n + 1):
|
|
66
|
+
for i in range(len(tokens) - n + 1):
|
|
67
|
+
ngram = " ".join(tokens[i:i + n])
|
|
68
|
+
ngrams.append(ngram)
|
|
69
|
+
|
|
70
|
+
return ngrams
|
|
71
|
+
|
|
72
|
+
def fit_transform(self, documents: list[str]) -> list[dict[str, float]]:
|
|
73
|
+
"""Fit the vectorizer on *documents* and return their TF-IDF vectors.
|
|
74
|
+
|
|
75
|
+
Args:
|
|
76
|
+
documents: List of text strings.
|
|
77
|
+
|
|
78
|
+
Returns:
|
|
79
|
+
List of sparse TF-IDF vectors (dicts mapping terms to weights).
|
|
80
|
+
"""
|
|
81
|
+
self.n_documents = len(documents)
|
|
82
|
+
document_term_counts = []
|
|
83
|
+
doc_frequency: dict[str, int] = defaultdict(int)
|
|
84
|
+
|
|
85
|
+
# Stage 1: Tokenize and compute term frequencies per document
|
|
86
|
+
for doc in documents:
|
|
87
|
+
tokens = self._tokenize_and_ngrams(doc)
|
|
88
|
+
term_counts: dict[str, int] = defaultdict(int)
|
|
89
|
+
for token in tokens:
|
|
90
|
+
term_counts[token] += 1
|
|
91
|
+
|
|
92
|
+
document_term_counts.append(term_counts)
|
|
93
|
+
|
|
94
|
+
# Count document frequency
|
|
95
|
+
for unique_term in set(tokens):
|
|
96
|
+
doc_frequency[unique_term] += 1
|
|
97
|
+
self.vocabulary.add(unique_term)
|
|
98
|
+
|
|
99
|
+
# Stage 2: Compute IDF weights
|
|
100
|
+
for term, df in doc_frequency.items():
|
|
101
|
+
self.idf_weights[term] = math.log(self.n_documents / df)
|
|
102
|
+
|
|
103
|
+
# Stage 3: Compute TF-IDF vectors
|
|
104
|
+
tfidf_vectors = []
|
|
105
|
+
for term_counts in document_term_counts:
|
|
106
|
+
vector = {}
|
|
107
|
+
for term, count in term_counts.items():
|
|
108
|
+
if count > 0:
|
|
109
|
+
tf = 1.0 + math.log(count)
|
|
110
|
+
vector[term] = tf * self.idf_weights[term]
|
|
111
|
+
tfidf_vectors.append(vector)
|
|
112
|
+
|
|
113
|
+
return tfidf_vectors
|
|
114
|
+
|
|
115
|
+
def transform(self, documents: list[str], partial_match: bool = False) -> list[dict[str, float]]:
|
|
116
|
+
"""Transform new documents into the fitted TF-IDF space.
|
|
117
|
+
|
|
118
|
+
Args:
|
|
119
|
+
documents: List of text strings to transform.
|
|
120
|
+
partial_match: If True, enable bi-directional prefix matching
|
|
121
|
+
(e.g., "print" matches "printing" and vice versa).
|
|
122
|
+
|
|
123
|
+
Returns:
|
|
124
|
+
List of sparse TF-IDF vectors.
|
|
125
|
+
"""
|
|
126
|
+
tfidf_vectors = []
|
|
127
|
+
for doc in documents:
|
|
128
|
+
tokens = self._tokenize_and_ngrams(doc)
|
|
129
|
+
term_counts: dict[str, int] = defaultdict(int)
|
|
130
|
+
for token in tokens:
|
|
131
|
+
if token in self.vocabulary:
|
|
132
|
+
term_counts[token] += 1
|
|
133
|
+
|
|
134
|
+
# Bi-directional prefix matching
|
|
135
|
+
if partial_match and len(token) >= 3:
|
|
136
|
+
for vocab_term in self.vocabulary:
|
|
137
|
+
if vocab_term != token and (vocab_term.startswith(token) or token.startswith(vocab_term)):
|
|
138
|
+
term_counts[vocab_term] += 1
|
|
139
|
+
|
|
140
|
+
vector = {}
|
|
141
|
+
for term, count in term_counts.items():
|
|
142
|
+
if count > 0:
|
|
143
|
+
tf = 1.0 + math.log(count)
|
|
144
|
+
vector[term] = tf * self.idf_weights[term]
|
|
145
|
+
tfidf_vectors.append(vector)
|
|
146
|
+
|
|
147
|
+
return tfidf_vectors
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def build_tfidf_matrix(documents: list[str]) -> tuple[CustomTfidfVectorizer, list[dict[str, float]]]:
|
|
151
|
+
"""Fit a custom TF-IDF vectorizer on *documents* and return the vectors.
|
|
152
|
+
|
|
153
|
+
Args:
|
|
154
|
+
documents: List of text strings.
|
|
155
|
+
|
|
156
|
+
Returns:
|
|
157
|
+
Tuple of (vectorizer, tfidf_vectors).
|
|
158
|
+
"""
|
|
159
|
+
vectorizer = CustomTfidfVectorizer(ngram_range=(1, 2))
|
|
160
|
+
tfidf_matrix = vectorizer.fit_transform(documents)
|
|
161
|
+
return vectorizer, tfidf_matrix
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def transform_query(vectorizer: CustomTfidfVectorizer, query: str) -> list[dict[str, float]]:
|
|
165
|
+
"""Transform a user query string into the fitted TF-IDF vector space.
|
|
166
|
+
|
|
167
|
+
Args:
|
|
168
|
+
vectorizer: A fitted ``CustomTfidfVectorizer``.
|
|
169
|
+
query: The user's search query text.
|
|
170
|
+
|
|
171
|
+
Returns:
|
|
172
|
+
List containing one sparse TF-IDF vector for the query.
|
|
173
|
+
"""
|
|
174
|
+
return vectorizer.transform([query], partial_match=True)
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""
|
|
2
|
+
ACUITY Framework — Scraper Module
|
|
3
|
+
|
|
4
|
+
Data collection from Facebook community groups using undetected_chromedriver.
|
|
5
|
+
|
|
6
|
+
Requires the ``scraper`` extra: ``pip install acuity-framework[scraper]``
|
|
7
|
+
|
|
8
|
+
Quick Start:
|
|
9
|
+
>>> from acuity.scraper import FacebookScraper
|
|
10
|
+
>>> scraper = FacebookScraper()
|
|
11
|
+
>>> posts = scraper.run(["https://facebook.com/groups/example"])
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from .scraper import FacebookScraper
|
|
15
|
+
|
|
16
|
+
__all__ = ["FacebookScraper"]
|