avito-sdk 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- avito_sdk/__init__.py +77 -0
- avito_sdk/async_client.py +213 -0
- avito_sdk/cli.py +123 -0
- avito_sdk/client.py +264 -0
- avito_sdk/export.py +203 -0
- avito_sdk/extractors.py +428 -0
- avito_sdk/http.py +210 -0
- avito_sdk/models.py +115 -0
- avito_sdk/py.typed +1 -0
- avito_sdk/tracker.py +185 -0
- avito_sdk/url.py +97 -0
- avito_sdk-0.1.0.dist-info/METADATA +237 -0
- avito_sdk-0.1.0.dist-info/RECORD +16 -0
- avito_sdk-0.1.0.dist-info/WHEEL +4 -0
- avito_sdk-0.1.0.dist-info/entry_points.txt +4 -0
- avito_sdk-0.1.0.dist-info/licenses/LICENSE +21 -0
avito_sdk/__init__.py
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""
|
|
2
|
+
avito-sdk: High-performance Avito scraping & data extraction SDK for Python.
|
|
3
|
+
|
|
4
|
+
Includes enhancements from Duff89/parser_avito:
|
|
5
|
+
- PR #334: Real-time price change tracking and seller name extraction
|
|
6
|
+
- PR #337: Deep parameters and characteristics parsing («О помещении», авто, etc.)
|
|
7
|
+
- PR #329 / Issue #305: Full description parsing from microdata & mobile API
|
|
8
|
+
- Asynchronous and Synchronous clients
|
|
9
|
+
- Headless execution (Zero GUI / Zero Tkinter dependencies)
|
|
10
|
+
- Universal runtime support (Python 3.8-3.16, Free-Threaded No-GIL, PyPy)
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from avito_sdk.async_client import AsyncAvitoClient
|
|
16
|
+
from avito_sdk.client import AvitoClient
|
|
17
|
+
from avito_sdk.export import (
|
|
18
|
+
to_csv,
|
|
19
|
+
to_dataframe,
|
|
20
|
+
to_excel,
|
|
21
|
+
to_json,
|
|
22
|
+
to_jsonl,
|
|
23
|
+
)
|
|
24
|
+
from avito_sdk.extractors import (
|
|
25
|
+
extract_catalog_items,
|
|
26
|
+
extract_description,
|
|
27
|
+
extract_params,
|
|
28
|
+
extract_seller_id,
|
|
29
|
+
extract_seller_name,
|
|
30
|
+
extract_views,
|
|
31
|
+
parse_raw_item,
|
|
32
|
+
)
|
|
33
|
+
from avito_sdk.models import (
|
|
34
|
+
Item,
|
|
35
|
+
PriceRecord,
|
|
36
|
+
SearchFilter,
|
|
37
|
+
SearchPage,
|
|
38
|
+
)
|
|
39
|
+
from avito_sdk.tracker import PriceTracker
|
|
40
|
+
from avito_sdk.url import (
|
|
41
|
+
build_item_api_url,
|
|
42
|
+
build_item_web_url,
|
|
43
|
+
build_page_url,
|
|
44
|
+
build_search_url,
|
|
45
|
+
normalize_region,
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
__version__ = "0.1.0"
|
|
49
|
+
__author__ = "eminsk"
|
|
50
|
+
|
|
51
|
+
__all__ = [
|
|
52
|
+
"__version__",
|
|
53
|
+
"AvitoClient",
|
|
54
|
+
"AsyncAvitoClient",
|
|
55
|
+
"Item",
|
|
56
|
+
"SearchFilter",
|
|
57
|
+
"SearchPage",
|
|
58
|
+
"PriceRecord",
|
|
59
|
+
"PriceTracker",
|
|
60
|
+
"extract_params",
|
|
61
|
+
"extract_seller_name",
|
|
62
|
+
"extract_seller_id",
|
|
63
|
+
"extract_description",
|
|
64
|
+
"extract_views",
|
|
65
|
+
"parse_raw_item",
|
|
66
|
+
"extract_catalog_items",
|
|
67
|
+
"build_search_url",
|
|
68
|
+
"build_item_api_url",
|
|
69
|
+
"build_item_web_url",
|
|
70
|
+
"build_page_url",
|
|
71
|
+
"normalize_region",
|
|
72
|
+
"to_excel",
|
|
73
|
+
"to_json",
|
|
74
|
+
"to_jsonl",
|
|
75
|
+
"to_csv",
|
|
76
|
+
"to_dataframe",
|
|
77
|
+
]
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Asynchronous high-level client for Avito (AsyncAvitoClient).
|
|
3
|
+
Designed for modern asyncio backends (FastAPI, aiohttp, Aiogram, Celery).
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import logging
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import AsyncGenerator, List, Optional, Union
|
|
11
|
+
|
|
12
|
+
from bs4 import BeautifulSoup
|
|
13
|
+
|
|
14
|
+
from avito_sdk.extractors import (
|
|
15
|
+
extract_catalog_items,
|
|
16
|
+
extract_description,
|
|
17
|
+
extract_embedded_state,
|
|
18
|
+
extract_params,
|
|
19
|
+
extract_seller_id,
|
|
20
|
+
extract_seller_name,
|
|
21
|
+
extract_views,
|
|
22
|
+
)
|
|
23
|
+
from avito_sdk.http import AsyncHttpTransport
|
|
24
|
+
from avito_sdk.models import Item, SearchFilter
|
|
25
|
+
from avito_sdk.tracker import PriceTracker
|
|
26
|
+
from avito_sdk.url import (
|
|
27
|
+
build_item_web_url,
|
|
28
|
+
build_page_url,
|
|
29
|
+
build_search_url,
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
logger = logging.getLogger("avito_sdk")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class AsyncAvitoClient:
|
|
36
|
+
"""
|
|
37
|
+
High-level asynchronous Avito SDK client.
|
|
38
|
+
|
|
39
|
+
Example:
|
|
40
|
+
>>> async with AsyncAvitoClient() as client:
|
|
41
|
+
... async for item in client.search("rtx 4070", limit=10):
|
|
42
|
+
... print(item.title, item.price)
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
def __init__(
|
|
46
|
+
self,
|
|
47
|
+
proxy: Optional[str] = None,
|
|
48
|
+
timeout: int = 25,
|
|
49
|
+
max_retries: int = 3,
|
|
50
|
+
tracker_db: Optional[Union[str, Path]] = "avito_prices.db",
|
|
51
|
+
enable_tracking: bool = True,
|
|
52
|
+
):
|
|
53
|
+
self.transport = AsyncHttpTransport(
|
|
54
|
+
proxy=proxy,
|
|
55
|
+
timeout=timeout,
|
|
56
|
+
max_retries=max_retries,
|
|
57
|
+
)
|
|
58
|
+
self.tracker = PriceTracker(db_path=tracker_db) if (enable_tracking and tracker_db) else None
|
|
59
|
+
|
|
60
|
+
async def search(
|
|
61
|
+
self,
|
|
62
|
+
query: str = "",
|
|
63
|
+
region: Optional[str] = "rossiya",
|
|
64
|
+
min_price: Optional[int] = None,
|
|
65
|
+
max_price: Optional[int] = None,
|
|
66
|
+
sort: str = "date",
|
|
67
|
+
category: Optional[str] = None,
|
|
68
|
+
with_delivery: bool = False,
|
|
69
|
+
enrich_details: bool = False,
|
|
70
|
+
limit: Optional[int] = None,
|
|
71
|
+
max_pages: int = 5,
|
|
72
|
+
) -> AsyncGenerator[Item, None]:
|
|
73
|
+
"""
|
|
74
|
+
Asynchronously search Avito and stream items.
|
|
75
|
+
"""
|
|
76
|
+
search_filter = SearchFilter(
|
|
77
|
+
query=query,
|
|
78
|
+
region=region,
|
|
79
|
+
min_price=min_price,
|
|
80
|
+
max_price=max_price,
|
|
81
|
+
sort=sort,
|
|
82
|
+
category=category,
|
|
83
|
+
with_delivery=with_delivery,
|
|
84
|
+
page=1,
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
yielded_count = 0
|
|
88
|
+
for page_num in range(1, max_pages + 1):
|
|
89
|
+
search_filter.page = page_num
|
|
90
|
+
url = build_search_url(search_filter)
|
|
91
|
+
|
|
92
|
+
items = await self.scrape_page_items(url)
|
|
93
|
+
if not items:
|
|
94
|
+
break
|
|
95
|
+
|
|
96
|
+
for item in items:
|
|
97
|
+
if self.tracker:
|
|
98
|
+
self.tracker.check_and_update(item)
|
|
99
|
+
|
|
100
|
+
if enrich_details:
|
|
101
|
+
await self.enrich_item(item)
|
|
102
|
+
|
|
103
|
+
yield item
|
|
104
|
+
yielded_count += 1
|
|
105
|
+
if limit and yielded_count >= limit:
|
|
106
|
+
return
|
|
107
|
+
|
|
108
|
+
async def scrape_page_items(self, url: str) -> List[Item]:
|
|
109
|
+
"""Fetch and extract items from a single page URL asynchronously."""
|
|
110
|
+
html_text = await self.transport.fetch_html(url)
|
|
111
|
+
soup = BeautifulSoup(html_text, "html.parser")
|
|
112
|
+
embedded = extract_embedded_state(soup)
|
|
113
|
+
|
|
114
|
+
if embedded:
|
|
115
|
+
items = extract_catalog_items(embedded)
|
|
116
|
+
if items:
|
|
117
|
+
return items
|
|
118
|
+
|
|
119
|
+
# Fallback to HTML elements
|
|
120
|
+
items: List[Item] = []
|
|
121
|
+
for el in soup.select('[data-marker="item"]'):
|
|
122
|
+
item_id_str = el.get("data-item-id") or el.get("id")
|
|
123
|
+
if not item_id_str:
|
|
124
|
+
continue
|
|
125
|
+
digits = "".join(filter(str.isdigit, str(item_id_str)))
|
|
126
|
+
if not digits:
|
|
127
|
+
continue
|
|
128
|
+
item_id = int(digits)
|
|
129
|
+
|
|
130
|
+
title_el = el.select_one('[data-marker="item-title"]') or el.select_one("h3")
|
|
131
|
+
title = title_el.get_text(strip=True) if title_el else ""
|
|
132
|
+
|
|
133
|
+
price_el = el.select_one('[data-marker="item-price"]') or el.select_one('[itemprop="price"]')
|
|
134
|
+
price = 0
|
|
135
|
+
if price_el:
|
|
136
|
+
price_digits = "".join(filter(str.isdigit, price_el.get_text()))
|
|
137
|
+
price = int(price_digits) if price_digits else 0
|
|
138
|
+
|
|
139
|
+
url_el = el.select_one('a[data-marker="item-title"]') or el.select_one('a[itemprop="url"]')
|
|
140
|
+
link = url_el.get("href") if url_el else ""
|
|
141
|
+
if link and link.startswith("/"):
|
|
142
|
+
link = f"https://www.avito.ru{link}"
|
|
143
|
+
|
|
144
|
+
seller_name = extract_seller_name({}, html_text=str(el))
|
|
145
|
+
|
|
146
|
+
items.append(
|
|
147
|
+
Item(
|
|
148
|
+
id=item_id,
|
|
149
|
+
title=title,
|
|
150
|
+
price=price,
|
|
151
|
+
url=link,
|
|
152
|
+
seller_name=seller_name,
|
|
153
|
+
)
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
return items
|
|
157
|
+
|
|
158
|
+
async def get_item(self, item_id: int) -> Item:
|
|
159
|
+
"""Fetch full details for an item by ID asynchronously."""
|
|
160
|
+
item = Item(id=item_id, url=build_item_web_url(item_id))
|
|
161
|
+
await self.enrich_item(item)
|
|
162
|
+
return item
|
|
163
|
+
|
|
164
|
+
async def enrich_item(self, item: Item) -> Item:
|
|
165
|
+
"""Enrich an item with card parameters, seller, description, and views asynchronously."""
|
|
166
|
+
try:
|
|
167
|
+
payload = await self.transport.fetch_item_card(item.id)
|
|
168
|
+
if payload:
|
|
169
|
+
params = extract_params(payload)
|
|
170
|
+
if params:
|
|
171
|
+
item.params = params
|
|
172
|
+
|
|
173
|
+
if not item.seller_name:
|
|
174
|
+
item.seller_name = extract_seller_name(payload)
|
|
175
|
+
if not item.seller_id:
|
|
176
|
+
item.seller_id = extract_seller_id(payload)
|
|
177
|
+
|
|
178
|
+
desc = extract_description(payload)
|
|
179
|
+
if desc:
|
|
180
|
+
item.description = desc
|
|
181
|
+
|
|
182
|
+
total_views, today_views = extract_views(payload)
|
|
183
|
+
if total_views is not None:
|
|
184
|
+
item.total_views = total_views
|
|
185
|
+
if today_views is not None:
|
|
186
|
+
item.today_views = today_views
|
|
187
|
+
except Exception as err:
|
|
188
|
+
logger.debug(f"Async API card fetch failed for item {item.id}: {err}")
|
|
189
|
+
try:
|
|
190
|
+
web_url = item.url or build_item_web_url(item.id)
|
|
191
|
+
html_text = await self.transport.fetch_html(web_url)
|
|
192
|
+
if not item.seller_name:
|
|
193
|
+
item.seller_name = extract_seller_name({}, html_text=html_text)
|
|
194
|
+
if not item.description:
|
|
195
|
+
item.description = extract_description({}, html_text=html_text)
|
|
196
|
+
total_views, today_views = extract_views({}, html_text=html_text)
|
|
197
|
+
if total_views is not None:
|
|
198
|
+
item.total_views = total_views
|
|
199
|
+
if today_views is not None:
|
|
200
|
+
item.today_views = today_views
|
|
201
|
+
except Exception as html_err:
|
|
202
|
+
logger.warning(f"Could not enrich item {item.id} asynchronously: {html_err}")
|
|
203
|
+
|
|
204
|
+
return item
|
|
205
|
+
|
|
206
|
+
async def close(self) -> None:
|
|
207
|
+
await self.transport.close()
|
|
208
|
+
|
|
209
|
+
async def __aenter__(self):
|
|
210
|
+
return self
|
|
211
|
+
|
|
212
|
+
async def __aexit__(self, exc_type, exc_val, exc_tb):
|
|
213
|
+
await self.close()
|
avito_sdk/cli.py
ADDED
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Command-line interface (CLI) for avito-sdk.
|
|
3
|
+
Run queries, inspect items, monitor prices, and export datasets directly from terminal.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import argparse
|
|
9
|
+
import sys
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from avito_sdk.client import AvitoClient
|
|
13
|
+
from avito_sdk.export import to_csv, to_excel, to_json
|
|
14
|
+
from avito_sdk.tracker import PriceTracker
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def main(argv=None) -> int:
|
|
18
|
+
parser = argparse.ArgumentParser(
|
|
19
|
+
prog="avito-sdk",
|
|
20
|
+
description="High-performance Avito Scraper & SDK with price tracking and parameters extraction",
|
|
21
|
+
)
|
|
22
|
+
subparsers = parser.add_subparsers(dest="command", help="Available subcommands")
|
|
23
|
+
|
|
24
|
+
# Search command
|
|
25
|
+
search_p = subparsers.add_parser("search", help="Search Avito listings")
|
|
26
|
+
search_p.add_argument("query", help="Search query (e.g. 'ноутбук lenovo')")
|
|
27
|
+
search_p.add_argument("-r", "--region", default="rossiya", help="Region slug or city (default: rossiya)")
|
|
28
|
+
search_p.add_argument("--min-price", type=int, default=None, help="Minimum price filter in RUB")
|
|
29
|
+
search_p.add_argument("--max-price", type=int, default=None, help="Maximum price filter in RUB")
|
|
30
|
+
search_p.add_argument("-n", "--limit", type=int, default=20, help="Maximum number of items to fetch")
|
|
31
|
+
search_p.add_argument("-p", "--pages", type=int, default=3, help="Max search result pages to scan")
|
|
32
|
+
search_p.add_argument("-o", "--output", help="Save output to file (.json, .csv, .xlsx)")
|
|
33
|
+
search_p.add_argument("--enrich", action="store_true", help="Fetch detailed parameters and full descriptions")
|
|
34
|
+
|
|
35
|
+
# Item command
|
|
36
|
+
item_p = subparsers.add_parser("item", help="Fetch detailed info for a single item")
|
|
37
|
+
item_p.add_argument("item_id", type=int, help="Avito numeric item ID")
|
|
38
|
+
item_p.add_argument("-o", "--output", help="Save output to JSON file")
|
|
39
|
+
|
|
40
|
+
# Drops command
|
|
41
|
+
drops_p = subparsers.add_parser("drops", help="Show all tracked items whose price has dropped")
|
|
42
|
+
drops_p.add_argument("--db", default="avito_prices.db", help="Path to SQLite price database")
|
|
43
|
+
|
|
44
|
+
args = parser.parse_args(argv)
|
|
45
|
+
|
|
46
|
+
if not args.command:
|
|
47
|
+
parser.print_help()
|
|
48
|
+
return 0
|
|
49
|
+
|
|
50
|
+
if args.command == "search":
|
|
51
|
+
print(f"[*] Searching for '{args.query}' in '{args.region}' (limit={args.limit})...")
|
|
52
|
+
client = AvitoClient()
|
|
53
|
+
items = list(
|
|
54
|
+
client.search(
|
|
55
|
+
query=args.query,
|
|
56
|
+
region=args.region,
|
|
57
|
+
min_price=args.min_price,
|
|
58
|
+
max_price=args.max_price,
|
|
59
|
+
enrich_details=args.enrich,
|
|
60
|
+
limit=args.limit,
|
|
61
|
+
max_pages=args.pages,
|
|
62
|
+
)
|
|
63
|
+
)
|
|
64
|
+
print(f"[+] Found {len(items)} items:")
|
|
65
|
+
for idx, item in enumerate(items, 1):
|
|
66
|
+
price_str = f"{item.price:,} ₽"
|
|
67
|
+
if item.old_price:
|
|
68
|
+
price_str += f" (было: {item.old_price:,} ₽, снижение на {item.price_drop:,} ₽)"
|
|
69
|
+
seller_str = f" | {item.seller_name}" if item.seller_name else ""
|
|
70
|
+
print(f" {idx:2d}. [{item.id}] {item.title[:45]} - {price_str}{seller_str}")
|
|
71
|
+
|
|
72
|
+
if args.output:
|
|
73
|
+
out_path = Path(args.output)
|
|
74
|
+
suffix = out_path.suffix.lower()
|
|
75
|
+
if suffix == ".xlsx":
|
|
76
|
+
to_excel(items, out_path)
|
|
77
|
+
elif suffix == ".csv":
|
|
78
|
+
to_csv(items, out_path)
|
|
79
|
+
else:
|
|
80
|
+
to_json(items, out_path)
|
|
81
|
+
print(f"[✓] Saved {len(items)} items to {out_path}")
|
|
82
|
+
|
|
83
|
+
elif args.command == "item":
|
|
84
|
+
print(f"[*] Fetching item {args.item_id}...")
|
|
85
|
+
client = AvitoClient()
|
|
86
|
+
item = client.get_item(args.item_id)
|
|
87
|
+
print(f"\nID: {item.id}")
|
|
88
|
+
print(f"Title: {item.title}")
|
|
89
|
+
print(f"Price: {item.price:,} ₽")
|
|
90
|
+
if item.seller_name:
|
|
91
|
+
print(f"Seller: {item.seller_name} (ID: {item.seller_id})")
|
|
92
|
+
if item.total_views is not None:
|
|
93
|
+
print(f"Views: {item.total_views} (today: {item.today_views})")
|
|
94
|
+
if item.params:
|
|
95
|
+
print("Parameters:")
|
|
96
|
+
for k, v in item.params.items():
|
|
97
|
+
print(f" - {k}: {v}")
|
|
98
|
+
if item.description:
|
|
99
|
+
print(f"\nDescription:\n{item.description[:300]}...")
|
|
100
|
+
|
|
101
|
+
if args.output:
|
|
102
|
+
to_json([item], args.output)
|
|
103
|
+
print(f"[✓] Saved to {args.output}")
|
|
104
|
+
|
|
105
|
+
elif args.command == "drops":
|
|
106
|
+
tracker = PriceTracker(db_path=args.db)
|
|
107
|
+
drops = tracker.get_price_drops()
|
|
108
|
+
if not drops:
|
|
109
|
+
print("[-] No price drops recorded yet in database.")
|
|
110
|
+
else:
|
|
111
|
+
print(f"[+] Found {len(drops)} items with price drops:")
|
|
112
|
+
for d in drops:
|
|
113
|
+
diff = d['initial_price'] - d['current_price']
|
|
114
|
+
print(
|
|
115
|
+
f" - [{d['item_id']}] {d.get('title', '')[:40]}: "
|
|
116
|
+
f"{d['initial_price']:,} -> {d['current_price']:,} ₽ (-{diff:,} ₽)"
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
return 0
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
if __name__ == "__main__":
|
|
123
|
+
sys.exit(main())
|
avito_sdk/client.py
ADDED
|
@@ -0,0 +1,264 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Synchronous high-level client for Avito (AvitoClient).
|
|
3
|
+
Provides intuitive methods for search, item enrichment, price tracking, and export.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import logging
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Generator, Iterable, List, Optional, Union
|
|
11
|
+
|
|
12
|
+
from bs4 import BeautifulSoup
|
|
13
|
+
|
|
14
|
+
from avito_sdk.export import to_csv, to_dataframe, to_excel, to_json, to_jsonl
|
|
15
|
+
from avito_sdk.extractors import (
|
|
16
|
+
extract_catalog_items,
|
|
17
|
+
extract_description,
|
|
18
|
+
extract_embedded_state,
|
|
19
|
+
extract_params,
|
|
20
|
+
extract_seller_id,
|
|
21
|
+
extract_seller_name,
|
|
22
|
+
extract_views,
|
|
23
|
+
parse_raw_item,
|
|
24
|
+
)
|
|
25
|
+
from avito_sdk.http import SyncHttpTransport
|
|
26
|
+
from avito_sdk.models import Item, SearchFilter, SearchPage
|
|
27
|
+
from avito_sdk.tracker import PriceTracker
|
|
28
|
+
from avito_sdk.url import (
|
|
29
|
+
build_item_api_url,
|
|
30
|
+
build_item_web_url,
|
|
31
|
+
build_page_url,
|
|
32
|
+
build_search_url,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
logger = logging.getLogger("avito_sdk")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class AvitoClient:
|
|
39
|
+
"""
|
|
40
|
+
High-level synchronous Avito SDK client.
|
|
41
|
+
|
|
42
|
+
Example:
|
|
43
|
+
>>> from avito_sdk import AvitoClient
|
|
44
|
+
>>> client = AvitoClient()
|
|
45
|
+
>>> for item in client.search("ThinkPad", region="moskva", limit=10):
|
|
46
|
+
... print(item.title, item.price, item.seller_name)
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
def __init__(
|
|
50
|
+
self,
|
|
51
|
+
proxy: Optional[str] = None,
|
|
52
|
+
timeout: int = 25,
|
|
53
|
+
max_retries: int = 3,
|
|
54
|
+
tracker_db: Optional[Union[str, Path]] = "avito_prices.db",
|
|
55
|
+
enable_tracking: bool = True,
|
|
56
|
+
):
|
|
57
|
+
self.transport = SyncHttpTransport(
|
|
58
|
+
proxy=proxy,
|
|
59
|
+
timeout=timeout,
|
|
60
|
+
max_retries=max_retries,
|
|
61
|
+
)
|
|
62
|
+
self.tracker = PriceTracker(db_path=tracker_db) if (enable_tracking and tracker_db) else None
|
|
63
|
+
|
|
64
|
+
def search(
|
|
65
|
+
self,
|
|
66
|
+
query: str = "",
|
|
67
|
+
region: Optional[str] = "rossiya",
|
|
68
|
+
min_price: Optional[int] = None,
|
|
69
|
+
max_price: Optional[int] = None,
|
|
70
|
+
sort: str = "date",
|
|
71
|
+
category: Optional[str] = None,
|
|
72
|
+
with_delivery: bool = False,
|
|
73
|
+
enrich_details: bool = False,
|
|
74
|
+
limit: Optional[int] = None,
|
|
75
|
+
max_pages: int = 5,
|
|
76
|
+
) -> Generator[Item, None, None]:
|
|
77
|
+
"""
|
|
78
|
+
Search Avito for items matching specified criteria.
|
|
79
|
+
Yields Item objects one by one.
|
|
80
|
+
"""
|
|
81
|
+
search_filter = SearchFilter(
|
|
82
|
+
query=query,
|
|
83
|
+
region=region,
|
|
84
|
+
min_price=min_price,
|
|
85
|
+
max_price=max_price,
|
|
86
|
+
sort=sort,
|
|
87
|
+
category=category,
|
|
88
|
+
with_delivery=with_delivery,
|
|
89
|
+
page=1,
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
yielded_count = 0
|
|
93
|
+
for page_num in range(1, max_pages + 1):
|
|
94
|
+
search_filter.page = page_num
|
|
95
|
+
url = build_search_url(search_filter)
|
|
96
|
+
logger.debug(f"Fetching search page {page_num}: {url}")
|
|
97
|
+
|
|
98
|
+
items = self.scrape_page_items(url)
|
|
99
|
+
if not items:
|
|
100
|
+
break
|
|
101
|
+
|
|
102
|
+
for item in items:
|
|
103
|
+
if self.tracker:
|
|
104
|
+
self.tracker.check_and_update(item)
|
|
105
|
+
|
|
106
|
+
if enrich_details:
|
|
107
|
+
self.enrich_item(item)
|
|
108
|
+
|
|
109
|
+
yield item
|
|
110
|
+
yielded_count += 1
|
|
111
|
+
if limit and yielded_count >= limit:
|
|
112
|
+
return
|
|
113
|
+
|
|
114
|
+
def scrape_url(
|
|
115
|
+
self,
|
|
116
|
+
url: str,
|
|
117
|
+
max_pages: int = 1,
|
|
118
|
+
enrich_details: bool = False,
|
|
119
|
+
) -> List[Item]:
|
|
120
|
+
"""Scrape items from an existing Avito search or catalog URL across multiple pages."""
|
|
121
|
+
results: List[Item] = []
|
|
122
|
+
for page in range(1, max_pages + 1):
|
|
123
|
+
page_url = build_page_url(url, page)
|
|
124
|
+
page_items = self.scrape_page_items(page_url)
|
|
125
|
+
if not page_items:
|
|
126
|
+
break
|
|
127
|
+
|
|
128
|
+
for item in page_items:
|
|
129
|
+
if self.tracker:
|
|
130
|
+
self.tracker.check_and_update(item)
|
|
131
|
+
if enrich_details:
|
|
132
|
+
self.enrich_item(item)
|
|
133
|
+
results.append(item)
|
|
134
|
+
|
|
135
|
+
return results
|
|
136
|
+
|
|
137
|
+
def scrape_page_items(self, url: str) -> List[Item]:
|
|
138
|
+
"""Fetch and extract items from a single page URL."""
|
|
139
|
+
html_text = self.transport.fetch_html(url)
|
|
140
|
+
soup = BeautifulSoup(html_text, "html.parser")
|
|
141
|
+
embedded = extract_embedded_state(soup)
|
|
142
|
+
|
|
143
|
+
if embedded:
|
|
144
|
+
items = extract_catalog_items(embedded)
|
|
145
|
+
if items:
|
|
146
|
+
return items
|
|
147
|
+
|
|
148
|
+
# Fallback to HTML elements
|
|
149
|
+
items: List[Item] = []
|
|
150
|
+
for el in soup.select('[data-marker="item"]'):
|
|
151
|
+
item_id_str = el.get("data-item-id") or el.get("id")
|
|
152
|
+
if not item_id_str:
|
|
153
|
+
continue
|
|
154
|
+
digits = "".join(filter(str.isdigit, str(item_id_str)))
|
|
155
|
+
if not digits:
|
|
156
|
+
continue
|
|
157
|
+
item_id = int(digits)
|
|
158
|
+
|
|
159
|
+
title_el = el.select_one('[data-marker="item-title"]') or el.select_one("h3")
|
|
160
|
+
title = title_el.get_text(strip=True) if title_el else ""
|
|
161
|
+
|
|
162
|
+
price_el = el.select_one('[data-marker="item-price"]') or el.select_one('[itemprop="price"]')
|
|
163
|
+
price = 0
|
|
164
|
+
if price_el:
|
|
165
|
+
price_digits = "".join(filter(str.isdigit, price_el.get_text()))
|
|
166
|
+
price = int(price_digits) if price_digits else 0
|
|
167
|
+
|
|
168
|
+
url_el = el.select_one('a[data-marker="item-title"]') or el.select_one('a[itemprop="url"]')
|
|
169
|
+
link = url_el.get("href") if url_el else ""
|
|
170
|
+
if link and link.startswith("/"):
|
|
171
|
+
link = f"https://www.avito.ru{link}"
|
|
172
|
+
|
|
173
|
+
seller_name = extract_seller_name({}, html_text=str(el))
|
|
174
|
+
|
|
175
|
+
items.append(
|
|
176
|
+
Item(
|
|
177
|
+
id=item_id,
|
|
178
|
+
title=title,
|
|
179
|
+
price=price,
|
|
180
|
+
url=link,
|
|
181
|
+
seller_name=seller_name,
|
|
182
|
+
)
|
|
183
|
+
)
|
|
184
|
+
|
|
185
|
+
return items
|
|
186
|
+
|
|
187
|
+
def get_item(self, item_id: int) -> Item:
|
|
188
|
+
"""
|
|
189
|
+
Fetch full details for an item by ID, including:
|
|
190
|
+
- Full description (Issue #305)
|
|
191
|
+
- Parameters and specs (PR #337)
|
|
192
|
+
- Seller name and ID (PR #334)
|
|
193
|
+
- View counts (total and today)
|
|
194
|
+
"""
|
|
195
|
+
item = Item(id=item_id, url=build_item_web_url(item_id))
|
|
196
|
+
self.enrich_item(item)
|
|
197
|
+
return item
|
|
198
|
+
|
|
199
|
+
def enrich_item(self, item: Item) -> Item:
|
|
200
|
+
"""Enrich an existing Item instance with deep card parameters and views."""
|
|
201
|
+
try:
|
|
202
|
+
payload = self.transport.fetch_item_card(item.id)
|
|
203
|
+
if payload:
|
|
204
|
+
# Parameters (PR #337)
|
|
205
|
+
params = extract_params(payload)
|
|
206
|
+
if params:
|
|
207
|
+
item.params = params
|
|
208
|
+
|
|
209
|
+
# Seller info (PR #334)
|
|
210
|
+
if not item.seller_name:
|
|
211
|
+
item.seller_name = extract_seller_name(payload)
|
|
212
|
+
if not item.seller_id:
|
|
213
|
+
item.seller_id = extract_seller_id(payload)
|
|
214
|
+
|
|
215
|
+
# Description and views
|
|
216
|
+
desc = extract_description(payload)
|
|
217
|
+
if desc:
|
|
218
|
+
item.description = desc
|
|
219
|
+
|
|
220
|
+
total_views, today_views = extract_views(payload)
|
|
221
|
+
if total_views is not None:
|
|
222
|
+
item.total_views = total_views
|
|
223
|
+
if today_views is not None:
|
|
224
|
+
item.today_views = today_views
|
|
225
|
+
except Exception as err:
|
|
226
|
+
logger.debug(f"API card fetch failed for item {item.id}, attempting HTML fallback: {err}")
|
|
227
|
+
try:
|
|
228
|
+
web_url = item.url or build_item_web_url(item.id)
|
|
229
|
+
html_text = self.transport.fetch_html(web_url)
|
|
230
|
+
if not item.seller_name:
|
|
231
|
+
item.seller_name = extract_seller_name({}, html_text=html_text)
|
|
232
|
+
if not item.description:
|
|
233
|
+
item.description = extract_description({}, html_text=html_text)
|
|
234
|
+
total_views, today_views = extract_views({}, html_text=html_text)
|
|
235
|
+
if total_views is not None:
|
|
236
|
+
item.total_views = total_views
|
|
237
|
+
if today_views is not None:
|
|
238
|
+
item.today_views = today_views
|
|
239
|
+
except Exception as html_err:
|
|
240
|
+
logger.warning(f"Could not enrich item {item.id}: {html_err}")
|
|
241
|
+
|
|
242
|
+
return item
|
|
243
|
+
|
|
244
|
+
# Export helper shortcuts
|
|
245
|
+
def export_excel(self, items: List[Item], filepath: Union[str, Path]) -> None:
|
|
246
|
+
to_excel(items, filepath)
|
|
247
|
+
|
|
248
|
+
def export_json(self, items: List[Item], filepath: Union[str, Path], indent: int = 2) -> None:
|
|
249
|
+
to_json(items, filepath, indent=indent)
|
|
250
|
+
|
|
251
|
+
def export_csv(self, items: List[Item], filepath: Union[str, Path]) -> None:
|
|
252
|
+
to_csv(items, filepath)
|
|
253
|
+
|
|
254
|
+
def export_dataframe(self, items: List[Item]) -> Any:
|
|
255
|
+
return to_dataframe(items)
|
|
256
|
+
|
|
257
|
+
def close(self) -> None:
|
|
258
|
+
self.transport.close()
|
|
259
|
+
|
|
260
|
+
def __enter__(self):
|
|
261
|
+
return self
|
|
262
|
+
|
|
263
|
+
def __exit__(self, exc_type, exc_val, exc_tb):
|
|
264
|
+
self.close()
|