new web ract

This commit is contained in:
jze9
2026-05-15 03:31:28 +05:00
parent de78624495
commit 2335497226
58 changed files with 7397 additions and 406 deletions

179
api/vk_scraper.py Normal file
View File

@@ -0,0 +1,179 @@
"""
Парсинг публичных групп ВКонтакте без API-ключа.
Использует мобильную версию m.vk.com — простой HTML без JS-рендеринга.
"""
import asyncio
import re
from datetime import datetime, timezone, timedelta
import httpx
from bs4 import BeautifulSoup
# Имитируем мобильный браузер
_HEADERS = {
"User-Agent": (
"Mozilla/5.0 (Linux; Android 13; Pixel 7) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/120.0.0.0 Mobile Safari/537.36"
),
"Accept-Language": "ru-RU,ru;q=0.9",
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
}
_BASE = "https://m.vk.com"
# Разбираем русские сокращения дат вида "14 мая в 10:35" или "вчера в 18:00"
_MONTHS = {
"янв": 1, "фев": 2, "мар": 3, "апр": 4, "май": 5, "маи": 5,
"июн": 6, "июл": 7, "авг": 8, "сен": 9, "окт": 10, "ноя": 11, "дек": 12,
}
def _parse_vk_date(text: str) -> datetime:
"""Конвертирует строку даты VK в datetime UTC."""
now = datetime.now(timezone.utc)
text = text.strip().lower()
if "сегодня" in text or "today" in text:
time_match = re.search(r"(\d{1,2}):(\d{2})", text)
if time_match:
return now.replace(hour=int(time_match[1]), minute=int(time_match[2]), second=0, microsecond=0)
return now
if "вчера" in text or "yesterday" in text:
yesterday = now - timedelta(days=1)
time_match = re.search(r"(\d{1,2}):(\d{2})", text)
if time_match:
return yesterday.replace(hour=int(time_match[1]), minute=int(time_match[2]), second=0, microsecond=0)
return yesterday
# "14 мая в 10:35" или "14 мая 2023 в 10:35"
match = re.search(r"(\d{1,2})\s+([а-яё]+)(?:\s+(\d{4}))?(?:\s+в\s+(\d{1,2}):(\d{2}))?", text)
if match:
day = int(match[1])
month = _MONTHS.get(match[2][:3], 1)
year = int(match[3]) if match[3] else now.year
hour = int(match[4]) if match[4] else 12
minute= int(match[5]) if match[5] else 0
try:
return datetime(year, month, day, hour, minute, tzinfo=timezone.utc)
except ValueError:
pass
return now
async def fetch_group_posts_html(group_id: str, count: int = 20) -> list[dict]:
"""
Скачивает и парсит посты публичной группы ВКонтакте.
group_id — числовой ID без минуса (например "12345").
Возвращает список словарей совместимых с форматом vk_parser.
"""
url = f"{_BASE}/wall-{group_id}"
posts = []
try:
async with httpx.AsyncClient(
headers=_HEADERS,
timeout=20,
follow_redirects=True,
) as client:
r = await client.get(url)
if r.status_code != 200:
print(f"[VK scraper] HTTP {r.status_code} для группы {group_id}")
return []
soup = BeautifulSoup(r.text, "lxml")
items = soup.select("div.wall_item, div._post, article.post")
if not items:
# Попробуем запасной селектор
items = soup.select("[id^='post-']")
for item in items[:count]:
post = _parse_post(item, group_id)
if post:
posts.append(post)
except Exception as exc:
print(f"[VK scraper] Ошибка группы {group_id}: {exc}")
return posts
def _parse_post(item, group_id: str) -> dict | None:
"""Парсит один пост из HTML-элемента."""
# ── ID поста ──────────────────────────────────────────────────────────────
post_id = None
elem_id = item.get("id", "")
id_match = re.search(r"(\d+)_(\d+)$", elem_id)
if id_match:
post_id = int(id_match[2])
else:
# Ищем ссылку на пост вида /wall-12345_67890
link = item.select_one("a[href*='wall-']")
if link:
lm = re.search(r"wall-?\d+_(\d+)", link["href"])
if lm:
post_id = int(lm[1])
if not post_id:
return None
owner_id = int(group_id)
# ── Текст ─────────────────────────────────────────────────────────────────
text_el = (
item.select_one(".pi_text")
or item.select_one(".wall_post_text")
or item.select_one("._post_content")
or item.select_one(".post_text")
)
text = text_el.get_text(separator="\n").strip() if text_el else ""
# ── Дата ──────────────────────────────────────────────────────────────────
date_el = item.select_one(".pi_date, time, .post_date, ._date")
ts = int(datetime.now(timezone.utc).timestamp())
if date_el:
dt = _parse_vk_date(date_el.get_text())
ts = int(dt.timestamp())
# ── Изображения ───────────────────────────────────────────────────────────
attachments = []
for img in item.select("img.pi_img, .wall_post_img img, img._photo_img"):
src = img.get("src") or img.get("data-src") or ""
if src and "userapi.com" in src or "vk.com" in src:
attachments.append({"type": "photo", "photo": {"sizes": [{"type": "x", "url": src}]}})
# ── Видео ─────────────────────────────────────────────────────────────────
for video_link in item.select("a[href*='/video']"):
href = video_link.get("href", "")
vid_match = re.search(r"/video(-?\d+)_(\d+)", href)
if vid_match:
oid = vid_match[1]
vid = vid_match[2]
player = f"https://vk.com/video_ext.php?oid={oid}&id={vid}"
title = video_link.get_text(strip=True) or "Видео"
attachments.append({"type": "video", "video": {"player": player, "title": title}})
# ── Ссылки ────────────────────────────────────────────────────────────────
for a in item.select("a.pi_snippet, a._snippet"):
href = a.get("href", "")
label = a.select_one(".pi_snippet_title, .snippet_title")
if href and not href.startswith("/"):
attachments.append({
"type": "link",
"link": {"url": href, "title": label.get_text(strip=True) if label else href},
})
if not text and not attachments:
return None
return {
"id": post_id,
"owner_id": -owner_id,
"date": ts,
"text": text,
"attachments": attachments,
}