Files
rf4-spotter/rf4_research/community_sources.py
T
ik 722c88d436
CI / backend-and-migrations (push) Canceled after 0s
CI / astro-build (push) Canceled after 0s
CI / dependency-audit (push) Canceled after 0s
CI / compose-e2e (push) Canceled after 0s
sync
2026-09-17 07:35:04 +07:00

518 lines
22 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
from __future__ import annotations
import json
import re
from dataclasses import dataclass
from datetime import datetime, time, timezone
from urllib.parse import urljoin
from bs4 import BeautifulSoup, Tag
class CommunityParseError(ValueError):
pass
@dataclass(frozen=True, slots=True)
class ExternalCatch:
source_system: str
source_external_id: str
source_url: str
fish: str
fish_external_id: str | None
waterbody: str
waterbody_external_id: str | None
x: int | None
y: int | None
weight_g: int | None
bait: str | None
rig_type: str | None
game_time: time | None
published_at: datetime | None
player_name: str | None
weather: str | None
water_temperature_c: float | None
clip: str | None
fishing_style: str | None
evidence_urls: tuple[str, ...]
coordinate_raw: str | None = None
@dataclass(frozen=True, slots=True)
class EquipmentItem:
kind: str
name: str
external_id: str | None
@dataclass(frozen=True, slots=True)
class RF4DBCatchDetail:
source_external_id: str
source_url: str
wind: str | None
line_release_m: float | None
clip_m: float | None
cast_direction_deg: float | None
equipment: tuple[EquipmentItem, ...]
@dataclass(frozen=True, slots=True)
class RF4DBWaterbody:
source_system: str
source_external_id: str
source_url: str
name: str
unlock_level: int | None
unlock_label: str
fish_species_count: int
image_url: str | None
@dataclass(frozen=True, slots=True)
class RF4DBWaterbodyDetail:
source_system: str
source_external_id: str
source_url: str
name: str
description: str | None
aliases: tuple[str, ...]
fish_species: tuple[str, ...]
fish_external_ids: tuple[str | None, ...]
image_urls: tuple[str, ...]
point_urls: tuple[str, ...]
def _text(node: Tag | None) -> str:
return " ".join(node.get_text(" ", strip=True).split()) if node else ""
def _fish_name(node: Tag) -> str:
"""Read a fish label while dropping the optional trophy weight suffix."""
return re.sub(r"\s*\d+(?:[.,]\d+)?\s*(?:кг|kg|г|g)\s*$", "", _text(node), flags=re.I).strip()
def _key(href: str | None) -> str | None:
if not href:
return None
return href.rstrip("/").rsplit("/", 1)[-1]
def _coordinates(raw: str) -> tuple[int | None, int | None]:
match = re.fullmatch(r"\s*(-?\d{1,5}):(-?\d{1,5})\s*", raw)
return (int(match.group(1)), int(match.group(2))) if match else (None, None)
def _game_time(raw: str) -> time | None:
match = re.search(r"Игровое время\s*(\d{1,2}):(\d{2})", raw)
return time(int(match.group(1)), int(match.group(2))) if match else None
def _weight(raw: str) -> int | None:
match = re.search(r"([\d\s.,]+)\s*(кг|kg|г|g)\b", raw.casefold())
if not match:
return None
number, unit = match.groups()
compact = number.replace(" ", "").replace(",", ".")
return round(float(compact) * 1000) if unit in {"кг", "kg"} else int(compact.replace(".", ""))
def _next_payloads(html: str) -> list[str]:
"""Decode the string payloads emitted by Next.js' server components."""
soup = BeautifulSoup(html, "html.parser")
result: list[str] = []
pattern = re.compile(r'^self\.__next_f\.push\(\[1,("(?:\\.|[^"\\])*")\]\)$', re.DOTALL)
for script in soup.find_all("script"):
raw = script.string or ""
match = pattern.fullmatch(raw.strip())
if not match:
continue
try:
result.append(json.loads(match.group(1)))
except json.JSONDecodeError:
continue
return result
def _json_after_marker(
payloads: list[str], marker: str, *, required_key: str | None = None,
) -> dict[str, object] | None:
decoder = json.JSONDecoder()
for payload in payloads:
offset = 0
while (start := payload.find(marker, offset)) >= 0:
offset = start + len(marker)
try:
value, _ = decoder.raw_decode(payload[offset:])
except json.JSONDecodeError:
continue
if isinstance(value, dict) and (required_key is None or required_key in value):
return value
return None
def _datetime(raw: object) -> datetime | None:
if not isinstance(raw, str):
return None
try:
return datetime.fromisoformat(raw.replace("Z", "+00:00"))
except ValueError:
return None
def parse_rf4db_waterbodies(
html: str, *, source_url: str = "https://rf4db.com/ru/maps", expected_count: int = 19,
) -> list[RF4DBWaterbody]:
"""Parse the public RF4DB waterbody index without assigning image roles."""
soup = BeautifulSoup(html, "html.parser")
result: list[RF4DBWaterbody] = []
seen: set[str] = set()
for link in soup.select('a[href*="/ru/maps/level_"]'):
href = link.get("href")
if not isinstance(href, str):
continue
external_id = _key(href)
if not external_id or external_id in seen:
continue
name = _text(link)
if not name:
continue
card = link.find_parent(["article", "li"])
if card is None:
card = link.find_parent("div")
card_text = _text(card)
level_match = re.search(r"(?:Уровень|Level)\s*(?:Lv\.?\s*)?(Старт|Start|\d+)", card_text, re.I)
fish_match = re.search(r"(?:Рыбы|Fish(?:es)?)\s*(\d+)", card_text, re.I)
if not level_match or not fish_match:
continue
unlock_label = level_match.group(1)
unlock_level = int(unlock_label) if unlock_label.isdigit() else None
image = card.select_one("img[src], img[data-src]") if card else None
image_raw = image.get("src") or image.get("data-src") if image else None
image_url = urljoin(source_url, str(image_raw)) if image_raw else None
result.append(RF4DBWaterbody(
source_system="rf4db",
source_external_id=external_id,
source_url=urljoin(source_url, href),
name=name,
unlock_level=unlock_level,
unlock_label=unlock_label,
fish_species_count=int(fish_match.group(1)),
image_url=image_url,
))
seen.add(external_id)
if not result:
raise CommunityParseError("RF4DB waterbody cards not found")
if len(result) != expected_count:
raise CommunityParseError(
f"RF4DB waterbody catalog is incomplete: expected {expected_count}, got {len(result)}"
)
return result
def parse_rf4db_waterbody_detail(
html: str, *, source_url: str,
) -> RF4DBWaterbodyDetail:
"""Parse one RF4DB waterbody page without assigning media or coordinates.
The detail page is accepted only when its localized heading and fish list
are present. Images and point links remain source candidates; review and
canonical crosswalks happen in later pipeline stages.
"""
soup = BeautifulSoup(html, "html.parser")
root = soup.select_one("article.waterbody-detail, main[data-waterbody-detail]") or soup
external_id = _key(source_url)
name = _text(root.select_one("h1"))
fish_nodes = root.select(".waterbody-fish a[href], [data-fish-list] a[href], a[href*='/fishes/']")
if not fish_nodes:
fish_nodes = root.select(".fish-list a[href], ul.fish a[href]")
fish_species: list[str] = []
fish_external_ids: list[str | None] = []
seen_fish: set[str] = set()
for node in fish_nodes:
fish_name = _fish_name(node)
fish_id = _key(node.get("href"))
identity = fish_id or fish_name.casefold()
if not fish_name or identity in seen_fish:
continue
fish_species.append(fish_name)
fish_external_ids.append(fish_id)
seen_fish.add(identity)
if not external_id or not name or not fish_species:
raise CommunityParseError("RF4DB waterbody detail not found or incomplete")
description_node = root.select_one("[data-description], .waterbody-description, .description")
description = _text(description_node) or None
aliases = tuple(dict.fromkeys(
_text(node) for node in root.select("[data-alias], .waterbody-aliases li, .aliases li")
if _text(node) and _text(node) != name
))
image_urls = tuple(dict.fromkeys(
urljoin(source_url, str(node.get("src") or node.get("data-src")))
for node in root.select("img[src], img[data-src]")
if (node.get("src") or node.get("data-src"))
and "/fish/" not in str(node.get("src") or node.get("data-src"))
))
point_urls = tuple(dict.fromkeys(
urljoin(source_url, str(node.get("href")))
for node in root.select('a[href*="/spots/"], a[href*="/points/"]')
if node.get("href")
))
return RF4DBWaterbodyDetail(
source_system="rf4db", source_external_id=external_id,
source_url=source_url, name=name, description=description,
aliases=aliases, fish_species=tuple(fish_species),
fish_external_ids=tuple(fish_external_ids), image_urls=image_urls,
point_urls=point_urls,
)
def parse_rf4db_catches(html: str, *, base_url: str = "https://rf4db.com") -> list[ExternalCatch]:
soup = BeautifulSoup(html, "html.parser")
result: list[ExternalCatch] = []
for card in soup.select("article.catch-card"):
detail = card.select_one('a.catch-card__time[href*="/catches/"]')
fish_link = card.select_one("h2 a")
map_link = card.select_one('.catch-card__place a[href*="/maps/"]')
if not detail or not fish_link or not map_link:
continue
external_id = _key(detail.get("href"))
if not external_id:
continue
coordinate_raw = _text(card.select_one(".catch-card__place b")) or None
x, y = _coordinates(coordinate_raw or "")
bait_link = card.select_one('.catch-card__place a[href*="/wiki/baits/"]')
badges = card.select(".catch-badge")
weather = next((_text(b.select_one("b")) for b in badges if _text(b).startswith("Погода")), None)
temperature_text = next((_text(b.select_one("b")) for b in badges if _text(b).startswith("Темп. воды")), "")
temperature_match = re.search(r"-?\d+(?:[.,]\d+)?", temperature_text)
rig = card.select_one(".catch-badge--rig")
result.append(ExternalCatch(
source_system="rf4db", source_external_id=external_id,
source_url=urljoin(base_url, str(detail.get("href"))),
fish=_text(fish_link), fish_external_id=_key(fish_link.get("href")),
waterbody=_text(map_link), waterbody_external_id=_key(map_link.get("href")),
x=x, y=y, weight_g=None, bait=_text(bait_link) or None,
rig_type=_text(rig) or None,
game_time=_game_time(_text(card.select_one(".catch-card__place"))),
published_at=None, player_name=None, weather=weather or None,
water_temperature_c=float(temperature_match.group().replace(",", ".")) if temperature_match else None,
clip=None, fishing_style=None, evidence_urls=(), coordinate_raw=coordinate_raw,
))
if not result:
raise CommunityParseError("RF4DB catch cards not found")
return result
def parse_rf4db_detail(html: str, *, source_url: str) -> RF4DBCatchDetail:
soup = BeautifulSoup(html, "html.parser")
root = soup.select_one("article.catch-detail")
if root is None:
raise CommunityParseError("RF4DB catch detail not found")
external_id = _key(source_url)
if not external_id:
raise CommunityParseError("RF4DB catch detail URL has no ID")
facts = {_text(row.select_one("dt")): _text(row.select_one("dd")) for row in root.select(".catch-facts > div")}
def number(label: str) -> float | None:
match = re.search(r"-?\d+(?:[.,]\d+)?", facts.get(label, ""))
return float(match.group().replace(",", ".")) if match else None
equipment: list[EquipmentItem] = []
for item in root.select(".catch-equipment article"):
link = item.select_one("a[href]")
name = _text(link)
if not name:
continue
equipment.append(EquipmentItem(
kind=_text(item.select_one("small")) or "Снаряжение",
name=name,
external_id=_key(link.get("href")) if link else None,
))
return RF4DBCatchDetail(
source_external_id=external_id,
source_url=source_url,
wind=facts.get("Ветер") or None,
line_release_m=number("Выпуск лески"),
clip_m=number("Клипса"),
cast_direction_deg=number("Направление заброса"),
equipment=tuple(equipment),
)
def _rf4stat_date(raw_date: str, raw_time: str, *, now: datetime) -> datetime | None:
try:
day, month = (int(part) for part in raw_date.split("."))
hour, minute = (int(part) for part in raw_time.split(":"))
value = datetime(now.year, month, day, hour, minute, tzinfo=timezone.utc)
if value > now:
value = value.replace(year=value.year - 1)
return value
except (TypeError, ValueError):
return None
def parse_rf4stat_fishing(
html: str, *, base_url: str = "https://rf4-stat.ru/", now: datetime | None = None,
) -> list[ExternalCatch]:
soup = BeautifulSoup(html, "html.parser")
now = now or datetime.now(timezone.utc)
result: list[ExternalCatch] = []
for row in soup.select("tr.load-row"):
id_link = row.select_one('a.share[href^="fishing/"]:not(.hide-print)')
fish_link = row.select_one('.fish-icon a[href^="fish/"]')
if not id_link or not fish_link:
continue
external_id = _key(id_link.get("href"))
if not external_id:
continue
position = row.select_one(".list-position .position:not(.position-locked)")
coordinate_raw = _text(row.select_one(".list-position .position")) or None
x, y = _coordinates(_text(position))
style = row.select_one(".post-style-icon[title]")
style_text = str(style.get("title", "")) if style else ""
result.append(ExternalCatch(
source_system="rf4stat-fishing", source_external_id=external_id,
source_url=urljoin(base_url, str(id_link.get("href"))),
fish=_text(row.select_one("td.fish a")), fish_external_id=_key(fish_link.get("href")),
waterbody=_text(row.select_one("td.mobile-location")), waterbody_external_id=None,
x=x, y=y, weight_g=_weight(_text(row.select_one("td.fish small"))),
bait=_text(row.select_one('[data-filter="bait"]')) or None,
rig_type=None, game_time=None,
published_at=_rf4stat_date(_text(row.select_one("td.time small")), _text(row.select_one("td.time > div")), now=now),
player_name=_text(row.select_one("td.list-gamer a")) or None,
weather=None, water_temperature_c=None,
clip=_text(row.select_one(".clip")) or None,
fishing_style=style_text.removeprefix("Вид ловли:").strip() or None,
evidence_urls=(urljoin(base_url, str(row.select_one("a.share.hide-print").get("href"))),) if row.select_one("a.share.hide-print") else (),
coordinate_raw=coordinate_raw,
))
if not result:
raise CommunityParseError("RF4-STAT fishing rows not found")
return result
def parse_rf4stat_posts(
html: str, *, base_url: str = "https://rf4-stat.ru/",
) -> list[ExternalCatch]:
soup = BeautifulSoup(html, "html.parser")
result: list[ExternalCatch] = []
for post in soup.select(".post-item.load-row"):
post_id = str(post.get("data-post-id", "")).strip()
if not post_id:
continue
published_raw = str(post.get("data-published-at", ""))
published_at = datetime.fromtimestamp(int(published_raw), tz=timezone.utc) if published_raw.isdigit() else None
position = post.select_one(".spot-col .position:not(.position-locked)")
coordinate_raw = _text(post.select_one(".spot-col .position")) or None
x, y = _coordinates(_text(position))
style = post.select_one(".post-style-icon[title]")
style_text = str(style.get("title", "")) if style else ""
evidence = tuple(urljoin(base_url, str(image.get("src"))) for image in post.select(".screen-file-slot img[src]"))
baits = _text(post.select_one(".bait-text")) or None
for index, fish in enumerate(post.select(".fish-icon")):
fish_link = fish.select_one('a[href^="fish/"]')
fish_name = _text(fish.select_one(".fish-name"))
if not fish_name:
continue
result.append(ExternalCatch(
source_system="rf4stat-post", source_external_id=f"{post_id}:{index}",
source_url=urljoin(base_url, f"posts/{post_id}"),
fish=fish_name, fish_external_id=_key(fish_link.get("href")) if fish_link else None,
waterbody=_text(post.select_one(".location")), waterbody_external_id=None,
x=x, y=y, weight_g=_weight(_text(fish.select_one(".weight"))),
bait=baits, rig_type=None, game_time=None, published_at=published_at,
player_name=_text(post.select_one(".player")) or None,
weather=None, water_temperature_c=None,
clip=_text(post.select_one(".clip")) or None,
fishing_style=style_text.removeprefix("Вид ловли:").strip() or None,
evidence_urls=evidence, coordinate_raw=coordinate_raw,
))
if not result:
raise CommunityParseError("RF4-STAT posts not found")
return result
def parse_rf4map_point(html: str, *, source_url: str) -> list[ExternalCatch]:
soup = BeautifulSoup(html, "html.parser")
lake_link = soup.select_one('a[href^="/lakes/"]')
waterbody = _text(lake_link)
waterbody_id = _key(lake_link.get("href")) if lake_link else None
points = _json_after_marker(_next_payloads(html), '"points":')
items = points.get("items") if points else None
if not waterbody or not waterbody_id or not isinstance(items, list):
raise CommunityParseError("RF4MAP point payload not found")
result: list[ExternalCatch] = []
for item in items:
if not isinstance(item, dict) or not isinstance(item.get("id"), int):
continue
fish = item.get("fish")
bait = item.get("bait")
evidence = item.get("imageUrls")
if not isinstance(fish, dict) or not isinstance(fish.get("name"), str):
continue
if not isinstance(evidence, list):
evidence = []
result.append(ExternalCatch(
source_system="rf4map", source_external_id=str(item["id"]), source_url=source_url,
fish=str(fish["name"]), fish_external_id=str(fish["id"]) if isinstance(fish.get("id"), int) else None,
waterbody=waterbody, waterbody_external_id=waterbody_id,
x=item.get("positionX") if isinstance(item.get("positionX"), int) else None,
y=item.get("positionY") if isinstance(item.get("positionY"), int) else None,
weight_g=None,
bait=str(bait["name"]) if isinstance(bait, dict) and isinstance(bait.get("name"), str) else None,
rig_type=None, game_time=None, published_at=_datetime(item.get("createdAt")),
player_name=item.get("authorName") if isinstance(item.get("authorName"), str) else None,
weather=None, water_temperature_c=None,
clip=str(item["clip"]) if isinstance(item.get("clip"), (int, float)) else None,
fishing_style=None,
evidence_urls=tuple(url for url in evidence or [] if isinstance(url, str)),
coordinate_raw=f"{item['positionX']}:{item['positionY']}" if isinstance(item.get("positionX"), int) and isinstance(item.get("positionY"), int) else None,
))
if not result:
raise CommunityParseError("RF4MAP point observations not found")
return result
def parse_rf4posts_spot(html: str, *, source_url: str) -> list[ExternalCatch]:
soup = BeautifulSoup(html, "html.parser")
spot = _json_after_marker(_next_payloads(html), '"spot":', required_key="id")
species = spot.get("fishSpecies") if spot else None
spot_id = spot.get("id") if spot else None
if not isinstance(spot_id, str) or not isinstance(species, list) or not species:
raise CommunityParseError("RF4 Posts spot payload not found")
waterbody = _text(soup.select_one("h1"))
fish_list = soup.select_one('[role="list"][aria-label]')
fish_names = [_text(node) for node in fish_list.select('[role="listitem"]')] if fish_list else []
if not waterbody or len(fish_names) != len(species):
raise CommunityParseError("RF4 Posts localized spot labels not found")
x, y = _coordinates(str(spot.get("coordinates", "")))
screenshots = spot.get("screenshots")
if not isinstance(screenshots, list):
screenshots = []
evidence = tuple(
str(item["url"]) for item in screenshots or []
if isinstance(item, dict) and isinstance(item.get("url"), str)
)
result: list[ExternalCatch] = []
for fish_id, fish_name in zip(species, fish_names, strict=True):
if not isinstance(fish_id, str):
continue
result.append(ExternalCatch(
source_system="rf4posts-spot", source_external_id=f"{spot_id}:{fish_id}", source_url=source_url,
fish=fish_name, fish_external_id=fish_id,
waterbody=waterbody,
waterbody_external_id=str(spot["waterBody"]) if isinstance(spot.get("waterBody"), str) else None,
x=x, y=y, weight_g=None, bait=None,
rig_type=str(spot["bottomRigType"]) if isinstance(spot.get("bottomRigType"), str) else None,
game_time=None, published_at=_datetime(spot.get("createdAt")), player_name=None,
weather=None, water_temperature_c=None,
clip=str(spot["clip"]) if isinstance(spot.get("clip"), (int, float)) else None,
fishing_style=str(spot["tackleType"]) if isinstance(spot.get("tackleType"), str) else None,
evidence_urls=evidence, coordinate_raw=str(spot["coordinates"]) if isinstance(spot.get("coordinates"), str) else None,
))
if not result:
raise CommunityParseError("RF4 Posts fish species not found")
return result