from __future__ import annotations import json import re from dataclasses import dataclass from datetime import datetime, time, timezone from urllib.parse import urljoin from bs4 import BeautifulSoup, Tag class CommunityParseError(ValueError): pass @dataclass(frozen=True, slots=True) class ExternalCatch: source_system: str source_external_id: str source_url: str fish: str fish_external_id: str | None waterbody: str waterbody_external_id: str | None x: int | None y: int | None weight_g: int | None bait: str | None rig_type: str | None game_time: time | None published_at: datetime | None player_name: str | None weather: str | None water_temperature_c: float | None clip: str | None fishing_style: str | None evidence_urls: tuple[str, ...] coordinate_raw: str | None = None @dataclass(frozen=True, slots=True) class EquipmentItem: kind: str name: str external_id: str | None GEAR_CATEGORIES = frozenset({ "bait", "lure", "rod", "reel", "line", "hook", "rig", "float", "sinker", "other", }) @dataclass(frozen=True, slots=True) class RF4DBGearItem: source_system: str source_external_id: str source_url: str name: str category: str subcategory: str | None brand: str | None family: str | None unlock_level: int | None image_url: str | None @dataclass(frozen=True, slots=True) class RF4DBGearAttribute: key: str state: str value: str | int | float | None source_text: str | None @dataclass(frozen=True, slots=True) class RF4DBGearDetail: source_system: str source_external_id: str source_url: str name: str category: str subcategory: str | None brand: str | None family: str | None unlock_level: int | None attributes: tuple[RF4DBGearAttribute, ...] variants: tuple[str, ...] compatible_with: tuple[str, ...] rig_types: tuple[str, ...] image_urls: tuple[str, ...] @dataclass(frozen=True, slots=True) class RF4DBCatchDetail: source_external_id: str source_url: str wind: str | None line_release_m: float | None clip_m: float | None cast_direction_deg: float | None equipment: tuple[EquipmentItem, ...] @dataclass(frozen=True, slots=True) class RF4DBWaterbody: source_system: str source_external_id: str source_url: str name: str unlock_level: int | None unlock_label: str fish_species_count: int image_url: str | None @dataclass(frozen=True, slots=True) class RF4DBWaterbodyDetail: source_system: str source_external_id: str source_url: str name: str description: str | None aliases: tuple[str, ...] fish_species: tuple[str, ...] fish_external_ids: tuple[str | None, ...] image_urls: tuple[str, ...] point_urls: tuple[str, ...] def _text(node: Tag | None) -> str: return " ".join(node.get_text(" ", strip=True).split()) if node else "" def _fish_name(node: Tag) -> str: """Read a fish label while dropping the optional trophy weight suffix.""" return re.sub(r"\s*\d+(?:[.,]\d+)?\s*(?:кг|kg|г|g)\s*$", "", _text(node), flags=re.I).strip() def _key(href: str | None) -> str | None: if not href: return None return href.rstrip("/").rsplit("/", 1)[-1] def _coordinates(raw: str) -> tuple[int | None, int | None]: match = re.fullmatch(r"\s*(-?\d{1,5}):(-?\d{1,5})\s*", raw) return (int(match.group(1)), int(match.group(2))) if match else (None, None) def _game_time(raw: str) -> time | None: match = re.search(r"Игровое время\s*(\d{1,2}):(\d{2})", raw) return time(int(match.group(1)), int(match.group(2))) if match else None def _weight(raw: str) -> int | None: match = re.search(r"([\d\s.,]+)\s*(кг|kg|г|g)\b", raw.casefold()) if not match: return None number, unit = match.groups() compact = number.replace(" ", "").replace(",", ".") return round(float(compact) * 1000) if unit in {"кг", "kg"} else int(compact.replace(".", "")) def _next_payloads(html: str) -> list[str]: """Decode the string payloads emitted by Next.js' server components.""" soup = BeautifulSoup(html, "html.parser") result: list[str] = [] pattern = re.compile(r'^self\.__next_f\.push\(\[1,("(?:\\.|[^"\\])*")\]\)$', re.DOTALL) for script in soup.find_all("script"): raw = script.string or "" match = pattern.fullmatch(raw.strip()) if not match: continue try: result.append(json.loads(match.group(1))) except json.JSONDecodeError: continue return result def _json_after_marker( payloads: list[str], marker: str, *, required_key: str | None = None, ) -> dict[str, object] | None: decoder = json.JSONDecoder() for payload in payloads: offset = 0 while (start := payload.find(marker, offset)) >= 0: offset = start + len(marker) try: value, _ = decoder.raw_decode(payload[offset:]) except json.JSONDecodeError: continue if isinstance(value, dict) and (required_key is None or required_key in value): return value return None def _datetime(raw: object) -> datetime | None: if not isinstance(raw, str): return None try: return datetime.fromisoformat(raw.replace("Z", "+00:00")) except ValueError: return None def parse_rf4db_waterbodies( html: str, *, source_url: str = "https://rf4db.com/ru/maps", expected_count: int = 19, ) -> list[RF4DBWaterbody]: """Parse the public RF4DB waterbody index without assigning image roles.""" soup = BeautifulSoup(html, "html.parser") result: list[RF4DBWaterbody] = [] seen: set[str] = set() for link in soup.select('a[href*="/ru/maps/level_"]'): href = link.get("href") if not isinstance(href, str): continue external_id = _key(href) if not external_id or external_id in seen: continue name = _text(link) if not name: continue card = link.find_parent(["article", "li"]) if card is None: card = link.find_parent("div") card_text = _text(card) level_match = re.search(r"(?:Уровень|Level)\s*(?:Lv\.?\s*)?(Старт|Start|\d+)", card_text, re.I) fish_match = re.search(r"(?:Рыбы|Fish(?:es)?)\s*(\d+)", card_text, re.I) if not level_match or not fish_match: continue unlock_label = level_match.group(1) unlock_level = int(unlock_label) if unlock_label.isdigit() else None image = card.select_one("img[src], img[data-src]") if card else None image_raw = image.get("src") or image.get("data-src") if image else None image_url = urljoin(source_url, str(image_raw)) if image_raw else None result.append(RF4DBWaterbody( source_system="rf4db", source_external_id=external_id, source_url=urljoin(source_url, href), name=name, unlock_level=unlock_level, unlock_label=unlock_label, fish_species_count=int(fish_match.group(1)), image_url=image_url, )) seen.add(external_id) if not result: raise CommunityParseError("RF4DB waterbody cards not found") if len(result) != expected_count: raise CommunityParseError( f"RF4DB waterbody catalog is incomplete: expected {expected_count}, got {len(result)}" ) return result def _gear_level(value: str | None) -> int | None: text = (value or "").strip() if not text: return None if not re.fullmatch(r"\d+", text): raise CommunityParseError(f"invalid gear unlock level: {text!r}") return int(text) def _gear_category(value: str | None) -> str: category = (value or "").strip().casefold() if category not in GEAR_CATEGORIES: raise CommunityParseError(f"invalid or missing gear category: {value!r}") return category def _gear_value(node: Tag) -> str | int | float | None: text = _text(node) or None if text is None: return None value_type = str(node.get("data-type") or "text").casefold() if value_type == "number": try: return float(text) if "." in text else int(text) except ValueError as exc: raise CommunityParseError(f"invalid numeric gear attribute: {text!r}") from exc return text def parse_rf4db_gear_catalog( html: str, *, source_url: str = "https://rf4db.com/ru/wiki/gear", expected_count: int | None = None, ) -> list[RF4DBGearItem]: """Parse a gear index while keeping an unknown total explicitly unknown.""" soup = BeautifulSoup(html, "html.parser") cards = soup.select("article.gear-card, [data-gear-card]") result: list[RF4DBGearItem] = [] seen: set[str] = set() for card in cards: link = card.select_one("a[href]") external_id = str(card.get("data-gear-id") or _key(link.get("href") if link else None) or "") name = _text(card.select_one("[data-name], h2, h3, .gear-name")) category = _gear_category(card.get("data-category")) if not external_id or not name: raise CommunityParseError("RF4DB gear card is missing id or name") if external_id in seen: raise CommunityParseError(f"duplicate RF4DB gear id: {external_id}") seen.add(external_id) item_url = urljoin(source_url, str(link.get("href"))) if link and link.get("href") else source_url image = card.select_one("img[src], img[data-src]") image_url = urljoin(item_url, str(image.get("src") or image.get("data-src"))) if image else None result.append(RF4DBGearItem( source_system="rf4db", source_external_id=external_id, source_url=item_url, name=name, category=category, subcategory=_text(card.select_one("[data-subcategory], .gear-subcategory")) or None, brand=_text(card.select_one("[data-brand], .gear-brand")) or None, family=_text(card.select_one("[data-family], .gear-family")) or None, unlock_level=_gear_level(card.get("data-unlock-level")), image_url=image_url, )) if not result: raise CommunityParseError("RF4DB gear cards not found") if expected_count is not None and len(result) != expected_count: raise CommunityParseError(f"RF4DB gear catalog count mismatch: expected {expected_count}, got {len(result)}") return result def parse_rf4db_gear_detail( html: str, *, source_url: str, ) -> RF4DBGearDetail: """Parse one gear detail page with explicit missing/not-applicable/value states.""" soup = BeautifulSoup(html, "html.parser") root = soup.select_one("main[data-gear-detail], article.gear-detail") or soup external_id = _key(source_url) name = _text(root.select_one("h1, [data-name]")) category = _gear_category(root.get("data-category")) if not external_id or not name: raise CommunityParseError("RF4DB gear detail is missing id or name") attributes: list[RF4DBGearAttribute] = [] for node in root.select("[data-gear-attribute]"): key = str(node.get("data-gear-attribute") or "").strip() state = str(node.get("data-state") or "value").strip().casefold() if not key or state not in {"value", "not_applicable", "missing"}: raise CommunityParseError("invalid RF4DB gear attribute state") value = None if state != "value" else _gear_value(node) attributes.append(RF4DBGearAttribute(key=key, state=state, value=value, source_text=_text(node) or None)) images = tuple(dict.fromkeys( urljoin(source_url, str(node.get("src") or node.get("data-src"))) for node in root.select("img[src], img[data-src]") if node.get("src") or node.get("data-src") )) return RF4DBGearDetail( source_system="rf4db", source_external_id=external_id, source_url=source_url, name=name, category=category, subcategory=_text(root.select_one("[data-subcategory], .gear-subcategory")) or None, brand=_text(root.select_one("[data-brand], .gear-brand")) or None, family=_text(root.select_one("[data-family], .gear-family")) or None, unlock_level=_gear_level(root.get("data-unlock-level")), attributes=tuple(attributes), variants=tuple(dict.fromkeys(_text(node) for node in root.select("[data-gear-variant]") if _text(node))), compatible_with=tuple(dict.fromkeys(_text(node) for node in root.select("[data-compatible-with]") if _text(node))), rig_types=tuple(dict.fromkeys(_text(node) for node in root.select("[data-rig-type]") if _text(node))), image_urls=images, ) def parse_rf4db_waterbody_detail( html: str, *, source_url: str, ) -> RF4DBWaterbodyDetail: """Parse one RF4DB waterbody page without assigning media or coordinates. The detail page is accepted only when its localized heading and fish list are present. Images and point links remain source candidates; review and canonical crosswalks happen in later pipeline stages. """ soup = BeautifulSoup(html, "html.parser") root = soup.select_one("article.waterbody-detail, main[data-waterbody-detail]") or soup external_id = _key(source_url) name = _text(root.select_one("h1")) fish_nodes = root.select(".waterbody-fish a[href], [data-fish-list] a[href], a[href*='/fishes/']") if not fish_nodes: fish_nodes = root.select(".fish-list a[href], ul.fish a[href]") fish_species: list[str] = [] fish_external_ids: list[str | None] = [] seen_fish: set[str] = set() for node in fish_nodes: fish_name = _fish_name(node) fish_id = _key(node.get("href")) identity = fish_id or fish_name.casefold() if not fish_name or identity in seen_fish: continue fish_species.append(fish_name) fish_external_ids.append(fish_id) seen_fish.add(identity) if not external_id or not name or not fish_species: raise CommunityParseError("RF4DB waterbody detail not found or incomplete") description_node = root.select_one("[data-description], .waterbody-description, .description") description = _text(description_node) or None aliases = tuple(dict.fromkeys( _text(node) for node in root.select("[data-alias], .waterbody-aliases li, .aliases li") if _text(node) and _text(node) != name )) image_urls = tuple(dict.fromkeys( urljoin(source_url, str(node.get("src") or node.get("data-src"))) for node in root.select("img[src], img[data-src]") if (node.get("src") or node.get("data-src")) and "/fish/" not in str(node.get("src") or node.get("data-src")) )) point_urls = tuple(dict.fromkeys( urljoin(source_url, str(node.get("href"))) for node in root.select('a[href*="/spots/"], a[href*="/points/"]') if node.get("href") )) return RF4DBWaterbodyDetail( source_system="rf4db", source_external_id=external_id, source_url=source_url, name=name, description=description, aliases=aliases, fish_species=tuple(fish_species), fish_external_ids=tuple(fish_external_ids), image_urls=image_urls, point_urls=point_urls, ) def parse_rf4db_catches(html: str, *, base_url: str = "https://rf4db.com") -> list[ExternalCatch]: soup = BeautifulSoup(html, "html.parser") result: list[ExternalCatch] = [] for card in soup.select("article.catch-card"): detail = card.select_one('a.catch-card__time[href*="/catches/"]') fish_link = card.select_one("h2 a") map_link = card.select_one('.catch-card__place a[href*="/maps/"]') if not detail or not fish_link or not map_link: continue external_id = _key(detail.get("href")) if not external_id: continue coordinate_raw = _text(card.select_one(".catch-card__place b")) or None x, y = _coordinates(coordinate_raw or "") bait_link = card.select_one('.catch-card__place a[href*="/wiki/baits/"]') badges = card.select(".catch-badge") weather = next((_text(b.select_one("b")) for b in badges if _text(b).startswith("Погода")), None) temperature_text = next((_text(b.select_one("b")) for b in badges if _text(b).startswith("Темп. воды")), "") temperature_match = re.search(r"-?\d+(?:[.,]\d+)?", temperature_text) rig = card.select_one(".catch-badge--rig") result.append(ExternalCatch( source_system="rf4db", source_external_id=external_id, source_url=urljoin(base_url, str(detail.get("href"))), fish=_text(fish_link), fish_external_id=_key(fish_link.get("href")), waterbody=_text(map_link), waterbody_external_id=_key(map_link.get("href")), x=x, y=y, weight_g=None, bait=_text(bait_link) or None, rig_type=_text(rig) or None, game_time=_game_time(_text(card.select_one(".catch-card__place"))), published_at=None, player_name=None, weather=weather or None, water_temperature_c=float(temperature_match.group().replace(",", ".")) if temperature_match else None, clip=None, fishing_style=None, evidence_urls=(), coordinate_raw=coordinate_raw, )) if not result: raise CommunityParseError("RF4DB catch cards not found") return result def parse_rf4db_detail(html: str, *, source_url: str) -> RF4DBCatchDetail: soup = BeautifulSoup(html, "html.parser") root = soup.select_one("article.catch-detail") if root is None: raise CommunityParseError("RF4DB catch detail not found") external_id = _key(source_url) if not external_id: raise CommunityParseError("RF4DB catch detail URL has no ID") facts = {_text(row.select_one("dt")): _text(row.select_one("dd")) for row in root.select(".catch-facts > div")} def number(label: str) -> float | None: match = re.search(r"-?\d+(?:[.,]\d+)?", facts.get(label, "")) return float(match.group().replace(",", ".")) if match else None equipment: list[EquipmentItem] = [] for item in root.select(".catch-equipment article"): link = item.select_one("a[href]") name = _text(link) if not name: continue equipment.append(EquipmentItem( kind=_text(item.select_one("small")) or "Снаряжение", name=name, external_id=_key(link.get("href")) if link else None, )) return RF4DBCatchDetail( source_external_id=external_id, source_url=source_url, wind=facts.get("Ветер") or None, line_release_m=number("Выпуск лески"), clip_m=number("Клипса"), cast_direction_deg=number("Направление заброса"), equipment=tuple(equipment), ) def _rf4stat_date(raw_date: str, raw_time: str, *, now: datetime) -> datetime | None: try: day, month = (int(part) for part in raw_date.split(".")) hour, minute = (int(part) for part in raw_time.split(":")) value = datetime(now.year, month, day, hour, minute, tzinfo=timezone.utc) if value > now: value = value.replace(year=value.year - 1) return value except (TypeError, ValueError): return None def parse_rf4stat_fishing( html: str, *, base_url: str = "https://rf4-stat.ru/", now: datetime | None = None, ) -> list[ExternalCatch]: soup = BeautifulSoup(html, "html.parser") now = now or datetime.now(timezone.utc) result: list[ExternalCatch] = [] for row in soup.select("tr.load-row"): id_link = row.select_one('a.share[href^="fishing/"]:not(.hide-print)') fish_link = row.select_one('.fish-icon a[href^="fish/"]') if not id_link or not fish_link: continue external_id = _key(id_link.get("href")) if not external_id: continue position = row.select_one(".list-position .position:not(.position-locked)") coordinate_raw = _text(row.select_one(".list-position .position")) or None x, y = _coordinates(_text(position)) style = row.select_one(".post-style-icon[title]") style_text = str(style.get("title", "")) if style else "" result.append(ExternalCatch( source_system="rf4stat-fishing", source_external_id=external_id, source_url=urljoin(base_url, str(id_link.get("href"))), fish=_text(row.select_one("td.fish a")), fish_external_id=_key(fish_link.get("href")), waterbody=_text(row.select_one("td.mobile-location")), waterbody_external_id=None, x=x, y=y, weight_g=_weight(_text(row.select_one("td.fish small"))), bait=_text(row.select_one('[data-filter="bait"]')) or None, rig_type=None, game_time=None, published_at=_rf4stat_date(_text(row.select_one("td.time small")), _text(row.select_one("td.time > div")), now=now), player_name=_text(row.select_one("td.list-gamer a")) or None, weather=None, water_temperature_c=None, clip=_text(row.select_one(".clip")) or None, fishing_style=style_text.removeprefix("Вид ловли:").strip() or None, evidence_urls=(urljoin(base_url, str(row.select_one("a.share.hide-print").get("href"))),) if row.select_one("a.share.hide-print") else (), coordinate_raw=coordinate_raw, )) if not result: raise CommunityParseError("RF4-STAT fishing rows not found") return result def parse_rf4stat_posts( html: str, *, base_url: str = "https://rf4-stat.ru/", ) -> list[ExternalCatch]: soup = BeautifulSoup(html, "html.parser") result: list[ExternalCatch] = [] for post in soup.select(".post-item.load-row"): post_id = str(post.get("data-post-id", "")).strip() if not post_id: continue published_raw = str(post.get("data-published-at", "")) published_at = datetime.fromtimestamp(int(published_raw), tz=timezone.utc) if published_raw.isdigit() else None position = post.select_one(".spot-col .position:not(.position-locked)") coordinate_raw = _text(post.select_one(".spot-col .position")) or None x, y = _coordinates(_text(position)) style = post.select_one(".post-style-icon[title]") style_text = str(style.get("title", "")) if style else "" evidence = tuple(urljoin(base_url, str(image.get("src"))) for image in post.select(".screen-file-slot img[src]")) baits = _text(post.select_one(".bait-text")) or None for index, fish in enumerate(post.select(".fish-icon")): fish_link = fish.select_one('a[href^="fish/"]') fish_name = _text(fish.select_one(".fish-name")) if not fish_name: continue result.append(ExternalCatch( source_system="rf4stat-post", source_external_id=f"{post_id}:{index}", source_url=urljoin(base_url, f"posts/{post_id}"), fish=fish_name, fish_external_id=_key(fish_link.get("href")) if fish_link else None, waterbody=_text(post.select_one(".location")), waterbody_external_id=None, x=x, y=y, weight_g=_weight(_text(fish.select_one(".weight"))), bait=baits, rig_type=None, game_time=None, published_at=published_at, player_name=_text(post.select_one(".player")) or None, weather=None, water_temperature_c=None, clip=_text(post.select_one(".clip")) or None, fishing_style=style_text.removeprefix("Вид ловли:").strip() or None, evidence_urls=evidence, coordinate_raw=coordinate_raw, )) if not result: raise CommunityParseError("RF4-STAT posts not found") return result def parse_rf4map_point(html: str, *, source_url: str) -> list[ExternalCatch]: soup = BeautifulSoup(html, "html.parser") lake_link = soup.select_one('a[href^="/lakes/"]') waterbody = _text(lake_link) waterbody_id = _key(lake_link.get("href")) if lake_link else None points = _json_after_marker(_next_payloads(html), '"points":') items = points.get("items") if points else None if not waterbody or not waterbody_id or not isinstance(items, list): raise CommunityParseError("RF4MAP point payload not found") result: list[ExternalCatch] = [] for item in items: if not isinstance(item, dict) or not isinstance(item.get("id"), int): continue fish = item.get("fish") bait = item.get("bait") evidence = item.get("imageUrls") if not isinstance(fish, dict) or not isinstance(fish.get("name"), str): continue if not isinstance(evidence, list): evidence = [] result.append(ExternalCatch( source_system="rf4map", source_external_id=str(item["id"]), source_url=source_url, fish=str(fish["name"]), fish_external_id=str(fish["id"]) if isinstance(fish.get("id"), int) else None, waterbody=waterbody, waterbody_external_id=waterbody_id, x=item.get("positionX") if isinstance(item.get("positionX"), int) else None, y=item.get("positionY") if isinstance(item.get("positionY"), int) else None, weight_g=None, bait=str(bait["name"]) if isinstance(bait, dict) and isinstance(bait.get("name"), str) else None, rig_type=None, game_time=None, published_at=_datetime(item.get("createdAt")), player_name=item.get("authorName") if isinstance(item.get("authorName"), str) else None, weather=None, water_temperature_c=None, clip=str(item["clip"]) if isinstance(item.get("clip"), (int, float)) else None, fishing_style=None, evidence_urls=tuple(url for url in evidence or [] if isinstance(url, str)), coordinate_raw=f"{item['positionX']}:{item['positionY']}" if isinstance(item.get("positionX"), int) and isinstance(item.get("positionY"), int) else None, )) if not result: raise CommunityParseError("RF4MAP point observations not found") return result def parse_rf4posts_spot(html: str, *, source_url: str) -> list[ExternalCatch]: soup = BeautifulSoup(html, "html.parser") spot = _json_after_marker(_next_payloads(html), '"spot":', required_key="id") species = spot.get("fishSpecies") if spot else None spot_id = spot.get("id") if spot else None if not isinstance(spot_id, str) or not isinstance(species, list) or not species: raise CommunityParseError("RF4 Posts spot payload not found") waterbody = _text(soup.select_one("h1")) fish_list = soup.select_one('[role="list"][aria-label]') fish_names = [_text(node) for node in fish_list.select('[role="listitem"]')] if fish_list else [] if not waterbody or len(fish_names) != len(species): raise CommunityParseError("RF4 Posts localized spot labels not found") x, y = _coordinates(str(spot.get("coordinates", ""))) screenshots = spot.get("screenshots") if not isinstance(screenshots, list): screenshots = [] evidence = tuple( str(item["url"]) for item in screenshots or [] if isinstance(item, dict) and isinstance(item.get("url"), str) ) result: list[ExternalCatch] = [] for fish_id, fish_name in zip(species, fish_names, strict=True): if not isinstance(fish_id, str): continue result.append(ExternalCatch( source_system="rf4posts-spot", source_external_id=f"{spot_id}:{fish_id}", source_url=source_url, fish=fish_name, fish_external_id=fish_id, waterbody=waterbody, waterbody_external_id=str(spot["waterBody"]) if isinstance(spot.get("waterBody"), str) else None, x=x, y=y, weight_g=None, bait=None, rig_type=str(spot["bottomRigType"]) if isinstance(spot.get("bottomRigType"), str) else None, game_time=None, published_at=_datetime(spot.get("createdAt")), player_name=None, weather=None, water_temperature_c=None, clip=str(spot["clip"]) if isinstance(spot.get("clip"), (int, float)) else None, fishing_style=str(spot["tackleType"]) if isinstance(spot.get("tackleType"), str) else None, evidence_urls=evidence, coordinate_raw=str(spot["coordinates"]) if isinstance(spot.get("coordinates"), str) else None, )) if not result: raise CommunityParseError("RF4 Posts fish species not found") return result