import json from pathlib import Path import pytest from rf4_research import community_cli from rf4_research.community_cli import enforce_fetch_interval, fetch_site_key, mark_fetch from rf4_research.community_sources import RF4DBWaterbodyDetail def test_fetch_cooldown_is_persistent_per_source(tmp_path: Path) -> None: """A02: check_and_reserve is atomic — one reservation per interval per source.""" from rf4_research.community_cli import check_and_reserve state_file = tmp_path / "fetch-state.json" check_and_reserve("rf4map-point", state_file=state_file, now=1_000) # Second call for same source within interval should fail with pytest.raises(RuntimeError, match="retry in 1800 seconds"): check_and_reserve("rf4map-point", state_file=state_file, now=1_000) # Different source should succeed check_and_reserve("rf4posts-spot", state_file=state_file, now=1_000) # After interval expires, should succeed again check_and_reserve("rf4map-point", state_file=state_file, now=2_800) def test_fetch_site_key_groups_endpoints_and_normalizes_www() -> None: assert fetch_site_key("https://rf4-stat.ru/fishing/") == "rf4-stat.ru" assert fetch_site_key("https://www.rf4-stat.ru/posts/") == "rf4-stat.ru" def test_fetch_site_key_normalizes_common_subdomains() -> None: """A02: download/api/cdn subdomains share the base domain key.""" assert fetch_site_key("https://download.rf4db.com/ru/catches") == "rf4db.com" assert fetch_site_key("https://api.rf4db.com/v1/catches") == "rf4db.com" assert fetch_site_key("https://cdn.rf4db.com/assets/img.jpg") == "rf4db.com" assert fetch_site_key("https://oss.rf4db.com/game/fish/pike.webp") == "rf4db.com" assert fetch_site_key("https://www.rf4db.com/") == "rf4db.com" # Base domain stays the same assert fetch_site_key("https://rf4db.com/") == "rf4db.com" assert fetch_site_key("https://gw.rf4map.ru/public/images/fish.webp") == "rf4map.ru" # Non-normalized subdomains stay as-is assert fetch_site_key("https://rf4map.ru/point/1") == "rf4map.ru" assert fetch_site_key("https://rf4-posts.com/spots/") == "rf4-posts.com" def test_failed_fetch_still_reserves_site_cooldown(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: state_file = tmp_path / "fetch-state.json" def fail(_url: str) -> str: raise OSError("offline") monkeypatch.setattr(community_cli, "fetch_html", fail) assert community_cli.main(["rf4db", "--state-file", str(state_file)]) == 1 # A02: download.rf4db.com normalized to rf4db.com assert "rf4db.com" in json.loads(state_file.read_text(encoding="utf-8")) def test_reserve_only_does_not_make_http_request( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], ) -> None: state_file = tmp_path / "fetch-state.json" def fail(_url: str) -> str: raise AssertionError("reserve-only must stop before HTTP") monkeypatch.setattr(community_cli, "fetch_html", fail) assert community_cli.main([ "rf4db-waterbodies", "--state-file", str(state_file), "--reserve-only", ]) == 0 payload = json.loads(capsys.readouterr().out) assert payload == {"reserved": True, "source": "rf4db-waterbodies", "site_key": "rf4db.com"} def test_detail_fetch_serializes_single_record( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], ) -> None: state_file = tmp_path / "fetch-state.json" source_url = "https://download.rf4db.com/ru/maps/level_002_kirzah_river" detail = RF4DBWaterbodyDetail( source_system="rf4db", source_external_id="level_002_kirzah_river", source_url=source_url, name="р. Вьюнок", description=None, aliases=(), fish_species=("Щука обыкновенная",), fish_external_ids=("n.pike",), image_urls=(), point_urls=(), ) monkeypatch.setattr(community_cli, "fetch_html", lambda _url: "") monkeypatch.setitem( community_cli.DETAIL_SOURCES, "rf4db-waterbody", lambda _html, *, source_url: detail, ) assert community_cli.main([ "rf4db-waterbody", "--url", source_url, "--state-file", str(state_file), ]) == 0 payload = json.loads(capsys.readouterr().out) assert payload["source_external_id"] == "level_002_kirzah_river" assert payload["fish_species"] == ["Щука обыкновенная"] def test_validate_url_host_rejects_disallowed_hosts() -> None: from rf4_research.community_cli import _validate_url_host # Allowed hosts pass assert _validate_url_host("https://download.rf4db.com/ru/catches") == "download.rf4db.com" assert _validate_url_host("https://www.rf4-stat.ru/posts/") == "rf4-stat.ru" assert _validate_url_host("https://rf4map.ru/point/123") == "rf4map.ru" # Disallowed hosts raise ValueError before network I/O with pytest.raises(ValueError, match="not in allowlist"): _validate_url_host("https://169.254.169.254/latest/meta-data/") with pytest.raises(ValueError, match="not in allowlist"): _validate_url_host("https://internal-service.corp/api") # Port validation also works with pytest.raises(ValueError, match="not in allowed ports"): _validate_url_host("https://download.rf4db.com:9999/admin") def test_validate_url_host_rejects_missing_hostname() -> None: from rf4_research.community_cli import _validate_url_host with pytest.raises(ValueError, match="valid hostname"): _validate_url_host("https://") with pytest.raises(ValueError, match="scheme.*not allowed"): _validate_url_host("ftp://rf4db.com/file") with pytest.raises(ValueError, match="HTTPS"): _validate_url_host("http://rf4db.com/file") def test_write_state_is_atomic_with_flush(tmp_path: Path) -> None: """A02: _write_state uses exclusive lock, flush, and atomic rename.""" from rf4_research.community_cli import _read_state, _write_state state_file = tmp_path / "state.json" _write_state(state_file, {"key1": 1000.0}) assert state_file.exists() assert not (state_file.with_suffix(".tmp")).exists() state = _read_state(state_file) assert state == {"key1": 1000.0} _write_state(state_file, {"key1": 2000.0, "key2": 3000.0}) state = _read_state(state_file) assert state == {"key1": 2000.0, "key2": 3000.0} assert not (state_file.with_suffix(".tmp")).exists() def test_check_and_reserve_allows_only_one_per_interval(tmp_path: Path) -> None: """A02: Atomic check-and-reserve — only one success per interval.""" from rf4_research.community_cli import check_and_reserve state_file = tmp_path / "state.json" base_time = 1000.0 # First call should succeed check_and_reserve("test-source", state_file=state_file, now=base_time) state = json.loads(state_file.read_text(encoding="utf-8")) assert state["test-source"] == base_time # Second call within interval should fail with pytest.raises(RuntimeError, match="retry in 1799"): check_and_reserve("test-source", state_file=state_file, now=base_time + 1) # After interval expires, should succeed again check_and_reserve("test-source", state_file=state_file, now=base_time + 1800) def _try_reserve_for_test(args: tuple) -> tuple: """Helper for multiprocessing — must be at module level.""" pid, state_file_str, results_list = args from rf4_research.community_cli import check_and_reserve from pathlib import Path try: check_and_reserve("shared-source", state_file=Path(state_file_str), now=1000.0) results_list.append((pid, "ok")) except RuntimeError as e: results_list.append((pid, str(e))) return (pid, "ok") def test_check_and_reserve_atomic_under_concurrent_access(tmp_path: Path) -> None: """A02: Real multi-process test — concurrent processes get at most one reservation.""" import multiprocessing state_file = tmp_path / "concurrent.json" # Explicitly shut down the manager and never leak a timed-out child. with multiprocessing.Manager() as manager: results = manager.list() processes = [] for i in range(3): args = (i, str(state_file), results) p = multiprocessing.Process(target=_try_reserve_for_test, args=(args,)) processes.append(p) for p in processes: p.start() for p in processes: p.join(timeout=10) for p in processes: if p.is_alive(): p.terminate() p.join(timeout=2) # At most one should succeed ok_count = sum(1 for _, r in results if r == "ok") assert ok_count == 1, f"Expected exactly 1 ok, got {ok_count}: {results}" denied_count = sum(1 for _, r in results if "cooldown" in r) assert denied_count == 2, f"Expected 2 denied, got {denied_count}: {results}" # A03: Manual redirect control tests def test_validate_url_before_io_rejects_http(tmp_path: Path) -> None: """A03: HTTP scheme rejected even for allowed hosts.""" from rf4_research.community_cli import _validate_url_before_io with pytest.raises(ValueError, match="Only HTTPS"): _validate_url_before_io("http://rf4-stat.ru/fishing/") def test_validate_url_before_io_rejects_bad_ports(tmp_path: Path) -> None: """A03: Non-standard ports rejected.""" from rf4_research.community_cli import _validate_url_before_io with pytest.raises(ValueError, match="not in allowed ports"): _validate_url_before_io("https://rf4-stat.ru:8080/path") with pytest.raises(ValueError, match="not in allowed ports"): _validate_url_before_io("https://rf4-stat.ru:4443/path") def test_validate_url_before_io_rejects_disallowed_hosts(tmp_path: Path) -> None: """A03: Disallowed hosts rejected before network I/O.""" from rf4_research.community_cli import _validate_url_before_io with pytest.raises(ValueError, match="not in allowlist"): _validate_url_before_io("https://evil.com/phishing") with pytest.raises(ValueError, match="not in allowlist"): _validate_url_before_io("https://169.254.169.254/metadata") def test_validate_url_before_io_allows_valid_urls() -> None: """A03: Valid allowed hosts pass validation.""" from rf4_research.community_cli import _validate_url_before_io hostname, scheme = _validate_url_before_io("https://rf4-stat.ru/fishing/") assert hostname == "rf4-stat.ru" assert scheme == "https" hostname, scheme = _validate_url_before_io("https://download.rf4db.com/ru/catches") assert hostname == "download.rf4db.com" assert scheme == "https" def test_validate_url_before_io_normalizes_www() -> None: """A03: www. prefix is stripped from hostname.""" from rf4_research.community_cli import _validate_url_before_io hostname, _ = _validate_url_before_io("https://www.rf4-stat.ru/fishing/") assert hostname == "rf4-stat.ru" def test_fetch_html_redirect_to_disallowed_host_rejected(monkeypatch: pytest.MonkeyPatch) -> None: """A03: Redirect to disallowed host raises before making request.""" from rf4_research.community_cli import fetch_html import urllib.error # Mock build_opener to return an opener that raises 302 redirect from unittest.mock import Mock from rf4_research.community_cli import _StrictRedirectHandler opener_instance = Mock() def mock_open(request, timeout=None): # Simulate a redirect response exc = urllib.error.HTTPError( url=str(request.full_url), code=302, msg="Found", hdrs=Mock(), fp=None ) exc.headers = {"Location": "https://evil.com/phishing"} raise exc opener_instance.open = mock_open def mock_build_opener(*args, **kwargs): return opener_instance monkeypatch.setattr("rf4_research.community_cli.build_opener", mock_build_opener) # Should raise during redirect validation (evil.com not in allowlist) with pytest.raises(ValueError, match="not in allowlist"): fetch_html("https://rf4-stat.ru/redirect-to-evil", _redirects=0) def test_fetch_html_redirect_chain_limit() -> None: """A03: Redirect chain exceeding MAX_REDIRECT_HOPS raises ValueError.""" from rf4_research.community_cli import fetch_html, MAX_REDIRECT_HOPS with pytest.raises(ValueError, match=f"Redirect chain exceeds {MAX_REDIRECT_HOPS} hops"): fetch_html("https://rf4-stat.ru/", _redirects=MAX_REDIRECT_HOPS + 1) def test_fetch_html_allows_valid_https() -> None: """A03: Valid HTTPS URLs pass scheme validation.""" from rf4_research.community_cli import _validate_url_before_io hostname, scheme = _validate_url_before_io("https://rf4-stat.ru/fishing/") assert scheme == "https" assert hostname == "rf4-stat.ru" def test_extract_redirect_url_from_headers() -> None: """A03: _extract_redirect_url handles both Location and location headers.""" from rf4_research.community_cli import _extract_redirect_url # Standard capitalization headers = {"Location": "https://rf4-stat.ru/new-path"} assert _extract_redirect_url(headers) == "https://rf4-stat.ru/new-path" # Lowercase (some servers use this) headers = {"location": "https://rf4-stat.ru/other-path"} assert _extract_redirect_url(headers) == "https://rf4-stat.ru/other-path" # Missing header headers = {} assert _extract_redirect_url(headers) is None def test_urljoin_resolves_relative_redirects() -> None: """A03: Relative redirect URLs are resolved against the base URL.""" from urllib.parse import urljoin # Relative path assert urljoin("https://rf4-stat.ru/old", "/new") == "https://rf4-stat.ru/new" # Relative without leading slash assert urljoin("https://rf4-stat.ru/old/path", "new") == "https://rf4-stat.ru/old/new" # Absolute URL assert urljoin("https://rf4-stat.ru/old", "https://rf4-stat.ru/absolute") == "https://rf4-stat.ru/absolute" # A02: State validation tests def test_read_state_rejects_corrupt_json(tmp_path: Path) -> None: """A02: Corrupt JSON returns empty dict.""" from rf4_research.community_cli import _read_state state_file = tmp_path / "corrupt.json" state_file.write_text("not valid json {{{") assert _read_state(state_file) == {} def test_read_state_rejects_invalid_structure(tmp_path: Path) -> None: """A02: Non-dict or non-flat structures return empty dict.""" from rf4_research.community_cli import _read_state state_file = tmp_path / "invalid.json" # List instead of dict state_file.write_text("[1, 2, 3]") assert _read_state(state_file) == {} # Dict with non-number values state_file.write_text('{"key": "value"}') assert _read_state(state_file) == {} # Dict with nested dict state_file.write_text('{"key": {"nested": true}}') assert _read_state(state_file) == {} # Dict with non-string keys (JSON always has string keys, but validate anyway) state_file.write_text('{"valid": 123}') assert _read_state(state_file) == {"valid": 123} def test_read_state_handles_missing_file() -> None: """A02: Missing state file returns empty dict.""" from rf4_research.community_cli import _read_state assert _read_state(Path("/nonexistent/state.json")) == {}