from __future__ import annotations import argparse from datetime import datetime, timezone import json from pathlib import Path import urllib.error import urllib.request from .community_cli import MIN_FETCH_INTERVAL_SECONDS, USER_AGENT, _StrictRedirectHandler, _read_state, _validate_url_before_io, check_and_reserve, fetch_html, fetch_site_key from .media_assets import approve_stored_assets, audit_media_catalog, extract_media_candidates, media_coverage, media_quality_report, merge_manifest, reconcile_queued_duplicates, reclassify_manifest, review_asset, store_asset DEFAULT_ROOT = Path("data/media") MAX_ASSET_BYTES = 15 * 1024 * 1024 MEDIA_PRIORITY = {"waterbody": 0, "fish": 1, "tackle": 2, "reference": 3} MAX_BATCH_ASSETS_PER_DOMAIN = 40 MAX_CONSECUTIVE_FAILURES = 3 def _queue_sort_key(item: dict) -> tuple: return ( MEDIA_PRIORITY.get(str(item.get("entity_type")), 99), not bool(item.get("label")), str(item.get("label") or "").casefold(), str(item.get("asset_url")), ) def media_queue_plan(root: Path, state_file: Path, *, now: float | None = None) -> dict: """Plan the next useful download per domain without network access.""" manifest = json.loads((root / "manifest.json").read_text(encoding="utf-8")) state = _read_state(state_file) current = datetime.now(timezone.utc).timestamp() if now is None else now grouped: dict[str, list[dict]] = {} for item in manifest.get("assets", []): if item.get("status") == "queued": grouped.setdefault(fetch_site_key(item["asset_url"]), []).append(item) domains: dict[str, dict] = {} for site, items in sorted(grouped.items()): items.sort(key=_queue_sort_key) last_attempt = state.get(site) retry_in = max(0, round(float(last_attempt) + MIN_FETCH_INTERVAL_SECONDS - current)) if isinstance(last_attempt, (int, float)) else 0 next_item = items[0] queued_by_type: dict[str, int] = {} for item in items: entity_type = str(item.get("entity_type") or "unknown") queued_by_type[entity_type] = queued_by_type.get(entity_type, 0) + 1 domains[site] = { "queued": len(items), "queued_by_type": dict(sorted(queued_by_type.items())), "ready": retry_in == 0, "retry_in_seconds": retry_in, "next_asset": { "asset_url": next_item["asset_url"], "entity_type": next_item.get("entity_type"), "label": next_item.get("label"), "external_id": next_item.get("external_id"), "source_page": next_item.get("source_page"), }, } coverage = media_coverage(root) return { "generated_at": datetime.fromtimestamp(current, timezone.utc).isoformat(), "network_requests": 0, "cooldown_seconds": MIN_FETCH_INTERVAL_SECONDS, "batch_limit_per_domain": MAX_BATCH_ASSETS_PER_DOMAIN, "queued_total": sum(group["queued"] for group in domains.values()), "domains": domains, "coverage": coverage["entities"], "exact_catalog_gaps_known": False, "catalog_gap_note": "Baseline stores verified totals, not a canonical name list; exact missing entity names cannot be claimed yet.", } def _save_manifest(path: Path, manifest: dict) -> None: path.write_text(json.dumps(manifest, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") def _attempt_asset(root: Path, manifest_path: Path, manifest: dict, queued: dict) -> tuple[str, Exception | None]: url = queued["asset_url"] attempted_at = datetime.now(timezone.utc).isoformat() try: request = urllib.request.Request(url, headers={"User-Agent": USER_AGENT, "Accept": "image/*"}) opener = urllib.request.build_opener(_StrictRedirectHandler()) with opener.open(request, timeout=30) as response: content_type = response.headers.get_content_type() if not content_type.startswith("image/"): raise ValueError(f"expected image, got {content_type}") body = response.read(MAX_ASSET_BYTES + 1) if len(body) > MAX_ASSET_BYTES: raise ValueError("asset exceeded 15MB limit") digest, relative, width, height, detected_mime = store_asset(root, body, content_type=content_type, source_url=url) except Exception as exc: code = exc.code if isinstance(exc, urllib.error.HTTPError) else None status = "missing" if code in {404, 410} else "blocked" if code in {401, 403, 429} else "invalid" if isinstance(exc, ValueError) else "queued" queued.update({"status": status, "last_attempt_at": attempted_at, "last_error": str(exc)[:500]}) _save_manifest(manifest_path, manifest) return f"{status} {url}: {str(exc)[:200]}", exc queued.update({"status": "stored", "sha256": digest, "local_path": relative, "content_type": detected_mime, "bytes": len(body), "width": width, "height": height, "fetched_at": attempted_at}) _save_manifest(manifest_path, manifest) return f"stored {url} as {relative}", None def _download_one(root: Path, state_file: Path, asset_url: str | None = None) -> str: manifest_path = root / "manifest.json" manifest = json.loads(manifest_path.read_text(encoding="utf-8")) queued = next((item for item in manifest["assets"] if item.get("status") == "queued" and (asset_url is None or item["asset_url"] == asset_url)), None) if queued is None: return "queue is empty" if asset_url is None else "asset is not queued" url = queued["asset_url"] _validate_url_before_io(url) check_and_reserve(fetch_site_key(url), state_file=state_file) message, error = _attempt_asset(root, manifest_path, manifest, queued) if error is not None: raise error return message def _download_batch(root: Path, state_file: Path, *, limit: int = MAX_BATCH_ASSETS_PER_DOMAIN) -> dict: """Download up to ``limit`` assets per domain under one reserved window.""" if not 1 <= limit <= MAX_BATCH_ASSETS_PER_DOMAIN: raise ValueError(f"batch limit must be 1..{MAX_BATCH_ASSETS_PER_DOMAIN}") manifest_path = root / "manifest.json" manifest = json.loads(manifest_path.read_text(encoding="utf-8")) grouped: dict[str, list[dict]] = {} for item in manifest.get("assets", []): if item.get("status") != "queued": continue _validate_url_before_io(item["asset_url"]) grouped.setdefault(fetch_site_key(item["asset_url"]), []).append(item) domains: dict[str, dict] = {} for site, items in sorted(grouped.items()): items.sort(key=_queue_sort_key) try: # One reservation governs the explicitly allowed asset batch. check_and_reserve(site, state_file=state_file) except RuntimeError as exc: domains[site] = {"attempted": 0, "stored": 0, "failed": 0, "skipped": str(exc)} continue stored = failed = consecutive_failures = 0 messages: list[str] = [] stopped_reason: str | None = None for item in items[:limit]: message, error = _attempt_asset(root, manifest_path, manifest, item) messages.append(message) if error is None: stored += 1 consecutive_failures = 0 continue failed += 1 consecutive_failures += 1 status = item.get("status") if status in {"blocked", "queued"}: stopped_reason = f"domain stopped after {status} response" break if consecutive_failures >= MAX_CONSECUTIVE_FAILURES: stopped_reason = f"domain stopped after {MAX_CONSECUTIVE_FAILURES} consecutive failures" break domains[site] = { "attempted": stored + failed, "stored": stored, "failed": failed, "stopped_reason": stopped_reason, "messages": messages, } return { "batch_limit_per_domain": limit, "cooldown_seconds": MIN_FETCH_INTERVAL_SECONDS, "domains": domains, "attempted_total": sum(item["attempted"] for item in domains.values()), "stored_total": sum(item["stored"] for item in domains.values()), "failed_total": sum(item["failed"] for item in domains.values()), } def main(argv: list[str] | None = None) -> int: parser = argparse.ArgumentParser(description="Index authorized RF4 media without hotlinking") parser.add_argument("url", nargs="?") download_mode = parser.add_mutually_exclusive_group() download_mode.add_argument("--download-one", action="store_true", help="Store one queued asset while respecting site cooldown") download_mode.add_argument("--download-batch", action="store_true", help="Store up to 40 queued assets per domain in one reserved window") parser.add_argument("--batch-limit", type=int, default=MAX_BATCH_ASSETS_PER_DOMAIN, choices=range(1, MAX_BATCH_ASSETS_PER_DOMAIN + 1), metavar="1..40") parser.add_argument("--asset-url", help="Download this exact queued asset instead of the first queue item") parser.add_argument("--reclassify", action="store_true", help="Reapply current conservative classifier without network access") parser.add_argument("--dedupe-queue", action="store_true", help="Mark queued label matches to stored assets as duplicates without network access") parser.add_argument("--approve-stored", action="store_true", help="Publish all stored assets after explicit owner approval") parser.add_argument("--review-url", help="Review an asset already present in the manifest") parser.add_argument("--decision", choices=("approved", "rejected")) parser.add_argument("--entity-type", choices=("fish", "waterbody", "tackle", "reference")) parser.add_argument("--entity-key") parser.add_argument("--note") parser.add_argument("--audit", action="store_true", help="Verify manifest metadata, hashes and local files without network access") parser.add_argument("--coverage", action="store_true", help="Compare candidates and approvals with the catalog baseline") parser.add_argument("--quality-report", action="store_true", help="Report low-resolution published fish and known alternatives without network access") parser.add_argument("--queue-plan", action="store_true", help="Show the next useful queued asset per domain without network access") parser.add_argument("--root", type=Path, default=DEFAULT_ROOT) parser.add_argument("--state-file", type=Path, default=Path(".cache/community-fetch-state.json")) args = parser.parse_args(argv) try: if args.audit: report = audit_media_catalog(args.root) print(json.dumps(report, ensure_ascii=False, indent=2)) return 1 if report["issues"] or report["orphaned_files"] else 0 if args.coverage: print(json.dumps(media_coverage(args.root), ensure_ascii=False, indent=2)) return 0 if args.quality_report: print(json.dumps(media_quality_report(args.root), ensure_ascii=False, indent=2)) return 0 if args.queue_plan: print(json.dumps(media_queue_plan(args.root, args.state_file), ensure_ascii=False, indent=2)) return 0 if args.review_url: if not args.decision: parser.error("--decision is required with --review-url") item = review_asset(args.root / "manifest.json", asset_url=args.review_url, decision=args.decision, entity_type=args.entity_type, entity_key=args.entity_key, note=args.note) print(f"{item['status']} {item['asset_url']}") return 0 if args.reclassify: manifest = reclassify_manifest(args.root / "manifest.json") print(f"reclassified {len(manifest['assets'])} candidates") return 0 if args.dedupe_queue: print(json.dumps(reconcile_queued_duplicates(args.root / "manifest.json"), ensure_ascii=False, indent=2)) return 0 if args.approve_stored: if not args.note: parser.error("--note is required with --approve-stored") print(json.dumps(approve_stored_assets(args.root / "manifest.json", note=args.note), ensure_ascii=False, indent=2)) return 0 if args.download_one: print(_download_one(args.root, args.state_file, args.asset_url)) return 0 if args.download_batch: if args.asset_url: parser.error("--asset-url is only valid with --download-one") report = _download_batch(args.root, args.state_file, limit=args.batch_limit) print(json.dumps(report, ensure_ascii=False, indent=2)) return 1 if any(item.get("stopped_reason") for item in report["domains"].values()) else 0 if not args.url: parser.error("url is required unless a download or offline mode is used") check_and_reserve(fetch_site_key(args.url), state_file=args.state_file) html = fetch_html(args.url) candidates = extract_media_candidates(html, source_page=args.url) manifest = merge_manifest(args.root / "manifest.json", candidates) except Exception as exc: parser.exit(1, f"media index failed: {exc}\n") print(f"indexed {len(candidates)} candidates; manifest contains {len(manifest['assets'])}") return 0 if __name__ == "__main__": raise SystemExit(main())