"""NRO delegated stats prefix geography collector. Parses the delegated extended/statistics file and stores coarse registry allocation geography as prefix-centric fallback hints. """ from __future__ import annotations import ipaddress import hashlib import json import tempfile import time from datetime import UTC, datetime from pathlib import Path from typing import Any import httpx from app.services.collectors.base import BaseCollector class NRODelegatedPrefixGeoCollector(BaseCollector): name = "nro_delegated_prefix_geo" priority = "P1" module = "L3" frequency_hours = 24 data_type = "prefix_geography" fail_on_empty = True _cache_dir = Path(tempfile.gettempdir()) / "planet-download-cache" / "nro" @staticmethod def _cache_key(url: str) -> str: return hashlib.sha1(url.encode("utf-8")).hexdigest()[:16] @classmethod def _cache_paths(cls, url: str) -> tuple[Path, Path, Path]: key = cls._cache_key(url) txt_path = cls._cache_dir / f"{key}.txt" part_path = cls._cache_dir / f"{key}.txt.part" meta_path = cls._cache_dir / f"{key}.meta.json" return txt_path, part_path, meta_path @staticmethod def _load_meta(meta_path: Path) -> dict[str, Any]: if not meta_path.exists(): return {} try: return json.loads(meta_path.read_text(encoding="utf-8")) except (json.JSONDecodeError, OSError): return {} @staticmethod def _save_meta(meta_path: Path, payload: dict[str, Any]) -> None: meta_path.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8") @staticmethod def _validators_match(meta: dict[str, Any], remote: dict[str, Any]) -> bool: etag = str(remote.get("etag") or "").strip() last_modified = str(remote.get("last_modified") or "").strip() if etag: return etag == str(meta.get("etag") or "").strip() if last_modified: return last_modified == str(meta.get("last_modified") or "").strip() return True async def _fetch_remote_info(self, client: httpx.AsyncClient, url: str) -> dict[str, Any]: try: response = await client.head(url) if response.status_code >= 400: return {} content_length_raw = response.headers.get("content-length") content_length = int(content_length_raw) if content_length_raw else None return { "etag": response.headers.get("etag"), "last_modified": response.headers.get("last-modified"), "content_length": content_length, "accept_ranges": (response.headers.get("accept-ranges") or "").lower(), } except (httpx.HTTPError, ValueError): return {} async def _download_body_with_resume(self, client: httpx.AsyncClient, url: str) -> str: self._cache_dir.mkdir(parents=True, exist_ok=True) txt_path, part_path, meta_path = self._cache_paths(url) meta = self._load_meta(meta_path) remote = await self._fetch_remote_info(client, url) if txt_path.exists(): local_size = txt_path.stat().st_size remote_size = remote.get("content_length") if self._validators_match(meta, remote) and (remote_size is None or local_size == remote_size): if remote_size > 0: await self.update_progress(remote_size, commit=True, force=True) return txt_path.read_text(encoding="utf-8", errors="replace") expected_size = remote.get("content_length") can_resume = (remote.get("accept_ranges") or "") == "bytes" resume_from = part_path.stat().st_size if part_path.exists() else 0 if expected_size is not None and resume_from > expected_size: part_path.unlink(missing_ok=True) resume_from = 0 if not self._validators_match(meta, remote): part_path.unlink(missing_ok=True) resume_from = 0 headers = { "User-Agent": "Planet-Intelligence-System/1.0 (Python/collector)", "Accept": "text/plain,*/*", } if txt_path.exists(): if meta.get("etag"): headers["If-None-Match"] = str(meta.get("etag")) elif meta.get("last_modified"): headers["If-Modified-Since"] = str(meta.get("last_modified")) if can_resume and resume_from > 0: headers["Range"] = f"bytes={resume_from}-" if remote.get("etag"): headers["If-Range"] = str(remote.get("etag")) elif remote.get("last_modified"): headers["If-Range"] = str(remote.get("last_modified")) async with client.stream("GET", url, headers=headers) as response: if response.status_code == 304 and txt_path.exists(): if expected_size and expected_size > 0: await self.update_progress(expected_size, commit=True, force=True) return txt_path.read_text(encoding="utf-8", errors="replace") response.raise_for_status() if response.status_code == 206 and resume_from > 0: mode = "ab" else: mode = "wb" resume_from = 0 downloaded = resume_from last_emit = 0 last_emit_time = time.monotonic() min_emit_bytes = max(expected_size // 150, 512 * 1024) if expected_size and expected_size > 0 else 1024 * 1024 with part_path.open(mode) as f: if downloaded > 0 and expected_size and expected_size > 0: await self.update_progress(min(downloaded, expected_size), commit=True) async for chunk in response.aiter_bytes(): if not chunk: continue f.write(chunk) downloaded += len(chunk) if not expected_size or expected_size <= 0: continue now = time.monotonic() should_emit = ( downloaded >= expected_size or downloaded - last_emit >= min_emit_bytes or now - last_emit_time >= 2.0 ) if should_emit: last_emit = downloaded last_emit_time = now await self.update_progress(min(downloaded, expected_size), commit=True) final_size = part_path.stat().st_size if part_path.exists() else 0 if expected_size is not None and final_size != expected_size: raise RuntimeError( f"NRO download incomplete for {url}: expected={expected_size}, got={final_size}" ) part_path.replace(txt_path) self._save_meta( meta_path, { "url": url, "etag": remote.get("etag"), "last_modified": remote.get("last_modified"), "content_length": expected_size, "updated_at": datetime.now(UTC).isoformat(), }, ) if expected_size and expected_size > 0: await self.update_progress(expected_size, commit=True, force=True) return txt_path.read_text(encoding="utf-8", errors="replace") async def fetch(self) -> list[dict[str, Any]]: if not self._resolved_url: raise RuntimeError("NRO delegated stats URL is not configured") async with httpx.AsyncClient(timeout=180.0, follow_redirects=True) as client: remote = await self._fetch_remote_info(client, self._resolved_url) total_expected = remote.get("content_length") or 0 if total_expected > 0 and self._current_task and self._db_session: self._current_task.total_records = total_expected self._current_task.records_processed = 0 self._current_task.progress = 0.0 await self._db_session.commit() await self._publish_task_update(force=True) try: body = await self._download_body_with_resume(client, self._resolved_url) except Exception: txt_path, part_path, meta_path = self._cache_paths(self._resolved_url) txt_path.unlink(missing_ok=True) part_path.unlink(missing_ok=True) meta_path.unlink(missing_ok=True) body = await self._download_body_with_resume(client, self._resolved_url) rows: list[dict[str, Any]] = [] for raw_line in body.splitlines(): line = raw_line.strip() if not line or line.startswith("#"): continue parts = line.split("|") if len(parts) < 7: continue rir = (parts[0] or "").strip().lower() country_code = (parts[1] or "").strip().upper() record_type = (parts[2] or "").strip().lower() start = (parts[3] or "").strip() value = (parts[4] or "").strip() allocated_date = (parts[5] or "").strip() status = (parts[6] or "").strip().lower() if record_type not in {"ipv4", "ipv6"}: continue if not start or not value: continue rows.append( { "rir": rir, "country_code": country_code, "type": record_type, "start": start, "value": value, "allocated_date": allocated_date, "status": status, } ) return rows def transform(self, raw_data: list[dict[str, Any]]) -> list[dict[str, Any]]: reference_date = datetime.now(UTC).isoformat() transformed: list[dict[str, Any]] = [] for item in raw_data: record_type = str(item.get("type") or "").strip().lower() start = str(item.get("start") or "").strip() value = str(item.get("value") or "").strip() country_code = str(item.get("country_code") or "").strip().upper() try: if record_type == "ipv4": start_ip = ipaddress.ip_address(start) count = int(value) if count <= 0: continue end_ip_int = int(start_ip) + count - 1 end_ip = ipaddress.ip_address(end_ip_int) network = list(ipaddress.summarize_address_range(start_ip, end_ip))[0] elif record_type == "ipv6": prefixlen = int(value) network = ipaddress.ip_network(f"{start}/{prefixlen}", strict=False) start_ip = network.network_address end_ip = network.broadcast_address else: continue except (ValueError, TypeError): continue family = f"ipv{network.version}" prefix = str(network) transformed.append( { "source_id": f"{item.get('rir')}:{family}:{prefix}:{country_code}", "name": prefix, "title": f"{prefix} {country_code}".strip(), "country": country_code, "city": "", "latitude": None, "longitude": None, "metadata": { "family": family, "prefix": prefix, "range_start": str(start_ip), "range_end": str(end_ip), "country_code": country_code, "rir": item.get("rir"), "status": item.get("status"), "allocated_date": item.get("allocated_date"), "source_dataset": "nro_delegated_stats", "confidence": "registry_allocated", }, "reference_date": reference_date, } ) return transformed