fix(bgp): add opengeofeed+nro collectors and bump version to 0.22.14
This commit is contained in:
303
backend/app/services/collectors/nro_delegated.py
Normal file
303
backend/app/services/collectors/nro_delegated.py
Normal file
@@ -0,0 +1,303 @@
|
|||||||
|
"""NRO delegated stats prefix geography collector.
|
||||||
|
|
||||||
|
Parses the delegated extended/statistics file and stores coarse registry
|
||||||
|
allocation geography as prefix-centric fallback hints.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import ipaddress
|
||||||
|
import hashlib
|
||||||
|
import json
|
||||||
|
import tempfile
|
||||||
|
import time
|
||||||
|
from datetime import UTC, datetime
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from app.services.collectors.base import BaseCollector
|
||||||
|
|
||||||
|
|
||||||
|
class NRODelegatedPrefixGeoCollector(BaseCollector):
|
||||||
|
name = "nro_delegated_prefix_geo"
|
||||||
|
priority = "P1"
|
||||||
|
module = "L3"
|
||||||
|
frequency_hours = 24
|
||||||
|
data_type = "prefix_geography"
|
||||||
|
fail_on_empty = True
|
||||||
|
_cache_dir = Path(tempfile.gettempdir()) / "planet-download-cache" / "nro"
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _cache_key(url: str) -> str:
|
||||||
|
return hashlib.sha1(url.encode("utf-8")).hexdigest()[:16]
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def _cache_paths(cls, url: str) -> tuple[Path, Path, Path]:
|
||||||
|
key = cls._cache_key(url)
|
||||||
|
txt_path = cls._cache_dir / f"{key}.txt"
|
||||||
|
part_path = cls._cache_dir / f"{key}.txt.part"
|
||||||
|
meta_path = cls._cache_dir / f"{key}.meta.json"
|
||||||
|
return txt_path, part_path, meta_path
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _load_meta(meta_path: Path) -> dict[str, Any]:
|
||||||
|
if not meta_path.exists():
|
||||||
|
return {}
|
||||||
|
try:
|
||||||
|
return json.loads(meta_path.read_text(encoding="utf-8"))
|
||||||
|
except (json.JSONDecodeError, OSError):
|
||||||
|
return {}
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _save_meta(meta_path: Path, payload: dict[str, Any]) -> None:
|
||||||
|
meta_path.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8")
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _validators_match(meta: dict[str, Any], remote: dict[str, Any]) -> bool:
|
||||||
|
etag = str(remote.get("etag") or "").strip()
|
||||||
|
last_modified = str(remote.get("last_modified") or "").strip()
|
||||||
|
if etag:
|
||||||
|
return etag == str(meta.get("etag") or "").strip()
|
||||||
|
if last_modified:
|
||||||
|
return last_modified == str(meta.get("last_modified") or "").strip()
|
||||||
|
return True
|
||||||
|
|
||||||
|
async def _fetch_remote_info(self, client: httpx.AsyncClient, url: str) -> dict[str, Any]:
|
||||||
|
try:
|
||||||
|
response = await client.head(url)
|
||||||
|
if response.status_code >= 400:
|
||||||
|
return {}
|
||||||
|
content_length_raw = response.headers.get("content-length")
|
||||||
|
content_length = int(content_length_raw) if content_length_raw else None
|
||||||
|
return {
|
||||||
|
"etag": response.headers.get("etag"),
|
||||||
|
"last_modified": response.headers.get("last-modified"),
|
||||||
|
"content_length": content_length,
|
||||||
|
"accept_ranges": (response.headers.get("accept-ranges") or "").lower(),
|
||||||
|
}
|
||||||
|
except (httpx.HTTPError, ValueError):
|
||||||
|
return {}
|
||||||
|
|
||||||
|
async def _download_body_with_resume(self, client: httpx.AsyncClient, url: str) -> str:
|
||||||
|
self._cache_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
txt_path, part_path, meta_path = self._cache_paths(url)
|
||||||
|
meta = self._load_meta(meta_path)
|
||||||
|
remote = await self._fetch_remote_info(client, url)
|
||||||
|
|
||||||
|
if txt_path.exists():
|
||||||
|
local_size = txt_path.stat().st_size
|
||||||
|
remote_size = remote.get("content_length")
|
||||||
|
if self._validators_match(meta, remote) and (remote_size is None or local_size == remote_size):
|
||||||
|
if remote_size > 0:
|
||||||
|
await self.update_progress(remote_size, commit=True, force=True)
|
||||||
|
return txt_path.read_text(encoding="utf-8", errors="replace")
|
||||||
|
|
||||||
|
expected_size = remote.get("content_length")
|
||||||
|
can_resume = (remote.get("accept_ranges") or "") == "bytes"
|
||||||
|
resume_from = part_path.stat().st_size if part_path.exists() else 0
|
||||||
|
if expected_size is not None and resume_from > expected_size:
|
||||||
|
part_path.unlink(missing_ok=True)
|
||||||
|
resume_from = 0
|
||||||
|
if not self._validators_match(meta, remote):
|
||||||
|
part_path.unlink(missing_ok=True)
|
||||||
|
resume_from = 0
|
||||||
|
|
||||||
|
headers = {
|
||||||
|
"User-Agent": "Planet-Intelligence-System/1.0 (Python/collector)",
|
||||||
|
"Accept": "text/plain,*/*",
|
||||||
|
}
|
||||||
|
if txt_path.exists():
|
||||||
|
if meta.get("etag"):
|
||||||
|
headers["If-None-Match"] = str(meta.get("etag"))
|
||||||
|
elif meta.get("last_modified"):
|
||||||
|
headers["If-Modified-Since"] = str(meta.get("last_modified"))
|
||||||
|
|
||||||
|
if can_resume and resume_from > 0:
|
||||||
|
headers["Range"] = f"bytes={resume_from}-"
|
||||||
|
if remote.get("etag"):
|
||||||
|
headers["If-Range"] = str(remote.get("etag"))
|
||||||
|
elif remote.get("last_modified"):
|
||||||
|
headers["If-Range"] = str(remote.get("last_modified"))
|
||||||
|
|
||||||
|
async with client.stream("GET", url, headers=headers) as response:
|
||||||
|
if response.status_code == 304 and txt_path.exists():
|
||||||
|
if expected_size and expected_size > 0:
|
||||||
|
await self.update_progress(expected_size, commit=True, force=True)
|
||||||
|
return txt_path.read_text(encoding="utf-8", errors="replace")
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
if response.status_code == 206 and resume_from > 0:
|
||||||
|
mode = "ab"
|
||||||
|
else:
|
||||||
|
mode = "wb"
|
||||||
|
resume_from = 0
|
||||||
|
|
||||||
|
downloaded = resume_from
|
||||||
|
last_emit = 0
|
||||||
|
last_emit_time = time.monotonic()
|
||||||
|
min_emit_bytes = max(expected_size // 150, 512 * 1024) if expected_size and expected_size > 0 else 1024 * 1024
|
||||||
|
|
||||||
|
with part_path.open(mode) as f:
|
||||||
|
if downloaded > 0 and expected_size and expected_size > 0:
|
||||||
|
await self.update_progress(min(downloaded, expected_size), commit=True)
|
||||||
|
async for chunk in response.aiter_bytes():
|
||||||
|
if not chunk:
|
||||||
|
continue
|
||||||
|
f.write(chunk)
|
||||||
|
downloaded += len(chunk)
|
||||||
|
if not expected_size or expected_size <= 0:
|
||||||
|
continue
|
||||||
|
now = time.monotonic()
|
||||||
|
should_emit = (
|
||||||
|
downloaded >= expected_size
|
||||||
|
or downloaded - last_emit >= min_emit_bytes
|
||||||
|
or now - last_emit_time >= 2.0
|
||||||
|
)
|
||||||
|
if should_emit:
|
||||||
|
last_emit = downloaded
|
||||||
|
last_emit_time = now
|
||||||
|
await self.update_progress(min(downloaded, expected_size), commit=True)
|
||||||
|
|
||||||
|
final_size = part_path.stat().st_size if part_path.exists() else 0
|
||||||
|
if expected_size is not None and final_size != expected_size:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"NRO download incomplete for {url}: expected={expected_size}, got={final_size}"
|
||||||
|
)
|
||||||
|
|
||||||
|
part_path.replace(txt_path)
|
||||||
|
self._save_meta(
|
||||||
|
meta_path,
|
||||||
|
{
|
||||||
|
"url": url,
|
||||||
|
"etag": remote.get("etag"),
|
||||||
|
"last_modified": remote.get("last_modified"),
|
||||||
|
"content_length": expected_size,
|
||||||
|
"updated_at": datetime.now(UTC).isoformat(),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
if expected_size and expected_size > 0:
|
||||||
|
await self.update_progress(expected_size, commit=True, force=True)
|
||||||
|
|
||||||
|
return txt_path.read_text(encoding="utf-8", errors="replace")
|
||||||
|
|
||||||
|
async def fetch(self) -> list[dict[str, Any]]:
|
||||||
|
if not self._resolved_url:
|
||||||
|
raise RuntimeError("NRO delegated stats URL is not configured")
|
||||||
|
|
||||||
|
async with httpx.AsyncClient(timeout=180.0, follow_redirects=True) as client:
|
||||||
|
remote = await self._fetch_remote_info(client, self._resolved_url)
|
||||||
|
total_expected = remote.get("content_length") or 0
|
||||||
|
if total_expected > 0 and self._current_task and self._db_session:
|
||||||
|
self._current_task.total_records = total_expected
|
||||||
|
self._current_task.records_processed = 0
|
||||||
|
self._current_task.progress = 0.0
|
||||||
|
await self._db_session.commit()
|
||||||
|
await self._publish_task_update(force=True)
|
||||||
|
|
||||||
|
try:
|
||||||
|
body = await self._download_body_with_resume(client, self._resolved_url)
|
||||||
|
except Exception:
|
||||||
|
txt_path, part_path, meta_path = self._cache_paths(self._resolved_url)
|
||||||
|
txt_path.unlink(missing_ok=True)
|
||||||
|
part_path.unlink(missing_ok=True)
|
||||||
|
meta_path.unlink(missing_ok=True)
|
||||||
|
body = await self._download_body_with_resume(client, self._resolved_url)
|
||||||
|
|
||||||
|
rows: list[dict[str, Any]] = []
|
||||||
|
for raw_line in body.splitlines():
|
||||||
|
line = raw_line.strip()
|
||||||
|
if not line or line.startswith("#"):
|
||||||
|
continue
|
||||||
|
|
||||||
|
parts = line.split("|")
|
||||||
|
if len(parts) < 7:
|
||||||
|
continue
|
||||||
|
|
||||||
|
rir = (parts[0] or "").strip().lower()
|
||||||
|
country_code = (parts[1] or "").strip().upper()
|
||||||
|
record_type = (parts[2] or "").strip().lower()
|
||||||
|
start = (parts[3] or "").strip()
|
||||||
|
value = (parts[4] or "").strip()
|
||||||
|
allocated_date = (parts[5] or "").strip()
|
||||||
|
status = (parts[6] or "").strip().lower()
|
||||||
|
|
||||||
|
if record_type not in {"ipv4", "ipv6"}:
|
||||||
|
continue
|
||||||
|
if not start or not value:
|
||||||
|
continue
|
||||||
|
|
||||||
|
rows.append(
|
||||||
|
{
|
||||||
|
"rir": rir,
|
||||||
|
"country_code": country_code,
|
||||||
|
"type": record_type,
|
||||||
|
"start": start,
|
||||||
|
"value": value,
|
||||||
|
"allocated_date": allocated_date,
|
||||||
|
"status": status,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
return rows
|
||||||
|
|
||||||
|
def transform(self, raw_data: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||||
|
reference_date = datetime.now(UTC).isoformat()
|
||||||
|
transformed: list[dict[str, Any]] = []
|
||||||
|
|
||||||
|
for item in raw_data:
|
||||||
|
record_type = str(item.get("type") or "").strip().lower()
|
||||||
|
start = str(item.get("start") or "").strip()
|
||||||
|
value = str(item.get("value") or "").strip()
|
||||||
|
country_code = str(item.get("country_code") or "").strip().upper()
|
||||||
|
|
||||||
|
try:
|
||||||
|
if record_type == "ipv4":
|
||||||
|
start_ip = ipaddress.ip_address(start)
|
||||||
|
count = int(value)
|
||||||
|
if count <= 0:
|
||||||
|
continue
|
||||||
|
end_ip_int = int(start_ip) + count - 1
|
||||||
|
end_ip = ipaddress.ip_address(end_ip_int)
|
||||||
|
network = list(ipaddress.summarize_address_range(start_ip, end_ip))[0]
|
||||||
|
elif record_type == "ipv6":
|
||||||
|
prefixlen = int(value)
|
||||||
|
network = ipaddress.ip_network(f"{start}/{prefixlen}", strict=False)
|
||||||
|
start_ip = network.network_address
|
||||||
|
end_ip = network.broadcast_address
|
||||||
|
else:
|
||||||
|
continue
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
continue
|
||||||
|
|
||||||
|
family = f"ipv{network.version}"
|
||||||
|
prefix = str(network)
|
||||||
|
|
||||||
|
transformed.append(
|
||||||
|
{
|
||||||
|
"source_id": f"{item.get('rir')}:{family}:{prefix}:{country_code}",
|
||||||
|
"name": prefix,
|
||||||
|
"title": f"{prefix} {country_code}".strip(),
|
||||||
|
"country": country_code,
|
||||||
|
"city": "",
|
||||||
|
"latitude": None,
|
||||||
|
"longitude": None,
|
||||||
|
"metadata": {
|
||||||
|
"family": family,
|
||||||
|
"prefix": prefix,
|
||||||
|
"range_start": str(start_ip),
|
||||||
|
"range_end": str(end_ip),
|
||||||
|
"country_code": country_code,
|
||||||
|
"rir": item.get("rir"),
|
||||||
|
"status": item.get("status"),
|
||||||
|
"allocated_date": item.get("allocated_date"),
|
||||||
|
"source_dataset": "nro_delegated_stats",
|
||||||
|
"confidence": "registry_allocated",
|
||||||
|
},
|
||||||
|
"reference_date": reference_date,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
return transformed
|
||||||
118
backend/app/services/collectors/opengeofeed.py
Normal file
118
backend/app/services/collectors/opengeofeed.py
Normal file
@@ -0,0 +1,118 @@
|
|||||||
|
"""OpenGeoFeed prefix geography collector.
|
||||||
|
|
||||||
|
Fetches public OpenGeoFeed CSV data and stores higher-confidence
|
||||||
|
prefix-to-location hints for BGP prefix-centric enrichment.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import csv
|
||||||
|
import ipaddress
|
||||||
|
from datetime import UTC, datetime
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from app.services.collectors.base import BaseCollector
|
||||||
|
|
||||||
|
|
||||||
|
class OpenGeoFeedPrefixGeoCollector(BaseCollector):
|
||||||
|
name = "opengeofeed_prefix_geo"
|
||||||
|
priority = "P1"
|
||||||
|
module = "L3"
|
||||||
|
frequency_hours = 24
|
||||||
|
data_type = "prefix_geography"
|
||||||
|
fail_on_empty = True
|
||||||
|
|
||||||
|
async def fetch(self) -> list[dict[str, Any]]:
|
||||||
|
if not self._resolved_url:
|
||||||
|
raise RuntimeError("OpenGeoFeed URL is not configured")
|
||||||
|
|
||||||
|
async with httpx.AsyncClient(timeout=180.0, follow_redirects=True) as client:
|
||||||
|
response = await client.get(
|
||||||
|
self._resolved_url,
|
||||||
|
headers={
|
||||||
|
"User-Agent": "Planet-Intelligence-System/1.0 (Python/collector)",
|
||||||
|
"Accept": "text/csv,*/*",
|
||||||
|
},
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
body = response.text
|
||||||
|
|
||||||
|
rows: list[dict[str, Any]] = []
|
||||||
|
reader = csv.reader(body.splitlines())
|
||||||
|
for fields in reader:
|
||||||
|
if not fields:
|
||||||
|
continue
|
||||||
|
first = (fields[0] or "").strip().lower()
|
||||||
|
if not first or first.startswith("#") or first == "prefix":
|
||||||
|
continue
|
||||||
|
|
||||||
|
prefix = (fields[0] or "").strip()
|
||||||
|
country_code = (fields[1] if len(fields) > 1 else "").strip()
|
||||||
|
region = (fields[2] if len(fields) > 2 else "").strip()
|
||||||
|
city = (fields[3] if len(fields) > 3 else "").strip()
|
||||||
|
postal_code = (fields[4] if len(fields) > 4 else "").strip()
|
||||||
|
|
||||||
|
# Keep additional columns for future enrichment without breaking
|
||||||
|
# current normalized schema.
|
||||||
|
extras = [value.strip() for value in fields[5:]] if len(fields) > 5 else []
|
||||||
|
|
||||||
|
rows.append(
|
||||||
|
{
|
||||||
|
"prefix": prefix,
|
||||||
|
"country_code": country_code,
|
||||||
|
"region": region,
|
||||||
|
"city": city,
|
||||||
|
"postal_code": postal_code,
|
||||||
|
"extra_columns": extras,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return rows
|
||||||
|
|
||||||
|
def transform(self, raw_data: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||||
|
reference_date = datetime.now(UTC).isoformat()
|
||||||
|
transformed: list[dict[str, Any]] = []
|
||||||
|
|
||||||
|
for item in raw_data:
|
||||||
|
prefix = str(item.get("prefix") or "").strip()
|
||||||
|
if not prefix:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
network = ipaddress.ip_network(prefix, strict=False)
|
||||||
|
except ValueError:
|
||||||
|
continue
|
||||||
|
|
||||||
|
family = f"ipv{network.version}"
|
||||||
|
country_code = str(item.get("country_code") or "").strip().upper()
|
||||||
|
region = str(item.get("region") or "").strip()
|
||||||
|
city = str(item.get("city") or "").strip()
|
||||||
|
postal_code = str(item.get("postal_code") or "").strip()
|
||||||
|
|
||||||
|
transformed.append(
|
||||||
|
{
|
||||||
|
"source_id": f"{family}:{prefix}:{country_code}:{region}:{city}",
|
||||||
|
"name": prefix,
|
||||||
|
"title": f"{prefix} {country_code}".strip(),
|
||||||
|
"country": country_code,
|
||||||
|
"city": city,
|
||||||
|
"latitude": None,
|
||||||
|
"longitude": None,
|
||||||
|
"metadata": {
|
||||||
|
"family": family,
|
||||||
|
"prefix": prefix,
|
||||||
|
"range_start": str(network.network_address),
|
||||||
|
"range_end": str(network.broadcast_address),
|
||||||
|
"country_code": country_code,
|
||||||
|
"region": region,
|
||||||
|
"city": city,
|
||||||
|
"postal_code": postal_code,
|
||||||
|
"extra_columns": item.get("extra_columns") or [],
|
||||||
|
"source_dataset": "opengeofeed_public",
|
||||||
|
"confidence": "geofeed",
|
||||||
|
},
|
||||||
|
"reference_date": reference_date,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
return transformed
|
||||||
@@ -7,6 +7,29 @@ This project follows the repository versioning rule:
|
|||||||
- `feature` -> `+0.1.0`
|
- `feature` -> `+0.1.0`
|
||||||
- `bugfix` -> `+0.0.1`
|
- `bugfix` -> `+0.0.1`
|
||||||
|
|
||||||
|
## 0.22.14
|
||||||
|
|
||||||
|
Released: 2026-04-07
|
||||||
|
|
||||||
|
### Highlights
|
||||||
|
|
||||||
|
- Completed the pending prefix-geography collector add-ons by shipping dedicated `OpenGeoFeed` and `NRO delegated stats` collectors.
|
||||||
|
- Finalized this slice as a bugfix release (`+0.0.1`) with version metadata synchronized across backend/frontend lockfiles.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- Added [opengeofeed.py](/home/ray/dev/linkong/planet/backend/app/services/collectors/opengeofeed.py), introducing `opengeofeed_prefix_geo` ingestion for high-confidence geofeed-backed prefix geography overrides.
|
||||||
|
- Added [nro_delegated.py](/home/ray/dev/linkong/planet/backend/app/services/collectors/nro_delegated.py), introducing `nro_delegated_prefix_geo` ingestion for registry-allocation fallback geography.
|
||||||
|
|
||||||
|
### Improved
|
||||||
|
|
||||||
|
- Improved collector registration wiring to include the new prefix-geography sources in the runtime collector registry and datasource catalog integration flow.
|
||||||
|
- Improved BGP roadmap traceability by aligning shipped collector capabilities with the staged prefix-geography strategy (override source + registry fallback source).
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- Fixed repository drift where datasource mappings and enrichment priority chain referenced `OpenGeoFeed/NRO` source names before the corresponding collector modules were fully committed in-tree.
|
||||||
|
|
||||||
## 0.22.13
|
## 0.22.13
|
||||||
|
|
||||||
Released: 2026-04-07
|
Released: 2026-04-07
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "planet-frontend",
|
"name": "planet-frontend",
|
||||||
"version": "0.22.13",
|
"version": "0.22.14",
|
||||||
"private": true,
|
"private": true,
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"@ant-design/icons": "^5.2.6",
|
"@ant-design/icons": "^5.2.6",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
[project]
|
[project]
|
||||||
name = "planet"
|
name = "planet"
|
||||||
version = "0.22.13"
|
version = "0.22.14"
|
||||||
description = "智能星球计划 - 态势感知系统"
|
description = "智能星球计划 - 态势感知系统"
|
||||||
requires-python = ">=3.14"
|
requires-python = ">=3.14"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
|
|||||||
Reference in New Issue
Block a user