#!/usr/bin/env python3 """Validate the round-1 network address book without third-party dependencies.""" from __future__ import annotations import csv import re import sys from collections import Counter from datetime import date from pathlib import Path from urllib.parse import urlparse ROOT = Path(__file__).resolve().parents[1] DATA = ROOT / "data" / "entities.csv" IP_DATA = ROOT / "data" / "ip-addresses.csv" REQUIRED = [ "id", "name", "category", "scope", "region", "homepage", "directory_url", "role", "source_url", "status", "verified_at", "notes", ] CATEGORIES = { "governance-identifiers", "standards", "registry", "dns", "interconnection", "routing-security", "incident-response", "physical-infrastructure", "operator-community", } SCOPES = {"global", "regional", "regional-community"} STATUSES = {"verified", "partial", "lead", "stale"} ID_RE = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$") IP_REQUIRED = [ "address", "exact_prefix", "parent_prefix", "classification", "registry", "registered_to", "announced_prefix", "origin_asn", "origin_holder", "operator", "service", "globally_routable", "source_registry", "source_routing", "source_service", "verified_at", "notes", ] def valid_url(value: str) -> bool: parsed = urlparse(value) return parsed.scheme == "https" and bool(parsed.netloc) def main() -> int: errors: list[str] = [] with DATA.open(newline="", encoding="utf-8") as handle: reader = csv.DictReader(handle) if reader.fieldnames != REQUIRED: errors.append(f"CSV columns differ from required schema: {reader.fieldnames!r}") rows = list(reader) ids = Counter(row.get("id", "") for row in rows) for row_number, row in enumerate(rows, start=2): ident = row.get("id", "") prefix = f"row {row_number} ({ident or 'missing-id'})" for field in REQUIRED[:-1]: # notes may be blank if not row.get(field, "").strip(): errors.append(f"{prefix}: missing {field}") if not ID_RE.fullmatch(ident): errors.append(f"{prefix}: invalid id") if row.get("category") not in CATEGORIES: errors.append(f"{prefix}: unknown category {row.get('category')!r}") if row.get("scope") not in SCOPES: errors.append(f"{prefix}: unknown scope {row.get('scope')!r}") if row.get("status") not in STATUSES: errors.append(f"{prefix}: unknown status {row.get('status')!r}") for field in ("homepage", "directory_url", "source_url"): if not valid_url(row.get(field, "")): errors.append(f"{prefix}: {field} must be an https URL") try: checked = date.fromisoformat(row.get("verified_at", "")) if checked > date.today(): errors.append(f"{prefix}: verified_at is in the future") except ValueError: errors.append(f"{prefix}: verified_at must be YYYY-MM-DD") for ident, count in ids.items(): if count > 1: errors.append(f"duplicate id {ident!r}: {count} rows") with IP_DATA.open(newline="", encoding="utf-8") as handle: ip_reader = csv.DictReader(handle) if ip_reader.fieldnames != IP_REQUIRED: errors.append(f"IP CSV columns differ from required schema: {ip_reader.fieldnames!r}") ip_rows = list(ip_reader) ip_addresses = Counter(row.get("address", "") for row in ip_rows) for row_number, row in enumerate(ip_rows, start=2): address = row.get("address", "") prefix = f"IP row {row_number} ({address or 'missing-address'})" for field in ("address", "exact_prefix", "parent_prefix", "classification", "registry", "registered_to", "service", "globally_routable", "source_registry", "verified_at"): if not row.get(field, "").strip(): errors.append(f"{prefix}: missing {field}") try: import ipaddress parsed = ipaddress.ip_address(address) exact = ipaddress.ip_network(row.get("exact_prefix", ""), strict=False) parent = ipaddress.ip_network(row.get("parent_prefix", ""), strict=False) if parsed not in exact or parsed not in parent: errors.append(f"{prefix}: address must be contained by exact and parent prefixes") except ValueError as exc: errors.append(f"{prefix}: invalid address/prefix: {exc}") if row.get("globally_routable") not in {"true", "false"}: errors.append(f"{prefix}: globally_routable must be true or false") for field in ("source_registry", "source_routing", "source_service"): value = row.get(field, "").strip() if value and not valid_url(value): errors.append(f"{prefix}: {field} must be an https URL when present") try: checked = date.fromisoformat(row.get("verified_at", "")) if checked > date.today(): errors.append(f"{prefix}: verified_at is in the future") except ValueError: errors.append(f"{prefix}: verified_at must be YYYY-MM-DD") for address, count in ip_addresses.items(): if count > 1: errors.append(f"duplicate IP address {address!r}: {count} rows") required_docs = [ ROOT / "README.md", ROOT / "docs" / "02-first-round-investigation.md", ROOT / "address-book" / "README.md", ROOT / "address-book" / "ip-addresses.md", ROOT / "docs" / "04-ip-address-model.md", ROOT / "registry-ips" / "README.md", ROOT / "registry-ips" / "iana--0-0-0-0--8.md", ROOT / "registry-ips" / "apnic--1-0-0-0--16.md", ROOT / "registry-ips" / "apnic--1-1-0-0--24.md", ROOT / "registry-ips" / "apnic--1-1-1-0--24.md", ROOT / "research" / "next-round.md", ] for path in required_docs: if not path.is_file(): errors.append(f"missing required document: {path.relative_to(ROOT)}") if errors: print("Validation failed:") for error in errors: print(f"- {error}") return 1 counts = Counter(row["category"] for row in rows) print(f"Validated {len(rows)} entities across {len(counts)} categories and {len(ip_rows)} IP address records.") for category, count in sorted(counts.items()): print(f"- {category}: {count}") return 0 if __name__ == "__main__": sys.exit(main())