Files
network-v1/scripts/validate.py
T

167 lines
7.4 KiB
Python

#!/usr/bin/env python3
"""Validate the round-1 network address book without third-party dependencies."""
from __future__ import annotations
import csv
import re
import sys
from collections import Counter
from datetime import date
from pathlib import Path
from urllib.parse import urlparse
ROOT = Path(__file__).resolve().parents[1]
DATA = ROOT / "data" / "entities.csv"
IP_DATA = ROOT / "data" / "ip-addresses.csv"
REQUIRED = [
"id", "name", "category", "scope", "region", "homepage",
"directory_url", "role", "source_url", "status", "verified_at", "notes",
]
CATEGORIES = {
"governance-identifiers", "standards", "registry", "dns", "interconnection",
"routing-security", "incident-response", "physical-infrastructure", "operator-community",
}
SCOPES = {"global", "regional", "regional-community"}
STATUSES = {"verified", "partial", "lead", "stale"}
ID_RE = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$")
IP_REQUIRED = [
"address", "exact_prefix", "parent_prefix", "classification", "registry",
"registered_to", "announced_prefix", "origin_asn", "origin_holder", "operator",
"service", "globally_routable", "source_registry", "source_routing",
"source_service", "verified_at", "notes",
]
def valid_url(value: str) -> bool:
parsed = urlparse(value)
return parsed.scheme == "https" and bool(parsed.netloc)
def main() -> int:
errors: list[str] = []
with DATA.open(newline="", encoding="utf-8") as handle:
reader = csv.DictReader(handle)
if reader.fieldnames != REQUIRED:
errors.append(f"CSV columns differ from required schema: {reader.fieldnames!r}")
rows = list(reader)
ids = Counter(row.get("id", "") for row in rows)
for row_number, row in enumerate(rows, start=2):
ident = row.get("id", "")
prefix = f"row {row_number} ({ident or 'missing-id'})"
for field in REQUIRED[:-1]: # notes may be blank
if not row.get(field, "").strip():
errors.append(f"{prefix}: missing {field}")
if not ID_RE.fullmatch(ident):
errors.append(f"{prefix}: invalid id")
if row.get("category") not in CATEGORIES:
errors.append(f"{prefix}: unknown category {row.get('category')!r}")
if row.get("scope") not in SCOPES:
errors.append(f"{prefix}: unknown scope {row.get('scope')!r}")
if row.get("status") not in STATUSES:
errors.append(f"{prefix}: unknown status {row.get('status')!r}")
for field in ("homepage", "directory_url", "source_url"):
if not valid_url(row.get(field, "")):
errors.append(f"{prefix}: {field} must be an https URL")
try:
checked = date.fromisoformat(row.get("verified_at", ""))
if checked > date.today():
errors.append(f"{prefix}: verified_at is in the future")
except ValueError:
errors.append(f"{prefix}: verified_at must be YYYY-MM-DD")
for ident, count in ids.items():
if count > 1:
errors.append(f"duplicate id {ident!r}: {count} rows")
with IP_DATA.open(newline="", encoding="utf-8") as handle:
ip_reader = csv.DictReader(handle)
if ip_reader.fieldnames != IP_REQUIRED:
errors.append(f"IP CSV columns differ from required schema: {ip_reader.fieldnames!r}")
ip_rows = list(ip_reader)
ip_addresses = Counter(row.get("address", "") for row in ip_rows)
for row_number, row in enumerate(ip_rows, start=2):
address = row.get("address", "")
prefix = f"IP row {row_number} ({address or 'missing-address'})"
for field in ("address", "exact_prefix", "parent_prefix", "classification", "registry",
"registered_to", "service", "globally_routable", "source_registry", "verified_at"):
if not row.get(field, "").strip():
errors.append(f"{prefix}: missing {field}")
try:
import ipaddress
parsed = ipaddress.ip_address(address)
exact = ipaddress.ip_network(row.get("exact_prefix", ""), strict=False)
parent = ipaddress.ip_network(row.get("parent_prefix", ""), strict=False)
if parsed not in exact or parsed not in parent:
errors.append(f"{prefix}: address must be contained by exact and parent prefixes")
except ValueError as exc:
errors.append(f"{prefix}: invalid address/prefix: {exc}")
if row.get("globally_routable") not in {"true", "false"}:
errors.append(f"{prefix}: globally_routable must be true or false")
for field in ("source_registry", "source_routing", "source_service"):
value = row.get(field, "").strip()
if value and not valid_url(value):
errors.append(f"{prefix}: {field} must be an https URL when present")
try:
checked = date.fromisoformat(row.get("verified_at", ""))
if checked > date.today():
errors.append(f"{prefix}: verified_at is in the future")
except ValueError:
errors.append(f"{prefix}: verified_at must be YYYY-MM-DD")
for address, count in ip_addresses.items():
if count > 1:
errors.append(f"duplicate IP address {address!r}: {count} rows")
required_docs = [
ROOT / "README.md",
ROOT / "docs" / "02-first-round-investigation.md",
ROOT / "address-book" / "README.md",
ROOT / "address-book" / "ip-addresses.md",
ROOT / "address-book" / "ip" / "TEMPLATE.md",
ROOT / "docs" / "04-ip-address-model.md",
ROOT / "registry-ips" / "README.md",
ROOT / "registry-ips" / "TEMPLATE.md",
ROOT / "registry-ips" / "000-000-000-000--08--iana.md",
ROOT / "registry-ips" / "001-000-000-000--16--placeholder.md",
ROOT / "registry-ips" / "001-001-000-000--24--placeholder.md",
ROOT / "registry-ips" / "001-001-001-000--24--apnic.md",
ROOT / "research" / "next-round.md",
]
for path in required_docs:
if not path.is_file():
errors.append(f"missing required document: {path.relative_to(ROOT)}")
range_records = sorted((ROOT / "registry-ips").glob("*.md"))
for path in range_records:
if path.name in {"README.md", "TEMPLATE.md"}:
continue
headings = [line.strip() for line in path.read_text(encoding="utf-8").splitlines() if line.startswith("## ")]
if not headings or headings[0] != "## Who is this IP range?":
errors.append(f"{path.relative_to(ROOT)}: first section must be '## Who is this IP range?'")
address_records = sorted((ROOT / "address-book" / "ip").glob("*.md"))
for path in address_records:
if path.name == "TEMPLATE.md":
continue
headings = [line.strip() for line in path.read_text(encoding="utf-8").splitlines() if line.startswith("## ")]
if not headings or headings[0] != "## Who is this IP?":
errors.append(f"{path.relative_to(ROOT)}: first section must be '## Who is this IP?'")
if errors:
print("Validation failed:")
for error in errors:
print(f"- {error}")
return 1
counts = Counter(row["category"] for row in rows)
print(f"Validated {len(rows)} entities across {len(counts)} categories and {len(ip_rows)} IP address records.")
for category, count in sorted(counts.items()):
print(f"- {category}: {count}")
return 0
if __name__ == "__main__":
sys.exit(main())