research: begin IP address ownership and routing map

This commit is contained in:
2026-07-20 18:52:14 +00:00
parent 4752d490d4
commit ced31f7d8b
11 changed files with 215 additions and 5 deletions
+49 -1
View File
@@ -13,6 +13,7 @@ from urllib.parse import urlparse
ROOT = Path(__file__).resolve().parents[1]
DATA = ROOT / "data" / "entities.csv"
IP_DATA = ROOT / "data" / "ip-addresses.csv"
REQUIRED = [
"id", "name", "category", "scope", "region", "homepage",
"directory_url", "role", "source_url", "status", "verified_at", "notes",
@@ -24,6 +25,12 @@ CATEGORIES = {
SCOPES = {"global", "regional", "regional-community"}
STATUSES = {"verified", "partial", "lead", "stale"}
ID_RE = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$")
IP_REQUIRED = [
"address", "exact_prefix", "parent_prefix", "classification", "registry",
"registered_to", "announced_prefix", "origin_asn", "origin_holder", "operator",
"service", "globally_routable", "source_registry", "source_routing",
"source_service", "verified_at", "notes",
]
def valid_url(value: str) -> bool:
@@ -68,10 +75,51 @@ def main() -> int:
if count > 1:
errors.append(f"duplicate id {ident!r}: {count} rows")
with IP_DATA.open(newline="", encoding="utf-8") as handle:
ip_reader = csv.DictReader(handle)
if ip_reader.fieldnames != IP_REQUIRED:
errors.append(f"IP CSV columns differ from required schema: {ip_reader.fieldnames!r}")
ip_rows = list(ip_reader)
ip_addresses = Counter(row.get("address", "") for row in ip_rows)
for row_number, row in enumerate(ip_rows, start=2):
address = row.get("address", "")
prefix = f"IP row {row_number} ({address or 'missing-address'})"
for field in ("address", "exact_prefix", "parent_prefix", "classification", "registry",
"registered_to", "service", "globally_routable", "source_registry", "verified_at"):
if not row.get(field, "").strip():
errors.append(f"{prefix}: missing {field}")
try:
import ipaddress
parsed = ipaddress.ip_address(address)
exact = ipaddress.ip_network(row.get("exact_prefix", ""), strict=False)
parent = ipaddress.ip_network(row.get("parent_prefix", ""), strict=False)
if parsed not in exact or parsed not in parent:
errors.append(f"{prefix}: address must be contained by exact and parent prefixes")
except ValueError as exc:
errors.append(f"{prefix}: invalid address/prefix: {exc}")
if row.get("globally_routable") not in {"true", "false"}:
errors.append(f"{prefix}: globally_routable must be true or false")
for field in ("source_registry", "source_routing", "source_service"):
value = row.get(field, "").strip()
if value and not valid_url(value):
errors.append(f"{prefix}: {field} must be an https URL when present")
try:
checked = date.fromisoformat(row.get("verified_at", ""))
if checked > date.today():
errors.append(f"{prefix}: verified_at is in the future")
except ValueError:
errors.append(f"{prefix}: verified_at must be YYYY-MM-DD")
for address, count in ip_addresses.items():
if count > 1:
errors.append(f"duplicate IP address {address!r}: {count} rows")
required_docs = [
ROOT / "README.md",
ROOT / "docs" / "02-first-round-investigation.md",
ROOT / "address-book" / "README.md",
ROOT / "address-book" / "ip-addresses.md",
ROOT / "docs" / "04-ip-address-model.md",
ROOT / "research" / "next-round.md",
]
for path in required_docs:
@@ -85,7 +133,7 @@ def main() -> int:
return 1
counts = Counter(row["category"] for row in rows)
print(f"Validated {len(rows)} entities across {len(counts)} categories.")
print(f"Validated {len(rows)} entities across {len(counts)} categories and {len(ip_rows)} IP address records.")
for category, count in sorted(counts.items()):
print(f"- {category}: {count}")
return 0