#!/usr/bin/env python3 """Synchronize authoritative IANA IPv4 range registries into Markdown and CSV.""" from __future__ import annotations import argparse import csv import io import re import urllib.request from collections import Counter from datetime import datetime, timezone from ipaddress import IPv4Network, ip_network from pathlib import Path ROOT = Path(__file__).resolve().parents[1] RANGE_DIR = ROOT / "registry-ips" DATA_DIR = ROOT / "data" GENERATED_MARKER = "" TOP_LEVEL_URL = "https://www.iana.org/assignments/ipv4-address-space/ipv4-address-space.csv" TOP_LEVEL_SOURCE = "https://www.iana.org/assignments/ipv4-address-space/ipv4-address-space.xhtml" SPECIAL_URL = "https://www.iana.org/assignments/iana-ipv4-special-registry/iana-ipv4-special-registry-1.csv" SPECIAL_SOURCE = "https://www.iana.org/assignments/iana-ipv4-special-registry/iana-ipv4-special-registry.xhtml" TOP_FIELDS = ["Prefix", "Designation", "Date", "WHOIS", "RDAP", "Status [1]", "Note"] SPECIAL_FIELDS = [ "Address Block", "Name", "RFC", "Allocation Date", "Termination Date", "Source", "Destination", "Forwardable", "Globally Reachable", "Reserved-by-Protocol", ] def top_level_network(value: str) -> IPv4Network: match = re.fullmatch(r"(\d{3})/8", value.strip()) if not match: raise ValueError(f"invalid IANA top-level prefix: {value!r}") first = int(match.group(1)) if not 0 <= first <= 255: raise ValueError(f"invalid first octet: {first}") return IPv4Network(f"{first}.0.0.0/8") def registry_slug(designation: str, rdap: str) -> str: value = f"{designation} {rdap}".lower() if "afrinic" in value: return "afrinic" if "apnic" in value: return "apnic" if "ripe" in value: return "ripe-ncc" if "lacnic" in value: return "lacnic" if "arin" in value: return "arin" if not rdap.strip() or "iana" in value: return "iana" raise ValueError(f"cannot derive registry from designation={designation!r}, rdap={rdap!r}") def clean_rdap(value: str) -> str: cleaned = value.strip() boundaries = [position for scheme in ("https://", "http://") if (position := cleaned.find(scheme, 1)) > 0] return cleaned[:min(boundaries)] if boundaries else cleaned def range_filename(network: IPv4Network, suffix: str) -> str: octets = "-".join(f"{int(part):03d}" for part in str(network.network_address).split(".")) return f"{octets}--{network.prefixlen:02d}--{suffix}.md" def expand_address_blocks(value: str) -> list[IPv4Network]: networks = [] for part in value.split(","): cleaned = re.sub(r"\s+\[\d+\]\s*$", "", part.strip()) parsed = ip_network(cleaned, strict=True) if not isinstance(parsed, IPv4Network): raise ValueError(f"not IPv4: {part!r}") networks.append(parsed) return networks def parse_top_level_csv(text: str) -> list[dict[str, object]]: reader = csv.DictReader(io.StringIO(text.lstrip("\ufeff"))) if reader.fieldnames != TOP_FIELDS: raise ValueError(f"unexpected top-level columns: {reader.fieldnames!r}") rows: list[dict[str, object]] = [] for raw in reader: network = top_level_network(raw["Prefix"]) rows.append({**raw, "network": network, "registry": registry_slug(raw["Designation"], raw["RDAP"])}) networks = {row["network"] for row in rows} expected = {IPv4Network(f"{first}.0.0.0/8") for first in range(256)} if len(rows) != 256 or networks != expected: raise ValueError(f"top-level registry must contain exactly 256 unique /8 ranges; found {len(rows)} rows") return sorted(rows, key=lambda row: int(row["network"].network_address)) def parse_special_csv(text: str) -> list[dict[str, object]]: reader = csv.DictReader(io.StringIO(text.lstrip("\ufeff"))) if reader.fieldnames != SPECIAL_FIELDS: raise ValueError(f"unexpected special-purpose columns: {reader.fieldnames!r}") rows: list[dict[str, object]] = [] for raw in reader: for network in expand_address_blocks(raw["Address Block"]): rows.append({**raw, "network": network}) prefixes = [row["network"] for row in rows] if len(prefixes) != len(set(prefixes)): raise ValueError("special-purpose registry contains duplicate expanded prefixes") return sorted(rows, key=lambda row: (int(row["network"].network_address), row["network"].prefixlen)) def fetch_text(url: str) -> str: request = urllib.request.Request(url, headers={"User-Agent": "network-v1-registry-sync/1.0"}) with urllib.request.urlopen(request, timeout=60) as response: return response.read().decode("utf-8-sig") def write_generated(path: Path, content: str) -> bool: if path.exists() and GENERATED_MARKER not in path.read_text(encoding="utf-8"): return False path.parent.mkdir(parents=True, exist_ok=True) if path.exists() and path.read_text(encoding="utf-8") == content: return False path.write_text(content, encoding="utf-8") return True def attribution_text(row: dict[str, object]) -> str: network = row["network"] designation = str(row["Designation"]).strip() or "not stated" registry = str(row["registry"]) status = str(row["Status [1]"]).lower() if registry == "iana": return ( f"IANA records **{network}** as **{designation}** with status **{status}**. " "It is not presented here as an ordinary allocation with one network operator." ) return ( f"At the top-level `/8` registry layer, IANA designates **{network}** to **{designation}** " f"and the RDAP authority maps to **{registry}**. This does not identify one holder or operator " "for every address inside the block; detailed attribution requires more-specific RDAP and BGP research." ) def render_top_level(row: dict[str, object], checked: str) -> str: network = row["network"] registry = str(row["registry"]) designation = str(row["Designation"]).strip() or "Not stated" return f"""{GENERATED_MARKER} # {network} — IANA top-level IPv4 registry ## Who is this IP range? {attribution_text(row)} ## Registry record - **Range:** `{network}` - **IANA designation:** {designation} - **Registry suffix:** `{registry}` - **IANA status:** {row['Status [1]']} - **Allocation/record date:** {row['Date'] or 'Not stated'} - **WHOIS authority:** {row['WHOIS'] or 'Not stated'} - **RDAP authority:** {clean_rdap(str(row['RDAP'])) or 'Not stated'} - **IANA note:** {row['Note'] or 'None'} - **Checked:** {checked} ## Scope and limitations This record identifies the authoritative **top-level `/8` disposition**. It does not enumerate the many more-specific allocations, assignments, route origins, operators, services, or exact addresses inside the range. A registry designation is not proof of legal ownership, current routing, physical location, or service operation. Use more-specific RDAP, BGP, RPKI, and operator sources before making those claims. ## Source - [IANA IPv4 Address Space registry]({TOP_LEVEL_SOURCE}), checked {checked} """ def clean_name(value: str) -> str: return " ".join(value.strip().strip('"').split()) def special_filename(network: IPv4Network) -> str: if network == IPv4Network("0.0.0.0/8"): return range_filename(network, "iana") return range_filename(network, "iana-special") def render_special(row: dict[str, object], checked: str) -> str: network = row["network"] name = clean_name(str(row["Name"])) return f"""{GENERATED_MARKER} # {network} — {name} ## Who is this IP range? IANA identifies **{network}** as **{name}** special-purpose IPv4 space. This is a protocol or standards classification, not an assertion that one company owns or operates every address in the range. ## Special-purpose properties - **Range:** `{network}` - **Name:** {name} - **RFC/reference:** {clean_name(str(row['RFC'])) or 'Not stated'} - **Allocation date:** {row['Allocation Date'] or 'Not stated'} - **Termination date:** {row['Termination Date'] or 'Not stated'} - **Valid as source:** {row['Source'] or 'Not stated'} - **Valid as destination:** {row['Destination'] or 'Not stated'} - **Forwardable:** {row['Forwardable'] or 'Not stated'} - **Globally reachable:** {row['Globally Reachable'] or 'Not stated'} - **Reserved by protocol:** {row['Reserved-by-Protocol'] or 'Not stated'} - **Checked:** {checked} ## Interpretation The properties above are copied from the authoritative IANA special-purpose registry. Footnote markers are retained because exceptions and qualifications matter. Routing or operational claims require separate current evidence. ## Source - [IANA IPv4 Special-Purpose Address Registry]({SPECIAL_SOURCE}), checked {checked} """ def write_csv(path: Path, fieldnames: list[str], rows: list[dict[str, str]]) -> None: path.parent.mkdir(parents=True, exist_ok=True) with path.open("w", newline="", encoding="utf-8") as handle: writer = csv.DictWriter(handle, fieldnames=fieldnames, lineterminator="\n") writer.writeheader() writer.writerows(rows) def render_index(top_rows: list[dict[str, object]], special_rows: list[dict[str, object]], checked: str) -> str: registry_counts = Counter(str(row["registry"]) for row in top_rows) status_counts = Counter(str(row["Status [1]"]).lower() for row in top_rows) lines = [ GENERATED_MARKER, "# Complete IANA IPv4 range index", "", f"Generated from authoritative IANA registries and checked **{checked}**.", "", "## Coverage", "", f"- **{len(top_rows)} top-level `/8` records**, covering all IPv4 addresses from `0.0.0.0` through `255.255.255.255` without gaps.", f"- **{len(special_rows)} expanded special-purpose prefixes** from the IANA IPv4 Special-Purpose Address Registry.", "- More-specific manual research records and placeholders remain alongside these authoritative baselines.", "", "## Top-level counts", "", "### By RDAP registry authority", "", ] lines.extend(f"- `{name}`: {count}" for name, count in sorted(registry_counts.items())) lines.extend(["", "### By IANA status", ""]) lines.extend(f"- `{name}`: {count}" for name, count in sorted(status_counts.items())) lines.extend([ "", "## All top-level IPv4 /8 ranges", "", "| Range | IANA designation | Status | Registry/RDAP | Record |", "|---|---|---|---|---|", ]) for row in top_rows: network = row["network"] filename = range_filename(network, str(row["registry"])) designation = str(row["Designation"]).replace("|", "\\|") lines.append(f"| `{network}` | {designation} | {row['Status [1]']} | `{row['registry']}` | [record]({filename}) |") lines.extend([ "", "## IANA special-purpose ranges", "", "Special-purpose prefixes can overlap top-level `/8` records and each other because they express more-specific protocol semantics.", "", "| Range | Name | Globally reachable | Record |", "|---|---|---|---|", ]) for row in special_rows: network = row["network"] name = clean_name(str(row["Name"])).replace("|", "\\|") lines.append(f"| `{network}` | {name} | {row['Globally Reachable'] or 'Not stated'} | [record]({special_filename(network)}) |") return "\n".join(lines) + "\n" def sync(top_text: str, special_text: str, checked: str) -> tuple[int, int]: top_rows = parse_top_level_csv(top_text) special_rows = parse_special_csv(special_text) expected_generated: set[Path] = set() top_csv_rows: list[dict[str, str]] = [] for row in top_rows: network = row["network"] filename = range_filename(network, str(row["registry"])) path = RANGE_DIR / filename write_generated(path, render_top_level(row, checked)) if path.exists() and GENERATED_MARKER in path.read_text(encoding="utf-8"): expected_generated.add(path) top_csv_rows.append({ "prefix": str(network), "designation": str(row["Designation"]), "registry": str(row["registry"]), "status": str(row["Status [1]"]).lower(), "allocation_date": str(row["Date"]), "whois": str(row["WHOIS"]), "rdap": clean_rdap(str(row["RDAP"])), "note": str(row["Note"]), "source_url": TOP_LEVEL_SOURCE, "verified_at": checked, "filename": filename, }) special_csv_rows: list[dict[str, str]] = [] for row in special_rows: network = row["network"] filename = special_filename(network) path = RANGE_DIR / filename write_generated(path, render_special(row, checked)) if path.exists() and GENERATED_MARKER in path.read_text(encoding="utf-8"): expected_generated.add(path) special_csv_rows.append({ "prefix": str(network), "name": clean_name(str(row["Name"])), "rfc": clean_name(str(row["RFC"])), "allocation_date": str(row["Allocation Date"]), "termination_date": str(row["Termination Date"]), "source": str(row["Source"]), "destination": str(row["Destination"]), "forwardable": str(row["Forwardable"]), "globally_reachable": str(row["Globally Reachable"]), "reserved_by_protocol": str(row["Reserved-by-Protocol"]), "source_url": SPECIAL_SOURCE, "verified_at": checked, "filename": filename, }) for path in RANGE_DIR.glob("*.md"): if path not in expected_generated and path.name != "IPV4-INDEX.md" and GENERATED_MARKER in path.read_text(encoding="utf-8"): path.unlink() write_csv( DATA_DIR / "iana-ipv4-top-level-ranges.csv", ["prefix", "designation", "registry", "status", "allocation_date", "whois", "rdap", "note", "source_url", "verified_at", "filename"], top_csv_rows, ) write_csv( DATA_DIR / "iana-ipv4-special-ranges.csv", ["prefix", "name", "rfc", "allocation_date", "termination_date", "source", "destination", "forwardable", "globally_reachable", "reserved_by_protocol", "source_url", "verified_at", "filename"], special_csv_rows, ) (RANGE_DIR / "IPV4-INDEX.md").write_text(render_index(top_rows, special_rows, checked), encoding="utf-8") return len(top_rows), len(special_rows) def main() -> int: parser = argparse.ArgumentParser() parser.add_argument("--top-level-file", type=Path) parser.add_argument("--special-file", type=Path) parser.add_argument("--checked", default=datetime.now(timezone.utc).date().isoformat()) args = parser.parse_args() top_text = args.top_level_file.read_text(encoding="utf-8-sig") if args.top_level_file else fetch_text(TOP_LEVEL_URL) special_text = args.special_file.read_text(encoding="utf-8-sig") if args.special_file else fetch_text(SPECIAL_URL) top_count, special_count = sync(top_text, special_text, args.checked) print(f"Synchronized {top_count} top-level IPv4 ranges and {special_count} special-purpose prefixes.") return 0 if __name__ == "__main__": raise SystemExit(main())