Files
network-v1/scripts/sync_iana_ipv4.py
T

346 lines
15 KiB
Python

#!/usr/bin/env python3
"""Synchronize authoritative IANA IPv4 range registries into Markdown and CSV."""
from __future__ import annotations
import argparse
import csv
import io
import re
import urllib.request
from collections import Counter
from datetime import datetime, timezone
from ipaddress import IPv4Network, ip_network
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
RANGE_DIR = ROOT / "registry-ips"
DATA_DIR = ROOT / "data"
GENERATED_MARKER = "<!-- generated by scripts/sync_iana_ipv4.py; do not edit manually -->"
TOP_LEVEL_URL = "https://www.iana.org/assignments/ipv4-address-space/ipv4-address-space.csv"
TOP_LEVEL_SOURCE = "https://www.iana.org/assignments/ipv4-address-space/ipv4-address-space.xhtml"
SPECIAL_URL = "https://www.iana.org/assignments/iana-ipv4-special-registry/iana-ipv4-special-registry-1.csv"
SPECIAL_SOURCE = "https://www.iana.org/assignments/iana-ipv4-special-registry/iana-ipv4-special-registry.xhtml"
TOP_FIELDS = ["Prefix", "Designation", "Date", "WHOIS", "RDAP", "Status [1]", "Note"]
SPECIAL_FIELDS = [
"Address Block", "Name", "RFC", "Allocation Date", "Termination Date",
"Source", "Destination", "Forwardable", "Globally Reachable", "Reserved-by-Protocol",
]
def top_level_network(value: str) -> IPv4Network:
match = re.fullmatch(r"(\d{3})/8", value.strip())
if not match:
raise ValueError(f"invalid IANA top-level prefix: {value!r}")
first = int(match.group(1))
if not 0 <= first <= 255:
raise ValueError(f"invalid first octet: {first}")
return IPv4Network(f"{first}.0.0.0/8")
def registry_slug(designation: str, rdap: str) -> str:
value = f"{designation} {rdap}".lower()
if "afrinic" in value:
return "afrinic"
if "apnic" in value:
return "apnic"
if "ripe" in value:
return "ripe-ncc"
if "lacnic" in value:
return "lacnic"
if "arin" in value:
return "arin"
if not rdap.strip() or "iana" in value:
return "iana"
raise ValueError(f"cannot derive registry from designation={designation!r}, rdap={rdap!r}")
def clean_rdap(value: str) -> str:
cleaned = value.strip()
boundaries = [position for scheme in ("https://", "http://") if (position := cleaned.find(scheme, 1)) > 0]
return cleaned[:min(boundaries)] if boundaries else cleaned
def range_filename(network: IPv4Network, suffix: str) -> str:
octets = "-".join(f"{int(part):03d}" for part in str(network.network_address).split("."))
return f"{octets}--{network.prefixlen:02d}--{suffix}.md"
def expand_address_blocks(value: str) -> list[IPv4Network]:
networks = []
for part in value.split(","):
cleaned = re.sub(r"\s+\[\d+\]\s*$", "", part.strip())
parsed = ip_network(cleaned, strict=True)
if not isinstance(parsed, IPv4Network):
raise ValueError(f"not IPv4: {part!r}")
networks.append(parsed)
return networks
def parse_top_level_csv(text: str) -> list[dict[str, object]]:
reader = csv.DictReader(io.StringIO(text.lstrip("\ufeff")))
if reader.fieldnames != TOP_FIELDS:
raise ValueError(f"unexpected top-level columns: {reader.fieldnames!r}")
rows: list[dict[str, object]] = []
for raw in reader:
network = top_level_network(raw["Prefix"])
rows.append({**raw, "network": network, "registry": registry_slug(raw["Designation"], raw["RDAP"])})
networks = {row["network"] for row in rows}
expected = {IPv4Network(f"{first}.0.0.0/8") for first in range(256)}
if len(rows) != 256 or networks != expected:
raise ValueError(f"top-level registry must contain exactly 256 unique /8 ranges; found {len(rows)} rows")
return sorted(rows, key=lambda row: int(row["network"].network_address))
def parse_special_csv(text: str) -> list[dict[str, object]]:
reader = csv.DictReader(io.StringIO(text.lstrip("\ufeff")))
if reader.fieldnames != SPECIAL_FIELDS:
raise ValueError(f"unexpected special-purpose columns: {reader.fieldnames!r}")
rows: list[dict[str, object]] = []
for raw in reader:
for network in expand_address_blocks(raw["Address Block"]):
rows.append({**raw, "network": network})
prefixes = [row["network"] for row in rows]
if len(prefixes) != len(set(prefixes)):
raise ValueError("special-purpose registry contains duplicate expanded prefixes")
return sorted(rows, key=lambda row: (int(row["network"].network_address), row["network"].prefixlen))
def fetch_text(url: str) -> str:
request = urllib.request.Request(url, headers={"User-Agent": "network-v1-registry-sync/1.0"})
with urllib.request.urlopen(request, timeout=60) as response:
return response.read().decode("utf-8-sig")
def write_generated(path: Path, content: str) -> bool:
if path.exists() and GENERATED_MARKER not in path.read_text(encoding="utf-8"):
return False
path.parent.mkdir(parents=True, exist_ok=True)
if path.exists() and path.read_text(encoding="utf-8") == content:
return False
path.write_text(content, encoding="utf-8")
return True
def attribution_text(row: dict[str, object]) -> str:
network = row["network"]
designation = str(row["Designation"]).strip() or "not stated"
registry = str(row["registry"])
status = str(row["Status [1]"]).lower()
if registry == "iana":
return (
f"IANA records **{network}** as **{designation}** with status **{status}**. "
"It is not presented here as an ordinary allocation with one network operator."
)
return (
f"At the top-level `/8` registry layer, IANA designates **{network}** to **{designation}** "
f"and the RDAP authority maps to **{registry}**. This does not identify one holder or operator "
"for every address inside the block; detailed attribution requires more-specific RDAP and BGP research."
)
def render_top_level(row: dict[str, object], checked: str) -> str:
network = row["network"]
registry = str(row["registry"])
designation = str(row["Designation"]).strip() or "Not stated"
return f"""{GENERATED_MARKER}
# {network} — IANA top-level IPv4 registry
## Who is this IP range?
{attribution_text(row)}
## Registry record
- **Range:** `{network}`
- **IANA designation:** {designation}
- **Registry suffix:** `{registry}`
- **IANA status:** {row['Status [1]']}
- **Allocation/record date:** {row['Date'] or 'Not stated'}
- **WHOIS authority:** {row['WHOIS'] or 'Not stated'}
- **RDAP authority:** {clean_rdap(str(row['RDAP'])) or 'Not stated'}
- **IANA note:** {row['Note'] or 'None'}
- **Checked:** {checked}
## Scope and limitations
This record identifies the authoritative **top-level `/8` disposition**. It does not enumerate the many more-specific allocations, assignments, route origins, operators, services, or exact addresses inside the range.
A registry designation is not proof of legal ownership, current routing, physical location, or service operation. Use more-specific RDAP, BGP, RPKI, and operator sources before making those claims.
## Source
- [IANA IPv4 Address Space registry]({TOP_LEVEL_SOURCE}), checked {checked}
"""
def clean_name(value: str) -> str:
return " ".join(value.strip().strip('"').split())
def special_filename(network: IPv4Network) -> str:
if network == IPv4Network("0.0.0.0/8"):
return range_filename(network, "iana")
return range_filename(network, "iana-special")
def render_special(row: dict[str, object], checked: str) -> str:
network = row["network"]
name = clean_name(str(row["Name"]))
return f"""{GENERATED_MARKER}
# {network} — {name}
## Who is this IP range?
IANA identifies **{network}** as **{name}** special-purpose IPv4 space. This is a protocol or standards classification, not an assertion that one company owns or operates every address in the range.
## Special-purpose properties
- **Range:** `{network}`
- **Name:** {name}
- **RFC/reference:** {clean_name(str(row['RFC'])) or 'Not stated'}
- **Allocation date:** {row['Allocation Date'] or 'Not stated'}
- **Termination date:** {row['Termination Date'] or 'Not stated'}
- **Valid as source:** {row['Source'] or 'Not stated'}
- **Valid as destination:** {row['Destination'] or 'Not stated'}
- **Forwardable:** {row['Forwardable'] or 'Not stated'}
- **Globally reachable:** {row['Globally Reachable'] or 'Not stated'}
- **Reserved by protocol:** {row['Reserved-by-Protocol'] or 'Not stated'}
- **Checked:** {checked}
## Interpretation
The properties above are copied from the authoritative IANA special-purpose registry. Footnote markers are retained because exceptions and qualifications matter. Routing or operational claims require separate current evidence.
## Source
- [IANA IPv4 Special-Purpose Address Registry]({SPECIAL_SOURCE}), checked {checked}
"""
def write_csv(path: Path, fieldnames: list[str], rows: list[dict[str, str]]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
with path.open("w", newline="", encoding="utf-8") as handle:
writer = csv.DictWriter(handle, fieldnames=fieldnames, lineterminator="\n")
writer.writeheader()
writer.writerows(rows)
def render_index(top_rows: list[dict[str, object]], special_rows: list[dict[str, object]], checked: str) -> str:
registry_counts = Counter(str(row["registry"]) for row in top_rows)
status_counts = Counter(str(row["Status [1]"]).lower() for row in top_rows)
lines = [
GENERATED_MARKER,
"# Complete IANA IPv4 range index",
"",
f"Generated from authoritative IANA registries and checked **{checked}**.",
"",
"## Coverage",
"",
f"- **{len(top_rows)} top-level `/8` records**, covering all IPv4 addresses from `0.0.0.0` through `255.255.255.255` without gaps.",
f"- **{len(special_rows)} expanded special-purpose prefixes** from the IANA IPv4 Special-Purpose Address Registry.",
"- More-specific manual research records and placeholders remain alongside these authoritative baselines.",
"",
"## Top-level counts",
"",
"### By RDAP registry authority",
"",
]
lines.extend(f"- `{name}`: {count}" for name, count in sorted(registry_counts.items()))
lines.extend(["", "### By IANA status", ""])
lines.extend(f"- `{name}`: {count}" for name, count in sorted(status_counts.items()))
lines.extend([
"", "## All top-level IPv4 /8 ranges", "",
"| Range | IANA designation | Status | Registry/RDAP | Record |",
"|---|---|---|---|---|",
])
for row in top_rows:
network = row["network"]
filename = range_filename(network, str(row["registry"]))
designation = str(row["Designation"]).replace("|", "\\|")
lines.append(f"| `{network}` | {designation} | {row['Status [1]']} | `{row['registry']}` | [record]({filename}) |")
lines.extend([
"", "## IANA special-purpose ranges", "",
"Special-purpose prefixes can overlap top-level `/8` records and each other because they express more-specific protocol semantics.",
"", "| Range | Name | Globally reachable | Record |",
"|---|---|---|---|",
])
for row in special_rows:
network = row["network"]
name = clean_name(str(row["Name"])).replace("|", "\\|")
lines.append(f"| `{network}` | {name} | {row['Globally Reachable'] or 'Not stated'} | [record]({special_filename(network)}) |")
return "\n".join(lines) + "\n"
def sync(top_text: str, special_text: str, checked: str) -> tuple[int, int]:
top_rows = parse_top_level_csv(top_text)
special_rows = parse_special_csv(special_text)
expected_generated: set[Path] = set()
top_csv_rows: list[dict[str, str]] = []
for row in top_rows:
network = row["network"]
filename = range_filename(network, str(row["registry"]))
path = RANGE_DIR / filename
write_generated(path, render_top_level(row, checked))
if path.exists() and GENERATED_MARKER in path.read_text(encoding="utf-8"):
expected_generated.add(path)
top_csv_rows.append({
"prefix": str(network), "designation": str(row["Designation"]),
"registry": str(row["registry"]), "status": str(row["Status [1]"]).lower(),
"allocation_date": str(row["Date"]), "whois": str(row["WHOIS"]),
"rdap": clean_rdap(str(row["RDAP"])), "note": str(row["Note"]),
"source_url": TOP_LEVEL_SOURCE, "verified_at": checked, "filename": filename,
})
special_csv_rows: list[dict[str, str]] = []
for row in special_rows:
network = row["network"]
filename = special_filename(network)
path = RANGE_DIR / filename
write_generated(path, render_special(row, checked))
if path.exists() and GENERATED_MARKER in path.read_text(encoding="utf-8"):
expected_generated.add(path)
special_csv_rows.append({
"prefix": str(network), "name": clean_name(str(row["Name"])), "rfc": clean_name(str(row["RFC"])),
"allocation_date": str(row["Allocation Date"]), "termination_date": str(row["Termination Date"]),
"source": str(row["Source"]), "destination": str(row["Destination"]),
"forwardable": str(row["Forwardable"]), "globally_reachable": str(row["Globally Reachable"]),
"reserved_by_protocol": str(row["Reserved-by-Protocol"]), "source_url": SPECIAL_SOURCE,
"verified_at": checked, "filename": filename,
})
for path in RANGE_DIR.glob("*.md"):
if path not in expected_generated and path.name != "IPV4-INDEX.md" and GENERATED_MARKER in path.read_text(encoding="utf-8"):
path.unlink()
write_csv(
DATA_DIR / "iana-ipv4-top-level-ranges.csv",
["prefix", "designation", "registry", "status", "allocation_date", "whois", "rdap", "note", "source_url", "verified_at", "filename"],
top_csv_rows,
)
write_csv(
DATA_DIR / "iana-ipv4-special-ranges.csv",
["prefix", "name", "rfc", "allocation_date", "termination_date", "source", "destination", "forwardable", "globally_reachable", "reserved_by_protocol", "source_url", "verified_at", "filename"],
special_csv_rows,
)
(RANGE_DIR / "IPV4-INDEX.md").write_text(render_index(top_rows, special_rows, checked), encoding="utf-8")
return len(top_rows), len(special_rows)
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--top-level-file", type=Path)
parser.add_argument("--special-file", type=Path)
parser.add_argument("--checked", default=datetime.now(timezone.utc).date().isoformat())
args = parser.parse_args()
top_text = args.top_level_file.read_text(encoding="utf-8-sig") if args.top_level_file else fetch_text(TOP_LEVEL_URL)
special_text = args.special_file.read_text(encoding="utf-8-sig") if args.special_file else fetch_text(SPECIAL_URL)
top_count, special_count = sync(top_text, special_text, args.checked)
print(f"Synchronized {top_count} top-level IPv4 ranges and {special_count} special-purpose prefixes.")
return 0
if __name__ == "__main__":
raise SystemExit(main())