research: index complete IANA IPv4 range baseline
This commit is contained in:
@@ -0,0 +1,345 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Synchronize authoritative IANA IPv4 range registries into Markdown and CSV."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import csv
|
||||
import io
|
||||
import re
|
||||
import urllib.request
|
||||
from collections import Counter
|
||||
from datetime import datetime, timezone
|
||||
from ipaddress import IPv4Network, ip_network
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
RANGE_DIR = ROOT / "registry-ips"
|
||||
DATA_DIR = ROOT / "data"
|
||||
GENERATED_MARKER = "<!-- generated by scripts/sync_iana_ipv4.py; do not edit manually -->"
|
||||
TOP_LEVEL_URL = "https://www.iana.org/assignments/ipv4-address-space/ipv4-address-space.csv"
|
||||
TOP_LEVEL_SOURCE = "https://www.iana.org/assignments/ipv4-address-space/ipv4-address-space.xhtml"
|
||||
SPECIAL_URL = "https://www.iana.org/assignments/iana-ipv4-special-registry/iana-ipv4-special-registry-1.csv"
|
||||
SPECIAL_SOURCE = "https://www.iana.org/assignments/iana-ipv4-special-registry/iana-ipv4-special-registry.xhtml"
|
||||
TOP_FIELDS = ["Prefix", "Designation", "Date", "WHOIS", "RDAP", "Status [1]", "Note"]
|
||||
SPECIAL_FIELDS = [
|
||||
"Address Block", "Name", "RFC", "Allocation Date", "Termination Date",
|
||||
"Source", "Destination", "Forwardable", "Globally Reachable", "Reserved-by-Protocol",
|
||||
]
|
||||
|
||||
|
||||
def top_level_network(value: str) -> IPv4Network:
|
||||
match = re.fullmatch(r"(\d{3})/8", value.strip())
|
||||
if not match:
|
||||
raise ValueError(f"invalid IANA top-level prefix: {value!r}")
|
||||
first = int(match.group(1))
|
||||
if not 0 <= first <= 255:
|
||||
raise ValueError(f"invalid first octet: {first}")
|
||||
return IPv4Network(f"{first}.0.0.0/8")
|
||||
|
||||
|
||||
def registry_slug(designation: str, rdap: str) -> str:
|
||||
value = f"{designation} {rdap}".lower()
|
||||
if "afrinic" in value:
|
||||
return "afrinic"
|
||||
if "apnic" in value:
|
||||
return "apnic"
|
||||
if "ripe" in value:
|
||||
return "ripe-ncc"
|
||||
if "lacnic" in value:
|
||||
return "lacnic"
|
||||
if "arin" in value:
|
||||
return "arin"
|
||||
if not rdap.strip() or "iana" in value:
|
||||
return "iana"
|
||||
raise ValueError(f"cannot derive registry from designation={designation!r}, rdap={rdap!r}")
|
||||
|
||||
|
||||
def clean_rdap(value: str) -> str:
|
||||
cleaned = value.strip()
|
||||
boundaries = [position for scheme in ("https://", "http://") if (position := cleaned.find(scheme, 1)) > 0]
|
||||
return cleaned[:min(boundaries)] if boundaries else cleaned
|
||||
|
||||
|
||||
def range_filename(network: IPv4Network, suffix: str) -> str:
|
||||
octets = "-".join(f"{int(part):03d}" for part in str(network.network_address).split("."))
|
||||
return f"{octets}--{network.prefixlen:02d}--{suffix}.md"
|
||||
|
||||
|
||||
def expand_address_blocks(value: str) -> list[IPv4Network]:
|
||||
networks = []
|
||||
for part in value.split(","):
|
||||
cleaned = re.sub(r"\s+\[\d+\]\s*$", "", part.strip())
|
||||
parsed = ip_network(cleaned, strict=True)
|
||||
if not isinstance(parsed, IPv4Network):
|
||||
raise ValueError(f"not IPv4: {part!r}")
|
||||
networks.append(parsed)
|
||||
return networks
|
||||
|
||||
|
||||
def parse_top_level_csv(text: str) -> list[dict[str, object]]:
|
||||
reader = csv.DictReader(io.StringIO(text.lstrip("\ufeff")))
|
||||
if reader.fieldnames != TOP_FIELDS:
|
||||
raise ValueError(f"unexpected top-level columns: {reader.fieldnames!r}")
|
||||
rows: list[dict[str, object]] = []
|
||||
for raw in reader:
|
||||
network = top_level_network(raw["Prefix"])
|
||||
rows.append({**raw, "network": network, "registry": registry_slug(raw["Designation"], raw["RDAP"])})
|
||||
networks = {row["network"] for row in rows}
|
||||
expected = {IPv4Network(f"{first}.0.0.0/8") for first in range(256)}
|
||||
if len(rows) != 256 or networks != expected:
|
||||
raise ValueError(f"top-level registry must contain exactly 256 unique /8 ranges; found {len(rows)} rows")
|
||||
return sorted(rows, key=lambda row: int(row["network"].network_address))
|
||||
|
||||
|
||||
def parse_special_csv(text: str) -> list[dict[str, object]]:
|
||||
reader = csv.DictReader(io.StringIO(text.lstrip("\ufeff")))
|
||||
if reader.fieldnames != SPECIAL_FIELDS:
|
||||
raise ValueError(f"unexpected special-purpose columns: {reader.fieldnames!r}")
|
||||
rows: list[dict[str, object]] = []
|
||||
for raw in reader:
|
||||
for network in expand_address_blocks(raw["Address Block"]):
|
||||
rows.append({**raw, "network": network})
|
||||
prefixes = [row["network"] for row in rows]
|
||||
if len(prefixes) != len(set(prefixes)):
|
||||
raise ValueError("special-purpose registry contains duplicate expanded prefixes")
|
||||
return sorted(rows, key=lambda row: (int(row["network"].network_address), row["network"].prefixlen))
|
||||
|
||||
|
||||
def fetch_text(url: str) -> str:
|
||||
request = urllib.request.Request(url, headers={"User-Agent": "network-v1-registry-sync/1.0"})
|
||||
with urllib.request.urlopen(request, timeout=60) as response:
|
||||
return response.read().decode("utf-8-sig")
|
||||
|
||||
|
||||
def write_generated(path: Path, content: str) -> bool:
|
||||
if path.exists() and GENERATED_MARKER not in path.read_text(encoding="utf-8"):
|
||||
return False
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
if path.exists() and path.read_text(encoding="utf-8") == content:
|
||||
return False
|
||||
path.write_text(content, encoding="utf-8")
|
||||
return True
|
||||
|
||||
|
||||
def attribution_text(row: dict[str, object]) -> str:
|
||||
network = row["network"]
|
||||
designation = str(row["Designation"]).strip() or "not stated"
|
||||
registry = str(row["registry"])
|
||||
status = str(row["Status [1]"]).lower()
|
||||
if registry == "iana":
|
||||
return (
|
||||
f"IANA records **{network}** as **{designation}** with status **{status}**. "
|
||||
"It is not presented here as an ordinary allocation with one network operator."
|
||||
)
|
||||
return (
|
||||
f"At the top-level `/8` registry layer, IANA designates **{network}** to **{designation}** "
|
||||
f"and the RDAP authority maps to **{registry}**. This does not identify one holder or operator "
|
||||
"for every address inside the block; detailed attribution requires more-specific RDAP and BGP research."
|
||||
)
|
||||
|
||||
|
||||
def render_top_level(row: dict[str, object], checked: str) -> str:
|
||||
network = row["network"]
|
||||
registry = str(row["registry"])
|
||||
designation = str(row["Designation"]).strip() or "Not stated"
|
||||
return f"""{GENERATED_MARKER}
|
||||
# {network} — IANA top-level IPv4 registry
|
||||
|
||||
## Who is this IP range?
|
||||
|
||||
{attribution_text(row)}
|
||||
|
||||
## Registry record
|
||||
|
||||
- **Range:** `{network}`
|
||||
- **IANA designation:** {designation}
|
||||
- **Registry suffix:** `{registry}`
|
||||
- **IANA status:** {row['Status [1]']}
|
||||
- **Allocation/record date:** {row['Date'] or 'Not stated'}
|
||||
- **WHOIS authority:** {row['WHOIS'] or 'Not stated'}
|
||||
- **RDAP authority:** {clean_rdap(str(row['RDAP'])) or 'Not stated'}
|
||||
- **IANA note:** {row['Note'] or 'None'}
|
||||
- **Checked:** {checked}
|
||||
|
||||
## Scope and limitations
|
||||
|
||||
This record identifies the authoritative **top-level `/8` disposition**. It does not enumerate the many more-specific allocations, assignments, route origins, operators, services, or exact addresses inside the range.
|
||||
|
||||
A registry designation is not proof of legal ownership, current routing, physical location, or service operation. Use more-specific RDAP, BGP, RPKI, and operator sources before making those claims.
|
||||
|
||||
## Source
|
||||
|
||||
- [IANA IPv4 Address Space registry]({TOP_LEVEL_SOURCE}), checked {checked}
|
||||
"""
|
||||
|
||||
|
||||
def clean_name(value: str) -> str:
|
||||
return " ".join(value.strip().strip('"').split())
|
||||
|
||||
|
||||
def special_filename(network: IPv4Network) -> str:
|
||||
if network == IPv4Network("0.0.0.0/8"):
|
||||
return range_filename(network, "iana")
|
||||
return range_filename(network, "iana-special")
|
||||
|
||||
|
||||
def render_special(row: dict[str, object], checked: str) -> str:
|
||||
network = row["network"]
|
||||
name = clean_name(str(row["Name"]))
|
||||
return f"""{GENERATED_MARKER}
|
||||
# {network} — {name}
|
||||
|
||||
## Who is this IP range?
|
||||
|
||||
IANA identifies **{network}** as **{name}** special-purpose IPv4 space. This is a protocol or standards classification, not an assertion that one company owns or operates every address in the range.
|
||||
|
||||
## Special-purpose properties
|
||||
|
||||
- **Range:** `{network}`
|
||||
- **Name:** {name}
|
||||
- **RFC/reference:** {clean_name(str(row['RFC'])) or 'Not stated'}
|
||||
- **Allocation date:** {row['Allocation Date'] or 'Not stated'}
|
||||
- **Termination date:** {row['Termination Date'] or 'Not stated'}
|
||||
- **Valid as source:** {row['Source'] or 'Not stated'}
|
||||
- **Valid as destination:** {row['Destination'] or 'Not stated'}
|
||||
- **Forwardable:** {row['Forwardable'] or 'Not stated'}
|
||||
- **Globally reachable:** {row['Globally Reachable'] or 'Not stated'}
|
||||
- **Reserved by protocol:** {row['Reserved-by-Protocol'] or 'Not stated'}
|
||||
- **Checked:** {checked}
|
||||
|
||||
## Interpretation
|
||||
|
||||
The properties above are copied from the authoritative IANA special-purpose registry. Footnote markers are retained because exceptions and qualifications matter. Routing or operational claims require separate current evidence.
|
||||
|
||||
## Source
|
||||
|
||||
- [IANA IPv4 Special-Purpose Address Registry]({SPECIAL_SOURCE}), checked {checked}
|
||||
"""
|
||||
|
||||
|
||||
def write_csv(path: Path, fieldnames: list[str], rows: list[dict[str, str]]) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with path.open("w", newline="", encoding="utf-8") as handle:
|
||||
writer = csv.DictWriter(handle, fieldnames=fieldnames, lineterminator="\n")
|
||||
writer.writeheader()
|
||||
writer.writerows(rows)
|
||||
|
||||
|
||||
def render_index(top_rows: list[dict[str, object]], special_rows: list[dict[str, object]], checked: str) -> str:
|
||||
registry_counts = Counter(str(row["registry"]) for row in top_rows)
|
||||
status_counts = Counter(str(row["Status [1]"]).lower() for row in top_rows)
|
||||
lines = [
|
||||
GENERATED_MARKER,
|
||||
"# Complete IANA IPv4 range index",
|
||||
"",
|
||||
f"Generated from authoritative IANA registries and checked **{checked}**.",
|
||||
"",
|
||||
"## Coverage",
|
||||
"",
|
||||
f"- **{len(top_rows)} top-level `/8` records**, covering all IPv4 addresses from `0.0.0.0` through `255.255.255.255` without gaps.",
|
||||
f"- **{len(special_rows)} expanded special-purpose prefixes** from the IANA IPv4 Special-Purpose Address Registry.",
|
||||
"- More-specific manual research records and placeholders remain alongside these authoritative baselines.",
|
||||
"",
|
||||
"## Top-level counts",
|
||||
"",
|
||||
"### By RDAP registry authority",
|
||||
"",
|
||||
]
|
||||
lines.extend(f"- `{name}`: {count}" for name, count in sorted(registry_counts.items()))
|
||||
lines.extend(["", "### By IANA status", ""])
|
||||
lines.extend(f"- `{name}`: {count}" for name, count in sorted(status_counts.items()))
|
||||
lines.extend([
|
||||
"", "## All top-level IPv4 /8 ranges", "",
|
||||
"| Range | IANA designation | Status | Registry/RDAP | Record |",
|
||||
"|---|---|---|---|---|",
|
||||
])
|
||||
for row in top_rows:
|
||||
network = row["network"]
|
||||
filename = range_filename(network, str(row["registry"]))
|
||||
designation = str(row["Designation"]).replace("|", "\\|")
|
||||
lines.append(f"| `{network}` | {designation} | {row['Status [1]']} | `{row['registry']}` | [record]({filename}) |")
|
||||
lines.extend([
|
||||
"", "## IANA special-purpose ranges", "",
|
||||
"Special-purpose prefixes can overlap top-level `/8` records and each other because they express more-specific protocol semantics.",
|
||||
"", "| Range | Name | Globally reachable | Record |",
|
||||
"|---|---|---|---|",
|
||||
])
|
||||
for row in special_rows:
|
||||
network = row["network"]
|
||||
name = clean_name(str(row["Name"])).replace("|", "\\|")
|
||||
lines.append(f"| `{network}` | {name} | {row['Globally Reachable'] or 'Not stated'} | [record]({special_filename(network)}) |")
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def sync(top_text: str, special_text: str, checked: str) -> tuple[int, int]:
|
||||
top_rows = parse_top_level_csv(top_text)
|
||||
special_rows = parse_special_csv(special_text)
|
||||
expected_generated: set[Path] = set()
|
||||
|
||||
top_csv_rows: list[dict[str, str]] = []
|
||||
for row in top_rows:
|
||||
network = row["network"]
|
||||
filename = range_filename(network, str(row["registry"]))
|
||||
path = RANGE_DIR / filename
|
||||
write_generated(path, render_top_level(row, checked))
|
||||
if path.exists() and GENERATED_MARKER in path.read_text(encoding="utf-8"):
|
||||
expected_generated.add(path)
|
||||
top_csv_rows.append({
|
||||
"prefix": str(network), "designation": str(row["Designation"]),
|
||||
"registry": str(row["registry"]), "status": str(row["Status [1]"]).lower(),
|
||||
"allocation_date": str(row["Date"]), "whois": str(row["WHOIS"]),
|
||||
"rdap": clean_rdap(str(row["RDAP"])), "note": str(row["Note"]),
|
||||
"source_url": TOP_LEVEL_SOURCE, "verified_at": checked, "filename": filename,
|
||||
})
|
||||
|
||||
special_csv_rows: list[dict[str, str]] = []
|
||||
for row in special_rows:
|
||||
network = row["network"]
|
||||
filename = special_filename(network)
|
||||
path = RANGE_DIR / filename
|
||||
write_generated(path, render_special(row, checked))
|
||||
if path.exists() and GENERATED_MARKER in path.read_text(encoding="utf-8"):
|
||||
expected_generated.add(path)
|
||||
special_csv_rows.append({
|
||||
"prefix": str(network), "name": clean_name(str(row["Name"])), "rfc": clean_name(str(row["RFC"])),
|
||||
"allocation_date": str(row["Allocation Date"]), "termination_date": str(row["Termination Date"]),
|
||||
"source": str(row["Source"]), "destination": str(row["Destination"]),
|
||||
"forwardable": str(row["Forwardable"]), "globally_reachable": str(row["Globally Reachable"]),
|
||||
"reserved_by_protocol": str(row["Reserved-by-Protocol"]), "source_url": SPECIAL_SOURCE,
|
||||
"verified_at": checked, "filename": filename,
|
||||
})
|
||||
|
||||
for path in RANGE_DIR.glob("*.md"):
|
||||
if path not in expected_generated and path.name != "IPV4-INDEX.md" and GENERATED_MARKER in path.read_text(encoding="utf-8"):
|
||||
path.unlink()
|
||||
|
||||
write_csv(
|
||||
DATA_DIR / "iana-ipv4-top-level-ranges.csv",
|
||||
["prefix", "designation", "registry", "status", "allocation_date", "whois", "rdap", "note", "source_url", "verified_at", "filename"],
|
||||
top_csv_rows,
|
||||
)
|
||||
write_csv(
|
||||
DATA_DIR / "iana-ipv4-special-ranges.csv",
|
||||
["prefix", "name", "rfc", "allocation_date", "termination_date", "source", "destination", "forwardable", "globally_reachable", "reserved_by_protocol", "source_url", "verified_at", "filename"],
|
||||
special_csv_rows,
|
||||
)
|
||||
(RANGE_DIR / "IPV4-INDEX.md").write_text(render_index(top_rows, special_rows, checked), encoding="utf-8")
|
||||
return len(top_rows), len(special_rows)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--top-level-file", type=Path)
|
||||
parser.add_argument("--special-file", type=Path)
|
||||
parser.add_argument("--checked", default=datetime.now(timezone.utc).date().isoformat())
|
||||
args = parser.parse_args()
|
||||
top_text = args.top_level_file.read_text(encoding="utf-8-sig") if args.top_level_file else fetch_text(TOP_LEVEL_URL)
|
||||
special_text = args.special_file.read_text(encoding="utf-8-sig") if args.special_file else fetch_text(SPECIAL_URL)
|
||||
top_count, special_count = sync(top_text, special_text, args.checked)
|
||||
print(f"Synchronized {top_count} top-level IPv4 ranges and {special_count} special-purpose prefixes.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
+5
-1
@@ -123,6 +123,10 @@ def main() -> int:
|
||||
ROOT / "docs" / "04-ip-address-model.md",
|
||||
ROOT / "registry-ips" / "README.md",
|
||||
ROOT / "registry-ips" / "TEMPLATE.md",
|
||||
ROOT / "registry-ips" / "IPV4-INDEX.md",
|
||||
ROOT / "data" / "iana-ipv4-top-level-ranges.csv",
|
||||
ROOT / "data" / "iana-ipv4-special-ranges.csv",
|
||||
ROOT / "scripts" / "sync_iana_ipv4.py",
|
||||
ROOT / "registry-ips" / "000-000-000-000--08--iana.md",
|
||||
ROOT / "registry-ips" / "001-000-000-000--16--placeholder.md",
|
||||
ROOT / "registry-ips" / "001-001-000-000--24--placeholder.md",
|
||||
@@ -135,7 +139,7 @@ def main() -> int:
|
||||
|
||||
range_records = sorted((ROOT / "registry-ips").glob("*.md"))
|
||||
for path in range_records:
|
||||
if path.name in {"README.md", "TEMPLATE.md"}:
|
||||
if path.name in {"README.md", "TEMPLATE.md", "IPV4-INDEX.md"}:
|
||||
continue
|
||||
headings = [line.strip() for line in path.read_text(encoding="utf-8").splitlines() if line.startswith("## ")]
|
||||
if not headings or headings[0] != "## Who is this IP range?":
|
||||
|
||||
Reference in New Issue
Block a user