research: establish global network address book baseline

This commit is contained in:
2026-07-20 18:18:54 +00:00
parent 94489db524
commit 4752d490d4
17 changed files with 538 additions and 1 deletions
+95
View File
@@ -0,0 +1,95 @@
#!/usr/bin/env python3
"""Validate the round-1 network address book without third-party dependencies."""
from __future__ import annotations
import csv
import re
import sys
from collections import Counter
from datetime import date
from pathlib import Path
from urllib.parse import urlparse
ROOT = Path(__file__).resolve().parents[1]
DATA = ROOT / "data" / "entities.csv"
REQUIRED = [
"id", "name", "category", "scope", "region", "homepage",
"directory_url", "role", "source_url", "status", "verified_at", "notes",
]
CATEGORIES = {
"governance-identifiers", "standards", "registry", "dns", "interconnection",
"routing-security", "incident-response", "physical-infrastructure", "operator-community",
}
SCOPES = {"global", "regional", "regional-community"}
STATUSES = {"verified", "partial", "lead", "stale"}
ID_RE = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$")
def valid_url(value: str) -> bool:
parsed = urlparse(value)
return parsed.scheme == "https" and bool(parsed.netloc)
def main() -> int:
errors: list[str] = []
with DATA.open(newline="", encoding="utf-8") as handle:
reader = csv.DictReader(handle)
if reader.fieldnames != REQUIRED:
errors.append(f"CSV columns differ from required schema: {reader.fieldnames!r}")
rows = list(reader)
ids = Counter(row.get("id", "") for row in rows)
for row_number, row in enumerate(rows, start=2):
ident = row.get("id", "")
prefix = f"row {row_number} ({ident or 'missing-id'})"
for field in REQUIRED[:-1]: # notes may be blank
if not row.get(field, "").strip():
errors.append(f"{prefix}: missing {field}")
if not ID_RE.fullmatch(ident):
errors.append(f"{prefix}: invalid id")
if row.get("category") not in CATEGORIES:
errors.append(f"{prefix}: unknown category {row.get('category')!r}")
if row.get("scope") not in SCOPES:
errors.append(f"{prefix}: unknown scope {row.get('scope')!r}")
if row.get("status") not in STATUSES:
errors.append(f"{prefix}: unknown status {row.get('status')!r}")
for field in ("homepage", "directory_url", "source_url"):
if not valid_url(row.get(field, "")):
errors.append(f"{prefix}: {field} must be an https URL")
try:
checked = date.fromisoformat(row.get("verified_at", ""))
if checked > date.today():
errors.append(f"{prefix}: verified_at is in the future")
except ValueError:
errors.append(f"{prefix}: verified_at must be YYYY-MM-DD")
for ident, count in ids.items():
if count > 1:
errors.append(f"duplicate id {ident!r}: {count} rows")
required_docs = [
ROOT / "README.md",
ROOT / "docs" / "02-first-round-investigation.md",
ROOT / "address-book" / "README.md",
ROOT / "research" / "next-round.md",
]
for path in required_docs:
if not path.is_file():
errors.append(f"missing required document: {path.relative_to(ROOT)}")
if errors:
print("Validation failed:")
for error in errors:
print(f"- {error}")
return 1
counts = Counter(row["category"] for row in rows)
print(f"Validated {len(rows)} entities across {len(counts)} categories.")
for category, count in sorted(counts.items()):
print(f"- {category}: {count}")
return 0
if __name__ == "__main__":
sys.exit(main())