96 lines
3.4 KiB
Python
96 lines
3.4 KiB
Python
#!/usr/bin/env python3
|
|
"""Validate the round-1 network address book without third-party dependencies."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import csv
|
|
import re
|
|
import sys
|
|
from collections import Counter
|
|
from datetime import date
|
|
from pathlib import Path
|
|
from urllib.parse import urlparse
|
|
|
|
ROOT = Path(__file__).resolve().parents[1]
|
|
DATA = ROOT / "data" / "entities.csv"
|
|
REQUIRED = [
|
|
"id", "name", "category", "scope", "region", "homepage",
|
|
"directory_url", "role", "source_url", "status", "verified_at", "notes",
|
|
]
|
|
CATEGORIES = {
|
|
"governance-identifiers", "standards", "registry", "dns", "interconnection",
|
|
"routing-security", "incident-response", "physical-infrastructure", "operator-community",
|
|
}
|
|
SCOPES = {"global", "regional", "regional-community"}
|
|
STATUSES = {"verified", "partial", "lead", "stale"}
|
|
ID_RE = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$")
|
|
|
|
|
|
def valid_url(value: str) -> bool:
|
|
parsed = urlparse(value)
|
|
return parsed.scheme == "https" and bool(parsed.netloc)
|
|
|
|
|
|
def main() -> int:
|
|
errors: list[str] = []
|
|
with DATA.open(newline="", encoding="utf-8") as handle:
|
|
reader = csv.DictReader(handle)
|
|
if reader.fieldnames != REQUIRED:
|
|
errors.append(f"CSV columns differ from required schema: {reader.fieldnames!r}")
|
|
rows = list(reader)
|
|
|
|
ids = Counter(row.get("id", "") for row in rows)
|
|
for row_number, row in enumerate(rows, start=2):
|
|
ident = row.get("id", "")
|
|
prefix = f"row {row_number} ({ident or 'missing-id'})"
|
|
for field in REQUIRED[:-1]: # notes may be blank
|
|
if not row.get(field, "").strip():
|
|
errors.append(f"{prefix}: missing {field}")
|
|
if not ID_RE.fullmatch(ident):
|
|
errors.append(f"{prefix}: invalid id")
|
|
if row.get("category") not in CATEGORIES:
|
|
errors.append(f"{prefix}: unknown category {row.get('category')!r}")
|
|
if row.get("scope") not in SCOPES:
|
|
errors.append(f"{prefix}: unknown scope {row.get('scope')!r}")
|
|
if row.get("status") not in STATUSES:
|
|
errors.append(f"{prefix}: unknown status {row.get('status')!r}")
|
|
for field in ("homepage", "directory_url", "source_url"):
|
|
if not valid_url(row.get(field, "")):
|
|
errors.append(f"{prefix}: {field} must be an https URL")
|
|
try:
|
|
checked = date.fromisoformat(row.get("verified_at", ""))
|
|
if checked > date.today():
|
|
errors.append(f"{prefix}: verified_at is in the future")
|
|
except ValueError:
|
|
errors.append(f"{prefix}: verified_at must be YYYY-MM-DD")
|
|
|
|
for ident, count in ids.items():
|
|
if count > 1:
|
|
errors.append(f"duplicate id {ident!r}: {count} rows")
|
|
|
|
required_docs = [
|
|
ROOT / "README.md",
|
|
ROOT / "docs" / "02-first-round-investigation.md",
|
|
ROOT / "address-book" / "README.md",
|
|
ROOT / "research" / "next-round.md",
|
|
]
|
|
for path in required_docs:
|
|
if not path.is_file():
|
|
errors.append(f"missing required document: {path.relative_to(ROOT)}")
|
|
|
|
if errors:
|
|
print("Validation failed:")
|
|
for error in errors:
|
|
print(f"- {error}")
|
|
return 1
|
|
|
|
counts = Counter(row["category"] for row in rows)
|
|
print(f"Validated {len(rows)} entities across {len(counts)} categories.")
|
|
for category, count in sorted(counts.items()):
|
|
print(f"- {category}: {count}")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|