diff --git a/Makefile b/Makefile index e6a55df..af82359 100644 --- a/Makefile +++ b/Makefile @@ -5,3 +5,4 @@ validate: summary: @python3 -c "import csv,collections; r=list(csv.DictReader(open('data/entities.csv', encoding='utf-8'))); print(f'{len(r)} entities'); [print(f'{n:3} {k}') for k,n in sorted(collections.Counter(x['category'] for x in r).items())]" + @python3 -c "import csv; r=list(csv.DictReader(open('data/ip-addresses.csv', encoding='utf-8'))); print(f'{len(r)} IP address records: ' + ', '.join(x['address'] for x in r))" diff --git a/README.md b/README.md index 164818c..bb1ffb3 100644 --- a/README.md +++ b/README.md @@ -8,8 +8,9 @@ A source-backed address book and evolving map of the organizations, coordination 1. Read the [first-round investigation](docs/02-first-round-investigation.md). 2. Browse the human-readable [address book](address-book/README.md). -3. Use [`data/entities.csv`](data/entities.csv) for machine-readable records. -4. Review the [next research round](research/next-round.md). +3. Start the IP-level map with [0.0.0.0 and 1.1.1.1](address-book/ip-addresses.md). +4. Use [`data/entities.csv`](data/entities.csv) and [`data/ip-addresses.csv`](data/ip-addresses.csv) for machine-readable records. +5. Review the [next research round](research/next-round.md) and [IP network roadmap](research/ip-network-roadmap.md). ## What “global network” means here @@ -31,7 +32,7 @@ It deliberately maps institutions and trusted public directories before attempti . ├── README.md # orientation and status ├── address-book/ # concise human-readable directories -├── data/entities.csv # canonical round-1 structured dataset +├── data/ # canonical entity and IP address datasets ├── docs/ # scope, method, findings, taxonomy ├── research/ # unanswered questions and next rounds ├── scripts/validate.py # dependency-free data/link-shape checks diff --git a/address-book/README.md b/address-book/README.md index c3673fd..bdc3794 100644 --- a/address-book/README.md +++ b/address-book/README.md @@ -8,6 +8,7 @@ This is the readable view of the canonical records in [`data/entities.csv`](../d | [Regional Internet Registries](regional-registries.md) | The five RIR service regions and public resource directories | | [Infrastructure and operations](infrastructure-and-operations.md) | DNS root, IXPs/peering, routing visibility, physical transport, operator communities | | [Security and resilience](security-and-resilience.md) | Incident-response directories and routing-security coordination | +| [IP address map](ip-addresses.md) | Registration, route origin, operator, and service for individual addresses | ## How to use it @@ -20,3 +21,4 @@ Start with the coordinating body or directory for the function you need. For exa - incident-response team discovery → FIRST or Trusted Introducer; - BGP visibility → Route Views or RIPE RIS; - submarine cable discovery → TeleGeography map, followed by operator/regulator verification. +- exact IP investigation → special-purpose registry, RDAP, BGP origin, and operator evidence kept as separate claims. diff --git a/address-book/ip-addresses.md b/address-book/ip-addresses.md new file mode 100644 index 0000000..38b7c20 --- /dev/null +++ b/address-book/ip-addresses.md @@ -0,0 +1,24 @@ +# IP address map + +The question “who owns this IP?” usually has several different answers. This directory separates registration, routing, operation, and service instead of forcing them into one unreliable `owner` field. + +## Initial records + +| Address | Registry classification | Registered resource holder | Current route origin | Operator/service | Globally routable | +|---|---|---|---|---|---| +| [`0.0.0.0`](ip/0.0.0.0.md) | IANA special-purpose | Not organizationally assigned | None | Unspecified address / “this host on this network” | No | +| [`1.1.1.1`](ip/1.1.1.1.md) | APNIC public resource (`1.1.1.0/24`) | APNIC Research and Development | AS13335, Cloudflare | Cloudflare 1.1.1.1 recursive DNS service; APNIC–Cloudflare project | Yes | + +## Ownership model + +For every address, investigate these independently: + +1. **Classification** — public unicast, private, loopback, documentation, multicast, reserved, or another special purpose. +2. **Registry chain** — IANA → RIR → allocated/assigned resource holder. +3. **Route origin** — currently announced covering prefix and origin ASN. +4. **Network operator** — entity operating the route or service. +5. **Service** — what the exact address does, if publicly documented. +6. **Infrastructure relationships** — IXPs, facilities, upstreams, cables, and geography where evidence permits. +7. **Time and provenance** — registry and routing facts can change, so every claim needs a source and verification date. + +The canonical structured records are in [`data/ip-addresses.csv`](../data/ip-addresses.csv). diff --git a/address-book/ip/0.0.0.0.md b/address-book/ip/0.0.0.0.md new file mode 100644 index 0000000..b6778f3 --- /dev/null +++ b/address-book/ip/0.0.0.0.md @@ -0,0 +1,26 @@ +# 0.0.0.0 + +## Answer + +`0.0.0.0` is **not an ordinary public address owned by a company**. IANA records both the exact `0.0.0.0/32` address and its `0.0.0.0/8` parent as special-purpose protocol space. + +| Axis | Finding | +|---|---| +| Exact registry entry | `0.0.0.0/32` | +| IANA name | “This host on this network” | +| Parent block | `0.0.0.0/8` — “This network” | +| Organizational holder | None; reserved by the IP protocol | +| Destination allowed | No | +| Forwardable | No | +| Globally reachable | No | +| Normal route origin | None expected | + +## Meaning + +The exact all-zero address is used as an unspecified source address in limited host-initialization contexts. It must not be treated as a normal globally reachable destination or as an address assigned to an ISP, cloud, or person. + +## Sources + +- [IANA IPv4 Special-Purpose Address Space](https://www.iana.org/assignments/iana-ipv4-special-registry/iana-ipv4-special-registry.xhtml), checked 2026-07-20 +- [RFC 1122, section 3.2.1.3](https://www.rfc-editor.org/rfc/rfc1122.html#section-3.2.1.3) +- [RFC 791](https://www.rfc-editor.org/rfc/rfc791.html) diff --git a/address-book/ip/1.1.1.1.md b/address-book/ip/1.1.1.1.md new file mode 100644 index 0000000..3ccf705 --- /dev/null +++ b/address-book/ip/1.1.1.1.md @@ -0,0 +1,34 @@ +# 1.1.1.1 + +## Answer + +`1.1.1.1` demonstrates why “owner” must be split into multiple relationships. + +| Axis | Finding | +|---|---| +| Exact address | `1.1.1.1` | +| Registered covering prefix | `1.1.1.0/24` | +| Registry | APNIC | +| RDAP network name/type | `APNIC-LABS` / `ASSIGNED PORTABLE` | +| RDAP registrant | APNIC Research and Development | +| Current announced prefix | `1.1.1.0/24` | +| Current route origin | `AS13335` | +| Origin holder | Cloudflare, Inc. (`CLOUDFLARENET`) | +| Operator/service | Cloudflare 1.1.1.1 public recursive DNS resolver | +| Project relationship | APNIC–Cloudflare DNS Resolver project | +| Globally routed | Yes | + +## Interpretation + +- **Registration:** APNIC RDAP identifies the resource as `APNIC-LABS` and the registrant as APNIC Research and Development. +- **Routing:** RIPEstat reports `1.1.1.0/24` announced by `AS13335`, Cloudflare. +- **Operation/service:** Cloudflare publicly documents `1.1.1.1` as its consumer recursive DNS service. +- **Relationship:** APNIC RDAP explicitly describes an APNIC and Cloudflare DNS Resolver project and says the research prefix is routed globally by AS13335/Cloudflare. + +Therefore, saying only “Cloudflare owns 1.1.1.1” loses important registry context. A more accurate statement is: **the covering prefix is registered through APNIC to APNIC Research and Development, while Cloudflare originates the route and operates the public resolver service as part of the documented collaboration.** + +## Sources + +- [APNIC RDAP record](https://rdap.apnic.net/ip/1.1.1.1), checked 2026-07-20 +- [RIPEstat prefix overview](https://stat.ripe.net/data/prefix-overview/data.json?resource=1.1.1.1), checked 2026-07-20 +- [Cloudflare launch announcement](https://blog.cloudflare.com/announcing-1111/) diff --git a/data/README.md b/data/README.md index 1f26574..e13c613 100644 --- a/data/README.md +++ b/data/README.md @@ -1,6 +1,6 @@ # Data directory -`entities.csv` is the canonical round-1 structured address book. +`entities.csv` is the canonical round-1 ecosystem address book. `ip-addresses.csv` begins the address-level map and deliberately separates registration, routing, operation, and service. ## Columns @@ -20,3 +20,5 @@ | `notes` | Important caveat; blank when none | CSV was chosen for round 1 because it is inspectable in Git, opens in spreadsheets, and requires no custom parser. If nested relationships become central, introduce versioned YAML or JSON without deleting CSV history. + +The IP-specific columns are documented in [`docs/04-ip-address-model.md`](../docs/04-ip-address-model.md). diff --git a/data/ip-addresses.csv b/data/ip-addresses.csv new file mode 100644 index 0000000..f9e9c7f --- /dev/null +++ b/data/ip-addresses.csv @@ -0,0 +1,3 @@ +address,exact_prefix,parent_prefix,classification,registry,registered_to,announced_prefix,origin_asn,origin_holder,operator,service,globally_routable,source_registry,source_routing,source_service,verified_at,notes +0.0.0.0,0.0.0.0/32,0.0.0.0/8,special-purpose,IANA,Not organizationally assigned,,,,,Unspecified address / this host on this network,false,https://www.iana.org/assignments/iana-ipv4-special-registry/iana-ipv4-special-registry.xhtml,,,2026-07-20,"Reserved by the IP protocol. IANA marks destination, forwardable, and globally reachable as false. The parent 0.0.0.0/8 is named This network." +1.1.1.1,1.1.1.1/32,1.1.1.0/24,public-unicast,APNIC,APNIC Research and Development,1.1.1.0/24,AS13335,"CLOUDFLARENET - Cloudflare, Inc.",Cloudflare,1.1.1.1 public recursive DNS resolver,true,https://rdap.apnic.net/ip/1.1.1.1,https://stat.ripe.net/data/prefix-overview/data.json?resource=1.1.1.1,https://blog.cloudflare.com/announcing-1111/,2026-07-20,"APNIC RDAP describes an APNIC and Cloudflare DNS Resolver project, says the research prefix is routed globally by AS13335/Cloudflare, and identifies APNIC Research and Development as registrant." diff --git a/docs/04-ip-address-model.md b/docs/04-ip-address-model.md new file mode 100644 index 0000000..c0e0eb6 --- /dev/null +++ b/docs/04-ip-address-model.md @@ -0,0 +1,39 @@ +# IP address record model + +## Purpose + +An IP address does not have one universally meaningful “owner.” Registration, legal control, route origination, network operation, and service branding may belong to different entities. The data model preserves those distinctions. + +## Round-1 fields + +| Field | Meaning | +|---|---| +| `address` | Exact queried IPv4 or IPv6 address | +| `exact_prefix` | Host-length prefix for the exact address | +| `parent_prefix` | Relevant special-purpose or registration block | +| `classification` | Public-unicast or special-purpose category | +| `registry` | IANA or responsible RIR | +| `registered_to` | Public resource registrant/holder from authoritative registry data | +| `announced_prefix` | Currently observed covering BGP prefix, if any | +| `origin_asn` | Current observed origin ASN, if any | +| `origin_holder` | Published holder/name associated with the origin ASN | +| `operator` | Publicly documented service/network operator | +| `service` | Function of the exact address, if established | +| `globally_routable` | Whether evidence indicates ordinary global routing | +| `source_*` | Registry, routing, and service provenance | +| `verified_at` | UTC date on which dynamic claims were checked | +| `notes` | Caveats and relationships that do not fit a scalar field | + +## Resolution workflow + +1. Check the IANA IPv4/IPv6 special-purpose registries first. +2. If ordinary public space, follow the IANA/RIR RDAP bootstrap to the authoritative RDAP service. +3. Record the allocation/assignment covering prefix and public entity roles. +4. Query at least one current BGP visibility source for covering prefix and origin ASN. +5. Identify the operator/service only from an operator or similarly authoritative source. +6. Preserve disagreements and multiple origins; never choose one silently. +7. Store the verification date because routing and registration are time-dependent. + +## Future relationships + +Later versions should normalize addresses, prefixes, ASNs, organizations, services, facilities, IXPs, geographies, RPKI objects, and source observations into separate records. CSV remains appropriate while the first set is small and auditable. diff --git a/research/ip-network-roadmap.md b/research/ip-network-roadmap.md new file mode 100644 index 0000000..1f15189 --- /dev/null +++ b/research/ip-network-roadmap.md @@ -0,0 +1,30 @@ +# IP network research roadmap + +## Next addresses and ranges + +Expand by **semantically important ranges**, not by blindly incrementing every IPv4 address: + +1. Special-purpose baseline: `0.0.0.0/8`, RFC 1918 private ranges, shared address space, loopback, link-local, documentation, benchmarking, multicast, and reserved space. +2. Public resolver landmarks: `1.1.1.1`, `8.8.8.8`, `9.9.9.9`, and their IPv6 counterparts. +3. DNS root-server identities and their anycast origin/operator relationships. +4. Representative addresses from each RIR service region. +5. Major cloud, CDN, transit, access, and public-sector networks selected by transparent criteria. + +## Automation backlog + +- [ ] Download IANA IPv4 and IPv6 special-purpose registries. +- [ ] Implement IANA RDAP bootstrap discovery instead of hard-coding one RIR. +- [ ] Cache source observations with timestamps and respectful rate limits. +- [ ] Add BGP multi-origin and more-specific-prefix handling. +- [ ] Add RPKI validity and ROA evidence. +- [ ] Normalize ASN and organization records instead of repeating names. +- [ ] Generate human-readable address pages from reviewed data. +- [ ] Add tests for IPv4/IPv6 parsing and special-purpose precedence. + +## Guardrails + +- Do not scan hosts or ports merely to populate the directory. +- Do not equate WHOIS/RDAP registration with legal ownership or current operation. +- Do not publish private contact details. +- Do not infer physical location from registration country alone. +- Do not claim one BGP collector provides complete global truth. diff --git a/scripts/validate.py b/scripts/validate.py index 1adf7fe..5c6cf05 100644 --- a/scripts/validate.py +++ b/scripts/validate.py @@ -13,6 +13,7 @@ from urllib.parse import urlparse ROOT = Path(__file__).resolve().parents[1] DATA = ROOT / "data" / "entities.csv" +IP_DATA = ROOT / "data" / "ip-addresses.csv" REQUIRED = [ "id", "name", "category", "scope", "region", "homepage", "directory_url", "role", "source_url", "status", "verified_at", "notes", @@ -24,6 +25,12 @@ CATEGORIES = { SCOPES = {"global", "regional", "regional-community"} STATUSES = {"verified", "partial", "lead", "stale"} ID_RE = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$") +IP_REQUIRED = [ + "address", "exact_prefix", "parent_prefix", "classification", "registry", + "registered_to", "announced_prefix", "origin_asn", "origin_holder", "operator", + "service", "globally_routable", "source_registry", "source_routing", + "source_service", "verified_at", "notes", +] def valid_url(value: str) -> bool: @@ -68,10 +75,51 @@ def main() -> int: if count > 1: errors.append(f"duplicate id {ident!r}: {count} rows") + with IP_DATA.open(newline="", encoding="utf-8") as handle: + ip_reader = csv.DictReader(handle) + if ip_reader.fieldnames != IP_REQUIRED: + errors.append(f"IP CSV columns differ from required schema: {ip_reader.fieldnames!r}") + ip_rows = list(ip_reader) + + ip_addresses = Counter(row.get("address", "") for row in ip_rows) + for row_number, row in enumerate(ip_rows, start=2): + address = row.get("address", "") + prefix = f"IP row {row_number} ({address or 'missing-address'})" + for field in ("address", "exact_prefix", "parent_prefix", "classification", "registry", + "registered_to", "service", "globally_routable", "source_registry", "verified_at"): + if not row.get(field, "").strip(): + errors.append(f"{prefix}: missing {field}") + try: + import ipaddress + parsed = ipaddress.ip_address(address) + exact = ipaddress.ip_network(row.get("exact_prefix", ""), strict=False) + parent = ipaddress.ip_network(row.get("parent_prefix", ""), strict=False) + if parsed not in exact or parsed not in parent: + errors.append(f"{prefix}: address must be contained by exact and parent prefixes") + except ValueError as exc: + errors.append(f"{prefix}: invalid address/prefix: {exc}") + if row.get("globally_routable") not in {"true", "false"}: + errors.append(f"{prefix}: globally_routable must be true or false") + for field in ("source_registry", "source_routing", "source_service"): + value = row.get(field, "").strip() + if value and not valid_url(value): + errors.append(f"{prefix}: {field} must be an https URL when present") + try: + checked = date.fromisoformat(row.get("verified_at", "")) + if checked > date.today(): + errors.append(f"{prefix}: verified_at is in the future") + except ValueError: + errors.append(f"{prefix}: verified_at must be YYYY-MM-DD") + for address, count in ip_addresses.items(): + if count > 1: + errors.append(f"duplicate IP address {address!r}: {count} rows") + required_docs = [ ROOT / "README.md", ROOT / "docs" / "02-first-round-investigation.md", ROOT / "address-book" / "README.md", + ROOT / "address-book" / "ip-addresses.md", + ROOT / "docs" / "04-ip-address-model.md", ROOT / "research" / "next-round.md", ] for path in required_docs: @@ -85,7 +133,7 @@ def main() -> int: return 1 counts = Counter(row["category"] for row in rows) - print(f"Validated {len(rows)} entities across {len(counts)} categories.") + print(f"Validated {len(rows)} entities across {len(counts)} categories and {len(ip_rows)} IP address records.") for category, count in sorted(counts.items()): print(f"- {category}: {count}") return 0