Drop a filter that matched nothing, keep the 79 ports it hid, and exit on a value the registers should not hold

This commit is contained in:
2026-09-20 00:16:59 +02:00
parent ec6f0e2edb
commit baed259802
7 changed files with 5957 additions and 5849 deletions
+3 -3
View File
@@ -45,9 +45,9 @@ replacement, and each removed path, column or flag.
`model` column. `model` column.
- `misc.httpmethod`, `misc.protocol` and `misc.port` are IANA's registries as tables. - `misc.httpmethod`, `misc.protocol` and `misc.port` are IANA's registries as tables.
`misc.httpmethod` carries `safe` and `idempotent` beside the method; `misc.protocol` `misc.httpmethod` carries `safe` and `idempotent` beside the method; `misc.protocol`
the keyword, its `number` and its `name`; `misc.port` renders a TCP port number, the keyword, its `number` and its `name`, selectable by either; `misc.port` renders a
with the service `name` beside it, keeping the first service the registry lists for TCP port number, with the IANA `service` name beside it, keeping the first service the
a port so a number selects one row. registry describes for a port so a number selects one row.
- `geo.SE` and `geo.US`: five linked tables per country, `region`, `municipality`, - `geo.SE` and `geo.US`: five linked tables per country, `region`, `municipality`,
`locality`, `postal-code` and `street`, weighted by population and address counts `locality`, `postal-code` and `street`, weighted by population and address counts
and built from SCB, GeoNames, Trafikverket NVDB and the US Census Bureau, and an and built from SCB, GeoNames, Trafikverket NVDB and the US Census Bureau, and an
+10 -10
View File
@@ -164,16 +164,16 @@ Each locale carries `address`, `color`, `company`, `date`, `email`, `first-name`
carries `car`, `coordinate`, `creditcard` (Luhn-valid), `currency` (ISO 4217), carries `car`, `coordinate`, `creditcard` (Luhn-valid), `currency` (ISO 4217),
`datetime` (RFC 3339), `emoji`, `httpmethod`, `httpstatus`, `language` (ISO 639-1, `datetime` (RFC 3339), `emoji`, `httpmethod`, `httpstatus`, `language` (ISO 639-1,
with its 639-2/T code), `mac`, `mimetype`, `objectid`, `port`, `protocol`, with its 639-2/T code), `mac`, `mimetype`, `objectid`, `port`, `protocol`,
`territory` (ISO 3166-1), `timezone` (IANA), `useragent` and `uuid` (v4). Many carry sub-fields — `misc.currency.symbol`, `territory` (ISO 3166-1), `timezone` (IANA), `useragent` and `uuid` (v4). Many carry
`misc.territory.alpha2`, `misc.httpstatus.code` — which `--list` shows. `car`, sub-fields — `misc.currency.symbol`, `misc.territory.alpha2`, `misc.httpstatus.code` —
`currency`, `httpstatus`, `language`, `mimetype`, `territory`, `timezone` and which `--list` shows. `car`, `currency`, `httpmethod`, `httpstatus`, `language`,
`httpmethod`, `port`, `protocol` and `useragent` are [tables](#table), so `mimetype`, `port`, `protocol`, `territory`, `timezone` and `useragent` are
`misc.territory[SE].capital` and [tables](#table), so `misc.territory[SE].capital` and `misc.currency[Euro].symbol`
`misc.currency[Euro].symbol` select a row; `car` and `useragent` carry no key or select a row; `car` and `useragent` carry no key or name, so they are drawn from rather
name, so they are drawn from rather than selected in. `misc.httpmethod`, than selected in. `misc.httpmethod`, `misc.protocol` and `misc.port` are IANA's
`misc.protocol` and `misc.port` are IANA's registries: a method carries whether it registries: a method carries the register's `yes` or `no` for `safe` and `idempotent`,
is `safe` and `idempotent`, a protocol its `number`, and `misc.port` renders the a protocol its `number` and its spelled-out `name`, and `misc.port` renders the number
number a port field holds, with the service `name` beside it. a port field holds, with the IANA `service` name beside it.
[`DATA-LICENSES.md`](DATA-LICENSES.md) names each table's source and licence. [`DATA-LICENSES.md`](DATA-LICENSES.md) names each table's source and licence.
`misc.timezone` is every zone tzdb gives a shipped territory, from one apiece for most `misc.timezone` is every zone tzdb gives a shipped territory, from one apiece for most
+12 -5
View File
@@ -3,13 +3,14 @@
data-import/httpmethod.py [--source URL_OR_FILE] [--cache DIR] [--out FILE] data-import/httpmethod.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
A method whose name is not letters and hyphens does not ship, which drops the `*` the A row needs a name of letters and hyphens, which drops the registry's `*`, and both of
registry holds to stop anyone registering it. the answers the registry gives for it.
""" """
import argparse import argparse
import csv import csv
import io import io
import re import re
import sys
from pathlib import Path from pathlib import Path
import source import source
@@ -19,13 +20,19 @@ SOURCE = "https://www.iana.org/assignments/http-methods/methods.csv"
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "httpmethod.tsv" OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "httpmethod.tsv"
CACHE = Path(__file__).resolve().parent / "cache" CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["idempotent", "method", "safe"] COLUMNS = ["idempotent", "method", "safe"]
ANSWERS = ("yes", "no")
def rows(text): def rows(text):
for r in csv.DictReader(io.StringIO(text)): for r in csv.DictReader(io.StringIO(text)):
method, safe, idempotent = r["Method Name"].strip(), r["Safe"].strip(), r["Idempotent"].strip() method = (r["Method Name"] or "").strip()
if re.match(r"^[A-Za-z][A-Za-z-]*$", method) and safe and idempotent: safe, idempotent = (r["Safe"] or "").strip(), (r["Idempotent"] or "").strip()
yield {"idempotent": idempotent, "method": method, "safe": safe} if not re.fullmatch(r"[A-Za-z][A-Za-z-]*", method) or not safe or not idempotent:
continue
for column, answer in (("Safe", safe), ("Idempotent", idempotent)):
if answer not in ANSWERS:
sys.exit(f"{method}: {column} {answer!r} is neither yes nor no")
yield {"idempotent": idempotent, "method": method, "safe": safe}
def main(): def main():
+20 -14
View File
@@ -3,13 +3,14 @@
data-import/port.py [--source URL_OR_FILE] [--cache DIR] [--out FILE] data-import/port.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
What ships is the TCP assignments in use: a row needs a service name, a port number and A row needs a service name and a numeric TCP port. A port the registry lists more than
a description the registry has filled in. A port the registry lists more than once keeps once keeps the first service it describes, or the first of them where it describes none,
the first service, so a number selects one row. so a number selects one row.
""" """
import argparse import argparse
import csv import csv
import io import io
import sys
from pathlib import Path from pathlib import Path
import source import source
@@ -18,21 +19,24 @@ import tsv
SOURCE = "https://www.iana.org/assignments/service-names-port-numbers/service-names-port-numbers.csv" SOURCE = "https://www.iana.org/assignments/service-names-port-numbers/service-names-port-numbers.csv"
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "port.tsv" OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "port.tsv"
CACHE = Path(__file__).resolve().parent / "cache" CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["name", "number"] COLUMNS = ["number", "service"]
UNUSED = ("Reserved", "Unassigned", "IANA assigned this well-formed service name to replace an unregistered or squatted port.") FLOOR = 4000
def rows(text): def rows(text):
seen = set() best, described = {}, set()
for r in csv.DictReader(io.StringIO(text)): for r in csv.DictReader(io.StringIO(text)):
name, number = r["Service Name"].strip(), r["Port Number"].strip() service, number = (r["Service Name"] or "").strip(), (r["Port Number"] or "").strip()
described = (r["Description"] or "").strip() transport, description = (r["Transport Protocol"] or "").strip(), (r["Description"] or "").strip()
if r["Transport Protocol"].strip() != "tcp" or not name or not number.isdigit(): if transport != "tcp" or not service or not number.isdigit():
continue continue
if not described or described in UNUSED or number in seen: if not 1 <= int(number) <= 65535:
continue sys.exit(f"{service}: port {number} is outside 1-65535")
seen.add(number) if number not in best or (description and number not in described):
yield {"name": name, "number": number} best[number] = {"number": number, "service": service}
if description:
described.add(number)
return sorted(best.values(), key=lambda r: int(r["number"]))
def main(): def main():
@@ -41,7 +45,9 @@ def main():
p.add_argument("--source", default=SOURCE) p.add_argument("--source", default=SOURCE)
p.add_argument("--out", default=str(OUT)) p.add_argument("--out", default=str(OUT))
a = p.parse_args() a = p.parse_args()
table = sorted(rows(source.fetch(a.source, a.cache, "service-names-port-numbers.csv").decode("utf-8")), key=lambda r: int(r["number"])) table = rows(source.fetch(a.source, a.cache, "service-names-port-numbers.csv").decode("utf-8"))
if len(table) < FLOOR:
sys.exit(f"only {len(table)} ports named a TCP service; the registry's columns have moved")
tsv.write(a.out, COLUMNS, table) tsv.write(a.out, COLUMNS, table)
+21 -5
View File
@@ -3,12 +3,14 @@
data-import/protocol.py [--source URL_OR_FILE] [--cache DIR] [--out FILE] data-import/protocol.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
A number the registry leaves unassigned, reserves or marks deprecated does not ship. A row needs a number and a keyword, which drops the seven numbers assigned to a class of
Where it names no protocol beside the keyword, the keyword is the name. protocols rather than to one; a keyword the registry reserves or marks deprecated goes
too. Where the registry names no protocol beside the keyword, the keyword is the name.
""" """
import argparse import argparse
import csv import csv
import io import io
import sys
from pathlib import Path from pathlib import Path
import source import source
@@ -18,15 +20,28 @@ SOURCE = "https://www.iana.org/assignments/protocol-numbers/protocol-numbers-1.c
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "protocol.tsv" OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "protocol.tsv"
CACHE = Path(__file__).resolve().parent / "cache" CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["keyword", "name", "number"] COLUMNS = ["keyword", "name", "number"]
SKIP = ("Unassigned", "Reserved", "deprecated") SKIP = ("Reserved", "deprecated")
def rows(text): def rows(text):
for r in csv.DictReader(io.StringIO(text)): for r in csv.DictReader(io.StringIO(text)):
number, keyword = r["Decimal"].strip(), r["Keyword"].strip() number, keyword = (r["Decimal"] or "").strip(), (r["Keyword"] or "").strip()
if not number.isdigit() or not keyword or any(s in keyword for s in SKIP): if not number.isdigit() or not keyword or any(s in keyword for s in SKIP):
continue continue
yield {"keyword": keyword, "name": " ".join(r["Protocol"].split()) or keyword, "number": number} if not 0 <= int(number) <= 255:
sys.exit(f"{keyword}: protocol number {number} is outside 0-255")
yield {"keyword": keyword, "name": " ".join((r["Protocol"] or "").split()) or keyword, "number": number}
def refuse_a_name_the_loader_would(table):
"""A name spelling another row's keyword, or a second row's name, is a load error for every consumer."""
keywords, seen = {r["keyword"] for r in table}, {}
for r in table:
if r["name"] != r["keyword"] and r["name"] in keywords:
sys.exit(f"{r['keyword']}: name {r['name']!r} is another protocol's keyword")
if r["name"] in seen:
sys.exit(f"{r['keyword']}: name {r['name']!r} repeats {seen[r['name']]}'s")
seen[r["name"]] = r["keyword"]
def main(): def main():
@@ -36,6 +51,7 @@ def main():
p.add_argument("--out", default=str(OUT)) p.add_argument("--out", default=str(OUT))
a = p.parse_args() a = p.parse_args()
table = sorted(rows(source.fetch(a.source, a.cache, "protocol-numbers-1.csv").decode("utf-8")), key=lambda r: int(r["number"])) table = sorted(rows(source.fetch(a.source, a.cache, "protocol-numbers-1.csv").decode("utf-8")), key=lambda r: int(r["number"]))
refuse_a_name_the_loader_would(table)
tsv.write(a.out, COLUMNS, table) tsv.write(a.out, COLUMNS, table)
+5890 -5811
View File
File diff suppressed because it is too large Load Diff
+1 -1
View File
@@ -1 +1 @@
{ "format": "{keyword}", "rows": "protocol.tsv", "key": "keyword" } { "format": "{keyword}", "rows": "protocol.tsv", "key": "keyword", "name": "name" }