Read misc.httpmethod, misc.protocol and misc.port from their IANA registries

This commit is contained in:
2026-09-20 00:01:57 +02:00
parent 9feadf4d15
commit 2016a7408c
13 changed files with 6152 additions and 7 deletions
+42
View File
@@ -0,0 +1,42 @@
#!/usr/bin/env python3
"""Rebuild data/misc/httpmethod.tsv from the IANA HTTP Method Registry.
data-import/httpmethod.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
A method whose name is not letters and hyphens does not ship, which drops the `*` the
registry holds to stop anyone registering it.
"""
import argparse
import csv
import io
import re
from pathlib import Path
import source
import tsv
SOURCE = "https://www.iana.org/assignments/http-methods/methods.csv"
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "httpmethod.tsv"
CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["idempotent", "method", "safe"]
def rows(text):
for r in csv.DictReader(io.StringIO(text)):
method, safe, idempotent = r["Method Name"].strip(), r["Safe"].strip(), r["Idempotent"].strip()
if re.match(r"^[A-Za-z][A-Za-z-]*$", method) and safe and idempotent:
yield {"idempotent": idempotent, "method": method, "safe": safe}
def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--cache", default=str(CACHE))
p.add_argument("--source", default=SOURCE)
p.add_argument("--out", default=str(OUT))
a = p.parse_args()
table = sorted(rows(source.fetch(a.source, a.cache, "http-methods.csv").decode("utf-8")), key=lambda r: r["method"])
tsv.write(a.out, COLUMNS, table)
if __name__ == "__main__":
main()
+49
View File
@@ -0,0 +1,49 @@
#!/usr/bin/env python3
"""Rebuild data/misc/port.tsv from the IANA Service Name and Transport Protocol Port Number Registry.
data-import/port.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
What ships is the TCP assignments in use: a row needs a service name, a port number and
a description the registry has filled in. A port the registry lists more than once keeps
the first service, so a number selects one row.
"""
import argparse
import csv
import io
from pathlib import Path
import source
import tsv
SOURCE = "https://www.iana.org/assignments/service-names-port-numbers/service-names-port-numbers.csv"
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "port.tsv"
CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["name", "number"]
UNUSED = ("Reserved", "Unassigned", "IANA assigned this well-formed service name to replace an unregistered or squatted port.")
def rows(text):
seen = set()
for r in csv.DictReader(io.StringIO(text)):
name, number = r["Service Name"].strip(), r["Port Number"].strip()
described = (r["Description"] or "").strip()
if r["Transport Protocol"].strip() != "tcp" or not name or not number.isdigit():
continue
if not described or described in UNUSED or number in seen:
continue
seen.add(number)
yield {"name": name, "number": number}
def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--cache", default=str(CACHE))
p.add_argument("--source", default=SOURCE)
p.add_argument("--out", default=str(OUT))
a = p.parse_args()
table = sorted(rows(source.fetch(a.source, a.cache, "service-names-port-numbers.csv").decode("utf-8")), key=lambda r: int(r["number"]))
tsv.write(a.out, COLUMNS, table)
if __name__ == "__main__":
main()
+43
View File
@@ -0,0 +1,43 @@
#!/usr/bin/env python3
"""Rebuild data/misc/protocol.tsv from the IANA Protocol Numbers registry.
data-import/protocol.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
A number the registry leaves unassigned, reserves or marks deprecated does not ship.
Where it names no protocol beside the keyword, the keyword is the name.
"""
import argparse
import csv
import io
from pathlib import Path
import source
import tsv
SOURCE = "https://www.iana.org/assignments/protocol-numbers/protocol-numbers-1.csv"
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "protocol.tsv"
CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["keyword", "name", "number"]
SKIP = ("Unassigned", "Reserved", "deprecated")
def rows(text):
for r in csv.DictReader(io.StringIO(text)):
number, keyword = r["Decimal"].strip(), r["Keyword"].strip()
if not number.isdigit() or not keyword or any(s in keyword for s in SKIP):
continue
yield {"keyword": keyword, "name": " ".join(r["Protocol"].split()) or keyword, "number": number}
def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--cache", default=str(CACHE))
p.add_argument("--source", default=SOURCE)
p.add_argument("--out", default=str(OUT))
a = p.parse_args()
table = sorted(rows(source.fetch(a.source, a.cache, "protocol-numbers-1.csv").decode("utf-8")), key=lambda r: int(r["number"]))
tsv.write(a.out, COLUMNS, table)
if __name__ == "__main__":
main()