Drop a filter that matched nothing, keep the 79 ports it hid, and exit on a value the registers should not hold
This commit is contained in:
+3
-3
@@ -45,9 +45,9 @@ replacement, and each removed path, column or flag.
|
|||||||
`model` column.
|
`model` column.
|
||||||
- `misc.httpmethod`, `misc.protocol` and `misc.port` are IANA's registries as tables.
|
- `misc.httpmethod`, `misc.protocol` and `misc.port` are IANA's registries as tables.
|
||||||
`misc.httpmethod` carries `safe` and `idempotent` beside the method; `misc.protocol`
|
`misc.httpmethod` carries `safe` and `idempotent` beside the method; `misc.protocol`
|
||||||
the keyword, its `number` and its `name`; `misc.port` renders a TCP port number,
|
the keyword, its `number` and its `name`, selectable by either; `misc.port` renders a
|
||||||
with the service `name` beside it, keeping the first service the registry lists for
|
TCP port number, with the IANA `service` name beside it, keeping the first service the
|
||||||
a port so a number selects one row.
|
registry describes for a port so a number selects one row.
|
||||||
- `geo.SE` and `geo.US`: five linked tables per country, `region`, `municipality`,
|
- `geo.SE` and `geo.US`: five linked tables per country, `region`, `municipality`,
|
||||||
`locality`, `postal-code` and `street`, weighted by population and address counts
|
`locality`, `postal-code` and `street`, weighted by population and address counts
|
||||||
and built from SCB, GeoNames, Trafikverket NVDB and the US Census Bureau, and an
|
and built from SCB, GeoNames, Trafikverket NVDB and the US Census Bureau, and an
|
||||||
|
|||||||
@@ -164,16 +164,16 @@ Each locale carries `address`, `color`, `company`, `date`, `email`, `first-name`
|
|||||||
carries `car`, `coordinate`, `creditcard` (Luhn-valid), `currency` (ISO 4217),
|
carries `car`, `coordinate`, `creditcard` (Luhn-valid), `currency` (ISO 4217),
|
||||||
`datetime` (RFC 3339), `emoji`, `httpmethod`, `httpstatus`, `language` (ISO 639-1,
|
`datetime` (RFC 3339), `emoji`, `httpmethod`, `httpstatus`, `language` (ISO 639-1,
|
||||||
with its 639-2/T code), `mac`, `mimetype`, `objectid`, `port`, `protocol`,
|
with its 639-2/T code), `mac`, `mimetype`, `objectid`, `port`, `protocol`,
|
||||||
`territory` (ISO 3166-1), `timezone` (IANA), `useragent` and `uuid` (v4). Many carry sub-fields — `misc.currency.symbol`,
|
`territory` (ISO 3166-1), `timezone` (IANA), `useragent` and `uuid` (v4). Many carry
|
||||||
`misc.territory.alpha2`, `misc.httpstatus.code` — which `--list` shows. `car`,
|
sub-fields — `misc.currency.symbol`, `misc.territory.alpha2`, `misc.httpstatus.code` —
|
||||||
`currency`, `httpstatus`, `language`, `mimetype`, `territory`, `timezone` and
|
which `--list` shows. `car`, `currency`, `httpmethod`, `httpstatus`, `language`,
|
||||||
`httpmethod`, `port`, `protocol` and `useragent` are [tables](#table), so
|
`mimetype`, `port`, `protocol`, `territory`, `timezone` and `useragent` are
|
||||||
`misc.territory[SE].capital` and
|
[tables](#table), so `misc.territory[SE].capital` and `misc.currency[Euro].symbol`
|
||||||
`misc.currency[Euro].symbol` select a row; `car` and `useragent` carry no key or
|
select a row; `car` and `useragent` carry no key or name, so they are drawn from rather
|
||||||
name, so they are drawn from rather than selected in. `misc.httpmethod`,
|
than selected in. `misc.httpmethod`, `misc.protocol` and `misc.port` are IANA's
|
||||||
`misc.protocol` and `misc.port` are IANA's registries: a method carries whether it
|
registries: a method carries the register's `yes` or `no` for `safe` and `idempotent`,
|
||||||
is `safe` and `idempotent`, a protocol its `number`, and `misc.port` renders the
|
a protocol its `number` and its spelled-out `name`, and `misc.port` renders the number
|
||||||
number a port field holds, with the service `name` beside it.
|
a port field holds, with the IANA `service` name beside it.
|
||||||
[`DATA-LICENSES.md`](DATA-LICENSES.md) names each table's source and licence.
|
[`DATA-LICENSES.md`](DATA-LICENSES.md) names each table's source and licence.
|
||||||
|
|
||||||
`misc.timezone` is every zone tzdb gives a shipped territory, from one apiece for most
|
`misc.timezone` is every zone tzdb gives a shipped territory, from one apiece for most
|
||||||
|
|||||||
@@ -3,13 +3,14 @@
|
|||||||
|
|
||||||
data-import/httpmethod.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
|
data-import/httpmethod.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
|
||||||
|
|
||||||
A method whose name is not letters and hyphens does not ship, which drops the `*` the
|
A row needs a name of letters and hyphens, which drops the registry's `*`, and both of
|
||||||
registry holds to stop anyone registering it.
|
the answers the registry gives for it.
|
||||||
"""
|
"""
|
||||||
import argparse
|
import argparse
|
||||||
import csv
|
import csv
|
||||||
import io
|
import io
|
||||||
import re
|
import re
|
||||||
|
import sys
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
import source
|
import source
|
||||||
@@ -19,12 +20,18 @@ SOURCE = "https://www.iana.org/assignments/http-methods/methods.csv"
|
|||||||
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "httpmethod.tsv"
|
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "httpmethod.tsv"
|
||||||
CACHE = Path(__file__).resolve().parent / "cache"
|
CACHE = Path(__file__).resolve().parent / "cache"
|
||||||
COLUMNS = ["idempotent", "method", "safe"]
|
COLUMNS = ["idempotent", "method", "safe"]
|
||||||
|
ANSWERS = ("yes", "no")
|
||||||
|
|
||||||
|
|
||||||
def rows(text):
|
def rows(text):
|
||||||
for r in csv.DictReader(io.StringIO(text)):
|
for r in csv.DictReader(io.StringIO(text)):
|
||||||
method, safe, idempotent = r["Method Name"].strip(), r["Safe"].strip(), r["Idempotent"].strip()
|
method = (r["Method Name"] or "").strip()
|
||||||
if re.match(r"^[A-Za-z][A-Za-z-]*$", method) and safe and idempotent:
|
safe, idempotent = (r["Safe"] or "").strip(), (r["Idempotent"] or "").strip()
|
||||||
|
if not re.fullmatch(r"[A-Za-z][A-Za-z-]*", method) or not safe or not idempotent:
|
||||||
|
continue
|
||||||
|
for column, answer in (("Safe", safe), ("Idempotent", idempotent)):
|
||||||
|
if answer not in ANSWERS:
|
||||||
|
sys.exit(f"{method}: {column} {answer!r} is neither yes nor no")
|
||||||
yield {"idempotent": idempotent, "method": method, "safe": safe}
|
yield {"idempotent": idempotent, "method": method, "safe": safe}
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+20
-14
@@ -3,13 +3,14 @@
|
|||||||
|
|
||||||
data-import/port.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
|
data-import/port.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
|
||||||
|
|
||||||
What ships is the TCP assignments in use: a row needs a service name, a port number and
|
A row needs a service name and a numeric TCP port. A port the registry lists more than
|
||||||
a description the registry has filled in. A port the registry lists more than once keeps
|
once keeps the first service it describes, or the first of them where it describes none,
|
||||||
the first service, so a number selects one row.
|
so a number selects one row.
|
||||||
"""
|
"""
|
||||||
import argparse
|
import argparse
|
||||||
import csv
|
import csv
|
||||||
import io
|
import io
|
||||||
|
import sys
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
import source
|
import source
|
||||||
@@ -18,21 +19,24 @@ import tsv
|
|||||||
SOURCE = "https://www.iana.org/assignments/service-names-port-numbers/service-names-port-numbers.csv"
|
SOURCE = "https://www.iana.org/assignments/service-names-port-numbers/service-names-port-numbers.csv"
|
||||||
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "port.tsv"
|
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "port.tsv"
|
||||||
CACHE = Path(__file__).resolve().parent / "cache"
|
CACHE = Path(__file__).resolve().parent / "cache"
|
||||||
COLUMNS = ["name", "number"]
|
COLUMNS = ["number", "service"]
|
||||||
UNUSED = ("Reserved", "Unassigned", "IANA assigned this well-formed service name to replace an unregistered or squatted port.")
|
FLOOR = 4000
|
||||||
|
|
||||||
|
|
||||||
def rows(text):
|
def rows(text):
|
||||||
seen = set()
|
best, described = {}, set()
|
||||||
for r in csv.DictReader(io.StringIO(text)):
|
for r in csv.DictReader(io.StringIO(text)):
|
||||||
name, number = r["Service Name"].strip(), r["Port Number"].strip()
|
service, number = (r["Service Name"] or "").strip(), (r["Port Number"] or "").strip()
|
||||||
described = (r["Description"] or "").strip()
|
transport, description = (r["Transport Protocol"] or "").strip(), (r["Description"] or "").strip()
|
||||||
if r["Transport Protocol"].strip() != "tcp" or not name or not number.isdigit():
|
if transport != "tcp" or not service or not number.isdigit():
|
||||||
continue
|
continue
|
||||||
if not described or described in UNUSED or number in seen:
|
if not 1 <= int(number) <= 65535:
|
||||||
continue
|
sys.exit(f"{service}: port {number} is outside 1-65535")
|
||||||
seen.add(number)
|
if number not in best or (description and number not in described):
|
||||||
yield {"name": name, "number": number}
|
best[number] = {"number": number, "service": service}
|
||||||
|
if description:
|
||||||
|
described.add(number)
|
||||||
|
return sorted(best.values(), key=lambda r: int(r["number"]))
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
@@ -41,7 +45,9 @@ def main():
|
|||||||
p.add_argument("--source", default=SOURCE)
|
p.add_argument("--source", default=SOURCE)
|
||||||
p.add_argument("--out", default=str(OUT))
|
p.add_argument("--out", default=str(OUT))
|
||||||
a = p.parse_args()
|
a = p.parse_args()
|
||||||
table = sorted(rows(source.fetch(a.source, a.cache, "service-names-port-numbers.csv").decode("utf-8")), key=lambda r: int(r["number"]))
|
table = rows(source.fetch(a.source, a.cache, "service-names-port-numbers.csv").decode("utf-8"))
|
||||||
|
if len(table) < FLOOR:
|
||||||
|
sys.exit(f"only {len(table)} ports named a TCP service; the registry's columns have moved")
|
||||||
tsv.write(a.out, COLUMNS, table)
|
tsv.write(a.out, COLUMNS, table)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+21
-5
@@ -3,12 +3,14 @@
|
|||||||
|
|
||||||
data-import/protocol.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
|
data-import/protocol.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
|
||||||
|
|
||||||
A number the registry leaves unassigned, reserves or marks deprecated does not ship.
|
A row needs a number and a keyword, which drops the seven numbers assigned to a class of
|
||||||
Where it names no protocol beside the keyword, the keyword is the name.
|
protocols rather than to one; a keyword the registry reserves or marks deprecated goes
|
||||||
|
too. Where the registry names no protocol beside the keyword, the keyword is the name.
|
||||||
"""
|
"""
|
||||||
import argparse
|
import argparse
|
||||||
import csv
|
import csv
|
||||||
import io
|
import io
|
||||||
|
import sys
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
import source
|
import source
|
||||||
@@ -18,15 +20,28 @@ SOURCE = "https://www.iana.org/assignments/protocol-numbers/protocol-numbers-1.c
|
|||||||
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "protocol.tsv"
|
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "protocol.tsv"
|
||||||
CACHE = Path(__file__).resolve().parent / "cache"
|
CACHE = Path(__file__).resolve().parent / "cache"
|
||||||
COLUMNS = ["keyword", "name", "number"]
|
COLUMNS = ["keyword", "name", "number"]
|
||||||
SKIP = ("Unassigned", "Reserved", "deprecated")
|
SKIP = ("Reserved", "deprecated")
|
||||||
|
|
||||||
|
|
||||||
def rows(text):
|
def rows(text):
|
||||||
for r in csv.DictReader(io.StringIO(text)):
|
for r in csv.DictReader(io.StringIO(text)):
|
||||||
number, keyword = r["Decimal"].strip(), r["Keyword"].strip()
|
number, keyword = (r["Decimal"] or "").strip(), (r["Keyword"] or "").strip()
|
||||||
if not number.isdigit() or not keyword or any(s in keyword for s in SKIP):
|
if not number.isdigit() or not keyword or any(s in keyword for s in SKIP):
|
||||||
continue
|
continue
|
||||||
yield {"keyword": keyword, "name": " ".join(r["Protocol"].split()) or keyword, "number": number}
|
if not 0 <= int(number) <= 255:
|
||||||
|
sys.exit(f"{keyword}: protocol number {number} is outside 0-255")
|
||||||
|
yield {"keyword": keyword, "name": " ".join((r["Protocol"] or "").split()) or keyword, "number": number}
|
||||||
|
|
||||||
|
|
||||||
|
def refuse_a_name_the_loader_would(table):
|
||||||
|
"""A name spelling another row's keyword, or a second row's name, is a load error for every consumer."""
|
||||||
|
keywords, seen = {r["keyword"] for r in table}, {}
|
||||||
|
for r in table:
|
||||||
|
if r["name"] != r["keyword"] and r["name"] in keywords:
|
||||||
|
sys.exit(f"{r['keyword']}: name {r['name']!r} is another protocol's keyword")
|
||||||
|
if r["name"] in seen:
|
||||||
|
sys.exit(f"{r['keyword']}: name {r['name']!r} repeats {seen[r['name']]}'s")
|
||||||
|
seen[r["name"]] = r["keyword"]
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
@@ -36,6 +51,7 @@ def main():
|
|||||||
p.add_argument("--out", default=str(OUT))
|
p.add_argument("--out", default=str(OUT))
|
||||||
a = p.parse_args()
|
a = p.parse_args()
|
||||||
table = sorted(rows(source.fetch(a.source, a.cache, "protocol-numbers-1.csv").decode("utf-8")), key=lambda r: int(r["number"]))
|
table = sorted(rows(source.fetch(a.source, a.cache, "protocol-numbers-1.csv").decode("utf-8")), key=lambda r: int(r["number"]))
|
||||||
|
refuse_a_name_the_loader_would(table)
|
||||||
tsv.write(a.out, COLUMNS, table)
|
tsv.write(a.out, COLUMNS, table)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+5890
-5811
File diff suppressed because it is too large
Load Diff
@@ -1 +1 @@
|
|||||||
{ "format": "{keyword}", "rows": "protocol.tsv", "key": "keyword" }
|
{ "format": "{keyword}", "rows": "protocol.tsv", "key": "keyword", "name": "name" }
|
||||||
|
|||||||
Reference in New Issue
Block a user