Read misc.language, httpstatus and mimetype from their registers, and keep a country's currency a column

This commit is contained in:
2026-09-18 21:44:51 +02:00
parent 3e1d3914c2
commit 8432a7c4ae
10 changed files with 1067 additions and 29 deletions
+40
View File
@@ -0,0 +1,40 @@
#!/usr/bin/env python3
"""Rebuild data/misc/httpstatus.tsv from the IANA HTTP Status Code Registry.
data-import/httpstatus.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
What ships is the codes in use: a range, an unassigned code and a qualified reason are all skipped.
"""
import argparse
import csv
import io
from pathlib import Path
import source
import tsv
SOURCE = "https://www.iana.org/assignments/http-status-codes/http-status-codes-1.csv"
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "httpstatus.tsv"
CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["code", "reason"]
def rows(text):
for r in csv.DictReader(io.StringIO(text)):
code, reason = r["Value"].strip(), r["Description"].strip()
if code.isdigit() and reason and "(" not in reason and reason != "Unassigned":
yield {"code": code, "reason": reason}
def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--cache", default=str(CACHE))
p.add_argument("--source", default=SOURCE)
p.add_argument("--out", default=str(OUT))
a = p.parse_args()
table = rows(source.fetch(a.source, a.cache, "http-status-codes-1.csv").decode("utf-8"))
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: int(r["code"])))
if __name__ == "__main__":
main()
+45
View File
@@ -0,0 +1,45 @@
#!/usr/bin/env python3
"""Rebuild data/misc/language.tsv from datasets/language-codes (PDDL), the LoC ISO 639-2 register.
data-import/language.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
Only the 639-2 entries carrying a 639-1 code ship, since the alpha-2 code is the key.
"""
import argparse
import csv
import io
import re
from pathlib import Path
import source
import tsv
SOURCE = "https://raw.githubusercontent.com/datasets/language-codes/main/data/language-codes-full.csv"
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "language.tsv"
CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["code", "code3", "name"]
def name(english):
"""The register lists every synonym; the first, without its qualifier, is the one a selector spells."""
return re.sub(r"\s*\([^)]*\)", "", english.split(";")[0]).strip()
def rows(text):
for r in csv.DictReader(io.StringIO(text)):
if r["alpha2"]:
yield {"code": r["alpha2"], "code3": r["alpha3-t"] or r["alpha3-b"], "name": name(r["English"])}
def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--cache", default=str(CACHE))
p.add_argument("--source", default=SOURCE)
p.add_argument("--out", default=str(OUT))
a = p.parse_args()
table = rows(source.fetch(a.source, a.cache, "language-codes-full.csv").decode("utf-8"))
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["code"]))
if __name__ == "__main__":
main()
+38
View File
@@ -0,0 +1,38 @@
#!/usr/bin/env python3
"""Rebuild data/misc/mimetype.tsv from mime-db (MIT), the IANA media type registry with extensions.
data-import/mimetype.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
Only IANA-registered types with a filename extension ship; the first extension is the one listed.
"""
import argparse
import json
from pathlib import Path
import source
import tsv
SOURCE = "https://raw.githubusercontent.com/jshttp/mime-db/master/db.json"
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "mimetype.tsv"
CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["ext", "type"]
def rows(db):
for mimetype, entry in db.items():
if entry.get("source") == "iana" and entry.get("extensions"):
yield {"ext": "." + entry["extensions"][0], "type": mimetype}
def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--cache", default=str(CACHE))
p.add_argument("--source", default=SOURCE)
p.add_argument("--out", default=str(OUT))
a = p.parse_args()
table = rows(json.loads(source.fetch(a.source, a.cache, "mime-db.json").decode("utf-8")))
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["type"]))
if __name__ == "__main__":
main()