Read misc.language, httpstatus and mimetype from their registers, and keep a country's currency a column
This commit is contained in:
@@ -0,0 +1,40 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Rebuild data/misc/httpstatus.tsv from the IANA HTTP Status Code Registry.
|
||||
|
||||
data-import/httpstatus.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
|
||||
|
||||
What ships is the codes in use: a range, an unassigned code and a qualified reason are all skipped.
|
||||
"""
|
||||
import argparse
|
||||
import csv
|
||||
import io
|
||||
from pathlib import Path
|
||||
|
||||
import source
|
||||
import tsv
|
||||
|
||||
SOURCE = "https://www.iana.org/assignments/http-status-codes/http-status-codes-1.csv"
|
||||
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "httpstatus.tsv"
|
||||
CACHE = Path(__file__).resolve().parent / "cache"
|
||||
COLUMNS = ["code", "reason"]
|
||||
|
||||
|
||||
def rows(text):
|
||||
for r in csv.DictReader(io.StringIO(text)):
|
||||
code, reason = r["Value"].strip(), r["Description"].strip()
|
||||
if code.isdigit() and reason and "(" not in reason and reason != "Unassigned":
|
||||
yield {"code": code, "reason": reason}
|
||||
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||
p.add_argument("--cache", default=str(CACHE))
|
||||
p.add_argument("--source", default=SOURCE)
|
||||
p.add_argument("--out", default=str(OUT))
|
||||
a = p.parse_args()
|
||||
table = rows(source.fetch(a.source, a.cache, "http-status-codes-1.csv").decode("utf-8"))
|
||||
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: int(r["code"])))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,45 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Rebuild data/misc/language.tsv from datasets/language-codes (PDDL), the LoC ISO 639-2 register.
|
||||
|
||||
data-import/language.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
|
||||
|
||||
Only the 639-2 entries carrying a 639-1 code ship, since the alpha-2 code is the key.
|
||||
"""
|
||||
import argparse
|
||||
import csv
|
||||
import io
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
import source
|
||||
import tsv
|
||||
|
||||
SOURCE = "https://raw.githubusercontent.com/datasets/language-codes/main/data/language-codes-full.csv"
|
||||
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "language.tsv"
|
||||
CACHE = Path(__file__).resolve().parent / "cache"
|
||||
COLUMNS = ["code", "code3", "name"]
|
||||
|
||||
|
||||
def name(english):
|
||||
"""The register lists every synonym; the first, without its qualifier, is the one a selector spells."""
|
||||
return re.sub(r"\s*\([^)]*\)", "", english.split(";")[0]).strip()
|
||||
|
||||
|
||||
def rows(text):
|
||||
for r in csv.DictReader(io.StringIO(text)):
|
||||
if r["alpha2"]:
|
||||
yield {"code": r["alpha2"], "code3": r["alpha3-t"] or r["alpha3-b"], "name": name(r["English"])}
|
||||
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||
p.add_argument("--cache", default=str(CACHE))
|
||||
p.add_argument("--source", default=SOURCE)
|
||||
p.add_argument("--out", default=str(OUT))
|
||||
a = p.parse_args()
|
||||
table = rows(source.fetch(a.source, a.cache, "language-codes-full.csv").decode("utf-8"))
|
||||
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["code"]))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,38 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Rebuild data/misc/mimetype.tsv from mime-db (MIT), the IANA media type registry with extensions.
|
||||
|
||||
data-import/mimetype.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
|
||||
|
||||
Only IANA-registered types with a filename extension ship; the first extension is the one listed.
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import source
|
||||
import tsv
|
||||
|
||||
SOURCE = "https://raw.githubusercontent.com/jshttp/mime-db/master/db.json"
|
||||
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "mimetype.tsv"
|
||||
CACHE = Path(__file__).resolve().parent / "cache"
|
||||
COLUMNS = ["ext", "type"]
|
||||
|
||||
|
||||
def rows(db):
|
||||
for mimetype, entry in db.items():
|
||||
if entry.get("source") == "iana" and entry.get("extensions"):
|
||||
yield {"ext": "." + entry["extensions"][0], "type": mimetype}
|
||||
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||
p.add_argument("--cache", default=str(CACHE))
|
||||
p.add_argument("--source", default=SOURCE)
|
||||
p.add_argument("--out", default=str(OUT))
|
||||
a = p.parse_args()
|
||||
table = rows(json.loads(source.fetch(a.source, a.cache, "mime-db.json").decode("utf-8")))
|
||||
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["type"]))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user