Add DATA-LICENSES.md, the data-import scripts and their compose service

This commit is contained in:
2026-09-17 12:28:52 +02:00
parent 51b9385aee
commit e0aa3ac8a4
4 changed files with 189 additions and 0 deletions
+81
View File
@@ -0,0 +1,81 @@
#!/usr/bin/env python3
"""Rebuild data/misc/country.tsv from datasets/country-codes (PDDL).
data-import/country.py [--source URL_OR_FILE] [--out FILE]
"""
import argparse
import csv
import io
import re
import sys
import urllib.request
from pathlib import Path
SOURCE = "https://raw.githubusercontent.com/datasets/country-codes/main/data/country-codes.csv"
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "country.tsv"
COLUMNS = ["alpha2", "alpha3", "calling-code", "capital", "currency", "flag", "languages", "name", "numeric", "tld"]
# Gaps in the source, keyed by alpha2.
FIXUPS = {"TR": {"currency": "TRY"}}
def read(source):
if re.match(r"^https?://", source):
with urllib.request.urlopen(source, timeout=60) as r:
return r.read().decode("utf-8")
return Path(source).read_text(encoding="utf-8")
def flag(alpha2):
return "".join(chr(0x1F1E6 + ord(c) - ord("A")) for c in alpha2)
def calling_code(dial):
first = re.sub(r"[^0-9-]", "", dial.split(",")[0])
return first if re.match(r"^\d+-\d{3}$", first) else first.split("-")[0]
def languages(field):
return ",".join(p for p in field.split(",") if p)
def rows(text):
for r in csv.DictReader(io.StringIO(text)):
alpha2 = r["ISO3166-1-Alpha-2"]
row = {
"alpha2": alpha2,
"alpha3": r["ISO3166-1-Alpha-3"],
"calling-code": calling_code(r["Dial"]),
"capital": r["Capital"],
"currency": r["ISO4217-currency_alphabetic_code"].split(",")[0],
"flag": flag(alpha2),
"languages": languages(r["Languages"]),
"name": r["CLDR display name"] or r["official_name_en"],
"numeric": r["ISO3166-1-numeric"].zfill(3),
"tld": r["TLD"],
}
row.update(FIXUPS.get(alpha2, {}))
if all(row[c] for c in ("alpha2", "alpha3", "calling-code", "capital", "currency", "name", "tld")):
yield row
def write(out, table):
lines = ["\t".join(COLUMNS)]
for row in sorted(table, key=lambda r: r["alpha2"]):
cells = [row[c] for c in COLUMNS]
assert not any("\t" in c or "\n" in c for c in cells), row
lines.append("\t".join(cells))
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
return len(lines) - 1
def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--source", default=SOURCE)
p.add_argument("--out", default=str(OUT))
a = p.parse_args()
n = write(a.out, rows(read(a.source)))
print(f"{a.out}: {n} rows", file=sys.stderr)
if __name__ == "__main__":
main()