diff --git a/DATA-LICENSES.md b/DATA-LICENSES.md new file mode 100644 index 0000000..c514066 --- /dev/null +++ b/DATA-LICENSES.md @@ -0,0 +1,18 @@ +# Data licenses + +Every shipped dataset, its source, its licence and the attribution it asks for. A +`data-import/` script rebuilds each sourced table; a curated one is hand-written. + +| Table | Source | Licence | Attribution | Rebuild | +|-------|--------|---------|-------------|---------| +| `misc/country.tsv` | [datasets/country-codes](https://github.com/datasets/country-codes) | [PDDL 1.0](https://opendatacommons.org/licenses/pddl/1-0/) | none required | `data-import/country.py` | +| `misc/currency.tsv` | [datasets/currency-codes](https://github.com/datasets/currency-codes); symbols from [Unicode CLDR](https://github.com/unicode-org/cldr) `en.xml` and `root.xml` | PDDL 1.0; [Unicode License v3](https://www.unicode.org/license.txt) | CLDR: "Copyright © 1991-2025 Unicode, Inc. Unicode and the Unicode Logo are registered trademarks of Unicode, Inc. in the United States and other countries." | `data-import/currency.py` | +| `misc/httpstatus.tsv` | curated (IANA HTTP status codes are facts) | — | — | — | +| `misc/language.tsv` | curated (ISO 639-1 codes are facts) | — | — | — | +| `misc/mimetype.tsv` | curated (IANA media types are facts) | — | — | — | + +Every other category is hand-written JSON under [`data/`](data), MIT like the code. + +```sh +docker compose run --rm --user "$(id -u):$(id -g)" data-import data-import/country.py +``` diff --git a/compose.yaml b/compose.yaml index ca3af11..1d19a46 100644 --- a/compose.yaml +++ b/compose.yaml @@ -64,5 +64,13 @@ services: <<: *go command: sh + # Rebuilds a shipped TSV from its source: docker compose run --rm data-import data-import/country.py + data-import: + image: python:3.14.7-slim + working_dir: /app + volumes: + - .:/app + entrypoint: python3 + volumes: gocache: diff --git a/data-import/country.py b/data-import/country.py new file mode 100644 index 0000000..326ab57 --- /dev/null +++ b/data-import/country.py @@ -0,0 +1,81 @@ +#!/usr/bin/env python3 +"""Rebuild data/misc/country.tsv from datasets/country-codes (PDDL). + + data-import/country.py [--source URL_OR_FILE] [--out FILE] +""" +import argparse +import csv +import io +import re +import sys +import urllib.request +from pathlib import Path + +SOURCE = "https://raw.githubusercontent.com/datasets/country-codes/main/data/country-codes.csv" +OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "country.tsv" +COLUMNS = ["alpha2", "alpha3", "calling-code", "capital", "currency", "flag", "languages", "name", "numeric", "tld"] +# Gaps in the source, keyed by alpha2. +FIXUPS = {"TR": {"currency": "TRY"}} + + +def read(source): + if re.match(r"^https?://", source): + with urllib.request.urlopen(source, timeout=60) as r: + return r.read().decode("utf-8") + return Path(source).read_text(encoding="utf-8") + + +def flag(alpha2): + return "".join(chr(0x1F1E6 + ord(c) - ord("A")) for c in alpha2) + + +def calling_code(dial): + first = re.sub(r"[^0-9-]", "", dial.split(",")[0]) + return first if re.match(r"^\d+-\d{3}$", first) else first.split("-")[0] + + +def languages(field): + return ",".join(p for p in field.split(",") if p) + + +def rows(text): + for r in csv.DictReader(io.StringIO(text)): + alpha2 = r["ISO3166-1-Alpha-2"] + row = { + "alpha2": alpha2, + "alpha3": r["ISO3166-1-Alpha-3"], + "calling-code": calling_code(r["Dial"]), + "capital": r["Capital"], + "currency": r["ISO4217-currency_alphabetic_code"].split(",")[0], + "flag": flag(alpha2), + "languages": languages(r["Languages"]), + "name": r["CLDR display name"] or r["official_name_en"], + "numeric": r["ISO3166-1-numeric"].zfill(3), + "tld": r["TLD"], + } + row.update(FIXUPS.get(alpha2, {})) + if all(row[c] for c in ("alpha2", "alpha3", "calling-code", "capital", "currency", "name", "tld")): + yield row + + +def write(out, table): + lines = ["\t".join(COLUMNS)] + for row in sorted(table, key=lambda r: r["alpha2"]): + cells = [row[c] for c in COLUMNS] + assert not any("\t" in c or "\n" in c for c in cells), row + lines.append("\t".join(cells)) + Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8") + return len(lines) - 1 + + +def main(): + p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + p.add_argument("--source", default=SOURCE) + p.add_argument("--out", default=str(OUT)) + a = p.parse_args() + n = write(a.out, rows(read(a.source))) + print(f"{a.out}: {n} rows", file=sys.stderr) + + +if __name__ == "__main__": + main() diff --git a/data-import/currency.py b/data-import/currency.py new file mode 100644 index 0000000..9389e3e --- /dev/null +++ b/data-import/currency.py @@ -0,0 +1,82 @@ +#!/usr/bin/env python3 +"""Rebuild data/misc/currency.tsv from datasets/currency-codes (PDDL) and CLDR's symbols (Unicode). + + data-import/currency.py [--source URL_OR_FILE] [--symbols URL_OR_FILE ...] [--out FILE] + +The symbols come from the first locale file that has one, narrow symbols before wide. +""" +import argparse +import csv +import io +import re +import sys +import urllib.request +import xml.etree.ElementTree as ET +from pathlib import Path + +SOURCE = "https://raw.githubusercontent.com/datasets/currency-codes/main/data/codes-all.csv" +SYMBOLS = [ + "https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/en.xml", + "https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/root.xml", +] +OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "currency.tsv" +COLUMNS = ["code", "decimals", "name", "numeric", "symbol"] + + +def read(source): + if re.match(r"^https?://", source): + with urllib.request.urlopen(source, timeout=60) as r: + return r.read().decode("utf-8") + return Path(source).read_text(encoding="utf-8") + + +def symbols(xml_texts): + """CLDR's symbol per code: the first locale's narrow symbol, else the first locale's wide one.""" + narrow, wide = {}, {} + for text in xml_texts: + for c in ET.fromstring(text).iter("currency"): + for s in c.findall("symbol"): + into = narrow if s.get("alt") == "narrow" else wide if s.get("alt") is None else None + if into is not None: + into.setdefault(c.get("type"), s.text) + return {**wide, **narrow} + + +def rows(text, symbol): + seen = set() + for r in csv.DictReader(io.StringIO(text)): + code = r["AlphabeticCode"] + if not code or code in seen or r["WithdrawalDate"] or r["Entity"].startswith("ZZ"): + continue + seen.add(code) + yield { + "code": code, + "decimals": r["MinorUnit"], + "name": r["Currency"], + "numeric": r["NumericCode"].zfill(3), + "symbol": symbol.get(code, code), + } + + +def write(out, table): + lines = ["\t".join(COLUMNS)] + for row in sorted(table, key=lambda r: r["code"]): + cells = [row[c] for c in COLUMNS] + assert all(cells) and not any("\t" in c or "\n" in c for c in cells), row + lines.append("\t".join(cells)) + Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8") + return len(lines) - 1 + + +def main(): + p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + p.add_argument("--source", default=SOURCE) + p.add_argument("--symbols", nargs="+", default=SYMBOLS) + p.add_argument("--out", default=str(OUT)) + a = p.parse_args() + n = write(a.out, rows(read(a.source), symbols(read(s) for s in a.symbols))) + print(f"{a.out}: {n} rows", file=sys.stderr) + + +if __name__ == "__main__": + main()