Add DATA-LICENSES.md, the data-import scripts and their compose service

This commit is contained in:
2026-09-17 12:28:52 +02:00
parent 51b9385aee
commit e0aa3ac8a4
4 changed files with 189 additions and 0 deletions
+18
View File
@@ -0,0 +1,18 @@
# Data licenses
Every shipped dataset, its source, its licence and the attribution it asks for. A
`data-import/` script rebuilds each sourced table; a curated one is hand-written.
| Table | Source | Licence | Attribution | Rebuild |
|-------|--------|---------|-------------|---------|
| `misc/country.tsv` | [datasets/country-codes](https://github.com/datasets/country-codes) | [PDDL 1.0](https://opendatacommons.org/licenses/pddl/1-0/) | none required | `data-import/country.py` |
| `misc/currency.tsv` | [datasets/currency-codes](https://github.com/datasets/currency-codes); symbols from [Unicode CLDR](https://github.com/unicode-org/cldr) `en.xml` and `root.xml` | PDDL 1.0; [Unicode License v3](https://www.unicode.org/license.txt) | CLDR: "Copyright © 1991-2025 Unicode, Inc. Unicode and the Unicode Logo are registered trademarks of Unicode, Inc. in the United States and other countries." | `data-import/currency.py` |
| `misc/httpstatus.tsv` | curated (IANA HTTP status codes are facts) | — | — | — |
| `misc/language.tsv` | curated (ISO 639-1 codes are facts) | — | — | — |
| `misc/mimetype.tsv` | curated (IANA media types are facts) | — | — | — |
Every other category is hand-written JSON under [`data/`](data), MIT like the code.
```sh
docker compose run --rm --user "$(id -u):$(id -g)" data-import data-import/country.py
```
+8
View File
@@ -64,5 +64,13 @@ services:
<<: *go <<: *go
command: sh command: sh
# Rebuilds a shipped TSV from its source: docker compose run --rm data-import data-import/country.py
data-import:
image: python:3.14.7-slim
working_dir: /app
volumes:
- .:/app
entrypoint: python3
volumes: volumes:
gocache: gocache:
+81
View File
@@ -0,0 +1,81 @@
#!/usr/bin/env python3
"""Rebuild data/misc/country.tsv from datasets/country-codes (PDDL).
data-import/country.py [--source URL_OR_FILE] [--out FILE]
"""
import argparse
import csv
import io
import re
import sys
import urllib.request
from pathlib import Path
SOURCE = "https://raw.githubusercontent.com/datasets/country-codes/main/data/country-codes.csv"
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "country.tsv"
COLUMNS = ["alpha2", "alpha3", "calling-code", "capital", "currency", "flag", "languages", "name", "numeric", "tld"]
# Gaps in the source, keyed by alpha2.
FIXUPS = {"TR": {"currency": "TRY"}}
def read(source):
if re.match(r"^https?://", source):
with urllib.request.urlopen(source, timeout=60) as r:
return r.read().decode("utf-8")
return Path(source).read_text(encoding="utf-8")
def flag(alpha2):
return "".join(chr(0x1F1E6 + ord(c) - ord("A")) for c in alpha2)
def calling_code(dial):
first = re.sub(r"[^0-9-]", "", dial.split(",")[0])
return first if re.match(r"^\d+-\d{3}$", first) else first.split("-")[0]
def languages(field):
return ",".join(p for p in field.split(",") if p)
def rows(text):
for r in csv.DictReader(io.StringIO(text)):
alpha2 = r["ISO3166-1-Alpha-2"]
row = {
"alpha2": alpha2,
"alpha3": r["ISO3166-1-Alpha-3"],
"calling-code": calling_code(r["Dial"]),
"capital": r["Capital"],
"currency": r["ISO4217-currency_alphabetic_code"].split(",")[0],
"flag": flag(alpha2),
"languages": languages(r["Languages"]),
"name": r["CLDR display name"] or r["official_name_en"],
"numeric": r["ISO3166-1-numeric"].zfill(3),
"tld": r["TLD"],
}
row.update(FIXUPS.get(alpha2, {}))
if all(row[c] for c in ("alpha2", "alpha3", "calling-code", "capital", "currency", "name", "tld")):
yield row
def write(out, table):
lines = ["\t".join(COLUMNS)]
for row in sorted(table, key=lambda r: r["alpha2"]):
cells = [row[c] for c in COLUMNS]
assert not any("\t" in c or "\n" in c for c in cells), row
lines.append("\t".join(cells))
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
return len(lines) - 1
def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--source", default=SOURCE)
p.add_argument("--out", default=str(OUT))
a = p.parse_args()
n = write(a.out, rows(read(a.source)))
print(f"{a.out}: {n} rows", file=sys.stderr)
if __name__ == "__main__":
main()
+82
View File
@@ -0,0 +1,82 @@
#!/usr/bin/env python3
"""Rebuild data/misc/currency.tsv from datasets/currency-codes (PDDL) and CLDR's symbols (Unicode).
data-import/currency.py [--source URL_OR_FILE] [--symbols URL_OR_FILE ...] [--out FILE]
The symbols come from the first locale file that has one, narrow symbols before wide.
"""
import argparse
import csv
import io
import re
import sys
import urllib.request
import xml.etree.ElementTree as ET
from pathlib import Path
SOURCE = "https://raw.githubusercontent.com/datasets/currency-codes/main/data/codes-all.csv"
SYMBOLS = [
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/en.xml",
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/root.xml",
]
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "currency.tsv"
COLUMNS = ["code", "decimals", "name", "numeric", "symbol"]
def read(source):
if re.match(r"^https?://", source):
with urllib.request.urlopen(source, timeout=60) as r:
return r.read().decode("utf-8")
return Path(source).read_text(encoding="utf-8")
def symbols(xml_texts):
"""CLDR's symbol per code: the first locale's narrow symbol, else the first locale's wide one."""
narrow, wide = {}, {}
for text in xml_texts:
for c in ET.fromstring(text).iter("currency"):
for s in c.findall("symbol"):
into = narrow if s.get("alt") == "narrow" else wide if s.get("alt") is None else None
if into is not None:
into.setdefault(c.get("type"), s.text)
return {**wide, **narrow}
def rows(text, symbol):
seen = set()
for r in csv.DictReader(io.StringIO(text)):
code = r["AlphabeticCode"]
if not code or code in seen or r["WithdrawalDate"] or r["Entity"].startswith("ZZ"):
continue
seen.add(code)
yield {
"code": code,
"decimals": r["MinorUnit"],
"name": r["Currency"],
"numeric": r["NumericCode"].zfill(3),
"symbol": symbol.get(code, code),
}
def write(out, table):
lines = ["\t".join(COLUMNS)]
for row in sorted(table, key=lambda r: r["code"]):
cells = [row[c] for c in COLUMNS]
assert all(cells) and not any("\t" in c or "\n" in c for c in cells), row
lines.append("\t".join(cells))
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
return len(lines) - 1
def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--source", default=SOURCE)
p.add_argument("--symbols", nargs="+", default=SYMBOLS)
p.add_argument("--out", default=str(OUT))
a = p.parse_args()
n = write(a.out, rows(read(a.source), symbols(read(s) for s in a.symbols)))
print(f"{a.out}: {n} rows", file=sys.stderr)
if __name__ == "__main__":
main()