Add DATA-LICENSES.md, the data-import scripts and their compose service
This commit is contained in:
@@ -0,0 +1,18 @@
|
|||||||
|
# Data licenses
|
||||||
|
|
||||||
|
Every shipped dataset, its source, its licence and the attribution it asks for. A
|
||||||
|
`data-import/` script rebuilds each sourced table; a curated one is hand-written.
|
||||||
|
|
||||||
|
| Table | Source | Licence | Attribution | Rebuild |
|
||||||
|
|-------|--------|---------|-------------|---------|
|
||||||
|
| `misc/country.tsv` | [datasets/country-codes](https://github.com/datasets/country-codes) | [PDDL 1.0](https://opendatacommons.org/licenses/pddl/1-0/) | none required | `data-import/country.py` |
|
||||||
|
| `misc/currency.tsv` | [datasets/currency-codes](https://github.com/datasets/currency-codes); symbols from [Unicode CLDR](https://github.com/unicode-org/cldr) `en.xml` and `root.xml` | PDDL 1.0; [Unicode License v3](https://www.unicode.org/license.txt) | CLDR: "Copyright © 1991-2025 Unicode, Inc. Unicode and the Unicode Logo are registered trademarks of Unicode, Inc. in the United States and other countries." | `data-import/currency.py` |
|
||||||
|
| `misc/httpstatus.tsv` | curated (IANA HTTP status codes are facts) | — | — | — |
|
||||||
|
| `misc/language.tsv` | curated (ISO 639-1 codes are facts) | — | — | — |
|
||||||
|
| `misc/mimetype.tsv` | curated (IANA media types are facts) | — | — | — |
|
||||||
|
|
||||||
|
Every other category is hand-written JSON under [`data/`](data), MIT like the code.
|
||||||
|
|
||||||
|
```sh
|
||||||
|
docker compose run --rm --user "$(id -u):$(id -g)" data-import data-import/country.py
|
||||||
|
```
|
||||||
@@ -64,5 +64,13 @@ services:
|
|||||||
<<: *go
|
<<: *go
|
||||||
command: sh
|
command: sh
|
||||||
|
|
||||||
|
# Rebuilds a shipped TSV from its source: docker compose run --rm data-import data-import/country.py
|
||||||
|
data-import:
|
||||||
|
image: python:3.14.7-slim
|
||||||
|
working_dir: /app
|
||||||
|
volumes:
|
||||||
|
- .:/app
|
||||||
|
entrypoint: python3
|
||||||
|
|
||||||
volumes:
|
volumes:
|
||||||
gocache:
|
gocache:
|
||||||
|
|||||||
@@ -0,0 +1,81 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Rebuild data/misc/country.tsv from datasets/country-codes (PDDL).
|
||||||
|
|
||||||
|
data-import/country.py [--source URL_OR_FILE] [--out FILE]
|
||||||
|
"""
|
||||||
|
import argparse
|
||||||
|
import csv
|
||||||
|
import io
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
import urllib.request
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
SOURCE = "https://raw.githubusercontent.com/datasets/country-codes/main/data/country-codes.csv"
|
||||||
|
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "country.tsv"
|
||||||
|
COLUMNS = ["alpha2", "alpha3", "calling-code", "capital", "currency", "flag", "languages", "name", "numeric", "tld"]
|
||||||
|
# Gaps in the source, keyed by alpha2.
|
||||||
|
FIXUPS = {"TR": {"currency": "TRY"}}
|
||||||
|
|
||||||
|
|
||||||
|
def read(source):
|
||||||
|
if re.match(r"^https?://", source):
|
||||||
|
with urllib.request.urlopen(source, timeout=60) as r:
|
||||||
|
return r.read().decode("utf-8")
|
||||||
|
return Path(source).read_text(encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def flag(alpha2):
|
||||||
|
return "".join(chr(0x1F1E6 + ord(c) - ord("A")) for c in alpha2)
|
||||||
|
|
||||||
|
|
||||||
|
def calling_code(dial):
|
||||||
|
first = re.sub(r"[^0-9-]", "", dial.split(",")[0])
|
||||||
|
return first if re.match(r"^\d+-\d{3}$", first) else first.split("-")[0]
|
||||||
|
|
||||||
|
|
||||||
|
def languages(field):
|
||||||
|
return ",".join(p for p in field.split(",") if p)
|
||||||
|
|
||||||
|
|
||||||
|
def rows(text):
|
||||||
|
for r in csv.DictReader(io.StringIO(text)):
|
||||||
|
alpha2 = r["ISO3166-1-Alpha-2"]
|
||||||
|
row = {
|
||||||
|
"alpha2": alpha2,
|
||||||
|
"alpha3": r["ISO3166-1-Alpha-3"],
|
||||||
|
"calling-code": calling_code(r["Dial"]),
|
||||||
|
"capital": r["Capital"],
|
||||||
|
"currency": r["ISO4217-currency_alphabetic_code"].split(",")[0],
|
||||||
|
"flag": flag(alpha2),
|
||||||
|
"languages": languages(r["Languages"]),
|
||||||
|
"name": r["CLDR display name"] or r["official_name_en"],
|
||||||
|
"numeric": r["ISO3166-1-numeric"].zfill(3),
|
||||||
|
"tld": r["TLD"],
|
||||||
|
}
|
||||||
|
row.update(FIXUPS.get(alpha2, {}))
|
||||||
|
if all(row[c] for c in ("alpha2", "alpha3", "calling-code", "capital", "currency", "name", "tld")):
|
||||||
|
yield row
|
||||||
|
|
||||||
|
|
||||||
|
def write(out, table):
|
||||||
|
lines = ["\t".join(COLUMNS)]
|
||||||
|
for row in sorted(table, key=lambda r: r["alpha2"]):
|
||||||
|
cells = [row[c] for c in COLUMNS]
|
||||||
|
assert not any("\t" in c or "\n" in c for c in cells), row
|
||||||
|
lines.append("\t".join(cells))
|
||||||
|
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
|
||||||
|
return len(lines) - 1
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||||
|
p.add_argument("--source", default=SOURCE)
|
||||||
|
p.add_argument("--out", default=str(OUT))
|
||||||
|
a = p.parse_args()
|
||||||
|
n = write(a.out, rows(read(a.source)))
|
||||||
|
print(f"{a.out}: {n} rows", file=sys.stderr)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,82 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Rebuild data/misc/currency.tsv from datasets/currency-codes (PDDL) and CLDR's symbols (Unicode).
|
||||||
|
|
||||||
|
data-import/currency.py [--source URL_OR_FILE] [--symbols URL_OR_FILE ...] [--out FILE]
|
||||||
|
|
||||||
|
The symbols come from the first locale file that has one, narrow symbols before wide.
|
||||||
|
"""
|
||||||
|
import argparse
|
||||||
|
import csv
|
||||||
|
import io
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
import urllib.request
|
||||||
|
import xml.etree.ElementTree as ET
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
SOURCE = "https://raw.githubusercontent.com/datasets/currency-codes/main/data/codes-all.csv"
|
||||||
|
SYMBOLS = [
|
||||||
|
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/en.xml",
|
||||||
|
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/root.xml",
|
||||||
|
]
|
||||||
|
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "currency.tsv"
|
||||||
|
COLUMNS = ["code", "decimals", "name", "numeric", "symbol"]
|
||||||
|
|
||||||
|
|
||||||
|
def read(source):
|
||||||
|
if re.match(r"^https?://", source):
|
||||||
|
with urllib.request.urlopen(source, timeout=60) as r:
|
||||||
|
return r.read().decode("utf-8")
|
||||||
|
return Path(source).read_text(encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def symbols(xml_texts):
|
||||||
|
"""CLDR's symbol per code: the first locale's narrow symbol, else the first locale's wide one."""
|
||||||
|
narrow, wide = {}, {}
|
||||||
|
for text in xml_texts:
|
||||||
|
for c in ET.fromstring(text).iter("currency"):
|
||||||
|
for s in c.findall("symbol"):
|
||||||
|
into = narrow if s.get("alt") == "narrow" else wide if s.get("alt") is None else None
|
||||||
|
if into is not None:
|
||||||
|
into.setdefault(c.get("type"), s.text)
|
||||||
|
return {**wide, **narrow}
|
||||||
|
|
||||||
|
|
||||||
|
def rows(text, symbol):
|
||||||
|
seen = set()
|
||||||
|
for r in csv.DictReader(io.StringIO(text)):
|
||||||
|
code = r["AlphabeticCode"]
|
||||||
|
if not code or code in seen or r["WithdrawalDate"] or r["Entity"].startswith("ZZ"):
|
||||||
|
continue
|
||||||
|
seen.add(code)
|
||||||
|
yield {
|
||||||
|
"code": code,
|
||||||
|
"decimals": r["MinorUnit"],
|
||||||
|
"name": r["Currency"],
|
||||||
|
"numeric": r["NumericCode"].zfill(3),
|
||||||
|
"symbol": symbol.get(code, code),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def write(out, table):
|
||||||
|
lines = ["\t".join(COLUMNS)]
|
||||||
|
for row in sorted(table, key=lambda r: r["code"]):
|
||||||
|
cells = [row[c] for c in COLUMNS]
|
||||||
|
assert all(cells) and not any("\t" in c or "\n" in c for c in cells), row
|
||||||
|
lines.append("\t".join(cells))
|
||||||
|
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
|
||||||
|
return len(lines) - 1
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||||
|
p.add_argument("--source", default=SOURCE)
|
||||||
|
p.add_argument("--symbols", nargs="+", default=SYMBOLS)
|
||||||
|
p.add_argument("--out", default=str(OUT))
|
||||||
|
a = p.parse_args()
|
||||||
|
n = write(a.out, rows(read(a.source), symbols(read(s) for s in a.symbols)))
|
||||||
|
print(f"{a.out}: {n} rows", file=sys.stderr)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user