Add the table node, row selection and linked tables, and ship the misc registers as tables #19
@@ -0,0 +1,18 @@
|
||||
# Data licenses
|
||||
|
||||
Every shipped dataset, its source, its licence and the attribution it asks for. A
|
||||
`data-import/` script rebuilds each sourced table; a curated one is hand-written.
|
||||
|
||||
| Table | Source | Licence | Attribution | Rebuild |
|
||||
|-------|--------|---------|-------------|---------|
|
||||
| `misc/country.tsv` | [datasets/country-codes](https://github.com/datasets/country-codes) | [PDDL 1.0](https://opendatacommons.org/licenses/pddl/1-0/) | none required | `data-import/country.py` |
|
||||
| `misc/currency.tsv` | [datasets/currency-codes](https://github.com/datasets/currency-codes); symbols from [Unicode CLDR](https://github.com/unicode-org/cldr) `en.xml` and `root.xml` | PDDL 1.0; [Unicode License v3](https://www.unicode.org/license.txt) | CLDR: "Copyright © 1991-2025 Unicode, Inc. Unicode and the Unicode Logo are registered trademarks of Unicode, Inc. in the United States and other countries." | `data-import/currency.py` |
|
||||
| `misc/httpstatus.tsv` | curated (IANA HTTP status codes are facts) | — | — | — |
|
||||
| `misc/language.tsv` | curated (ISO 639-1 codes are facts) | — | — | — |
|
||||
| `misc/mimetype.tsv` | curated (IANA media types are facts) | — | — | — |
|
||||
|
||||
Every other category is hand-written JSON under [`data/`](data), MIT like the code.
|
||||
|
||||
```sh
|
||||
docker compose run --rm --user "$(id -u):$(id -g)" data-import data-import/country.py
|
||||
```
|
||||
@@ -64,5 +64,13 @@ services:
|
||||
<<: *go
|
||||
command: sh
|
||||
|
||||
# Rebuilds a shipped TSV from its source: docker compose run --rm data-import data-import/country.py
|
||||
data-import:
|
||||
image: python:3.14.7-slim
|
||||
working_dir: /app
|
||||
volumes:
|
||||
- .:/app
|
||||
entrypoint: python3
|
||||
|
||||
volumes:
|
||||
gocache:
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Rebuild data/misc/country.tsv from datasets/country-codes (PDDL).
|
||||
|
||||
data-import/country.py [--source URL_OR_FILE] [--out FILE]
|
||||
"""
|
||||
import argparse
|
||||
import csv
|
||||
import io
|
||||
import re
|
||||
import sys
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
SOURCE = "https://raw.githubusercontent.com/datasets/country-codes/main/data/country-codes.csv"
|
||||
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "country.tsv"
|
||||
COLUMNS = ["alpha2", "alpha3", "calling-code", "capital", "currency", "flag", "languages", "name", "numeric", "tld"]
|
||||
# Gaps in the source, keyed by alpha2.
|
||||
FIXUPS = {"TR": {"currency": "TRY"}}
|
||||
|
||||
|
||||
def read(source):
|
||||
if re.match(r"^https?://", source):
|
||||
with urllib.request.urlopen(source, timeout=60) as r:
|
||||
return r.read().decode("utf-8")
|
||||
return Path(source).read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def flag(alpha2):
|
||||
return "".join(chr(0x1F1E6 + ord(c) - ord("A")) for c in alpha2)
|
||||
|
||||
|
||||
def calling_code(dial):
|
||||
first = re.sub(r"[^0-9-]", "", dial.split(",")[0])
|
||||
return first if re.match(r"^\d+-\d{3}$", first) else first.split("-")[0]
|
||||
|
||||
|
||||
def languages(field):
|
||||
return ",".join(p for p in field.split(",") if p)
|
||||
|
||||
|
||||
def rows(text):
|
||||
for r in csv.DictReader(io.StringIO(text)):
|
||||
alpha2 = r["ISO3166-1-Alpha-2"]
|
||||
row = {
|
||||
"alpha2": alpha2,
|
||||
"alpha3": r["ISO3166-1-Alpha-3"],
|
||||
"calling-code": calling_code(r["Dial"]),
|
||||
"capital": r["Capital"],
|
||||
"currency": r["ISO4217-currency_alphabetic_code"].split(",")[0],
|
||||
"flag": flag(alpha2),
|
||||
"languages": languages(r["Languages"]),
|
||||
"name": r["CLDR display name"] or r["official_name_en"],
|
||||
"numeric": r["ISO3166-1-numeric"].zfill(3),
|
||||
"tld": r["TLD"],
|
||||
}
|
||||
row.update(FIXUPS.get(alpha2, {}))
|
||||
if all(row[c] for c in ("alpha2", "alpha3", "calling-code", "capital", "currency", "name", "tld")):
|
||||
yield row
|
||||
|
||||
|
||||
def write(out, table):
|
||||
lines = ["\t".join(COLUMNS)]
|
||||
for row in sorted(table, key=lambda r: r["alpha2"]):
|
||||
cells = [row[c] for c in COLUMNS]
|
||||
assert not any("\t" in c or "\n" in c for c in cells), row
|
||||
lines.append("\t".join(cells))
|
||||
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
|
||||
return len(lines) - 1
|
||||
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||
p.add_argument("--source", default=SOURCE)
|
||||
p.add_argument("--out", default=str(OUT))
|
||||
a = p.parse_args()
|
||||
n = write(a.out, rows(read(a.source)))
|
||||
print(f"{a.out}: {n} rows", file=sys.stderr)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,82 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Rebuild data/misc/currency.tsv from datasets/currency-codes (PDDL) and CLDR's symbols (Unicode).
|
||||
|
||||
data-import/currency.py [--source URL_OR_FILE] [--symbols URL_OR_FILE ...] [--out FILE]
|
||||
|
||||
The symbols come from the first locale file that has one, narrow symbols before wide.
|
||||
"""
|
||||
import argparse
|
||||
import csv
|
||||
import io
|
||||
import re
|
||||
import sys
|
||||
import urllib.request
|
||||
import xml.etree.ElementTree as ET
|
||||
from pathlib import Path
|
||||
|
||||
SOURCE = "https://raw.githubusercontent.com/datasets/currency-codes/main/data/codes-all.csv"
|
||||
SYMBOLS = [
|
||||
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/en.xml",
|
||||
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/root.xml",
|
||||
]
|
||||
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "currency.tsv"
|
||||
COLUMNS = ["code", "decimals", "name", "numeric", "symbol"]
|
||||
|
||||
|
||||
def read(source):
|
||||
if re.match(r"^https?://", source):
|
||||
with urllib.request.urlopen(source, timeout=60) as r:
|
||||
return r.read().decode("utf-8")
|
||||
return Path(source).read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def symbols(xml_texts):
|
||||
"""CLDR's symbol per code: the first locale's narrow symbol, else the first locale's wide one."""
|
||||
narrow, wide = {}, {}
|
||||
for text in xml_texts:
|
||||
for c in ET.fromstring(text).iter("currency"):
|
||||
for s in c.findall("symbol"):
|
||||
into = narrow if s.get("alt") == "narrow" else wide if s.get("alt") is None else None
|
||||
if into is not None:
|
||||
into.setdefault(c.get("type"), s.text)
|
||||
return {**wide, **narrow}
|
||||
|
||||
|
||||
def rows(text, symbol):
|
||||
seen = set()
|
||||
for r in csv.DictReader(io.StringIO(text)):
|
||||
code = r["AlphabeticCode"]
|
||||
if not code or code in seen or r["WithdrawalDate"] or r["Entity"].startswith("ZZ"):
|
||||
continue
|
||||
seen.add(code)
|
||||
yield {
|
||||
"code": code,
|
||||
"decimals": r["MinorUnit"],
|
||||
"name": r["Currency"],
|
||||
"numeric": r["NumericCode"].zfill(3),
|
||||
"symbol": symbol.get(code, code),
|
||||
}
|
||||
|
||||
|
||||
def write(out, table):
|
||||
lines = ["\t".join(COLUMNS)]
|
||||
for row in sorted(table, key=lambda r: r["code"]):
|
||||
cells = [row[c] for c in COLUMNS]
|
||||
assert all(cells) and not any("\t" in c or "\n" in c for c in cells), row
|
||||
lines.append("\t".join(cells))
|
||||
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
|
||||
return len(lines) - 1
|
||||
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||
p.add_argument("--source", default=SOURCE)
|
||||
p.add_argument("--symbols", nargs="+", default=SYMBOLS)
|
||||
p.add_argument("--out", default=str(OUT))
|
||||
a = p.parse_args()
|
||||
n = write(a.out, rows(read(a.source), symbols(read(s) for s in a.symbols)))
|
||||
print(f"{a.out}: {n} rows", file=sys.stderr)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user