Add DATA-LICENSES.md, the data-import scripts and their compose service
This commit is contained in:
@@ -0,0 +1,82 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Rebuild data/misc/currency.tsv from datasets/currency-codes (PDDL) and CLDR's symbols (Unicode).
|
||||
|
||||
data-import/currency.py [--source URL_OR_FILE] [--symbols URL_OR_FILE ...] [--out FILE]
|
||||
|
||||
The symbols come from the first locale file that has one, narrow symbols before wide.
|
||||
"""
|
||||
import argparse
|
||||
import csv
|
||||
import io
|
||||
import re
|
||||
import sys
|
||||
import urllib.request
|
||||
import xml.etree.ElementTree as ET
|
||||
from pathlib import Path
|
||||
|
||||
SOURCE = "https://raw.githubusercontent.com/datasets/currency-codes/main/data/codes-all.csv"
|
||||
SYMBOLS = [
|
||||
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/en.xml",
|
||||
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/root.xml",
|
||||
]
|
||||
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "currency.tsv"
|
||||
COLUMNS = ["code", "decimals", "name", "numeric", "symbol"]
|
||||
|
||||
|
||||
def read(source):
|
||||
if re.match(r"^https?://", source):
|
||||
with urllib.request.urlopen(source, timeout=60) as r:
|
||||
return r.read().decode("utf-8")
|
||||
return Path(source).read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def symbols(xml_texts):
|
||||
"""CLDR's symbol per code: the first locale's narrow symbol, else the first locale's wide one."""
|
||||
narrow, wide = {}, {}
|
||||
for text in xml_texts:
|
||||
for c in ET.fromstring(text).iter("currency"):
|
||||
for s in c.findall("symbol"):
|
||||
into = narrow if s.get("alt") == "narrow" else wide if s.get("alt") is None else None
|
||||
if into is not None:
|
||||
into.setdefault(c.get("type"), s.text)
|
||||
return {**wide, **narrow}
|
||||
|
||||
|
||||
def rows(text, symbol):
|
||||
seen = set()
|
||||
for r in csv.DictReader(io.StringIO(text)):
|
||||
code = r["AlphabeticCode"]
|
||||
if not code or code in seen or r["WithdrawalDate"] or r["Entity"].startswith("ZZ"):
|
||||
continue
|
||||
seen.add(code)
|
||||
yield {
|
||||
"code": code,
|
||||
"decimals": r["MinorUnit"],
|
||||
"name": r["Currency"],
|
||||
"numeric": r["NumericCode"].zfill(3),
|
||||
"symbol": symbol.get(code, code),
|
||||
}
|
||||
|
||||
|
||||
def write(out, table):
|
||||
lines = ["\t".join(COLUMNS)]
|
||||
for row in sorted(table, key=lambda r: r["code"]):
|
||||
cells = [row[c] for c in COLUMNS]
|
||||
assert all(cells) and not any("\t" in c or "\n" in c for c in cells), row
|
||||
lines.append("\t".join(cells))
|
||||
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
|
||||
return len(lines) - 1
|
||||
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||
p.add_argument("--source", default=SOURCE)
|
||||
p.add_argument("--symbols", nargs="+", default=SYMBOLS)
|
||||
p.add_argument("--out", default=str(OUT))
|
||||
a = p.parse_args()
|
||||
n = write(a.out, rows(read(a.source), symbols(read(s) for s in a.symbols)))
|
||||
print(f"{a.out}: {n} rows", file=sys.stderr)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user