Share one fetch and one TSV writer across the import scripts, and name the geo scripts' stages
This commit is contained in:
+8
-23
@@ -1,35 +1,28 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Rebuild data/misc/currency.tsv from datasets/currency-codes (PDDL) and CLDR's symbols (Unicode).
|
||||
|
||||
data-import/currency.py [--source URL_OR_FILE] [--symbols URL_OR_FILE ...] [--out FILE]
|
||||
data-import/currency.py [--source URL_OR_FILE] [--symbols URL_OR_FILE ...] [--cache DIR] [--out FILE]
|
||||
|
||||
The symbols come from the first locale file that has one, narrow symbols before wide.
|
||||
"""
|
||||
import argparse
|
||||
import csv
|
||||
import io
|
||||
import re
|
||||
import sys
|
||||
import urllib.request
|
||||
import xml.etree.ElementTree as ET
|
||||
from pathlib import Path
|
||||
|
||||
import tsv
|
||||
|
||||
SOURCE = "https://raw.githubusercontent.com/datasets/currency-codes/main/data/codes-all.csv"
|
||||
SYMBOLS = [
|
||||
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/en.xml",
|
||||
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/root.xml",
|
||||
]
|
||||
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "currency.tsv"
|
||||
CACHE = Path(__file__).resolve().parent / "cache"
|
||||
COLUMNS = ["code", "decimals", "name", "numeric", "symbol"]
|
||||
|
||||
|
||||
def read(source):
|
||||
if re.match(r"^https?://", source):
|
||||
with urllib.request.urlopen(source, timeout=60) as r:
|
||||
return r.read().decode("utf-8")
|
||||
return Path(source).read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def symbols(xml_texts):
|
||||
"""CLDR's symbol per code: the first locale's narrow symbol, else the first locale's wide one."""
|
||||
narrow, wide = {}, {}
|
||||
@@ -58,24 +51,16 @@ def rows(text, symbol):
|
||||
}
|
||||
|
||||
|
||||
def write(out, table):
|
||||
lines = ["\t".join(COLUMNS)]
|
||||
for row in sorted(table, key=lambda r: r["code"]):
|
||||
cells = [row[c] for c in COLUMNS]
|
||||
assert all(cells) and not any("\t" in c or "\n" in c for c in cells), row
|
||||
lines.append("\t".join(cells))
|
||||
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
|
||||
return len(lines) - 1
|
||||
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||
p.add_argument("--cache", default=str(CACHE))
|
||||
p.add_argument("--source", default=SOURCE)
|
||||
p.add_argument("--symbols", nargs="+", default=SYMBOLS)
|
||||
p.add_argument("--out", default=str(OUT))
|
||||
a = p.parse_args()
|
||||
n = write(a.out, rows(read(a.source), symbols(read(s) for s in a.symbols)))
|
||||
print(f"{a.out}: {n} rows", file=sys.stderr)
|
||||
symbol = symbols(tsv.fetch(s, a.cache, Path(s).name).decode("utf-8") for s in a.symbols)
|
||||
table = rows(tsv.fetch(a.source, a.cache, "codes-all.csv").decode("utf-8"), symbol)
|
||||
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["code"]))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
Reference in New Issue
Block a user