Share one fetch and one TSV writer across the import scripts, and name the geo scripts' stages

This commit is contained in:
2026-09-18 01:10:44 +02:00
parent 9e37f3dc51
commit 92f58edc2d
5 changed files with 170 additions and 194 deletions
+8 -23
View File
@@ -1,35 +1,28 @@
#!/usr/bin/env python3
"""Rebuild data/misc/currency.tsv from datasets/currency-codes (PDDL) and CLDR's symbols (Unicode).
data-import/currency.py [--source URL_OR_FILE] [--symbols URL_OR_FILE ...] [--out FILE]
data-import/currency.py [--source URL_OR_FILE] [--symbols URL_OR_FILE ...] [--cache DIR] [--out FILE]
The symbols come from the first locale file that has one, narrow symbols before wide.
"""
import argparse
import csv
import io
import re
import sys
import urllib.request
import xml.etree.ElementTree as ET
from pathlib import Path
import tsv
SOURCE = "https://raw.githubusercontent.com/datasets/currency-codes/main/data/codes-all.csv"
SYMBOLS = [
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/en.xml",
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/root.xml",
]
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "currency.tsv"
CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["code", "decimals", "name", "numeric", "symbol"]
def read(source):
if re.match(r"^https?://", source):
with urllib.request.urlopen(source, timeout=60) as r:
return r.read().decode("utf-8")
return Path(source).read_text(encoding="utf-8")
def symbols(xml_texts):
"""CLDR's symbol per code: the first locale's narrow symbol, else the first locale's wide one."""
narrow, wide = {}, {}
@@ -58,24 +51,16 @@ def rows(text, symbol):
}
def write(out, table):
lines = ["\t".join(COLUMNS)]
for row in sorted(table, key=lambda r: r["code"]):
cells = [row[c] for c in COLUMNS]
assert all(cells) and not any("\t" in c or "\n" in c for c in cells), row
lines.append("\t".join(cells))
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
return len(lines) - 1
def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--cache", default=str(CACHE))
p.add_argument("--source", default=SOURCE)
p.add_argument("--symbols", nargs="+", default=SYMBOLS)
p.add_argument("--out", default=str(OUT))
a = p.parse_args()
n = write(a.out, rows(read(a.source), symbols(read(s) for s in a.symbols)))
print(f"{a.out}: {n} rows", file=sys.stderr)
symbol = symbols(tsv.fetch(s, a.cache, Path(s).name).decode("utf-8") for s in a.symbols)
table = rows(tsv.fetch(a.source, a.cache, "codes-all.csv").decode("utf-8"), symbol)
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["code"]))
if __name__ == "__main__":