Move the birth number into its own table under sex, drop the sexed US titles, and split the import helper by content

This commit is contained in:
2026-09-18 18:27:11 +02:00
parent 7d50f43d0b
commit 02eb95b94b
19 changed files with 102 additions and 80 deletions
+2 -1
View File
@@ -9,6 +9,7 @@ import io
import re
from pathlib import Path
import source
import tsv
SOURCE = "https://raw.githubusercontent.com/datasets/country-codes/main/data/country-codes.csv"
@@ -58,7 +59,7 @@ def main():
p.add_argument("--source", default=SOURCE)
p.add_argument("--out", default=str(OUT))
a = p.parse_args()
table = rows(tsv.fetch(a.source, a.cache, "country-codes.csv").decode("utf-8"))
table = rows(source.fetch(a.source, a.cache, "country-codes.csv").decode("utf-8"))
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["alpha2"]))
+3 -2
View File
@@ -11,6 +11,7 @@ import io
import xml.etree.ElementTree as ET
from pathlib import Path
import source
import tsv
SOURCE = "https://raw.githubusercontent.com/datasets/currency-codes/main/data/codes-all.csv"
@@ -58,8 +59,8 @@ def main():
p.add_argument("--symbols", nargs="+", default=SYMBOLS)
p.add_argument("--out", default=str(OUT))
a = p.parse_args()
symbol = symbols(tsv.fetch(s, a.cache, Path(s).name).decode("utf-8") for s in a.symbols)
table = rows(tsv.fetch(a.source, a.cache, "codes-all.csv").decode("utf-8"), symbol)
symbol = symbols(source.fetch(s, a.cache, Path(s).name).decode("utf-8") for s in a.symbols)
table = rows(source.fetch(a.source, a.cache, "codes-all.csv").decode("utf-8"), symbol)
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["code"]))
+6 -4
View File
@@ -16,6 +16,8 @@ import urllib.request
import zipfile
from pathlib import Path
import source
import xlsx
import tsv
CODES = "https://www.scb.se/contentassets/7a89e48960f741e08918e489ea36354a/kommunlankod-2026.xlsx"
@@ -39,7 +41,7 @@ ONE_POSITION = {"Stockholm", "Göteborg", "Malmö"}
UNMATCHED_POPULATION = 200
def scb_codes(cache):
regions, municipalities = {}, {}
for cells in tsv.xlsx_rows(tsv.fetch(CODES, cache, "kommunlankod.xlsx", magic=b"PK")):
for cells in xlsx.rows(source.fetch(CODES, cache, "kommunlankod.xlsx", magic=b"PK")):
if len(cells) < 2 or not re.fullmatch(r"\d{2}|\d{4}", cells[0]):
continue
(regions if len(cells[0]) == 2 else municipalities)[cells[0]] = cells[1].strip()
@@ -48,12 +50,12 @@ def scb_codes(cache):
def scb_population(cache):
body = json.dumps(POPULATION_QUERY).encode()
data = tsv.fetch(POPULATION, cache, "befolkning.json", data=body, headers={"Content-Type": "application/json"})
data = source.fetch(POPULATION, cache, "befolkning.json", data=body, headers={"Content-Type": "application/json"})
return {row["key"][0]: row["values"][0] for row in json.loads(data.decode("utf-8-sig"))["data"]}
def scb_tatorter(cache):
text = tsv.fetch(TATORTER, cache, "tatorter.csv").decode("utf-8")
text = source.fetch(TATORTER, cache, "tatorter.csv").decode("utf-8")
by_name = collections.defaultdict(list)
for r in csv.DictReader(io.StringIO(text)):
by_name[r["tatort"]].append((r["kommun"], int(r["bef"])))
@@ -61,7 +63,7 @@ def scb_tatorter(cache):
def geonames(cache):
z = zipfile.ZipFile(io.BytesIO(tsv.fetch(POSTAL_CODES, cache, "SE.zip", magic=b"PK")))
z = zipfile.ZipFile(io.BytesIO(source.fetch(POSTAL_CODES, cache, "SE.zip", magic=b"PK")))
rows = []
for line in z.read("SE.txt").decode("utf-8").splitlines():
f = line.split("\t")
+5 -4
View File
@@ -14,6 +14,7 @@ import sys
import zipfile
from pathlib import Path
import source
import tsv
GAZETTEER = "https://www2.census.gov/geo/docs/maps-data/data/gazetteer/2026_Gazetteer/2026_Gaz_{}_national.zip"
@@ -56,14 +57,14 @@ def text(data):
def gazetteer(cache, kind):
z = zipfile.ZipFile(io.BytesIO(tsv.fetch(GAZETTEER.format(kind), cache, f"gaz_{kind}.zip", magic=b"PK")))
z = zipfile.ZipFile(io.BytesIO(source.fetch(GAZETTEER.format(kind), cache, f"gaz_{kind}.zip", magic=b"PK")))
rows = text(z.read(z.namelist()[0])).splitlines()
header = [h.strip() for h in rows[0].split("|")]
return [dict(zip(header, (c.strip() for c in row.split("|")))) for row in rows[1:]]
def csv_rows(cache, url, name):
return list(csv.DictReader(io.StringIO(text(tsv.fetch(url, cache, name)))))
return list(csv.DictReader(io.StringIO(text(source.fetch(url, cache, name)))))
def dbf_rows(data, wanted):
@@ -89,7 +90,7 @@ def dbf_rows(data, wanted):
def tiger_zip(cache, kind, county):
return tsv.fetch(TIGER.format(kind.upper(), county, kind), cache, f"tl_{county}_{kind}.zip", magic=b"PK")
return source.fetch(TIGER.format(kind.upper(), county, kind), cache, f"tl_{county}_{kind}.zip", magic=b"PK")
def tiger(cache, kind, county, wanted):
@@ -130,7 +131,7 @@ def localities(cache, min_population, counties):
def postal_codes(cache, localities):
"""Each ZCTA whose largest part inside an incorporated place lies in a shipped place."""
parts = {}
for r in csv.DictReader(io.StringIO(text(tsv.fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"):
for r in csv.DictReader(io.StringIO(text(source.fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"):
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"] and not r["NAMELSAD_PLACE_20"].endswith(" CDP"):
parts.setdefault(r["GEOID_ZCTA5_20"], []).append((int(r["AREALAND_PART"]), r["GEOID_PLACE_20"]))
largest = {zcta: max(p)[1] for zcta, p in parts.items()}
+5 -3
View File
@@ -9,6 +9,8 @@ import argparse
import re
from pathlib import Path
import source
import xlsx
import tsv
SOURCE = "https://www.scb.se/contentassets/9fe7dbb460994c72b835163dbc491ef9/namn-med-minst-tva-barare-31-december-2022.xlsx"
@@ -40,9 +42,9 @@ def main():
p.add_argument("--out", default=str(OUT))
p.add_argument("--source", default=SOURCE)
a = p.parse_args()
data = tsv.fetch(a.source, a.cache, "scb-namn-2022.xlsx", magic=b"PK")
first = [{"name": cased(n), "sex": sex, "count": c} for sex, sheet in SHEETS.items() for n, c in counted(tsv.xlsx_rows(data, sheet), a.first)]
last = [{"name": cased(n), "count": c} for n, c in counted(tsv.xlsx_rows(data, SURNAMES), a.last)]
data = source.fetch(a.source, a.cache, "scb-namn-2022.xlsx", magic=b"PK")
first = [{"name": cased(n), "sex": sex, "count": c} for sex, sheet in SHEETS.items() for n, c in counted(xlsx.rows(data, sheet), a.first)]
last = [{"name": cased(n), "count": c} for n, c in counted(xlsx.rows(data, SURNAMES), a.last)]
out = Path(a.out)
tsv.write(out / "first-name.tsv", ["name", "sex", "count"], sorted(first, key=lambda r: (r["name"], r["sex"])))
tsv.write(out / "last-name.tsv", ["name", "count"], sorted(last, key=lambda r: r["name"]))
+5 -4
View File
@@ -15,6 +15,7 @@ import re
import zipfile
from pathlib import Path
import source
import tsv
NAMES = "https://raw.githubusercontent.com/hackerb9/ssa-baby-names/master/alldata.txt"
@@ -24,7 +25,7 @@ CACHE = Path(__file__).resolve().parent / "cache"
MC = re.compile(r"^Mc([a-z])")
def csv_rows(data, member):
def csv_or_zip_rows(data, member):
"""The rows of a CSV, or of every member of a zip named like member, name,sex,count[,year]."""
if data.startswith(b"PK"):
z = zipfile.ZipFile(io.BytesIO(data))
@@ -39,7 +40,7 @@ def csv_rows(data, member):
def given(data, from_year):
counts = collections.Counter()
for r in csv_rows(data, r"yob\d{4}\.txt"):
for r in csv_or_zip_rows(data, r"yob\d{4}\.txt"):
if len(r) >= 4 and r[3].isdigit() and int(r[3]) >= from_year:
counts[(r[0], r[1].lower())] += int(r[2])
return counts
@@ -60,12 +61,12 @@ def main():
p.add_argument("--out", default=str(OUT))
p.add_argument("--surnames", default=SURNAMES)
a = p.parse_args()
counts = given(tsv.fetch(a.names, a.cache, "ssa-names.txt"), a.from_year)
counts = given(source.fetch(a.names, a.cache, "ssa-names.txt"), a.from_year)
first = []
for sex in ("f", "m"):
top = sorted(((n, c) for (n, s), c in counts.items() if s == sex), key=lambda n: (-n[1], n[0]))[:a.first]
first += [{"name": n, "sex": sex, "count": c} for n, c in top]
rows = csv_rows(tsv.fetch(a.surnames, a.cache, "census-surnames-2010.zip"), r"Names_2010Census\.csv")
rows = csv_or_zip_rows(source.fetch(a.surnames, a.cache, "census-surnames-2010.zip"), r"Names_2010Census\.csv")
last = [{"name": surname(r[0]), "count": int(r[2])} for r in rows if len(r) >= 3 and r[2].isdigit() and r[0].isalpha()]
last = sorted(last, key=lambda r: (-r["count"], r["name"]))[:a.last]
out = Path(a.out)
+28
View File
@@ -0,0 +1,28 @@
"""A source fetched once into the cache."""
import re
import sys
import time
import urllib.request
from pathlib import Path
def fetch(source, cache, name, magic=b"", data=None, headers=None):
"""The bytes of a URL, downloaded into cache/name once, or of a local file."""
if not re.match(r"^https?://", source):
return Path(source).read_bytes()
path = Path(cache) / name
for attempt in range(1, 6):
if path.exists():
return path.read_bytes()
req = urllib.request.Request(source, data=data, headers={"User-Agent": "fejkdata data-import", **(headers or {})})
try:
with urllib.request.urlopen(req, timeout=600) as r:
body = r.read()
except OSError:
body = b""
if body and body.startswith(magic) and b"Request Rejected" not in body[:512]:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(body)
elif attempt < 5:
time.sleep(10 * attempt)
sys.exit(f"{source}: no valid download in 5 attempts")
+1 -45
View File
@@ -1,52 +1,8 @@
"""A source fetched once into the cache, and a table written as the loader admits it."""
import io
"""A table written as the loader admits it."""
import re
import sys
import time
import urllib.request
import xml.etree.ElementTree as ET
import zipfile
from pathlib import Path
XLSX_NS = {"m": "http://schemas.openxmlformats.org/spreadsheetml/2006/main", "r": "http://schemas.openxmlformats.org/officeDocument/2006/relationships"}
def fetch(source, cache, name, magic=b"", data=None, headers=None):
"""The bytes of a URL, downloaded into cache/name once, or of a local file."""
if not re.match(r"^https?://", source):
return Path(source).read_bytes()
path = Path(cache) / name
for attempt in range(1, 6):
if path.exists():
return path.read_bytes()
req = urllib.request.Request(source, data=data, headers={"User-Agent": "fejkdata data-import", **(headers or {})})
try:
with urllib.request.urlopen(req, timeout=600) as r:
body = r.read()
except OSError:
body = b""
if body and body.startswith(magic) and b"Request Rejected" not in body[:512]:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(body)
elif attempt < 5:
time.sleep(10 * attempt)
sys.exit(f"{source}: no valid download in 5 attempts")
def xlsx_rows(data, sheet=None):
"""The rows of an xlsx sheet named sheet, the first sheet by default, each a list of cell texts."""
z = zipfile.ZipFile(io.BytesIO(data))
strings = ["".join(t.text or "" for t in si.iter("{%s}t" % XLSX_NS["m"])) for si in ET.fromstring(z.read("xl/sharedStrings.xml")).findall("m:si", XLSX_NS)]
rels = {r.get("Id"): r.get("Target") for r in ET.fromstring(z.read("xl/_rels/workbook.xml.rels"))}
sheets = {s.get("name"): rels[s.get("{%s}id" % XLSX_NS["r"])] for s in ET.fromstring(z.read("xl/workbook.xml")).iter("{%s}sheet" % XLSX_NS["m"])}
target = sheets[sheet] if sheet else next(iter(sheets.values()))
for row in ET.fromstring(z.read("xl/" + target)).findall(".//m:row", XLSX_NS):
cells = []
for c in row.findall("m:c", XLSX_NS):
v = c.find("m:v", XLSX_NS)
cells.append("" if v is None else strings[int(v.text)] if c.get("t") == "s" else v.text)
yield cells
def write(path, columns, rows):
"""Write the rows as a TSV; every cell must be non-empty and free of tabs, newlines and braces."""
+21
View File
@@ -0,0 +1,21 @@
"""The rows of an xlsx sheet."""
import io
import xml.etree.ElementTree as ET
import zipfile
NS = {"m": "http://schemas.openxmlformats.org/spreadsheetml/2006/main", "r": "http://schemas.openxmlformats.org/officeDocument/2006/relationships"}
def rows(data, sheet=None):
"""The rows of an xlsx sheet named sheet, the first sheet by default, each a list of cell texts."""
z = zipfile.ZipFile(io.BytesIO(data))
strings = ["".join(t.text or "" for t in si.iter("{%s}t" % NS["m"])) for si in ET.fromstring(z.read("xl/sharedStrings.xml")).findall("m:si", NS)]
rels = {r.get("Id"): r.get("Target") for r in ET.fromstring(z.read("xl/_rels/workbook.xml.rels"))}
sheets = {s.get("name"): rels[s.get("{%s}id" % NS["r"])] for s in ET.fromstring(z.read("xl/workbook.xml")).iter("{%s}sheet" % NS["m"])}
target = sheets[sheet] if sheet else next(iter(sheets.values()))
for row in ET.fromstring(z.read("xl/" + target)).findall(".//m:row", NS):
cells = []
for c in row.findall("m:c", NS):
v = c.find("m:v", NS)
cells.append("" if v is None else strings[int(v.text)] if c.get("t") == "s" else v.text)
yield cells