Weighted first and last names from SCB, the SSA and the Census under a sex table, personnummer, samordningsnummer, SSN and ITIN in their valid ranges, and date, time and datetime over date()
This commit is contained in:
+1
-17
@@ -13,7 +13,6 @@ import os
|
||||
import re
|
||||
import sys
|
||||
import urllib.request
|
||||
import xml.etree.ElementTree as ET
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
@@ -38,24 +37,9 @@ CACHE = Path(__file__).resolve().parent / "cache"
|
||||
TIMEZONE = "Europe/Stockholm"
|
||||
ONE_POSITION = {"Stockholm", "Göteborg", "Malmö"}
|
||||
UNMATCHED_POPULATION = 200
|
||||
XLSX_NS = {"m": "http://schemas.openxmlformats.org/spreadsheetml/2006/main"}
|
||||
|
||||
|
||||
def xlsx_rows(data):
|
||||
z = zipfile.ZipFile(io.BytesIO(data))
|
||||
strings = ["".join(t.text or "" for t in si.iter("{%s}t" % XLSX_NS["m"])) for si in ET.fromstring(z.read("xl/sharedStrings.xml")).findall("m:si", XLSX_NS)]
|
||||
sheet = ET.fromstring(z.read("xl/worksheets/sheet1.xml"))
|
||||
for row in sheet.findall(".//m:row", XLSX_NS):
|
||||
cells = []
|
||||
for c in row.findall("m:c", XLSX_NS):
|
||||
v = c.find("m:v", XLSX_NS)
|
||||
cells.append("" if v is None else strings[int(v.text)] if c.get("t") == "s" else v.text)
|
||||
yield cells
|
||||
|
||||
|
||||
def scb_codes(cache):
|
||||
regions, municipalities = {}, {}
|
||||
for cells in xlsx_rows(tsv.fetch(CODES, cache, "kommunlankod.xlsx", magic=b"PK")):
|
||||
for cells in tsv.xlsx_rows(tsv.fetch(CODES, cache, "kommunlankod.xlsx", magic=b"PK")):
|
||||
if len(cells) < 2 or not re.fullmatch(r"\d{2}|\d{4}", cells[0]):
|
||||
continue
|
||||
(regions if len(cells[0]) == 2 else municipalities)[cells[0]] = cells[1].strip()
|
||||
|
||||
@@ -0,0 +1,52 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Rebuild data/sv_SE/first-name.tsv and last-name.tsv from SCB's whole-population name counts (CC0, "Källa: SCB").
|
||||
|
||||
data-import/names-se.py [--source URL_OR_FILE] [--first N] [--last N] [--cache DIR] [--out DIR]
|
||||
|
||||
A tilltalsnamn goes under each sex that carries it, so a name both sexes carry has two rows.
|
||||
"""
|
||||
import argparse
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
import tsv
|
||||
|
||||
SOURCE = "https://www.scb.se/contentassets/9fe7dbb460994c72b835163dbc491ef9/namn-med-minst-tva-barare-31-december-2022.xlsx"
|
||||
OUT = Path(__file__).resolve().parent.parent / "data" / "sv_SE"
|
||||
CACHE = Path(__file__).resolve().parent / "cache"
|
||||
SHEETS = {"f": "Tilltalsnamn kvinnor", "m": "Tilltalsnamn män"}
|
||||
SURNAMES = "Efternamn"
|
||||
NAME = re.compile(r"^[^\W\d_]{2,}([ '-][^\W\d_]{2,})*$")
|
||||
PARTICLES = {"af", "av", "de", "den", "der", "di", "du", "la", "le", "van", "von"}
|
||||
|
||||
|
||||
def cased(name):
|
||||
"""SCB's uppercase name as it is written: each part capitalised, a particle before another part lowercased."""
|
||||
parts = name.lower().split(" ")
|
||||
return " ".join(w if w in PARTICLES and len(parts) > 1 else w.title() for w in parts)
|
||||
|
||||
|
||||
def counted(rows, top):
|
||||
"""The top names of a sheet by bearers, initials and single letters dropped."""
|
||||
names = [(cells[0], int(cells[1])) for cells in rows if len(cells) >= 2 and cells[1].isdigit() and NAME.match(cells[0])]
|
||||
return sorted(names, key=lambda n: (-n[1], n[0]))[:top]
|
||||
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||
p.add_argument("--cache", default=str(CACHE))
|
||||
p.add_argument("--first", type=int, default=2000, help="names per sex")
|
||||
p.add_argument("--last", type=int, default=5000)
|
||||
p.add_argument("--out", default=str(OUT))
|
||||
p.add_argument("--source", default=SOURCE)
|
||||
a = p.parse_args()
|
||||
data = tsv.fetch(a.source, a.cache, "scb-namn-2022.xlsx", magic=b"PK")
|
||||
first = [{"name": cased(n), "sex": sex, "count": c} for sex, sheet in SHEETS.items() for n, c in counted(tsv.xlsx_rows(data, sheet), a.first)]
|
||||
last = [{"name": cased(n), "count": c} for n, c in counted(tsv.xlsx_rows(data, SURNAMES), a.last)]
|
||||
out = Path(a.out)
|
||||
tsv.write(out / "first-name.tsv", ["name", "sex", "count"], sorted(first, key=lambda r: (r["name"], r["sex"])))
|
||||
tsv.write(out / "last-name.tsv", ["name", "count"], sorted(last, key=lambda r: r["name"]))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,77 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Rebuild data/en_US/first-name.tsv and last-name.tsv from SSA baby names and the Census 2010 surnames (public domain).
|
||||
|
||||
data-import/names-us.py [--names URL_OR_FILE] [--surnames URL_OR_FILE] [--from-year YEAR] [--first N] [--last N] [--cache DIR] [--out DIR]
|
||||
|
||||
A given name is counted over the births of --from-year and later, under each sex it was given to, so a
|
||||
name both sexes carry has two rows. --names takes SSA's names.zip or a merged name,sex,count,year file;
|
||||
--surnames takes the Census names.zip or its Names_2010Census.csv.
|
||||
"""
|
||||
import argparse
|
||||
import collections
|
||||
import csv
|
||||
import io
|
||||
import re
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
import tsv
|
||||
|
||||
NAMES = "https://raw.githubusercontent.com/hackerb9/ssa-baby-names/master/alldata.txt"
|
||||
SURNAMES = "https://www2.census.gov/topics/genealogy/2010surnames/names.zip"
|
||||
OUT = Path(__file__).resolve().parent.parent / "data" / "en_US"
|
||||
CACHE = Path(__file__).resolve().parent / "cache"
|
||||
MC = re.compile(r"^Mc([a-z])")
|
||||
|
||||
|
||||
def csv_rows(data, member):
|
||||
"""The rows of a CSV, or of every member of a zip named like member, name,sex,count[,year]."""
|
||||
if data.startswith(b"PK"):
|
||||
z = zipfile.ZipFile(io.BytesIO(data))
|
||||
for n in sorted(z.namelist()):
|
||||
if re.fullmatch(member, n):
|
||||
year = re.sub(r"\D", "", n)
|
||||
for r in csv.reader(io.StringIO(z.read(n).decode("utf-8-sig"))):
|
||||
yield r + [year] if year else r
|
||||
return
|
||||
yield from csv.reader(io.StringIO(data.decode("utf-8-sig")))
|
||||
|
||||
|
||||
def given(data, from_year):
|
||||
counts = collections.Counter()
|
||||
for r in csv_rows(data, r"yob\d{4}\.txt"):
|
||||
if len(r) >= 4 and r[3].isdigit() and int(r[3]) >= from_year:
|
||||
counts[(r[0], r[1].lower())] += int(r[2])
|
||||
return counts
|
||||
|
||||
|
||||
def surname(name):
|
||||
"""A Census uppercase surname as it is written: capitalised, and Mc before a capital."""
|
||||
return MC.sub(lambda m: "Mc" + m.group(1).upper(), name.title())
|
||||
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||
p.add_argument("--cache", default=str(CACHE))
|
||||
p.add_argument("--first", type=int, default=2000, help="names per sex")
|
||||
p.add_argument("--from-year", type=int, default=1930)
|
||||
p.add_argument("--last", type=int, default=5000)
|
||||
p.add_argument("--names", default=NAMES)
|
||||
p.add_argument("--out", default=str(OUT))
|
||||
p.add_argument("--surnames", default=SURNAMES)
|
||||
a = p.parse_args()
|
||||
counts = given(tsv.fetch(a.names, a.cache, "ssa-names.txt"), a.from_year)
|
||||
first = []
|
||||
for sex in ("f", "m"):
|
||||
top = sorted(((n, c) for (n, s), c in counts.items() if s == sex), key=lambda n: (-n[1], n[0]))[:a.first]
|
||||
first += [{"name": n, "sex": sex, "count": c} for n, c in top]
|
||||
rows = csv_rows(tsv.fetch(a.surnames, a.cache, "census-surnames-2010.zip"), r"Names_2010Census\.csv")
|
||||
last = [{"name": surname(r[0]), "count": int(r[2])} for r in rows if len(r) >= 3 and r[2].isdigit() and r[0].isalpha()]
|
||||
last = sorted(last, key=lambda r: (-r["count"], r["name"]))[:a.last]
|
||||
out = Path(a.out)
|
||||
tsv.write(out / "first-name.tsv", ["name", "sex", "count"], sorted(first, key=lambda r: (r["name"], r["sex"])))
|
||||
tsv.write(out / "last-name.tsv", ["name", "count"], sorted(last, key=lambda r: r["name"]))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,10 +1,15 @@
|
||||
"""A source fetched once into the cache, and a table written as the loader admits it."""
|
||||
import io
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
import urllib.request
|
||||
import xml.etree.ElementTree as ET
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
XLSX_NS = {"m": "http://schemas.openxmlformats.org/spreadsheetml/2006/main", "r": "http://schemas.openxmlformats.org/officeDocument/2006/relationships"}
|
||||
|
||||
|
||||
def fetch(source, cache, name, magic=b"", data=None, headers=None):
|
||||
"""The bytes of a URL, downloaded into cache/name once, or of a local file."""
|
||||
@@ -28,6 +33,21 @@ def fetch(source, cache, name, magic=b"", data=None, headers=None):
|
||||
sys.exit(f"{source}: no valid download in 5 attempts")
|
||||
|
||||
|
||||
def xlsx_rows(data, sheet=None):
|
||||
"""The rows of an xlsx sheet named sheet, the first sheet by default, each a list of cell texts."""
|
||||
z = zipfile.ZipFile(io.BytesIO(data))
|
||||
strings = ["".join(t.text or "" for t in si.iter("{%s}t" % XLSX_NS["m"])) for si in ET.fromstring(z.read("xl/sharedStrings.xml")).findall("m:si", XLSX_NS)]
|
||||
rels = {r.get("Id"): r.get("Target") for r in ET.fromstring(z.read("xl/_rels/workbook.xml.rels"))}
|
||||
sheets = {s.get("name"): rels[s.get("{%s}id" % XLSX_NS["r"])] for s in ET.fromstring(z.read("xl/workbook.xml")).iter("{%s}sheet" % XLSX_NS["m"])}
|
||||
target = sheets[sheet] if sheet else next(iter(sheets.values()))
|
||||
for row in ET.fromstring(z.read("xl/" + target)).findall(".//m:row", XLSX_NS):
|
||||
cells = []
|
||||
for c in row.findall("m:c", XLSX_NS):
|
||||
v = c.find("m:v", XLSX_NS)
|
||||
cells.append("" if v is None else strings[int(v.text)] if c.get("t") == "s" else v.text)
|
||||
yield cells
|
||||
|
||||
|
||||
def write(path, columns, rows):
|
||||
"""Write the rows as a TSV; every cell must be non-empty and free of tabs, newlines and braces."""
|
||||
lines = ["\t".join(columns)]
|
||||
|
||||
Reference in New Issue
Block a user