Weighted person names, valid personal ids and date() in both locales #21

Merged
lilleman merged 38 commits from person-ids-date into main 2026-09-18 20:45:51 +02:00
28 changed files with 18217 additions and 124 deletions
Showing only changes of commit a136be6059 - Show all commits
+1 -17
View File
@@ -13,7 +13,6 @@ import os
import re
import sys
import urllib.request
import xml.etree.ElementTree as ET
import zipfile
from pathlib import Path
@@ -38,24 +37,9 @@ CACHE = Path(__file__).resolve().parent / "cache"
TIMEZONE = "Europe/Stockholm"
ONE_POSITION = {"Stockholm", "Göteborg", "Malmö"}
UNMATCHED_POPULATION = 200
XLSX_NS = {"m": "http://schemas.openxmlformats.org/spreadsheetml/2006/main"}
def xlsx_rows(data):
z = zipfile.ZipFile(io.BytesIO(data))
strings = ["".join(t.text or "" for t in si.iter("{%s}t" % XLSX_NS["m"])) for si in ET.fromstring(z.read("xl/sharedStrings.xml")).findall("m:si", XLSX_NS)]
sheet = ET.fromstring(z.read("xl/worksheets/sheet1.xml"))
for row in sheet.findall(".//m:row", XLSX_NS):
cells = []
for c in row.findall("m:c", XLSX_NS):
v = c.find("m:v", XLSX_NS)
cells.append("" if v is None else strings[int(v.text)] if c.get("t") == "s" else v.text)
yield cells
def scb_codes(cache):
regions, municipalities = {}, {}
for cells in xlsx_rows(tsv.fetch(CODES, cache, "kommunlankod.xlsx", magic=b"PK")):
for cells in tsv.xlsx_rows(tsv.fetch(CODES, cache, "kommunlankod.xlsx", magic=b"PK")):
if len(cells) < 2 or not re.fullmatch(r"\d{2}|\d{4}", cells[0]):
continue
(regions if len(cells[0]) == 2 else municipalities)[cells[0]] = cells[1].strip()
+52
View File
@@ -0,0 +1,52 @@
#!/usr/bin/env python3
"""Rebuild data/sv_SE/first-name.tsv and last-name.tsv from SCB's whole-population name counts (CC0, "Källa: SCB").
data-import/names-se.py [--source URL_OR_FILE] [--first N] [--last N] [--cache DIR] [--out DIR]
A tilltalsnamn goes under each sex that carries it, so a name both sexes carry has two rows.
"""
import argparse
import re
from pathlib import Path
import tsv
SOURCE = "https://www.scb.se/contentassets/9fe7dbb460994c72b835163dbc491ef9/namn-med-minst-tva-barare-31-december-2022.xlsx"
OUT = Path(__file__).resolve().parent.parent / "data" / "sv_SE"
CACHE = Path(__file__).resolve().parent / "cache"
SHEETS = {"f": "Tilltalsnamn kvinnor", "m": "Tilltalsnamn män"}
SURNAMES = "Efternamn"
NAME = re.compile(r"^[^\W\d_]{2,}([ '-][^\W\d_]{2,})*$")
PARTICLES = {"af", "av", "de", "den", "der", "di", "du", "la", "le", "van", "von"}
def cased(name):
"""SCB's uppercase name as it is written: each part capitalised, a particle before another part lowercased."""
parts = name.lower().split(" ")
return " ".join(w if w in PARTICLES and len(parts) > 1 else w.title() for w in parts)
def counted(rows, top):
"""The top names of a sheet by bearers, initials and single letters dropped."""
names = [(cells[0], int(cells[1])) for cells in rows if len(cells) >= 2 and cells[1].isdigit() and NAME.match(cells[0])]
return sorted(names, key=lambda n: (-n[1], n[0]))[:top]
def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--cache", default=str(CACHE))
p.add_argument("--first", type=int, default=2000, help="names per sex")
p.add_argument("--last", type=int, default=5000)
p.add_argument("--out", default=str(OUT))
p.add_argument("--source", default=SOURCE)
a = p.parse_args()
data = tsv.fetch(a.source, a.cache, "scb-namn-2022.xlsx", magic=b"PK")
first = [{"name": cased(n), "sex": sex, "count": c} for sex, sheet in SHEETS.items() for n, c in counted(tsv.xlsx_rows(data, sheet), a.first)]
last = [{"name": cased(n), "count": c} for n, c in counted(tsv.xlsx_rows(data, SURNAMES), a.last)]
out = Path(a.out)
tsv.write(out / "first-name.tsv", ["name", "sex", "count"], sorted(first, key=lambda r: (r["name"], r["sex"])))
tsv.write(out / "last-name.tsv", ["name", "count"], sorted(last, key=lambda r: r["name"]))
if __name__ == "__main__":
main()
+77
View File
@@ -0,0 +1,77 @@
#!/usr/bin/env python3
"""Rebuild data/en_US/first-name.tsv and last-name.tsv from SSA baby names and the Census 2010 surnames (public domain).
data-import/names-us.py [--names URL_OR_FILE] [--surnames URL_OR_FILE] [--from-year YEAR] [--first N] [--last N] [--cache DIR] [--out DIR]
A given name is counted over the births of --from-year and later, under each sex it was given to, so a
name both sexes carry has two rows. --names takes SSA's names.zip or a merged name,sex,count,year file;
--surnames takes the Census names.zip or its Names_2010Census.csv.
"""
import argparse
import collections
import csv
import io
import re
import zipfile
from pathlib import Path
import tsv
NAMES = "https://raw.githubusercontent.com/hackerb9/ssa-baby-names/master/alldata.txt"
SURNAMES = "https://www2.census.gov/topics/genealogy/2010surnames/names.zip"
OUT = Path(__file__).resolve().parent.parent / "data" / "en_US"
CACHE = Path(__file__).resolve().parent / "cache"
MC = re.compile(r"^Mc([a-z])")
def csv_rows(data, member):
"""The rows of a CSV, or of every member of a zip named like member, name,sex,count[,year]."""
if data.startswith(b"PK"):
z = zipfile.ZipFile(io.BytesIO(data))
for n in sorted(z.namelist()):
if re.fullmatch(member, n):
year = re.sub(r"\D", "", n)
for r in csv.reader(io.StringIO(z.read(n).decode("utf-8-sig"))):
yield r + [year] if year else r
return
yield from csv.reader(io.StringIO(data.decode("utf-8-sig")))
def given(data, from_year):
counts = collections.Counter()
for r in csv_rows(data, r"yob\d{4}\.txt"):
if len(r) >= 4 and r[3].isdigit() and int(r[3]) >= from_year:
counts[(r[0], r[1].lower())] += int(r[2])
return counts
def surname(name):
"""A Census uppercase surname as it is written: capitalised, and Mc before a capital."""
return MC.sub(lambda m: "Mc" + m.group(1).upper(), name.title())
def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--cache", default=str(CACHE))
p.add_argument("--first", type=int, default=2000, help="names per sex")
p.add_argument("--from-year", type=int, default=1930)
p.add_argument("--last", type=int, default=5000)
p.add_argument("--names", default=NAMES)
p.add_argument("--out", default=str(OUT))
p.add_argument("--surnames", default=SURNAMES)
a = p.parse_args()
counts = given(tsv.fetch(a.names, a.cache, "ssa-names.txt"), a.from_year)
first = []
for sex in ("f", "m"):
top = sorted(((n, c) for (n, s), c in counts.items() if s == sex), key=lambda n: (-n[1], n[0]))[:a.first]
first += [{"name": n, "sex": sex, "count": c} for n, c in top]
rows = csv_rows(tsv.fetch(a.surnames, a.cache, "census-surnames-2010.zip"), r"Names_2010Census\.csv")
last = [{"name": surname(r[0]), "count": int(r[2])} for r in rows if len(r) >= 3 and r[2].isdigit() and r[0].isalpha()]
last = sorted(last, key=lambda r: (-r["count"], r["name"]))[:a.last]
out = Path(a.out)
tsv.write(out / "first-name.tsv", ["name", "sex", "count"], sorted(first, key=lambda r: (r["name"], r["sex"])))
tsv.write(out / "last-name.tsv", ["name", "count"], sorted(last, key=lambda r: r["name"]))
if __name__ == "__main__":
main()
+20
View File
@@ -1,10 +1,15 @@
"""A source fetched once into the cache, and a table written as the loader admits it."""
import io
import re
import sys
import time
import urllib.request
import xml.etree.ElementTree as ET
import zipfile
from pathlib import Path
XLSX_NS = {"m": "http://schemas.openxmlformats.org/spreadsheetml/2006/main", "r": "http://schemas.openxmlformats.org/officeDocument/2006/relationships"}
def fetch(source, cache, name, magic=b"", data=None, headers=None):
"""The bytes of a URL, downloaded into cache/name once, or of a local file."""
@@ -28,6 +33,21 @@ def fetch(source, cache, name, magic=b"", data=None, headers=None):
sys.exit(f"{source}: no valid download in 5 attempts")
def xlsx_rows(data, sheet=None):
"""The rows of an xlsx sheet named sheet, the first sheet by default, each a list of cell texts."""
z = zipfile.ZipFile(io.BytesIO(data))
strings = ["".join(t.text or "" for t in si.iter("{%s}t" % XLSX_NS["m"])) for si in ET.fromstring(z.read("xl/sharedStrings.xml")).findall("m:si", XLSX_NS)]
rels = {r.get("Id"): r.get("Target") for r in ET.fromstring(z.read("xl/_rels/workbook.xml.rels"))}
sheets = {s.get("name"): rels[s.get("{%s}id" % XLSX_NS["r"])] for s in ET.fromstring(z.read("xl/workbook.xml")).iter("{%s}sheet" % XLSX_NS["m"])}
target = sheets[sheet] if sheet else next(iter(sheets.values()))
for row in ET.fromstring(z.read("xl/" + target)).findall(".//m:row", XLSX_NS):
cells = []
for c in row.findall("m:c", XLSX_NS):
v = c.find("m:v", XLSX_NS)
cells.append("" if v is None else strings[int(v.text)] if c.get("t") == "s" else v.text)
yield cells
def write(path, columns, rows):
"""Write the rows as a TSV; every cell must be non-empty and free of tabs, newlines and braces."""
lines = ["\t".join(columns)]
+1 -13
View File
@@ -1,13 +1 @@
{
"format": "{month}/{day}/{year}",
"month": ["01", "02", "03", "04", "05", "06", "07", "08", "09", "10", "11", "12"],
"day": ["01", "02", "03", "04", "05", "06", "07", "08", "09", "10", "11", "12", "13", "14", "15", "16", "17", "18", "19", "20", "21", "22", "23", "24", "25", "26", "27", "28"],
"year": [
"197{digits(1)}",
"198{digits(1)}",
{ "format": "199{digits(1)}", "weight": 2 },
{ "format": "200{digits(1)}", "weight": 3 },
{ "format": "201{digits(1)}", "weight": 3 },
{ "format": "202{digits(1)}", "weight": 2 }
]
}
"{date(1970-01-01,2029-12-31,'01/02/2006')}"
+1
View File
@@ -0,0 +1 @@
{ "format": "{name}", "rows": "first-name.tsv", "name": "name", "parent": "sex", "weight": "count" }
File diff suppressed because it is too large Load Diff
+15
View File
@@ -0,0 +1,15 @@
{
"format": "9{digits(2)}-{group}-{serial}",
"group": [
{ "format": "{int(50,65)}", "weight": 16 },
{ "format": "{int(70,88)}", "weight": 19 },
{ "format": "{int(90,92)}", "weight": 3 },
{ "format": "{int(94,99)}", "weight": 6 }
],
"serial": [
{ "format": "000{int(1,9)}", "weight": 9 },
{ "format": "00{int(10,99)}", "weight": 90 },
{ "format": "0{int(100,999)}", "weight": 900 },
{ "format": "{int(1000,9999)}", "weight": 9000 }
]
}
+1
View File
@@ -0,0 +1 @@
{ "format": "{name}", "rows": "last-name.tsv", "key": "name", "weight": "count" }
File diff suppressed because it is too large Load Diff
+5 -5
View File
@@ -1,7 +1,7 @@
{
"format": "{prefix}{femalefirst|malefirst} {last}",
"femalefirst": ["Abigail", "Addison", "Amelia", "Aria", "Aubrey", "Audrey", "Aurora", "Ava", "Bella", "Brooklyn", "Camila", "Caroline", "Charlotte", "Chloe", "Claire", "Eleanor", "Elizabeth", "Ella", "Ellie", "Emily", "Emma", "Evelyn", "Gianna", "Grace", "Hannah", "Harper", "Hazel", "Isabella", "Layla", "Leah", "Lillian", "Lily", "Lucy", "Luna", "Madison", "Maya", "Mia", "Mila", "Naomi", "Natalie", "Nora", "Olivia", "Paisley", "Penelope", "Riley", "Savannah", "Scarlett", "Sofia", "Sophia", "Stella", "Victoria", "Violet", "Zoe"],
"malefirst": ["Aiden", "Alexander", "Andrew", "Anthony", "Asher", "Benjamin", "Caleb", "Carter", "Charles", "Christopher", "Daniel", "David", "Dylan", "Elijah", "Ethan", "Ezra", "Gabriel", "Grayson", "Henry", "Isaac", "Jack", "Jackson", "Jacob", "James", "Jayden", "John", "Joseph", "Joshua", "Julian", "Levi", "Liam", "Lincoln", "Logan", "Lucas", "Luke", "Mason", "Mateo", "Matthew", "Michael", "Nathan", "Noah", "Oliver", "Owen", "Samuel", "Sebastian", "Theodore", "Thomas", "William", "Wyatt"],
"last": ["Adams", "Allen", "Anderson", "Bailey", "Baker", "Bell", "Bennett", "Brooks", "Brown", "Campbell", "Carter", "Clark", "Collins", "Cook", "Cooper", "Cox", "Davis", "Edwards", "Evans", "Flores", "Foster", "Garcia", "Gonzalez", "Gray", "Green", "Hall", "Harris", "Hayes", "Hernandez", "Hill", "Howard", "Hughes", "Jackson", "James", "Jenkins", "Johnson", "Jones", "Kelly", "King", "Lee", "Lewis", "Long", "Lopez", "Martin", "Martinez", "Miller", "Mitchell", "Moore", "Morgan", "Morris", "Murphy", "Nelson", "Parker", "Perez", "Perry", "Peterson", "Phillips", "Powell", "Price", "Ramirez", "Reed", "Richardson", "Rivera", "Roberts", "Robinson", "Rodriguez", "Rogers", "Ross", "Russell", "Sanchez", "Sanders", "Scott", "Smith", "Stewart", "Sullivan", "Taylor", "Thomas", "Thompson", "Torres", "Turner", "Walker", "Ward", "Watson", "White", "Williams", "Wilson", "Wood", "Wright", "Young"],
"prefix": ["", { "format": "{title} ", "title": ["Dr", "Miss", "Mr", "Mrs", "Ms", "Mx", "Prof"], "weight": 0.1 }]
"format": "{prefix}{first} {last}",
"first": "{.first-name.name}",
"last": "{.last-name.name}",
"prefix": ["", { "format": "{title} ", "title": ["Dr", "Miss", "Mr", "Mrs", "Ms", "Mx", "Prof"], "weight": 0.1 }],
"sex": "{.sex.name}"
}
+1
View File
@@ -0,0 +1 @@
{ "format": "{name}", "rows": "sex.tsv", "key": "code", "name": "name" }
+3
View File
@@ -0,0 +1,3 @@
code name
f female
m male
1 code name
2 f female
3 m male
+19 -1
View File
@@ -1 +1,19 @@
"{int(100,999)}-{digits(2)}-{digits(4)}"
{
"format": "{area}-{group}-{serial}",
"area": [
{ "format": "00{int(1,9)}", "weight": 9 },
{ "format": "0{int(10,99)}", "weight": 90 },
{ "format": "{int(100,665)}", "weight": 566 },
{ "format": "{int(667,899)}", "weight": 233 }
],
"group": [
{ "format": "0{int(1,9)}", "weight": 9 },
{ "format": "{int(10,99)}", "weight": 90 }
],
"serial": [
{ "format": "000{int(1,9)}", "weight": 9 },
{ "format": "00{int(10,99)}", "weight": 90 },
{ "format": "0{int(100,999)}", "weight": 900 },
{ "format": "{int(1000,9999)}", "weight": 9000 }
]
}
+1 -6
View File
@@ -1,6 +1 @@
{
"format": "{hour}:{minute} {ampm}",
"hour": ["1", "2", "3", "4", "5", "6", "7", "8", "9", "10", "11", "12"],
"minute": { "format": "{t}{digits(1)}", "t": ["0", "1", "2", "3", "4", "5"] },
"ampm": ["AM", "PM"]
}
"{time('3:04 PM')}"
+1
View File
@@ -0,0 +1 @@
"{date(2000-01-01,2029-12-31,'2006-01-02T15:04:05Z')}"
+1 -13
View File
@@ -1,13 +1 @@
{
"format": "{year}-{month}-{day}",
"month": ["01", "02", "03", "04", "05", "06", "07", "08", "09", "10", "11", "12"],
"day": ["01", "02", "03", "04", "05", "06", "07", "08", "09", "10", "11", "12", "13", "14", "15", "16", "17", "18", "19", "20", "21", "22", "23", "24", "25", "26", "27", "28"],
"year": [
"197{digits(1)}",
"198{digits(1)}",
{ "format": "199{digits(1)}", "weight": 2 },
{ "format": "200{digits(1)}", "weight": 3 },
{ "format": "201{digits(1)}", "weight": 3 },
{ "format": "202{digits(1)}", "weight": 2 }
]
}
"{date(1970-01-01,2029-12-31,'2006-01-02')}"
+1
View File
@@ -0,0 +1 @@
{ "format": "{name}", "rows": "first-name.tsv", "name": "name", "parent": "sex", "weight": "count" }
File diff suppressed because it is too large Load Diff
+1
View File
@@ -0,0 +1 @@
{ "format": "{name}", "rows": "last-name.tsv", "key": "name", "weight": "count" }
File diff suppressed because it is too large Load Diff
+5 -30
View File
@@ -1,32 +1,7 @@
{
"format": "{prefix}{femalefirst|malefirst} {last}",
"femalefirst": ["Agnes", "Alice", "Alicia", "Alma", "Amanda", "Anna", "Astrid", "Cornelia", "Ebba", "Elin", "Ella", "Ellen", "Elsa", "Emilia", "Emma", "Ester", "Eva", "Frida", "Hanna", "Hedda", "Ida", "Ingrid", "Isabelle", "Julia", "Karin", "Klara", "Kristina", "Lena", "Linnéa", "Lova", "Maja", "Maria", "Moa", "Molly", "Märta", "Nellie", "Nora", "Olivia", "Saga", "Sara", "Selma", "Signe", "Siri", "Sofia", "Stina", "Tilde", "Tuva", "Wilma", "Ylva", "Åsa"],
"malefirst": ["Adam", "Albin", "Alexander", "Alfred", "Anders", "Anton", "Arvid", "Axel", "Bengt", "Bo", "Carl", "David", "Edvin", "Elias", "Emil", "Erik", "Filip", "Folke", "Fredrik", "Gustav", "Göran", "Hampus", "Hans", "Henrik", "Hugo", "Isak", "Johan", "Jonas", "Karl", "Kjell", "Lars", "Leo", "Linus", "Love", "Magnus", "Mats", "Mattias", "Mikael", "Nils", "Olle", "Oskar", "Otto", "Patrik", "Per", "Pontus", "Rasmus", "Samuel", "Sixten", "Stefan", "Sten", "Sven", "Theodor", "Tobias", "Viktor", "William", "Åke"],
"last": [
{
"format": "{first}sson",
"weight": 5,
"first": ["Ander", "Arvid", "Bengt", "Daniel", "David", "Erik", "Fredrik", "Gustaf", "Henrik", "Håkan", "Isak", "Jakob", "Johan", "Jon", "Jön", "Karl", "Lar", "Magnu", "Matt", "Mikael", "Mårten", "Nil", "Ol", "Per", "Petter", "Samuel", "Sven"]
},
{
"format": "{first}{last}",
"weight": 3.6,
"first": ["Ahl", "Berg", "Blom", "Ceder", "Dahl", "Ek", "Eng", "Falk", "Fors", "Gran", "Hag", "Hed", "Holm", "Lind", "Lund", "Mal", "Ny", "Rosen", "Sand", "Sjö", "Skog", "Söder", "Sten", "Strand", "Sund", "Wik", "Öst"],
"last": ["berg", "blad", "crona", "dahl", "ed", "fors", "gren", "holm", "in", "kvist", "löf", "lund", "man", "mark", "qvist", "roth", "stedt", "sten", "strand", "ström", "vall"]
},
{
"format": "{first}{last}",
"weight": 0.267,
"first": ["Hell", "Wall"],
"last": ["berg", "blad", "crona", "dahl", "ed", "fors", "gren", "holm", "in", "kvist", "man", "mark", "qvist", "roth", "stedt", "sten", "strand", "ström", "vall"]
},
{
"format": "{first}{last}",
"weight": 0.133,
"first": "Norr",
"last": ["berg", "blad", "crona", "dahl", "ed", "fors", "gren", "holm", "in", "kvist", "löf", "lund", "man", "mark", "qvist", "stedt", "sten", "strand", "ström", "vall"]
},
["Berg", "Blom", "Eismar", "Falk", "Holm", "Lind", "Norberg", "Strand", "Ström", "von Flemming", "Åberg", "Öberg"]
],
"prefix": ["", { "format": "{string} ", "string": ["dr", "prof"], "weight": 0.05 }]
"format": "{prefix}{first} {last}",
"first": "{.first-name.name}",
"last": "{.last-name.name}",
"prefix": ["", { "format": "{string} ", "string": ["dr", "prof"], "weight": 0.05 }],
"sex": "{.sex.name}"
}
+1
View File
@@ -0,0 +1 @@
"{date(1930-01-01,2025-12-31,'060102')}-{.sex.birth-number}{luhn()}"
+1
View File
@@ -0,0 +1 @@
"{date(1930-01-01,2025-12-31,'0601')}{int(61,88)}-{.sex.birth-number}{luhn()}"
+1
View File
@@ -0,0 +1 @@
{ "format": "{name}", "rows": "sex.tsv", "key": "code", "name": "name" }
+3
View File
@@ -0,0 +1,3 @@
code name birth-number
f kvinna 238
m man 239
1 code name birth-number
2 f kvinna 238
3 m man 239
-22
View File
@@ -1,22 +0,0 @@
{
"format": "{digits(2)}{mmdd}-{digits(3)}{luhn()}",
"mmdd": [
{
"format": "{m}{d}",
"weight": 7,
"m": ["01", "03", "05", "07", "08", "10", "12"],
"d": ["01", "02", "03", "04", "05", "06", "07", "08", "09", "10", "11", "12", "13", "14", "15", "16", "17", "18", "19", "20", "21", "22", "23", "24", "25", "26", "27", "28", "29", "30", "31"]
},
{
"format": "{m}{d}",
"weight": 4,
"m": ["04", "06", "09", "11"],
"d": ["01", "02", "03", "04", "05", "06", "07", "08", "09", "10", "11", "12", "13", "14", "15", "16", "17", "18", "19", "20", "21", "22", "23", "24", "25", "26", "27", "28", "29", "30"]
},
{
"format": "{m}{d}",
"m": "02",
"d": ["01", "02", "03", "04", "05", "06", "07", "08", "09", "10", "11", "12", "13", "14", "15", "16", "17", "18", "19", "20", "21", "22", "23", "24", "25", "26", "27", "28"]
}
]
}
+1 -17
View File
@@ -1,17 +1 @@
{
"format": "{hour}:{minute}{sec}",
"hour": [
{ "format": "0{digits(1)}", "weight": 4 },
{ "format": "1{digits(1)}", "weight": 4 },
{ "format": "2{t}", "t": ["0", "1", "2", "3"], "weight": 2 }
],
"minute": { "format": "{t}{digits(1)}", "t": ["0", "1", "2", "3", "4", "5"] },
"sec": [
"",
{
"format": ":{s}",
"s": { "format": "{t}{digits(1)}", "t": ["0", "1", "2", "3", "4", "5"] },
"weight": 0.4
}
]
}
"{time('15:04')}"