Weighted first and last names from SCB, the SSA and the Census under a sex table, personnummer, samordningsnummer, SSN and ITIN in their valid ranges, and date, time and datetime over date()

This commit is contained in:
2026-09-18 18:10:59 +02:00
parent 289d5edc31
commit a136be6059
28 changed files with 18217 additions and 124 deletions
+20
View File
@@ -1,10 +1,15 @@
"""A source fetched once into the cache, and a table written as the loader admits it."""
import io
import re
import sys
import time
import urllib.request
import xml.etree.ElementTree as ET
import zipfile
from pathlib import Path
XLSX_NS = {"m": "http://schemas.openxmlformats.org/spreadsheetml/2006/main", "r": "http://schemas.openxmlformats.org/officeDocument/2006/relationships"}
def fetch(source, cache, name, magic=b"", data=None, headers=None):
"""The bytes of a URL, downloaded into cache/name once, or of a local file."""
@@ -28,6 +33,21 @@ def fetch(source, cache, name, magic=b"", data=None, headers=None):
sys.exit(f"{source}: no valid download in 5 attempts")
def xlsx_rows(data, sheet=None):
"""The rows of an xlsx sheet named sheet, the first sheet by default, each a list of cell texts."""
z = zipfile.ZipFile(io.BytesIO(data))
strings = ["".join(t.text or "" for t in si.iter("{%s}t" % XLSX_NS["m"])) for si in ET.fromstring(z.read("xl/sharedStrings.xml")).findall("m:si", XLSX_NS)]
rels = {r.get("Id"): r.get("Target") for r in ET.fromstring(z.read("xl/_rels/workbook.xml.rels"))}
sheets = {s.get("name"): rels[s.get("{%s}id" % XLSX_NS["r"])] for s in ET.fromstring(z.read("xl/workbook.xml")).iter("{%s}sheet" % XLSX_NS["m"])}
target = sheets[sheet] if sheet else next(iter(sheets.values()))
for row in ET.fromstring(z.read("xl/" + target)).findall(".//m:row", XLSX_NS):
cells = []
for c in row.findall("m:c", XLSX_NS):
v = c.find("m:v", XLSX_NS)
cells.append("" if v is None else strings[int(v.text)] if c.get("t") == "s" else v.text)
yield cells
def write(path, columns, rows):
"""Write the rows as a TSV; every cell must be non-empty and free of tabs, newlines and braces."""
lines = ["\t".join(columns)]