Move the birth number into its own table under sex, drop the sexed US titles, and split the import helper by content
This commit is contained in:
@@ -9,6 +9,7 @@ import io
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
import source
|
||||
import tsv
|
||||
|
||||
SOURCE = "https://raw.githubusercontent.com/datasets/country-codes/main/data/country-codes.csv"
|
||||
@@ -58,7 +59,7 @@ def main():
|
||||
p.add_argument("--source", default=SOURCE)
|
||||
p.add_argument("--out", default=str(OUT))
|
||||
a = p.parse_args()
|
||||
table = rows(tsv.fetch(a.source, a.cache, "country-codes.csv").decode("utf-8"))
|
||||
table = rows(source.fetch(a.source, a.cache, "country-codes.csv").decode("utf-8"))
|
||||
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["alpha2"]))
|
||||
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ import io
|
||||
import xml.etree.ElementTree as ET
|
||||
from pathlib import Path
|
||||
|
||||
import source
|
||||
import tsv
|
||||
|
||||
SOURCE = "https://raw.githubusercontent.com/datasets/currency-codes/main/data/codes-all.csv"
|
||||
@@ -58,8 +59,8 @@ def main():
|
||||
p.add_argument("--symbols", nargs="+", default=SYMBOLS)
|
||||
p.add_argument("--out", default=str(OUT))
|
||||
a = p.parse_args()
|
||||
symbol = symbols(tsv.fetch(s, a.cache, Path(s).name).decode("utf-8") for s in a.symbols)
|
||||
table = rows(tsv.fetch(a.source, a.cache, "codes-all.csv").decode("utf-8"), symbol)
|
||||
symbol = symbols(source.fetch(s, a.cache, Path(s).name).decode("utf-8") for s in a.symbols)
|
||||
table = rows(source.fetch(a.source, a.cache, "codes-all.csv").decode("utf-8"), symbol)
|
||||
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["code"]))
|
||||
|
||||
|
||||
|
||||
@@ -16,6 +16,8 @@ import urllib.request
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
import source
|
||||
import xlsx
|
||||
import tsv
|
||||
|
||||
CODES = "https://www.scb.se/contentassets/7a89e48960f741e08918e489ea36354a/kommunlankod-2026.xlsx"
|
||||
@@ -39,7 +41,7 @@ ONE_POSITION = {"Stockholm", "Göteborg", "Malmö"}
|
||||
UNMATCHED_POPULATION = 200
|
||||
def scb_codes(cache):
|
||||
regions, municipalities = {}, {}
|
||||
for cells in tsv.xlsx_rows(tsv.fetch(CODES, cache, "kommunlankod.xlsx", magic=b"PK")):
|
||||
for cells in xlsx.rows(source.fetch(CODES, cache, "kommunlankod.xlsx", magic=b"PK")):
|
||||
if len(cells) < 2 or not re.fullmatch(r"\d{2}|\d{4}", cells[0]):
|
||||
continue
|
||||
(regions if len(cells[0]) == 2 else municipalities)[cells[0]] = cells[1].strip()
|
||||
@@ -48,12 +50,12 @@ def scb_codes(cache):
|
||||
|
||||
def scb_population(cache):
|
||||
body = json.dumps(POPULATION_QUERY).encode()
|
||||
data = tsv.fetch(POPULATION, cache, "befolkning.json", data=body, headers={"Content-Type": "application/json"})
|
||||
data = source.fetch(POPULATION, cache, "befolkning.json", data=body, headers={"Content-Type": "application/json"})
|
||||
return {row["key"][0]: row["values"][0] for row in json.loads(data.decode("utf-8-sig"))["data"]}
|
||||
|
||||
|
||||
def scb_tatorter(cache):
|
||||
text = tsv.fetch(TATORTER, cache, "tatorter.csv").decode("utf-8")
|
||||
text = source.fetch(TATORTER, cache, "tatorter.csv").decode("utf-8")
|
||||
by_name = collections.defaultdict(list)
|
||||
for r in csv.DictReader(io.StringIO(text)):
|
||||
by_name[r["tatort"]].append((r["kommun"], int(r["bef"])))
|
||||
@@ -61,7 +63,7 @@ def scb_tatorter(cache):
|
||||
|
||||
|
||||
def geonames(cache):
|
||||
z = zipfile.ZipFile(io.BytesIO(tsv.fetch(POSTAL_CODES, cache, "SE.zip", magic=b"PK")))
|
||||
z = zipfile.ZipFile(io.BytesIO(source.fetch(POSTAL_CODES, cache, "SE.zip", magic=b"PK")))
|
||||
rows = []
|
||||
for line in z.read("SE.txt").decode("utf-8").splitlines():
|
||||
f = line.split("\t")
|
||||
|
||||
@@ -14,6 +14,7 @@ import sys
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
import source
|
||||
import tsv
|
||||
|
||||
GAZETTEER = "https://www2.census.gov/geo/docs/maps-data/data/gazetteer/2026_Gazetteer/2026_Gaz_{}_national.zip"
|
||||
@@ -56,14 +57,14 @@ def text(data):
|
||||
|
||||
|
||||
def gazetteer(cache, kind):
|
||||
z = zipfile.ZipFile(io.BytesIO(tsv.fetch(GAZETTEER.format(kind), cache, f"gaz_{kind}.zip", magic=b"PK")))
|
||||
z = zipfile.ZipFile(io.BytesIO(source.fetch(GAZETTEER.format(kind), cache, f"gaz_{kind}.zip", magic=b"PK")))
|
||||
rows = text(z.read(z.namelist()[0])).splitlines()
|
||||
header = [h.strip() for h in rows[0].split("|")]
|
||||
return [dict(zip(header, (c.strip() for c in row.split("|")))) for row in rows[1:]]
|
||||
|
||||
|
||||
def csv_rows(cache, url, name):
|
||||
return list(csv.DictReader(io.StringIO(text(tsv.fetch(url, cache, name)))))
|
||||
return list(csv.DictReader(io.StringIO(text(source.fetch(url, cache, name)))))
|
||||
|
||||
|
||||
def dbf_rows(data, wanted):
|
||||
@@ -89,7 +90,7 @@ def dbf_rows(data, wanted):
|
||||
|
||||
|
||||
def tiger_zip(cache, kind, county):
|
||||
return tsv.fetch(TIGER.format(kind.upper(), county, kind), cache, f"tl_{county}_{kind}.zip", magic=b"PK")
|
||||
return source.fetch(TIGER.format(kind.upper(), county, kind), cache, f"tl_{county}_{kind}.zip", magic=b"PK")
|
||||
|
||||
|
||||
def tiger(cache, kind, county, wanted):
|
||||
@@ -130,7 +131,7 @@ def localities(cache, min_population, counties):
|
||||
def postal_codes(cache, localities):
|
||||
"""Each ZCTA whose largest part inside an incorporated place lies in a shipped place."""
|
||||
parts = {}
|
||||
for r in csv.DictReader(io.StringIO(text(tsv.fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"):
|
||||
for r in csv.DictReader(io.StringIO(text(source.fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"):
|
||||
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"] and not r["NAMELSAD_PLACE_20"].endswith(" CDP"):
|
||||
parts.setdefault(r["GEOID_ZCTA5_20"], []).append((int(r["AREALAND_PART"]), r["GEOID_PLACE_20"]))
|
||||
largest = {zcta: max(p)[1] for zcta, p in parts.items()}
|
||||
|
||||
@@ -9,6 +9,8 @@ import argparse
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
import source
|
||||
import xlsx
|
||||
import tsv
|
||||
|
||||
SOURCE = "https://www.scb.se/contentassets/9fe7dbb460994c72b835163dbc491ef9/namn-med-minst-tva-barare-31-december-2022.xlsx"
|
||||
@@ -40,9 +42,9 @@ def main():
|
||||
p.add_argument("--out", default=str(OUT))
|
||||
p.add_argument("--source", default=SOURCE)
|
||||
a = p.parse_args()
|
||||
data = tsv.fetch(a.source, a.cache, "scb-namn-2022.xlsx", magic=b"PK")
|
||||
first = [{"name": cased(n), "sex": sex, "count": c} for sex, sheet in SHEETS.items() for n, c in counted(tsv.xlsx_rows(data, sheet), a.first)]
|
||||
last = [{"name": cased(n), "count": c} for n, c in counted(tsv.xlsx_rows(data, SURNAMES), a.last)]
|
||||
data = source.fetch(a.source, a.cache, "scb-namn-2022.xlsx", magic=b"PK")
|
||||
first = [{"name": cased(n), "sex": sex, "count": c} for sex, sheet in SHEETS.items() for n, c in counted(xlsx.rows(data, sheet), a.first)]
|
||||
last = [{"name": cased(n), "count": c} for n, c in counted(xlsx.rows(data, SURNAMES), a.last)]
|
||||
out = Path(a.out)
|
||||
tsv.write(out / "first-name.tsv", ["name", "sex", "count"], sorted(first, key=lambda r: (r["name"], r["sex"])))
|
||||
tsv.write(out / "last-name.tsv", ["name", "count"], sorted(last, key=lambda r: r["name"]))
|
||||
|
||||
@@ -15,6 +15,7 @@ import re
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
import source
|
||||
import tsv
|
||||
|
||||
NAMES = "https://raw.githubusercontent.com/hackerb9/ssa-baby-names/master/alldata.txt"
|
||||
@@ -24,7 +25,7 @@ CACHE = Path(__file__).resolve().parent / "cache"
|
||||
MC = re.compile(r"^Mc([a-z])")
|
||||
|
||||
|
||||
def csv_rows(data, member):
|
||||
def csv_or_zip_rows(data, member):
|
||||
"""The rows of a CSV, or of every member of a zip named like member, name,sex,count[,year]."""
|
||||
if data.startswith(b"PK"):
|
||||
z = zipfile.ZipFile(io.BytesIO(data))
|
||||
@@ -39,7 +40,7 @@ def csv_rows(data, member):
|
||||
|
||||
def given(data, from_year):
|
||||
counts = collections.Counter()
|
||||
for r in csv_rows(data, r"yob\d{4}\.txt"):
|
||||
for r in csv_or_zip_rows(data, r"yob\d{4}\.txt"):
|
||||
if len(r) >= 4 and r[3].isdigit() and int(r[3]) >= from_year:
|
||||
counts[(r[0], r[1].lower())] += int(r[2])
|
||||
return counts
|
||||
@@ -60,12 +61,12 @@ def main():
|
||||
p.add_argument("--out", default=str(OUT))
|
||||
p.add_argument("--surnames", default=SURNAMES)
|
||||
a = p.parse_args()
|
||||
counts = given(tsv.fetch(a.names, a.cache, "ssa-names.txt"), a.from_year)
|
||||
counts = given(source.fetch(a.names, a.cache, "ssa-names.txt"), a.from_year)
|
||||
first = []
|
||||
for sex in ("f", "m"):
|
||||
top = sorted(((n, c) for (n, s), c in counts.items() if s == sex), key=lambda n: (-n[1], n[0]))[:a.first]
|
||||
first += [{"name": n, "sex": sex, "count": c} for n, c in top]
|
||||
rows = csv_rows(tsv.fetch(a.surnames, a.cache, "census-surnames-2010.zip"), r"Names_2010Census\.csv")
|
||||
rows = csv_or_zip_rows(source.fetch(a.surnames, a.cache, "census-surnames-2010.zip"), r"Names_2010Census\.csv")
|
||||
last = [{"name": surname(r[0]), "count": int(r[2])} for r in rows if len(r) >= 3 and r[2].isdigit() and r[0].isalpha()]
|
||||
last = sorted(last, key=lambda r: (-r["count"], r["name"]))[:a.last]
|
||||
out = Path(a.out)
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
"""A source fetched once into the cache."""
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def fetch(source, cache, name, magic=b"", data=None, headers=None):
|
||||
"""The bytes of a URL, downloaded into cache/name once, or of a local file."""
|
||||
if not re.match(r"^https?://", source):
|
||||
return Path(source).read_bytes()
|
||||
path = Path(cache) / name
|
||||
for attempt in range(1, 6):
|
||||
if path.exists():
|
||||
return path.read_bytes()
|
||||
req = urllib.request.Request(source, data=data, headers={"User-Agent": "fejkdata data-import", **(headers or {})})
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=600) as r:
|
||||
body = r.read()
|
||||
except OSError:
|
||||
body = b""
|
||||
if body and body.startswith(magic) and b"Request Rejected" not in body[:512]:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_bytes(body)
|
||||
elif attempt < 5:
|
||||
time.sleep(10 * attempt)
|
||||
sys.exit(f"{source}: no valid download in 5 attempts")
|
||||
+1
-45
@@ -1,52 +1,8 @@
|
||||
"""A source fetched once into the cache, and a table written as the loader admits it."""
|
||||
import io
|
||||
"""A table written as the loader admits it."""
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
import urllib.request
|
||||
import xml.etree.ElementTree as ET
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
XLSX_NS = {"m": "http://schemas.openxmlformats.org/spreadsheetml/2006/main", "r": "http://schemas.openxmlformats.org/officeDocument/2006/relationships"}
|
||||
|
||||
|
||||
def fetch(source, cache, name, magic=b"", data=None, headers=None):
|
||||
"""The bytes of a URL, downloaded into cache/name once, or of a local file."""
|
||||
if not re.match(r"^https?://", source):
|
||||
return Path(source).read_bytes()
|
||||
path = Path(cache) / name
|
||||
for attempt in range(1, 6):
|
||||
if path.exists():
|
||||
return path.read_bytes()
|
||||
req = urllib.request.Request(source, data=data, headers={"User-Agent": "fejkdata data-import", **(headers or {})})
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=600) as r:
|
||||
body = r.read()
|
||||
except OSError:
|
||||
body = b""
|
||||
if body and body.startswith(magic) and b"Request Rejected" not in body[:512]:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_bytes(body)
|
||||
elif attempt < 5:
|
||||
time.sleep(10 * attempt)
|
||||
sys.exit(f"{source}: no valid download in 5 attempts")
|
||||
|
||||
|
||||
def xlsx_rows(data, sheet=None):
|
||||
"""The rows of an xlsx sheet named sheet, the first sheet by default, each a list of cell texts."""
|
||||
z = zipfile.ZipFile(io.BytesIO(data))
|
||||
strings = ["".join(t.text or "" for t in si.iter("{%s}t" % XLSX_NS["m"])) for si in ET.fromstring(z.read("xl/sharedStrings.xml")).findall("m:si", XLSX_NS)]
|
||||
rels = {r.get("Id"): r.get("Target") for r in ET.fromstring(z.read("xl/_rels/workbook.xml.rels"))}
|
||||
sheets = {s.get("name"): rels[s.get("{%s}id" % XLSX_NS["r"])] for s in ET.fromstring(z.read("xl/workbook.xml")).iter("{%s}sheet" % XLSX_NS["m"])}
|
||||
target = sheets[sheet] if sheet else next(iter(sheets.values()))
|
||||
for row in ET.fromstring(z.read("xl/" + target)).findall(".//m:row", XLSX_NS):
|
||||
cells = []
|
||||
for c in row.findall("m:c", XLSX_NS):
|
||||
v = c.find("m:v", XLSX_NS)
|
||||
cells.append("" if v is None else strings[int(v.text)] if c.get("t") == "s" else v.text)
|
||||
yield cells
|
||||
|
||||
|
||||
def write(path, columns, rows):
|
||||
"""Write the rows as a TSV; every cell must be non-empty and free of tabs, newlines and braces."""
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
"""The rows of an xlsx sheet."""
|
||||
import io
|
||||
import xml.etree.ElementTree as ET
|
||||
import zipfile
|
||||
|
||||
NS = {"m": "http://schemas.openxmlformats.org/spreadsheetml/2006/main", "r": "http://schemas.openxmlformats.org/officeDocument/2006/relationships"}
|
||||
|
||||
|
||||
def rows(data, sheet=None):
|
||||
"""The rows of an xlsx sheet named sheet, the first sheet by default, each a list of cell texts."""
|
||||
z = zipfile.ZipFile(io.BytesIO(data))
|
||||
strings = ["".join(t.text or "" for t in si.iter("{%s}t" % NS["m"])) for si in ET.fromstring(z.read("xl/sharedStrings.xml")).findall("m:si", NS)]
|
||||
rels = {r.get("Id"): r.get("Target") for r in ET.fromstring(z.read("xl/_rels/workbook.xml.rels"))}
|
||||
sheets = {s.get("name"): rels[s.get("{%s}id" % NS["r"])] for s in ET.fromstring(z.read("xl/workbook.xml")).iter("{%s}sheet" % NS["m"])}
|
||||
target = sheets[sheet] if sheet else next(iter(sheets.values()))
|
||||
for row in ET.fromstring(z.read("xl/" + target)).findall(".//m:row", NS):
|
||||
cells = []
|
||||
for c in row.findall("m:c", NS):
|
||||
v = c.find("m:v", NS)
|
||||
cells.append("" if v is None else strings[int(v.text)] if c.get("t") == "s" else v.text)
|
||||
yield cells
|
||||
Reference in New Issue
Block a user