Share one fetch and one TSV writer across the import scripts, and name the geo scripts' stages

This commit is contained in:
2026-09-18 01:10:44 +02:00
parent 9e37f3dc51
commit 92f58edc2d
5 changed files with 170 additions and 194 deletions
+7 -22
View File
@@ -1,30 +1,24 @@
#!/usr/bin/env python3 #!/usr/bin/env python3
"""Rebuild data/misc/country.tsv from datasets/country-codes (PDDL). """Rebuild data/misc/country.tsv from datasets/country-codes (PDDL).
data-import/country.py [--source URL_OR_FILE] [--out FILE] data-import/country.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
""" """
import argparse import argparse
import csv import csv
import io import io
import re import re
import sys
import urllib.request
from pathlib import Path from pathlib import Path
import tsv
SOURCE = "https://raw.githubusercontent.com/datasets/country-codes/main/data/country-codes.csv" SOURCE = "https://raw.githubusercontent.com/datasets/country-codes/main/data/country-codes.csv"
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "country.tsv" OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "country.tsv"
CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["alpha2", "alpha3", "calling-code", "capital", "currency", "flag", "languages", "name", "numeric", "tld"] COLUMNS = ["alpha2", "alpha3", "calling-code", "capital", "currency", "flag", "languages", "name", "numeric", "tld"]
# Gaps in the source, keyed by alpha2. # Gaps in the source, keyed by alpha2.
FIXUPS = {"TR": {"currency": "TRY"}} FIXUPS = {"TR": {"currency": "TRY"}}
def read(source):
if re.match(r"^https?://", source):
with urllib.request.urlopen(source, timeout=60) as r:
return r.read().decode("utf-8")
return Path(source).read_text(encoding="utf-8")
def flag(alpha2): def flag(alpha2):
return "".join(chr(0x1F1E6 + ord(c) - ord("A")) for c in alpha2) return "".join(chr(0x1F1E6 + ord(c) - ord("A")) for c in alpha2)
@@ -58,23 +52,14 @@ def rows(text):
yield row yield row
def write(out, table):
lines = ["\t".join(COLUMNS)]
for row in sorted(table, key=lambda r: r["alpha2"]):
cells = [row[c] for c in COLUMNS]
assert not any("\t" in c or "\n" in c for c in cells), row
lines.append("\t".join(cells))
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
return len(lines) - 1
def main(): def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--cache", default=str(CACHE))
p.add_argument("--source", default=SOURCE) p.add_argument("--source", default=SOURCE)
p.add_argument("--out", default=str(OUT)) p.add_argument("--out", default=str(OUT))
a = p.parse_args() a = p.parse_args()
n = write(a.out, rows(read(a.source))) table = rows(tsv.fetch(a.source, a.cache, "country-codes.csv").decode("utf-8"))
print(f"{a.out}: {n} rows", file=sys.stderr) tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["alpha2"]))
if __name__ == "__main__": if __name__ == "__main__":
+8 -23
View File
@@ -1,35 +1,28 @@
#!/usr/bin/env python3 #!/usr/bin/env python3
"""Rebuild data/misc/currency.tsv from datasets/currency-codes (PDDL) and CLDR's symbols (Unicode). """Rebuild data/misc/currency.tsv from datasets/currency-codes (PDDL) and CLDR's symbols (Unicode).
data-import/currency.py [--source URL_OR_FILE] [--symbols URL_OR_FILE ...] [--out FILE] data-import/currency.py [--source URL_OR_FILE] [--symbols URL_OR_FILE ...] [--cache DIR] [--out FILE]
The symbols come from the first locale file that has one, narrow symbols before wide. The symbols come from the first locale file that has one, narrow symbols before wide.
""" """
import argparse import argparse
import csv import csv
import io import io
import re
import sys
import urllib.request
import xml.etree.ElementTree as ET import xml.etree.ElementTree as ET
from pathlib import Path from pathlib import Path
import tsv
SOURCE = "https://raw.githubusercontent.com/datasets/currency-codes/main/data/codes-all.csv" SOURCE = "https://raw.githubusercontent.com/datasets/currency-codes/main/data/codes-all.csv"
SYMBOLS = [ SYMBOLS = [
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/en.xml", "https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/en.xml",
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/root.xml", "https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/root.xml",
] ]
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "currency.tsv" OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "currency.tsv"
CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["code", "decimals", "name", "numeric", "symbol"] COLUMNS = ["code", "decimals", "name", "numeric", "symbol"]
def read(source):
if re.match(r"^https?://", source):
with urllib.request.urlopen(source, timeout=60) as r:
return r.read().decode("utf-8")
return Path(source).read_text(encoding="utf-8")
def symbols(xml_texts): def symbols(xml_texts):
"""CLDR's symbol per code: the first locale's narrow symbol, else the first locale's wide one.""" """CLDR's symbol per code: the first locale's narrow symbol, else the first locale's wide one."""
narrow, wide = {}, {} narrow, wide = {}, {}
@@ -58,24 +51,16 @@ def rows(text, symbol):
} }
def write(out, table):
lines = ["\t".join(COLUMNS)]
for row in sorted(table, key=lambda r: r["code"]):
cells = [row[c] for c in COLUMNS]
assert all(cells) and not any("\t" in c or "\n" in c for c in cells), row
lines.append("\t".join(cells))
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
return len(lines) - 1
def main(): def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--cache", default=str(CACHE))
p.add_argument("--source", default=SOURCE) p.add_argument("--source", default=SOURCE)
p.add_argument("--symbols", nargs="+", default=SYMBOLS) p.add_argument("--symbols", nargs="+", default=SYMBOLS)
p.add_argument("--out", default=str(OUT)) p.add_argument("--out", default=str(OUT))
a = p.parse_args() a = p.parse_args()
n = write(a.out, rows(read(a.source), symbols(read(s) for s in a.symbols))) symbol = symbols(tsv.fetch(s, a.cache, Path(s).name).decode("utf-8") for s in a.symbols)
print(f"{a.out}: {n} rows", file=sys.stderr) table = rows(tsv.fetch(a.source, a.cache, "codes-all.csv").decode("utf-8"), symbol)
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["code"]))
if __name__ == "__main__": if __name__ == "__main__":
+46 -60
View File
@@ -23,6 +23,8 @@ import xml.etree.ElementTree as ET
import zipfile import zipfile
from pathlib import Path from pathlib import Path
import tsv
CODES = "https://www.scb.se/contentassets/7a89e48960f741e08918e489ea36354a/kommunlankod-2026.xlsx" CODES = "https://www.scb.se/contentassets/7a89e48960f741e08918e489ea36354a/kommunlankod-2026.xlsx"
POPULATION = "https://api.scb.se/OV0104/v1/doris/sv/ssd/START/BE/BE0101/BE0101A/BefolkningNy" POPULATION = "https://api.scb.se/OV0104/v1/doris/sv/ssd/START/BE/BE0101/BE0101A/BefolkningNy"
POPULATION_QUERY = { POPULATION_QUERY = {
@@ -45,15 +47,6 @@ UNMATCHED_POPULATION = 200
XLSX_NS = {"m": "http://schemas.openxmlformats.org/spreadsheetml/2006/main"} XLSX_NS = {"m": "http://schemas.openxmlformats.org/spreadsheetml/2006/main"}
def fetch(url, cache, name, data=None, headers=None):
path = cache / name
if not path.exists():
req = urllib.request.Request(url, data=data, headers=headers or {})
with urllib.request.urlopen(req, timeout=300) as r:
path.write_bytes(r.read())
return path.read_bytes()
def xlsx_rows(data): def xlsx_rows(data):
z = zipfile.ZipFile(io.BytesIO(data)) z = zipfile.ZipFile(io.BytesIO(data))
strings = ["".join(t.text or "" for t in si.iter("{%s}t" % XLSX_NS["m"])) for si in ET.fromstring(z.read("xl/sharedStrings.xml")).findall("m:si", XLSX_NS)] strings = ["".join(t.text or "" for t in si.iter("{%s}t" % XLSX_NS["m"])) for si in ET.fromstring(z.read("xl/sharedStrings.xml")).findall("m:si", XLSX_NS)]
@@ -68,7 +61,7 @@ def xlsx_rows(data):
def scb_codes(cache): def scb_codes(cache):
regions, municipalities = {}, {} regions, municipalities = {}, {}
for cells in xlsx_rows(fetch(CODES, cache, "kommunlankod.xlsx")): for cells in xlsx_rows(tsv.fetch(CODES, cache, "kommunlankod.xlsx", magic=b"PK")):
if len(cells) < 2 or not re.fullmatch(r"\d{2}|\d{4}", cells[0]): if len(cells) < 2 or not re.fullmatch(r"\d{2}|\d{4}", cells[0]):
continue continue
(regions if len(cells[0]) == 2 else municipalities)[cells[0]] = cells[1].strip() (regions if len(cells[0]) == 2 else municipalities)[cells[0]] = cells[1].strip()
@@ -77,12 +70,12 @@ def scb_codes(cache):
def scb_population(cache): def scb_population(cache):
body = json.dumps(POPULATION_QUERY).encode() body = json.dumps(POPULATION_QUERY).encode()
data = fetch(POPULATION, cache, "befolkning.json", data=body, headers={"Content-Type": "application/json"}) data = tsv.fetch(POPULATION, cache, "befolkning.json", data=body, headers={"Content-Type": "application/json"})
return {row["key"][0]: row["values"][0] for row in json.loads(data.decode("utf-8-sig"))["data"]} return {row["key"][0]: row["values"][0] for row in json.loads(data.decode("utf-8-sig"))["data"]}
def scb_tatorter(cache): def scb_tatorter(cache):
text = fetch(TATORTER, cache, "tatorter.csv").decode("utf-8") text = tsv.fetch(TATORTER, cache, "tatorter.csv").decode("utf-8")
by_name = collections.defaultdict(list) by_name = collections.defaultdict(list)
for r in csv.DictReader(io.StringIO(text)): for r in csv.DictReader(io.StringIO(text)):
by_name[r["tatort"]].append((r["kommun"], int(r["bef"]))) by_name[r["tatort"]].append((r["kommun"], int(r["bef"])))
@@ -90,7 +83,7 @@ def scb_tatorter(cache):
def geonames(cache): def geonames(cache):
z = zipfile.ZipFile(io.BytesIO(fetch(POSTAL_CODES, cache, "SE.zip"))) z = zipfile.ZipFile(io.BytesIO(tsv.fetch(POSTAL_CODES, cache, "SE.zip", magic=b"PK")))
rows = [] rows = []
for line in z.read("SE.txt").decode("utf-8").splitlines(): for line in z.read("SE.txt").decode("utf-8").splitlines():
f = line.split("\t") f = line.split("\t")
@@ -190,14 +183,36 @@ def well_cased(name):
return all(part[:1].isupper() and (len(part) == 1 or not part.isupper()) for part in re.split(r"[ -]", name)) return all(part[:1].isupper() and (len(part) == 1 or not part.isupper()) for part in re.split(r"[ -]", name))
def write(path, columns, rows): def localities(codes, tatorter, municipalities, population):
lines = ["\t".join(columns)] """Each postort with its municipality, weight, centroid and street-delivery codes."""
for row in rows: by_locality = collections.defaultdict(list)
cells = [str(row[c]) for c in columns] for r in codes:
assert not any(re.search(r"[\t\n{}]", c) for c in cells), row by_locality[r["locality"]].append(r)
lines.append("\t".join(cells)) out, how = {}, collections.Counter()
path.write_text("\n".join(lines) + "\n", encoding="utf-8") for name, rows in by_locality.items():
print(f"{path}: {len(rows)} rows", file=sys.stderr) municipality, method = municipality_of(name, rows, tatorter, municipalities)
how[method] += 1
kept = street_delivery(name, [r["code"].replace(" ", "") for r in rows])
with_point = [r for r in rows if r["lat"] is not None]
if municipality is None or not kept or not with_point or not well_cased(name):
continue
lat = sum(r["lat"] for r in with_point) / len(with_point)
lon = sum(r["lon"] for r in with_point) / len(with_point)
out[name] = {"name": name, "municipality": municipality, "population": population_of(name, municipality, tatorter, municipalities, population), "lat": f"{lat:.4f}", "lon": f"{lon:.4f}", "codes": kept}
print(f"municipality by {dict(how)}; {len(by_locality) - len(out)} postorter dropped", file=sys.stderr)
return out
def streets(segments, codes, localities, per_locality):
"""The names with most segments per locality, each segment at its nearest code centroid."""
nearest = Nearest((r["lat"], r["lon"], r["locality"]) for r in codes if r["lat"] is not None and r["locality"] in localities)
count = collections.Counter()
for name, lat, lon in segments:
count[(nearest.find(lat, lon), name)] += 1
of = collections.defaultdict(list)
for (locality, name), n in count.items():
of[locality].append((n, name))
return {locality: [{"name": name, "locality": locality, "segments": n} for n, name in sorted(named, key=lambda s: (-s[0], s[1]))[:per_locality]] for locality, named in of.items()}
def main(): def main():
@@ -216,48 +231,19 @@ def main():
regions, municipalities = scb_codes(cache) regions, municipalities = scb_codes(cache)
population = scb_population(cache) population = scb_population(cache)
tatorter = scb_tatorter(cache)
codes = geonames(cache) codes = geonames(cache)
places = localities(codes, scb_tatorter(cache), municipalities, population)
by_locality = collections.defaultdict(list) named = streets(nvdb_segments(cache, key), codes, places, a.streets_per_locality)
for r in codes: places = {name: l for name, l in places.items() if name in named}
by_locality[r["locality"]].append(r) empty = sorted(m for m in municipalities if not any(l["municipality"] == m for l in places.values()))
localities, how = {}, collections.Counter()
for name, rows in by_locality.items():
municipality, method = municipality_of(name, rows, tatorter, municipalities)
how[method] += 1
kept = street_delivery(name, [r["code"].replace(" ", "") for r in rows])
with_point = [r for r in rows if r["lat"] is not None]
if municipality is None or not kept or not with_point or not well_cased(name):
continue
lat = sum(r["lat"] for r in with_point) / len(with_point)
lon = sum(r["lon"] for r in with_point) / len(with_point)
localities[name] = {"name": name, "municipality": municipality, "population": population_of(name, municipality, tatorter, municipalities, population), "lat": f"{lat:.4f}", "lon": f"{lon:.4f}", "codes": kept}
print(f"municipality by {dict(how)}; {len(by_locality) - len(localities)} postorter dropped", file=sys.stderr)
centroids = [(r["lat"], r["lon"], r["locality"]) for r in codes if r["lat"] is not None and r["locality"] in localities]
nearest_locality = Nearest(centroids)
segments = collections.Counter()
for name, lat, lon in nvdb_segments(cache, key):
segments[(nearest_locality.find(lat, lon), name)] += 1
streets_of = collections.defaultdict(list)
for (locality, name), n in segments.items():
streets_of[locality].append((n, name))
streets = []
for locality in sorted(streets_of):
for n, name in sorted(streets_of[locality], key=lambda s: (-s[0], s[1]))[: a.streets_per_locality]:
streets.append({"name": name, "locality": locality, "segments": n})
for name in [l for l in localities if l not in streets_of]:
del localities[name]
empty = sorted(m for m in municipalities if not any(l["municipality"] == m for l in localities.values()))
if empty: if empty:
sys.exit(f"municipalities without a locality: {empty}") sys.exit(f"municipalities without a locality: {empty}")
write(out / "region.tsv", ["code", "name", "population", "timezone"], [{"code": c, "name": n, "population": population[c], "timezone": TIMEZONE} for c, n in sorted(regions.items())])
write(out / "municipality.tsv", ["code", "name", "region", "population"], [{"code": c, "name": n, "region": c[:2], "population": population[c]} for c, n in sorted(municipalities.items())]) tsv.write(out / "region.tsv", ["code", "name", "population", "timezone"], [{"code": c, "name": n, "population": population[c], "timezone": TIMEZONE} for c, n in sorted(regions.items())])
write(out / "locality.tsv", ["name", "municipality", "population", "lat", "lon"], [l for _, l in sorted(localities.items())]) tsv.write(out / "municipality.tsv", ["code", "name", "region", "population"], [{"code": c, "name": n, "region": c[:2], "population": population[c]} for c, n in sorted(municipalities.items())])
write(out / "postal-code.tsv", ["code", "locality"], sorted(({"code": f"{c[:3]} {c[3:]}", "locality": l["name"]} for l in localities.values() for c in l["codes"]), key=lambda r: r["code"])) tsv.write(out / "locality.tsv", ["name", "municipality", "population", "lat", "lon"], [l for _, l in sorted(places.items())])
write(out / "street.tsv", ["name", "locality", "segments"], streets) tsv.write(out / "postal-code.tsv", ["code", "locality"], sorted(({"code": f"{c[:3]} {c[3:]}", "locality": l["name"]} for l in places.values() for c in l["codes"]), key=lambda r: r["code"]))
tsv.write(out / "street.tsv", ["name", "locality", "segments"], [s for locality in sorted(named) for s in named[locality]])
if __name__ == "__main__": if __name__ == "__main__":
+70 -89
View File
@@ -16,11 +16,11 @@ import io
import re import re
import struct import struct
import sys import sys
import time
import urllib.request
import zipfile import zipfile
from pathlib import Path from pathlib import Path
import tsv
GAZETTEER = "https://www2.census.gov/geo/docs/maps-data/data/gazetteer/2026_Gazetteer/2026_Gaz_{}_national.zip" GAZETTEER = "https://www2.census.gov/geo/docs/maps-data/data/gazetteer/2026_Gazetteer/2026_Gaz_{}_national.zip"
POPULATION = "https://www2.census.gov/programs-surveys/popest/datasets/2020-2025/{}" POPULATION = "https://www2.census.gov/programs-surveys/popest/datasets/2020-2025/{}"
STATES = POPULATION.format("state/totals/NST-EST2025-ALLDATA.csv") STATES = POPULATION.format("state/totals/NST-EST2025-ALLDATA.csv")
@@ -35,6 +35,7 @@ CDP = "57"
SUFFIX = re.compile(r" (city and borough|city|town|village|borough|municipality|comunidad|zona urbana|metropolitan government|metro government|consolidated government|unified government|urban county|corporation|plantation)( \(balance\))?$") SUFFIX = re.compile(r" (city and borough|city|town|village|borough|municipality|comunidad|zona urbana|metropolitan government|metro government|consolidated government|unified government|urban county|corporation|plantation)( \(balance\))?$")
# Places whose Census name is a merged government's; the postal city is what an address carries. # Places whose Census name is a merged government's; the postal city is what an address carries.
NAMES = {"1303440": "Athens", "1304204": "Augusta", "1349008": "Macon", "2148006": "Louisville", "3011397": "Butte", "4732742": "Hartsville", "4752006": "Nashville"} NAMES = {"1303440": "Athens", "1304204": "Augusta", "1349008": "Macon", "2148006": "Louisville", "3011397": "Butte", "4732742": "Hartsville", "4752006": "Nashville"}
# The predominant zone of each state.
TIMEZONES = { TIMEZONES = {
"AK": "America/Anchorage", "AL": "America/Chicago", "AR": "America/Chicago", "AZ": "America/Phoenix", "AK": "America/Anchorage", "AL": "America/Chicago", "AR": "America/Chicago", "AZ": "America/Phoenix",
"CA": "America/Los_Angeles", "CO": "America/Denver", "CT": "America/New_York", "DC": "America/New_York", "CA": "America/Los_Angeles", "CO": "America/Denver", "CT": "America/New_York", "DC": "America/New_York",
@@ -52,24 +53,6 @@ TIMEZONES = {
} }
def fetch(url, cache, name, magic=b""):
path = cache / name
for attempt in range(1, 6):
if path.exists():
return path.read_bytes()
req = urllib.request.Request(url, headers={"User-Agent": "fejkdata data-import"})
try:
with urllib.request.urlopen(req, timeout=600) as r:
data = r.read()
except OSError:
data = b""
if data.startswith(magic) and b"Request Rejected" not in data[:512]:
path.write_bytes(data)
else:
time.sleep(10 * attempt)
sys.exit(f"{url}: no valid download in 5 attempts")
def text(data): def text(data):
try: try:
return data.decode("utf-8-sig") return data.decode("utf-8-sig")
@@ -78,14 +61,14 @@ def text(data):
def gazetteer(cache, kind): def gazetteer(cache, kind):
z = zipfile.ZipFile(io.BytesIO(fetch(GAZETTEER.format(kind), cache, f"gaz_{kind}.zip", magic=b"PK"))) z = zipfile.ZipFile(io.BytesIO(tsv.fetch(GAZETTEER.format(kind), cache, f"gaz_{kind}.zip", magic=b"PK")))
rows = text(z.read(z.namelist()[0])).splitlines() rows = text(z.read(z.namelist()[0])).splitlines()
header = [h.strip() for h in rows[0].split("|")] header = [h.strip() for h in rows[0].split("|")]
return [dict(zip(header, (c.strip() for c in row.split("|")))) for row in rows[1:]] return [dict(zip(header, (c.strip() for c in row.split("|")))) for row in rows[1:]]
def csv_rows(cache, url, name): def csv_rows(cache, url, name):
return list(csv.DictReader(io.StringIO(text(fetch(url, cache, name))))) return list(csv.DictReader(io.StringIO(text(tsv.fetch(url, cache, name)))))
def dbf_rows(data, wanted): def dbf_rows(data, wanted):
@@ -111,7 +94,7 @@ def dbf_rows(data, wanted):
def tiger_zip(cache, kind, county): def tiger_zip(cache, kind, county):
return fetch(TIGER.format(kind.upper(), county, kind), cache, f"tl_{county}_{kind}.zip", magic=b"PK") return tsv.fetch(TIGER.format(kind.upper(), county, kind), cache, f"tl_{county}_{kind}.zip", magic=b"PK")
def tiger(cache, kind, county, wanted): def tiger(cache, kind, county, wanted):
@@ -128,14 +111,56 @@ def place_name(geoid, name):
return stripped return stripped
def write(path, columns, rows): def localities(cache, min_population, counties):
lines = ["\t".join(columns)] """Each shipped place with its county, population and centroid."""
for row in rows: place_population, county_part = {}, collections.defaultdict(list)
cells = [str(row[c]) for c in columns] for r in csv_rows(cache, PLACES, "sub-est2025.csv"):
assert not any(re.search(r"[\t\n{}]", c) for c in cells), row if r["SUMLEV"] == "162":
lines.append("\t".join(cells)) place_population[r["STATE"] + r["PLACE"]] = int(r[ESTIMATE])
path.write_text("\n".join(lines) + "\n", encoding="utf-8") elif r["SUMLEV"] == "157":
print(f"{path}: {len(rows)} rows", file=sys.stderr) county_part[r["STATE"] + r["PLACE"]].append((int(r[ESTIMATE]), r["STATE"] + r["COUNTY"]))
out = {}
for r in gazetteer(cache, "place"):
geoid, population = r["GEOID"], place_population.get(r["GEOID"], 0)
if r["FUNCSTAT"] not in "AFN" or r["LSAD"] == CDP or population < min_population or not county_part.get(geoid):
continue
county = max(county_part[geoid])[1]
if county not in counties:
print(f"{geoid} {r['NAME']}: county {county} unknown, dropped", file=sys.stderr)
continue
out[geoid] = {"code": geoid, "name": place_name(geoid, r["NAME"]), "municipality": county, "population": population, "lat": r["INTPTLAT"], "lon": r["INTPTLONG"]}
return out
def postal_codes(cache, localities):
"""Each ZCTA and the shipped place holding most of its land."""
parts = {}
for r in csv.DictReader(io.StringIO(text(tsv.fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"):
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"] in localities:
parts.setdefault(r["GEOID_ZCTA5_20"], []).append((int(r["AREALAND_PART"]), r["GEOID_PLACE_20"]))
return {zcta: max(p)[1] for zcta, p in parts.items()}
def streets(cache, counties, locality_of_zcta, per_locality):
"""The names with most TIGER address ranges per place, and the address ranges per ZCTA."""
with concurrent.futures.ThreadPoolExecutor(3) as pool:
list(pool.map(lambda c: (tiger_zip(cache, "addr", c), tiger_zip(cache, "featnames", c)), counties))
addresses, count = collections.Counter(), collections.Counter()
for county in counties:
zips = collections.defaultdict(set)
for r in tiger(cache, "addr", county, {"TLID", "ZIP"}):
if r["ZIP"] in locality_of_zcta:
zips[r["TLID"]].add(r["ZIP"])
addresses[r["ZIP"]] += 1
for r in tiger(cache, "featnames", county, {"TLID", "FULLNAME", "PAFLAG"}):
if r["PAFLAG"] == "P" and r["FULLNAME"]:
for z in zips.get(r["TLID"], ()):
count[(locality_of_zcta[z], r["FULLNAME"])] += 1
of = collections.defaultdict(list)
for (locality, name), n in count.items():
of[locality].append((n, name))
named = {locality: [{"name": name, "locality": locality, "addresses": n} for n, name in sorted(ranked, key=lambda s: (-s[0], s[1]))[:per_locality]] for locality, ranked in of.items()}
return named, addresses
def main(): def main():
@@ -150,17 +175,8 @@ def main():
out.mkdir(parents=True, exist_ok=True) out.mkdir(parents=True, exist_ok=True)
state_population = {r["STATE"]: r[ESTIMATE] for r in csv_rows(cache, STATES, "nst-est2025.csv") if r["SUMLEV"] == "040"} state_population = {r["STATE"]: r[ESTIMATE] for r in csv_rows(cache, STATES, "nst-est2025.csv") if r["SUMLEV"] == "040"}
regions = {r["USPS"]: {"abbr": r["USPS"], "code": r["GEOID"], "name": r["NAME"], "population": state_population[r["GEOID"]], "timezone": TIMEZONES[r["USPS"]]} for r in gazetteer(cache, "state")}
county_population = {r["STATE"] + r["COUNTY"]: r[ESTIMATE] for r in csv_rows(cache, COUNTIES, "co-est2025.csv") if r["SUMLEV"] == "050"} county_population = {r["STATE"] + r["COUNTY"]: r[ESTIMATE] for r in csv_rows(cache, COUNTIES, "co-est2025.csv") if r["SUMLEV"] == "050"}
place_population, county_part = {}, collections.defaultdict(list)
for r in csv_rows(cache, PLACES, "sub-est2025.csv"):
if r["SUMLEV"] == "162":
place_population[r["STATE"] + r["PLACE"]] = int(r[ESTIMATE])
elif r["SUMLEV"] == "157":
county_part[r["STATE"] + r["PLACE"]].append((int(r[ESTIMATE]), r["STATE"] + r["COUNTY"]))
regions = {}
for r in gazetteer(cache, "state"):
regions[r["USPS"]] = {"abbr": r["USPS"], "code": r["GEOID"], "name": r["NAME"], "population": state_population[r["GEOID"]], "timezone": TIMEZONES[r["USPS"]]}
counties, unestimated = {}, collections.Counter() counties, unestimated = {}, collections.Counter()
for r in gazetteer(cache, "counties"): for r in gazetteer(cache, "counties"):
if r["GEOID"] not in county_population: if r["GEOID"] not in county_population:
@@ -168,56 +184,21 @@ def main():
continue continue
counties[r["GEOID"]] = {"code": r["GEOID"], "name": r["NAME"], "region": r["USPS"], "population": county_population[r["GEOID"]]} counties[r["GEOID"]] = {"code": r["GEOID"], "name": r["NAME"], "region": r["USPS"], "population": county_population[r["GEOID"]]}
print(f"counties without a population estimate, dropped: {dict(unestimated)}", file=sys.stderr) print(f"counties without a population estimate, dropped: {dict(unestimated)}", file=sys.stderr)
localities = {}
for r in gazetteer(cache, "place"):
geoid, population = r["GEOID"], place_population.get(r["GEOID"], 0)
if r["FUNCSTAT"] not in "AFN" or r["LSAD"] == CDP or population < a.min_population or not county_part.get(geoid):
continue
county = max(county_part[geoid])[1]
if county not in counties:
print(f"{geoid} {r['NAME']}: county {county} unknown, dropped", file=sys.stderr)
continue
localities[geoid] = {"code": geoid, "name": place_name(geoid, r["NAME"]), "municipality": county, "population": population, "lat": r["INTPTLAT"], "lon": r["INTPTLONG"]}
zcta_of = {} places = localities(cache, a.min_population, counties)
for r in csv.DictReader(io.StringIO(text(fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"): locality_of_zcta = postal_codes(cache, places)
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"] in localities: named, addresses = streets(cache, sorted({l["municipality"] for l in places.values()}), locality_of_zcta, a.streets_per_locality)
zcta_of.setdefault(r["GEOID_ZCTA5_20"], []).append((int(r["AREALAND_PART"]), r["GEOID_PLACE_20"])) for geoid in [l for l in places if l not in named]:
locality_of_zcta = {zcta: max(parts)[1] for zcta, parts in zcta_of.items()} print(f"{geoid} {places[geoid]['name']}: no streets, dropped", file=sys.stderr)
del places[geoid]
needed = sorted({l["municipality"] for l in localities.values()}) kept_counties = {l["municipality"] for l in places.values()}
with concurrent.futures.ThreadPoolExecutor(3) as pool:
list(pool.map(lambda c: (tiger_zip(cache, "addr", c), tiger_zip(cache, "featnames", c)), needed))
addresses, streets_of = collections.Counter(), collections.Counter()
for county in needed:
zips = collections.defaultdict(set)
for r in tiger(cache, "addr", county, {"TLID", "ZIP"}):
if r["ZIP"] in locality_of_zcta:
zips[r["TLID"]].add(r["ZIP"])
addresses[r["ZIP"]] += 1
for r in tiger(cache, "featnames", county, {"TLID", "FULLNAME", "PAFLAG"}):
if r["PAFLAG"] == "P" and r["FULLNAME"]:
for z in zips.get(r["TLID"], ()):
streets_of[(locality_of_zcta[z], r["FULLNAME"])] += 1
by_locality = collections.defaultdict(list)
for (locality, name), n in streets_of.items():
by_locality[locality].append((n, name))
streets = []
for locality in sorted(by_locality):
for n, name in sorted(by_locality[locality], key=lambda s: (-s[0], s[1]))[: a.streets_per_locality]:
streets.append({"name": name, "locality": locality, "addresses": n})
for geoid in [l for l in localities if l not in by_locality]:
print(f"{geoid} {localities[geoid]['name']}: no streets, dropped", file=sys.stderr)
del localities[geoid]
postal_codes = [{"code": z, "locality": l, "addresses": addresses[z]} for z, l in sorted(locality_of_zcta.items()) if addresses[z] and l in localities]
kept_counties = {l["municipality"] for l in localities.values()}
kept_regions = {counties[c]["region"] for c in kept_counties} kept_regions = {counties[c]["region"] for c in kept_counties}
write(out / "region.tsv", ["abbr", "code", "name", "population", "timezone"], [r for _, r in sorted(regions.items()) if r["abbr"] in kept_regions])
write(out / "municipality.tsv", ["code", "name", "region", "population"], [c for _, c in sorted(counties.items()) if c["code"] in kept_counties]) tsv.write(out / "region.tsv", ["abbr", "code", "name", "population", "timezone"], [r for _, r in sorted(regions.items()) if r["abbr"] in kept_regions])
write(out / "locality.tsv", ["code", "name", "municipality", "population", "lat", "lon"], [l for _, l in sorted(localities.items())]) tsv.write(out / "municipality.tsv", ["code", "name", "region", "population"], [c for _, c in sorted(counties.items()) if c["code"] in kept_counties])
write(out / "postal-code.tsv", ["code", "locality", "addresses"], postal_codes) tsv.write(out / "locality.tsv", ["code", "name", "municipality", "population", "lat", "lon"], [l for _, l in sorted(places.items())])
write(out / "street.tsv", ["name", "locality", "addresses"], streets) tsv.write(out / "postal-code.tsv", ["code", "locality", "addresses"], [{"code": z, "locality": l, "addresses": addresses[z]} for z, l in sorted(locality_of_zcta.items()) if addresses[z] and l in places])
tsv.write(out / "street.tsv", ["name", "locality", "addresses"], [s for locality in sorted(named) for s in named[locality]])
if __name__ == "__main__": if __name__ == "__main__":
+39
View File
@@ -0,0 +1,39 @@
"""A source fetched once into the cache, and a table written as the loader admits it."""
import re
import sys
import time
import urllib.request
from pathlib import Path
def fetch(source, cache, name, magic=b"", data=None, headers=None):
"""The bytes of a URL, downloaded into cache/name once, or of a local file."""
if not re.match(r"^https?://", source):
return Path(source).read_bytes()
path = Path(cache) / name
for attempt in range(1, 6):
if path.exists():
return path.read_bytes()
req = urllib.request.Request(source, data=data, headers={"User-Agent": "fejkdata data-import", **(headers or {})})
try:
with urllib.request.urlopen(req, timeout=600) as r:
body = r.read()
except OSError:
body = b""
if body.startswith(magic) and b"Request Rejected" not in body[:512]:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(body)
else:
time.sleep(10 * attempt)
sys.exit(f"{source}: no valid download in 5 attempts")
def write(path, columns, rows):
"""Write the rows as a TSV; every cell must be non-empty and free of tabs, newlines and braces."""
lines = ["\t".join(columns)]
for row in rows:
cells = [str(row[c]) for c in columns]
assert all(cells) and not any(re.search(r"[\t\n{}]", c) for c in cells), row
lines.append("\t".join(cells))
Path(path).write_text("\n".join(lines) + "\n", encoding="utf-8")
print(f"{path}: {len(lines) - 1} rows", file=sys.stderr)