Share one fetch and one TSV writer across the import scripts, and name the geo scripts' stages

This commit is contained in:
2026-09-18 01:10:44 +02:00
parent 9e37f3dc51
commit 92f58edc2d
5 changed files with 170 additions and 194 deletions
+70 -89
View File
@@ -16,11 +16,11 @@ import io
import re
import struct
import sys
import time
import urllib.request
import zipfile
from pathlib import Path
import tsv
GAZETTEER = "https://www2.census.gov/geo/docs/maps-data/data/gazetteer/2026_Gazetteer/2026_Gaz_{}_national.zip"
POPULATION = "https://www2.census.gov/programs-surveys/popest/datasets/2020-2025/{}"
STATES = POPULATION.format("state/totals/NST-EST2025-ALLDATA.csv")
@@ -35,6 +35,7 @@ CDP = "57"
SUFFIX = re.compile(r" (city and borough|city|town|village|borough|municipality|comunidad|zona urbana|metropolitan government|metro government|consolidated government|unified government|urban county|corporation|plantation)( \(balance\))?$")
# Places whose Census name is a merged government's; the postal city is what an address carries.
NAMES = {"1303440": "Athens", "1304204": "Augusta", "1349008": "Macon", "2148006": "Louisville", "3011397": "Butte", "4732742": "Hartsville", "4752006": "Nashville"}
# The predominant zone of each state.
TIMEZONES = {
"AK": "America/Anchorage", "AL": "America/Chicago", "AR": "America/Chicago", "AZ": "America/Phoenix",
"CA": "America/Los_Angeles", "CO": "America/Denver", "CT": "America/New_York", "DC": "America/New_York",
@@ -52,24 +53,6 @@ TIMEZONES = {
}
def fetch(url, cache, name, magic=b""):
path = cache / name
for attempt in range(1, 6):
if path.exists():
return path.read_bytes()
req = urllib.request.Request(url, headers={"User-Agent": "fejkdata data-import"})
try:
with urllib.request.urlopen(req, timeout=600) as r:
data = r.read()
except OSError:
data = b""
if data.startswith(magic) and b"Request Rejected" not in data[:512]:
path.write_bytes(data)
else:
time.sleep(10 * attempt)
sys.exit(f"{url}: no valid download in 5 attempts")
def text(data):
try:
return data.decode("utf-8-sig")
@@ -78,14 +61,14 @@ def text(data):
def gazetteer(cache, kind):
z = zipfile.ZipFile(io.BytesIO(fetch(GAZETTEER.format(kind), cache, f"gaz_{kind}.zip", magic=b"PK")))
z = zipfile.ZipFile(io.BytesIO(tsv.fetch(GAZETTEER.format(kind), cache, f"gaz_{kind}.zip", magic=b"PK")))
rows = text(z.read(z.namelist()[0])).splitlines()
header = [h.strip() for h in rows[0].split("|")]
return [dict(zip(header, (c.strip() for c in row.split("|")))) for row in rows[1:]]
def csv_rows(cache, url, name):
return list(csv.DictReader(io.StringIO(text(fetch(url, cache, name)))))
return list(csv.DictReader(io.StringIO(text(tsv.fetch(url, cache, name)))))
def dbf_rows(data, wanted):
@@ -111,7 +94,7 @@ def dbf_rows(data, wanted):
def tiger_zip(cache, kind, county):
return fetch(TIGER.format(kind.upper(), county, kind), cache, f"tl_{county}_{kind}.zip", magic=b"PK")
return tsv.fetch(TIGER.format(kind.upper(), county, kind), cache, f"tl_{county}_{kind}.zip", magic=b"PK")
def tiger(cache, kind, county, wanted):
@@ -128,14 +111,56 @@ def place_name(geoid, name):
return stripped
def write(path, columns, rows):
lines = ["\t".join(columns)]
for row in rows:
cells = [str(row[c]) for c in columns]
assert not any(re.search(r"[\t\n{}]", c) for c in cells), row
lines.append("\t".join(cells))
path.write_text("\n".join(lines) + "\n", encoding="utf-8")
print(f"{path}: {len(rows)} rows", file=sys.stderr)
def localities(cache, min_population, counties):
"""Each shipped place with its county, population and centroid."""
place_population, county_part = {}, collections.defaultdict(list)
for r in csv_rows(cache, PLACES, "sub-est2025.csv"):
if r["SUMLEV"] == "162":
place_population[r["STATE"] + r["PLACE"]] = int(r[ESTIMATE])
elif r["SUMLEV"] == "157":
county_part[r["STATE"] + r["PLACE"]].append((int(r[ESTIMATE]), r["STATE"] + r["COUNTY"]))
out = {}
for r in gazetteer(cache, "place"):
geoid, population = r["GEOID"], place_population.get(r["GEOID"], 0)
if r["FUNCSTAT"] not in "AFN" or r["LSAD"] == CDP or population < min_population or not county_part.get(geoid):
continue
county = max(county_part[geoid])[1]
if county not in counties:
print(f"{geoid} {r['NAME']}: county {county} unknown, dropped", file=sys.stderr)
continue
out[geoid] = {"code": geoid, "name": place_name(geoid, r["NAME"]), "municipality": county, "population": population, "lat": r["INTPTLAT"], "lon": r["INTPTLONG"]}
return out
def postal_codes(cache, localities):
"""Each ZCTA and the shipped place holding most of its land."""
parts = {}
for r in csv.DictReader(io.StringIO(text(tsv.fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"):
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"] in localities:
parts.setdefault(r["GEOID_ZCTA5_20"], []).append((int(r["AREALAND_PART"]), r["GEOID_PLACE_20"]))
return {zcta: max(p)[1] for zcta, p in parts.items()}
def streets(cache, counties, locality_of_zcta, per_locality):
"""The names with most TIGER address ranges per place, and the address ranges per ZCTA."""
with concurrent.futures.ThreadPoolExecutor(3) as pool:
list(pool.map(lambda c: (tiger_zip(cache, "addr", c), tiger_zip(cache, "featnames", c)), counties))
addresses, count = collections.Counter(), collections.Counter()
for county in counties:
zips = collections.defaultdict(set)
for r in tiger(cache, "addr", county, {"TLID", "ZIP"}):
if r["ZIP"] in locality_of_zcta:
zips[r["TLID"]].add(r["ZIP"])
addresses[r["ZIP"]] += 1
for r in tiger(cache, "featnames", county, {"TLID", "FULLNAME", "PAFLAG"}):
if r["PAFLAG"] == "P" and r["FULLNAME"]:
for z in zips.get(r["TLID"], ()):
count[(locality_of_zcta[z], r["FULLNAME"])] += 1
of = collections.defaultdict(list)
for (locality, name), n in count.items():
of[locality].append((n, name))
named = {locality: [{"name": name, "locality": locality, "addresses": n} for n, name in sorted(ranked, key=lambda s: (-s[0], s[1]))[:per_locality]] for locality, ranked in of.items()}
return named, addresses
def main():
@@ -150,17 +175,8 @@ def main():
out.mkdir(parents=True, exist_ok=True)
state_population = {r["STATE"]: r[ESTIMATE] for r in csv_rows(cache, STATES, "nst-est2025.csv") if r["SUMLEV"] == "040"}
regions = {r["USPS"]: {"abbr": r["USPS"], "code": r["GEOID"], "name": r["NAME"], "population": state_population[r["GEOID"]], "timezone": TIMEZONES[r["USPS"]]} for r in gazetteer(cache, "state")}
county_population = {r["STATE"] + r["COUNTY"]: r[ESTIMATE] for r in csv_rows(cache, COUNTIES, "co-est2025.csv") if r["SUMLEV"] == "050"}
place_population, county_part = {}, collections.defaultdict(list)
for r in csv_rows(cache, PLACES, "sub-est2025.csv"):
if r["SUMLEV"] == "162":
place_population[r["STATE"] + r["PLACE"]] = int(r[ESTIMATE])
elif r["SUMLEV"] == "157":
county_part[r["STATE"] + r["PLACE"]].append((int(r[ESTIMATE]), r["STATE"] + r["COUNTY"]))
regions = {}
for r in gazetteer(cache, "state"):
regions[r["USPS"]] = {"abbr": r["USPS"], "code": r["GEOID"], "name": r["NAME"], "population": state_population[r["GEOID"]], "timezone": TIMEZONES[r["USPS"]]}
counties, unestimated = {}, collections.Counter()
for r in gazetteer(cache, "counties"):
if r["GEOID"] not in county_population:
@@ -168,56 +184,21 @@ def main():
continue
counties[r["GEOID"]] = {"code": r["GEOID"], "name": r["NAME"], "region": r["USPS"], "population": county_population[r["GEOID"]]}
print(f"counties without a population estimate, dropped: {dict(unestimated)}", file=sys.stderr)
localities = {}
for r in gazetteer(cache, "place"):
geoid, population = r["GEOID"], place_population.get(r["GEOID"], 0)
if r["FUNCSTAT"] not in "AFN" or r["LSAD"] == CDP or population < a.min_population or not county_part.get(geoid):
continue
county = max(county_part[geoid])[1]
if county not in counties:
print(f"{geoid} {r['NAME']}: county {county} unknown, dropped", file=sys.stderr)
continue
localities[geoid] = {"code": geoid, "name": place_name(geoid, r["NAME"]), "municipality": county, "population": population, "lat": r["INTPTLAT"], "lon": r["INTPTLONG"]}
zcta_of = {}
for r in csv.DictReader(io.StringIO(text(fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"):
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"] in localities:
zcta_of.setdefault(r["GEOID_ZCTA5_20"], []).append((int(r["AREALAND_PART"]), r["GEOID_PLACE_20"]))
locality_of_zcta = {zcta: max(parts)[1] for zcta, parts in zcta_of.items()}
needed = sorted({l["municipality"] for l in localities.values()})
with concurrent.futures.ThreadPoolExecutor(3) as pool:
list(pool.map(lambda c: (tiger_zip(cache, "addr", c), tiger_zip(cache, "featnames", c)), needed))
addresses, streets_of = collections.Counter(), collections.Counter()
for county in needed:
zips = collections.defaultdict(set)
for r in tiger(cache, "addr", county, {"TLID", "ZIP"}):
if r["ZIP"] in locality_of_zcta:
zips[r["TLID"]].add(r["ZIP"])
addresses[r["ZIP"]] += 1
for r in tiger(cache, "featnames", county, {"TLID", "FULLNAME", "PAFLAG"}):
if r["PAFLAG"] == "P" and r["FULLNAME"]:
for z in zips.get(r["TLID"], ()):
streets_of[(locality_of_zcta[z], r["FULLNAME"])] += 1
by_locality = collections.defaultdict(list)
for (locality, name), n in streets_of.items():
by_locality[locality].append((n, name))
streets = []
for locality in sorted(by_locality):
for n, name in sorted(by_locality[locality], key=lambda s: (-s[0], s[1]))[: a.streets_per_locality]:
streets.append({"name": name, "locality": locality, "addresses": n})
for geoid in [l for l in localities if l not in by_locality]:
print(f"{geoid} {localities[geoid]['name']}: no streets, dropped", file=sys.stderr)
del localities[geoid]
postal_codes = [{"code": z, "locality": l, "addresses": addresses[z]} for z, l in sorted(locality_of_zcta.items()) if addresses[z] and l in localities]
kept_counties = {l["municipality"] for l in localities.values()}
places = localities(cache, a.min_population, counties)
locality_of_zcta = postal_codes(cache, places)
named, addresses = streets(cache, sorted({l["municipality"] for l in places.values()}), locality_of_zcta, a.streets_per_locality)
for geoid in [l for l in places if l not in named]:
print(f"{geoid} {places[geoid]['name']}: no streets, dropped", file=sys.stderr)
del places[geoid]
kept_counties = {l["municipality"] for l in places.values()}
kept_regions = {counties[c]["region"] for c in kept_counties}
write(out / "region.tsv", ["abbr", "code", "name", "population", "timezone"], [r for _, r in sorted(regions.items()) if r["abbr"] in kept_regions])
write(out / "municipality.tsv", ["code", "name", "region", "population"], [c for _, c in sorted(counties.items()) if c["code"] in kept_counties])
write(out / "locality.tsv", ["code", "name", "municipality", "population", "lat", "lon"], [l for _, l in sorted(localities.items())])
write(out / "postal-code.tsv", ["code", "locality", "addresses"], postal_codes)
write(out / "street.tsv", ["name", "locality", "addresses"], streets)
tsv.write(out / "region.tsv", ["abbr", "code", "name", "population", "timezone"], [r for _, r in sorted(regions.items()) if r["abbr"] in kept_regions])
tsv.write(out / "municipality.tsv", ["code", "name", "region", "population"], [c for _, c in sorted(counties.items()) if c["code"] in kept_counties])
tsv.write(out / "locality.tsv", ["code", "name", "municipality", "population", "lat", "lon"], [l for _, l in sorted(places.items())])
tsv.write(out / "postal-code.tsv", ["code", "locality", "addresses"], [{"code": z, "locality": l, "addresses": addresses[z]} for z, l in sorted(locality_of_zcta.items()) if addresses[z] and l in places])
tsv.write(out / "street.tsv", ["name", "locality", "addresses"], [s for locality in sorted(named) for s in named[locality]])
if __name__ == "__main__":