#!/usr/bin/env python3 """Rebuild data/geo/US/*.tsv from the Census Bureau's Gazetteer, population estimates, ZCTA relationships and TIGER/Line files (public domain). data-import/geo-us.py [--cache DIR] [--min-population N] [--streets-per-locality N] [--out DIR] A locality is an incorporated place, or a consolidated city's balance, of at least N people, in the county holding most of it. Its postal codes are the ZCTAs mostly inside it, weighted by their TIGER address ranges, and its streets the N names with most address ranges in those codes. """ import argparse import collections import concurrent.futures import csv import io import re import struct import sys import zipfile from pathlib import Path import tsv GAZETTEER = "https://www2.census.gov/geo/docs/maps-data/data/gazetteer/2026_Gazetteer/2026_Gaz_{}_national.zip" POPULATION = "https://www2.census.gov/programs-surveys/popest/datasets/2020-2025/{}" STATES = POPULATION.format("state/totals/NST-EST2025-ALLDATA.csv") COUNTIES = POPULATION.format("counties/totals/co-est2025-alldata.csv") PLACES = POPULATION.format("cities/totals/sub-est2025.csv") ZCTA_PLACE = "https://www2.census.gov/geo/docs/maps-data/data/rel2020/zcta520/tab20_zcta520_place20_natl.txt" TIGER = "https://www2.census.gov/geo/tiger/TIGER2025/{0}/tl_2025_{1}_{2}.zip" OUT = Path(__file__).resolve().parent.parent / "data" / "geo" / "US" CACHE = Path(__file__).resolve().parent / "cache" ESTIMATE = "POPESTIMATE2025" CDP = "57" SUFFIX = re.compile(r" (city and borough|city|town|village|borough|municipality|comunidad|zona urbana|metropolitan government|metro government|consolidated government|unified government|urban county|corporation|plantation)( \(balance\))?$") # Places whose Census name is a merged government's; the postal city is what an address carries. NAMES = {"1303440": "Athens", "1304204": "Augusta", "1349008": "Macon", "2148006": "Louisville", "3011397": "Butte", "4732742": "Hartsville", "4752006": "Nashville"} # The predominant zone of each state. TIMEZONES = { "AK": "America/Anchorage", "AL": "America/Chicago", "AR": "America/Chicago", "AZ": "America/Phoenix", "CA": "America/Los_Angeles", "CO": "America/Denver", "CT": "America/New_York", "DC": "America/New_York", "DE": "America/New_York", "FL": "America/New_York", "GA": "America/New_York", "HI": "Pacific/Honolulu", "IA": "America/Chicago", "ID": "America/Boise", "IL": "America/Chicago", "IN": "America/Indiana/Indianapolis", "KS": "America/Chicago", "KY": "America/New_York", "LA": "America/Chicago", "MA": "America/New_York", "MD": "America/New_York", "ME": "America/New_York", "MI": "America/Detroit", "MN": "America/Chicago", "MO": "America/Chicago", "MS": "America/Chicago", "MT": "America/Denver", "NC": "America/New_York", "ND": "America/Chicago", "NE": "America/Chicago", "NH": "America/New_York", "NJ": "America/New_York", "NM": "America/Denver", "NV": "America/Los_Angeles", "NY": "America/New_York", "OH": "America/New_York", "OK": "America/Chicago", "OR": "America/Los_Angeles", "PA": "America/New_York", "PR": "America/Puerto_Rico", "RI": "America/New_York", "SC": "America/New_York", "SD": "America/Chicago", "TN": "America/Chicago", "TX": "America/Chicago", "UT": "America/Denver", "VA": "America/New_York", "VT": "America/New_York", "WA": "America/Los_Angeles", "WI": "America/Chicago", "WV": "America/New_York", "WY": "America/Denver", } def text(data): try: return data.decode("utf-8-sig") except UnicodeDecodeError: return data.decode("latin-1") def gazetteer(cache, kind): z = zipfile.ZipFile(io.BytesIO(tsv.fetch(GAZETTEER.format(kind), cache, f"gaz_{kind}.zip", magic=b"PK"))) rows = text(z.read(z.namelist()[0])).splitlines() header = [h.strip() for h in rows[0].split("|")] return [dict(zip(header, (c.strip() for c in row.split("|")))) for row in rows[1:]] def csv_rows(cache, url, name): return list(csv.DictReader(io.StringIO(text(tsv.fetch(url, cache, name))))) def dbf_rows(data, wanted): """The records of a dBASE file, the wanted fields only.""" count, header_len, record_len = struct.unpack("