Refuse an empty download, finish the NVDB pull atomically, drop route designations and unlettered names from the streets, and give a ZCTA to the place holding most of it

This commit is contained in:
2026-09-18 01:34:11 +02:00
parent 8e30d05a0e
commit 8c2b51ebed
6 changed files with 29 additions and 29 deletions
+4 -8
View File
@@ -2,12 +2,6 @@
"""Rebuild data/geo/SE/*.tsv from SCB (CC0), GeoNames (CC BY 4.0) and Trafikverket NVDB (CC0).
TRAFIKVERKET_API_KEY=… data-import/geo-se.py [--key-file FILE] [--cache DIR] [--streets-per-locality N] [--out DIR]
A locality is a GeoNames postort, placed in the municipality it names, else of its
tätort, else of most of its codes, and weighted by its tätort's population, else its
municipality's, else 200. Box codes are dropped by the digit after the postort's own
prefix. Each NVDB street segment goes to the nearest postal code centroid; a locality
keeps the N names with most segments.
"""
import argparse
import collections
@@ -95,7 +89,7 @@ def geonames(cache):
def nvdb_segments(cache, key):
path = cache / "nvdb-gatunamn.tsv"
if not path.exists():
with open(path, "w", encoding="utf-8") as out:
with open(path.with_suffix(".part"), "w", encoding="utf-8") as out:
change = "0"
while True:
query = (
@@ -116,6 +110,7 @@ def nvdb_segments(cache, key):
change = result["INFO"]["LASTCHANGEID"]
if len(rows) < NVDB_PAGE:
break
path.with_suffix(".part").rename(path)
for line in path.read_text(encoding="utf-8").splitlines():
name, lon, lat = line.split("\t")
yield name, float(lat), float(lon)
@@ -208,7 +203,8 @@ def streets(segments, codes, localities, per_locality):
nearest = Nearest((r["lat"], r["lon"], r["locality"]) for r in codes if r["lat"] is not None and r["locality"] in localities)
count = collections.Counter()
for name, lat, lon in segments:
count[(nearest.find(lat, lon), name)] += 1
if name[0].isalpha():
count[(nearest.find(lat, lon), name)] += 1
of = collections.defaultdict(list)
for (locality, name), n in count.items():
of[locality].append((n, name))
+7 -11
View File
@@ -2,11 +2,6 @@
"""Rebuild data/geo/US/*.tsv from the Census Bureau's Gazetteer, population estimates, ZCTA relationships and TIGER/Line files (public domain).
data-import/geo-us.py [--cache DIR] [--min-population N] [--streets-per-locality N] [--out DIR]
A locality is an incorporated place, or a consolidated city's balance, of at least N
people, in the county holding most of it. Its postal codes are the ZCTAs mostly inside
it, weighted by their TIGER address ranges, and its streets the N names with most
address ranges in those codes.
"""
import argparse
import collections
@@ -32,10 +27,10 @@ OUT = Path(__file__).resolve().parent.parent / "data" / "geo" / "US"
CACHE = Path(__file__).resolve().parent / "cache"
ESTIMATE = "POPESTIMATE2025"
CDP = "57"
HIGHWAY = re.compile(r"\b(I- |Hwy |Rte |Route |Rd )\d")
SUFFIX = re.compile(r" (city and borough|city|town|village|borough|municipality|comunidad|zona urbana|metropolitan government|metro government|consolidated government|unified government|urban county|corporation|plantation)( \(balance\))?$")
# Places whose Census name is a merged government's; the postal city is what an address carries.
NAMES = {"1303440": "Athens", "1304204": "Augusta", "1349008": "Macon", "2148006": "Louisville", "3011397": "Butte", "4732742": "Hartsville", "4752006": "Nashville"}
# The predominant zone of each state.
TIMEZONES = {
"AK": "America/Anchorage", "AL": "America/Chicago", "AR": "America/Chicago", "AZ": "America/Phoenix",
"CA": "America/Los_Angeles", "CO": "America/Denver", "CT": "America/New_York", "DC": "America/New_York",
@@ -122,7 +117,7 @@ def localities(cache, min_population, counties):
out = {}
for r in gazetteer(cache, "place"):
geoid, population = r["GEOID"], place_population.get(r["GEOID"], 0)
if r["FUNCSTAT"] not in "AFN" or r["LSAD"] == CDP or population < min_population or not county_part.get(geoid):
if r["FUNCSTAT"] not in ("A", "F", "N") or r["LSAD"] == CDP or population < min_population or not county_part.get(geoid):
continue
county = max(county_part[geoid])[1]
if county not in counties:
@@ -133,12 +128,13 @@ def localities(cache, min_population, counties):
def postal_codes(cache, localities):
"""Each ZCTA and the shipped place holding most of its land."""
"""Each ZCTA whose largest part inside any place lies in a shipped place."""
parts = {}
for r in csv.DictReader(io.StringIO(text(tsv.fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"):
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"] in localities:
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"]:
parts.setdefault(r["GEOID_ZCTA5_20"], []).append((int(r["AREALAND_PART"]), r["GEOID_PLACE_20"]))
return {zcta: max(p)[1] for zcta, p in parts.items()}
largest = {zcta: max(p)[1] for zcta, p in parts.items()}
return {zcta: place for zcta, place in largest.items() if place in localities}
def streets(cache, counties, locality_of_zcta, per_locality):
@@ -153,7 +149,7 @@ def streets(cache, counties, locality_of_zcta, per_locality):
zips[r["TLID"]].add(r["ZIP"])
addresses[r["ZIP"]] += 1
for r in tiger(cache, "featnames", county, {"TLID", "FULLNAME", "PAFLAG"}):
if r["PAFLAG"] == "P" and r["FULLNAME"]:
if r["PAFLAG"] == "P" and r["FULLNAME"] and not HIGHWAY.search(r["FULLNAME"]):
for z in zips.get(r["TLID"], ()):
count[(locality_of_zcta[z], r["FULLNAME"])] += 1
of = collections.defaultdict(list)
+4 -3
View File
@@ -20,10 +20,10 @@ def fetch(source, cache, name, magic=b"", data=None, headers=None):
body = r.read()
except OSError:
body = b""
if body.startswith(magic) and b"Request Rejected" not in body[:512]:
if body and body.startswith(magic) and b"Request Rejected" not in body[:512]:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(body)
else:
elif attempt < 5:
time.sleep(10 * attempt)
sys.exit(f"{source}: no valid download in 5 attempts")
@@ -33,7 +33,8 @@ def write(path, columns, rows):
lines = ["\t".join(columns)]
for row in rows:
cells = [str(row[c]) for c in columns]
assert all(cells) and not any(re.search(r"[\t\n{}]", c) for c in cells), row
if not all(cells) or any(re.search(r"[\t\n{}]", c) for c in cells):
raise ValueError(f"{path}: a cell is empty or holds a tab, newline or brace: {row}")
lines.append("\t".join(cells))
Path(path).write_text("\n".join(lines) + "\n", encoding="utf-8")
print(f"{path}: {len(lines) - 1} rows", file=sys.stderr)