Share one fetch and one TSV writer across the import scripts, and name the geo scripts' stages

This commit is contained in:
2026-09-18 01:10:44 +02:00
parent 9e37f3dc51
commit 92f58edc2d
5 changed files with 170 additions and 194 deletions
+7 -22
View File
@@ -1,30 +1,24 @@
#!/usr/bin/env python3
"""Rebuild data/misc/country.tsv from datasets/country-codes (PDDL).
data-import/country.py [--source URL_OR_FILE] [--out FILE]
data-import/country.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
"""
import argparse
import csv
import io
import re
import sys
import urllib.request
from pathlib import Path
import tsv
SOURCE = "https://raw.githubusercontent.com/datasets/country-codes/main/data/country-codes.csv"
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "country.tsv"
CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["alpha2", "alpha3", "calling-code", "capital", "currency", "flag", "languages", "name", "numeric", "tld"]
# Gaps in the source, keyed by alpha2.
FIXUPS = {"TR": {"currency": "TRY"}}
def read(source):
if re.match(r"^https?://", source):
with urllib.request.urlopen(source, timeout=60) as r:
return r.read().decode("utf-8")
return Path(source).read_text(encoding="utf-8")
def flag(alpha2):
return "".join(chr(0x1F1E6 + ord(c) - ord("A")) for c in alpha2)
@@ -58,23 +52,14 @@ def rows(text):
yield row
def write(out, table):
lines = ["\t".join(COLUMNS)]
for row in sorted(table, key=lambda r: r["alpha2"]):
cells = [row[c] for c in COLUMNS]
assert not any("\t" in c or "\n" in c for c in cells), row
lines.append("\t".join(cells))
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
return len(lines) - 1
def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--cache", default=str(CACHE))
p.add_argument("--source", default=SOURCE)
p.add_argument("--out", default=str(OUT))
a = p.parse_args()
n = write(a.out, rows(read(a.source)))
print(f"{a.out}: {n} rows", file=sys.stderr)
table = rows(tsv.fetch(a.source, a.cache, "country-codes.csv").decode("utf-8"))
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["alpha2"]))
if __name__ == "__main__":
+8 -23
View File
@@ -1,35 +1,28 @@
#!/usr/bin/env python3
"""Rebuild data/misc/currency.tsv from datasets/currency-codes (PDDL) and CLDR's symbols (Unicode).
data-import/currency.py [--source URL_OR_FILE] [--symbols URL_OR_FILE ...] [--out FILE]
data-import/currency.py [--source URL_OR_FILE] [--symbols URL_OR_FILE ...] [--cache DIR] [--out FILE]
The symbols come from the first locale file that has one, narrow symbols before wide.
"""
import argparse
import csv
import io
import re
import sys
import urllib.request
import xml.etree.ElementTree as ET
from pathlib import Path
import tsv
SOURCE = "https://raw.githubusercontent.com/datasets/currency-codes/main/data/codes-all.csv"
SYMBOLS = [
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/en.xml",
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/root.xml",
]
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "currency.tsv"
CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["code", "decimals", "name", "numeric", "symbol"]
def read(source):
if re.match(r"^https?://", source):
with urllib.request.urlopen(source, timeout=60) as r:
return r.read().decode("utf-8")
return Path(source).read_text(encoding="utf-8")
def symbols(xml_texts):
"""CLDR's symbol per code: the first locale's narrow symbol, else the first locale's wide one."""
narrow, wide = {}, {}
@@ -58,24 +51,16 @@ def rows(text, symbol):
}
def write(out, table):
lines = ["\t".join(COLUMNS)]
for row in sorted(table, key=lambda r: r["code"]):
cells = [row[c] for c in COLUMNS]
assert all(cells) and not any("\t" in c or "\n" in c for c in cells), row
lines.append("\t".join(cells))
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
return len(lines) - 1
def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--cache", default=str(CACHE))
p.add_argument("--source", default=SOURCE)
p.add_argument("--symbols", nargs="+", default=SYMBOLS)
p.add_argument("--out", default=str(OUT))
a = p.parse_args()
n = write(a.out, rows(read(a.source), symbols(read(s) for s in a.symbols)))
print(f"{a.out}: {n} rows", file=sys.stderr)
symbol = symbols(tsv.fetch(s, a.cache, Path(s).name).decode("utf-8") for s in a.symbols)
table = rows(tsv.fetch(a.source, a.cache, "codes-all.csv").decode("utf-8"), symbol)
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["code"]))
if __name__ == "__main__":
+46 -60
View File
@@ -23,6 +23,8 @@ import xml.etree.ElementTree as ET
import zipfile
from pathlib import Path
import tsv
CODES = "https://www.scb.se/contentassets/7a89e48960f741e08918e489ea36354a/kommunlankod-2026.xlsx"
POPULATION = "https://api.scb.se/OV0104/v1/doris/sv/ssd/START/BE/BE0101/BE0101A/BefolkningNy"
POPULATION_QUERY = {
@@ -45,15 +47,6 @@ UNMATCHED_POPULATION = 200
XLSX_NS = {"m": "http://schemas.openxmlformats.org/spreadsheetml/2006/main"}
def fetch(url, cache, name, data=None, headers=None):
path = cache / name
if not path.exists():
req = urllib.request.Request(url, data=data, headers=headers or {})
with urllib.request.urlopen(req, timeout=300) as r:
path.write_bytes(r.read())
return path.read_bytes()
def xlsx_rows(data):
z = zipfile.ZipFile(io.BytesIO(data))
strings = ["".join(t.text or "" for t in si.iter("{%s}t" % XLSX_NS["m"])) for si in ET.fromstring(z.read("xl/sharedStrings.xml")).findall("m:si", XLSX_NS)]
@@ -68,7 +61,7 @@ def xlsx_rows(data):
def scb_codes(cache):
regions, municipalities = {}, {}
for cells in xlsx_rows(fetch(CODES, cache, "kommunlankod.xlsx")):
for cells in xlsx_rows(tsv.fetch(CODES, cache, "kommunlankod.xlsx", magic=b"PK")):
if len(cells) < 2 or not re.fullmatch(r"\d{2}|\d{4}", cells[0]):
continue
(regions if len(cells[0]) == 2 else municipalities)[cells[0]] = cells[1].strip()
@@ -77,12 +70,12 @@ def scb_codes(cache):
def scb_population(cache):
body = json.dumps(POPULATION_QUERY).encode()
data = fetch(POPULATION, cache, "befolkning.json", data=body, headers={"Content-Type": "application/json"})
data = tsv.fetch(POPULATION, cache, "befolkning.json", data=body, headers={"Content-Type": "application/json"})
return {row["key"][0]: row["values"][0] for row in json.loads(data.decode("utf-8-sig"))["data"]}
def scb_tatorter(cache):
text = fetch(TATORTER, cache, "tatorter.csv").decode("utf-8")
text = tsv.fetch(TATORTER, cache, "tatorter.csv").decode("utf-8")
by_name = collections.defaultdict(list)
for r in csv.DictReader(io.StringIO(text)):
by_name[r["tatort"]].append((r["kommun"], int(r["bef"])))
@@ -90,7 +83,7 @@ def scb_tatorter(cache):
def geonames(cache):
z = zipfile.ZipFile(io.BytesIO(fetch(POSTAL_CODES, cache, "SE.zip")))
z = zipfile.ZipFile(io.BytesIO(tsv.fetch(POSTAL_CODES, cache, "SE.zip", magic=b"PK")))
rows = []
for line in z.read("SE.txt").decode("utf-8").splitlines():
f = line.split("\t")
@@ -190,14 +183,36 @@ def well_cased(name):
return all(part[:1].isupper() and (len(part) == 1 or not part.isupper()) for part in re.split(r"[ -]", name))
def write(path, columns, rows):
lines = ["\t".join(columns)]
for row in rows:
cells = [str(row[c]) for c in columns]
assert not any(re.search(r"[\t\n{}]", c) for c in cells), row
lines.append("\t".join(cells))
path.write_text("\n".join(lines) + "\n", encoding="utf-8")
print(f"{path}: {len(rows)} rows", file=sys.stderr)
def localities(codes, tatorter, municipalities, population):
"""Each postort with its municipality, weight, centroid and street-delivery codes."""
by_locality = collections.defaultdict(list)
for r in codes:
by_locality[r["locality"]].append(r)
out, how = {}, collections.Counter()
for name, rows in by_locality.items():
municipality, method = municipality_of(name, rows, tatorter, municipalities)
how[method] += 1
kept = street_delivery(name, [r["code"].replace(" ", "") for r in rows])
with_point = [r for r in rows if r["lat"] is not None]
if municipality is None or not kept or not with_point or not well_cased(name):
continue
lat = sum(r["lat"] for r in with_point) / len(with_point)
lon = sum(r["lon"] for r in with_point) / len(with_point)
out[name] = {"name": name, "municipality": municipality, "population": population_of(name, municipality, tatorter, municipalities, population), "lat": f"{lat:.4f}", "lon": f"{lon:.4f}", "codes": kept}
print(f"municipality by {dict(how)}; {len(by_locality) - len(out)} postorter dropped", file=sys.stderr)
return out
def streets(segments, codes, localities, per_locality):
"""The names with most segments per locality, each segment at its nearest code centroid."""
nearest = Nearest((r["lat"], r["lon"], r["locality"]) for r in codes if r["lat"] is not None and r["locality"] in localities)
count = collections.Counter()
for name, lat, lon in segments:
count[(nearest.find(lat, lon), name)] += 1
of = collections.defaultdict(list)
for (locality, name), n in count.items():
of[locality].append((n, name))
return {locality: [{"name": name, "locality": locality, "segments": n} for n, name in sorted(named, key=lambda s: (-s[0], s[1]))[:per_locality]] for locality, named in of.items()}
def main():
@@ -216,48 +231,19 @@ def main():
regions, municipalities = scb_codes(cache)
population = scb_population(cache)
tatorter = scb_tatorter(cache)
codes = geonames(cache)
by_locality = collections.defaultdict(list)
for r in codes:
by_locality[r["locality"]].append(r)
localities, how = {}, collections.Counter()
for name, rows in by_locality.items():
municipality, method = municipality_of(name, rows, tatorter, municipalities)
how[method] += 1
kept = street_delivery(name, [r["code"].replace(" ", "") for r in rows])
with_point = [r for r in rows if r["lat"] is not None]
if municipality is None or not kept or not with_point or not well_cased(name):
continue
lat = sum(r["lat"] for r in with_point) / len(with_point)
lon = sum(r["lon"] for r in with_point) / len(with_point)
localities[name] = {"name": name, "municipality": municipality, "population": population_of(name, municipality, tatorter, municipalities, population), "lat": f"{lat:.4f}", "lon": f"{lon:.4f}", "codes": kept}
print(f"municipality by {dict(how)}; {len(by_locality) - len(localities)} postorter dropped", file=sys.stderr)
centroids = [(r["lat"], r["lon"], r["locality"]) for r in codes if r["lat"] is not None and r["locality"] in localities]
nearest_locality = Nearest(centroids)
segments = collections.Counter()
for name, lat, lon in nvdb_segments(cache, key):
segments[(nearest_locality.find(lat, lon), name)] += 1
streets_of = collections.defaultdict(list)
for (locality, name), n in segments.items():
streets_of[locality].append((n, name))
streets = []
for locality in sorted(streets_of):
for n, name in sorted(streets_of[locality], key=lambda s: (-s[0], s[1]))[: a.streets_per_locality]:
streets.append({"name": name, "locality": locality, "segments": n})
for name in [l for l in localities if l not in streets_of]:
del localities[name]
empty = sorted(m for m in municipalities if not any(l["municipality"] == m for l in localities.values()))
places = localities(codes, scb_tatorter(cache), municipalities, population)
named = streets(nvdb_segments(cache, key), codes, places, a.streets_per_locality)
places = {name: l for name, l in places.items() if name in named}
empty = sorted(m for m in municipalities if not any(l["municipality"] == m for l in places.values()))
if empty:
sys.exit(f"municipalities without a locality: {empty}")
write(out / "region.tsv", ["code", "name", "population", "timezone"], [{"code": c, "name": n, "population": population[c], "timezone": TIMEZONE} for c, n in sorted(regions.items())])
write(out / "municipality.tsv", ["code", "name", "region", "population"], [{"code": c, "name": n, "region": c[:2], "population": population[c]} for c, n in sorted(municipalities.items())])
write(out / "locality.tsv", ["name", "municipality", "population", "lat", "lon"], [l for _, l in sorted(localities.items())])
write(out / "postal-code.tsv", ["code", "locality"], sorted(({"code": f"{c[:3]} {c[3:]}", "locality": l["name"]} for l in localities.values() for c in l["codes"]), key=lambda r: r["code"]))
write(out / "street.tsv", ["name", "locality", "segments"], streets)
tsv.write(out / "region.tsv", ["code", "name", "population", "timezone"], [{"code": c, "name": n, "population": population[c], "timezone": TIMEZONE} for c, n in sorted(regions.items())])
tsv.write(out / "municipality.tsv", ["code", "name", "region", "population"], [{"code": c, "name": n, "region": c[:2], "population": population[c]} for c, n in sorted(municipalities.items())])
tsv.write(out / "locality.tsv", ["name", "municipality", "population", "lat", "lon"], [l for _, l in sorted(places.items())])
tsv.write(out / "postal-code.tsv", ["code", "locality"], sorted(({"code": f"{c[:3]} {c[3:]}", "locality": l["name"]} for l in places.values() for c in l["codes"]), key=lambda r: r["code"]))
tsv.write(out / "street.tsv", ["name", "locality", "segments"], [s for locality in sorted(named) for s in named[locality]])
if __name__ == "__main__":
+70 -89
View File
@@ -16,11 +16,11 @@ import io
import re
import struct
import sys
import time
import urllib.request
import zipfile
from pathlib import Path
import tsv
GAZETTEER = "https://www2.census.gov/geo/docs/maps-data/data/gazetteer/2026_Gazetteer/2026_Gaz_{}_national.zip"
POPULATION = "https://www2.census.gov/programs-surveys/popest/datasets/2020-2025/{}"
STATES = POPULATION.format("state/totals/NST-EST2025-ALLDATA.csv")
@@ -35,6 +35,7 @@ CDP = "57"
SUFFIX = re.compile(r" (city and borough|city|town|village|borough|municipality|comunidad|zona urbana|metropolitan government|metro government|consolidated government|unified government|urban county|corporation|plantation)( \(balance\))?$")
# Places whose Census name is a merged government's; the postal city is what an address carries.
NAMES = {"1303440": "Athens", "1304204": "Augusta", "1349008": "Macon", "2148006": "Louisville", "3011397": "Butte", "4732742": "Hartsville", "4752006": "Nashville"}
# The predominant zone of each state.
TIMEZONES = {
"AK": "America/Anchorage", "AL": "America/Chicago", "AR": "America/Chicago", "AZ": "America/Phoenix",
"CA": "America/Los_Angeles", "CO": "America/Denver", "CT": "America/New_York", "DC": "America/New_York",
@@ -52,24 +53,6 @@ TIMEZONES = {
}
def fetch(url, cache, name, magic=b""):
path = cache / name
for attempt in range(1, 6):
if path.exists():
return path.read_bytes()
req = urllib.request.Request(url, headers={"User-Agent": "fejkdata data-import"})
try:
with urllib.request.urlopen(req, timeout=600) as r:
data = r.read()
except OSError:
data = b""
if data.startswith(magic) and b"Request Rejected" not in data[:512]:
path.write_bytes(data)
else:
time.sleep(10 * attempt)
sys.exit(f"{url}: no valid download in 5 attempts")
def text(data):
try:
return data.decode("utf-8-sig")
@@ -78,14 +61,14 @@ def text(data):
def gazetteer(cache, kind):
z = zipfile.ZipFile(io.BytesIO(fetch(GAZETTEER.format(kind), cache, f"gaz_{kind}.zip", magic=b"PK")))
z = zipfile.ZipFile(io.BytesIO(tsv.fetch(GAZETTEER.format(kind), cache, f"gaz_{kind}.zip", magic=b"PK")))
rows = text(z.read(z.namelist()[0])).splitlines()
header = [h.strip() for h in rows[0].split("|")]
return [dict(zip(header, (c.strip() for c in row.split("|")))) for row in rows[1:]]
def csv_rows(cache, url, name):
return list(csv.DictReader(io.StringIO(text(fetch(url, cache, name)))))
return list(csv.DictReader(io.StringIO(text(tsv.fetch(url, cache, name)))))
def dbf_rows(data, wanted):
@@ -111,7 +94,7 @@ def dbf_rows(data, wanted):
def tiger_zip(cache, kind, county):
return fetch(TIGER.format(kind.upper(), county, kind), cache, f"tl_{county}_{kind}.zip", magic=b"PK")
return tsv.fetch(TIGER.format(kind.upper(), county, kind), cache, f"tl_{county}_{kind}.zip", magic=b"PK")
def tiger(cache, kind, county, wanted):
@@ -128,14 +111,56 @@ def place_name(geoid, name):
return stripped
def write(path, columns, rows):
lines = ["\t".join(columns)]
for row in rows:
cells = [str(row[c]) for c in columns]
assert not any(re.search(r"[\t\n{}]", c) for c in cells), row
lines.append("\t".join(cells))
path.write_text("\n".join(lines) + "\n", encoding="utf-8")
print(f"{path}: {len(rows)} rows", file=sys.stderr)
def localities(cache, min_population, counties):
"""Each shipped place with its county, population and centroid."""
place_population, county_part = {}, collections.defaultdict(list)
for r in csv_rows(cache, PLACES, "sub-est2025.csv"):
if r["SUMLEV"] == "162":
place_population[r["STATE"] + r["PLACE"]] = int(r[ESTIMATE])
elif r["SUMLEV"] == "157":
county_part[r["STATE"] + r["PLACE"]].append((int(r[ESTIMATE]), r["STATE"] + r["COUNTY"]))
out = {}
for r in gazetteer(cache, "place"):
geoid, population = r["GEOID"], place_population.get(r["GEOID"], 0)
if r["FUNCSTAT"] not in "AFN" or r["LSAD"] == CDP or population < min_population or not county_part.get(geoid):
continue
county = max(county_part[geoid])[1]
if county not in counties:
print(f"{geoid} {r['NAME']}: county {county} unknown, dropped", file=sys.stderr)
continue
out[geoid] = {"code": geoid, "name": place_name(geoid, r["NAME"]), "municipality": county, "population": population, "lat": r["INTPTLAT"], "lon": r["INTPTLONG"]}
return out
def postal_codes(cache, localities):
"""Each ZCTA and the shipped place holding most of its land."""
parts = {}
for r in csv.DictReader(io.StringIO(text(tsv.fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"):
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"] in localities:
parts.setdefault(r["GEOID_ZCTA5_20"], []).append((int(r["AREALAND_PART"]), r["GEOID_PLACE_20"]))
return {zcta: max(p)[1] for zcta, p in parts.items()}
def streets(cache, counties, locality_of_zcta, per_locality):
"""The names with most TIGER address ranges per place, and the address ranges per ZCTA."""
with concurrent.futures.ThreadPoolExecutor(3) as pool:
list(pool.map(lambda c: (tiger_zip(cache, "addr", c), tiger_zip(cache, "featnames", c)), counties))
addresses, count = collections.Counter(), collections.Counter()
for county in counties:
zips = collections.defaultdict(set)
for r in tiger(cache, "addr", county, {"TLID", "ZIP"}):
if r["ZIP"] in locality_of_zcta:
zips[r["TLID"]].add(r["ZIP"])
addresses[r["ZIP"]] += 1
for r in tiger(cache, "featnames", county, {"TLID", "FULLNAME", "PAFLAG"}):
if r["PAFLAG"] == "P" and r["FULLNAME"]:
for z in zips.get(r["TLID"], ()):
count[(locality_of_zcta[z], r["FULLNAME"])] += 1
of = collections.defaultdict(list)
for (locality, name), n in count.items():
of[locality].append((n, name))
named = {locality: [{"name": name, "locality": locality, "addresses": n} for n, name in sorted(ranked, key=lambda s: (-s[0], s[1]))[:per_locality]] for locality, ranked in of.items()}
return named, addresses
def main():
@@ -150,17 +175,8 @@ def main():
out.mkdir(parents=True, exist_ok=True)
state_population = {r["STATE"]: r[ESTIMATE] for r in csv_rows(cache, STATES, "nst-est2025.csv") if r["SUMLEV"] == "040"}
regions = {r["USPS"]: {"abbr": r["USPS"], "code": r["GEOID"], "name": r["NAME"], "population": state_population[r["GEOID"]], "timezone": TIMEZONES[r["USPS"]]} for r in gazetteer(cache, "state")}
county_population = {r["STATE"] + r["COUNTY"]: r[ESTIMATE] for r in csv_rows(cache, COUNTIES, "co-est2025.csv") if r["SUMLEV"] == "050"}
place_population, county_part = {}, collections.defaultdict(list)
for r in csv_rows(cache, PLACES, "sub-est2025.csv"):
if r["SUMLEV"] == "162":
place_population[r["STATE"] + r["PLACE"]] = int(r[ESTIMATE])
elif r["SUMLEV"] == "157":
county_part[r["STATE"] + r["PLACE"]].append((int(r[ESTIMATE]), r["STATE"] + r["COUNTY"]))
regions = {}
for r in gazetteer(cache, "state"):
regions[r["USPS"]] = {"abbr": r["USPS"], "code": r["GEOID"], "name": r["NAME"], "population": state_population[r["GEOID"]], "timezone": TIMEZONES[r["USPS"]]}
counties, unestimated = {}, collections.Counter()
for r in gazetteer(cache, "counties"):
if r["GEOID"] not in county_population:
@@ -168,56 +184,21 @@ def main():
continue
counties[r["GEOID"]] = {"code": r["GEOID"], "name": r["NAME"], "region": r["USPS"], "population": county_population[r["GEOID"]]}
print(f"counties without a population estimate, dropped: {dict(unestimated)}", file=sys.stderr)
localities = {}
for r in gazetteer(cache, "place"):
geoid, population = r["GEOID"], place_population.get(r["GEOID"], 0)
if r["FUNCSTAT"] not in "AFN" or r["LSAD"] == CDP or population < a.min_population or not county_part.get(geoid):
continue
county = max(county_part[geoid])[1]
if county not in counties:
print(f"{geoid} {r['NAME']}: county {county} unknown, dropped", file=sys.stderr)
continue
localities[geoid] = {"code": geoid, "name": place_name(geoid, r["NAME"]), "municipality": county, "population": population, "lat": r["INTPTLAT"], "lon": r["INTPTLONG"]}
zcta_of = {}
for r in csv.DictReader(io.StringIO(text(fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"):
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"] in localities:
zcta_of.setdefault(r["GEOID_ZCTA5_20"], []).append((int(r["AREALAND_PART"]), r["GEOID_PLACE_20"]))
locality_of_zcta = {zcta: max(parts)[1] for zcta, parts in zcta_of.items()}
needed = sorted({l["municipality"] for l in localities.values()})
with concurrent.futures.ThreadPoolExecutor(3) as pool:
list(pool.map(lambda c: (tiger_zip(cache, "addr", c), tiger_zip(cache, "featnames", c)), needed))
addresses, streets_of = collections.Counter(), collections.Counter()
for county in needed:
zips = collections.defaultdict(set)
for r in tiger(cache, "addr", county, {"TLID", "ZIP"}):
if r["ZIP"] in locality_of_zcta:
zips[r["TLID"]].add(r["ZIP"])
addresses[r["ZIP"]] += 1
for r in tiger(cache, "featnames", county, {"TLID", "FULLNAME", "PAFLAG"}):
if r["PAFLAG"] == "P" and r["FULLNAME"]:
for z in zips.get(r["TLID"], ()):
streets_of[(locality_of_zcta[z], r["FULLNAME"])] += 1
by_locality = collections.defaultdict(list)
for (locality, name), n in streets_of.items():
by_locality[locality].append((n, name))
streets = []
for locality in sorted(by_locality):
for n, name in sorted(by_locality[locality], key=lambda s: (-s[0], s[1]))[: a.streets_per_locality]:
streets.append({"name": name, "locality": locality, "addresses": n})
for geoid in [l for l in localities if l not in by_locality]:
print(f"{geoid} {localities[geoid]['name']}: no streets, dropped", file=sys.stderr)
del localities[geoid]
postal_codes = [{"code": z, "locality": l, "addresses": addresses[z]} for z, l in sorted(locality_of_zcta.items()) if addresses[z] and l in localities]
kept_counties = {l["municipality"] for l in localities.values()}
places = localities(cache, a.min_population, counties)
locality_of_zcta = postal_codes(cache, places)
named, addresses = streets(cache, sorted({l["municipality"] for l in places.values()}), locality_of_zcta, a.streets_per_locality)
for geoid in [l for l in places if l not in named]:
print(f"{geoid} {places[geoid]['name']}: no streets, dropped", file=sys.stderr)
del places[geoid]
kept_counties = {l["municipality"] for l in places.values()}
kept_regions = {counties[c]["region"] for c in kept_counties}
write(out / "region.tsv", ["abbr", "code", "name", "population", "timezone"], [r for _, r in sorted(regions.items()) if r["abbr"] in kept_regions])
write(out / "municipality.tsv", ["code", "name", "region", "population"], [c for _, c in sorted(counties.items()) if c["code"] in kept_counties])
write(out / "locality.tsv", ["code", "name", "municipality", "population", "lat", "lon"], [l for _, l in sorted(localities.items())])
write(out / "postal-code.tsv", ["code", "locality", "addresses"], postal_codes)
write(out / "street.tsv", ["name", "locality", "addresses"], streets)
tsv.write(out / "region.tsv", ["abbr", "code", "name", "population", "timezone"], [r for _, r in sorted(regions.items()) if r["abbr"] in kept_regions])
tsv.write(out / "municipality.tsv", ["code", "name", "region", "population"], [c for _, c in sorted(counties.items()) if c["code"] in kept_counties])
tsv.write(out / "locality.tsv", ["code", "name", "municipality", "population", "lat", "lon"], [l for _, l in sorted(places.items())])
tsv.write(out / "postal-code.tsv", ["code", "locality", "addresses"], [{"code": z, "locality": l, "addresses": addresses[z]} for z, l in sorted(locality_of_zcta.items()) if addresses[z] and l in places])
tsv.write(out / "street.tsv", ["name", "locality", "addresses"], [s for locality in sorted(named) for s in named[locality]])
if __name__ == "__main__":
+39
View File
@@ -0,0 +1,39 @@
"""A source fetched once into the cache, and a table written as the loader admits it."""
import re
import sys
import time
import urllib.request
from pathlib import Path
def fetch(source, cache, name, magic=b"", data=None, headers=None):
"""The bytes of a URL, downloaded into cache/name once, or of a local file."""
if not re.match(r"^https?://", source):
return Path(source).read_bytes()
path = Path(cache) / name
for attempt in range(1, 6):
if path.exists():
return path.read_bytes()
req = urllib.request.Request(source, data=data, headers={"User-Agent": "fejkdata data-import", **(headers or {})})
try:
with urllib.request.urlopen(req, timeout=600) as r:
body = r.read()
except OSError:
body = b""
if body.startswith(magic) and b"Request Rejected" not in body[:512]:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(body)
else:
time.sleep(10 * attempt)
sys.exit(f"{source}: no valid download in 5 attempts")
def write(path, columns, rows):
"""Write the rows as a TSV; every cell must be non-empty and free of tabs, newlines and braces."""
lines = ["\t".join(columns)]
for row in rows:
cells = [str(row[c]) for c in columns]
assert all(cells) and not any(re.search(r"[\t\n{}]", c) for c in cells), row
lines.append("\t".join(cells))
Path(path).write_text("\n".join(lines) + "\n", encoding="utf-8")
print(f"{path}: {len(lines) - 1} rows", file=sys.stderr)