Share one fetch and one TSV writer across the import scripts, and name the geo scripts' stages
This commit is contained in:
+7
-22
@@ -1,30 +1,24 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""Rebuild data/misc/country.tsv from datasets/country-codes (PDDL).
|
"""Rebuild data/misc/country.tsv from datasets/country-codes (PDDL).
|
||||||
|
|
||||||
data-import/country.py [--source URL_OR_FILE] [--out FILE]
|
data-import/country.py [--source URL_OR_FILE] [--cache DIR] [--out FILE]
|
||||||
"""
|
"""
|
||||||
import argparse
|
import argparse
|
||||||
import csv
|
import csv
|
||||||
import io
|
import io
|
||||||
import re
|
import re
|
||||||
import sys
|
|
||||||
import urllib.request
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
|
import tsv
|
||||||
|
|
||||||
SOURCE = "https://raw.githubusercontent.com/datasets/country-codes/main/data/country-codes.csv"
|
SOURCE = "https://raw.githubusercontent.com/datasets/country-codes/main/data/country-codes.csv"
|
||||||
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "country.tsv"
|
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "country.tsv"
|
||||||
|
CACHE = Path(__file__).resolve().parent / "cache"
|
||||||
COLUMNS = ["alpha2", "alpha3", "calling-code", "capital", "currency", "flag", "languages", "name", "numeric", "tld"]
|
COLUMNS = ["alpha2", "alpha3", "calling-code", "capital", "currency", "flag", "languages", "name", "numeric", "tld"]
|
||||||
# Gaps in the source, keyed by alpha2.
|
# Gaps in the source, keyed by alpha2.
|
||||||
FIXUPS = {"TR": {"currency": "TRY"}}
|
FIXUPS = {"TR": {"currency": "TRY"}}
|
||||||
|
|
||||||
|
|
||||||
def read(source):
|
|
||||||
if re.match(r"^https?://", source):
|
|
||||||
with urllib.request.urlopen(source, timeout=60) as r:
|
|
||||||
return r.read().decode("utf-8")
|
|
||||||
return Path(source).read_text(encoding="utf-8")
|
|
||||||
|
|
||||||
|
|
||||||
def flag(alpha2):
|
def flag(alpha2):
|
||||||
return "".join(chr(0x1F1E6 + ord(c) - ord("A")) for c in alpha2)
|
return "".join(chr(0x1F1E6 + ord(c) - ord("A")) for c in alpha2)
|
||||||
|
|
||||||
@@ -58,23 +52,14 @@ def rows(text):
|
|||||||
yield row
|
yield row
|
||||||
|
|
||||||
|
|
||||||
def write(out, table):
|
|
||||||
lines = ["\t".join(COLUMNS)]
|
|
||||||
for row in sorted(table, key=lambda r: r["alpha2"]):
|
|
||||||
cells = [row[c] for c in COLUMNS]
|
|
||||||
assert not any("\t" in c or "\n" in c for c in cells), row
|
|
||||||
lines.append("\t".join(cells))
|
|
||||||
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
|
|
||||||
return len(lines) - 1
|
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||||
|
p.add_argument("--cache", default=str(CACHE))
|
||||||
p.add_argument("--source", default=SOURCE)
|
p.add_argument("--source", default=SOURCE)
|
||||||
p.add_argument("--out", default=str(OUT))
|
p.add_argument("--out", default=str(OUT))
|
||||||
a = p.parse_args()
|
a = p.parse_args()
|
||||||
n = write(a.out, rows(read(a.source)))
|
table = rows(tsv.fetch(a.source, a.cache, "country-codes.csv").decode("utf-8"))
|
||||||
print(f"{a.out}: {n} rows", file=sys.stderr)
|
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["alpha2"]))
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|||||||
+8
-23
@@ -1,35 +1,28 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""Rebuild data/misc/currency.tsv from datasets/currency-codes (PDDL) and CLDR's symbols (Unicode).
|
"""Rebuild data/misc/currency.tsv from datasets/currency-codes (PDDL) and CLDR's symbols (Unicode).
|
||||||
|
|
||||||
data-import/currency.py [--source URL_OR_FILE] [--symbols URL_OR_FILE ...] [--out FILE]
|
data-import/currency.py [--source URL_OR_FILE] [--symbols URL_OR_FILE ...] [--cache DIR] [--out FILE]
|
||||||
|
|
||||||
The symbols come from the first locale file that has one, narrow symbols before wide.
|
The symbols come from the first locale file that has one, narrow symbols before wide.
|
||||||
"""
|
"""
|
||||||
import argparse
|
import argparse
|
||||||
import csv
|
import csv
|
||||||
import io
|
import io
|
||||||
import re
|
|
||||||
import sys
|
|
||||||
import urllib.request
|
|
||||||
import xml.etree.ElementTree as ET
|
import xml.etree.ElementTree as ET
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
|
import tsv
|
||||||
|
|
||||||
SOURCE = "https://raw.githubusercontent.com/datasets/currency-codes/main/data/codes-all.csv"
|
SOURCE = "https://raw.githubusercontent.com/datasets/currency-codes/main/data/codes-all.csv"
|
||||||
SYMBOLS = [
|
SYMBOLS = [
|
||||||
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/en.xml",
|
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/en.xml",
|
||||||
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/root.xml",
|
"https://raw.githubusercontent.com/unicode-org/cldr/main/common/main/root.xml",
|
||||||
]
|
]
|
||||||
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "currency.tsv"
|
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "currency.tsv"
|
||||||
|
CACHE = Path(__file__).resolve().parent / "cache"
|
||||||
COLUMNS = ["code", "decimals", "name", "numeric", "symbol"]
|
COLUMNS = ["code", "decimals", "name", "numeric", "symbol"]
|
||||||
|
|
||||||
|
|
||||||
def read(source):
|
|
||||||
if re.match(r"^https?://", source):
|
|
||||||
with urllib.request.urlopen(source, timeout=60) as r:
|
|
||||||
return r.read().decode("utf-8")
|
|
||||||
return Path(source).read_text(encoding="utf-8")
|
|
||||||
|
|
||||||
|
|
||||||
def symbols(xml_texts):
|
def symbols(xml_texts):
|
||||||
"""CLDR's symbol per code: the first locale's narrow symbol, else the first locale's wide one."""
|
"""CLDR's symbol per code: the first locale's narrow symbol, else the first locale's wide one."""
|
||||||
narrow, wide = {}, {}
|
narrow, wide = {}, {}
|
||||||
@@ -58,24 +51,16 @@ def rows(text, symbol):
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
def write(out, table):
|
|
||||||
lines = ["\t".join(COLUMNS)]
|
|
||||||
for row in sorted(table, key=lambda r: r["code"]):
|
|
||||||
cells = [row[c] for c in COLUMNS]
|
|
||||||
assert all(cells) and not any("\t" in c or "\n" in c for c in cells), row
|
|
||||||
lines.append("\t".join(cells))
|
|
||||||
Path(out).write_text("\n".join(lines) + "\n", encoding="utf-8")
|
|
||||||
return len(lines) - 1
|
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||||
|
p.add_argument("--cache", default=str(CACHE))
|
||||||
p.add_argument("--source", default=SOURCE)
|
p.add_argument("--source", default=SOURCE)
|
||||||
p.add_argument("--symbols", nargs="+", default=SYMBOLS)
|
p.add_argument("--symbols", nargs="+", default=SYMBOLS)
|
||||||
p.add_argument("--out", default=str(OUT))
|
p.add_argument("--out", default=str(OUT))
|
||||||
a = p.parse_args()
|
a = p.parse_args()
|
||||||
n = write(a.out, rows(read(a.source), symbols(read(s) for s in a.symbols)))
|
symbol = symbols(tsv.fetch(s, a.cache, Path(s).name).decode("utf-8") for s in a.symbols)
|
||||||
print(f"{a.out}: {n} rows", file=sys.stderr)
|
table = rows(tsv.fetch(a.source, a.cache, "codes-all.csv").decode("utf-8"), symbol)
|
||||||
|
tsv.write(a.out, COLUMNS, sorted(table, key=lambda r: r["code"]))
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|||||||
+46
-60
@@ -23,6 +23,8 @@ import xml.etree.ElementTree as ET
|
|||||||
import zipfile
|
import zipfile
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
|
import tsv
|
||||||
|
|
||||||
CODES = "https://www.scb.se/contentassets/7a89e48960f741e08918e489ea36354a/kommunlankod-2026.xlsx"
|
CODES = "https://www.scb.se/contentassets/7a89e48960f741e08918e489ea36354a/kommunlankod-2026.xlsx"
|
||||||
POPULATION = "https://api.scb.se/OV0104/v1/doris/sv/ssd/START/BE/BE0101/BE0101A/BefolkningNy"
|
POPULATION = "https://api.scb.se/OV0104/v1/doris/sv/ssd/START/BE/BE0101/BE0101A/BefolkningNy"
|
||||||
POPULATION_QUERY = {
|
POPULATION_QUERY = {
|
||||||
@@ -45,15 +47,6 @@ UNMATCHED_POPULATION = 200
|
|||||||
XLSX_NS = {"m": "http://schemas.openxmlformats.org/spreadsheetml/2006/main"}
|
XLSX_NS = {"m": "http://schemas.openxmlformats.org/spreadsheetml/2006/main"}
|
||||||
|
|
||||||
|
|
||||||
def fetch(url, cache, name, data=None, headers=None):
|
|
||||||
path = cache / name
|
|
||||||
if not path.exists():
|
|
||||||
req = urllib.request.Request(url, data=data, headers=headers or {})
|
|
||||||
with urllib.request.urlopen(req, timeout=300) as r:
|
|
||||||
path.write_bytes(r.read())
|
|
||||||
return path.read_bytes()
|
|
||||||
|
|
||||||
|
|
||||||
def xlsx_rows(data):
|
def xlsx_rows(data):
|
||||||
z = zipfile.ZipFile(io.BytesIO(data))
|
z = zipfile.ZipFile(io.BytesIO(data))
|
||||||
strings = ["".join(t.text or "" for t in si.iter("{%s}t" % XLSX_NS["m"])) for si in ET.fromstring(z.read("xl/sharedStrings.xml")).findall("m:si", XLSX_NS)]
|
strings = ["".join(t.text or "" for t in si.iter("{%s}t" % XLSX_NS["m"])) for si in ET.fromstring(z.read("xl/sharedStrings.xml")).findall("m:si", XLSX_NS)]
|
||||||
@@ -68,7 +61,7 @@ def xlsx_rows(data):
|
|||||||
|
|
||||||
def scb_codes(cache):
|
def scb_codes(cache):
|
||||||
regions, municipalities = {}, {}
|
regions, municipalities = {}, {}
|
||||||
for cells in xlsx_rows(fetch(CODES, cache, "kommunlankod.xlsx")):
|
for cells in xlsx_rows(tsv.fetch(CODES, cache, "kommunlankod.xlsx", magic=b"PK")):
|
||||||
if len(cells) < 2 or not re.fullmatch(r"\d{2}|\d{4}", cells[0]):
|
if len(cells) < 2 or not re.fullmatch(r"\d{2}|\d{4}", cells[0]):
|
||||||
continue
|
continue
|
||||||
(regions if len(cells[0]) == 2 else municipalities)[cells[0]] = cells[1].strip()
|
(regions if len(cells[0]) == 2 else municipalities)[cells[0]] = cells[1].strip()
|
||||||
@@ -77,12 +70,12 @@ def scb_codes(cache):
|
|||||||
|
|
||||||
def scb_population(cache):
|
def scb_population(cache):
|
||||||
body = json.dumps(POPULATION_QUERY).encode()
|
body = json.dumps(POPULATION_QUERY).encode()
|
||||||
data = fetch(POPULATION, cache, "befolkning.json", data=body, headers={"Content-Type": "application/json"})
|
data = tsv.fetch(POPULATION, cache, "befolkning.json", data=body, headers={"Content-Type": "application/json"})
|
||||||
return {row["key"][0]: row["values"][0] for row in json.loads(data.decode("utf-8-sig"))["data"]}
|
return {row["key"][0]: row["values"][0] for row in json.loads(data.decode("utf-8-sig"))["data"]}
|
||||||
|
|
||||||
|
|
||||||
def scb_tatorter(cache):
|
def scb_tatorter(cache):
|
||||||
text = fetch(TATORTER, cache, "tatorter.csv").decode("utf-8")
|
text = tsv.fetch(TATORTER, cache, "tatorter.csv").decode("utf-8")
|
||||||
by_name = collections.defaultdict(list)
|
by_name = collections.defaultdict(list)
|
||||||
for r in csv.DictReader(io.StringIO(text)):
|
for r in csv.DictReader(io.StringIO(text)):
|
||||||
by_name[r["tatort"]].append((r["kommun"], int(r["bef"])))
|
by_name[r["tatort"]].append((r["kommun"], int(r["bef"])))
|
||||||
@@ -90,7 +83,7 @@ def scb_tatorter(cache):
|
|||||||
|
|
||||||
|
|
||||||
def geonames(cache):
|
def geonames(cache):
|
||||||
z = zipfile.ZipFile(io.BytesIO(fetch(POSTAL_CODES, cache, "SE.zip")))
|
z = zipfile.ZipFile(io.BytesIO(tsv.fetch(POSTAL_CODES, cache, "SE.zip", magic=b"PK")))
|
||||||
rows = []
|
rows = []
|
||||||
for line in z.read("SE.txt").decode("utf-8").splitlines():
|
for line in z.read("SE.txt").decode("utf-8").splitlines():
|
||||||
f = line.split("\t")
|
f = line.split("\t")
|
||||||
@@ -190,14 +183,36 @@ def well_cased(name):
|
|||||||
return all(part[:1].isupper() and (len(part) == 1 or not part.isupper()) for part in re.split(r"[ -]", name))
|
return all(part[:1].isupper() and (len(part) == 1 or not part.isupper()) for part in re.split(r"[ -]", name))
|
||||||
|
|
||||||
|
|
||||||
def write(path, columns, rows):
|
def localities(codes, tatorter, municipalities, population):
|
||||||
lines = ["\t".join(columns)]
|
"""Each postort with its municipality, weight, centroid and street-delivery codes."""
|
||||||
for row in rows:
|
by_locality = collections.defaultdict(list)
|
||||||
cells = [str(row[c]) for c in columns]
|
for r in codes:
|
||||||
assert not any(re.search(r"[\t\n{}]", c) for c in cells), row
|
by_locality[r["locality"]].append(r)
|
||||||
lines.append("\t".join(cells))
|
out, how = {}, collections.Counter()
|
||||||
path.write_text("\n".join(lines) + "\n", encoding="utf-8")
|
for name, rows in by_locality.items():
|
||||||
print(f"{path}: {len(rows)} rows", file=sys.stderr)
|
municipality, method = municipality_of(name, rows, tatorter, municipalities)
|
||||||
|
how[method] += 1
|
||||||
|
kept = street_delivery(name, [r["code"].replace(" ", "") for r in rows])
|
||||||
|
with_point = [r for r in rows if r["lat"] is not None]
|
||||||
|
if municipality is None or not kept or not with_point or not well_cased(name):
|
||||||
|
continue
|
||||||
|
lat = sum(r["lat"] for r in with_point) / len(with_point)
|
||||||
|
lon = sum(r["lon"] for r in with_point) / len(with_point)
|
||||||
|
out[name] = {"name": name, "municipality": municipality, "population": population_of(name, municipality, tatorter, municipalities, population), "lat": f"{lat:.4f}", "lon": f"{lon:.4f}", "codes": kept}
|
||||||
|
print(f"municipality by {dict(how)}; {len(by_locality) - len(out)} postorter dropped", file=sys.stderr)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def streets(segments, codes, localities, per_locality):
|
||||||
|
"""The names with most segments per locality, each segment at its nearest code centroid."""
|
||||||
|
nearest = Nearest((r["lat"], r["lon"], r["locality"]) for r in codes if r["lat"] is not None and r["locality"] in localities)
|
||||||
|
count = collections.Counter()
|
||||||
|
for name, lat, lon in segments:
|
||||||
|
count[(nearest.find(lat, lon), name)] += 1
|
||||||
|
of = collections.defaultdict(list)
|
||||||
|
for (locality, name), n in count.items():
|
||||||
|
of[locality].append((n, name))
|
||||||
|
return {locality: [{"name": name, "locality": locality, "segments": n} for n, name in sorted(named, key=lambda s: (-s[0], s[1]))[:per_locality]] for locality, named in of.items()}
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
@@ -216,48 +231,19 @@ def main():
|
|||||||
|
|
||||||
regions, municipalities = scb_codes(cache)
|
regions, municipalities = scb_codes(cache)
|
||||||
population = scb_population(cache)
|
population = scb_population(cache)
|
||||||
tatorter = scb_tatorter(cache)
|
|
||||||
codes = geonames(cache)
|
codes = geonames(cache)
|
||||||
|
places = localities(codes, scb_tatorter(cache), municipalities, population)
|
||||||
by_locality = collections.defaultdict(list)
|
named = streets(nvdb_segments(cache, key), codes, places, a.streets_per_locality)
|
||||||
for r in codes:
|
places = {name: l for name, l in places.items() if name in named}
|
||||||
by_locality[r["locality"]].append(r)
|
empty = sorted(m for m in municipalities if not any(l["municipality"] == m for l in places.values()))
|
||||||
localities, how = {}, collections.Counter()
|
|
||||||
for name, rows in by_locality.items():
|
|
||||||
municipality, method = municipality_of(name, rows, tatorter, municipalities)
|
|
||||||
how[method] += 1
|
|
||||||
kept = street_delivery(name, [r["code"].replace(" ", "") for r in rows])
|
|
||||||
with_point = [r for r in rows if r["lat"] is not None]
|
|
||||||
if municipality is None or not kept or not with_point or not well_cased(name):
|
|
||||||
continue
|
|
||||||
lat = sum(r["lat"] for r in with_point) / len(with_point)
|
|
||||||
lon = sum(r["lon"] for r in with_point) / len(with_point)
|
|
||||||
localities[name] = {"name": name, "municipality": municipality, "population": population_of(name, municipality, tatorter, municipalities, population), "lat": f"{lat:.4f}", "lon": f"{lon:.4f}", "codes": kept}
|
|
||||||
print(f"municipality by {dict(how)}; {len(by_locality) - len(localities)} postorter dropped", file=sys.stderr)
|
|
||||||
|
|
||||||
centroids = [(r["lat"], r["lon"], r["locality"]) for r in codes if r["lat"] is not None and r["locality"] in localities]
|
|
||||||
nearest_locality = Nearest(centroids)
|
|
||||||
segments = collections.Counter()
|
|
||||||
for name, lat, lon in nvdb_segments(cache, key):
|
|
||||||
segments[(nearest_locality.find(lat, lon), name)] += 1
|
|
||||||
streets_of = collections.defaultdict(list)
|
|
||||||
for (locality, name), n in segments.items():
|
|
||||||
streets_of[locality].append((n, name))
|
|
||||||
streets = []
|
|
||||||
for locality in sorted(streets_of):
|
|
||||||
for n, name in sorted(streets_of[locality], key=lambda s: (-s[0], s[1]))[: a.streets_per_locality]:
|
|
||||||
streets.append({"name": name, "locality": locality, "segments": n})
|
|
||||||
for name in [l for l in localities if l not in streets_of]:
|
|
||||||
del localities[name]
|
|
||||||
|
|
||||||
empty = sorted(m for m in municipalities if not any(l["municipality"] == m for l in localities.values()))
|
|
||||||
if empty:
|
if empty:
|
||||||
sys.exit(f"municipalities without a locality: {empty}")
|
sys.exit(f"municipalities without a locality: {empty}")
|
||||||
write(out / "region.tsv", ["code", "name", "population", "timezone"], [{"code": c, "name": n, "population": population[c], "timezone": TIMEZONE} for c, n in sorted(regions.items())])
|
|
||||||
write(out / "municipality.tsv", ["code", "name", "region", "population"], [{"code": c, "name": n, "region": c[:2], "population": population[c]} for c, n in sorted(municipalities.items())])
|
tsv.write(out / "region.tsv", ["code", "name", "population", "timezone"], [{"code": c, "name": n, "population": population[c], "timezone": TIMEZONE} for c, n in sorted(regions.items())])
|
||||||
write(out / "locality.tsv", ["name", "municipality", "population", "lat", "lon"], [l for _, l in sorted(localities.items())])
|
tsv.write(out / "municipality.tsv", ["code", "name", "region", "population"], [{"code": c, "name": n, "region": c[:2], "population": population[c]} for c, n in sorted(municipalities.items())])
|
||||||
write(out / "postal-code.tsv", ["code", "locality"], sorted(({"code": f"{c[:3]} {c[3:]}", "locality": l["name"]} for l in localities.values() for c in l["codes"]), key=lambda r: r["code"]))
|
tsv.write(out / "locality.tsv", ["name", "municipality", "population", "lat", "lon"], [l for _, l in sorted(places.items())])
|
||||||
write(out / "street.tsv", ["name", "locality", "segments"], streets)
|
tsv.write(out / "postal-code.tsv", ["code", "locality"], sorted(({"code": f"{c[:3]} {c[3:]}", "locality": l["name"]} for l in places.values() for c in l["codes"]), key=lambda r: r["code"]))
|
||||||
|
tsv.write(out / "street.tsv", ["name", "locality", "segments"], [s for locality in sorted(named) for s in named[locality]])
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|||||||
+70
-89
@@ -16,11 +16,11 @@ import io
|
|||||||
import re
|
import re
|
||||||
import struct
|
import struct
|
||||||
import sys
|
import sys
|
||||||
import time
|
|
||||||
import urllib.request
|
|
||||||
import zipfile
|
import zipfile
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
|
import tsv
|
||||||
|
|
||||||
GAZETTEER = "https://www2.census.gov/geo/docs/maps-data/data/gazetteer/2026_Gazetteer/2026_Gaz_{}_national.zip"
|
GAZETTEER = "https://www2.census.gov/geo/docs/maps-data/data/gazetteer/2026_Gazetteer/2026_Gaz_{}_national.zip"
|
||||||
POPULATION = "https://www2.census.gov/programs-surveys/popest/datasets/2020-2025/{}"
|
POPULATION = "https://www2.census.gov/programs-surveys/popest/datasets/2020-2025/{}"
|
||||||
STATES = POPULATION.format("state/totals/NST-EST2025-ALLDATA.csv")
|
STATES = POPULATION.format("state/totals/NST-EST2025-ALLDATA.csv")
|
||||||
@@ -35,6 +35,7 @@ CDP = "57"
|
|||||||
SUFFIX = re.compile(r" (city and borough|city|town|village|borough|municipality|comunidad|zona urbana|metropolitan government|metro government|consolidated government|unified government|urban county|corporation|plantation)( \(balance\))?$")
|
SUFFIX = re.compile(r" (city and borough|city|town|village|borough|municipality|comunidad|zona urbana|metropolitan government|metro government|consolidated government|unified government|urban county|corporation|plantation)( \(balance\))?$")
|
||||||
# Places whose Census name is a merged government's; the postal city is what an address carries.
|
# Places whose Census name is a merged government's; the postal city is what an address carries.
|
||||||
NAMES = {"1303440": "Athens", "1304204": "Augusta", "1349008": "Macon", "2148006": "Louisville", "3011397": "Butte", "4732742": "Hartsville", "4752006": "Nashville"}
|
NAMES = {"1303440": "Athens", "1304204": "Augusta", "1349008": "Macon", "2148006": "Louisville", "3011397": "Butte", "4732742": "Hartsville", "4752006": "Nashville"}
|
||||||
|
# The predominant zone of each state.
|
||||||
TIMEZONES = {
|
TIMEZONES = {
|
||||||
"AK": "America/Anchorage", "AL": "America/Chicago", "AR": "America/Chicago", "AZ": "America/Phoenix",
|
"AK": "America/Anchorage", "AL": "America/Chicago", "AR": "America/Chicago", "AZ": "America/Phoenix",
|
||||||
"CA": "America/Los_Angeles", "CO": "America/Denver", "CT": "America/New_York", "DC": "America/New_York",
|
"CA": "America/Los_Angeles", "CO": "America/Denver", "CT": "America/New_York", "DC": "America/New_York",
|
||||||
@@ -52,24 +53,6 @@ TIMEZONES = {
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
def fetch(url, cache, name, magic=b""):
|
|
||||||
path = cache / name
|
|
||||||
for attempt in range(1, 6):
|
|
||||||
if path.exists():
|
|
||||||
return path.read_bytes()
|
|
||||||
req = urllib.request.Request(url, headers={"User-Agent": "fejkdata data-import"})
|
|
||||||
try:
|
|
||||||
with urllib.request.urlopen(req, timeout=600) as r:
|
|
||||||
data = r.read()
|
|
||||||
except OSError:
|
|
||||||
data = b""
|
|
||||||
if data.startswith(magic) and b"Request Rejected" not in data[:512]:
|
|
||||||
path.write_bytes(data)
|
|
||||||
else:
|
|
||||||
time.sleep(10 * attempt)
|
|
||||||
sys.exit(f"{url}: no valid download in 5 attempts")
|
|
||||||
|
|
||||||
|
|
||||||
def text(data):
|
def text(data):
|
||||||
try:
|
try:
|
||||||
return data.decode("utf-8-sig")
|
return data.decode("utf-8-sig")
|
||||||
@@ -78,14 +61,14 @@ def text(data):
|
|||||||
|
|
||||||
|
|
||||||
def gazetteer(cache, kind):
|
def gazetteer(cache, kind):
|
||||||
z = zipfile.ZipFile(io.BytesIO(fetch(GAZETTEER.format(kind), cache, f"gaz_{kind}.zip", magic=b"PK")))
|
z = zipfile.ZipFile(io.BytesIO(tsv.fetch(GAZETTEER.format(kind), cache, f"gaz_{kind}.zip", magic=b"PK")))
|
||||||
rows = text(z.read(z.namelist()[0])).splitlines()
|
rows = text(z.read(z.namelist()[0])).splitlines()
|
||||||
header = [h.strip() for h in rows[0].split("|")]
|
header = [h.strip() for h in rows[0].split("|")]
|
||||||
return [dict(zip(header, (c.strip() for c in row.split("|")))) for row in rows[1:]]
|
return [dict(zip(header, (c.strip() for c in row.split("|")))) for row in rows[1:]]
|
||||||
|
|
||||||
|
|
||||||
def csv_rows(cache, url, name):
|
def csv_rows(cache, url, name):
|
||||||
return list(csv.DictReader(io.StringIO(text(fetch(url, cache, name)))))
|
return list(csv.DictReader(io.StringIO(text(tsv.fetch(url, cache, name)))))
|
||||||
|
|
||||||
|
|
||||||
def dbf_rows(data, wanted):
|
def dbf_rows(data, wanted):
|
||||||
@@ -111,7 +94,7 @@ def dbf_rows(data, wanted):
|
|||||||
|
|
||||||
|
|
||||||
def tiger_zip(cache, kind, county):
|
def tiger_zip(cache, kind, county):
|
||||||
return fetch(TIGER.format(kind.upper(), county, kind), cache, f"tl_{county}_{kind}.zip", magic=b"PK")
|
return tsv.fetch(TIGER.format(kind.upper(), county, kind), cache, f"tl_{county}_{kind}.zip", magic=b"PK")
|
||||||
|
|
||||||
|
|
||||||
def tiger(cache, kind, county, wanted):
|
def tiger(cache, kind, county, wanted):
|
||||||
@@ -128,14 +111,56 @@ def place_name(geoid, name):
|
|||||||
return stripped
|
return stripped
|
||||||
|
|
||||||
|
|
||||||
def write(path, columns, rows):
|
def localities(cache, min_population, counties):
|
||||||
lines = ["\t".join(columns)]
|
"""Each shipped place with its county, population and centroid."""
|
||||||
for row in rows:
|
place_population, county_part = {}, collections.defaultdict(list)
|
||||||
cells = [str(row[c]) for c in columns]
|
for r in csv_rows(cache, PLACES, "sub-est2025.csv"):
|
||||||
assert not any(re.search(r"[\t\n{}]", c) for c in cells), row
|
if r["SUMLEV"] == "162":
|
||||||
lines.append("\t".join(cells))
|
place_population[r["STATE"] + r["PLACE"]] = int(r[ESTIMATE])
|
||||||
path.write_text("\n".join(lines) + "\n", encoding="utf-8")
|
elif r["SUMLEV"] == "157":
|
||||||
print(f"{path}: {len(rows)} rows", file=sys.stderr)
|
county_part[r["STATE"] + r["PLACE"]].append((int(r[ESTIMATE]), r["STATE"] + r["COUNTY"]))
|
||||||
|
out = {}
|
||||||
|
for r in gazetteer(cache, "place"):
|
||||||
|
geoid, population = r["GEOID"], place_population.get(r["GEOID"], 0)
|
||||||
|
if r["FUNCSTAT"] not in "AFN" or r["LSAD"] == CDP or population < min_population or not county_part.get(geoid):
|
||||||
|
continue
|
||||||
|
county = max(county_part[geoid])[1]
|
||||||
|
if county not in counties:
|
||||||
|
print(f"{geoid} {r['NAME']}: county {county} unknown, dropped", file=sys.stderr)
|
||||||
|
continue
|
||||||
|
out[geoid] = {"code": geoid, "name": place_name(geoid, r["NAME"]), "municipality": county, "population": population, "lat": r["INTPTLAT"], "lon": r["INTPTLONG"]}
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def postal_codes(cache, localities):
|
||||||
|
"""Each ZCTA and the shipped place holding most of its land."""
|
||||||
|
parts = {}
|
||||||
|
for r in csv.DictReader(io.StringIO(text(tsv.fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"):
|
||||||
|
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"] in localities:
|
||||||
|
parts.setdefault(r["GEOID_ZCTA5_20"], []).append((int(r["AREALAND_PART"]), r["GEOID_PLACE_20"]))
|
||||||
|
return {zcta: max(p)[1] for zcta, p in parts.items()}
|
||||||
|
|
||||||
|
|
||||||
|
def streets(cache, counties, locality_of_zcta, per_locality):
|
||||||
|
"""The names with most TIGER address ranges per place, and the address ranges per ZCTA."""
|
||||||
|
with concurrent.futures.ThreadPoolExecutor(3) as pool:
|
||||||
|
list(pool.map(lambda c: (tiger_zip(cache, "addr", c), tiger_zip(cache, "featnames", c)), counties))
|
||||||
|
addresses, count = collections.Counter(), collections.Counter()
|
||||||
|
for county in counties:
|
||||||
|
zips = collections.defaultdict(set)
|
||||||
|
for r in tiger(cache, "addr", county, {"TLID", "ZIP"}):
|
||||||
|
if r["ZIP"] in locality_of_zcta:
|
||||||
|
zips[r["TLID"]].add(r["ZIP"])
|
||||||
|
addresses[r["ZIP"]] += 1
|
||||||
|
for r in tiger(cache, "featnames", county, {"TLID", "FULLNAME", "PAFLAG"}):
|
||||||
|
if r["PAFLAG"] == "P" and r["FULLNAME"]:
|
||||||
|
for z in zips.get(r["TLID"], ()):
|
||||||
|
count[(locality_of_zcta[z], r["FULLNAME"])] += 1
|
||||||
|
of = collections.defaultdict(list)
|
||||||
|
for (locality, name), n in count.items():
|
||||||
|
of[locality].append((n, name))
|
||||||
|
named = {locality: [{"name": name, "locality": locality, "addresses": n} for n, name in sorted(ranked, key=lambda s: (-s[0], s[1]))[:per_locality]] for locality, ranked in of.items()}
|
||||||
|
return named, addresses
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
@@ -150,17 +175,8 @@ def main():
|
|||||||
out.mkdir(parents=True, exist_ok=True)
|
out.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
state_population = {r["STATE"]: r[ESTIMATE] for r in csv_rows(cache, STATES, "nst-est2025.csv") if r["SUMLEV"] == "040"}
|
state_population = {r["STATE"]: r[ESTIMATE] for r in csv_rows(cache, STATES, "nst-est2025.csv") if r["SUMLEV"] == "040"}
|
||||||
|
regions = {r["USPS"]: {"abbr": r["USPS"], "code": r["GEOID"], "name": r["NAME"], "population": state_population[r["GEOID"]], "timezone": TIMEZONES[r["USPS"]]} for r in gazetteer(cache, "state")}
|
||||||
county_population = {r["STATE"] + r["COUNTY"]: r[ESTIMATE] for r in csv_rows(cache, COUNTIES, "co-est2025.csv") if r["SUMLEV"] == "050"}
|
county_population = {r["STATE"] + r["COUNTY"]: r[ESTIMATE] for r in csv_rows(cache, COUNTIES, "co-est2025.csv") if r["SUMLEV"] == "050"}
|
||||||
place_population, county_part = {}, collections.defaultdict(list)
|
|
||||||
for r in csv_rows(cache, PLACES, "sub-est2025.csv"):
|
|
||||||
if r["SUMLEV"] == "162":
|
|
||||||
place_population[r["STATE"] + r["PLACE"]] = int(r[ESTIMATE])
|
|
||||||
elif r["SUMLEV"] == "157":
|
|
||||||
county_part[r["STATE"] + r["PLACE"]].append((int(r[ESTIMATE]), r["STATE"] + r["COUNTY"]))
|
|
||||||
|
|
||||||
regions = {}
|
|
||||||
for r in gazetteer(cache, "state"):
|
|
||||||
regions[r["USPS"]] = {"abbr": r["USPS"], "code": r["GEOID"], "name": r["NAME"], "population": state_population[r["GEOID"]], "timezone": TIMEZONES[r["USPS"]]}
|
|
||||||
counties, unestimated = {}, collections.Counter()
|
counties, unestimated = {}, collections.Counter()
|
||||||
for r in gazetteer(cache, "counties"):
|
for r in gazetteer(cache, "counties"):
|
||||||
if r["GEOID"] not in county_population:
|
if r["GEOID"] not in county_population:
|
||||||
@@ -168,56 +184,21 @@ def main():
|
|||||||
continue
|
continue
|
||||||
counties[r["GEOID"]] = {"code": r["GEOID"], "name": r["NAME"], "region": r["USPS"], "population": county_population[r["GEOID"]]}
|
counties[r["GEOID"]] = {"code": r["GEOID"], "name": r["NAME"], "region": r["USPS"], "population": county_population[r["GEOID"]]}
|
||||||
print(f"counties without a population estimate, dropped: {dict(unestimated)}", file=sys.stderr)
|
print(f"counties without a population estimate, dropped: {dict(unestimated)}", file=sys.stderr)
|
||||||
localities = {}
|
|
||||||
for r in gazetteer(cache, "place"):
|
|
||||||
geoid, population = r["GEOID"], place_population.get(r["GEOID"], 0)
|
|
||||||
if r["FUNCSTAT"] not in "AFN" or r["LSAD"] == CDP or population < a.min_population or not county_part.get(geoid):
|
|
||||||
continue
|
|
||||||
county = max(county_part[geoid])[1]
|
|
||||||
if county not in counties:
|
|
||||||
print(f"{geoid} {r['NAME']}: county {county} unknown, dropped", file=sys.stderr)
|
|
||||||
continue
|
|
||||||
localities[geoid] = {"code": geoid, "name": place_name(geoid, r["NAME"]), "municipality": county, "population": population, "lat": r["INTPTLAT"], "lon": r["INTPTLONG"]}
|
|
||||||
|
|
||||||
zcta_of = {}
|
places = localities(cache, a.min_population, counties)
|
||||||
for r in csv.DictReader(io.StringIO(text(fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"):
|
locality_of_zcta = postal_codes(cache, places)
|
||||||
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"] in localities:
|
named, addresses = streets(cache, sorted({l["municipality"] for l in places.values()}), locality_of_zcta, a.streets_per_locality)
|
||||||
zcta_of.setdefault(r["GEOID_ZCTA5_20"], []).append((int(r["AREALAND_PART"]), r["GEOID_PLACE_20"]))
|
for geoid in [l for l in places if l not in named]:
|
||||||
locality_of_zcta = {zcta: max(parts)[1] for zcta, parts in zcta_of.items()}
|
print(f"{geoid} {places[geoid]['name']}: no streets, dropped", file=sys.stderr)
|
||||||
|
del places[geoid]
|
||||||
needed = sorted({l["municipality"] for l in localities.values()})
|
kept_counties = {l["municipality"] for l in places.values()}
|
||||||
with concurrent.futures.ThreadPoolExecutor(3) as pool:
|
|
||||||
list(pool.map(lambda c: (tiger_zip(cache, "addr", c), tiger_zip(cache, "featnames", c)), needed))
|
|
||||||
addresses, streets_of = collections.Counter(), collections.Counter()
|
|
||||||
for county in needed:
|
|
||||||
zips = collections.defaultdict(set)
|
|
||||||
for r in tiger(cache, "addr", county, {"TLID", "ZIP"}):
|
|
||||||
if r["ZIP"] in locality_of_zcta:
|
|
||||||
zips[r["TLID"]].add(r["ZIP"])
|
|
||||||
addresses[r["ZIP"]] += 1
|
|
||||||
for r in tiger(cache, "featnames", county, {"TLID", "FULLNAME", "PAFLAG"}):
|
|
||||||
if r["PAFLAG"] == "P" and r["FULLNAME"]:
|
|
||||||
for z in zips.get(r["TLID"], ()):
|
|
||||||
streets_of[(locality_of_zcta[z], r["FULLNAME"])] += 1
|
|
||||||
by_locality = collections.defaultdict(list)
|
|
||||||
for (locality, name), n in streets_of.items():
|
|
||||||
by_locality[locality].append((n, name))
|
|
||||||
streets = []
|
|
||||||
for locality in sorted(by_locality):
|
|
||||||
for n, name in sorted(by_locality[locality], key=lambda s: (-s[0], s[1]))[: a.streets_per_locality]:
|
|
||||||
streets.append({"name": name, "locality": locality, "addresses": n})
|
|
||||||
for geoid in [l for l in localities if l not in by_locality]:
|
|
||||||
print(f"{geoid} {localities[geoid]['name']}: no streets, dropped", file=sys.stderr)
|
|
||||||
del localities[geoid]
|
|
||||||
postal_codes = [{"code": z, "locality": l, "addresses": addresses[z]} for z, l in sorted(locality_of_zcta.items()) if addresses[z] and l in localities]
|
|
||||||
|
|
||||||
kept_counties = {l["municipality"] for l in localities.values()}
|
|
||||||
kept_regions = {counties[c]["region"] for c in kept_counties}
|
kept_regions = {counties[c]["region"] for c in kept_counties}
|
||||||
write(out / "region.tsv", ["abbr", "code", "name", "population", "timezone"], [r for _, r in sorted(regions.items()) if r["abbr"] in kept_regions])
|
|
||||||
write(out / "municipality.tsv", ["code", "name", "region", "population"], [c for _, c in sorted(counties.items()) if c["code"] in kept_counties])
|
tsv.write(out / "region.tsv", ["abbr", "code", "name", "population", "timezone"], [r for _, r in sorted(regions.items()) if r["abbr"] in kept_regions])
|
||||||
write(out / "locality.tsv", ["code", "name", "municipality", "population", "lat", "lon"], [l for _, l in sorted(localities.items())])
|
tsv.write(out / "municipality.tsv", ["code", "name", "region", "population"], [c for _, c in sorted(counties.items()) if c["code"] in kept_counties])
|
||||||
write(out / "postal-code.tsv", ["code", "locality", "addresses"], postal_codes)
|
tsv.write(out / "locality.tsv", ["code", "name", "municipality", "population", "lat", "lon"], [l for _, l in sorted(places.items())])
|
||||||
write(out / "street.tsv", ["name", "locality", "addresses"], streets)
|
tsv.write(out / "postal-code.tsv", ["code", "locality", "addresses"], [{"code": z, "locality": l, "addresses": addresses[z]} for z, l in sorted(locality_of_zcta.items()) if addresses[z] and l in places])
|
||||||
|
tsv.write(out / "street.tsv", ["name", "locality", "addresses"], [s for locality in sorted(named) for s in named[locality]])
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|||||||
@@ -0,0 +1,39 @@
|
|||||||
|
"""A source fetched once into the cache, and a table written as the loader admits it."""
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
import urllib.request
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
def fetch(source, cache, name, magic=b"", data=None, headers=None):
|
||||||
|
"""The bytes of a URL, downloaded into cache/name once, or of a local file."""
|
||||||
|
if not re.match(r"^https?://", source):
|
||||||
|
return Path(source).read_bytes()
|
||||||
|
path = Path(cache) / name
|
||||||
|
for attempt in range(1, 6):
|
||||||
|
if path.exists():
|
||||||
|
return path.read_bytes()
|
||||||
|
req = urllib.request.Request(source, data=data, headers={"User-Agent": "fejkdata data-import", **(headers or {})})
|
||||||
|
try:
|
||||||
|
with urllib.request.urlopen(req, timeout=600) as r:
|
||||||
|
body = r.read()
|
||||||
|
except OSError:
|
||||||
|
body = b""
|
||||||
|
if body.startswith(magic) and b"Request Rejected" not in body[:512]:
|
||||||
|
path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
path.write_bytes(body)
|
||||||
|
else:
|
||||||
|
time.sleep(10 * attempt)
|
||||||
|
sys.exit(f"{source}: no valid download in 5 attempts")
|
||||||
|
|
||||||
|
|
||||||
|
def write(path, columns, rows):
|
||||||
|
"""Write the rows as a TSV; every cell must be non-empty and free of tabs, newlines and braces."""
|
||||||
|
lines = ["\t".join(columns)]
|
||||||
|
for row in rows:
|
||||||
|
cells = [str(row[c]) for c in columns]
|
||||||
|
assert all(cells) and not any(re.search(r"[\t\n{}]", c) for c in cells), row
|
||||||
|
lines.append("\t".join(cells))
|
||||||
|
Path(path).write_text("\n".join(lines) + "\n", encoding="utf-8")
|
||||||
|
print(f"{path}: {len(lines) - 1} rows", file=sys.stderr)
|
||||||
Reference in New Issue
Block a user