Add the geo trees for SE and US as linked tables, and build each locale's address on them #20

Merged
lilleman merged 14 commits from geo-tables into main 2026-09-18 01:44:02 +02:00
6 changed files with 29 additions and 29 deletions
Showing only changes of commit 8c2b51ebed - Show all commits
+10 -2
View File
@@ -1014,8 +1014,9 @@ renamed or retyped line is a major.
whole.** `data/sv_SE` alone no longer loads: a test loads `data` and prefixes
the locale, and `--no-shipped-data -d` takes the whole `data` folder or a set of
one's own.
- **The default embed holds every Swedish postort and the US places of 25,000 or
more.** Sweden fits whole in 700 KB; every US place of 10,000 would pass a
- **The default embed holds every Swedish postort the import can place and give a
street-delivery code and a street, and the US places of 25,000 or more.** Sweden
fits whole in 700 KB; every US place of 10,000 would pass a
megabyte and fetch 1,200 counties of TIGER files, so the threshold sits where the
two countries match in size, and `--min-population` and
`--streets-per-locality` on the import scripts build a fuller set. The two trees
@@ -1030,6 +1031,13 @@ renamed or retyped line is a major.
carries stale spellings; the nearest code across a border named the wrong kommun
half the time it was tried, so a postort none of the three rules place is
dropped, as is one not cased like a place name.
- **A highway designation is not a street, and a US postal code belongs to the place
holding most of its land inside places.** `I- 55 Bus` and `US Hwy 1` carry the
most address ranges in many places and would head every address, so the import
drops names spelled as a route. A ZCTA goes to the place its largest in-place part
lies in, and ships only when that place does; counting the land outside every
place too would drop a quarter of the places, whose codes straddle unincorporated
land, for a postal city the USPS mostly names the same way.
- **`List` advertises direct descents only.** `region.municipality.locality` is
listed, and `region.locality` resolves too but is not: the set of every descent
through a chain of five tables is every subsequence of it, and the direct chain is
+4 -8
View File
@@ -2,12 +2,6 @@
"""Rebuild data/geo/SE/*.tsv from SCB (CC0), GeoNames (CC BY 4.0) and Trafikverket NVDB (CC0).
TRAFIKVERKET_API_KEY=… data-import/geo-se.py [--key-file FILE] [--cache DIR] [--streets-per-locality N] [--out DIR]
A locality is a GeoNames postort, placed in the municipality it names, else of its
tätort, else of most of its codes, and weighted by its tätort's population, else its
municipality's, else 200. Box codes are dropped by the digit after the postort's own
prefix. Each NVDB street segment goes to the nearest postal code centroid; a locality
keeps the N names with most segments.
"""
import argparse
import collections
@@ -95,7 +89,7 @@ def geonames(cache):
def nvdb_segments(cache, key):
path = cache / "nvdb-gatunamn.tsv"
if not path.exists():
with open(path, "w", encoding="utf-8") as out:
with open(path.with_suffix(".part"), "w", encoding="utf-8") as out:
change = "0"
while True:
query = (
@@ -116,6 +110,7 @@ def nvdb_segments(cache, key):
change = result["INFO"]["LASTCHANGEID"]
if len(rows) < NVDB_PAGE:
break
path.with_suffix(".part").rename(path)
for line in path.read_text(encoding="utf-8").splitlines():
name, lon, lat = line.split("\t")
yield name, float(lat), float(lon)
@@ -208,7 +203,8 @@ def streets(segments, codes, localities, per_locality):
nearest = Nearest((r["lat"], r["lon"], r["locality"]) for r in codes if r["lat"] is not None and r["locality"] in localities)
count = collections.Counter()
for name, lat, lon in segments:
count[(nearest.find(lat, lon), name)] += 1
if name[0].isalpha():
count[(nearest.find(lat, lon), name)] += 1
of = collections.defaultdict(list)
for (locality, name), n in count.items():
of[locality].append((n, name))
+7 -11
View File
@@ -2,11 +2,6 @@
"""Rebuild data/geo/US/*.tsv from the Census Bureau's Gazetteer, population estimates, ZCTA relationships and TIGER/Line files (public domain).
data-import/geo-us.py [--cache DIR] [--min-population N] [--streets-per-locality N] [--out DIR]
A locality is an incorporated place, or a consolidated city's balance, of at least N
people, in the county holding most of it. Its postal codes are the ZCTAs mostly inside
it, weighted by their TIGER address ranges, and its streets the N names with most
address ranges in those codes.
"""
import argparse
import collections
@@ -32,10 +27,10 @@ OUT = Path(__file__).resolve().parent.parent / "data" / "geo" / "US"
CACHE = Path(__file__).resolve().parent / "cache"
ESTIMATE = "POPESTIMATE2025"
CDP = "57"
HIGHWAY = re.compile(r"\b(I- |Hwy |Rte |Route |Rd )\d")
SUFFIX = re.compile(r" (city and borough|city|town|village|borough|municipality|comunidad|zona urbana|metropolitan government|metro government|consolidated government|unified government|urban county|corporation|plantation)( \(balance\))?$")
# Places whose Census name is a merged government's; the postal city is what an address carries.
NAMES = {"1303440": "Athens", "1304204": "Augusta", "1349008": "Macon", "2148006": "Louisville", "3011397": "Butte", "4732742": "Hartsville", "4752006": "Nashville"}
# The predominant zone of each state.
TIMEZONES = {
"AK": "America/Anchorage", "AL": "America/Chicago", "AR": "America/Chicago", "AZ": "America/Phoenix",
"CA": "America/Los_Angeles", "CO": "America/Denver", "CT": "America/New_York", "DC": "America/New_York",
@@ -122,7 +117,7 @@ def localities(cache, min_population, counties):
out = {}
for r in gazetteer(cache, "place"):
geoid, population = r["GEOID"], place_population.get(r["GEOID"], 0)
if r["FUNCSTAT"] not in "AFN" or r["LSAD"] == CDP or population < min_population or not county_part.get(geoid):
if r["FUNCSTAT"] not in ("A", "F", "N") or r["LSAD"] == CDP or population < min_population or not county_part.get(geoid):
continue
county = max(county_part[geoid])[1]
if county not in counties:
@@ -133,12 +128,13 @@ def localities(cache, min_population, counties):
def postal_codes(cache, localities):
"""Each ZCTA and the shipped place holding most of its land."""
"""Each ZCTA whose largest part inside any place lies in a shipped place."""
parts = {}
for r in csv.DictReader(io.StringIO(text(tsv.fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"):
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"] in localities:
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"]:
parts.setdefault(r["GEOID_ZCTA5_20"], []).append((int(r["AREALAND_PART"]), r["GEOID_PLACE_20"]))
return {zcta: max(p)[1] for zcta, p in parts.items()}
largest = {zcta: max(p)[1] for zcta, p in parts.items()}
return {zcta: place for zcta, place in largest.items() if place in localities}
def streets(cache, counties, locality_of_zcta, per_locality):
@@ -153,7 +149,7 @@ def streets(cache, counties, locality_of_zcta, per_locality):
zips[r["TLID"]].add(r["ZIP"])
addresses[r["ZIP"]] += 1
for r in tiger(cache, "featnames", county, {"TLID", "FULLNAME", "PAFLAG"}):
if r["PAFLAG"] == "P" and r["FULLNAME"]:
if r["PAFLAG"] == "P" and r["FULLNAME"] and not HIGHWAY.search(r["FULLNAME"]):
for z in zips.get(r["TLID"], ()):
count[(locality_of_zcta[z], r["FULLNAME"])] += 1
of = collections.defaultdict(list)
+4 -3
View File
@@ -20,10 +20,10 @@ def fetch(source, cache, name, magic=b"", data=None, headers=None):
body = r.read()
except OSError:
body = b""
if body.startswith(magic) and b"Request Rejected" not in body[:512]:
if body and body.startswith(magic) and b"Request Rejected" not in body[:512]:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(body)
else:
elif attempt < 5:
time.sleep(10 * attempt)
sys.exit(f"{source}: no valid download in 5 attempts")
@@ -33,7 +33,8 @@ def write(path, columns, rows):
lines = ["\t".join(columns)]
for row in rows:
cells = [str(row[c]) for c in columns]
assert all(cells) and not any(re.search(r"[\t\n{}]", c) for c in cells), row
if not all(cells) or any(re.search(r"[\t\n{}]", c) for c in cells):
raise ValueError(f"{path}: a cell is empty or holds a tab, newline or brace: {row}")
lines.append("\t".join(cells))
Path(path).write_text("\n".join(lines) + "\n", encoding="utf-8")
print(f"{path}: {len(lines) - 1} rows", file=sys.stderr)
-1
View File
@@ -44,7 +44,6 @@ var (
regionOf = map[string]string{"0180": "01", "0184": "01", "1280": "12", "1281": "12", "1480": "14"}
)
// siblings adds two child tables under locality: postal-code, keyed, and street, keyless.
func siblings() map[string]string {
return with(geo(), map[string]string{
"postal-code.json": `{"format":"{code}","rows":"postal-code.tsv","key":"code","parent":"locality"}`,
+4 -4
View File
@@ -54,10 +54,10 @@ countries; the README maps each to the native term.
| Table | SE | US | Weight |
|---|---|---|---|
| `region` | län (21) | state and DC (50; Hawaii has no incorporated place) | population |
| `municipality` | kommun (290) | county with a shipped place (666) | population |
| `locality` | postort (1,522), tätort population | place of 25,000+ (1,601) | population |
| `postal-code` | postnummer with street delivery (13,712) | ZCTA of a shipped place (7,401) | one; address ranges |
| `street` | gatunamn, top 10 per postort (14,764) | street name, top 10 per place (16,010) | segments; address ranges |
| `municipality` | kommun (290) | county with a shipped place (663) | population |
| `locality` | postort (1,522), tätort population | place of 25,000+ (1,579) | population |
| `postal-code` | postnummer with street delivery (13,712) | ZCTA of a shipped place (5,946) | one; address ranges |
| `street` | gatunamn, top 10 per postort (14,764) | street name, top 10 per place (15,790) | segments; address ranges |
- Shipped in step 2, README Data. `geo.SE.address` is a record over one consistent
draw. Each region row carries its timezone, each locality its centroid.