Rank a ZCTA's parts among incorporated places only, drop numbered loops and highways, and say which places ship no address
This commit is contained in:
@@ -236,7 +236,7 @@ consistent draw of them, which the locale's `address` reads.
|
|||||||
|-------|----------|----------|--------|
|
|-------|----------|----------|--------|
|
||||||
| `region` | län, by code or name | state, by USPS abbreviation or name; `code` is the FIPS code | population |
|
| `region` | län, by code or name | state, by USPS abbreviation or name; `code` is the FIPS code | population |
|
||||||
| `municipality` | kommun, by code or name | county, by FIPS code or name | population |
|
| `municipality` | kommun, by code or name | county, by FIPS code or name | population |
|
||||||
| `locality` | postort, by name | incorporated place of 25,000 people or more, by GEOID or name; Hawaii has none | tätort population, the kommun's where the postort names it, else 200; place population |
|
| `locality` | postort, by name | incorporated place of 25,000 people or more with a postal code of its own, by GEOID or name; Hawaii has none | tätort population, the kommun's where the postort names it, else 200; place population |
|
||||||
| `postal-code` | postnummer with street delivery, by code | ZCTA, by code | one; address ranges |
|
| `postal-code` | postnummer with street delivery, by code | ZCTA, by code | one; address ranges |
|
||||||
| `street` | gatunamn, the ten with most road segments per postort | street name, the ten with most address ranges per place | segments; address ranges |
|
| `street` | gatunamn, the ten with most road segments per postort | street name, the ten with most address ranges per place | segments; address ranges |
|
||||||
|
|
||||||
@@ -1035,9 +1035,11 @@ renamed or retyped line is a major.
|
|||||||
holding most of its land inside places.** `I- 55 Bus` and `US Hwy 1` carry the
|
holding most of its land inside places.** `I- 55 Bus` and `US Hwy 1` carry the
|
||||||
most address ranges in many places and would head every address, so the import
|
most address ranges in many places and would head every address, so the import
|
||||||
drops names spelled as a route. A ZCTA goes to the place its largest in-place part
|
drops names spelled as a route. A ZCTA goes to the place its largest in-place part
|
||||||
lies in, and ships only when that place does; counting the land outside every
|
lies in, census-designated places left out since they never ship, and ships only
|
||||||
place too would drop a quarter of the places, whose codes straddle unincorporated
|
when that place does, so a few dozen places whose every code lies mostly in a
|
||||||
land, for a postal city the USPS mostly names the same way.
|
bigger neighbour ship no address; counting the land outside every place too would
|
||||||
|
drop a quarter of the places, whose codes straddle unincorporated land, for a
|
||||||
|
postal city the USPS mostly names the same way.
|
||||||
- **`List` advertises direct descents only.** `region.municipality.locality` is
|
- **`List` advertises direct descents only.** `region.municipality.locality` is
|
||||||
listed, and `region.locality` resolves too but is not: the set of every descent
|
listed, and `region.locality` resolves too but is not: the set of every descent
|
||||||
through a chain of five tables is every subsequence of it, and the direct chain is
|
through a chain of five tables is every subsequence of it, and the direct chain is
|
||||||
|
|||||||
@@ -27,7 +27,7 @@ OUT = Path(__file__).resolve().parent.parent / "data" / "geo" / "US"
|
|||||||
CACHE = Path(__file__).resolve().parent / "cache"
|
CACHE = Path(__file__).resolve().parent / "cache"
|
||||||
ESTIMATE = "POPESTIMATE2025"
|
ESTIMATE = "POPESTIMATE2025"
|
||||||
CDP = "57"
|
CDP = "57"
|
||||||
HIGHWAY = re.compile(r"\b(I- |Hwy |Rte |Route |Rd )\d")
|
HIGHWAY = re.compile(r"\b(I- |Hwy |Highway |Loop |Rte |Route |Rd )\d")
|
||||||
SUFFIX = re.compile(r" (city and borough|city|town|village|borough|municipality|comunidad|zona urbana|metropolitan government|metro government|consolidated government|unified government|urban county|corporation|plantation)( \(balance\))?$")
|
SUFFIX = re.compile(r" (city and borough|city|town|village|borough|municipality|comunidad|zona urbana|metropolitan government|metro government|consolidated government|unified government|urban county|corporation|plantation)( \(balance\))?$")
|
||||||
# Places whose Census name is a merged government's; the postal city is what an address carries.
|
# Places whose Census name is a merged government's; the postal city is what an address carries.
|
||||||
NAMES = {"1303440": "Athens", "1304204": "Augusta", "1349008": "Macon", "2148006": "Louisville", "3011397": "Butte", "4732742": "Hartsville", "4752006": "Nashville"}
|
NAMES = {"1303440": "Athens", "1304204": "Augusta", "1349008": "Macon", "2148006": "Louisville", "3011397": "Butte", "4732742": "Hartsville", "4752006": "Nashville"}
|
||||||
@@ -128,10 +128,10 @@ def localities(cache, min_population, counties):
|
|||||||
|
|
||||||
|
|
||||||
def postal_codes(cache, localities):
|
def postal_codes(cache, localities):
|
||||||
"""Each ZCTA whose largest part inside any place lies in a shipped place."""
|
"""Each ZCTA whose largest part inside an incorporated place lies in a shipped place."""
|
||||||
parts = {}
|
parts = {}
|
||||||
for r in csv.DictReader(io.StringIO(text(tsv.fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"):
|
for r in csv.DictReader(io.StringIO(text(tsv.fetch(ZCTA_PLACE, cache, "zcta-place.txt"))), delimiter="|"):
|
||||||
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"]:
|
if r["GEOID_ZCTA5_20"] and r["GEOID_PLACE_20"] and not r["NAMELSAD_PLACE_20"].endswith(" CDP"):
|
||||||
parts.setdefault(r["GEOID_ZCTA5_20"], []).append((int(r["AREALAND_PART"]), r["GEOID_PLACE_20"]))
|
parts.setdefault(r["GEOID_ZCTA5_20"], []).append((int(r["AREALAND_PART"]), r["GEOID_PLACE_20"]))
|
||||||
largest = {zcta: max(p)[1] for zcta, p in parts.items()}
|
largest = {zcta: max(p)[1] for zcta, p in parts.items()}
|
||||||
return {zcta: place for zcta, place in largest.items() if place in localities}
|
return {zcta: place for zcta, place in largest.items() if place in localities}
|
||||||
|
|||||||
@@ -54,10 +54,10 @@ countries; the README maps each to the native term.
|
|||||||
| Table | SE | US | Weight |
|
| Table | SE | US | Weight |
|
||||||
|---|---|---|---|
|
|---|---|---|---|
|
||||||
| `region` | län (21) | state and DC (50; Hawaii has no incorporated place) | population |
|
| `region` | län (21) | state and DC (50; Hawaii has no incorporated place) | population |
|
||||||
| `municipality` | kommun (290) | county with a shipped place (663) | population |
|
| `municipality` | kommun (290) | county with a shipped place | population |
|
||||||
| `locality` | postort (1,522), tätort population | place of 25,000+ (1,579) | population |
|
| `locality` | postort, tätort population | place of 25,000+ | population |
|
||||||
| `postal-code` | postnummer with street delivery (13,712) | ZCTA of a shipped place (5,946) | one; address ranges |
|
| `postal-code` | postnummer with street delivery | ZCTA of a shipped place | one; address ranges |
|
||||||
| `street` | gatunamn, top 10 per postort (14,764) | street name, top 10 per place (15,790) | segments; address ranges |
|
| `street` | gatunamn, top 10 per postort | street name, top 10 per place | segments; address ranges |
|
||||||
|
|
||||||
- Shipped in step 2, README Data. `geo.SE.address` is a record over one consistent
|
- Shipped in step 2, README Data. `geo.SE.address` is a record over one consistent
|
||||||
draw. Each region row carries its timezone, each locality its centroid.
|
draw. Each region row carries its timezone, each locality its centroid.
|
||||||
|
|||||||
Reference in New Issue
Block a user