diff --git a/DATA-LICENSES.md b/DATA-LICENSES.md index 8004007..9cf38a5 100644 --- a/DATA-LICENSES.md +++ b/DATA-LICENSES.md @@ -5,13 +5,16 @@ Every shipped dataset, its source, its licence and the attribution it asks for. The scripts run through `docker compose run --rm data-import data-import/.py` and cache their downloads under `data-import/cache/`; `geo-se.py` needs a Trafikverket API key, free at [data.trafikverket.se](https://data.trafikverket.se/), -in `TRAFIKVERKET_API_KEY` or a `--key-file`. +in `TRAFIKVERKET_API_KEY` or a `--key-file`; `geo-us.py` fetches two TIGER/Line +files per county it ships, a few hundred megabytes in all. | Table | Source | Licence | Attribution | Rebuild | |-------|--------|---------|-------------|---------| | `geo/SE/locality.tsv`, `postal-code.tsv` | [GeoNames](https://www.geonames.org/) postal codes for SE; populations from SCB tätorter 2023 | [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/); CC0 1.0 | "Postal codes from GeoNames, www.geonames.org" | `data-import/geo-se.py` | | `geo/SE/region.tsv`, `municipality.tsv` | [SCB](https://www.scb.se/) län and kommun codes 2026 and population 2024 | [CC0 1.0](https://creativecommons.org/publicdomain/zero/1.0/) | none required | `data-import/geo-se.py` | | `geo/SE/street.tsv` | [Trafikverket NVDB](https://www.trafikverket.se/) Gatunamn, through the open API | CC0 1.0 | none required | `data-import/geo-se.py` | +| `geo/US/region.tsv`, `municipality.tsv`, `locality.tsv` | [Census Bureau](https://www.census.gov/) Gazetteer 2026 and population estimates 2025 | [public domain](https://www.usa.gov/government-works) | none required | `data-import/geo-us.py` | +| `geo/US/postal-code.tsv`, `street.tsv` | Census Bureau ZCTA to place relationships 2020 and TIGER/Line 2025 address ranges and feature names | public domain | none required | `data-import/geo-us.py` | | `misc/country.tsv` | [datasets/country-codes](https://github.com/datasets/country-codes) | [PDDL 1.0](https://opendatacommons.org/licenses/pddl/1-0/) | none required | `data-import/country.py` | | `misc/currency.tsv` | [datasets/currency-codes](https://github.com/datasets/currency-codes); symbols from [Unicode CLDR](https://github.com/unicode-org/cldr) `en.xml` and `root.xml` | PDDL 1.0; [Unicode License v3](https://www.unicode.org/license.txt) | CLDR: "Copyright © 1991-2025 Unicode, Inc. Unicode and the Unicode Logo are registered trademarks of Unicode, Inc. in the United States and other countries." | `data-import/currency.py` | | `misc/httpstatus.tsv` | curated (IANA HTTP status codes are facts) | — | — | — | diff --git a/data-import/geo-us.py b/data-import/geo-us.py new file mode 100644 index 0000000..b1fd360 --- /dev/null +++ b/data-import/geo-us.py @@ -0,0 +1,224 @@ +#!/usr/bin/env python3 +"""Rebuild data/geo/US/*.tsv from the Census Bureau's Gazetteer, population estimates, ZCTA relationships and TIGER/Line files (public domain). + + data-import/geo-us.py [--cache DIR] [--min-population N] [--streets-per-locality N] [--out DIR] + +A locality is an incorporated place, or a consolidated city's balance, of at least N +people, in the county holding most of it. Its postal codes are the ZCTAs mostly inside +it, weighted by their TIGER address ranges, and its streets the N names with most +address ranges in those codes. +""" +import argparse +import collections +import concurrent.futures +import csv +import io +import re +import struct +import sys +import time +import urllib.request +import zipfile +from pathlib import Path + +GAZETTEER = "https://www2.census.gov/geo/docs/maps-data/data/gazetteer/2026_Gazetteer/2026_Gaz_{}_national.zip" +POPULATION = "https://www2.census.gov/programs-surveys/popest/datasets/2020-2025/{}" +STATES = POPULATION.format("state/totals/NST-EST2025-ALLDATA.csv") +COUNTIES = POPULATION.format("counties/totals/co-est2025-alldata.csv") +PLACES = POPULATION.format("cities/totals/sub-est2025.csv") +ZCTA_PLACE = "https://www2.census.gov/geo/docs/maps-data/data/rel2020/zcta520/tab20_zcta520_place20_natl.txt" +TIGER = "https://www2.census.gov/geo/tiger/TIGER2025/{0}/tl_2025_{1}_{2}.zip" +OUT = Path(__file__).resolve().parent.parent / "data" / "geo" / "US" +CACHE = Path(__file__).resolve().parent / "cache" +ESTIMATE = "POPESTIMATE2025" +CDP = "57" +SUFFIX = re.compile(r" (city and borough|city|town|village|borough|municipality|comunidad|zona urbana|metropolitan government|metro government|consolidated government|unified government|urban county|corporation|plantation)( \(balance\))?$") +# Places whose Census name is a merged government's; the postal city is what an address carries. +NAMES = {"1303440": "Athens", "1304204": "Augusta", "1349008": "Macon", "2148006": "Louisville", "3011397": "Butte", "4732742": "Hartsville", "4752006": "Nashville"} +TIMEZONES = { + "AK": "America/Anchorage", "AL": "America/Chicago", "AR": "America/Chicago", "AZ": "America/Phoenix", + "CA": "America/Los_Angeles", "CO": "America/Denver", "CT": "America/New_York", "DC": "America/New_York", + "DE": "America/New_York", "FL": "America/New_York", "GA": "America/New_York", "HI": "Pacific/Honolulu", + "IA": "America/Chicago", "ID": "America/Boise", "IL": "America/Chicago", "IN": "America/Indiana/Indianapolis", + "KS": "America/Chicago", "KY": "America/New_York", "LA": "America/Chicago", "MA": "America/New_York", + "MD": "America/New_York", "ME": "America/New_York", "MI": "America/Detroit", "MN": "America/Chicago", + "MO": "America/Chicago", "MS": "America/Chicago", "MT": "America/Denver", "NC": "America/New_York", + "ND": "America/Chicago", "NE": "America/Chicago", "NH": "America/New_York", "NJ": "America/New_York", + "NM": "America/Denver", "NV": "America/Los_Angeles", "NY": "America/New_York", "OH": "America/New_York", + "OK": "America/Chicago", "OR": "America/Los_Angeles", "PA": "America/New_York", "PR": "America/Puerto_Rico", + "RI": "America/New_York", "SC": "America/New_York", "SD": "America/Chicago", "TN": "America/Chicago", + "TX": "America/Chicago", "UT": "America/Denver", "VA": "America/New_York", "VT": "America/New_York", + "WA": "America/Los_Angeles", "WI": "America/Chicago", "WV": "America/New_York", "WY": "America/Denver", +} + + +def fetch(url, cache, name, magic=b""): + path = cache / name + for attempt in range(1, 6): + if path.exists(): + return path.read_bytes() + req = urllib.request.Request(url, headers={"User-Agent": "fejkdata data-import"}) + try: + with urllib.request.urlopen(req, timeout=600) as r: + data = r.read() + except OSError: + data = b"" + if data.startswith(magic) and b"Request Rejected" not in data[:512]: + path.write_bytes(data) + else: + time.sleep(10 * attempt) + sys.exit(f"{url}: no valid download in 5 attempts") + + +def text(data): + try: + return data.decode("utf-8-sig") + except UnicodeDecodeError: + return data.decode("latin-1") + + +def gazetteer(cache, kind): + z = zipfile.ZipFile(io.BytesIO(fetch(GAZETTEER.format(kind), cache, f"gaz_{kind}.zip", magic=b"PK"))) + rows = text(z.read(z.namelist()[0])).splitlines() + header = [h.strip() for h in rows[0].split("|")] + return [dict(zip(header, (c.strip() for c in row.split("|")))) for row in rows[1:]] + + +def csv_rows(cache, url, name): + return list(csv.DictReader(io.StringIO(text(fetch(url, cache, name))))) + + +def dbf_rows(data, wanted): + """The records of a dBASE file, the wanted fields only.""" + count, header_len, record_len = struct.unpack("