From 2ad00b3569402c1f6e858d2087fc555712f9622a Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Fri, 18 Sep 2026 23:22:55 +0200 Subject: [PATCH] Refuse an is_independent the script cannot read, name Edge on Android, and correct the claims review found false --- CHANGELOG.md | 6 ++++-- DATA-LICENSES.md | 9 ++++----- README.md | 27 ++++++++++++++------------- data-import/territory.py | 7 ++++++- data-import/timezone.py | 29 +++++++++++++++++++---------- data-import/useragent.py | 20 +++++++++++++++----- 6 files changed, 62 insertions(+), 36 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index fcc3113..5a62c38 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -34,11 +34,13 @@ replacement, and each removed path, column or flag. type mime-db records as IANA-registered and gives a filename extension. `DATA-LICENSES.md` lists each source. - `misc.timezone`, `misc.car` and `misc.useragent` are tables too. `misc.timezone` - is tzdb `zone.tab`'s zone for each shipped territory, with `offset` holding the + is every zone tzdb `zone.tab` gives a shipped territory, with `offset` holding the zone's standard UTC offset and `territory` linking to `misc.territory`, so `misc.territory[SE].timezone` draws `Europe/Stockholm`; zones of a territory `misc.territory` does not ship, Antarctica's among them, and the constant `UTC` - are gone. `misc.useragent` is the top-user-agents desktop and mobile lists with + are gone. It draws from 401 zones where it drew from 27 common ones, and evenly, + so a bare `misc.timezone` now renders a sub-zone such as `America/Indiana/Knox` + far more often than a capital's: select inside a territory, or pin the zone. `misc.useragent` is the top-user-agents desktop and mobile lists with `browser`, `device` and `os` columns. `misc.car` keeps its makes and models, with `misc.car.maker` renamed `misc.car.make`. - `geo.SE` and `geo.US`: five linked tables per country, `region`, `municipality`, diff --git a/DATA-LICENSES.md b/DATA-LICENSES.md index 64ac814..b44c4ff 100644 --- a/DATA-LICENSES.md +++ b/DATA-LICENSES.md @@ -14,14 +14,13 @@ Every shipped dataset, its source, its licence and the attribution it asks for. | `sv_SE/first-name.tsv`, `last-name.tsv` | [SCB](https://www.scb.se/) names with at least two bearers, 31 December 2022 | CC0 1.0 | "Källa: SCB" | `data-import/names-se.py` | | `en_US/first-name.tsv` | [SSA](https://www.ssa.gov/oact/babynames/) baby names, births 1930 to 2020, through [hackerb9/ssa-baby-names](https://github.com/hackerb9/ssa-baby-names) | public domain | none required | `data-import/names-us.py` | | `en_US/last-name.tsv` | Census Bureau surnames occurring 100 or more times, 2010 | public domain | none required | `data-import/names-us.py` | -| `misc/car.tsv` | curated makes and models, pending an international source (goal 10) | — | — | — | -| `sv_SE/sex.tsv`, `sv_SE/birth-number.tsv`, `sv_SE/title.tsv`, `en_US/sex.tsv`, `en_US/title.tsv` | curated (Skatteverket's test birth numbers are facts) | — | — | — | -| `misc/territory.tsv` | [datasets/country-codes](https://github.com/datasets/country-codes) | [PDDL 1.0](https://opendatacommons.org/licenses/pddl/1-0/) | none required | `data-import/territory.py` | +| `sv_SE/sex.tsv`, `sv_SE/birth-number.tsv`, `sv_SE/title.tsv`, `en_US/sex.tsv`, `en_US/title.tsv`, `misc/car.tsv` | curated; Skatteverket's test birth numbers are facts, and `car.tsv` waits for an international source (goal 10) | — | — | — | | `misc/currency.tsv` | [datasets/currency-codes](https://github.com/datasets/currency-codes); symbols from [Unicode CLDR](https://github.com/unicode-org/cldr) `en.xml` and `root.xml` | PDDL 1.0; [Unicode License v3](https://www.unicode.org/license.txt) | CLDR: "Copyright © 1991-2025 Unicode, Inc. Unicode and the Unicode Logo are registered trademarks of Unicode, Inc. in the United States and other countries." | `data-import/currency.py` | | `misc/httpstatus.tsv` | [IANA HTTP Status Code Registry](https://www.iana.org/assignments/http-status-codes/) | [public domain](https://www.iana.org/help/licensing-terms) | none required | `data-import/httpstatus.py` | | `misc/language.tsv` | [datasets/language-codes](https://github.com/datasets/language-codes), the [Library of Congress](https://www.loc.gov/standards/iso639-2/) ISO 639-2 register | [PDDL 1.0](https://opendatacommons.org/licenses/pddl/1-0/) | none required | `data-import/language.py` | | `misc/mimetype.tsv` | [mime-db](https://github.com/jshttp/mime-db), the IANA media type registry with filename extensions | [MIT](https://github.com/jshttp/mime-db/blob/master/LICENSE) | "Copyright (c) 2014 Jonathan Ong, Copyright (c) 2015-2022 Douglas Christopher Wilson" | `data-import/mimetype.py` | -| `misc/timezone.tsv` | [IANA tzdb](https://www.iana.org/time-zones) `zone.tab` and the standard offset of each zone | [public domain](https://data.iana.org/time-zones/tzdb/LICENSE) | none required | `data-import/timezone.py` | +| `misc/territory.tsv` | [datasets/country-codes](https://github.com/datasets/country-codes) | [PDDL 1.0](https://opendatacommons.org/licenses/pddl/1-0/) | none required | `data-import/territory.py` | +| `misc/timezone.tsv` | [IANA tzdb](https://www.iana.org/time-zones) 2026d `zone.tab` and the standard offset of each zone | [public domain](https://data.iana.org/time-zones/tzdb/LICENSE) | none required | `data-import/timezone.py` | | `misc/useragent.tsv` | [top-user-agents](https://github.com/microlinkhq/top-user-agents) desktop and mobile lists | [MIT](https://github.com/microlinkhq/top-user-agents/blob/master/LICENSE.md) | "Copyright © 2020 Kiko Beats" | `data-import/useragent.py` | -Every other category is hand-written JSON under [`data/`](data), MIT like the code. +Every other category is hand-written under [`data/`](data), MIT like the code. diff --git a/README.md b/README.md index 9f940c6..350c6a8 100644 --- a/README.md +++ b/README.md @@ -168,20 +168,21 @@ carries `car`, `coordinate`, `creditcard` (Luhn-valid), `currency` (ISO 4217), `misc.territory.alpha2`, `misc.httpstatus.code` — which `--list` shows. `car`, `currency`, `httpstatus`, `language`, `mimetype`, `territory`, `timezone` and `useragent` are [tables](#table), so `misc.territory[SE].capital` and -`misc.currency[Euro].symbol` select a row; +`misc.currency[Euro].symbol` select a row; `car` and `useragent` carry no key or +name, so they are drawn from rather than selected in. [`DATA-LICENSES.md`](DATA-LICENSES.md) names each table's source and licence. -`misc.timezone` is tzdb's zone for a territory, with the territory's code and the -zone's standard offset — not the offset in force on any given date, which a zone -name is what you store precisely to avoid. It links to `misc.territory`, so -`misc.territory[SE].timezone` is `Europe/Stockholm` and a drawn territory and zone -agree. `misc.useragent` carries `browser`, `device` and `os` beside the string, and -`misc.car` a `make` and a `model`. +`misc.timezone` is every zone tzdb gives a shipped territory — 401 of them, from one +apiece for most to 29 for the US — with the territory's code and the zone's standard +offset, not the offset in force on any given date, which a zone name is what you +store precisely to avoid. It links to `misc.territory`, so `misc.territory[SE].timezone` +is `Europe/Stockholm` and a drawn territory and zone agree. `misc.useragent` carries +`browser`, `device` and `os` beside the string, and `misc.car` a `make` and a `model`. ISO 3166-1 codes territories, not sovereign states, so that is what the table is called: Greenland and Åland have codes of their own, and `misc.territory.country` names the state each belongs to — `DK` for Greenland, `FI` for Åland, and its own -code for a sovereign one. +code for a sovereign one, or for one the register names no state for. `sex`, `first-name` and `last-name` are tables weighted by bearers, from SCB, the SSA and the Census Bureau. `first-name` links to `sex`, so `sv_SE.sex[f].first-name` @@ -1045,8 +1046,7 @@ App developers writing tests and fixtures, in Go and at a shell: A table is a category with a TSV beside its file, so only a root choice has the spelling the fence names; a nested choice of same-shaped templates and an inline one keep loading. Fields must all be strings because a cell is a string node: a - choice whose items carry a nested choice, as `misc.car` does, is not one table but - two linked ones, which a later conversion writes. + choice whose items carry a nested choice is not one table but two linked ones. - **A table is a record of string columns.** Its columns are the CSV header and the `INSERT` column list, fixed by the TSV header, so a table is a record by construction; every column is a string until a typed column option earns its place. @@ -1146,9 +1146,10 @@ App developers writing tests and fixtures, in Go and at a shell: load, since nothing could then select it. - **`misc` is what every locale shares.** A category whose facts differ by country belongs in that country's locale, read from the register that country's own - records use; `misc` takes only sources that are international. So NHTSA vPIC - builds `en_US.car` and Mobility Sweden's registrations `sv_SE.car`, never - `misc.car`. + records use; `misc` takes only sources that are international. NHTSA vPIC and + Mobility Sweden's registrations are national, so they build `en_US.car` and + `sv_SE.car`; `misc.car` waits for an international source rather than take one + of theirs. - **A register's canonical spelling loses to the one its domain writes.** Where a source offers several spellings of one fact, the shipped one is what records in that domain carry. `misc.timezone` reads `zone.tab` and not the `zone1970.tab` diff --git a/data-import/territory.py b/data-import/territory.py index 101b7ad..be8723b 100644 --- a/data-import/territory.py +++ b/data-import/territory.py @@ -18,6 +18,7 @@ OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "territory.tsv" CACHE = Path(__file__).resolve().parent / "cache" COLUMNS = ["alpha2", "alpha3", "calling-code", "capital", "country", "currency", "flag", "languages", "name", "numeric", "tld"] SOVEREIGN = re.compile(r"^(?:Part of|Territor(?:y|ies) of|Crown dependency of|Commonwealth of|Associated with) ([A-Z]{2})$") +STANDALONE = {"In contention", "International"} # Gaps in the source, keyed by alpha2. FIXUPS = {"TR": {"currency": "TRY"}} @@ -38,7 +39,11 @@ def languages(field): def country(alpha2, independent): """The sovereign state the register records; a territory it records none for stands alone.""" m = SOVEREIGN.match(independent) - return m.group(1) if m else alpha2 + if m: + return m.group(1) + if independent == "Yes" or independent in STANDALONE: + return alpha2 + sys.exit(f"{alpha2}: is_independent {independent!r} names no sovereign this script can read") def rows(text): diff --git a/data-import/timezone.py b/data-import/timezone.py index 868046f..947f370 100644 --- a/data-import/timezone.py +++ b/data-import/timezone.py @@ -3,9 +3,8 @@ data-import/timezone.py [--source URL_OR_FILE] [--cache DIR] [--out FILE] [--territories FILE] -zone.tab names one zone per territory, so Europe/Stockholm ships where zone1970.tab -would spell Sweden Europe/Berlin. The offset is the zone's standard offset, the first -field of its Zone rule's last continuation line, with a Link resolved to its target. +The offset is the zone's standard offset, the first field of its Zone rule's last +continuation line, with a Link resolved to its target. """ import argparse import csv @@ -26,12 +25,19 @@ COLUMNS = ["offset", "territory", "zone"] REGIONS = ["africa", "antarctica", "asia", "australasia", "backward", "etcetera", "europe", "northamerica", "southamerica"] +def member(tar, name): + try: + return tar.extractfile(name).read().decode("utf-8") + except KeyError: + sys.exit(f"{name}: the tarball no longer holds it; the tzdb layout has moved") + + def offsets(tar): """Every zone's standard offset, and every link's target.""" std, links = {}, {} for name in REGIONS: zone = None - for raw in tar.extractfile(name).read().decode("utf-8").splitlines(): + for raw in member(tar, name).splitlines(): line = raw.split("#")[0].rstrip() if not line.strip(): continue @@ -59,19 +65,22 @@ def resolve(zone, std, links): def utc_offset(raw): - """±HH:MM from a tzdb STDOFF field; a zone still off by seconds is not one we can spell.""" - m = re.match(r"^(-)?(\d{1,2}):(\d{2})(?::(\d{2}))?$", raw) + """±HH:MM from a tzdb STDOFF field, which writes the minutes and seconds only when it has them.""" + m = re.match(r"^(-)?(\d{1,2})(?::(\d{2})(?::(\d{2}))?)?$", raw) if not m or (m.group(4) or "00") != "00": return None - return f"{'-' if m.group(1) else '+'}{int(m.group(2)):02d}:{m.group(3)}" + return f"{'-' if m.group(1) else '+'}{int(m.group(2)):02d}:{m.group(3) or '00'}" def rows(tar, territories): std, links = offsets(tar) - for line in tar.extractfile("zone.tab").read().decode("utf-8").splitlines(): + for line in member(tar, "zone.tab").splitlines(): if line.startswith("#") or not line.strip(): continue - territory, _, zone = line.split("\t")[:3] + fields = line.split("\t") + if len(fields) < 3: + sys.exit(f"zone.tab: {line!r} has {len(fields)} fields; a row names a territory, a location and a zone") + territory, zone = fields[0], fields[2] if territory not in territories: continue raw = resolve(zone, std, links) @@ -79,7 +88,7 @@ def rows(tar, territories): sys.exit(f"{zone}: the tarball gives it no Zone rule and no Link to one") offset = utc_offset(raw) if offset is None: - sys.exit(f"{zone}: standard offset {raw!r} is not a whole number of minutes") + sys.exit(f"{zone}: standard offset {raw!r} is not a ±HH:MM the table can spell") yield {"offset": offset, "territory": territory, "zone": zone} diff --git a/data-import/useragent.py b/data-import/useragent.py index 9dddd94..5cfc436 100644 --- a/data-import/useragent.py +++ b/data-import/useragent.py @@ -9,6 +9,7 @@ parsed. A row ships when the string names a browser and an operating system both import argparse import json import re +import sys from pathlib import Path import source @@ -18,9 +19,9 @@ SOURCE = "https://raw.githubusercontent.com/microlinkhq/top-user-agents/master/s OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "useragent.tsv" CACHE = Path(__file__).resolve().parent / "cache" COLUMNS = ["browser", "device", "os", "ua"] -# First match wins: Edge, Opera and Samsung Internet all carry Chrome's token too, -# and an iPhone says "like Mac OS X". -BROWSERS = [("Edge", r"Edg(iOS)?/"), ("Opera", r"OPR/"), ("Samsung Internet", r"SamsungBrowser/"), +# First match wins: every Chromium fork carries Chrome's token too, so one this list +# does not name would ship as Chrome rather than be dropped. +BROWSERS = [("Edge", r"Edg(A|iOS)?/"), ("Opera", r"OPR/"), ("Samsung Internet", r"SamsungBrowser/"), ("Chrome", r"(Chrome|CriOS)/"), ("Firefox", r"(Firefox|FxiOS)/"), ("Safari", r"Version/[\d.]+ .*Safari")] SYSTEMS = [("iOS", r"iPhone|iPad|CPU OS "), ("Android", r"Android"), ("ChromeOS", r"CrOS"), ("Windows", r"Windows NT"), ("macOS", r"Macintosh|Mac OS X"), ("Linux", r"X11.*Linux|Ubuntu")] @@ -34,11 +35,20 @@ def named(table, ua): def rows(desktop, mobile): + seen, kept = set(), [] for device, uas in (("desktop", desktop), ("mobile", mobile)): for ua in uas: browser, os = named(BROWSERS, ua), named(SYSTEMS, ua) - if browser and os: - yield {"browser": browser, "device": device, "os": os, "ua": ua} + if not browser or not os: + print(f"dropped, naming no {'browser' if not browser else 'operating system'}: {ua}", file=sys.stderr) + continue + if ua in seen: + sys.exit(f"{ua}: listed twice, which would draw it twice as often") + seen.add(ua) + kept.append({"browser": browser, "device": device, "os": os, "ua": ua}) + if len(kept) < 0.8 * (len(desktop) + len(mobile)): + sys.exit(f"only {len(kept)} of {len(desktop) + len(mobile)} strings named both; the tokens have moved") + return kept def main():