Refuse an is_independent the script cannot read, name Edge on Android, and correct the claims review found false
Tests / vet + fmt + tests (pull_request) Successful in 1m23s
Tests / Gitea release from CHANGELOG.md (pull_request) Has been skipped

This commit is contained in:
2026-09-18 23:22:55 +02:00
parent 8a17fb1548
commit 2ad00b3569
6 changed files with 62 additions and 36 deletions
+6 -1
View File
@@ -18,6 +18,7 @@ OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "territory.tsv"
CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["alpha2", "alpha3", "calling-code", "capital", "country", "currency", "flag", "languages", "name", "numeric", "tld"]
SOVEREIGN = re.compile(r"^(?:Part of|Territor(?:y|ies) of|Crown dependency of|Commonwealth of|Associated with) ([A-Z]{2})$")
STANDALONE = {"In contention", "International"}
# Gaps in the source, keyed by alpha2.
FIXUPS = {"TR": {"currency": "TRY"}}
@@ -38,7 +39,11 @@ def languages(field):
def country(alpha2, independent):
"""The sovereign state the register records; a territory it records none for stands alone."""
m = SOVEREIGN.match(independent)
return m.group(1) if m else alpha2
if m:
return m.group(1)
if independent == "Yes" or independent in STANDALONE:
return alpha2
sys.exit(f"{alpha2}: is_independent {independent!r} names no sovereign this script can read")
def rows(text):
+19 -10
View File
@@ -3,9 +3,8 @@
data-import/timezone.py [--source URL_OR_FILE] [--cache DIR] [--out FILE] [--territories FILE]
zone.tab names one zone per territory, so Europe/Stockholm ships where zone1970.tab
would spell Sweden Europe/Berlin. The offset is the zone's standard offset, the first
field of its Zone rule's last continuation line, with a Link resolved to its target.
The offset is the zone's standard offset, the first field of its Zone rule's last
continuation line, with a Link resolved to its target.
"""
import argparse
import csv
@@ -26,12 +25,19 @@ COLUMNS = ["offset", "territory", "zone"]
REGIONS = ["africa", "antarctica", "asia", "australasia", "backward", "etcetera", "europe", "northamerica", "southamerica"]
def member(tar, name):
try:
return tar.extractfile(name).read().decode("utf-8")
except KeyError:
sys.exit(f"{name}: the tarball no longer holds it; the tzdb layout has moved")
def offsets(tar):
"""Every zone's standard offset, and every link's target."""
std, links = {}, {}
for name in REGIONS:
zone = None
for raw in tar.extractfile(name).read().decode("utf-8").splitlines():
for raw in member(tar, name).splitlines():
line = raw.split("#")[0].rstrip()
if not line.strip():
continue
@@ -59,19 +65,22 @@ def resolve(zone, std, links):
def utc_offset(raw):
"""±HH:MM from a tzdb STDOFF field; a zone still off by seconds is not one we can spell."""
m = re.match(r"^(-)?(\d{1,2}):(\d{2})(?::(\d{2}))?$", raw)
"""±HH:MM from a tzdb STDOFF field, which writes the minutes and seconds only when it has them."""
m = re.match(r"^(-)?(\d{1,2})(?::(\d{2})(?::(\d{2}))?)?$", raw)
if not m or (m.group(4) or "00") != "00":
return None
return f"{'-' if m.group(1) else '+'}{int(m.group(2)):02d}:{m.group(3)}"
return f"{'-' if m.group(1) else '+'}{int(m.group(2)):02d}:{m.group(3) or '00'}"
def rows(tar, territories):
std, links = offsets(tar)
for line in tar.extractfile("zone.tab").read().decode("utf-8").splitlines():
for line in member(tar, "zone.tab").splitlines():
if line.startswith("#") or not line.strip():
continue
territory, _, zone = line.split("\t")[:3]
fields = line.split("\t")
if len(fields) < 3:
sys.exit(f"zone.tab: {line!r} has {len(fields)} fields; a row names a territory, a location and a zone")
territory, zone = fields[0], fields[2]
if territory not in territories:
continue
raw = resolve(zone, std, links)
@@ -79,7 +88,7 @@ def rows(tar, territories):
sys.exit(f"{zone}: the tarball gives it no Zone rule and no Link to one")
offset = utc_offset(raw)
if offset is None:
sys.exit(f"{zone}: standard offset {raw!r} is not a whole number of minutes")
sys.exit(f"{zone}: standard offset {raw!r} is not a ±HH:MM the table can spell")
yield {"offset": offset, "territory": territory, "zone": zone}
+15 -5
View File
@@ -9,6 +9,7 @@ parsed. A row ships when the string names a browser and an operating system both
import argparse
import json
import re
import sys
from pathlib import Path
import source
@@ -18,9 +19,9 @@ SOURCE = "https://raw.githubusercontent.com/microlinkhq/top-user-agents/master/s
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "useragent.tsv"
CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["browser", "device", "os", "ua"]
# First match wins: Edge, Opera and Samsung Internet all carry Chrome's token too,
# and an iPhone says "like Mac OS X".
BROWSERS = [("Edge", r"Edg(iOS)?/"), ("Opera", r"OPR/"), ("Samsung Internet", r"SamsungBrowser/"),
# First match wins: every Chromium fork carries Chrome's token too, so one this list
# does not name would ship as Chrome rather than be dropped.
BROWSERS = [("Edge", r"Edg(A|iOS)?/"), ("Opera", r"OPR/"), ("Samsung Internet", r"SamsungBrowser/"),
("Chrome", r"(Chrome|CriOS)/"), ("Firefox", r"(Firefox|FxiOS)/"), ("Safari", r"Version/[\d.]+ .*Safari")]
SYSTEMS = [("iOS", r"iPhone|iPad|CPU OS "), ("Android", r"Android"), ("ChromeOS", r"CrOS"),
("Windows", r"Windows NT"), ("macOS", r"Macintosh|Mac OS X"), ("Linux", r"X11.*Linux|Ubuntu")]
@@ -34,11 +35,20 @@ def named(table, ua):
def rows(desktop, mobile):
seen, kept = set(), []
for device, uas in (("desktop", desktop), ("mobile", mobile)):
for ua in uas:
browser, os = named(BROWSERS, ua), named(SYSTEMS, ua)
if browser and os:
yield {"browser": browser, "device": device, "os": os, "ua": ua}
if not browser or not os:
print(f"dropped, naming no {'browser' if not browser else 'operating system'}: {ua}", file=sys.stderr)
continue
if ua in seen:
sys.exit(f"{ua}: listed twice, which would draw it twice as often")
seen.add(ua)
kept.append({"browser": browser, "device": device, "os": os, "ua": ua})
if len(kept) < 0.8 * (len(desktop) + len(mobile)):
sys.exit(f"only {len(kept)} of {len(desktop) + len(mobile)} strings named both; the tokens have moved")
return kept
def main():