Files
fejkdata/data-import/timezone.py
T
lilleman 2ad00b3569
Tests / vet + fmt + tests (pull_request) Successful in 1m23s
Tests / Gitea release from CHANGELOG.md (pull_request) Has been skipped
Refuse an is_independent the script cannot read, name Edge on Android, and correct the claims review found false
2026-09-18 23:22:55 +02:00

115 lines
4.1 KiB
Python

#!/usr/bin/env python3
"""Rebuild data/misc/timezone.tsv from the IANA tzdb tarball (public domain).
data-import/timezone.py [--source URL_OR_FILE] [--cache DIR] [--out FILE] [--territories FILE]
The offset is the zone's standard offset, the first field of its Zone rule's last
continuation line, with a Link resolved to its target.
"""
import argparse
import csv
import io
import re
import sys
import tarfile
from pathlib import Path
import source
import tsv
SOURCE = "https://data.iana.org/time-zones/tzdata-latest.tar.gz"
OUT = Path(__file__).resolve().parent.parent / "data" / "misc" / "timezone.tsv"
TERRITORIES = Path(__file__).resolve().parent.parent / "data" / "misc" / "territory.tsv"
CACHE = Path(__file__).resolve().parent / "cache"
COLUMNS = ["offset", "territory", "zone"]
REGIONS = ["africa", "antarctica", "asia", "australasia", "backward", "etcetera", "europe", "northamerica", "southamerica"]
def member(tar, name):
try:
return tar.extractfile(name).read().decode("utf-8")
except KeyError:
sys.exit(f"{name}: the tarball no longer holds it; the tzdb layout has moved")
def offsets(tar):
"""Every zone's standard offset, and every link's target."""
std, links = {}, {}
for name in REGIONS:
zone = None
for raw in member(tar, name).splitlines():
line = raw.split("#")[0].rstrip()
if not line.strip():
continue
if line.startswith("Zone"):
f = line.split()
zone, std[f[1]] = f[1], f[2]
elif line.startswith("Link"):
f = line.split()
links[f[2]], zone = f[1], None
elif line[0] in " \t" and zone:
std[zone] = line.split()[0]
else:
zone = None
return std, links
def resolve(zone, std, links):
for _ in range(10):
if zone in std:
return std[zone]
if zone not in links:
return None
zone = links[zone]
return None
def utc_offset(raw):
"""±HH:MM from a tzdb STDOFF field, which writes the minutes and seconds only when it has them."""
m = re.match(r"^(-)?(\d{1,2})(?::(\d{2})(?::(\d{2}))?)?$", raw)
if not m or (m.group(4) or "00") != "00":
return None
return f"{'-' if m.group(1) else '+'}{int(m.group(2)):02d}:{m.group(3) or '00'}"
def rows(tar, territories):
std, links = offsets(tar)
for line in member(tar, "zone.tab").splitlines():
if line.startswith("#") or not line.strip():
continue
fields = line.split("\t")
if len(fields) < 3:
sys.exit(f"zone.tab: {line!r} has {len(fields)} fields; a row names a territory, a location and a zone")
territory, zone = fields[0], fields[2]
if territory not in territories:
continue
raw = resolve(zone, std, links)
if raw is None:
sys.exit(f"{zone}: the tarball gives it no Zone rule and no Link to one")
offset = utc_offset(raw)
if offset is None:
sys.exit(f"{zone}: standard offset {raw!r} is not a ±HH:MM the table can spell")
yield {"offset": offset, "territory": territory, "zone": zone}
def main():
p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
p.add_argument("--cache", default=str(CACHE))
p.add_argument("--source", default=SOURCE)
p.add_argument("--out", default=str(OUT))
p.add_argument("--territories", default=str(TERRITORIES))
a = p.parse_args()
shipped = {r["alpha2"] for r in csv.DictReader(io.StringIO(Path(a.territories).read_text(encoding="utf-8")), delimiter="\t")}
body = source.fetch(a.source, a.cache, "tzdata-latest.tar.gz", magic=b"\x1f\x8b")
with tarfile.open(fileobj=io.BytesIO(body)) as tar:
table = sorted(rows(tar, shipped), key=lambda r: r["zone"])
print(f"tzdb {tar.extractfile('version').read().decode('utf-8').strip()}", file=sys.stderr)
linked = {r["territory"] for r in table}
if missing := shipped - linked:
sys.exit(f"no zone for {sorted(missing)}; a parent row without a child is a load error")
tsv.write(a.out, COLUMNS, table)
if __name__ == "__main__":
main()