diff --git a/AGENTS.md b/AGENTS.md index 5bab122..abdbe46 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,6 +1,6 @@ # Rules -- Go runs only through `docker compose run --rm `; the merge gate is `docker build .` plus CI's changelog check. +- Go runs only through `docker compose run --rm `; the merge gate is `docker build .` plus CI's changelog check. - Tests first, in their own commit; the implementation follows in the next. A re-pin of seeded output or of `testdata/shipped_shape.txt` is its own commit. - A change under `data/` or to the shape pin adds its `CHANGELOG.md` entry under `Unreleased` in the same PR, and so does a change to a flag, an exit code, an exported name, a fence, a builtin or the lowest Go; what is major is the README's Versioning table. - One-line commit messages: no ticket prefix, no repo name, no authorship trailers. @@ -8,4 +8,4 @@ - One spelling per result: reject the other at `New`, and let the error name the spelling to use. - A standing choice a reader would relitigate goes under Decisions in the README, not in a comment. - A README example is a `json` block that loads and renders as a category; `readme_test.go` runs every one. -- Cyclomatic complexity is gated at 14: the table-shaped dispatches (`eachToken`, `calc.factor`, `walkPath`) sit at 13–14 and stay whole; anything else that reaches 14 is decomposed. +- Cyclomatic complexity is gated at 14: the table-shaped dispatches (`linkParent`, `compileTemplate`, `renderEdges`) sit at it and stay whole; a function that would pass it is decomposed. diff --git a/CHANGELOG.md b/CHANGELOG.md index b83d133..d7d5cf2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -13,8 +13,9 @@ replacement, and each removed path, column or flag. `key`, `name`, `weight` and `parent`; a path selects a row by key or name, `misc.country[SE]`, and descends to a linked table by name; linked tables draw consistently within one render and draw group. `rows` is an option, so no - template may carry a field of that name. Refused at `New`: a `name` without a - `key`, a name spelling another row's key, a table named like a column of any + template may carry a field of that name. A `name` without a `key` resolves + inside the table's `parent`. Refused at `New`: a `name` without a `key` or a + `parent`, a name repeating inside one parent row, a name spelling another row's key, a table named like a column of any table above it, a table whose format or cell references its own family, and, within one render and draw group, a path drawing a table another path selects a row of, or two paths pinning different rows of one table. @@ -32,3 +33,31 @@ replacement, and each removed path, column or flag. `address` record over one consistent draw of them. `sv_SE.address` and `en_US.address` read those records, so `en_US.address.street` no longer carries `name` and `suffix`, and a locale folder loads only beside `geo`. +- `{date(from,to,'layout')}` and `{time('layout')}`: a second between two days, or + within one, in a single-quoted Go layout, drawn in UTC; `from` may equal `to`. + `sv_SE.date`, `en_US.date`, `sv_SE.time` and `en_US.time` render through them, so + `date.year`, `date.month`, `date.day`, `time.hour`, `time.minute`, `time.minute.t`, + `time.sec` and `time.ampm` are no longer paths — a part of a date is now its own + `{date(…,'2006')}`. `misc.datetime` is an RFC 3339 instant. +- `sex`, `first-name` and `last-name` tables in `sv_SE` and `en_US`, weighted by + bearers from SCB, the SSA and the Census Bureau; `first-name` links to `sex`, and a + name both sexes carry is a row under each. `person` reads them, so its columns are + `first`, `last`, `prefix` and `sex`; `person.femalefirst` and `person.malefirst` are + no longer paths — draw `sex[f].first-name` and `sex[m].first-name` instead. + `en_US.title` links to `sex` as well, so `en_US.person.prefix` draws `Mr` or `Ms` + without contradicting the record's `sex`; `sv_SE.title` is an unsexed table of the + same shape. +- `sv_SE.personnummer` and `sv_SE.samordningsnummer`, Skatteverket's test series + from a `sv_SE.birth-number` table under `sex`, in place of `sv_SE.ssn`, whose + `ssn.mmdd`, `ssn.mmdd.m` and `ssn.mmdd.d` go with it. `en_US.ssn` now draws the + ranges the SSA assigns and carries the columns `area`, `group` and `serial`, so + `--format csv en_US.ssn` writes a header where it used to fail; `en_US.itin` is new + and carries the same three. `person.prefix` is null where a person has no title, + where it used to be an empty string, so `--format sql` writes `NULL` and a `string` + struct field reading it becomes `*string`. +- `ErrNoColumns` is exported, so a caller can tell the one record fence a path can + answer from the rest. +- An error names a spelling that runs: a layout is named single-quoted and free of + its own quotes, a row of a table with no key is named as the path that selects it, + `sv_SE.sex[f].first-name[Kim]`, and a category with no columns names the record + that gives it one. diff --git a/DATA-LICENSES.md b/DATA-LICENSES.md index 3b63be0..1436f95 100644 --- a/DATA-LICENSES.md +++ b/DATA-LICENSES.md @@ -11,6 +11,10 @@ Every shipped dataset, its source, its licence and the attribution it asks for. | `geo/SE/street.tsv` | [Trafikverket NVDB](https://www.trafikverket.se/) Gatunamn, through the open API | CC0 1.0 | none required | `data-import/geo-se.py` | | `geo/US/region.tsv`, `municipality.tsv`, `locality.tsv` | [Census Bureau](https://www.census.gov/) Gazetteer 2026 and population estimates 2025 | [public domain](https://www.usa.gov/government-works) | none required | `data-import/geo-us.py` | | `geo/US/postal-code.tsv`, `street.tsv` | Census Bureau ZCTA to place relationships 2020 and TIGER/Line 2025 address ranges and feature names | public domain | none required | `data-import/geo-us.py` | +| `sv_SE/first-name.tsv`, `last-name.tsv` | [SCB](https://www.scb.se/) names with at least two bearers, 31 December 2022 | CC0 1.0 | "Källa: SCB" | `data-import/names-se.py` | +| `en_US/first-name.tsv` | [SSA](https://www.ssa.gov/oact/babynames/) baby names, births 1930 to 2020, through [hackerb9/ssa-baby-names](https://github.com/hackerb9/ssa-baby-names) | public domain | none required | `data-import/names-us.py` | +| `en_US/last-name.tsv` | Census Bureau surnames occurring 100 or more times, 2010 | public domain | none required | `data-import/names-us.py` | +| `sv_SE/sex.tsv`, `sv_SE/birth-number.tsv`, `sv_SE/title.tsv`, `en_US/sex.tsv`, `en_US/title.tsv` | curated (Skatteverket's test birth numbers are facts) | — | — | — | | `misc/country.tsv` | [datasets/country-codes](https://github.com/datasets/country-codes) | [PDDL 1.0](https://opendatacommons.org/licenses/pddl/1-0/) | none required | `data-import/country.py` | | `misc/currency.tsv` | [datasets/currency-codes](https://github.com/datasets/currency-codes); symbols from [Unicode CLDR](https://github.com/unicode-org/cldr) `en.xml` and `root.xml` | PDDL 1.0; [Unicode License v3](https://www.unicode.org/license.txt) | CLDR: "Copyright © 1991-2025 Unicode, Inc. Unicode and the Unicode Logo are registered trademarks of Unicode, Inc. in the United States and other countries." | `data-import/currency.py` | | `misc/httpstatus.tsv` | curated (IANA HTTP status codes are facts) | — | — | — | diff --git a/README.md b/README.md index bfd609c..69c6078 100644 --- a/README.md +++ b/README.md @@ -22,6 +22,7 @@ fejkdata 'geo.SE.locality[Lund].street' # Fjelievägen — a linked tabl fejkdata --data-path ./mydata sv_SE.word # layer a directory over the shipped data fejkdata --no-shipped-data -d ./mydata --list # only your data fejkdata 'name: {/sv_SE.person.last}' # name: — an inline template +fejkdata "{date(1990-01-01,2010-12-31,'2006-01-02')}" # 2003-11-27 — the argument in "…", the layout in '…' fejkdata '{"format":"name: {x}","x":["bosse","lina"]}' # name: bosse or name: lina ``` @@ -32,11 +33,9 @@ a JSON object, array or string, or that carries a `{` token, is instead an **inline template**: a format string or a JSON value compiled and rendered on the spot. Its tokens reach the data by reference from the root — `{/sv_SE.person.last}`, so shipped and `--data-path` categories are alike -available. An inline template sits in no folder, so the folder-relative `{.name}` -and `{..name}` are rejected naming the root spelling, and one reference alone — -`{/sv_SE.person}` — is the path written as a template, rejected naming the path, as is -a path written `/sv_SE.person`. A path never contains a brace or a quote, and a -bracket only as a selector after a name, so the two cannot collide (see [Decisions](#decisions)). +available. A path never contains a brace or a quote, and a bracket only as a selector +after a name, so the two spellings cannot collide; which spellings an inline template +rejects, and what each names instead, is under [Decisions](#decisions). | Flag | | |------|--| @@ -114,6 +113,12 @@ the template itself, which composes the format into one string rather than projecting columns — ask for more records with `--repeat`. A `repeat` on a column is fine. +Columns are written in name order, whatever order the fields appear in. A category +whose fields are the parts of one value — `sv_SE.price`, `sv_SE.version`, `misc.uuid` — +projects those parts rather than the value, so for one column holding what `Fake` +renders, write `{"format":"","price":"{/sv_SE.price}"}` as a fieldless category asks +for. + A column carrying a newline keeps it inside the quoted CSV field or the SQL string literal, so a row can span physical lines: read the stream with a CSV or SQL parser rather than splitting it on newlines. @@ -141,6 +146,72 @@ a value two fields share in its own category and reference that. A field hold, t operand ties fields together within one column as always (see [Correlated fields](#correlated-fields) and [Decisions](#decisions)). +## Data + +The shipped set under [`data/`](data) — one folder per locale (`en_US`, `sv_SE`) +plus a locale-neutral `misc` folder — is embedded, so the CLI and the library +work with no data on disk. A directory is a namespace: each JSON file is a +category named after the file, each subdirectory a dot-path segment, so +`mydata/sv_SE/person.json` is `sv_SE.person` and replaces the shipped one. +Sources merge in order; matching folders combine, any other clash is won by the +last loaded. Names may not use `.`, `|`, `(`, `{`, `}`, `[`, `]`, `"` or `/`, nor be +`-`, which a struct tag reserves; dot-prefixed entries are skipped, so a data directory can also be a checkout. + +Each locale carries `address`, `color`, `company`, `date`, `email`, `first-name`, +`ip`, `last-name`, `person`, `phone`, `price`, `sentence`, `sex`, `time`, `url`, +`username`, `version` and `word`, formatted per locale; `sv_SE` adds +`personnummer` and `samordningsnummer`, `en_US` adds `ssn` and `itin`. `misc` +carries `car`, `coordinate`, `country` (ISO 3166), `creditcard` (Luhn-valid), +`currency` (ISO 4217), `datetime` (RFC 3339), `emoji`, `httpstatus`, `language` +(ISO 639), `mac`, `mimetype`, `objectid`, `timezone` (IANA), `useragent` and +`uuid` (v4). Many carry sub-fields — `misc.currency.symbol`, +`misc.country.alpha2`, `misc.httpstatus.code` — which `--list` shows. `country`, +`currency`, `httpstatus`, `language` and `mimetype` are [tables](#table), so +`misc.country[SE].capital` and `misc.currency[Euro].symbol` select a row; +[`DATA-LICENSES.md`](DATA-LICENSES.md) names each table's source and licence. + +`sex`, `first-name` and `last-name` are tables weighted by bearers, from SCB, the +SSA and the Census Bureau. `first-name` links to `sex`, so `sv_SE.sex[f].first-name` +draws a woman's name, and a name both sexes carry is a row under each, so +`en_US.sex[m].first-name[Taylor]` names the one a `first-name[Taylor]` alone cannot. +`person` reads one draw of the three, so its `first` and `sex` columns agree, and so +does a `personnummer` in the same render: its birth number, `sv_SE.birth-number` +under `sex`, is Skatteverket's test series, 238 for a woman and 239 for a man, which no +real person is ever given. `en_US.title` links to `sex` too, so a person's prefix +never contradicts it. + +A person of a chosen sex is assembled from the tables — `sex[f].first-name` beside +`last-name` — while a shipped `personnummer` agrees with the sex its own render +*drew*, not with one a path selects. That test series is also small: a personnummer +is one of about 70,000 values, a day in 1930–2025 against the two birth numbers, so a +fixture past a few hundred rows repeats one and a `UNIQUE` column needs a category of +your own. `sv_SE.date` and `en_US.date` are uniform over 1970-01-01 to 2029-12-31, +`misc.datetime` over 2000-01-01 to 2029-12-31. What the two locales do not share: +`en_US.address` carries a `region` column the Swedish one has no use for, and +`sv_SE.title` has no `parent`, so it is selected as `sv_SE.title[dr]` rather than +inside a sex. + +A `geo` folder holds one tree per country under its alpha-2 code: five +[linked tables](#linked-tables) named alike, and an `address` record over one +consistent draw of them, which the locale's `address` reads. + +| Table | `geo.SE` | `geo.US` | Weight | +|-------|----------|----------|--------| +| `region` | län, by code or name | state, by USPS abbreviation or name; `code` is the FIPS code | population | +| `municipality` | kommun, by code or name | county, by FIPS code or name | population | +| `locality` | postort, by name | incorporated place of 25,000 people or more with a postal code of its own, by GEOID or name; Hawaii has none | tätort population, the kommun's where the postort names it, else 200; place population | +| `postal-code` | postnummer with street delivery, by code | ZCTA, by code | one; address ranges | +| `street` | gatunamn, the ten with most road segments per postort | street name, the ten with most address ranges per place | segments; address ranges | + +`geo.SE.region[Skåne län].municipality` draws a kommun in Skåne, +`geo.SE.locality[Lund].street` a street in Lund, and +`geo.US.region[IL].locality[Springfield]` settles which Springfield. A region row +carries its `timezone`, the state's predominant zone, and a locality its `lat` and +`lon`. What ports across countries is the five table names, the `name` column, +selection by name, and the `address` record's columns `street`, `street-number`, +`postal-code` and `locality`; every other column is the country's own, `code` on a +Swedish region but `abbr` on a US one. + ## Library ```sh @@ -206,49 +277,6 @@ A `*Generator` is safe for concurrent use; a seeded sequence is reproducible onl when drawn from one goroutine. Changing how a value is composed shifts the seeded stream for that value and everything drawn after it. -## Data - -The shipped set under [`data/`](data) — one folder per locale (`en_US`, `sv_SE`) -plus a locale-neutral `misc` folder — is embedded, so the CLI and the library -work with no data on disk. A directory is a namespace: each JSON file is a -category named after the file, each subdirectory a dot-path segment, so -`mydata/sv_SE/person.json` is `sv_SE.person` and replaces the shipped one. -Sources merge in order; matching folders combine, any other clash is won by the -last loaded. Names may not use `.`, `|`, `(`, `{`, `}`, `[`, `]`, `"` or `/`, nor be -`-`, which a struct tag reserves; dot-prefixed entries are skipped, so a data directory can also be a checkout. - -Each locale carries `address`, `color`, `company`, `date`, `email`, `ip`, -`person`, `phone`, `price`, `sentence`, `ssn`, `time`, `url`, `username`, -`version` and `word`, formatted per locale. `misc` carries `car`, `coordinate`, -`country` (ISO 3166), `creditcard` (Luhn-valid), `currency` (ISO 4217), `emoji`, -`httpstatus`, `language` (ISO 639), `mac`, `mimetype`, `objectid`, `timezone` -(IANA), `useragent` and `uuid` (v4). Many carry sub-fields — `misc.currency.symbol`, -`misc.country.alpha2`, `misc.httpstatus.code` — which `--list` shows. `country`, -`currency`, `httpstatus`, `language` and `mimetype` are [tables](#table), so -`misc.country[SE].capital` and `misc.currency[Euro].symbol` select a row; -[`DATA-LICENSES.md`](DATA-LICENSES.md) names each table's source and licence. - -A `geo` folder holds one tree per country under its alpha-2 code: five -[linked tables](#linked-tables) named alike, and an `address` record over one -consistent draw of them, which the locale's `address` reads. - -| Table | `geo.SE` | `geo.US` | Weight | -|-------|----------|----------|--------| -| `region` | län, by code or name | state, by USPS abbreviation or name; `code` is the FIPS code | population | -| `municipality` | kommun, by code or name | county, by FIPS code or name | population | -| `locality` | postort, by name | incorporated place of 25,000 people or more with a postal code of its own, by GEOID or name; Hawaii has none | tätort population, the kommun's where the postort names it, else 200; place population | -| `postal-code` | postnummer with street delivery, by code | ZCTA, by code | one; address ranges | -| `street` | gatunamn, the ten with most road segments per postort | street name, the ten with most address ranges per place | segments; address ranges | - -`geo.SE.region[Skåne län].municipality` draws a kommun in Skåne, -`geo.SE.locality[Lund].street` a street in Lund, and -`geo.US.region[IL].locality[Springfield]` settles which Springfield. A region row -carries its `timezone`, the state's predominant zone, and a locality its `lat` and -`lon`. What ports across countries is the five table names, the `name` column, -selection by name, and the `address` record's columns `street`, `street-number`, -`postal-code` and `locality`; every other column is the country's own, `code` on a -Swedish region but `abbr` on a US one. - ## Data format Every value is a **node**, nestable without limit: @@ -407,7 +435,9 @@ cell token, and refuses a TSV no category names, a key that is empty or repeats, weight that is not a positive number, and a key or name holding `[`, `]`, `{`, `}`, `"` or `|`, which a selector cannot spell; the rows are indexed on the first draw that selects one. A `name` needs a `key`, since a name naming several rows is reported by -their keys, and a name spelling another row's key is refused, since the key would +their keys, or a `parent`, inside whose row a name names one row, so `first-name[Kim]` +is settled by the `sex` selected before it and a name repeating inside one parent row +is refused; a name spelling another row's key is refused, since the key would select first and the name never. The table's options are its own — `rows`, `key`, `name`, `weight` and `parent` — so a column may be named `name`, as one usually is. @@ -505,21 +535,34 @@ stays reproducible. | `{ulid()}` | sample | ULID, 26 Crockford base32 chars | | `{nanoid(n)}` | sample | URL-safe Nano ID, `n` chars | | `{iban(CC)}` | sample | length- and mod-97-valid IBAN for BE, DE, DK, ES, FI, NO or SE | +| `{date(from,to,'layout')}` | sample | a second between two `YYYY-MM-DD` days, both included, in a quoted Go layout: `'2006-01-02'`, `'January 2, 2006'`, `'060102'`, `'2006-01-02T15:04:05Z'` | +| `{time('layout')}` | sample | a second within a day: `'15:04'`, `'3:04 PM'` | | `{seq()}`, `{seq(name)}` | counter | next integer from 1 in this generator; `name` selects an independent counter | | `{calc(expr)}`, `{calc(expr,dp)}` | computation | an arithmetic expression over sibling fields ([Computation](#computation)) | | `{lowercase(x)}`, `{uppercase(x)}`, `{ascii(x)}` | transform | a field's value rewritten ([Transforms](#transforms)) | A derivation reads what is to its left, so place it after its payload; the buffer is per expansion, so a nested template keeps fixed parts out of the sum. A -Swedish personnummer is a Luhn checksum over the nine digits before it: +Swedish personnummer is a Luhn checksum over the nine digits before it, six of +them a birthdate: ```json -{ "format": "{century}{core}", "century": ["19", "20"], - "core": { "format": "{digits(2)}{mmdd}-{digits(3)}{luhn()}", "mmdd": ["0115", "0704", "1218"] } } +{ "format": "{date(1930-01-01,2010-12-31,'060102')}-{birth}{luhn()}", "birth": ["238", "239"] } ``` -Renders e.g. `19811218-9876`. `{seq()}` spans `Fake` calls and `repeat`, resets -with a new generator, and is the natural primary key for the SQL example above. +Renders e.g. `811218-2389`. A layout is Go's: the reference time `Mon Jan 2 +15:04:05 MST 2006` spelled as the output should look, quoted, since a layout may +carry the comma that separates arguments, with English names. Every second +between the two days is reachable, so a layout with a clock draws the time too, and +`from` may equal `to`, which is that one day. The instant is UTC, so a zone in the +layout prints `UTC` or `Z`. +The quotes delimit a layout outside a selector only, so `[O'Fallon]` in an +argument stays a name. Rejected at `New`: a bound that is no calendar date, or not +before the other; an unquoted layout, naming the single-quoted one; a layout naming no field, which is +text, as is one day in a layout with no clock; for `date` a layout naming no date +field, naming `time`; and for `time` a layout naming a date field, naming `date`. `{seq()}` spans `Fake` +calls and `repeat`, resets with a new generator, and is the natural primary key for +the SQL example above. ### Computation @@ -573,9 +616,9 @@ without naming `sv_SE`: Renders e.g. `Hej, Pat Smith!`. A reference path into a category is held like a [correlated](#correlated-fields) path, but for the whole render — one `Fake`, or one -record — rather than one format: `{.person.femalefirst} {.person.last}` name one +record — rather than one format: `{.person.first} {.person.last}` name one person, as do the same two references in sibling fields or a nested template, and -`{lowercase(.person.femalefirst)}` reads that same draw. Each `repeat` iteration is +`{lowercase(.person.first)}` reads that same draw. Each `repeat` iteration is a render of its own, in no group, so it draws anew, and a [draw group](#draw-group) holds a draw apart. A bare reference names no field and makes its own picks each time — `{/misc.uuid} {/misc.uuid}` is two draws — while the reference paths inside what it @@ -594,8 +637,8 @@ its groups by name; the unnamed group spans them all. ```json { "format": "{payer} pays {payee}; signed {signature}", - "payer": { "format": "{/sv_SE.person.femalefirst} {/sv_SE.person.last}", "drawGroup": "payer" }, - "payee": "{/sv_SE.person.femalefirst} {/sv_SE.person.last}", + "payer": { "format": "{/sv_SE.person.first} {/sv_SE.person.last}", "drawGroup": "payer" }, + "payee": "{/sv_SE.person.first} {/sv_SE.person.last}", "signature": { "format": "{/sv_SE.person.last}", "drawGroup": "payer" } } ``` @@ -697,6 +740,15 @@ datatype and nullability, and each table's key, name, weight and parent columns; changes it or `data/` adds its `CHANGELOG.md` entry, which CI checks. A removed, renamed or retyped line is a major. +## Audience + +App developers writing tests and fixtures, in Go and at a shell: + +- a **bulk fixture author**, thousands of rows into CSV or SQL +- a **Go test author**, filling a struct with `FakeStruct` +- a **hand fixture author**, one value at a shell +- a **validator-facing author**, who needs a value a real checker accepts + ## Goals 1. **Valid by construction** — every value passes the check its real consumer @@ -728,8 +780,7 @@ renamed or retyped line is a major. key, or prefixing options, would tax every template to guard against a misspelt option. - **`{a|b}` stays beside nested choices.** `[[…], […]]` picks the same way, but - its arms are anonymous; `{femalefirst|malefirst}` keeps `person.femalefirst` - addressable. + its arms are anonymous; `{female|male}` keeps `person.female` addressable. - **Flags follow getopt_long.** `--name value` and `--name=value` both work; a short flag's value attaches or follows (`-s42`, `-s 42`) and short flags bundle (`-hn 3`), as every shell user expects. A single-dash long flag is rejected @@ -747,7 +798,9 @@ renamed or retyped line is a major. library's own advice reachable: the error for an object holding only a format names `"…"`, and that spelling has to work where it is printed. An argument or struct tag of one reference alone, `{/users}`, is refused naming the path `users`: both - render the same text, and only the path names a record. `IsTemplate` exports the + render the same text, and only the path names a record. A folder-relative `{.name}` + or `{..name}` is refused naming `{/name}`, since an inline template sits in no + folder. `IsTemplate` exports the rule, so the CLI, struct tags and any other caller read one. - **An inline template skips the cycle fence.** `New` proves the loaded tree acyclic, an inline node is a finite tree of its own, and nothing in the tree can @@ -1040,6 +1093,51 @@ renamed or retyped line is a major. bigger neighbour ship no address; counting the land outside every place too would drop a quarter of the places, whose codes straddle unincorporated land, for a postal city the USPS mostly names the same way. +- **`--list` stays a plain list of paths.** It is what a script reads, so every line + has to be a path that `Fake` takes; a marker for the tables a `[selector]` follows, + or a legend above them, would make the output something to parse before use. + `--help` names the selector spelling instead, and the Table section teaches it. +- **A layout is always quoted.** A layout may carry the comma that separates + arguments, `'January 2, 2006'`, and one spelling for every layout beats a rule + about which ones need the quotes, so the bare spelling is refused naming the + quoted one. The layout is Go's reference time because the library renders with + it and a Go caller already knows it; its names are English, and a locale's own + month and weekday names are data. +- **A title is a table under `sex`.** A prefix drawn apart would put `Mr` on a record + whose `sex` column says `female`, which is the disagreement the record exists to + prevent; the tables this set already has are what a title needs, so `en_US.title` + links to `sex` as `first-name` does. Swedish has no everyday sexed honorific, so + `sv_SE.title` is a table as well but carries no `parent`. Its weight column is + `share`, not the `count` a name table carries, because the values are a curated + proportion rather than bearers anyone counted. +- **A table owns the spelling of a selector on it.** A reference reaches a table by a + path that carries no selector — `sv_SE.person.first` reads `first-name` through + `sex` — so the walk that resolved a name cannot say where a reader would type one. + The table's own location can, which is why it keeps its path, and why an ambiguity + error names `sv_SE.sex[f].first-name[Kim]` rather than the table's own name. +- **No builtin reads the clock, so a date is bounded by days, never by an age.** + An `age(min,max)` would make a seeded fixture change with the day it runs on, + which is what a seed exists to prevent; a birthdate for someone 20 to 60 is + `date(1966-01-01,2006-12-31,…)`, re-pinned as any fixture is. +- **A name column without a key resolves inside its parent.** A given name both + sexes carry is a row under each, so `name` cannot be the key; the parent's row + tells the two apart, `sex[f].first-name[Kim]`, the ambiguity error spells each + row inside its parent, and a name repeating inside one parent row is refused at + load, since nothing could then select it. +- **The Swedish ids draw Skatteverket's test series.** A Luhn-valid personnummer + over a random birth number may be a living person's; 238 and 239 after any date + are blocked from assignment, so the shipped `personnummer` and + `samordningsnummer` use those. They sit in a `birth-number` table under `sex` + rather than as a column of it: the render's shared draw of the family is what + makes the number and the name agree on sex, and `sex` stays one shape across + locales instead of collecting every sex-keyed id fact. A samordningsnummer's + day, the birthday plus 60, is drawn from 61 to 88, valid in every month, rather + than computed from the date drawn. +- **The US given names come from a mirror of the SSA file.** ssa.gov refuses a + client outside the US, so `names-us.py` reads a GitHub copy that ends at 2020, + which a count over the births since 1930 barely feels; `--names` takes the + official zip. The SSA's placeholder rows are top-1000 entries that name nobody, so + the import drops them by name rather than by a rank a regeneration would move. - **`List` advertises direct descents only.** `region.municipality.locality` is listed, and `region.locality` resolves too but is not: the set of every descent through a chain of five tables is every subsequence of it, and the direct chain is @@ -1091,15 +1189,19 @@ A shipped table built from a source is rebuilt by its script under [`data-import/`](data-import), one command per dataset, fetching the source named in [`DATA-LICENSES.md`](DATA-LICENSES.md). Downloads are cached under `data-import/cache/`, so delete it to fetch afresh; `geo-us.py` fetches two -TIGER/Line files per county it ships, a few hundred megabytes, and `geo-se.py` needs +TIGER/Line files per county it ships, a few hundred megabytes, `geo-se.py` needs a Trafikverket API key, free at [data.trafikverket.se](https://data.trafikverket.se/), -in `TRAFIKVERKET_API_KEY` or a `--key-file`: +in `TRAFIKVERKET_API_KEY` or a `--key-file`, and the Census host behind `geo-us.py` +and `names-us.py` rejects a client for a while after a burst, so `--surnames` takes +a copy of the surname file: ```sh docker compose run --rm --user "$(id -u):$(id -g)" data-import data-import/country.py docker compose run --rm --user "$(id -u):$(id -g)" data-import data-import/currency.py docker compose run --rm --user "$(id -u):$(id -g)" data-import data-import/geo-us.py docker compose run --rm --user "$(id -u):$(id -g)" -e TRAFIKVERKET_API_KEY data-import data-import/geo-se.py +docker compose run --rm --user "$(id -u):$(id -g)" data-import data-import/names-se.py +docker compose run --rm --user "$(id -u):$(id -g)" data-import data-import/names-us.py ``` To release, head `CHANGELOG.md` with the version's section in place of `Unreleased` @@ -1125,6 +1227,9 @@ family.go a family of linked tables: the rows a render pins, and the fence reference.go reference sigils, and binding references across the tree graph.go the render graph: edges, cycles, the repeat bound, tree walks builtins.go the {name()} function registry and its implementations +layout.go date and time layouts: the instants one is proved against, and the two samples +checksum.go the check characters a derivation appends, and the IBAN they sit inside +transform.go the builtins that rewrite an operand's value, and the ASCII folding calc.go the {calc()} arithmetic evaluator: parser, eval, validation datatype.go column datatypes: DataType, where datatype and null may sit, a column's datatype value.go the value proof: what a typed column or calc operand holds, checked at load diff --git a/bench_test.go b/bench_test.go index e016d58..3df242a 100644 --- a/bench_test.go +++ b/bench_test.go @@ -26,12 +26,12 @@ func benchPath(b *testing.B, dir, path string) { } } -func BenchmarkPerson(b *testing.B) { benchPath(b, "data", "sv_SE.person") } -func BenchmarkAddress(b *testing.B) { benchPath(b, "data", "sv_SE.address") } -func BenchmarkWord(b *testing.B) { benchPath(b, "data", "sv_SE.word") } -func BenchmarkCreditcard(b *testing.B) { benchPath(b, "data", "misc.creditcard") } -func BenchmarkSSN(b *testing.B) { benchPath(b, "data", "sv_SE.ssn") } -func BenchmarkUUIDv7(b *testing.B) { benchPath(b, "data", "misc.uuid") } +func BenchmarkPerson(b *testing.B) { benchPath(b, "data", "sv_SE.person") } +func BenchmarkAddress(b *testing.B) { benchPath(b, "data", "sv_SE.address") } +func BenchmarkWord(b *testing.B) { benchPath(b, "data", "sv_SE.word") } +func BenchmarkCreditcard(b *testing.B) { benchPath(b, "data", "misc.creditcard") } +func BenchmarkPersonnummer(b *testing.B) { benchPath(b, "data", "sv_SE.personnummer") } +func BenchmarkUUIDv7(b *testing.B) { benchPath(b, "data", "misc.uuid") } func tmpData(b *testing.B, name, body string) string { b.Helper() diff --git a/builtins.go b/builtins.go index 6075668..c6d6c0d 100644 --- a/builtins.go +++ b/builtins.go @@ -7,7 +7,6 @@ import ( "math" "strconv" "strings" - "unicode" ) // maxLen caps sample output lengths (hex, nanoid, base64, digits, upper, lower) @@ -60,6 +59,8 @@ var builtins = map[string]builtin{ cc := a[0] return func(s *session, _ string, _ []string) string { return iban(s, cc) } }}, + "date": {arity: -1, check: dateArgs, prep: datePrep}, + "time": {arity: -1, check: timeArg, prep: timePrep}, "calc": {arity: -1, check: checkCalc, prep: calcPrep, operands: calcOperands}, "lowercase": {arity: 1, check: transformArg, prep: transformPrep(strings.ToLower), operands: transformOperand}, "uppercase": {arity: 1, check: transformArg, prep: transformPrep(strings.ToUpper), operands: transformOperand}, @@ -88,13 +89,11 @@ func derive(f func(emitted string) string) func([]string) callFn { return func(_ *session, emitted string, _ []string) string { return f(emitted) } } } - func sample(f func(rng) string) func([]string) callFn { return func([]string) callFn { return func(s *session, _ string, _ []string) string { return f(s) } } } - func chars(alphabet string) func([]string) callFn { return func(a []string) callFn { n := atoi(a[0]) @@ -113,96 +112,6 @@ func formatFloat(v float64, dp int) string { return s } -// transforms are the builtins that rewrite one operand's value; they nest, so -// {lowercase(ascii(x))} folds then lowers. -var transforms = map[string]func(string) string{ - "ascii": asciiFold, - "lowercase": strings.ToLower, - "uppercase": strings.ToUpper, -} - -// unwrapTransform peels nested transform calls off an operand arg, returning the -// field it finally names and the transforms to apply, innermost last. -func unwrapTransform(arg string) (leaf string, chain []func(string) string, err error) { - for { - name, args, isCall := funcCall(arg) - if !isCall { - return arg, chain, nil - } - fn, isTransform := transforms[name] - if !isTransform { - return "", nil, fmt.Errorf("%s(%s) is not a transform, so it cannot be an operand", name, strings.Join(args, ",")) - } - if len(args) != 1 { - return "", nil, fmt.Errorf("%s takes 1 arg, got %d", name, len(args)) - } - chain = append(chain, fn) - arg = args[0] - } -} - -func transformArg(fields map[string]node, a []string) error { - leaf, _, err := unwrapTransform(a[0]) - if err != nil { - return err - } - if isRef(leaf) { - _, _, err := refShape(leaf) - return err - } - return checkArm(leaf, fields, false) -} - -func transformOperand(a []string) []string { - leaf, _, err := unwrapTransform(a[0]) - if err != nil { - return nil - } - return []string{leaf} -} - -func transformPrep(outer func(string) string) func([]string) callFn { - return func(a []string) callFn { - _, chain, err := unwrapTransform(a[0]) - if err != nil { - panic(fmt.Sprintf("fejkdata: transform arg %q reached prep unvalidated: %v", a[0], err)) - } - return func(_ *session, _ string, operands []string) string { - v := operands[0] - for i := len(chain) - 1; i >= 0; i-- { - v = chain[i](v) - } - return outer(v) - } - } -} - -// asciiFolds maps the Latin letters with diacritics or ligatures to ASCII. -var asciiFolds = map[rune]string{ - 'À': "A", 'Á': "A", 'Â': "A", 'Ã': "A", 'Ä': "A", 'Å': "A", 'Æ': "AE", 'Ç': "C", - 'È': "E", 'É': "E", 'Ê': "E", 'Ë': "E", 'Ì': "I", 'Í': "I", 'Î': "I", 'Ï': "I", - 'Ð': "D", 'Ñ': "N", 'Ò': "O", 'Ó': "O", 'Ô': "O", 'Õ': "O", 'Ö': "O", 'Ø': "O", - 'Ù': "U", 'Ú': "U", 'Û': "U", 'Ü': "U", 'Ý': "Y", 'Þ': "Th", 'ß': "ss", 'Œ': "OE", - 'à': "a", 'á': "a", 'â': "a", 'ã': "a", 'ä': "a", 'å': "a", 'æ': "ae", 'ç': "c", - 'è': "e", 'é': "e", 'ê': "e", 'ë': "e", 'ì': "i", 'í': "i", 'î': "i", 'ï': "i", - 'ð': "d", 'ñ': "n", 'ò': "o", 'ó': "o", 'ô': "o", 'õ': "o", 'ö': "o", 'ø': "o", - 'ù': "u", 'ú': "u", 'û': "u", 'ü': "u", 'ý': "y", 'þ': "th", 'ÿ': "y", 'œ': "oe", -} - -// asciiFold rewrites s to ASCII: folded Latin letters stay, any other non-ASCII -// rune is dropped. -func asciiFold(s string) string { - var b strings.Builder - for _, r := range s { - if r <= unicode.MaxASCII { - b.WriteRune(r) - } else { - b.WriteString(asciiFolds[r]) - } - } - return b.String() -} - // atoi parses an arg a builtin's check already validated. It panics rather than // returning zero, so a check that stops covering its own args is a stack trace and // not a silently wrong length, range or decimal count. @@ -222,7 +131,6 @@ func atof(s string) float64 { } return f } - func randBytes(r rng, n int) []byte { b := make([]byte, n) for i := range b { @@ -230,7 +138,6 @@ func randBytes(r rng, n int) []byte { } return b } - func randChars(r rng, n int, alphabet string) string { b := make([]byte, n) for i := range b { @@ -253,7 +160,6 @@ func plainInt(s string) (int, error) { } return n, nil } - func posIntArg(_ map[string]node, a []string) error { n, err := plainInt(a[0]) if errors.Is(err, strconv.ErrRange) { @@ -270,7 +176,6 @@ func posIntArg(_ map[string]node, a []string) error { } return nil } - func intRangeArgs(_ map[string]node, a []string) error { lo, err := plainInt(a[0]) if err != nil { @@ -291,7 +196,6 @@ func intRangeArgs(_ map[string]node, a []string) error { } return nil } - func floatArgs(_ map[string]node, a []string) error { lo, e1 := strconv.ParseFloat(a[0], 64) hi, e2 := strconv.ParseFloat(a[1], 64) @@ -319,10 +223,9 @@ func floatArgs(_ map[string]node, a []string) error { } return nil } - func seqArg(_ map[string]node, a []string) error { if len(a) > 1 { - return fmt.Errorf("seq takes at most one name, got %d args", len(a)) + return fmt.Errorf("seq takes at most one name, got %d", len(a)) } if len(a) == 1 && a[0] == "" { return fmt.Errorf("seq name must not be empty") @@ -373,96 +276,3 @@ func ulid(r rng) string { } return string(out) } - -// luhnCheck returns the Luhn check digit (0-9) over the digits of s; non-digit -// runes are skipped. Doubling runs from the rightmost digit, so the result is -// correct whatever the payload length. -func luhnCheck(s string) int { - sum, double := 0, true - for i := len(s) - 1; i >= 0; i-- { - c := s[i] - if c < '0' || c > '9' { - continue - } - d := int(c - '0') - if double { - if d *= 2; d > 9 { - d -= 9 - } - } - double = !double - sum += d - } - return (10 - sum%10) % 10 -} - -// mod11Check returns the weighted mod-11 check character over the digits of s -// (weights 2..7 cycling from the right). A would-be value of 10 emits 'X', as in -// ISBN-10 / ISO 7064; non-digits are skipped. -func mod11Check(s string) string { - sum, w := 0, 2 - for i := len(s) - 1; i >= 0; i-- { - c := s[i] - if c < '0' || c > '9' { - continue - } - sum += int(c-'0') * w - if w++; w > 7 { - w = 2 - } - } - if chk := (11 - sum%11) % 11; chk != 10 { - return string(rune('0' + chk)) - } - return "X" -} - -// eanCheck returns the EAN-13 / UPC-A / ISBN-13 / GTIN check digit over the -// digits of s: weights 3 and 1 alternating from the rightmost digit, mod 10. -func eanCheck(s string) string { - sum, w := 0, 3 - for i := len(s) - 1; i >= 0; i-- { - c := s[i] - if c < '0' || c > '9' { - continue - } - sum += int(c-'0') * w - w = 4 - w // 3 <-> 1 - } - return string(rune('0' + (10-sum%10)%10)) -} - -// ibanLen maps a supported country code to the full IBAN length. The check digits -// sit between the country code and the BBAN, so — unlike luhn/ean — iban can't be -// a left-to-right derivation; it generates the whole value instead. -var ibanLen = map[string]int{"BE": 16, "DE": 22, "DK": 18, "ES": 24, "FI": 18, "NO": 15, "SE": 24} - -func ibanArg(_ map[string]node, a []string) error { - if _, ok := ibanLen[a[0]]; !ok { - return fmt.Errorf("iban(%q): unsupported country code", a[0]) - } - return nil -} - -// iban generates a structurally valid IBAN for cc: a numeric BBAN of the right -// length, then mod-97 check digits. Real bank/branch structure isn't modelled — -// the result passes length and checksum validation, which is what fake data needs. -func iban(r rng, cc string) string { - bban := make([]byte, ibanLen[cc]-4) - for i := range bban { - bban[i] = byte('0' + r.IntN(10)) - } - rem := 0 - feed := func(d int) { rem = (rem*10 + d) % 97 } - for _, c := range bban { - feed(int(c - '0')) - } - for i := 0; i < len(cc); i++ { // letters A-Z -> 10..35, fed as two digits - v := int(cc[i]-'A') + 10 - feed(v / 10) - feed(v % 10) - } - feed(0) - feed(0) - return fmt.Sprintf("%s%02d%s", cc, 98-rem, bban) -} diff --git a/builtins_test.go b/builtins_test.go index 31b2f72..adcf039 100644 --- a/builtins_test.go +++ b/builtins_test.go @@ -4,7 +4,9 @@ import ( "encoding/base64" "regexp" "strconv" + "strings" "testing" + "time" ) func TestBuiltinIDGenerators(t *testing.T) { @@ -41,6 +43,7 @@ func TestBuiltinSamplesReproducible(t *testing.T) { `"{uuid()}"`, `"{ulid()}"`, `"{nanoid(12)}"`, `"{int(1,1000000)}"`, `"{float(0,1,6)}"`, `"{base64(12)}"`, `"{iban(SE)}"`, + `"{date(2000-01-01,2020-12-31,'2006-01-02 15:04:05')}"`, `"{time('15:04:05')}"`, } { if a, b := mustRender(t, engine(7), tmpl), mustRender(t, engine(7), tmpl); a != b { t.Fatalf("%s not reproducible: %q != %q", tmpl, a, b) @@ -233,3 +236,135 @@ func TestClassBuiltinArgs(t *testing.T) { } } } + +// TestBuiltinDateAndTime pins the two clock-free samples: date draws a second in +// [from 00:00:00, to 23:59:59], both days reachable, and renders it in the quoted Go +// layout, commas and English names included; time draws a second within one day. +func TestBuiltinDateAndTime(t *testing.T) { + f := engine(1) + seen := map[string]bool{} + for i := 0; i < 500; i++ { + got := mustRender(t, f, `"{date(1990-01-01,1990-12-31,'2006-01-02')}"`) + if d, err := time.Parse("2006-01-02", got); err != nil || d.Year() != 1990 { + t.Fatalf("date = %q, want a 1990 calendar date (err %v)", got, err) + } + seen[got] = true + } + if len(seen) < 200 { + t.Fatalf("date drew %d distinct days of 365 in 500, want a uniform spread", len(seen)) + } + lo, hi := false, false + for i := 0; i < 200; i++ { + got := mustRender(t, f, `"{date(2020-02-28,2020-02-29,'2006-01-02')}"`) + if got != "2020-02-28" && got != "2020-02-29" { + t.Fatalf("date(2020-02-28,2020-02-29) = %q, out of range", got) + } + lo, hi = lo || got == "2020-02-28", hi || got == "2020-02-29" + } + if !lo || !hi { + t.Fatalf("date never hit a bound: lo=%v hi=%v (bounds must be inclusive)", lo, hi) + } + if got := mustRender(t, f, `"{date(2020-07-04,2020-07-05,'January 2, 2006')}"`); got != "July 4, 2020" && got != "July 5, 2020" { + t.Fatalf("date with a comma in its layout = %q", got) + } + rfc := regexp.MustCompile(`^2021-\d\d-\d\dT\d\d:\d\d:\d\dZ$`) + clock := regexp.MustCompile(`^([01]\d|2[0-3]):[0-5]\d$`) + ampm := regexp.MustCompile(`^(1[0-2]|[1-9]):[0-5]\d (AM|PM)$`) + seconds := map[string]bool{} + for i := 0; i < 300; i++ { + got := mustRender(t, f, `"{date(2021-01-01,2021-12-31,'2006-01-02T15:04:05Z07:00')}"`) + if !rfc.MatchString(got) { + t.Fatalf("date in an RFC 3339 layout = %q, want %s", got, rfc) + } + seconds[got[17:19]] = true + if got := mustRender(t, f, `"{time('15:04')}"`); !clock.MatchString(got) { + t.Fatalf("time('15:04') = %q, want %s", got, clock) + } + if got := mustRender(t, f, `"{time('3:04 PM')}"`); !ampm.MatchString(got) { + t.Fatalf("time('3:04 PM') = %q, want %s", got, ampm) + } + got = mustRender(t, f, `"{date(1950-01-01,2000-12-31,'060102')}-238{luhn()}"`) + if d := digitsOnly(got); len(d) != 10 || !luhnValid(d) { + t.Fatalf("a personnummer over date() = %q, want ten Luhn-valid digits", got) + } + } + if len(seconds) < 30 { + t.Fatalf("date drew %d distinct seconds in 300, want the whole day, not midnight", len(seconds)) + } +} + +// TestBuiltinDateArgs pins the New-time checks: bounds are calendar dates in order, +// the layout is quoted, names a field, and for time names no date field. +func TestBuiltinDateArgs(t *testing.T) { + for tmpl, want := range map[string]string{ + `"{date(1990-13-01,1990-12-31,'2006-01-02')}"`: "1990-13-01", + `"{date(1990-12-31,1990-01-01,'2006-01-02')}"`: "is after", + `"{date(1990-01-01,1990-01-01,'2006-01-02')}"`: "write it as text", + `"{date(1990-01-01,1990-12-31,2006-01-02)}"`: "'2006-01-02'", + `"{date(1990-01-01,1990-12-31,'January 2, 2006)}"`: "'", + `"{date(1990-01-01,1990-12-31,\"January 2, 2006\")}"`: "quoted: 'January 2, 2006'", + `"{date(1990-01-01,1990-12-31,'x')}"`: "text", + `"{date(1990-01-01,1990-12-31,'')}"`: "text", + `"{date(1990-01-01,1990-12-31)}"`: "3 arguments", + `"{date(1990-01-01,1990-12-31,January 2, 2006)}"`: "'January 2, 2006'", + `"{time(3:04 PM, Mon)}"`: "'3:04 PM, Mon'", + `"{time(15:04)}"`: "'15:04'", + `"{time('2006-01-02 15:04')}"`: "date(", + `"{time('x')}"`: "text", + `"{date(1990-01-01,1990-12-31,'15:04')}"`: "time('15:04')", + } { + _, err := compile(parse(t, tmpl)) + if err == nil || !strings.Contains(err.Error(), want) { + t.Errorf("compile(%s) = %v, want an error mentioning %q", tmpl, err, want) + } + } +} + +// TestBuiltinLayoutErrorsNameARunnableSpelling pins the two layout errors a user +// can follow: a double-quoted layout is named single-quoted, without its own +// quotes carried into the suggestion, whether it split on a comma or not. +func TestBuiltinLayoutErrorsNameARunnableSpelling(t *testing.T) { + _, err := compile(parse(t, `"{date(1990-01-01,1990-12-31,\"2006-01-02\")}"`)) + if err == nil || !strings.Contains(err.Error(), "write '2006-01-02'") { + t.Fatalf("a double-quoted layout = %v, want it named single-quoted", err) + } + if strings.Contains(err.Error(), `'"`) { + t.Errorf("%v names a layout that renders its own quotes", err) + } + if !strings.Contains(err.Error(), "is double-quoted") { + t.Errorf("%v does not say which quotes were wrong", err) + } + _, err = compile(parse(t, `"{time(15:04)}"`)) + if err == nil || !strings.Contains(err.Error(), "write '15:04'") { + t.Fatalf("a bare layout = %v, want it named quoted", err) + } + // The comma hint belongs to a layout that split, not to a call given extra args. + _, err = compile(parse(t, `"{time(0,12,'15:04')}"`)) + if err == nil || !strings.Contains(err.Error(), "takes 1 argument, got 3") { + t.Fatalf("time with three args = %v, want the count named in the singular", err) + } + if strings.Contains(err.Error(), "holding a comma") { + t.Errorf("%v offers the comma hint though the layout is already quoted", err) + } +} + +// TestBuiltinDateSpansOneDay pins that from == to is a day: with a clock layout it +// draws every second of it, and without one it could only emit one value. +func TestBuiltinDateSpansOneDay(t *testing.T) { + f := engine(1) + seen := map[string]bool{} + for i := 0; i < 500; i++ { + got := mustRender(t, f, `"{date(2026-01-01,2026-01-01,'2006-01-02 15:04:05')}"`) + if !strings.HasPrefix(got, "2026-01-01 ") { + t.Fatalf("date over one day = %q, out of range", got) + } + seen[got] = true + } + if len(seen) < 400 { + t.Fatalf("date over one day drew %d distinct seconds in 500, want the whole day", len(seen)) + } + _, err := compile(parse(t, `"{date(2026-01-01,2026-01-01,'2006-01-02')}"`)) + if err == nil || !strings.Contains(err.Error(), "write it as text") { + t.Fatalf("one day in a date-only layout = %v, want it named a constant", err) + } +} diff --git a/checksum.go b/checksum.go new file mode 100644 index 0000000..8029f75 --- /dev/null +++ b/checksum.go @@ -0,0 +1,96 @@ +package fejkdata + +import "fmt" + +// luhnCheck returns the Luhn check digit (0-9) over the digits of s; non-digit +// runes are skipped. Doubling runs from the rightmost digit, so the result is +// correct whatever the payload length. +func luhnCheck(s string) int { + sum, double := 0, true + for i := len(s) - 1; i >= 0; i-- { + c := s[i] + if c < '0' || c > '9' { + continue + } + d := int(c - '0') + if double { + if d *= 2; d > 9 { + d -= 9 + } + } + double = !double + sum += d + } + return (10 - sum%10) % 10 +} + +// mod11Check returns the weighted mod-11 check character over the digits of s +// (weights 2..7 cycling from the right). A would-be value of 10 emits 'X', as in +// ISBN-10 / ISO 7064; non-digits are skipped. +func mod11Check(s string) string { + sum, w := 0, 2 + for i := len(s) - 1; i >= 0; i-- { + c := s[i] + if c < '0' || c > '9' { + continue + } + sum += int(c-'0') * w + if w++; w > 7 { + w = 2 + } + } + if chk := (11 - sum%11) % 11; chk != 10 { + return string(rune('0' + chk)) + } + return "X" +} + +// eanCheck returns the EAN-13 / UPC-A / ISBN-13 / GTIN check digit over the +// digits of s: weights 3 and 1 alternating from the rightmost digit, mod 10. +func eanCheck(s string) string { + sum, w := 0, 3 + for i := len(s) - 1; i >= 0; i-- { + c := s[i] + if c < '0' || c > '9' { + continue + } + sum += int(c-'0') * w + w = 4 - w // 3 <-> 1 + } + return string(rune('0' + (10-sum%10)%10)) +} + +// ibanLen maps a supported country code to the full IBAN length. The check digits +// sit between the country code and the BBAN, so — unlike luhn/ean — iban can't be +// a left-to-right derivation; it generates the whole value instead. +var ibanLen = map[string]int{"BE": 16, "DE": 22, "DK": 18, "ES": 24, "FI": 18, "NO": 15, "SE": 24} + +func ibanArg(_ map[string]node, a []string) error { + if _, ok := ibanLen[a[0]]; !ok { + return fmt.Errorf("iban(%q): unsupported country code", a[0]) + } + return nil +} + +// iban generates a structurally valid IBAN for cc: a numeric BBAN of the right +// length, then mod-97 check digits. Real bank/branch structure isn't modelled — +// the result passes length and checksum validation, which is what fake data needs. +func iban(r rng, cc string) string { + bban := make([]byte, ibanLen[cc]-4) + for i := range bban { + bban[i] = byte('0' + r.IntN(10)) + } + rem := 0 + feed := func(d int) { rem = (rem*10 + d) % 97 } + for _, c := range bban { + feed(int(c - '0')) + } + for i := 0; i < len(cc); i++ { // letters A-Z -> 10..35, fed as two digits + v := int(cc[i]-'A') + 10 + feed(v / 10) + feed(v % 10) + } + feed(0) + feed(0) + return fmt.Sprintf("%s%02d%s", cc, 98-rem, bban) +} diff --git a/cmd/fejkdata/main.go b/cmd/fejkdata/main.go index dd3d0ef..57ccd79 100644 --- a/cmd/fejkdata/main.go +++ b/cmd/fejkdata/main.go @@ -28,6 +28,9 @@ const usage = `Usage: fejkdata [flags]