From 044294dd9084f44da129af07ed10cc4be0ecabe7 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 15 Sep 2026 11:17:21 +0200 Subject: [PATCH 01/10] Tests: typed columns, null, and the load check that every render parses as its datatype --- datatype_test.go | 161 +++++++++++++++++++++++++++++++++++++++++++ one_spelling_test.go | 13 ++-- readme_test.go | 41 +++++++++++ record_test.go | 51 +++++++++++--- 4 files changed, 252 insertions(+), 14 deletions(-) create mode 100644 datatype_test.go diff --git a/datatype_test.go b/datatype_test.go new file mode 100644 index 0000000..aa8c4a1 --- /dev/null +++ b/datatype_test.go @@ -0,0 +1,161 @@ +package fejkdata + +import ( + "encoding/json" + "regexp" + "slices" + "strings" + "testing" +) + +func TestDatatypeAndNullSitOnlyInAColumn(t *testing.T) { + for _, src := range []string{ + `{"format":"","age":{"format":"{int(18,99)}","datatype":"integer"}}`, + `{"format":"","n":{"format":"42","datatype":"integer"}}`, + `{"format":"","gone":null}`, + `{"format":"","middle":[null,"Ann","Eva"]}`, + `{"format":"","age":[null,{"format":"{int(18,99)}","datatype":"integer","weight":9}]}`, + `{"format":"","pick":[[null,"a"],"b"]}`, + } { + if _, err := compile(parse(t, src)); err != nil { + t.Errorf("compile(%s) = %v, want a column to take a datatype and null", src, err) + } + } + for src, want := range map[string]string{ + `{"format":"","n":{"format":"1","datatype":"int"}}`: `datatype takes "integer", "number" or "boolean", got "int"`, + `{"format":"","n":{"format":"1","datatype":1}}`: "datatype must be a string", + `{"format":"{int(1,9)}","datatype":"integer"}`: "datatype only types a record column", + `[{"format":"1","datatype":"integer"},"x"]`: "datatype only types a record column", + `{"format":"{p}","p":{"format":"{n}","n":{"format":"1","datatype":"integer"}}}`: "datatype only types a record column", + `{"format":"{n}","repeat":2,"n":{"format":"1","datatype":"integer"}}`: "datatype only types a record column", + `null`: `so write ""`, + `{"format":"{p}","p":{"format":"{x}","x":[null,"a"]}}`: `so write ""`, + `{"format":"","c":[{"format":"1","datatype":"integer"},"x"]}`: "a column holds one datatype", + } { + if _, err := compile(parse(t, src)); err == nil || !strings.Contains(err.Error(), want) { + t.Errorf("compile(%s) = %v, want an error containing %q", src, err, want) + } + } +} + +func TestDatatypeRejectsARenderItsTypeRejects(t *testing.T) { + cat := `[{"format":"{code}","code":"200"},{"format":"{code}","code":"2x"}]` + for _, c := range []struct{ name, column, want string }{ + {"a leading zero", `{"format":"{digits(3)}","datatype":"integer"}`, "which is not an integer"}, + {"a fraction", `{"format":"{v}","v":["1","1.5"],"datatype":"integer"}`, `can render "1.5", which is not an integer`}, + {"a signed sample before digits", `{"format":"{int(-5,5)}{digits(2)}","datatype":"integer"}`, "which is not an integer"}, + {"a separator", `{"format":"{int(1,9)}","repeat":2,"separator":",","datatype":"integer"}`, "which is not an integer"}, + {"through a reference", `{"format":"{/cat.code}","datatype":"integer"}`, `can render "2x"`}, + {"a bare dot", `{"format":".5","datatype":"number"}`, `can render ".5", which is not a number`}, + {"a trailing dot", `{"format":"{int(1,9)}.","datatype":"number"}`, "which is not a number"}, + {"a plus sign", `{"format":"+1","datatype":"number"}`, `can render "+1"`}, + {"a capital", `{"format":"{b}","b":["true","True"],"datatype":"boolean"}`, `can render "True", which is not a boolean`}, + {"an upper-casing transform", `{"format":"{uppercase(b)}","b":["true","false"],"datatype":"boolean"}`, "which is not a boolean"}, + {"an operand that is not always a number", `{"format":"{calc(a * 2)}","a":["1","x"],"datatype":"number"}`, `operand "a" can render "x"`}, + {"a divisor that can be zero", `{"format":"{calc(a / b)}","a":"{int(1,9)}","b":"{int(0,9)}","datatype":"number"}`, "divides by b, which can be zero"}, + {"an overflow", `{"format":"{calc(a * a)}","a":"{digits(200)}","datatype":"number"}`, "can overflow"}, + {"a division in an integer column", `{"format":"{calc(a / b)}","a":"{int(1,9)}","b":"{int(1,9)}","datatype":"integer"}`, "which is not an integer"}, + } { + row := `{"format":"","col":` + c.column + `}` + _, err := New(WithoutShippedData(), WithDataPath(writeData(t, map[string]string{"cat": cat, "row": row}))) + if err == nil || !strings.Contains(err.Error(), c.want) { + t.Errorf("%s: New = %v, want an error containing %q", c.name, err, c.want) + } + f := newGenerator(t, writeData(t, map[string]string{"cat": cat})) + if _, err := f.NewTemplate(row); err == nil || !strings.Contains(err.Error(), c.want) { + t.Errorf("%s: NewTemplate = %v, want the inline template refused the same way", c.name, err) + } + } +} + +var integerText = regexp.MustCompile(`^-?(0|[1-9][0-9]*)$`) + +func TestDatatypeAcceptsAColumnThatAlwaysParses(t *testing.T) { + cat := `[{"format":"{code}","code":"200"},{"format":"{code}","code":"404"}]` + for _, column := range []string{ + `{"format":"{int(1,99)}","datatype":"integer"}`, + `{"format":"-{int(1,9)}","datatype":"integer"}`, + `{"format":"1{digits(2)}","datatype":"integer"}`, + `{"format":"{seq()}","datatype":"integer"}`, + `{"format":"{/cat.code}","datatype":"integer"}`, + `{"format":"{float(-1,1,2)}","datatype":"number"}`, + `{"format":"{int(1,9)}e{int(1,9)}","datatype":"number"}`, + `{"format":"6.022e23","datatype":"number"}`, + `{"format":"{lowercase(b)}","b":["TRUE","False"],"datatype":"boolean"}`, + `{"format":"{calc(net * qty, 2)}","net":["19.99","5.00"],"qty":["3","7"],"datatype":"number"}`, + `{"format":"{calc(a + b)}","a":"{int(1,9)}","b":"{int(-9,9)}","datatype":"integer"}`, + `{"format":"{calc(a / (b + 1), 2)}","a":"{int(1,9)}","b":"{digits(2)}","datatype":"number"}`, + `{"format":"{calc(a / b, 0)}","a":"{int(1,9)}","b":"{int(1,9)}","datatype":"integer"}`, + `{"format":"{calc(sub * 1.25, 2)}","sub":{"format":"{calc(a * b)}","a":"{int(1,9)}","b":"{float(0,5,2)}"},"datatype":"number"}`, + } { + row := `{"format":"","col":` + column + `}` + f, err := New(WithoutShippedData(), WithDataPath(writeData(t, map[string]string{"cat": cat, "row": row})), WithSeed(1)) + if err != nil { + t.Errorf("%s: New = %v, want it loaded", column, err) + continue + } + for i := 0; i < 200; i++ { + r, err := f.FakeRecord("row") + if err != nil { + t.Fatal(err) + } + var m map[string]any + if err := json.Unmarshal([]byte(r.JSON()), &m); err != nil { + t.Errorf("%s: JSON() = %s is not JSON: %v", column, r.JSON(), err) + break + } + c := r.Columns()[0] + _, isBool := m["col"].(bool) + _, isNumber := m["col"].(float64) + if c.DataType == DataTypeBoolean && !isBool || c.DataType != DataTypeBoolean && !isNumber || c.DataType == DataTypeInteger && !integerText.MatchString(c.Value) { + t.Errorf("%s: column %+v written as %s, want its datatype", column, c, r.JSON()) + break + } + } + } +} + +func TestNullColumn(t *testing.T) { + dir := writeData(t, map[string]string{ + "row": `{"format":"[{middle}]","gone":null,"middle":[null,"Ann"],"score":[null,{"format":"{int(1,9)}","datatype":"integer"}]}`, + }) + f := newGenerator(t, dir, WithSeed(1)) + drew := map[bool]bool{} + for i := 0; i < 100; i++ { + r, err := f.FakeRecord("row") + if err != nil { + t.Fatal(err) + } + gone, middle, score := r.Columns()[0], r.Columns()[1], r.Columns()[2] + if !gone.Null || gone.Value != "" || gone.DataType != DataTypeString { + t.Fatalf("gone = %+v, want a null string column every draw", gone) + } + if middle.Null == (middle.Value == "Ann") { + t.Fatalf("middle = %+v, want null or Ann", middle) + } + if score.DataType != DataTypeInteger { + t.Fatalf("score = %+v, want the integer its non-null item declares, null or not", score) + } + drew[middle.Null] = true + } + if len(drew) != 2 { + t.Errorf("middle drew only null=%v in 100 records, want both", drew) + } + if v := fake(t, f, "row"); v != "[]" && v != "[Ann]" { + t.Errorf("Fake(row) = %q, want a null to render as \"\"", v) + } + if v := fake(t, f, "row.gone"); v != "" { + t.Errorf("Fake(row.gone) = %q, want \"\"", v) + } + if !slices.Contains(f.List(), "row.gone") { + t.Errorf("List() = %v, want the null column row.gone, which Fake accepts", f.List()) + } +} + +func TestEveryBuiltinSaysWhatItEmits(t *testing.T) { + for name, b := range builtins { + if _, isTransform := transforms[name]; b.emits == nil && name != "calc" && !isTransform { + t.Errorf("builtin %s declares no emits, so a typed column calling it cannot be checked", name) + } + } +} diff --git a/one_spelling_test.go b/one_spelling_test.go index d5bbd8b..ca1aeac 100644 --- a/one_spelling_test.go +++ b/one_spelling_test.go @@ -30,12 +30,13 @@ func TestRepeatedChoiceItemIsRejected(t *testing.T) { func TestInertObjectIsRejected(t *testing.T) { for src, want := range map[string]string{ - `{"format":"Malmö"}`: `write "Malmö"`, - `{"format":"{digits(3)}"}`: `write "{digits(3)}"`, - `[{"format":"a","weight":1},"b"]`: "weight 1", - `{"format":"{x}","x":"v","repeat":1}`: "repeat 1", - `{"format":"{x}","x":"v","separator":","}`: "separator", - `{"format":"{x}","x":"v","repeat":2,"separator":""}`: "default", + `{"format":"Malmö"}`: `write "Malmö"`, + `{"format":"{digits(3)}"}`: `write "{digits(3)}"`, + `[{"format":"a","weight":1},"b"]`: "weight 1", + `{"format":"{x}","x":"v","repeat":1}`: "repeat 1", + `{"format":"{x}","x":"v","separator":","}`: "separator", + `{"format":"{x}","x":"v","repeat":2,"separator":""}`: "default", + `{"format":"","n":{"format":"1","datatype":"string"}}`: `datatype "string" is the default`, } { if _, err := compile(parse(t, src)); err == nil || !strings.Contains(err.Error(), want) { t.Errorf("compile(%s) = %v, want an error mentioning %s", src, err, want) diff --git a/readme_test.go b/readme_test.go index 18790e9..c628709 100644 --- a/readme_test.go +++ b/readme_test.go @@ -1,6 +1,7 @@ package fejkdata import ( + "encoding/json" "os" "regexp" "strings" @@ -87,3 +88,43 @@ func TestReadmeSQLExampleOutput(t *testing.T) { t.Errorf("README SQL example with seed 1 = %q, README prints %q", got, want[1]) } } + +func TestReadmeDatatypeExample(t *testing.T) { + src := readme(t) + i := strings.Index(src, "### Datatype") + if i < 0 { + t.Fatal("README lost the Datatype section") + } + block := jsonBlock.FindStringSubmatch(src[i:]) + f, err := New(WithDataPath(writeData(t, map[string]string{"order": block[1]})), WithSeed(1)) + if err != nil { + t.Fatal(err) + } + r, err := f.FakeRecord("order") + if err != nil { + t.Fatal(err) + } + var m map[string]any + if err := json.Unmarshal([]byte(r.JSON()), &m); err != nil { + t.Fatalf("JSON() = %s: %v", r.JSON(), err) + } + shown := map[DataType]bool{} + for _, c := range r.Columns() { + shown[c.DataType] = true + var ok bool + switch c.DataType { + case DataTypeBoolean: + _, ok = m[c.Name].(bool) + case DataTypeInteger, DataTypeNumber: + _, ok = m[c.Name].(float64) + default: + _, ok = m[c.Name].(string) + } + if !ok { + t.Errorf("column %q, datatype %s, written as %s", c.Name, c.DataType, r.JSON()) + } + } + if !shown[DataTypeInteger] || !shown[DataTypeNumber] || !shown[DataTypeBoolean] { + t.Errorf("README Datatype example shows %v, want an integer, a number and a boolean column", shown) + } +} diff --git a/record_test.go b/record_test.go index ee2f754..520ceef 100644 --- a/record_test.go +++ b/record_test.go @@ -3,6 +3,7 @@ package fejkdata import ( "encoding/csv" "encoding/json" + "reflect" "strings" "testing" ) @@ -133,18 +134,52 @@ func TestRecordCSV(t *testing.T) { } func TestRecordCSVEmptyValueStaysARow(t *testing.T) { - dir := writeData(t, map[string]string{"blank": `{"format": "", "note": ""}`}) - f := newGenerator(t, dir, WithSeed(1)) - r, err := f.FakeRecord("blank") + for _, body := range []string{`{"format": "", "note": ""}`, `{"format": "", "note": null}`} { + f := newGenerator(t, writeData(t, map[string]string{"blank": body}), WithSeed(1)) + r, err := f.FakeRecord("blank") + if err != nil { + t.Fatal(err) + } + rows, err := csv.NewReader(strings.NewReader(r.CSVHeader() + "\n" + r.CSVLine() + "\n")).ReadAll() + if err != nil { + t.Fatalf("csv: %v", err) + } + if len(rows) != 2 || len(rows[1]) != 1 || rows[1][0] != "" { + t.Fatalf("%s: one empty column parsed to %v, want a header and one row of one empty field", body, rows) + } + } +} + +func TestRecordWritesTypedAndNullColumns(t *testing.T) { + dir := writeData(t, map[string]string{ + "row": `{"format":"","age":{"format":"42","datatype":"integer"},"gone":null,"name":"O'Brien","nick":"","paid":{"format":"true","datatype":"boolean"},"price":{"format":"19.99","datatype":"number"}}`, + }) + r, err := newGenerator(t, dir, WithSeed(1)).FakeRecord("row") if err != nil { t.Fatal(err) } - rows, err := csv.NewReader(strings.NewReader(r.CSVHeader() + "\n" + r.CSVLine() + "\n")).ReadAll() - if err != nil { - t.Fatalf("csv: %v", err) + want := []Column{ + {Name: "age", DataType: DataTypeInteger, Value: "42"}, + {Name: "gone", Null: true}, + {Name: "name", Value: "O'Brien"}, + {Name: "nick"}, + {Name: "paid", DataType: DataTypeBoolean, Value: "true"}, + {Name: "price", DataType: DataTypeNumber, Value: "19.99"}, } - if len(rows) != 2 || len(rows[1]) != 1 || rows[1][0] != "" { - t.Fatalf("one empty column parsed to %v, want a header and one row of one empty field", rows) + if got := r.Columns(); !reflect.DeepEqual(got, want) { + t.Errorf("Columns() = %+v, want %+v", got, want) + } + if got, want := r.JSON(), `{"age":42,"gone":null,"name":"O'Brien","nick":"","paid":true,"price":19.99}`; got != want { + t.Errorf("JSON() = %s, want %s", got, want) + } + if got, want := r.SQLInsert("t"), `INSERT INTO "t" ("age", "gone", "name", "nick", "paid", "price") VALUES (42, NULL, 'O''Brien', '', true, 19.99);`; got != want { + t.Errorf("SQLInsert() = %s, want %s", got, want) + } + if got, want := r.CSVLine(), `42,,O'Brien,"",true,19.99`; got != want { + t.Errorf("CSVLine() = %s, want %s: null an unquoted empty field, an empty string quoted", got, want) + } + if got := DataTypeNumber.String(); got != "number" { + t.Errorf("DataTypeNumber.String() = %q, want the data's spelling", got) } } -- 2.52.0 From 94dc562c6c9804c64f8d5b309870638721d12323 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 15 Sep 2026 11:23:57 +0200 Subject: [PATCH 02/10] Typed columns and null: a datatype option, null items, and a load check that every typed render parses --- README.md | 68 +++++++-- builtins.go | 122 ++++++++++++--- calc.go | 265 ++++++++++++++++++++++++++++++++- datatype.go | 140 ++++++++++++++++++ fejkdata.go | 2 + graph.go | 5 +- node.go | 71 ++++++--- path.go | 2 +- record.go | 125 ++++++++++------ render.go | 2 + renderlang.go | 403 ++++++++++++++++++++++++++++++++++++++++++++++++++ template.go | 3 + todo.md | 5 - 13 files changed, 1106 insertions(+), 107 deletions(-) create mode 100644 datatype.go create mode 100644 renderlang.go diff --git a/README.md b/README.md index 02c7f22..45a8b4b 100644 --- a/README.md +++ b/README.md @@ -82,8 +82,8 @@ For structured output a record writes the row for you. A record is a template seen as columns: its fields are the columns, its `format` the whole. `--format json|ndjson|csv|sql` writes the records; the library's -`FakeRecord` (below) hands back the columns. Every column is a string — typed scalars -are on the release checklist, see [`todo.md`](todo.md). Save +`FakeRecord` (below) hands back the columns. A column is a string unless it declares a +[datatype](#datatype), and a [`null`](#null) item draws it as null. Save `mydata/users.json`: ```json @@ -164,7 +164,8 @@ r, err = f.FakeRecordTemplate(`{"format":"{x}","x":["a","b"]}`) // compile + ren | `WithDataFS(fsys)` | layer an `fs.FS`, such as your own `embed.FS` | | `WithoutShippedData()` | load only what you give | -A `*Record` carries its columns via `Columns()`, and serializes them with `JSON()` +A `*Record` carries its columns via `Columns()` — each a `Column` of `Name`, +`DataType`, rendered `Value` and `Null` — and serializes them with `JSON()` (one object), `CSVHeader()`/`CSVLine()`, or `SQLInsert(table)` — the shapes the CLI's `--format` writes. `FakeRecord` and `FakeRecordTemplate` take a record; a path or template that is not one — a bare string, a choice, or a folder — errors. @@ -255,10 +256,53 @@ Renders e.g. `bar foo baz`. Rejected at load: a `separator` without a `repeat`, a `separator` of `""` (the default), and a `repeat` that multiplies to more than 1 048 576 renders along any path of nested repeats. +### Datatype + +A record column may declare `datatype` — `integer`, `number` or `boolean` — so `json` +writes `42` rather than `"42"` and `sql` a bare literal; a column without one is a +string: + +```json +{ "format": "", + "id": { "format": "{seq()}", "datatype": "integer" }, + "paid": { "format": "{p}", "p": ["true", "false"], "datatype": "boolean" }, + "total": { "format": "{calc(net * qty, 2)}", "net": ["19.99", "5.00"], "qty": ["3", "7"], "datatype": "number" } } +``` + +Writes e.g. `{"id":1,"paid":true,"total":59.97}`. A column is a field of the top-level +template, or an item of a choice standing in for one; `datatype` anywhere else is a +load error. So is a column that can render text its datatype rejects — `integer` takes +`-?(0|[1-9][0-9]*)`, `number` a JSON number, `boolean` `true` or `false` — and the +error shows such a render: + +```text +order.id: datatype integer, but it can render "000", which is not an integer +``` + +A `{calc()}` fills an `integer` or `number` column only where it provably prints no +`NaN` or `Inf`: each operand is a plain decimal — a sign, digits, one dot — of at most +300 bytes, or a field holding only such a calc, and no divisor can be zero. An +`integer` column also needs a decimals count of `0`, or integer operands and no `/`. + +### Null + +A `null` item draws a record column as null: `json` writes `null`, `sql` `NULL`, and +`csv` an empty field, with an empty string written `""` so PostgreSQL's `COPY … CSV` +reads both back. `Fake` renders a null as `""`. The other items' weights skew its +odds: + +```json +{ "format": "", "deleted_at": null, "middle": [null, { "format": "{n}", "n": ["Ann", "Eva"], "weight": 3 }] } +``` + +`deleted_at` is null every draw, `middle` a name three draws in four. Rejected at +load: `null` anywhere but a column, naming `""`, and a column whose items declare +different datatypes. + ### Options and fields -`format`, `weight`, `repeat` and `separator` are the only options; **any other -key is a field** (see [Decisions](#decisions)). An object that does nothing a +`format`, `weight`, `repeat`, `separator` and `datatype` are the only options; **any +other key is a field** (see [Decisions](#decisions)). An object that does nothing a string can't — only a `format` — is rejected naming the string, as is a one-item choice naming its item. @@ -437,8 +481,8 @@ tokens add cost in proportion to the output. ## Decisions -- **Options and fields share one namespace.** `format`, `weight`, `repeat` and - `separator` are reserved; every other key is a field. Nesting fields under a +- **Options and fields share one namespace.** `format`, `weight`, `repeat`, + `separator` and `datatype` are reserved; every other key is a field. Nesting fields under a key, or prefixing options, would tax every template to guard against a misspelt option. - **`{a|b}` stays beside nested choices.** `[[…], […]]` picks the same way, but @@ -518,7 +562,8 @@ tokens add cost in proportion to the output. so `a/(b*c)` with `b` fixed at `0` and `c` varying loads and prints `Inf` every draw — catching it needs zero-absorbing algebra for a shape nobody writes. - **In data, a default written out and a constant spelled as a sample are load - errors.** `weight: 1`, `repeat: 1`, `separator: ""`, `int(5,5)`, `float(1,1,2)`, + errors.** `weight: 1`, `repeat: 1`, `separator: ""`, `datatype: "string"`, + `int(5,5)`, `float(1,1,2)`, `+5` and `05` each spell what a shorter form already spells, so each is rejected naming that form. The CLI's numbers follow the shell instead: `--seed 007` and `--repeat +3` are 7 and 3, as every command line reads them. @@ -557,6 +602,9 @@ tokens add cost in proportion to the output. row — would vary per draw. A fixed column set is what the CSV and `INSERT` contracts rest on, so the restriction holds even where a particular choice would happen to agree. +- **Null is a `null` item, not a rate.** A null is one more outcome of a column's + draw, so a choice's weights skew it like any other; a null-rate option would be a + second way to state odds. - **The performance gate asserts allocations, not wall-clock time.** `AllocsPerRun` is deterministic across machines, so a ±10% ceiling does not flake under CI load, while time varies with the machine and its neighbours. A rendering slowdown @@ -612,7 +660,9 @@ hold.go the hold: one draw per expansion for paths and operands, and its reference.go reference sigils, and binding references across the tree graph.go the render graph: edges, cycles, the repeat bound, tree walks builtins.go the {name()} function registry and its implementations -calc.go the {calc()} arithmetic evaluator: parser, eval, validation +calc.go the {calc()} arithmetic evaluator: parser, eval, validation, and the proof a typed column's calc is finite +datatype.go column datatypes: DataType, where datatype and null may sit, and the load check every typed render passes +renderlang.go what text a node can render, as relations over a scalar's grammar data.go data loading: fs.FS folders/files -> namespace tree, multi-source merge cmd/fejkdata/ the fejkdata CLI data/ shipped data (JSON), embedded at build: locale folders + a misc folder diff --git a/builtins.go b/builtins.go index 8b9fec6..c958a54 100644 --- a/builtins.go +++ b/builtins.go @@ -5,6 +5,7 @@ import ( "errors" "fmt" "math" + "slices" "strconv" "strings" "unicode" @@ -24,36 +25,36 @@ const ( // samples read only the rng. A time-based id (uuid v7, ulid) draws its timestamp // from the rng, not the wall clock, so seeded output stays reproducible. var builtins = map[string]builtin{ - "luhn": {arity: 0, prep: derive(func(e string) string { return string(rune('0' + luhnCheck(e))) })}, - "mod11": {arity: 0, prep: derive(mod11Check)}, - "ean": {arity: 0, prep: derive(eanCheck)}, - "uuid": {arity: 0, prep: sample(uuidV7)}, - "ulid": {arity: 0, prep: sample(ulid)}, - "nanoid": {arity: 1, check: posIntArg, prep: chars(nanoidAlphabet)}, - "hex": {arity: 1, check: posIntArg, prep: chars(hexDigits)}, - "digits": {arity: 1, check: posIntArg, prep: chars("0123456789")}, - "upper": {arity: 1, check: posIntArg, prep: chars("ABCDEFGHIJKLMNOPQRSTUVWXYZ")}, - "lower": {arity: 1, check: posIntArg, prep: chars("abcdefghijklmnopqrstuvwxyz")}, + "luhn": {arity: 0, prep: derive(func(e string) string { return string(rune('0' + luhnCheck(e))) }), emits: always(textShape{{{decimalDigits, 1, 1}}})}, + "mod11": {arity: 0, prep: derive(mod11Check), emits: always(textShape{{{decimalDigits + "X", 1, 1}}})}, + "ean": {arity: 0, prep: derive(eanCheck), emits: always(textShape{{{decimalDigits, 1, 1}}})}, + "uuid": {arity: 0, prep: sample(uuidV7), emits: always(uuidShape)}, + "ulid": {arity: 0, prep: sample(ulid), emits: always(textShape{{{crockford[:8], 1, 1}, {crockford, 25, 25}}})}, + "nanoid": sampleOf(nanoidAlphabet), + "hex": sampleOf(hexDigits), + "digits": sampleOf(decimalDigits), + "upper": sampleOf("ABCDEFGHIJKLMNOPQRSTUVWXYZ"), + "lower": sampleOf("abcdefghijklmnopqrstuvwxyz"), "base64": {arity: 1, check: posIntArg, prep: func(a []string) callFn { n := atoi(a[0]) return func(s *session, _ string, _ []string) string { return base64.StdEncoding.EncodeToString(randBytes(s, n)) } - }}, + }, emits: base64Shape}, "int": {arity: 2, check: intRangeArgs, prep: func(a []string) callFn { lo, span := atoi(a[0]), atoi(a[1])-atoi(a[0])+1 return func(s *session, _ string, _ []string) string { return strconv.Itoa(lo + s.IntN(span)) } - }}, + }, emits: intShape}, "float": {arity: 3, check: floatArgs, prep: func(a []string) callFn { lo, hi, dp := atof(a[0]), atof(a[1]), atoi(a[2]) return func(s *session, _ string, _ []string) string { return strconv.FormatFloat(lo+s.Float64()*(hi-lo), 'f', dp, 64) } - }}, + }, emits: func(a []string) textShape { return printedFloat(atof(a[0]), atof(a[1]), atoi(a[2]), false) }}, "iban": {arity: 1, check: ibanArg, prep: func(a []string) callFn { cc := a[0] return func(s *session, _ string, _ []string) string { return iban(s, cc) } - }}, + }, emits: ibanShape}, "calc": {arity: -1, check: checkCalc, prep: calcPrep, operands: calcOperands}, "lowercase": {arity: 1, check: transformArg, prep: transformPrep(strings.ToLower), operands: transformOperand}, "uppercase": {arity: 1, check: transformArg, prep: transformPrep(strings.ToUpper), operands: transformOperand}, @@ -69,7 +70,7 @@ var builtins = map[string]builtin{ return func(s *session, _ string, _ []string) string { return strconv.FormatUint(s.next(key), 10) } - }}, + }, emits: always(textShape{{{nonZeroDigits, 1, 1}, {decimalDigits, 0, 19}}})}, } // derive and sample are the two argument-free builtin shapes: a derivation reads @@ -94,6 +95,82 @@ func chars(alphabet string) func([]string) callFn { } } +// sampleOf is the builtin that draws n characters from an alphabet. +func sampleOf(alphabet string) builtin { + return builtin{arity: 1, check: posIntArg, prep: chars(alphabet), emits: func(a []string) textShape { + n := atoi(a[0]) + return textShape{{{alphabet, n, n}}} + }} +} + +// always is the emits of a builtin whose args do not change what it can print. +func always(s textShape) func([]string) textShape { + return func([]string) textShape { return s } +} + +var uuidShape = textShape{{{hexDigits, 8, 8}, {"-", 1, 1}, {hexDigits, 4, 4}, {"-", 1, 1}, {"7", 1, 1}, {hexDigits, 3, 3}, {"-", 1, 1}, {"89ab", 1, 1}, {hexDigits, 3, 3}, {"-", 1, 1}, {hexDigits, 12, 12}}} + +const base64Alphabet = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/" + +func base64Shape(a []string) textShape { + n := atoi(a[0]) + pad := (3 - n%3) % 3 + size := 4*((n+2)/3) - pad + return textShape{{{base64Alphabet, size, size}, {"=", pad, pad}}} +} + +// intShape is what int prints: a sign only below zero, and no leading zero. +func intShape(a []string) textShape { + lo, hi := atoi(a[0]), atoi(a[1]) + var s textShape + if lo <= 0 && hi >= 0 { + s = append(s, []charRun{{"0", 1, 1}}) + } + if hi > 0 { + s = append(s, []charRun{{nonZeroDigits, 1, 1}, {decimalDigits, 0, len(a[1]) - 1}}) + } + if lo < 0 { + s = append(s, []charRun{{"-", 1, 1}, {nonZeroDigits, 1, 1}, {decimalDigits, 0, len(a[0]) - 2}}) + } + return s +} + +func ibanShape(a []string) textShape { + cc, digits := a[0], ibanLen[a[0]]-2 + return textShape{{{cc[:1], 1, 1}, {cc[1:], 1, 1}, {decimalDigits, digits, digits}}} +} + +// shortestFraction bounds the fraction FormatFloat's shortest form prints: at most 17 +// significant digits after up to 323 zeros. +const shortestFraction = 340 + +// printedFloat is what strconv.FormatFloat(v, 'f', dp, 64) prints for a v in [lo, hi] +// that is whole when integral. +func printedFloat(lo, hi float64, dp int, integral bool) textShape { + digits := len(strconv.FormatFloat(math.Floor(math.Max(math.Abs(lo), math.Abs(hi))), 'f', 0, 64)) + 1 // one more for a rounding carry + wholes := [][]charRun{{{"0", 1, 1}}, {{nonZeroDigits, 1, 1}, {decimalDigits, 0, digits - 1}}} + fractions := [][]charRun{nil} + switch { + case dp > 0: + fractions = [][]charRun{{{".", 1, 1}, {decimalDigits, dp, dp}}} + case dp < 0 && !integral: + fractions = append(fractions, []charRun{{".", 1, 1}, {decimalDigits, 1, shortestFraction}}) + } + signs := [][]charRun{nil} + if lo < 0 || math.Signbit(lo) { + signs = append(signs, []charRun{{"-", 1, 1}}) + } + var s textShape + for _, sign := range signs { + for _, whole := range wholes { + for _, fraction := range fractions { + s = append(s, slices.Concat(sign, whole, fraction)) + } + } + } + return s +} + const hexDigits = "0123456789abcdef" // transforms are the builtins that rewrite one operand's value; they nest, so @@ -106,20 +183,19 @@ var transforms = map[string]func(string) string{ // unwrapTransform peels nested transform calls off an operand arg, returning the // field it finally names and the transforms to apply, innermost last. -func unwrapTransform(arg string) (leaf string, chain []func(string) string, err error) { +func unwrapTransform(arg string) (leaf string, chain []string, err error) { for { name, args, isCall := funcCall(arg) if !isCall { return arg, chain, nil } - fn, isTransform := transforms[name] - if !isTransform { + if _, isTransform := transforms[name]; !isTransform { return "", nil, fmt.Errorf("%s(%s) is not a transform, so it cannot be an operand", name, strings.Join(args, ",")) } if len(args) != 1 { return "", nil, fmt.Errorf("%s takes 1 arg, got %d", name, len(args)) } - chain = append(chain, fn) + chain = append(chain, name) arg = args[0] } } @@ -150,10 +226,14 @@ func transformPrep(outer func(string) string) func([]string) callFn { if err != nil { panic(fmt.Sprintf("fejkdata: transform arg %q reached prep unvalidated: %v", a[0], err)) } + fns := make([]func(string) string, len(chain)) + for i, name := range chain { + fns[i] = transforms[name] + } return func(_ *session, _ string, operands []string) string { v := operands[0] - for i := len(chain) - 1; i >= 0; i-- { - v = chain[i](v) + for i := len(fns) - 1; i >= 0; i-- { + v = fns[i](v) } return outer(v) } diff --git a/calc.go b/calc.go index cd776b7..7d7f2cd 100644 --- a/calc.go +++ b/calc.go @@ -6,6 +6,7 @@ import ( "strconv" "strings" "unicode" + "unicode/utf8" ) // calcNode is a parsed expression node. It evaluates over the operand values expand @@ -154,10 +155,12 @@ func calcText(n calcNode) string { return "?" } -// neverNumeric reports a node no render of which is a number: fixed text that does -// not parse, or a choice of only such items. text is one such render. +// neverNumeric reports a node no render of which is a number: a null, fixed text that +// does not parse, or a choice of only such items. text is one such render. func neverNumeric(n node) (text string, never bool) { switch n := n.(type) { + case *null: + return "", true case *template: if !n.fixed || n.repeat > 1 { return "", false @@ -191,10 +194,7 @@ func calcPrep(args []string) callFn { at[name] = i } placed := indexVars(expr, at) - dp := -1 - if len(args) == 2 { - dp = atoi(args[1]) - } + dp := calcDecimals(args) return func(_ *session, _ string, operands []string) string { return strconv.FormatFloat(placed.eval(operands), 'f', dp, 64) } @@ -384,3 +384,256 @@ func contains(bs []byte, b byte) bool { } return false } + +// calcDecimals is a calc's decimals count, or -1 for the shortest form. +func calcDecimals(args []string) int { + if len(args) == 2 { + return atoi(args[1]) + } + return -1 +} + +// calcLimit is the largest magnitude a proof accepts as finite, far enough below +// math.MaxFloat64 that rounding in the bounds cannot hide an overflow. +const calcLimit = 1e300 + +// maxOperandLen is the longest operand text a proof bounds by its length, so that +// bound, 10^maxOperandLen, stays within calcLimit. +const maxOperandLen = 300 + +// calcBound is what a proof knows of every value a calc can take: it lies in [lo, hi], +// is at least nonZero from zero unless nonZero is 0, and is whole when integral. +type calcBound struct { + lo, hi, nonZero float64 + integral bool +} + +func magnitude(b calcBound) float64 { return math.Max(math.Abs(b.lo), math.Abs(b.hi)) } + +// doubt is why a proof could not show a calc finite, and the render that shows it. +type doubt struct{ render, why string } + +type bounded struct { + b calcBound + d *doubt +} + +// calcProof bounds a typed column's calcs from their operands' renders, to show each +// prints a number rather than NaN or Inf. +type calcProof struct { + decimal *textLanguage + operands map[node]bounded + lengths map[node]int +} + +func newCalcProof() *calcProof { + p := &calcProof{operands: map[node]bounded{}, lengths: map[node]int{}} + p.decimal = newTextLanguage(decimalGrammar, p) + return p +} + +// call bounds one calc token of t. +func (p *calcProof) call(t *template, args []string) (calcBound, *doubt) { + expr, err := parseCalc(args[0]) + if err != nil { + panic(fmt.Sprintf("fejkdata: calc(%q) reached a proof unparsed: %v", args[0], err)) + } + b, d := p.expr(expr, t.fields) + if d != nil { + return b, &doubt{d.render, fmt.Sprintf("{calc(%s)}: %s", strings.Join(args, ", "), d.why)} + } + return b, nil +} + +func (p *calcProof) expr(n calcNode, fields map[string]node) (calcBound, *doubt) { + switch n := n.(type) { + case calcNum: + v := float64(n) + return calcBound{v, v, v, v == math.Trunc(v)}, nil + case calcVar: + return p.operand(string(n), fields[string(n)]) + case calcNeg: + b, d := p.expr(n.x, fields) + return calcBound{-b.hi, -b.lo, b.nonZero, b.integral}, d + case calcBin: + l, d := p.expr(n.l, fields) + if d != nil { + return l, d + } + r, d := p.expr(n.r, fields) + if d != nil { + return r, d + } + return combine(n, l, r) + } + panic(fmt.Sprintf("fejkdata: calc node %T has no bound", n)) +} + +// combine bounds one operation from the bounds of its sides. +func combine(n calcBin, l, r calcBound) (calcBound, *doubt) { + b := calcBound{integral: l.integral && r.integral} + switch n.op { + case '+': + b.lo, b.hi = l.lo+r.lo, l.hi+r.hi + case '-': + b.lo, b.hi = l.lo-r.hi, l.hi-r.lo + case '*': + b.lo = min(l.lo*r.lo, l.lo*r.hi, l.hi*r.lo, l.hi*r.hi) + b.hi = max(l.lo*r.lo, l.lo*r.hi, l.hi*r.lo, l.hi*r.hi) + b.nonZero = l.nonZero * r.nonZero + default: + if r.nonZero == 0 { + return b, &doubt{"+Inf", fmt.Sprintf("divides by %s, which can be zero", calcText(n.r))} + } + m := magnitude(l) / r.nonZero + b = calcBound{lo: -m, hi: m, nonZero: l.nonZero / magnitude(r)} + } + if b.lo > 0 || b.hi < 0 { + b.nonZero = math.Max(b.nonZero, math.Min(math.Abs(b.lo), math.Abs(b.hi))) + } + if !(magnitude(b) <= calcLimit) { + return b, &doubt{"+Inf", calcText(n) + " can overflow"} + } + return b, nil +} + +// operand bounds a calc operand, once per node. +func (p *calcProof) operand(name string, n node) (calcBound, *doubt) { + if seen, done := p.operands[n]; done { + return seen.b, seen.d + } + b, d := p.measure(name, n) + p.operands[n] = bounded{b, d} + return b, d +} + +// measure bounds an operand through the calc it renders when that is all it renders, +// and otherwise from its text: a plain decimal of at most maxOperandLen bytes. +func (p *calcProof) measure(name string, n node) (calcBound, *doubt) { + if t, ok := n.(*template); ok { + if args, isCalc := soleCalc(t); isCalc { + b, d := p.call(t, args) + return rounded(b, calcDecimals(args)), d + } + } + text := p.decimal.node(n, nil) + if w, escapes := text.escape(decimalAccept); escapes { + why := fmt.Sprintf("operand %q can render %s, which is not a plain decimal", name, w) + if w.why != "" { + why += ": " + w.why + } + return calcBound{}, &doubt{"NaN", why} + } + size := p.length(n) + if size > maxOperandLen { + return calcBound{}, &doubt{"NaN", fmt.Sprintf("operand %q can render more than %d bytes, too many to bound", name, maxOperandLen)} + } + ends, m := text.to[1], math.Pow(10, float64(size)) + b := calcBound{hi: m, nonZero: 1 / m, integral: ends&decimalFractional == 0} + if ends&decimalNegative != 0 { + b.lo = -m + } + if ends&decimalZero != 0 { + b.nonZero = 0 + } + return b, nil +} + +// soleCalc reports a template that renders one calc and nothing else, with its args. +func soleCalc(t *template) ([]string, bool) { + if t.repeat != 1 || len(t.ops) != 1 || t.ops[0].kind != 'b' { + return nil, false + } + name, args, _ := funcCall(t.format[1 : len(t.format)-1]) + return args, name == "calc" +} + +// rounded is b once printed to dp decimals, which moves a value by up to half a unit. +func rounded(b calcBound, dp int) calcBound { + if dp < 0 { + return b + } + half := math.Pow(10, -float64(dp)) / 2 + return calcBound{b.lo - half, b.hi + half, math.Max(0, b.nonZero-half), b.integral || dp == 0} +} + +// length is the most bytes a render of n can take, anything past maxOperandLen +// reported as maxOperandLen+1. +func (p *calcProof) length(n node) int { + if size, done := p.lengths[n]; done { + return size + } + size := 0 + switch n := n.(type) { + case *choice: + for _, it := range n.items { + size = max(size, p.length(it)) + } + case *template: + size = p.formatLength(n)*n.repeat + len(n.separator)*(n.repeat-1) + } + size = min(size, maxOperandLen+1) + p.lengths[n] = size + return size +} + +func (p *calcProof) formatLength(t *template) int { + size := 0 + _ = eachToken(t.format, func(tok ftoken) error { + if tok.kind == 'l' { + size += utf8.RuneLen(tok.r) + } else { + size += p.tokenLength(t, tok.body) + } + size = min(size, maxOperandLen+1) + return nil + }) + return size +} + +// tokenLength is the most bytes one token can print. A transform never lengthens a +// render that reads as a decimal: it maps each non-ASCII rune, two bytes or more, to at +// most two ASCII letters. +func (p *calcProof) tokenLength(t *template, body string) int { + name, args, isFunc := funcCall(body) + var arms []arm + switch _, isTransform := transforms[name]; { + case !isFunc: + arms = splitArms(body, t.refs) + case isTransform: + leaf, _, _ := unwrapTransform(args[0]) + arms = []arm{splitArm(leaf, t.refs)} + case name == "calc": + b, d := p.call(t, args) + if d != nil { + return len(d.render) + } + return shapeLength(printedFloat(b.lo, b.hi, calcDecimals(args), b.integral)) + default: + return shapeLength(builtins[name].emits(args)) + } + size := 0 + for _, a := range arms { + for _, leaf := range pathLeaves(t.fields[a.key], a.tail) { + size = max(size, p.length(leaf)) + } + } + return size +} + +// shapeLength is the most bytes a shape can emit, anything past maxOperandLen reported +// as maxOperandLen+1. +func shapeLength(s textShape) int { + longest := 0 + for _, alt := range s { + size := 0 + for _, run := range alt { + if run.max < 0 { + return maxOperandLen + 1 + } + size += run.max + } + longest = max(longest, size) + } + return min(longest, maxOperandLen+1) +} diff --git a/datatype.go b/datatype.go new file mode 100644 index 0000000..b92a398 --- /dev/null +++ b/datatype.go @@ -0,0 +1,140 @@ +package fejkdata + +import ( + "errors" + "fmt" +) + +// DataType is what a record column holds, which decides how a record writes its value. +type DataType int + +// The datatypes a column declares with "datatype"; a column without one is a string. +const ( + DataTypeString DataType = iota + DataTypeInteger + DataTypeNumber + DataTypeBoolean +) + +var dataTypeNames = [...]string{"string", "integer", "number", "boolean"} + +// String is the datatype as data spells it. +func (d DataType) String() string { + if d < 0 || int(d) >= len(dataTypeNames) { + return fmt.Sprintf("DataType(%d)", int(d)) + } + return dataTypeNames[d] +} + +// position is where a JSON value sits, which decides whether it may carry a datatype or +// be null. +type position int + +const ( + inFormat position = iota // rendered by a format, so neither + atTop // a category or an inline template, whose fields are the columns + inColumn // a column, or a choice item standing in for one +) + +// datatypeOf reads a template's "datatype" (default DataTypeString). +func datatypeOf(m map[string]any, pos position) (DataType, error) { + v, ok := m["datatype"] + if !ok { + return DataTypeString, nil + } + name, ok := v.(string) + if !ok { + return 0, fmt.Errorf("datatype must be a string, got %T", v) + } + if name == DataTypeString.String() { + return 0, fmt.Errorf("datatype %q is the default, so it has no effect; drop it", name) + } + for d := DataTypeInteger; d <= DataTypeBoolean; d++ { + if name != d.String() { + continue + } + if pos != inColumn { + return 0, errors.New("datatype only types a record column — a field of the top-level template — so it has no effect here") + } + return d, nil + } + return 0, fmt.Errorf(`datatype takes "integer", "number" or "boolean", got %q`, name) +} + +// columnDatatype is the datatype a column's items declare. They must agree, since a +// column holds one; a column only ever null is a string. +func columnDatatype(n node) (DataType, error) { + var declared []DataType + var collect func(node) + collect = func(n node) { + switch n := n.(type) { + case *choice: + for _, it := range n.items { + collect(it) + } + case *template: + declared = append(declared, n.datatype) + } + } + collect(n) + if len(declared) == 0 { + return DataTypeString, nil + } + for _, d := range declared { + if d != declared[0] { + return declared[0], fmt.Errorf("its items declare %s and %s; a column holds one datatype, so give every item the same", declared[0], d) + } + } + return declared[0], nil +} + +// datatypeSpec is what a datatype's text must satisfy: a grammar, the states a render +// may end in, and how an error names the datatype. +type datatypeSpec struct { + grammar *grammar + accept uint32 + noun string +} + +var datatypeSpecs = map[DataType]datatypeSpec{ + DataTypeInteger: {numberGrammar, integerAccept, "an integer"}, + DataTypeNumber: {numberGrammar, numberAccept, "a number"}, + DataTypeBoolean: {booleanGrammar, booleanAccept, "a boolean"}, +} + +// datatypeCheck proves every render of a typed column is text its datatype takes. One +// check covers a scope, so a node several columns reach is read once per grammar. +type datatypeCheck struct { + languages map[*grammar]*textLanguage + proof *calcProof +} + +func (c *datatypeCheck) check(path string, n node) error { + t, ok := n.(*template) + if !ok || t.datatype == DataTypeString { + return nil + } + spec := datatypeSpecs[t.datatype] + w, escapes := c.language(spec.grammar).node(t, nil).escape(spec.accept) + if !escapes { + return nil + } + msg := fmt.Sprintf("%s: datatype %s, but it can render %s, which is not %s", path, t.datatype, w, spec.noun) + if w.why != "" { + msg += ": " + w.why + } + return errors.New(msg) +} + +func (c *datatypeCheck) language(g *grammar) *textLanguage { + if c.proof == nil { + c.proof = newCalcProof() + c.languages = map[*grammar]*textLanguage{} + } + l, made := c.languages[g] + if !made { + l = newTextLanguage(g, c.proof) + c.languages[g] = l + } + return l +} diff --git a/fejkdata.go b/fejkdata.go index c558d9b..35cd73e 100644 --- a/fejkdata.go +++ b/fejkdata.go @@ -162,6 +162,8 @@ func paths(n node) []string { } } return out + case *null: + return []string{""} case *choice: out := []string{""} for p := range n.shared { diff --git a/graph.go b/graph.go index cf5cd5b..f65627e 100644 --- a/graph.go +++ b/graph.go @@ -186,7 +186,10 @@ func checkScope(s nodeScope) error { if err := s(func(path string, n node) error { return repeatCheck(path, n, mem) }); err != nil { return err } - return s(heldCheck) + if err := s(heldCheck); err != nil { + return err + } + return s((&datatypeCheck{}).check) } type reachMemo map[node]int diff --git a/node.go b/node.go index 384a2c6..d5af8e3 100644 --- a/node.go +++ b/node.go @@ -32,6 +32,12 @@ type choice struct { func (*choice) isNode() {} +// null is a record column's missing value, rendered as "". It is not zero-sized, so two +// nulls are two map keys. +type null struct{ _ byte } + +func (*null) isNode() {} + // template renders a format string, substituting {tokens} from fields. A bare // JSON string is a template with no fields. repeat (default 1) renders that format // that many times and joins the results with separator (default ""), each render @@ -41,6 +47,7 @@ type template struct { fields map[string]node repeat int separator string + datatype DataType ops []op // format compiled once (see compileOps); what expand walks grow int // minimum output size, to size the render buffer fixed bool // no op varies, so every render is lit @@ -66,26 +73,37 @@ func (t *template) field(seg string) (node, bool) { return n, ok } -// compile converts parsed JSON into a node tree, validating structure up front. -// Only a choice's items carry a weight, so one here would be inert whatever its type. +// compile converts parsed JSON — a category or an inline template — into a node tree, +// validating structure up front. func compile(v any) (node, error) { + return compileAt(v, atTop) +} + +// compileAt compiles a node that is no choice's item. Only a choice's items carry a +// weight, so one here would be inert whatever its type. +func compileAt(v any, pos position) (node, error) { if m, ok := v.(map[string]any); ok { if _, weighted := m["weight"]; weighted { return nil, fmt.Errorf("weight only skews a choice's items, so it has no effect here; it is an option and can never be a field") } } - return compileItem(v) + return compileItem(v, pos) } // compileItem compiles one node, allowing the weight a choice item may carry. -func compileItem(v any) (node, error) { +func compileItem(v any, pos position) (node, error) { switch v := v.(type) { case string: return compileString(v) case []any: - return compileChoice(v) + return compileChoice(v, pos) case map[string]any: - return compileTemplate(v) + return compileTemplate(v, pos) + case nil: + if pos != inColumn { + return nil, fmt.Errorf(`null is a record column's value; here it only renders "", so write ""`) + } + return &null{}, nil default: return nil, fmt.Errorf("a template value must be a string, a list or an object, not %s", jsonKind(v)) } @@ -99,8 +117,6 @@ func jsonKind(v any) string { return "a number" case bool: return "a boolean" - case nil: - return "null" } return fmt.Sprintf("%T", v) } @@ -136,7 +152,11 @@ func (t *template) compileFormat() error { return checkNoRepeatedRead(t.format, c, t.refs) } -func compileChoice(items []any) (node, error) { +func compileChoice(items []any, pos position) (node, error) { + itemPos := inFormat + if pos == inColumn { + itemPos = inColumn + } if len(items) == 0 { return nil, fmt.Errorf("empty choice") } @@ -160,7 +180,7 @@ func compileChoice(items []any) (node, error) { } total += w cum[i] = total - n, err := compileItem(raw) + n, err := compileItem(raw, itemPos) if err != nil { return nil, err } @@ -203,22 +223,26 @@ func checkNoRepeatedItem(items []any) error { return nil } -func compileTemplate(m map[string]any) (node, error) { - o, err := readOptions(m) +func compileTemplate(m map[string]any, pos position) (node, error) { + o, err := readOptions(m, pos) if err != nil { return nil, err } - fields, err := compileFields(m) + fieldPos := inFormat + if pos == atTop && o.repeat == 1 { + fieldPos = inColumn + } + fields, err := compileFields(m, fieldPos) if err != nil { return nil, err } - if len(fields) == 0 && o.repeat == 1 && !o.weighted { + if len(fields) == 0 && o.repeat == 1 && !o.weighted && o.datatype == DataTypeString { return nil, fmt.Errorf("an object holding only a format is a string; write %q", o.format) } if err := checkTokens(o.format, fields); err != nil { return nil, err } - t := &template{format: o.format, fields: fields, repeat: o.repeat, separator: o.separator} + t := &template{format: o.format, fields: fields, repeat: o.repeat, separator: o.separator, datatype: o.datatype} if err := t.compileFormat(); err != nil { return nil, err } @@ -227,13 +251,14 @@ func compileTemplate(m map[string]any) (node, error) { // templateOptions is what a template object's option keys say. type templateOptions struct { + datatype DataType format string repeat int separator string weighted bool } -func readOptions(m map[string]any) (templateOptions, error) { +func readOptions(m map[string]any, pos position) (templateOptions, error) { var o templateOptions format, ok := m["format"].(string) if !ok { @@ -245,6 +270,9 @@ func readOptions(m map[string]any) (templateOptions, error) { return o, err } o.repeat = repeat + if o.datatype, err = datatypeOf(m, pos); err != nil { + return o, err + } if sv, ok := m["separator"]; ok { if o.separator, ok = sv.(string); !ok { return o, fmt.Errorf("separator must be a string, got %T", sv) @@ -262,7 +290,7 @@ func readOptions(m map[string]any) (templateOptions, error) { // compileFields compiles every non-option key of a template object, in name order // so which of several bad fields is reported does not vary. -func compileFields(m map[string]any) (map[string]node, error) { +func compileFields(m map[string]any, pos position) (map[string]node, error) { fields := make(map[string]node, len(m)) keys := make([]string, 0, len(m)) for k := range m { @@ -276,7 +304,10 @@ func compileFields(m map[string]any) (map[string]node, error) { if err := checkName(k); err != nil { return nil, fmt.Errorf("field %w", err) } - n, err := compile(m[k]) + n, err := compileAt(m[k], pos) + if err == nil && pos == inColumn { + _, err = columnDatatype(n) + } if err != nil { return nil, fmt.Errorf("field %q: %w", k, err) } @@ -360,10 +391,10 @@ func checkName(name string) error { } // isOption reports whether a template key configures the node instead of naming a -// field. These four names can never be fields. +// field. These names can never be fields. func isOption(name string) bool { switch name { - case "format", "repeat", "separator", "weight": + case "datatype", "format", "repeat", "separator", "weight": return true } return false diff --git a/path.go b/path.go index ee9b1c5..687d9b9 100644 --- a/path.go +++ b/path.go @@ -60,7 +60,7 @@ func walkPath(n node, tail []string, w pathWalk) error { } return nil } - return fmt.Errorf("cannot descend into %T at %q", n, tail[0]) + return fmt.Errorf("no field %q", tail[0]) } // carriedByAll is the choice rule a path that must resolve on every call obeys: diff --git a/record.go b/record.go index 803e041..fc53ee0 100644 --- a/record.go +++ b/record.go @@ -9,10 +9,14 @@ import ( "strings" ) -// Column is one rendered column of a record. +// Column is one rendered column of a record. Value is the rendered text, which a +// serializer quotes for DataTypeString and writes bare for any other datatype; a Null +// column has no Value. type Column struct { - Name string - Value string + Name string + DataType DataType + Value string + Null bool } // Record is one record rendered from a template: every direct field is a column, @@ -28,70 +32,93 @@ func (r *Record) Columns() []Column { return append([]Column(nil), r.columns...) } -// JSON renders the record as one JSON object, every column a string. +// JSON renders the record as one JSON object. func (r *Record) JSON() string { - m := make(map[string]string, len(r.columns)) - for _, c := range r.columns { - m[c.Name] = c.Value + var b strings.Builder + b.WriteByte('{') + for i, c := range r.columns { + if i > 0 { + b.WriteByte(',') + } + b.WriteString(jsonString(c.Name)) + b.WriteByte(':') + b.WriteString(literal(c, jsonString, "null")) } - b, _ := json.Marshal(m) + b.WriteByte('}') + return b.String() +} + +func jsonString(s string) string { + b, _ := json.Marshal(s) return string(b) } // CSVHeader renders the column names as one CSV header line. func (r *Record) CSVHeader() string { - return csvLine(r.names()) + fields := make([]string, len(r.columns)) + for i, c := range r.columns { + fields[i] = csvField(c.Name) + } + return strings.Join(fields, ",") } -// CSVLine renders the column values as one CSV row. +// CSVLine renders the column values as one CSV row: a null column an empty field and an +// empty string "", the convention PostgreSQL's COPY reads a null by. func (r *Record) CSVLine() string { - return csvLine(r.values()) -} - -func (r *Record) names() []string { - out := make([]string, len(r.columns)) + fields := make([]string, len(r.columns)) for i, c := range r.columns { - out[i] = c.Name + fields[i] = literal(c, csvField, "") } - return out + if line := strings.Join(fields, ","); line != "" { + return line + } + return `""` // a blank line is a row every CSV reader drops } -func (r *Record) values() []string { - out := make([]string, len(r.columns)) - for i, c := range r.columns { - out[i] = c.Value +func csvField(s string) string { + if s == "" { + return `""` } - return out -} - -func csvLine(cols []string) string { var b strings.Builder w := csv.NewWriter(&b) - _ = w.Write(cols) + _ = w.Write([]string{s}) w.Flush() - line := strings.TrimSuffix(b.String(), "\n") - if line == "" { - return `""` // a blank line is a row every CSV reader drops - } - return line + return strings.TrimSuffix(b.String(), "\n") } -// SQLInsert renders the record as one INSERT statement into table: identifiers in -// ANSI double quotes, every value a single-quoted string literal. +// SQLInsert renders the record as one INSERT statement into table, identifiers in ANSI +// double quotes. func (r *Record) SQLInsert(table string) string { cols := make([]string, len(r.columns)) vals := make([]string, len(r.columns)) for i, c := range r.columns { cols[i] = quoteIdent(c.Name) - vals[i] = "'" + strings.ReplaceAll(c.Value, "'", "''") + "'" + vals[i] = literal(c, sqlString, "NULL") } return fmt.Sprintf("INSERT INTO %s (%s) VALUES (%s);", quoteIdent(table), strings.Join(cols, ", "), strings.Join(vals, ", ")) } +func sqlString(s string) string { + return "'" + strings.ReplaceAll(s, "'", "''") + "'" +} + func quoteIdent(s string) string { return `"` + strings.ReplaceAll(s, `"`, `""`) + `"` } +// literal spells a column the way a serializer writes it: quoted for a string, bare for +// any other datatype, whose every render the load check proved a literal, and nullText +// for a null. +func literal(c Column, quote func(string) string, nullText string) string { + switch { + case c.Null: + return nullText + case c.DataType == DataTypeString: + return quote(c.Value) + } + return c.Value +} + // FakeRecord renders a path as one record: the template it names, with each direct // field drawn as a column. Only a category-level template is a record — a path // that descends into a field, or that names a folder or a choice, is an error. @@ -116,7 +143,7 @@ func (f *Generator) FakeRecord(path string) (*Record, error) { // columns, or why it is not a record. type recordShape struct { t *template - columns []string + columns []Column err error } @@ -140,7 +167,7 @@ func (f *Generator) recordShapeOf(n node) recordShape { type RecordTemplate struct { g *Generator t *template - columns []string + columns []Column } // Fake renders the record with one draw. @@ -175,7 +202,7 @@ func (f *Generator) FakeRecordTemplate(input string) (*Record, error) { // recordOf is the fence both record entry points pass. The columns come back with // the template, fixed for every draw the caller goes on to make. -func recordOf(n node) (*template, []string, error) { +func recordOf(n node) (*template, []Column, error) { t, ok := n.(*template) if !ok { return nil, nil, errors.New("names a choice, not a template; a record is a template whose fields are its columns") @@ -183,13 +210,18 @@ func recordOf(n node) (*template, []string, error) { if t.repeat != 1 { return nil, nil, fmt.Errorf("carries repeat %d, which composes its format into one string; a record projects columns instead — drop the repeat and render the record again for more rows", t.repeat) } - columns := recordColumns(t) - if len(columns) == 0 { + names := recordColumns(t) + if len(names) == 0 { return nil, nil, errors.New("has no fields, so no columns") } - if err := checkColumnRefs(t, columns); err != nil { + if err := checkColumnRefs(t, names); err != nil { return nil, nil, err } + columns := make([]Column, len(names)) + for i, name := range names { + datatype, _ := columnDatatype(t.fields[name]) // compile refused a column whose items disagree + columns[i] = Column{Name: name, DataType: datatype} + } return t, columns, nil } @@ -262,11 +294,16 @@ func columnRefs(t *template, columns []string) ([]columnRef, error) { // renderRecord draws each column once, in the name order recordOf fixed, over one // reference scope shared across them. -func renderRecord(s *session, t *template, columns []string) *Record { +func renderRecord(s *session, t *template, columns []Column) *Record { scope := &draws{variant: map[string]node{}, value: map[string]string{}} - r := &Record{columns: make([]Column, len(columns))} - for i, name := range columns { - r.columns[i] = Column{Name: name, Value: render(s, t.fields[name], scope)} + r := &Record{columns: append([]Column(nil), columns...)} + for i := range r.columns { + n := drawn(s, t.fields[r.columns[i].Name]) + if _, isNull := n.(*null); isNull { + r.columns[i].Null = true + } else { + r.columns[i].Value = render(s, n, scope) + } } return r } diff --git a/render.go b/render.go index 7a7af3a..330ed0f 100644 --- a/render.go +++ b/render.go @@ -56,6 +56,8 @@ func render(s *session, n node, refScope *draws) string { switch n := n.(type) { case *choice: return render(s, pick(s, n), refScope) + case *null: + return "" case *template: if n.repeat == 1 { if n.fixed { diff --git a/renderlang.go b/renderlang.go new file mode 100644 index 0000000..4d56a2d --- /dev/null +++ b/renderlang.go @@ -0,0 +1,403 @@ +package fejkdata + +import ( + "slices" + "strconv" + "strings" + "unicode/utf8" +) + +// grammar is a deterministic automaton over a scalar's text: state 0 is dead, 1 the +// start, and each state lists the runes that leave it and where they lead. +type grammar [][]arc + +type arc struct { + on string + to int +} + +func (g *grammar) run(q int, s string) int { + for _, r := range s { + if q = g.step(q, r); q == 0 { + return 0 + } + } + return q +} + +func (g *grammar) step(q int, r rune) int { + for _, a := range (*g)[q] { + if strings.ContainsRune(a.on, r) { + return a.to + } + } + return 0 +} + +const ( + decimalDigits = "0123456789" + nonZeroDigits = "123456789" +) + +// numberGrammar reads a JSON number. States: 2 "-", 3 "0", 4 more integer digits, 5 ".", +// 6 fraction digits, 7 "e", 8 its sign, 9 exponent digits. +var numberGrammar = &grammar{ + nil, + {{"-", 2}, {"0", 3}, {nonZeroDigits, 4}}, + {{"0", 3}, {nonZeroDigits, 4}}, + {{".", 5}, {"eE", 7}}, + {{decimalDigits, 4}, {".", 5}, {"eE", 7}}, + {{decimalDigits, 6}}, + {{decimalDigits, 6}, {"eE", 7}}, + {{"+-", 8}, {decimalDigits, 9}}, + {{decimalDigits, 9}}, + {{decimalDigits, 9}}, +} + +const ( + integerAccept uint32 = 1<<3 | 1<<4 + numberAccept = integerAccept | 1<<6 | 1<<9 +) + +var booleanGrammar = &grammar{ + nil, + {{"t", 2}, {"f", 6}}, + {{"r", 3}}, {{"u", 4}}, {{"e", 5}}, nil, + {{"a", 7}}, {{"l", 8}}, {{"s", 9}}, {{"e", 10}}, nil, +} + +const booleanAccept uint32 = 1<<5 | 1<<10 + +// decimalGrammar reads what a calc operand must render to be proven finite: a sign, +// digits and at most one dot. Past the sign, states 4–9 are positive and 10–15 their +// negatives: 4 zero digits, 5 a nonzero integer, 6 a leading dot, 7 zero with a dot, +// 8 a nonzero integer with a zero fraction, 9 a nonzero fraction. +var decimalGrammar = &grammar{ + nil, + {{"+", 2}, {"-", 3}, {"0", 4}, {nonZeroDigits, 5}, {".", 6}}, + {{"0", 4}, {nonZeroDigits, 5}, {".", 6}}, + {{"0", 10}, {nonZeroDigits, 11}, {".", 12}}, + {{"0", 4}, {nonZeroDigits, 5}, {".", 7}}, + {{decimalDigits, 5}, {".", 8}}, + {{"0", 7}, {nonZeroDigits, 9}}, + {{"0", 7}, {nonZeroDigits, 9}}, + {{"0", 8}, {nonZeroDigits, 9}}, + {{decimalDigits, 9}}, + {{"0", 10}, {nonZeroDigits, 11}, {".", 13}}, + {{decimalDigits, 11}, {".", 14}}, + {{"0", 13}, {nonZeroDigits, 15}}, + {{"0", 13}, {nonZeroDigits, 15}}, + {{"0", 14}, {nonZeroDigits, 15}}, + {{decimalDigits, 15}}, +} + +const ( + decimalAccept uint32 = 1<<4 | 1<<5 | 1<<7 | 1<<8 | 1<<9 | 1<<10 | 1<<11 | 1<<13 | 1<<14 | 1<<15 + decimalNegative uint32 = 0xfc00 + decimalZero uint32 = 1<<4 | 1<<7 | 1<<10 | 1<<13 + decimalFractional uint32 = 1<<9 | 1<<15 +) + +// relation is what a node's renders do to a grammar: from each state, the states a +// render can end in, and one render reaching each. +type relation struct { + g *grammar + to []uint32 + w []witness // w[from*len(to)+to] +} + +// witness is one render, cut past witnessCap bytes, and why it can occur when the text +// alone does not say. +type witness struct { + text string + cut bool + why string +} + +const witnessCap = 60 + +func (w witness) then(next witness) witness { + if w.why == "" { + w.why = next.why + } + if w.cut { + return w + } + w.text += next.text + w.cut = next.cut + if len(w.text) > witnessCap { + end := witnessCap + for !utf8.RuneStart(w.text[end]) { + end-- + } + w.text, w.cut = w.text[:end], true + } + return w +} + +func (w witness) String() string { + if w.cut { + return strconv.Quote(w.text + "…") + } + return strconv.Quote(w.text) +} + +func newRelation(g *grammar) *relation { + n := len(*g) + return &relation{g: g, to: make([]uint32, n), w: make([]witness, n*n)} +} + +func (r *relation) add(from, to int, w witness) { + if r.to[from]&(1<>= 1; k == 0 { + return out + } + } +} + +// closure is any number of renders of r in a row, where r includes the empty render. +func (r *relation) closure() *relation { + for { + next := r.then(r) + if slices.Equal(next.to, r.to) { + return r + } + r = next + } +} + +// escape finds a render from the start that ends outside accept, preferring one that +// carries a reason. +func (r *relation) escape(accept uint32) (witness, bool) { + var found witness + escapes := false + for to := range r.to { + if (r.to[1]&^accept)&(1< 1 { + r = r.then(l.text(f.apply(n.separator)).then(r).power(n.repeat - 1)) + } + } + l.memo[key] = r + return r +} + +func (l *textLanguage) text(s string) *relation { return textRelation(l.g, s, "") } + +// format reads a template's format the way expand renders it: literal runs and tokens +// in turn. +func (l *textLanguage) format(t *template, f fold) *relation { + r := l.empty + var lit strings.Builder + _ = eachToken(t.format, func(tok ftoken) error { + if tok.kind == 'l' { + lit.WriteRune(tok.r) + return nil + } + r = r.then(l.text(f.apply(lit.String()))).then(l.token(t, tok.body, f)) + lit.Reset() + return nil + }) + return r.then(l.text(f.apply(lit.String()))) +} + +// token reads one {…} token: a field read, a transform over one, a calc, or what a +// builtin emits. +func (l *textLanguage) token(t *template, body string, f fold) *relation { + name, args, isFunc := funcCall(body) + if !isFunc { + var r *relation + for _, a := range splitArms(body, t.refs) { + r = union(r, l.read(t, a, f)) + } + return r + } + if _, isTransform := transforms[name]; isTransform { + leaf, chain, _ := unwrapTransform(args[0]) + inner := slices.Clone(chain) + slices.Reverse(inner) + return l.read(t, splitArm(leaf, t.refs), append(append(inner, name), f...)) + } + if name == "calc" { + return l.calc(t, args, f) + } + return l.shape(builtins[name].emits(args), f) +} + +// read is one arm of a token: every node its path can land on. +func (l *textLanguage) read(t *template, a arm, f fold) *relation { + var r *relation + for _, leaf := range pathLeaves(t.fields[a.key], a.tail) { + r = union(r, l.node(leaf, f)) + } + return r +} + +func (l *textLanguage) calc(t *template, args []string, f fold) *relation { + b, d := l.proof.call(t, args) + if d != nil { + return textRelation(l.g, f.apply(d.render), d.why) + } + return l.shape(printedFloat(b.lo, b.hi, calcDecimals(args), b.integral), f) +} + +func (l *textLanguage) shape(s textShape, f fold) *relation { + var r *relation + for _, alt := range s { + seq := l.empty + for _, run := range alt { + seq = seq.then(l.run(run, f)) + } + r = union(r, seq) + } + return r +} + +// run reads a charRun: min characters, then up to max-min more. +func (l *textLanguage) run(c charRun, f fold) *relation { + one := newRelation(l.g) + for from := range one.to { + for _, ch := range c.chars { + s := f.apply(string(ch)) + one.add(from, l.g.run(from, s), witness{text: s}) + } + } + more := l.empty + switch optional := union(one, l.empty); { + case c.max < 0: + more = optional.closure() + case c.max > c.min: + more = optional.power(c.max - c.min) + } + if c.min == 0 { + return more + } + return one.power(c.min).then(more) +} diff --git a/template.go b/template.go index 7ef9336..a673002 100644 --- a/template.go +++ b/template.go @@ -72,6 +72,9 @@ type builtin struct { // operands names the fields the call reads, which expand renders for it; nil // for a builtin that reads none. operands func(args []string) []string + // emits is the text a call can print, for the datatype check; nil for calc and the + // transforms, whose text the check derives from what they read. + emits func(args []string) textShape } // funcCall splits a "{token}" body shaped name(args) into its parts; ok is false diff --git a/todo.md b/todo.md index 0c2b28f..3e670b2 100644 --- a/todo.md +++ b/todo.md @@ -6,11 +6,6 @@ The record API lands first, so the data update can use it. ### Record API -- Typed columns — a column declares its type, so `json` writes `42` rather than - `"42"` and `sql` an unquoted literal: string, integer, number, boolean, and a - way to write null. A template that can render a value its type rejects is a - load error. The option key is reserved from then on, so a common column name - like `type` is a poor pick. - Struct-filling — fill a Go struct from `fake:"…"` tags holding a path or an inline template, for parity with gofakeit and go-faker. The field's Go type is the column type, through the same conversion and load checks as typed columns, -- 2.52.0 From 770dca507aa9481fce1fb01f231d4beadaeeef18 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 15 Sep 2026 14:31:27 +0200 Subject: [PATCH 03/10] Tests: a typed column holds one value bounded by its arguments, and a lone null CSV column is a blank line --- datatype_test.go | 79 +++++++++++++++++++++++++++--------------------- record_test.go | 28 +++++++++-------- 2 files changed, 59 insertions(+), 48 deletions(-) diff --git a/datatype_test.go b/datatype_test.go index aa8c4a1..e65f711 100644 --- a/datatype_test.go +++ b/datatype_test.go @@ -4,6 +4,7 @@ import ( "encoding/json" "regexp" "slices" + "strconv" "strings" "testing" ) @@ -29,8 +30,9 @@ func TestDatatypeAndNullSitOnlyInAColumn(t *testing.T) { `{"format":"{p}","p":{"format":"{n}","n":{"format":"1","datatype":"integer"}}}`: "datatype only types a record column", `{"format":"{n}","repeat":2,"n":{"format":"1","datatype":"integer"}}`: "datatype only types a record column", `null`: `so write ""`, - `{"format":"{p}","p":{"format":"{x}","x":[null,"a"]}}`: `so write ""`, - `{"format":"","c":[{"format":"1","datatype":"integer"},"x"]}`: "a column holds one datatype", + `{"format":"{p}","p":{"format":"{x}","x":[null,"a"]}}`: `so write ""`, + `{"format":"","c":[{"format":"1","datatype":"integer"},"x"]}`: `write it as {"format":"x","datatype":"integer"}`, + `{"format":"","c":[{"format":"1","datatype":"integer"},{"format":"true","datatype":"boolean"}]}`: "a column holds one datatype", } { if _, err := compile(parse(t, src)); err == nil || !strings.Contains(err.Error(), want) { t.Errorf("compile(%s) = %v, want an error containing %q", src, err, want) @@ -38,52 +40,66 @@ func TestDatatypeAndNullSitOnlyInAColumn(t *testing.T) { } } -func TestDatatypeRejectsARenderItsTypeRejects(t *testing.T) { - cat := `[{"format":"{code}","code":"200"},{"format":"{code}","code":"2x"}]` +func TestDatatypeRejectsAValueItsTypeRejects(t *testing.T) { + tree := map[string]string{ + "cat": `[{"format":"{code}","code":"200"},{"format":"{code}","code":"2x"}]`, + "src": `{"format":"","score":[null,{"format":"{int(1,9)}","datatype":"integer"}]}`, + } for _, c := range []struct{ name, column, want string }{ - {"a leading zero", `{"format":"{digits(3)}","datatype":"integer"}`, "which is not an integer"}, - {"a fraction", `{"format":"{v}","v":["1","1.5"],"datatype":"integer"}`, `can render "1.5", which is not an integer`}, - {"a signed sample before digits", `{"format":"{int(-5,5)}{digits(2)}","datatype":"integer"}`, "which is not an integer"}, - {"a separator", `{"format":"{int(1,9)}","repeat":2,"separator":",","datatype":"integer"}`, "which is not an integer"}, - {"through a reference", `{"format":"{/cat.code}","datatype":"integer"}`, `can render "2x"`}, - {"a bare dot", `{"format":".5","datatype":"number"}`, `can render ".5", which is not a number`}, - {"a trailing dot", `{"format":"{int(1,9)}.","datatype":"number"}`, "which is not a number"}, - {"a plus sign", `{"format":"+1","datatype":"number"}`, `can render "+1"`}, - {"a capital", `{"format":"{b}","b":["true","True"],"datatype":"boolean"}`, `can render "True", which is not a boolean`}, - {"an upper-casing transform", `{"format":"{uppercase(b)}","b":["true","false"],"datatype":"boolean"}`, "which is not a boolean"}, - {"an operand that is not always a number", `{"format":"{calc(a * 2)}","a":["1","x"],"datatype":"number"}`, `operand "a" can render "x"`}, - {"a divisor that can be zero", `{"format":"{calc(a / b)}","a":"{int(1,9)}","b":"{int(0,9)}","datatype":"number"}`, "divides by b, which can be zero"}, - {"an overflow", `{"format":"{calc(a * a)}","a":"{digits(200)}","datatype":"number"}`, "can overflow"}, - {"a division in an integer column", `{"format":"{calc(a / b)}","a":"{int(1,9)}","b":"{int(1,9)}","datatype":"integer"}`, "which is not an integer"}, + {"a sample with leading zeros", `{"format":"{digits(3)}","datatype":"integer"}`, "{digits(3)} prints text, not an integer"}, + {"a fraction", `{"format":"{v}","v":["1","1.5"],"datatype":"integer"}`, `"1.5" is not an integer`}, + {"past int64", `{"format":"9223372036854775808","datatype":"integer"}`, "past the int64 range"}, + {"a float sample", `{"format":"{float(0,1,2)}","datatype":"integer"}`, "{float(0,1,2)} prints a number, not an integer"}, + {"composed digits", `{"format":"1{digits(2)}","datatype":"integer"}`, "is not one value"}, + {"a sign before a sample", `{"format":"-{int(1,9)}","datatype":"integer"}`, "is not one value"}, + {"a repeat", `{"format":"{int(1,9)}","repeat":2,"separator":",","datatype":"integer"}`, "carries a repeat"}, + {"through a reference", `{"format":"{/cat.code}","datatype":"integer"}`, `"2x" is not an integer`}, + {"a null through a reference", `{"format":"{/src.score}","datatype":"integer"}`, "reads a null"}, + {"a bare dot", `{"format":".5","datatype":"number"}`, `".5" is not a number`}, + {"a plus sign", `{"format":"+1","datatype":"number"}`, `"+1" is not a number`}, + {"a text sample", `{"format":"{hex(4)}","datatype":"number"}`, "{hex(4)} prints text, not a number"}, + {"a capital", `{"format":"{b}","b":["true","True"],"datatype":"boolean"}`, `"True" is not a boolean`}, + {"a transform", `{"format":"{lowercase(b)}","b":["TRUE","FALSE"],"datatype":"boolean"}`, "{lowercase(b)} rewrites text"}, + {"a number as a boolean", `{"format":"{int(0,1)}","datatype":"boolean"}`, "{int(0,1)} prints an integer, not a boolean"}, + {"an operand that is not always a number", `{"format":"{calc(a * 2)}","a":["1","x"],"datatype":"number"}`, `operand "a": "x" is not a number`}, + {"a divisor that can be zero", `{"format":"{calc(a / b)}","a":"{int(1,9)}","b":"{int(0,9)}","datatype":"number"}`, "divides by b, which is not proven nonzero"}, + {"an overflow", `{"format":"{calc(a * a)}","a":"{digits(200)}","datatype":"number"}`, "is not proven within 1e300"}, + {"a division in an integer column", `{"format":"{calc(a / b)}","a":"{int(1,9)}","b":"{int(1,9)}","datatype":"integer"}`, "prints a number, not an integer"}, } { row := `{"format":"","col":` + c.column + `}` - _, err := New(WithoutShippedData(), WithDataPath(writeData(t, map[string]string{"cat": cat, "row": row}))) + files := map[string]string{"row": row} + for name, body := range tree { + files[name] = body + } + _, err := New(WithoutShippedData(), WithDataPath(writeData(t, files))) if err == nil || !strings.Contains(err.Error(), c.want) { t.Errorf("%s: New = %v, want an error containing %q", c.name, err, c.want) } - f := newGenerator(t, writeData(t, map[string]string{"cat": cat})) + f := newGenerator(t, writeData(t, tree)) if _, err := f.NewTemplate(row); err == nil || !strings.Contains(err.Error(), c.want) { t.Errorf("%s: NewTemplate = %v, want the inline template refused the same way", c.name, err) } } } -var integerText = regexp.MustCompile(`^-?(0|[1-9][0-9]*)$`) +var jsonInteger = regexp.MustCompile(`^-?(0|[1-9][0-9]*)$`) func TestDatatypeAcceptsAColumnThatAlwaysParses(t *testing.T) { cat := `[{"format":"{code}","code":"200"},{"format":"{code}","code":"404"}]` for _, column := range []string{ `{"format":"{int(1,99)}","datatype":"integer"}`, - `{"format":"-{int(1,9)}","datatype":"integer"}`, - `{"format":"1{digits(2)}","datatype":"integer"}`, + `{"format":"{int(-9,-1)}","datatype":"integer"}`, `{"format":"{seq()}","datatype":"integer"}`, `{"format":"{/cat.code}","datatype":"integer"}`, + `{"format":"{a|b}","a":"1","b":"{int(5,9)}","datatype":"integer"}`, + `{"format":"{float(-1.5,9.5,0)}","datatype":"integer"}`, `{"format":"{float(-1,1,2)}","datatype":"number"}`, - `{"format":"{int(1,9)}e{int(1,9)}","datatype":"number"}`, - `{"format":"6.022e23","datatype":"number"}`, - `{"format":"{lowercase(b)}","b":["TRUE","False"],"datatype":"boolean"}`, + `{"format":"{v}","v":["1","2.5","6.022e23"],"datatype":"number"}`, + `{"format":"{b}","b":["true","false"],"datatype":"boolean"}`, `{"format":"{calc(net * qty, 2)}","net":["19.99","5.00"],"qty":["3","7"],"datatype":"number"}`, `{"format":"{calc(a + b)}","a":"{int(1,9)}","b":"{int(-9,9)}","datatype":"integer"}`, + `{"format":"{calc(x / (b - c), 2)}","x":"{int(1,9)}","b":"{int(10,20)}","c":"{int(1,5)}","datatype":"number"}`, + `{"format":"{calc(x * x * x * x, 2)}","x":"{float(0,1,80)}","datatype":"number"}`, `{"format":"{calc(a / (b + 1), 2)}","a":"{int(1,9)}","b":"{digits(2)}","datatype":"number"}`, `{"format":"{calc(a / b, 0)}","a":"{int(1,9)}","b":"{int(1,9)}","datatype":"integer"}`, `{"format":"{calc(sub * 1.25, 2)}","sub":{"format":"{calc(a * b)}","a":"{int(1,9)}","b":"{float(0,5,2)}"},"datatype":"number"}`, @@ -107,7 +123,8 @@ func TestDatatypeAcceptsAColumnThatAlwaysParses(t *testing.T) { c := r.Columns()[0] _, isBool := m["col"].(bool) _, isNumber := m["col"].(float64) - if c.DataType == DataTypeBoolean && !isBool || c.DataType != DataTypeBoolean && !isNumber || c.DataType == DataTypeInteger && !integerText.MatchString(c.Value) { + _, int64Err := strconv.ParseInt(c.Value, 10, 64) + if c.DataType == DataTypeBoolean && !isBool || c.DataType != DataTypeBoolean && !isNumber || c.DataType == DataTypeInteger && (!jsonInteger.MatchString(c.Value) || int64Err != nil) { t.Errorf("%s: column %+v written as %s, want its datatype", column, c, r.JSON()) break } @@ -151,11 +168,3 @@ func TestNullColumn(t *testing.T) { t.Errorf("List() = %v, want the null column row.gone, which Fake accepts", f.List()) } } - -func TestEveryBuiltinSaysWhatItEmits(t *testing.T) { - for name, b := range builtins { - if _, isTransform := transforms[name]; b.emits == nil && name != "calc" && !isTransform { - t.Errorf("builtin %s declares no emits, so a typed column calling it cannot be checked", name) - } - } -} diff --git a/record_test.go b/record_test.go index 520ceef..c5fb717 100644 --- a/record_test.go +++ b/record_test.go @@ -134,19 +134,21 @@ func TestRecordCSV(t *testing.T) { } func TestRecordCSVEmptyValueStaysARow(t *testing.T) { - for _, body := range []string{`{"format": "", "note": ""}`, `{"format": "", "note": null}`} { - f := newGenerator(t, writeData(t, map[string]string{"blank": body}), WithSeed(1)) - r, err := f.FakeRecord("blank") - if err != nil { - t.Fatal(err) - } - rows, err := csv.NewReader(strings.NewReader(r.CSVHeader() + "\n" + r.CSVLine() + "\n")).ReadAll() - if err != nil { - t.Fatalf("csv: %v", err) - } - if len(rows) != 2 || len(rows[1]) != 1 || rows[1][0] != "" { - t.Fatalf("%s: one empty column parsed to %v, want a header and one row of one empty field", body, rows) - } + dir := writeData(t, map[string]string{"blank": `{"format": "", "note": ""}`, "gone": `{"format": "", "note": null}`}) + f := newGenerator(t, dir, WithSeed(1)) + r, err := f.FakeRecord("blank") + if err != nil { + t.Fatal(err) + } + rows, err := csv.NewReader(strings.NewReader(r.CSVHeader() + "\n" + r.CSVLine() + "\n")).ReadAll() + if err != nil { + t.Fatalf("csv: %v", err) + } + if len(rows) != 2 || len(rows[1]) != 1 || rows[1][0] != "" { + t.Fatalf("one empty column parsed to %v, want a header and one row of one empty field", rows) + } + if r, err := f.FakeRecord("gone"); err != nil || r.CSVLine() != "" { + t.Fatalf("a lone null column wrote %q, %v; want the blank line PostgreSQL's COPY reads as null", r.CSVLine(), err) } } -- 2.52.0 From 5615ab6d89f285dd719bfab96663e1a33da17106 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 15 Sep 2026 14:33:04 +0200 Subject: [PATCH 04/10] Check a typed column's value, not its text: one literal, value call, calc or read, bounded by its arguments --- README.md | 57 ++++--- builtins.go | 130 ++++------------ calc.go | 262 +------------------------------- datatype.go | 79 +++------- graph.go | 2 +- node.go | 2 +- record.go | 14 +- renderlang.go | 403 -------------------------------------------------- template.go | 6 +- todo.md | 4 + value.go | 264 +++++++++++++++++++++++++++++++++ 11 files changed, 379 insertions(+), 844 deletions(-) delete mode 100644 renderlang.go create mode 100644 value.go diff --git a/README.md b/README.md index 45a8b4b..ae37af6 100644 --- a/README.md +++ b/README.md @@ -271,25 +271,31 @@ string: Writes e.g. `{"id":1,"paid":true,"total":59.97}`. A column is a field of the top-level template, or an item of a choice standing in for one; `datatype` anywhere else is a -load error. So is a column that can render text its datatype rejects — `integer` takes -`-?(0|[1-9][0-9]*)`, `number` a JSON number, `boolean` `true` or `false` — and the -error shows such a render: +load error. A typed column holds one value, alone in its format: a literal, one +`{int()}`, `{float()}`, `{seq()}` or `{calc()}` call, or a read that lands only on such +values. `integer` is an int64 written `-?(0|[1-9][0-9]*)` — `{float()}` prints one at +`0` decimals — `number` a JSON number, `boolean` `true` or `false`. A value its +datatype cannot hold is a load error naming it: ```text -order.id: datatype integer, but it can render "000", which is not an integer +order.id: datatype integer: {digits(3)} prints text, not an integer +order.id: datatype integer: "1{digits(2)}" is not one value; write one literal or one {int()}, {float()}, {seq()} or {calc()}, or read one ``` -A `{calc()}` fills an `integer` or `number` column only where it provably prints no -`NaN` or `Inf`: each operand is a plain decimal — a sign, digits, one dot — of at most -300 bytes, or a field holding only such a calc, and no divisor can be zero. An -`integer` column also needs a decimals count of `0`, or integer operands and no `/`. +A typed column's `{calc()}` must be proven to print a number: each operand a number +literal, an `{int()}`, `{float()}`, `{seq()}` or `{digits()}` call, a calc, or a read of +such values, whose bounds keep every divisor from zero and the result within `1e300`. +What the bounds cannot show is refused — `{calc(a / b)}: divides by b, which is not +proven nonzero`. The calc fills an `integer` column at `0` decimals, or over whole +operands with no `/`. ### Null A `null` item draws a record column as null: `json` writes `null`, `sql` `NULL`, and -`csv` an empty field, with an empty string written `""` so PostgreSQL's `COPY … CSV` -reads both back. `Fake` renders a null as `""`. The other items' weights skew its -odds: +`csv` an empty field, with an empty string written `""` — the convention PostgreSQL's +`COPY … CSV` reads. A record of one null column is a blank line, which `COPY` reads as +null but most CSV readers skip, so write such a record as `json` or `sql`. `Fake` +renders a null as `""`. The other items' weights skew its odds: ```json { "format": "", "deleted_at": null, "middle": [null, { "format": "{n}", "n": ["Ann", "Eva"], "weight": 3 }] } @@ -363,7 +369,7 @@ Renders e.g. `19.99 x 3 = 59.97`. An operand that can never be a number (`"abc"` or a choice of such) is rejected at load, as is a division by a constant zero (`1/0`, or a fixed `"0"` field); an operand that sometimes is not a number yields `NaN`, and a division by one that is not constant `Inf` — both print rather than -fail. +fail, except in a [typed column](#datatype), which must prove neither happens. ### Transforms @@ -556,11 +562,13 @@ tokens add cost in proportion to the output. - **64-bit targets only.** The gate builds amd64, and the buffer sizing a render pre-computes (renders × bytes) assumes a 64-bit int; on a 32-bit target it could overflow and panic. -- **A constant zero divisor is a load error; a divisor that is not constant prints - `Inf`.** `1/0` and a fixed `"0"` field are decidable, so they join the - never-numeric operand as a load error; the fold stops where an operand varies, +- **A constant zero divisor is a load error; in a string column a divisor that is not + constant prints `Inf`.** `1/0` and a fixed `"0"` field are decidable, so they join + the never-numeric operand as a load error; the fold stops where an operand varies, so `a/(b*c)` with `b` fixed at `0` and `c` varying loads and prints `Inf` every - draw — catching it needs zero-absorbing algebra for a shape nobody writes. + draw — catching it needs zero-absorbing algebra for a shape nobody writes. A + [typed column](#datatype) bounds its operands instead and refuses a divisor it + cannot keep from zero. - **In data, a default written out and a constant spelled as a sample are load errors.** `weight: 1`, `repeat: 1`, `separator: ""`, `datatype: "string"`, `int(5,5)`, `float(1,1,2)`, @@ -605,6 +613,17 @@ tokens add cost in proportion to the output. - **Null is a `null` item, not a rate.** A null is one more outcome of a column's draw, so a choice's weights skew it like any other; a null-rate option would be a second way to state odds. +- **A typed column holds one value, not composed text.** Its bounds come from a + literal or a call's arguments, so a load error names a real value, a range check is + one comparison, and `1{digits(2)}` is a second spelling of `{int(100,199)}`. +- **A typed column's calc is refused unless proven.** Operand bounds must keep each + divisor from zero and the result finite; what they cannot show is refused rather + than trusted, since a bare `NaN` breaks the JSON and SQL it lands in. +- **`Column` carries text, not a Go value.** `Value` is the rendered string beside + `DataType` and `Null`, which each serializer writes as the load check proved it; a + `Value any` would hand every caller a type switch. +- **The package stays flat.** Go ties a package to one directory, so folders would + split the API into packages. - **The performance gate asserts allocations, not wall-clock time.** `AllocsPerRun` is deterministic across machines, so a ±10% ceiling does not flake under CI load, while time varies with the machine and its neighbours. A rendering slowdown @@ -660,9 +679,9 @@ hold.go the hold: one draw per expansion for paths and operands, and its reference.go reference sigils, and binding references across the tree graph.go the render graph: edges, cycles, the repeat bound, tree walks builtins.go the {name()} function registry and its implementations -calc.go the {calc()} arithmetic evaluator: parser, eval, validation, and the proof a typed column's calc is finite -datatype.go column datatypes: DataType, where datatype and null may sit, and the load check every typed render passes -renderlang.go what text a node can render, as relations over a scalar's grammar +calc.go the {calc()} arithmetic evaluator: parser, eval, validation +datatype.go column datatypes: DataType, where datatype and null may sit, a column's datatype +value.go the value proof: what a typed column or calc operand holds, checked at load data.go data loading: fs.FS folders/files -> namespace tree, multi-source merge cmd/fejkdata/ the fejkdata CLI data/ shipped data (JSON), embedded at build: locale folders + a misc folder diff --git a/builtins.go b/builtins.go index c958a54..833c2e2 100644 --- a/builtins.go +++ b/builtins.go @@ -5,7 +5,6 @@ import ( "errors" "fmt" "math" - "slices" "strconv" "strings" "unicode" @@ -25,36 +24,42 @@ const ( // samples read only the rng. A time-based id (uuid v7, ulid) draws its timestamp // from the rng, not the wall clock, so seeded output stays reproducible. var builtins = map[string]builtin{ - "luhn": {arity: 0, prep: derive(func(e string) string { return string(rune('0' + luhnCheck(e))) }), emits: always(textShape{{{decimalDigits, 1, 1}}})}, - "mod11": {arity: 0, prep: derive(mod11Check), emits: always(textShape{{{decimalDigits + "X", 1, 1}}})}, - "ean": {arity: 0, prep: derive(eanCheck), emits: always(textShape{{{decimalDigits, 1, 1}}})}, - "uuid": {arity: 0, prep: sample(uuidV7), emits: always(uuidShape)}, - "ulid": {arity: 0, prep: sample(ulid), emits: always(textShape{{{crockford[:8], 1, 1}, {crockford, 25, 25}}})}, - "nanoid": sampleOf(nanoidAlphabet), - "hex": sampleOf(hexDigits), - "digits": sampleOf(decimalDigits), - "upper": sampleOf("ABCDEFGHIJKLMNOPQRSTUVWXYZ"), - "lower": sampleOf("abcdefghijklmnopqrstuvwxyz"), + "luhn": {arity: 0, prep: derive(func(e string) string { return string(rune('0' + luhnCheck(e))) })}, + "mod11": {arity: 0, prep: derive(mod11Check)}, + "ean": {arity: 0, prep: derive(eanCheck)}, + "uuid": {arity: 0, prep: sample(uuidV7)}, + "ulid": {arity: 0, prep: sample(ulid)}, + "nanoid": {arity: 1, check: posIntArg, prep: chars(nanoidAlphabet)}, + "hex": {arity: 1, check: posIntArg, prep: chars(hexDigits)}, + "digits": {arity: 1, check: posIntArg, prep: chars("0123456789"), number: func(a []string) (proven, DataType) { + return bounded(0, math.Pow(10, float64(atoi(a[0])))-1, true), DataTypeString + }}, + "upper": {arity: 1, check: posIntArg, prep: chars("ABCDEFGHIJKLMNOPQRSTUVWXYZ")}, + "lower": {arity: 1, check: posIntArg, prep: chars("abcdefghijklmnopqrstuvwxyz")}, "base64": {arity: 1, check: posIntArg, prep: func(a []string) callFn { n := atoi(a[0]) return func(s *session, _ string, _ []string) string { return base64.StdEncoding.EncodeToString(randBytes(s, n)) } - }, emits: base64Shape}, + }}, "int": {arity: 2, check: intRangeArgs, prep: func(a []string) callFn { lo, span := atoi(a[0]), atoi(a[1])-atoi(a[0])+1 return func(s *session, _ string, _ []string) string { return strconv.Itoa(lo + s.IntN(span)) } - }, emits: intShape}, + }, number: func(a []string) (proven, DataType) { + return bounded(float64(atoi(a[0])), float64(atoi(a[1])), true), DataTypeInteger + }}, "float": {arity: 3, check: floatArgs, prep: func(a []string) callFn { lo, hi, dp := atof(a[0]), atof(a[1]), atoi(a[2]) return func(s *session, _ string, _ []string) string { return strconv.FormatFloat(lo+s.Float64()*(hi-lo), 'f', dp, 64) } - }, emits: func(a []string) textShape { return printedFloat(atof(a[0]), atof(a[1]), atoi(a[2]), false) }}, + }, number: func(a []string) (proven, DataType) { + return printedNumber(bounded(atof(a[0]), atof(a[1]), false), atoi(a[2])) + }}, "iban": {arity: 1, check: ibanArg, prep: func(a []string) callFn { cc := a[0] return func(s *session, _ string, _ []string) string { return iban(s, cc) } - }, emits: ibanShape}, + }}, "calc": {arity: -1, check: checkCalc, prep: calcPrep, operands: calcOperands}, "lowercase": {arity: 1, check: transformArg, prep: transformPrep(strings.ToLower), operands: transformOperand}, "uppercase": {arity: 1, check: transformArg, prep: transformPrep(strings.ToUpper), operands: transformOperand}, @@ -70,7 +75,9 @@ var builtins = map[string]builtin{ return func(s *session, _ string, _ []string) string { return strconv.FormatUint(s.next(key), 10) } - }, emits: always(textShape{{{nonZeroDigits, 1, 1}, {decimalDigits, 0, 19}}})}, + }, number: func([]string) (proven, DataType) { + return bounded(1, math.MaxInt64, true), DataTypeInteger + }}, } // derive and sample are the two argument-free builtin shapes: a derivation reads @@ -95,82 +102,6 @@ func chars(alphabet string) func([]string) callFn { } } -// sampleOf is the builtin that draws n characters from an alphabet. -func sampleOf(alphabet string) builtin { - return builtin{arity: 1, check: posIntArg, prep: chars(alphabet), emits: func(a []string) textShape { - n := atoi(a[0]) - return textShape{{{alphabet, n, n}}} - }} -} - -// always is the emits of a builtin whose args do not change what it can print. -func always(s textShape) func([]string) textShape { - return func([]string) textShape { return s } -} - -var uuidShape = textShape{{{hexDigits, 8, 8}, {"-", 1, 1}, {hexDigits, 4, 4}, {"-", 1, 1}, {"7", 1, 1}, {hexDigits, 3, 3}, {"-", 1, 1}, {"89ab", 1, 1}, {hexDigits, 3, 3}, {"-", 1, 1}, {hexDigits, 12, 12}}} - -const base64Alphabet = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/" - -func base64Shape(a []string) textShape { - n := atoi(a[0]) - pad := (3 - n%3) % 3 - size := 4*((n+2)/3) - pad - return textShape{{{base64Alphabet, size, size}, {"=", pad, pad}}} -} - -// intShape is what int prints: a sign only below zero, and no leading zero. -func intShape(a []string) textShape { - lo, hi := atoi(a[0]), atoi(a[1]) - var s textShape - if lo <= 0 && hi >= 0 { - s = append(s, []charRun{{"0", 1, 1}}) - } - if hi > 0 { - s = append(s, []charRun{{nonZeroDigits, 1, 1}, {decimalDigits, 0, len(a[1]) - 1}}) - } - if lo < 0 { - s = append(s, []charRun{{"-", 1, 1}, {nonZeroDigits, 1, 1}, {decimalDigits, 0, len(a[0]) - 2}}) - } - return s -} - -func ibanShape(a []string) textShape { - cc, digits := a[0], ibanLen[a[0]]-2 - return textShape{{{cc[:1], 1, 1}, {cc[1:], 1, 1}, {decimalDigits, digits, digits}}} -} - -// shortestFraction bounds the fraction FormatFloat's shortest form prints: at most 17 -// significant digits after up to 323 zeros. -const shortestFraction = 340 - -// printedFloat is what strconv.FormatFloat(v, 'f', dp, 64) prints for a v in [lo, hi] -// that is whole when integral. -func printedFloat(lo, hi float64, dp int, integral bool) textShape { - digits := len(strconv.FormatFloat(math.Floor(math.Max(math.Abs(lo), math.Abs(hi))), 'f', 0, 64)) + 1 // one more for a rounding carry - wholes := [][]charRun{{{"0", 1, 1}}, {{nonZeroDigits, 1, 1}, {decimalDigits, 0, digits - 1}}} - fractions := [][]charRun{nil} - switch { - case dp > 0: - fractions = [][]charRun{{{".", 1, 1}, {decimalDigits, dp, dp}}} - case dp < 0 && !integral: - fractions = append(fractions, []charRun{{".", 1, 1}, {decimalDigits, 1, shortestFraction}}) - } - signs := [][]charRun{nil} - if lo < 0 || math.Signbit(lo) { - signs = append(signs, []charRun{{"-", 1, 1}}) - } - var s textShape - for _, sign := range signs { - for _, whole := range wholes { - for _, fraction := range fractions { - s = append(s, slices.Concat(sign, whole, fraction)) - } - } - } - return s -} - const hexDigits = "0123456789abcdef" // transforms are the builtins that rewrite one operand's value; they nest, so @@ -183,19 +114,20 @@ var transforms = map[string]func(string) string{ // unwrapTransform peels nested transform calls off an operand arg, returning the // field it finally names and the transforms to apply, innermost last. -func unwrapTransform(arg string) (leaf string, chain []string, err error) { +func unwrapTransform(arg string) (leaf string, chain []func(string) string, err error) { for { name, args, isCall := funcCall(arg) if !isCall { return arg, chain, nil } - if _, isTransform := transforms[name]; !isTransform { + fn, isTransform := transforms[name] + if !isTransform { return "", nil, fmt.Errorf("%s(%s) is not a transform, so it cannot be an operand", name, strings.Join(args, ",")) } if len(args) != 1 { return "", nil, fmt.Errorf("%s takes 1 arg, got %d", name, len(args)) } - chain = append(chain, name) + chain = append(chain, fn) arg = args[0] } } @@ -226,14 +158,10 @@ func transformPrep(outer func(string) string) func([]string) callFn { if err != nil { panic(fmt.Sprintf("fejkdata: transform arg %q reached prep unvalidated: %v", a[0], err)) } - fns := make([]func(string) string, len(chain)) - for i, name := range chain { - fns[i] = transforms[name] - } return func(_ *session, _ string, operands []string) string { v := operands[0] - for i := len(fns) - 1; i >= 0; i-- { - v = fns[i](v) + for i := len(chain) - 1; i >= 0; i-- { + v = chain[i](v) } return outer(v) } diff --git a/calc.go b/calc.go index 7d7f2cd..901e8c9 100644 --- a/calc.go +++ b/calc.go @@ -6,7 +6,6 @@ import ( "strconv" "strings" "unicode" - "unicode/utf8" ) // calcNode is a parsed expression node. It evaluates over the operand values expand @@ -200,6 +199,14 @@ func calcPrep(args []string) callFn { } } +// calcDecimals is a calc's decimals count, or -1 for the shortest form. +func calcDecimals(args []string) int { + if len(args) == 2 { + return atoi(args[1]) + } + return -1 +} + // indexVars replaces each operand name with its position in the values expand reads. // Both sides take that order from calcVars, so they cannot drift. func indexVars(n calcNode, at map[string]int) calcNode { @@ -384,256 +391,3 @@ func contains(bs []byte, b byte) bool { } return false } - -// calcDecimals is a calc's decimals count, or -1 for the shortest form. -func calcDecimals(args []string) int { - if len(args) == 2 { - return atoi(args[1]) - } - return -1 -} - -// calcLimit is the largest magnitude a proof accepts as finite, far enough below -// math.MaxFloat64 that rounding in the bounds cannot hide an overflow. -const calcLimit = 1e300 - -// maxOperandLen is the longest operand text a proof bounds by its length, so that -// bound, 10^maxOperandLen, stays within calcLimit. -const maxOperandLen = 300 - -// calcBound is what a proof knows of every value a calc can take: it lies in [lo, hi], -// is at least nonZero from zero unless nonZero is 0, and is whole when integral. -type calcBound struct { - lo, hi, nonZero float64 - integral bool -} - -func magnitude(b calcBound) float64 { return math.Max(math.Abs(b.lo), math.Abs(b.hi)) } - -// doubt is why a proof could not show a calc finite, and the render that shows it. -type doubt struct{ render, why string } - -type bounded struct { - b calcBound - d *doubt -} - -// calcProof bounds a typed column's calcs from their operands' renders, to show each -// prints a number rather than NaN or Inf. -type calcProof struct { - decimal *textLanguage - operands map[node]bounded - lengths map[node]int -} - -func newCalcProof() *calcProof { - p := &calcProof{operands: map[node]bounded{}, lengths: map[node]int{}} - p.decimal = newTextLanguage(decimalGrammar, p) - return p -} - -// call bounds one calc token of t. -func (p *calcProof) call(t *template, args []string) (calcBound, *doubt) { - expr, err := parseCalc(args[0]) - if err != nil { - panic(fmt.Sprintf("fejkdata: calc(%q) reached a proof unparsed: %v", args[0], err)) - } - b, d := p.expr(expr, t.fields) - if d != nil { - return b, &doubt{d.render, fmt.Sprintf("{calc(%s)}: %s", strings.Join(args, ", "), d.why)} - } - return b, nil -} - -func (p *calcProof) expr(n calcNode, fields map[string]node) (calcBound, *doubt) { - switch n := n.(type) { - case calcNum: - v := float64(n) - return calcBound{v, v, v, v == math.Trunc(v)}, nil - case calcVar: - return p.operand(string(n), fields[string(n)]) - case calcNeg: - b, d := p.expr(n.x, fields) - return calcBound{-b.hi, -b.lo, b.nonZero, b.integral}, d - case calcBin: - l, d := p.expr(n.l, fields) - if d != nil { - return l, d - } - r, d := p.expr(n.r, fields) - if d != nil { - return r, d - } - return combine(n, l, r) - } - panic(fmt.Sprintf("fejkdata: calc node %T has no bound", n)) -} - -// combine bounds one operation from the bounds of its sides. -func combine(n calcBin, l, r calcBound) (calcBound, *doubt) { - b := calcBound{integral: l.integral && r.integral} - switch n.op { - case '+': - b.lo, b.hi = l.lo+r.lo, l.hi+r.hi - case '-': - b.lo, b.hi = l.lo-r.hi, l.hi-r.lo - case '*': - b.lo = min(l.lo*r.lo, l.lo*r.hi, l.hi*r.lo, l.hi*r.hi) - b.hi = max(l.lo*r.lo, l.lo*r.hi, l.hi*r.lo, l.hi*r.hi) - b.nonZero = l.nonZero * r.nonZero - default: - if r.nonZero == 0 { - return b, &doubt{"+Inf", fmt.Sprintf("divides by %s, which can be zero", calcText(n.r))} - } - m := magnitude(l) / r.nonZero - b = calcBound{lo: -m, hi: m, nonZero: l.nonZero / magnitude(r)} - } - if b.lo > 0 || b.hi < 0 { - b.nonZero = math.Max(b.nonZero, math.Min(math.Abs(b.lo), math.Abs(b.hi))) - } - if !(magnitude(b) <= calcLimit) { - return b, &doubt{"+Inf", calcText(n) + " can overflow"} - } - return b, nil -} - -// operand bounds a calc operand, once per node. -func (p *calcProof) operand(name string, n node) (calcBound, *doubt) { - if seen, done := p.operands[n]; done { - return seen.b, seen.d - } - b, d := p.measure(name, n) - p.operands[n] = bounded{b, d} - return b, d -} - -// measure bounds an operand through the calc it renders when that is all it renders, -// and otherwise from its text: a plain decimal of at most maxOperandLen bytes. -func (p *calcProof) measure(name string, n node) (calcBound, *doubt) { - if t, ok := n.(*template); ok { - if args, isCalc := soleCalc(t); isCalc { - b, d := p.call(t, args) - return rounded(b, calcDecimals(args)), d - } - } - text := p.decimal.node(n, nil) - if w, escapes := text.escape(decimalAccept); escapes { - why := fmt.Sprintf("operand %q can render %s, which is not a plain decimal", name, w) - if w.why != "" { - why += ": " + w.why - } - return calcBound{}, &doubt{"NaN", why} - } - size := p.length(n) - if size > maxOperandLen { - return calcBound{}, &doubt{"NaN", fmt.Sprintf("operand %q can render more than %d bytes, too many to bound", name, maxOperandLen)} - } - ends, m := text.to[1], math.Pow(10, float64(size)) - b := calcBound{hi: m, nonZero: 1 / m, integral: ends&decimalFractional == 0} - if ends&decimalNegative != 0 { - b.lo = -m - } - if ends&decimalZero != 0 { - b.nonZero = 0 - } - return b, nil -} - -// soleCalc reports a template that renders one calc and nothing else, with its args. -func soleCalc(t *template) ([]string, bool) { - if t.repeat != 1 || len(t.ops) != 1 || t.ops[0].kind != 'b' { - return nil, false - } - name, args, _ := funcCall(t.format[1 : len(t.format)-1]) - return args, name == "calc" -} - -// rounded is b once printed to dp decimals, which moves a value by up to half a unit. -func rounded(b calcBound, dp int) calcBound { - if dp < 0 { - return b - } - half := math.Pow(10, -float64(dp)) / 2 - return calcBound{b.lo - half, b.hi + half, math.Max(0, b.nonZero-half), b.integral || dp == 0} -} - -// length is the most bytes a render of n can take, anything past maxOperandLen -// reported as maxOperandLen+1. -func (p *calcProof) length(n node) int { - if size, done := p.lengths[n]; done { - return size - } - size := 0 - switch n := n.(type) { - case *choice: - for _, it := range n.items { - size = max(size, p.length(it)) - } - case *template: - size = p.formatLength(n)*n.repeat + len(n.separator)*(n.repeat-1) - } - size = min(size, maxOperandLen+1) - p.lengths[n] = size - return size -} - -func (p *calcProof) formatLength(t *template) int { - size := 0 - _ = eachToken(t.format, func(tok ftoken) error { - if tok.kind == 'l' { - size += utf8.RuneLen(tok.r) - } else { - size += p.tokenLength(t, tok.body) - } - size = min(size, maxOperandLen+1) - return nil - }) - return size -} - -// tokenLength is the most bytes one token can print. A transform never lengthens a -// render that reads as a decimal: it maps each non-ASCII rune, two bytes or more, to at -// most two ASCII letters. -func (p *calcProof) tokenLength(t *template, body string) int { - name, args, isFunc := funcCall(body) - var arms []arm - switch _, isTransform := transforms[name]; { - case !isFunc: - arms = splitArms(body, t.refs) - case isTransform: - leaf, _, _ := unwrapTransform(args[0]) - arms = []arm{splitArm(leaf, t.refs)} - case name == "calc": - b, d := p.call(t, args) - if d != nil { - return len(d.render) - } - return shapeLength(printedFloat(b.lo, b.hi, calcDecimals(args), b.integral)) - default: - return shapeLength(builtins[name].emits(args)) - } - size := 0 - for _, a := range arms { - for _, leaf := range pathLeaves(t.fields[a.key], a.tail) { - size = max(size, p.length(leaf)) - } - } - return size -} - -// shapeLength is the most bytes a shape can emit, anything past maxOperandLen reported -// as maxOperandLen+1. -func shapeLength(s textShape) int { - longest := 0 - for _, alt := range s { - size := 0 - for _, run := range alt { - if run.max < 0 { - return maxOperandLen + 1 - } - size += run.max - } - longest = max(longest, size) - } - return min(longest, maxOperandLen+1) -} diff --git a/datatype.go b/datatype.go index b92a398..94873de 100644 --- a/datatype.go +++ b/datatype.go @@ -16,7 +16,10 @@ const ( DataTypeBoolean ) -var dataTypeNames = [...]string{"string", "integer", "number", "boolean"} +var ( + dataTypeNames = [...]string{"string", "integer", "number", "boolean"} + dataTypeNouns = [...]string{"text", "an integer", "a number", "a boolean"} +) // String is the datatype as data spells it. func (d DataType) String() string { @@ -32,7 +35,7 @@ type position int const ( inFormat position = iota // rendered by a format, so neither - atTop // a category or an inline template, whose fields are the columns + atTop // a category or an inline template, whose fields may be columns inColumn // a column, or a choice item standing in for one ) @@ -64,7 +67,7 @@ func datatypeOf(m map[string]any, pos position) (DataType, error) { // columnDatatype is the datatype a column's items declare. They must agree, since a // column holds one; a column only ever null is a string. func columnDatatype(n node) (DataType, error) { - var declared []DataType + var items []*template var collect func(node) collect = func(n node) { switch n := n.(type) { @@ -73,68 +76,32 @@ func columnDatatype(n node) (DataType, error) { collect(it) } case *template: - declared = append(declared, n.datatype) + items = append(items, n) } } collect(n) - if len(declared) == 0 { + if len(items) == 0 { return DataTypeString, nil } - for _, d := range declared { - if d != declared[0] { - return declared[0], fmt.Errorf("its items declare %s and %s; a column holds one datatype, so give every item the same", declared[0], d) + for _, t := range items[1:] { + if t.datatype != items[0].datatype { + return items[0].datatype, disagreement(items[0], t) } } - return declared[0], nil + return items[0].datatype, nil } -// datatypeSpec is what a datatype's text must satisfy: a grammar, the states a render -// may end in, and how an error names the datatype. -type datatypeSpec struct { - grammar *grammar - accept uint32 - noun string -} - -var datatypeSpecs = map[DataType]datatypeSpec{ - DataTypeInteger: {numberGrammar, integerAccept, "an integer"}, - DataTypeNumber: {numberGrammar, numberAccept, "a number"}, - DataTypeBoolean: {booleanGrammar, booleanAccept, "a boolean"}, -} - -// datatypeCheck proves every render of a typed column is text its datatype takes. One -// check covers a scope, so a node several columns reach is read once per grammar. -type datatypeCheck struct { - languages map[*grammar]*textLanguage - proof *calcProof -} - -func (c *datatypeCheck) check(path string, n node) error { - t, ok := n.(*template) - if !ok || t.datatype == DataTypeString { - return nil +// disagreement names the fix for two items of one column declaring different datatypes. +func disagreement(a, b *template) error { + typed, bare := a, b + if typed.datatype == DataTypeString { + typed, bare = b, a } - spec := datatypeSpecs[t.datatype] - w, escapes := c.language(spec.grammar).node(t, nil).escape(spec.accept) - if !escapes { - return nil + switch { + case bare.datatype != DataTypeString: + return fmt.Errorf("its items declare %s and %s; a column holds one datatype", a.datatype, b.datatype) + case len(bare.fields) == 0 && bare.repeat == 1: + return fmt.Errorf(`item %q declares no datatype, and a column holds one; write it as {"format":%q,"datatype":%q}`, bare.format, bare.format, typed.datatype) } - msg := fmt.Sprintf("%s: datatype %s, but it can render %s, which is not %s", path, t.datatype, w, spec.noun) - if w.why != "" { - msg += ": " + w.why - } - return errors.New(msg) -} - -func (c *datatypeCheck) language(g *grammar) *textLanguage { - if c.proof == nil { - c.proof = newCalcProof() - c.languages = map[*grammar]*textLanguage{} - } - l, made := c.languages[g] - if !made { - l = newTextLanguage(g, c.proof) - c.languages[g] = l - } - return l + return fmt.Errorf(`an item declares no datatype beside one declaring %s; a column holds one, so give it "datatype": %q`, typed.datatype, typed.datatype) } diff --git a/graph.go b/graph.go index f65627e..f8361e1 100644 --- a/graph.go +++ b/graph.go @@ -189,7 +189,7 @@ func checkScope(s nodeScope) error { if err := s(heldCheck); err != nil { return err } - return s((&datatypeCheck{}).check) + return s((&valueProof{}).checkDatatype) } type reachMemo map[node]int diff --git a/node.go b/node.go index d5af8e3..e73de11 100644 --- a/node.go +++ b/node.go @@ -229,7 +229,7 @@ func compileTemplate(m map[string]any, pos position) (node, error) { return nil, err } fieldPos := inFormat - if pos == atTop && o.repeat == 1 { + if pos == atTop && projectsColumns(o.repeat) { fieldPos = inColumn } fields, err := compileFields(m, fieldPos) diff --git a/record.go b/record.go index fc53ee0..2658343 100644 --- a/record.go +++ b/record.go @@ -63,16 +63,14 @@ func (r *Record) CSVHeader() string { } // CSVLine renders the column values as one CSV row: a null column an empty field and an -// empty string "", the convention PostgreSQL's COPY reads a null by. +// empty string "", the convention PostgreSQL's COPY reads a null by. A record of one null +// column is a blank line, which COPY reads as null and most CSV readers skip. func (r *Record) CSVLine() string { fields := make([]string, len(r.columns)) for i, c := range r.columns { fields[i] = literal(c, csvField, "") } - if line := strings.Join(fields, ","); line != "" { - return line - } - return `""` // a blank line is a row every CSV reader drops + return strings.Join(fields, ",") } func csvField(s string) string { @@ -207,7 +205,7 @@ func recordOf(n node) (*template, []Column, error) { if !ok { return nil, nil, errors.New("names a choice, not a template; a record is a template whose fields are its columns") } - if t.repeat != 1 { + if !projectsColumns(t.repeat) { return nil, nil, fmt.Errorf("carries repeat %d, which composes its format into one string; a record projects columns instead — drop the repeat and render the record again for more rows", t.repeat) } names := recordColumns(t) @@ -225,6 +223,10 @@ func recordOf(n node) (*template, []Column, error) { return t, columns, nil } +// projectsColumns reports whether a category or inline template with this repeat is a +// record, its fields the columns; a repeat composes the format into one string instead. +func projectsColumns(repeat int) bool { return repeat == 1 } + // checkColumnRefs rejects the reference reads a record's shared draw cannot answer // for: one column rendering a level another reads a path into, and a column // reading the record back through its own path. diff --git a/renderlang.go b/renderlang.go deleted file mode 100644 index 4d56a2d..0000000 --- a/renderlang.go +++ /dev/null @@ -1,403 +0,0 @@ -package fejkdata - -import ( - "slices" - "strconv" - "strings" - "unicode/utf8" -) - -// grammar is a deterministic automaton over a scalar's text: state 0 is dead, 1 the -// start, and each state lists the runes that leave it and where they lead. -type grammar [][]arc - -type arc struct { - on string - to int -} - -func (g *grammar) run(q int, s string) int { - for _, r := range s { - if q = g.step(q, r); q == 0 { - return 0 - } - } - return q -} - -func (g *grammar) step(q int, r rune) int { - for _, a := range (*g)[q] { - if strings.ContainsRune(a.on, r) { - return a.to - } - } - return 0 -} - -const ( - decimalDigits = "0123456789" - nonZeroDigits = "123456789" -) - -// numberGrammar reads a JSON number. States: 2 "-", 3 "0", 4 more integer digits, 5 ".", -// 6 fraction digits, 7 "e", 8 its sign, 9 exponent digits. -var numberGrammar = &grammar{ - nil, - {{"-", 2}, {"0", 3}, {nonZeroDigits, 4}}, - {{"0", 3}, {nonZeroDigits, 4}}, - {{".", 5}, {"eE", 7}}, - {{decimalDigits, 4}, {".", 5}, {"eE", 7}}, - {{decimalDigits, 6}}, - {{decimalDigits, 6}, {"eE", 7}}, - {{"+-", 8}, {decimalDigits, 9}}, - {{decimalDigits, 9}}, - {{decimalDigits, 9}}, -} - -const ( - integerAccept uint32 = 1<<3 | 1<<4 - numberAccept = integerAccept | 1<<6 | 1<<9 -) - -var booleanGrammar = &grammar{ - nil, - {{"t", 2}, {"f", 6}}, - {{"r", 3}}, {{"u", 4}}, {{"e", 5}}, nil, - {{"a", 7}}, {{"l", 8}}, {{"s", 9}}, {{"e", 10}}, nil, -} - -const booleanAccept uint32 = 1<<5 | 1<<10 - -// decimalGrammar reads what a calc operand must render to be proven finite: a sign, -// digits and at most one dot. Past the sign, states 4–9 are positive and 10–15 their -// negatives: 4 zero digits, 5 a nonzero integer, 6 a leading dot, 7 zero with a dot, -// 8 a nonzero integer with a zero fraction, 9 a nonzero fraction. -var decimalGrammar = &grammar{ - nil, - {{"+", 2}, {"-", 3}, {"0", 4}, {nonZeroDigits, 5}, {".", 6}}, - {{"0", 4}, {nonZeroDigits, 5}, {".", 6}}, - {{"0", 10}, {nonZeroDigits, 11}, {".", 12}}, - {{"0", 4}, {nonZeroDigits, 5}, {".", 7}}, - {{decimalDigits, 5}, {".", 8}}, - {{"0", 7}, {nonZeroDigits, 9}}, - {{"0", 7}, {nonZeroDigits, 9}}, - {{"0", 8}, {nonZeroDigits, 9}}, - {{decimalDigits, 9}}, - {{"0", 10}, {nonZeroDigits, 11}, {".", 13}}, - {{decimalDigits, 11}, {".", 14}}, - {{"0", 13}, {nonZeroDigits, 15}}, - {{"0", 13}, {nonZeroDigits, 15}}, - {{"0", 14}, {nonZeroDigits, 15}}, - {{decimalDigits, 15}}, -} - -const ( - decimalAccept uint32 = 1<<4 | 1<<5 | 1<<7 | 1<<8 | 1<<9 | 1<<10 | 1<<11 | 1<<13 | 1<<14 | 1<<15 - decimalNegative uint32 = 0xfc00 - decimalZero uint32 = 1<<4 | 1<<7 | 1<<10 | 1<<13 - decimalFractional uint32 = 1<<9 | 1<<15 -) - -// relation is what a node's renders do to a grammar: from each state, the states a -// render can end in, and one render reaching each. -type relation struct { - g *grammar - to []uint32 - w []witness // w[from*len(to)+to] -} - -// witness is one render, cut past witnessCap bytes, and why it can occur when the text -// alone does not say. -type witness struct { - text string - cut bool - why string -} - -const witnessCap = 60 - -func (w witness) then(next witness) witness { - if w.why == "" { - w.why = next.why - } - if w.cut { - return w - } - w.text += next.text - w.cut = next.cut - if len(w.text) > witnessCap { - end := witnessCap - for !utf8.RuneStart(w.text[end]) { - end-- - } - w.text, w.cut = w.text[:end], true - } - return w -} - -func (w witness) String() string { - if w.cut { - return strconv.Quote(w.text + "…") - } - return strconv.Quote(w.text) -} - -func newRelation(g *grammar) *relation { - n := len(*g) - return &relation{g: g, to: make([]uint32, n), w: make([]witness, n*n)} -} - -func (r *relation) add(from, to int, w witness) { - if r.to[from]&(1<>= 1; k == 0 { - return out - } - } -} - -// closure is any number of renders of r in a row, where r includes the empty render. -func (r *relation) closure() *relation { - for { - next := r.then(r) - if slices.Equal(next.to, r.to) { - return r - } - r = next - } -} - -// escape finds a render from the start that ends outside accept, preferring one that -// carries a reason. -func (r *relation) escape(accept uint32) (witness, bool) { - var found witness - escapes := false - for to := range r.to { - if (r.to[1]&^accept)&(1< 1 { - r = r.then(l.text(f.apply(n.separator)).then(r).power(n.repeat - 1)) - } - } - l.memo[key] = r - return r -} - -func (l *textLanguage) text(s string) *relation { return textRelation(l.g, s, "") } - -// format reads a template's format the way expand renders it: literal runs and tokens -// in turn. -func (l *textLanguage) format(t *template, f fold) *relation { - r := l.empty - var lit strings.Builder - _ = eachToken(t.format, func(tok ftoken) error { - if tok.kind == 'l' { - lit.WriteRune(tok.r) - return nil - } - r = r.then(l.text(f.apply(lit.String()))).then(l.token(t, tok.body, f)) - lit.Reset() - return nil - }) - return r.then(l.text(f.apply(lit.String()))) -} - -// token reads one {…} token: a field read, a transform over one, a calc, or what a -// builtin emits. -func (l *textLanguage) token(t *template, body string, f fold) *relation { - name, args, isFunc := funcCall(body) - if !isFunc { - var r *relation - for _, a := range splitArms(body, t.refs) { - r = union(r, l.read(t, a, f)) - } - return r - } - if _, isTransform := transforms[name]; isTransform { - leaf, chain, _ := unwrapTransform(args[0]) - inner := slices.Clone(chain) - slices.Reverse(inner) - return l.read(t, splitArm(leaf, t.refs), append(append(inner, name), f...)) - } - if name == "calc" { - return l.calc(t, args, f) - } - return l.shape(builtins[name].emits(args), f) -} - -// read is one arm of a token: every node its path can land on. -func (l *textLanguage) read(t *template, a arm, f fold) *relation { - var r *relation - for _, leaf := range pathLeaves(t.fields[a.key], a.tail) { - r = union(r, l.node(leaf, f)) - } - return r -} - -func (l *textLanguage) calc(t *template, args []string, f fold) *relation { - b, d := l.proof.call(t, args) - if d != nil { - return textRelation(l.g, f.apply(d.render), d.why) - } - return l.shape(printedFloat(b.lo, b.hi, calcDecimals(args), b.integral), f) -} - -func (l *textLanguage) shape(s textShape, f fold) *relation { - var r *relation - for _, alt := range s { - seq := l.empty - for _, run := range alt { - seq = seq.then(l.run(run, f)) - } - r = union(r, seq) - } - return r -} - -// run reads a charRun: min characters, then up to max-min more. -func (l *textLanguage) run(c charRun, f fold) *relation { - one := newRelation(l.g) - for from := range one.to { - for _, ch := range c.chars { - s := f.apply(string(ch)) - one.add(from, l.g.run(from, s), witness{text: s}) - } - } - more := l.empty - switch optional := union(one, l.empty); { - case c.max < 0: - more = optional.closure() - case c.max > c.min: - more = optional.power(c.max - c.min) - } - if c.min == 0 { - return more - } - return one.power(c.min).then(more) -} diff --git a/template.go b/template.go index a673002..0b9e76b 100644 --- a/template.go +++ b/template.go @@ -72,9 +72,9 @@ type builtin struct { // operands names the fields the call reads, which expand renders for it; nil // for a builtin that reads none. operands func(args []string) []string - // emits is the text a call can print, for the datatype check; nil for calc and the - // transforms, whose text the check derives from what they read. - emits func(args []string) textShape + // number bounds the number a call prints and names the datatype its text is; nil + // for a builtin that prints text. + number func(args []string) (proven, DataType) } // funcCall splits a "{token}" body shaped name(args) into its parts; ok is false diff --git a/todo.md b/todo.md index 3e670b2..f252672 100644 --- a/todo.md +++ b/todo.md @@ -22,6 +22,10 @@ The record API lands first, so the data update can use it. - `code` and `symbol` sibling fields reading `currency`, as `{code} {symbol}` → a matching pair - `{a} & {b}`, each reading `person` → one person, or two when `a` and `b` name different groups - two bare `{/sv_SE.word}` → two words +- Reference inheritance — settle whether a column that is exactly one reference to + another record's column, like `{/src.score}`, takes that column's datatype and + null. Today a null there writes `""`, and a typed column reading it is refused. + Settle before draw groups and the data update. ### Data diff --git a/value.go b/value.go new file mode 100644 index 0000000..56c3d93 --- /dev/null +++ b/value.go @@ -0,0 +1,264 @@ +package fejkdata + +import ( + "fmt" + "math" + "regexp" + "strconv" + "strings" +) + +// proven is what a proof knows of every render of a node: bounds on the number each +// reads as, and per datatype why some render's text is not one ("" when none). +type proven struct { + lo, hi float64 + nonZero float64 // every value is at least this far from zero; 0 when one can be zero + integral bool + notNumber string // why some render reads as no finite number, the way calc reads it + not [len(dataTypeNames)]string +} + +// valueProof proves what typed columns and their calc operands hold, each node once per +// scope. A typed column holds one value: a literal, one value builtin, one calc, or a +// read of such values. +type valueProof struct { + memo map[node]proven +} + +// checkDatatype rejects a typed column some render of which is not text of its datatype. +func (p *valueProof) checkDatatype(path string, n node) error { + t, ok := n.(*template) + if !ok || t.datatype == DataTypeString { + return nil + } + if err := p.prove(t, t.datatype); err != nil { + return fmt.Errorf("%s: %w", path, err) + } + return nil +} + +// prove reports why some render of n is not text of datatype d. +func (p *valueProof) prove(n node, d DataType) error { + if reason := p.of(n).not[d]; reason != "" { + return fmt.Errorf("datatype %s: %s", d, reason) + } + return nil +} + +func (p *valueProof) of(n node) proven { + if v, done := p.memo[n]; done { + return v + } + if p.memo == nil { + p.memo = map[node]proven{} + } + var v proven + switch n := n.(type) { + case *choice: + v = p.unite(n.items) + case *template: + v = p.template(n) + default: + v = unproven(`it reads a null, which renders "" outside its own column`) + } + p.memo[n] = v + return v +} + +func (p *valueProof) unite(nodes []node) proven { + v := p.of(nodes[0]) + for _, n := range nodes[1:] { + w := p.of(n) + v.lo, v.hi, v.nonZero = min(v.lo, w.lo), max(v.hi, w.hi), min(v.nonZero, w.nonZero) + v.integral = v.integral && w.integral + if v.notNumber == "" { + v.notNumber = w.notNumber + } + for d := range v.not { + if v.not[d] == "" { + v.not[d] = w.not[d] + } + } + } + return v +} + +// template proves a template that renders one value: fixed text, or a format that is +// one token alone. +func (p *valueProof) template(t *template) proven { + switch { + case t.repeat != 1: + return unproven(fmt.Sprintf("%q carries a repeat, which composes text rather than one value", t.format)) + case t.fixed: + return literalValue(t.lit) + case len(t.ops) != 1: + return unproven(fmt.Sprintf("%q is not one value; write one literal or one {int()}, {float()}, {seq()} or {calc()}, or read one", t.format)) + } + body := t.format[1 : len(t.format)-1] + name, args, isFunc := funcCall(body) + switch _, isTransform := transforms[name]; { + case !isFunc: + var leaves []node + for _, a := range splitArms(body, t.refs) { + leaves = append(leaves, pathLeaves(t.fields[a.key], a.tail)...) + } + return p.unite(leaves) + case name == "calc": + return p.calc(t, body, args) + case builtins[name].number != nil: + v, prints := builtins[name].number(args) + return printing(body, prints, v) + case isTransform: + return unproven(fmt.Sprintf("{%s} rewrites text rather than printing a value; write the values it would print", body)) + } + return printing(body, DataTypeString, proven{notNumber: fmt.Sprintf("{%s} prints text, not a number", body)}) +} + +func (p *valueProof) calc(t *template, body string, args []string) proven { + expr, err := parseCalc(args[0]) + if err != nil { + panic(fmt.Sprintf("fejkdata: calc(%q) reached a proof unparsed: %v", args[0], err)) + } + v, doubt := p.expr(expr, t.fields) + if doubt == "" && !(magnitude(v) <= calcLimit) { + doubt = calcText(expr) + " is not proven within 1e300" + } + if doubt != "" { + return unproven(fmt.Sprintf("{%s}: %s", body, doubt)) + } + v, prints := printedNumber(v, calcDecimals(args)) + return printing(body, prints, v) +} + +// calcLimit is the largest magnitude a proof accepts as finite, far enough below +// math.MaxFloat64 that rounding in the bounds cannot hide an overflow. +const calcLimit = 1e300 + +// expr bounds a calc expression from its operands, or says why it cannot. +func (p *valueProof) expr(n calcNode, fields map[string]node) (proven, string) { + switch n := n.(type) { + case calcNum: + v := float64(n) + return bounded(v, v, v == math.Trunc(v)), "" + case calcVar: + v := p.of(fields[string(n)]) + if v.notNumber != "" { + return proven{}, fmt.Sprintf("operand %q: %s", string(n), v.notNumber) + } + return proven{lo: v.lo, hi: v.hi, nonZero: v.nonZero, integral: v.integral}, "" + case calcNeg: + v, doubt := p.expr(n.x, fields) + v.lo, v.hi = -v.hi, -v.lo + return v, doubt + case calcBin: + l, doubt := p.expr(n.l, fields) + if doubt != "" { + return l, doubt + } + r, doubt := p.expr(n.r, fields) + if doubt != "" { + return r, doubt + } + return combine(n, l, r) + } + panic(fmt.Sprintf("fejkdata: calc node %T has no bound", n)) +} + +// combine bounds one operation from the bounds of its sides. +func combine(n calcBin, l, r proven) (proven, string) { + var v proven + integral := l.integral && r.integral + switch n.op { + case '+': + v = bounded(l.lo+r.lo, l.hi+r.hi, integral) + case '-': + v = bounded(l.lo-r.hi, l.hi-r.lo, integral) + case '*': + v = bounded(min(l.lo*r.lo, l.lo*r.hi, l.hi*r.lo, l.hi*r.hi), max(l.lo*r.lo, l.lo*r.hi, l.hi*r.lo, l.hi*r.hi), integral) + v.nonZero = max(v.nonZero, l.nonZero*r.nonZero) + default: + if r.nonZero == 0 { + return v, fmt.Sprintf("divides by %s, which is not proven nonzero", calcText(n.r)) + } + m := magnitude(l) / r.nonZero + v = proven{lo: -m, hi: m, nonZero: l.nonZero / magnitude(r)} + } + if !(magnitude(v) <= calcLimit) { + return v, calcText(n) + " is not proven within 1e300" + } + return v, "" +} + +// bounded is a number in [lo, hi], its distance from zero read off the bounds. +func bounded(lo, hi float64, integral bool) proven { + v := proven{lo: lo, hi: hi, integral: integral} + switch { + case lo > 0: + v.nonZero = lo + case hi < 0: + v.nonZero = -hi + } + return v +} + +func magnitude(v proven) float64 { return math.Max(math.Abs(v.lo), math.Abs(v.hi)) } + +// printedNumber is v once strconv.FormatFloat prints it to dp decimals, and the datatype +// that text is: an integer when whole and within int64, else a number. +func printedNumber(v proven, dp int) (proven, DataType) { + if dp >= 0 { + half := math.Pow(10, -float64(dp)) / 2 + v = proven{lo: v.lo - half, hi: v.hi + half, nonZero: math.Max(0, v.nonZero-half), integral: v.integral || dp == 0} + } + if (dp == 0 || dp < 0 && v.integral) && magnitude(v) < math.MaxInt64 { + return v, DataTypeInteger + } + return v, DataTypeNumber +} + +// printing is v for a token whose every render is text of datatype prints, with a reason +// against each datatype that text is not. +func printing(token string, prints DataType, v proven) proven { + for d := DataTypeInteger; d <= DataTypeBoolean; d++ { + if prints != d && !(prints == DataTypeInteger && d == DataTypeNumber) { + v.not[d] = fmt.Sprintf("{%s} prints %s, not %s", token, dataTypeNouns[prints], dataTypeNouns[d]) + } + } + return v +} + +// unproven is a render no datatype and no calc can take, for why. +func unproven(why string) proven { + v := proven{notNumber: why} + for d := DataTypeInteger; d <= DataTypeBoolean; d++ { + v.not[d] = why + } + return v +} + +var ( + integerText = regexp.MustCompile(`^-?(0|[1-9][0-9]*)$`) + numberText = regexp.MustCompile(`^-?(0|[1-9][0-9]*)(\.[0-9]+)?([eE][+-]?[0-9]+)?$`) +) + +// literalValue proves fixed text: the number calc reads it as, and each datatype it is. +func literalValue(text string) proven { + var v proven + if f, err := strconv.ParseFloat(strings.TrimSpace(text), 64); err != nil || math.IsNaN(f) || math.IsInf(f, 0) { + v.notNumber = fmt.Sprintf("%q is not a number", text) + } else { + v = bounded(f, f, f == math.Trunc(f)) + } + if _, err := strconv.ParseInt(text, 10, 64); !integerText.MatchString(text) { + v.not[DataTypeInteger] = fmt.Sprintf("%q is not an integer", text) + } else if err != nil { + v.not[DataTypeInteger] = fmt.Sprintf("%q is past the int64 range", text) + } + if v.notNumber != "" || !numberText.MatchString(text) { + v.not[DataTypeNumber] = fmt.Sprintf("%q is not a number", text) + } + if text != "true" && text != "false" { + v.not[DataTypeBoolean] = fmt.Sprintf("%q is not a boolean", text) + } + return v +} -- 2.52.0 From 4283af196876d0bfbd12267d37b15f759f74f75b Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 15 Sep 2026 14:39:20 +0200 Subject: [PATCH 05/10] Tests: a composed calc operand is refused naming digits among the spellings --- datatype_test.go | 1 + 1 file changed, 1 insertion(+) diff --git a/datatype_test.go b/datatype_test.go index e65f711..995baf2 100644 --- a/datatype_test.go +++ b/datatype_test.go @@ -65,6 +65,7 @@ func TestDatatypeRejectsAValueItsTypeRejects(t *testing.T) { {"a divisor that can be zero", `{"format":"{calc(a / b)}","a":"{int(1,9)}","b":"{int(0,9)}","datatype":"number"}`, "divides by b, which is not proven nonzero"}, {"an overflow", `{"format":"{calc(a * a)}","a":"{digits(200)}","datatype":"number"}`, "is not proven within 1e300"}, {"a division in an integer column", `{"format":"{calc(a / b)}","a":"{int(1,9)}","b":"{int(1,9)}","datatype":"integer"}`, "prints a number, not an integer"}, + {"a composed operand", `{"format":"{calc(n * 2)}","n":"{int(1,99)}.{digits(2)}","datatype":"number"}`, "{seq()}, {digits()} or {calc()}"}, } { row := `{"format":"","col":` + c.column + `}` files := map[string]string{"row": row} -- 2.52.0 From 8ca887aa7d6e75a9aafddb9a793a1272306d4e05 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 15 Sep 2026 14:39:25 +0200 Subject: [PATCH 06/10] Name digits among an operand's spellings, and document number by what its text is --- template.go | 2 +- value.go | 8 +++++++- 2 files changed, 8 insertions(+), 2 deletions(-) diff --git a/template.go b/template.go index 0b9e76b..1e5d3e6 100644 --- a/template.go +++ b/template.go @@ -73,7 +73,7 @@ type builtin struct { // for a builtin that reads none. operands func(args []string) []string // number bounds the number a call prints and names the datatype its text is; nil - // for a builtin that prints text. + // for a builtin whose text is no number. number func(args []string) (proven, DataType) } diff --git a/value.go b/value.go index 56c3d93..5d580b5 100644 --- a/value.go +++ b/value.go @@ -92,7 +92,9 @@ func (p *valueProof) template(t *template) proven { case t.fixed: return literalValue(t.lit) case len(t.ops) != 1: - return unproven(fmt.Sprintf("%q is not one value; write one literal or one {int()}, {float()}, {seq()} or {calc()}, or read one", t.format)) + v := unproven(notOneValue(t.format, "{int()}, {float()}, {seq()} or {calc()}")) + v.notNumber = notOneValue(t.format, "{int()}, {float()}, {seq()}, {digits()} or {calc()}") + return v } body := t.format[1 : len(t.format)-1] name, args, isFunc := funcCall(body) @@ -227,6 +229,10 @@ func printing(token string, prints DataType, v proven) proven { return v } +func notOneValue(format, calls string) string { + return fmt.Sprintf("%q is not one value; write one literal or one %s, or read one", format, calls) +} + // unproven is a render no datatype and no calc can take, for why. func unproven(why string) proven { v := proven{notNumber: why} -- 2.52.0 From 1ad9418a760f75eb155f9c3478e699cd12dbfaf9 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 15 Sep 2026 15:08:14 +0200 Subject: [PATCH 07/10] Tests: a printed zero is no divisor and prints unsigned, int64 refusals say so, and CSVLine allocates twice --- builtins_test.go | 3 +++ calc_test.go | 2 ++ datatype_test.go | 17 ++++++++++++++++- one_spelling_test.go | 1 + perf_test.go | 4 ++++ record_test.go | 12 ++++++++---- 6 files changed, 34 insertions(+), 5 deletions(-) diff --git a/builtins_test.go b/builtins_test.go index 7a63509..31b2f72 100644 --- a/builtins_test.go +++ b/builtins_test.go @@ -69,6 +69,9 @@ func TestBuiltinFloat(t *testing.T) { if got := mustRender(t, f, `"{float(1,2,3)}"`); !re.MatchString(got) { t.Fatalf("float(1,2,3) = %q, want d.ddd in [1,2]", got) } + if got := mustRender(t, f, `"{float(-1,1,0)}"`); got == "-0" { + t.Fatalf("float(-1,1,0) = %q, want a zero printed unsigned", got) + } } } diff --git a/calc_test.go b/calc_test.go index 77adf77..7a8b7bd 100644 --- a/calc_test.go +++ b/calc_test.go @@ -36,6 +36,7 @@ func TestCalcAuto(t *testing.T) { `"{calc(10 / 3)}"`: "3.3333333333333335", `"{calc(6 / 2)}"`: "3", `"{calc(1 / 4)}"`: "0.25", + `"{calc(0 * -1)}"`: "0", } for tmpl, want := range cases { if got := mustRender(t, f, tmpl); got != want { @@ -52,6 +53,7 @@ func TestCalcDecimals(t *testing.T) { `"{calc(10 / 3, 2)}"`: "3.33", `"{calc(10 / 3, 0)}"`: "3", `"{calc(2 * 3, 2)}"`: "6.00", + `"{calc(-0.001, 2)}"`: "0.00", } for tmpl, want := range cases { if got := mustRender(t, f, tmpl); got != want { diff --git a/datatype_test.go b/datatype_test.go index 995baf2..f2955cf 100644 --- a/datatype_test.go +++ b/datatype_test.go @@ -32,6 +32,7 @@ func TestDatatypeAndNullSitOnlyInAColumn(t *testing.T) { `null`: `so write ""`, `{"format":"{p}","p":{"format":"{x}","x":[null,"a"]}}`: `so write ""`, `{"format":"","c":[{"format":"1","datatype":"integer"},"x"]}`: `write it as {"format":"x","datatype":"integer"}`, + `{"format":"","c":[{"format":"1","datatype":"integer"},{"format":"2","weight":3}]}`: `give it "datatype": "integer"`, `{"format":"","c":[{"format":"1","datatype":"integer"},{"format":"true","datatype":"boolean"}]}`: "a column holds one datatype", } { if _, err := compile(parse(t, src)); err == nil || !strings.Contains(err.Error(), want) { @@ -66,6 +67,17 @@ func TestDatatypeRejectsAValueItsTypeRejects(t *testing.T) { {"an overflow", `{"format":"{calc(a * a)}","a":"{digits(200)}","datatype":"number"}`, "is not proven within 1e300"}, {"a division in an integer column", `{"format":"{calc(a / b)}","a":"{int(1,9)}","b":"{int(1,9)}","datatype":"integer"}`, "prints a number, not an integer"}, {"a composed operand", `{"format":"{calc(n * 2)}","n":"{int(1,99)}.{digits(2)}","datatype":"number"}`, "{seq()}, {digits()} or {calc()}"}, + {"a divisor that prints as zero", `{"format":"{calc(1 / b, 2)}","b":"{float(4.9999999999999994e-79,5e-79,78)}","datatype":"number"}`, "divides by b, which is not proven nonzero"}, + {"a divisor that rounds to zero", `{"format":"{calc(a / b)}","a":"{int(1,9)}","b":"{float(0.1,1,0)}","datatype":"number"}`, "divides by b, which is not proven nonzero"}, + {"a zero among a divisor's literals", `{"format":"{calc(a / b)}","a":"{int(1,9)}","b":["0","5"],"datatype":"number"}`, "divides by b, which is not proven nonzero"}, + {"a negated divisor crossing zero", `{"format":"{calc(a / (-b + 10))}","a":"{int(1,9)}","b":"{int(1,20)}","datatype":"number"}`, "which is not proven nonzero"}, + {"a subtracted divisor crossing zero", `{"format":"{calc(a / (10 - b))}","a":"{int(1,9)}","b":"{int(1,20)}","datatype":"number"}`, "which is not proven nonzero"}, + {"a quotient past the limit", `{"format":"{calc(a / b / b)}","a":"{digits(300)}","b":"{float(0.000001,1,6)}","datatype":"number"}`, "is not proven within 1e300"}, + {"an operand past the limit", `{"format":"{calc(a)}","a":"{digits(400)}","datatype":"number"}`, "is not proven within 1e300"}, + {"a whole calc past int64", `{"format":"{calc(a * 2)}","a":"{seq()}","datatype":"integer"}`, "{calc(a * 2)} is not proven within int64"}, + {"a whole float past int64", `{"format":"{float(0,1e19,0)}","datatype":"integer"}`, "{float(0,1e19,0)} is not proven within int64"}, + {"a signed zero integer", `{"format":"-0","datatype":"integer"}`, `"-0" is zero written with a sign; write "0"`}, + {"a signed zero number", `{"format":"-0.00","datatype":"number"}`, `"-0.00" is zero written with a sign; write "0.00"`}, } { row := `{"format":"","col":` + c.column + `}` files := map[string]string{"row": row} @@ -83,7 +95,7 @@ func TestDatatypeRejectsAValueItsTypeRejects(t *testing.T) { } } -var jsonInteger = regexp.MustCompile(`^-?(0|[1-9][0-9]*)$`) +var jsonInteger = regexp.MustCompile(`^(0|-?[1-9][0-9]*)$`) func TestDatatypeAcceptsAColumnThatAlwaysParses(t *testing.T) { cat := `[{"format":"{code}","code":"200"},{"format":"{code}","code":"404"}]` @@ -94,6 +106,9 @@ func TestDatatypeAcceptsAColumnThatAlwaysParses(t *testing.T) { `{"format":"{/cat.code}","datatype":"integer"}`, `{"format":"{a|b}","a":"1","b":"{int(5,9)}","datatype":"integer"}`, `{"format":"{float(-1.5,9.5,0)}","datatype":"integer"}`, + `{"format":"{float(-1,1,0)}","datatype":"integer"}`, + `{"format":"{calc(a * b)}","a":"{int(-9,-1)}","b":"{int(0,9)}","datatype":"integer"}`, + `{"format":"{calc(x + 1)}","x":"{float(0,9,0)}","datatype":"integer"}`, `{"format":"{float(-1,1,2)}","datatype":"number"}`, `{"format":"{v}","v":["1","2.5","6.022e23"],"datatype":"number"}`, `{"format":"{b}","b":["true","false"],"datatype":"boolean"}`, diff --git a/one_spelling_test.go b/one_spelling_test.go index ca1aeac..8253b1c 100644 --- a/one_spelling_test.go +++ b/one_spelling_test.go @@ -18,6 +18,7 @@ func TestRepeatedChoiceItemIsRejected(t *testing.T) { `["a", "a", "b"]`: `{ "format": "a", "weight": 2 }`, `[{"format":"{x}","x":"1"},{"format":"{x}","x":"1"}]`: "repeats item", `{"format":"{w}","w":["", "", "x"]}`: `{ "format": "", "weight": 2 }`, + `{"format":"","w":[null,null,"a"]}`: "a null takes no weight", } { if _, err := compile(parse(t, src)); err == nil || !strings.Contains(err.Error(), want) { t.Errorf("compile(%s) = %v, want an error naming %s", src, err, want) diff --git a/perf_test.go b/perf_test.go index a3b750d..0ee6dcb 100644 --- a/perf_test.go +++ b/perf_test.go @@ -68,6 +68,10 @@ func TestNoRecordAllocRegression(t *testing.T) { if allocs := testing.AllocsPerRun(10000, func() { f.FakeRecord("x") }); allocs > base*1.10 { t.Errorf("%s: %.1f allocs/op regressed past %.1f (baseline %.1f + 10%%); a record fence running per draw is the usual cause", s.name, allocs, base*1.10, base) } + r, _ := f.FakeRecord("x") + if allocs := testing.AllocsPerRun(10000, func() { _ = r.CSVLine() }); allocs > 2 { + t.Errorf("%s: CSVLine() makes %.1f allocs/op, want 2: the fields slice and the joined line", s.name, allocs) + } } } diff --git a/record_test.go b/record_test.go index c5fb717..c57bdec 100644 --- a/record_test.go +++ b/record_test.go @@ -134,7 +134,7 @@ func TestRecordCSV(t *testing.T) { } func TestRecordCSVEmptyValueStaysARow(t *testing.T) { - dir := writeData(t, map[string]string{"blank": `{"format": "", "note": ""}`, "gone": `{"format": "", "note": null}`}) + dir := writeData(t, map[string]string{"blank": `{"format": "", "note": ""}`}) f := newGenerator(t, dir, WithSeed(1)) r, err := f.FakeRecord("blank") if err != nil { @@ -147,9 +147,6 @@ func TestRecordCSVEmptyValueStaysARow(t *testing.T) { if len(rows) != 2 || len(rows[1]) != 1 || rows[1][0] != "" { t.Fatalf("one empty column parsed to %v, want a header and one row of one empty field", rows) } - if r, err := f.FakeRecord("gone"); err != nil || r.CSVLine() != "" { - t.Fatalf("a lone null column wrote %q, %v; want the blank line PostgreSQL's COPY reads as null", r.CSVLine(), err) - } } func TestRecordWritesTypedAndNullColumns(t *testing.T) { @@ -183,6 +180,13 @@ func TestRecordWritesTypedAndNullColumns(t *testing.T) { if got := DataTypeNumber.String(); got != "number" { t.Errorf("DataTypeNumber.String() = %q, want the data's spelling", got) } + lone, err := newGenerator(t, writeData(t, map[string]string{"gone": `{"format":"","note":null}`})).FakeRecord("gone") + if err != nil { + t.Fatal(err) + } + if got := lone.CSVLine(); got != "" { + t.Errorf("a lone null column wrote CSVLine() = %q, want the blank line PostgreSQL's COPY reads as null", got) + } } func TestRecordRejectsOverlappingReferenceColumns(t *testing.T) { -- 2.52.0 From 3e7127da71f167d86389fb8dacf831398ebfe403 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 15 Sep 2026 15:08:26 +0200 Subject: [PATCH 08/10] Parse a divisor's rounding half exactly, print zero unsigned, name int64 refusals, and quote CSV fields without a writer --- README.md | 9 +++---- builtins.go | 27 ++++++++++++++------- calc.go | 2 +- datatype.go | 2 +- node.go | 6 +++-- record.go | 20 ++++++++-------- template.go | 6 ++--- value.go | 69 +++++++++++++++++++++++++++++++---------------------- 8 files changed, 83 insertions(+), 58 deletions(-) diff --git a/README.md b/README.md index ae37af6..2cda366 100644 --- a/README.md +++ b/README.md @@ -273,8 +273,8 @@ Writes e.g. `{"id":1,"paid":true,"total":59.97}`. A column is a field of the top template, or an item of a choice standing in for one; `datatype` anywhere else is a load error. A typed column holds one value, alone in its format: a literal, one `{int()}`, `{float()}`, `{seq()}` or `{calc()}` call, or a read that lands only on such -values. `integer` is an int64 written `-?(0|[1-9][0-9]*)` — `{float()}` prints one at -`0` decimals — `number` a JSON number, `boolean` `true` or `false`. A value its +values. `integer` is an int64 written `0|-?[1-9][0-9]*` — `{float()}` prints one at +`0` decimals within int64 — `number` a JSON number, `boolean` `true` or `false`. A value its datatype cannot hold is a load error naming it: ```text @@ -287,7 +287,7 @@ literal, an `{int()}`, `{float()}`, `{seq()}` or `{digits()}` call, a calc, or a such values, whose bounds keep every divisor from zero and the result within `1e300`. What the bounds cannot show is refused — `{calc(a / b)}: divides by b, which is not proven nonzero`. The calc fills an `integer` column at `0` decimals, or over whole -operands with no `/`. +operands with no `/`, while its bounds stay within int64. ### Null @@ -365,7 +365,8 @@ hyphenated field can't be an operand. { "format": "{net} x {qty} = {calc(net * qty, 2)}", "net": ["19.99", "5.00"], "qty": ["3", "7"] } ``` -Renders e.g. `19.99 x 3 = 59.97`. An operand that can never be a number (`"abc"`, +Renders e.g. `19.99 x 3 = 59.97`. A result that rounds to zero prints unsigned — `0`, +`0.00` — as `{float()}`'s does. An operand that can never be a number (`"abc"`, or a choice of such) is rejected at load, as is a division by a constant zero (`1/0`, or a fixed `"0"` field); an operand that sometimes is not a number yields `NaN`, and a division by one that is not constant `Inf` — both print rather than diff --git a/builtins.go b/builtins.go index 833c2e2..6075668 100644 --- a/builtins.go +++ b/builtins.go @@ -31,8 +31,8 @@ var builtins = map[string]builtin{ "ulid": {arity: 0, prep: sample(ulid)}, "nanoid": {arity: 1, check: posIntArg, prep: chars(nanoidAlphabet)}, "hex": {arity: 1, check: posIntArg, prep: chars(hexDigits)}, - "digits": {arity: 1, check: posIntArg, prep: chars("0123456789"), number: func(a []string) (proven, DataType) { - return bounded(0, math.Pow(10, float64(atoi(a[0])))-1, true), DataTypeString + "digits": {arity: 1, check: posIntArg, prep: chars("0123456789"), number: func(token string, a []string) proven { + return printing(token, DataTypeString, bounded(0, math.Pow(10, float64(atoi(a[0])))-1, true)) }}, "upper": {arity: 1, check: posIntArg, prep: chars("ABCDEFGHIJKLMNOPQRSTUVWXYZ")}, "lower": {arity: 1, check: posIntArg, prep: chars("abcdefghijklmnopqrstuvwxyz")}, @@ -45,16 +45,16 @@ var builtins = map[string]builtin{ "int": {arity: 2, check: intRangeArgs, prep: func(a []string) callFn { lo, span := atoi(a[0]), atoi(a[1])-atoi(a[0])+1 return func(s *session, _ string, _ []string) string { return strconv.Itoa(lo + s.IntN(span)) } - }, number: func(a []string) (proven, DataType) { - return bounded(float64(atoi(a[0])), float64(atoi(a[1])), true), DataTypeInteger + }, number: func(token string, a []string) proven { + return printing(token, DataTypeInteger, bounded(float64(atoi(a[0])), float64(atoi(a[1])), true)) }}, "float": {arity: 3, check: floatArgs, prep: func(a []string) callFn { lo, hi, dp := atof(a[0]), atof(a[1]), atoi(a[2]) return func(s *session, _ string, _ []string) string { - return strconv.FormatFloat(lo+s.Float64()*(hi-lo), 'f', dp, 64) + return formatFloat(lo+s.Float64()*(hi-lo), dp) } - }, number: func(a []string) (proven, DataType) { - return printedNumber(bounded(atof(a[0]), atof(a[1]), false), atoi(a[2])) + }, number: func(token string, a []string) proven { + return printedNumber(token, bounded(atof(a[0]), atof(a[1]), false), atoi(a[2])) }}, "iban": {arity: 1, check: ibanArg, prep: func(a []string) callFn { cc := a[0] @@ -75,8 +75,8 @@ var builtins = map[string]builtin{ return func(s *session, _ string, _ []string) string { return strconv.FormatUint(s.next(key), 10) } - }, number: func([]string) (proven, DataType) { - return bounded(1, math.MaxInt64, true), DataTypeInteger + }, number: func(token string, _ []string) proven { + return printing(token, DataTypeInteger, bounded(1, math.MaxInt64, true)) }}, } @@ -104,6 +104,15 @@ func chars(alphabet string) func([]string) callFn { const hexDigits = "0123456789abcdef" +// formatFloat prints v to dp decimals, -1 for the shortest form, and a zero unsigned. +func formatFloat(v float64, dp int) string { + s := strconv.FormatFloat(v, 'f', dp, 64) + if strings.HasPrefix(s, "-") && strings.Trim(s, "-0.") == "" { + return s[1:] + } + return s +} + // transforms are the builtins that rewrite one operand's value; they nest, so // {lowercase(ascii(x))} folds then lowers. var transforms = map[string]func(string) string{ diff --git a/calc.go b/calc.go index 901e8c9..eb0c2c4 100644 --- a/calc.go +++ b/calc.go @@ -195,7 +195,7 @@ func calcPrep(args []string) callFn { placed := indexVars(expr, at) dp := calcDecimals(args) return func(_ *session, _ string, operands []string) string { - return strconv.FormatFloat(placed.eval(operands), 'f', dp, 64) + return formatFloat(placed.eval(operands), dp) } } diff --git a/datatype.go b/datatype.go index 94873de..03bdb7f 100644 --- a/datatype.go +++ b/datatype.go @@ -100,7 +100,7 @@ func disagreement(a, b *template) error { switch { case bare.datatype != DataTypeString: return fmt.Errorf("its items declare %s and %s; a column holds one datatype", a.datatype, b.datatype) - case len(bare.fields) == 0 && bare.repeat == 1: + case bare.fields == nil: // a JSON string; an object, which may carry a weight, has a fields map return fmt.Errorf(`item %q declares no datatype, and a column holds one; write it as {"format":%q,"datatype":%q}`, bare.format, bare.format, typed.datatype) } return fmt.Errorf(`an item declares no datatype beside one declaring %s; a column holds one, so give it "datatype": %q`, typed.datatype, typed.datatype) diff --git a/node.go b/node.go index e73de11..cef1bc7 100644 --- a/node.go +++ b/node.go @@ -32,8 +32,7 @@ type choice struct { func (*choice) isNode() {} -// null is a record column's missing value, rendered as "". It is not zero-sized, so two -// nulls are two map keys. +// null is a column's missing value, rendered ""; sized so two nulls are two map keys. type null struct{ _ byte } func (*null) isNode() {} @@ -216,6 +215,9 @@ func checkNoRepeatedItem(items []any) error { if s, isString := raw.(string); isString { return fmt.Errorf("choice item %q is repeated; skew the odds with a weight instead: { \"format\": %q, \"weight\": 2 }", s, s) } + if raw == nil { + return fmt.Errorf("choice item %d repeats null; a null takes no weight, so weight the other items instead", i) + } return fmt.Errorf("choice item %d repeats item %d; skew the odds with a weight on one of them instead", i, j) } seen[string(key)] = i diff --git a/record.go b/record.go index 2658343..84bf8dd 100644 --- a/record.go +++ b/record.go @@ -1,12 +1,13 @@ package fejkdata import ( - "encoding/csv" "encoding/json" "errors" "fmt" "sort" "strings" + "unicode" + "unicode/utf8" ) // Column is one rendered column of a record. Value is the rendered text, which a @@ -73,15 +74,16 @@ func (r *Record) CSVLine() string { return strings.Join(fields, ",") } +// csvField quotes a field where encoding/csv would, and an empty one too. func csvField(s string) string { - if s == "" { + first, _ := utf8.DecodeRuneInString(s) + switch { + case s == "": return `""` + case s == `\.` || strings.ContainsAny(s, "\",\r\n") || unicode.IsSpace(first): + return `"` + strings.ReplaceAll(s, `"`, `""`) + `"` } - var b strings.Builder - w := csv.NewWriter(&b) - _ = w.Write([]string{s}) - w.Flush() - return strings.TrimSuffix(b.String(), "\n") + return s } // SQLInsert renders the record as one INSERT statement into table, identifiers in ANSI @@ -104,9 +106,7 @@ func quoteIdent(s string) string { return `"` + strings.ReplaceAll(s, `"`, `""`) + `"` } -// literal spells a column the way a serializer writes it: quoted for a string, bare for -// any other datatype, whose every render the load check proved a literal, and nullText -// for a null. +// literal writes a string column quoted, a proven typed one bare, and a null as nullText. func literal(c Column, quote func(string) string, nullText string) string { switch { case c.Null: diff --git a/template.go b/template.go index 1e5d3e6..45bfcda 100644 --- a/template.go +++ b/template.go @@ -72,9 +72,9 @@ type builtin struct { // operands names the fields the call reads, which expand renders for it; nil // for a builtin that reads none. operands func(args []string) []string - // number bounds the number a call prints and names the datatype its text is; nil - // for a builtin whose text is no number. - number func(args []string) (proven, DataType) + // number proves what a call prints, token its body: the bounds of its number and the + // datatype of its text; nil for a builtin whose text is no number. + number func(token string, args []string) proven } // funcCall splits a "{token}" body shaped name(args) into its parts; ok is false diff --git a/value.go b/value.go index 5d580b5..6d19dc7 100644 --- a/value.go +++ b/value.go @@ -11,16 +11,14 @@ import ( // proven is what a proof knows of every render of a node: bounds on the number each // reads as, and per datatype why some render's text is not one ("" when none). type proven struct { - lo, hi float64 - nonZero float64 // every value is at least this far from zero; 0 when one can be zero - integral bool - notNumber string // why some render reads as no finite number, the way calc reads it - not [len(dataTypeNames)]string + lo, hi float64 + nonZero float64 // every value is at least this far from zero; 0 when one can be zero + integral bool + notOperand string // why some render reads as no finite number, the way calc reads it + not [len(dataTypeNames)]string } -// valueProof proves what typed columns and their calc operands hold, each node once per -// scope. A typed column holds one value: a literal, one value builtin, one calc, or a -// read of such values. +// valueProof proves what typed columns and their calc operands hold, each node once per scope. type valueProof struct { memo map[node]proven } @@ -71,8 +69,8 @@ func (p *valueProof) unite(nodes []node) proven { w := p.of(n) v.lo, v.hi, v.nonZero = min(v.lo, w.lo), max(v.hi, w.hi), min(v.nonZero, w.nonZero) v.integral = v.integral && w.integral - if v.notNumber == "" { - v.notNumber = w.notNumber + if v.notOperand == "" { + v.notOperand = w.notOperand } for d := range v.not { if v.not[d] == "" { @@ -93,7 +91,7 @@ func (p *valueProof) template(t *template) proven { return literalValue(t.lit) case len(t.ops) != 1: v := unproven(notOneValue(t.format, "{int()}, {float()}, {seq()} or {calc()}")) - v.notNumber = notOneValue(t.format, "{int()}, {float()}, {seq()}, {digits()} or {calc()}") + v.notOperand = notOneValue(t.format, "{int()}, {float()}, {seq()}, {digits()} or {calc()}") return v } body := t.format[1 : len(t.format)-1] @@ -108,12 +106,11 @@ func (p *valueProof) template(t *template) proven { case name == "calc": return p.calc(t, body, args) case builtins[name].number != nil: - v, prints := builtins[name].number(args) - return printing(body, prints, v) + return builtins[name].number(body, args) case isTransform: return unproven(fmt.Sprintf("{%s} rewrites text rather than printing a value; write the values it would print", body)) } - return printing(body, DataTypeString, proven{notNumber: fmt.Sprintf("{%s} prints text, not a number", body)}) + return printing(body, DataTypeString, proven{notOperand: fmt.Sprintf("{%s} prints text, not a number", body)}) } func (p *valueProof) calc(t *template, body string, args []string) proven { @@ -128,8 +125,7 @@ func (p *valueProof) calc(t *template, body string, args []string) proven { if doubt != "" { return unproven(fmt.Sprintf("{%s}: %s", body, doubt)) } - v, prints := printedNumber(v, calcDecimals(args)) - return printing(body, prints, v) + return printedNumber(body, v, calcDecimals(args)) } // calcLimit is the largest magnitude a proof accepts as finite, far enough below @@ -144,8 +140,8 @@ func (p *valueProof) expr(n calcNode, fields map[string]node) (proven, string) { return bounded(v, v, v == math.Trunc(v)), "" case calcVar: v := p.of(fields[string(n)]) - if v.notNumber != "" { - return proven{}, fmt.Sprintf("operand %q: %s", string(n), v.notNumber) + if v.notOperand != "" { + return proven{}, fmt.Sprintf("operand %q: %s", string(n), v.notOperand) } return proven{lo: v.lo, hi: v.hi, nonZero: v.nonZero, integral: v.integral}, "" case calcNeg: @@ -205,17 +201,21 @@ func bounded(lo, hi float64, integral bool) proven { func magnitude(v proven) float64 { return math.Max(math.Abs(v.lo), math.Abs(v.hi)) } -// printedNumber is v once strconv.FormatFloat prints it to dp decimals, and the datatype -// that text is: an integer when whole and within int64, else a number. -func printedNumber(v proven, dp int) (proven, DataType) { +// printedNumber is what a token printing v to dp decimals holds: an integer when whole +// and within int64, else a number. +func printedNumber(token string, v proven, dp int) proven { if dp >= 0 { - half := math.Pow(10, -float64(dp)) / 2 + half, _ := strconv.ParseFloat("5e-"+strconv.Itoa(dp+1), 64) // math.Pow(10, -dp) can land below the tie and let a printed zero through v = proven{lo: v.lo - half, hi: v.hi + half, nonZero: math.Max(0, v.nonZero-half), integral: v.integral || dp == 0} } - if (dp == 0 || dp < 0 && v.integral) && magnitude(v) < math.MaxInt64 { - return v, DataTypeInteger + if dp != 0 && !(dp < 0 && v.integral) { + return printing(token, DataTypeNumber, v) } - return v, DataTypeNumber + v = printing(token, DataTypeInteger, v) + if !(magnitude(v) < math.MaxInt64) { + v.not[DataTypeInteger] = fmt.Sprintf("{%s} is not proven within int64", token) + } + return v } // printing is v for a token whose every render is text of datatype prints, with a reason @@ -235,7 +235,7 @@ func notOneValue(format, calls string) string { // unproven is a render no datatype and no calc can take, for why. func unproven(why string) proven { - v := proven{notNumber: why} + v := proven{notOperand: why} for d := DataTypeInteger; d <= DataTypeBoolean; d++ { v.not[d] = why } @@ -251,7 +251,7 @@ var ( func literalValue(text string) proven { var v proven if f, err := strconv.ParseFloat(strings.TrimSpace(text), 64); err != nil || math.IsNaN(f) || math.IsInf(f, 0) { - v.notNumber = fmt.Sprintf("%q is not a number", text) + v.notOperand = fmt.Sprintf("%q is not a number", text) } else { v = bounded(f, f, f == math.Trunc(f)) } @@ -260,11 +260,24 @@ func literalValue(text string) proven { } else if err != nil { v.not[DataTypeInteger] = fmt.Sprintf("%q is past the int64 range", text) } - if v.notNumber != "" || !numberText.MatchString(text) { + if v.notOperand != "" || !numberText.MatchString(text) { v.not[DataTypeNumber] = fmt.Sprintf("%q is not a number", text) } if text != "true" && text != "false" { v.not[DataTypeBoolean] = fmt.Sprintf("%q is not a boolean", text) } + return signedZero(text, v) +} + +// signedZero refuses a zero written with a sign as a typed value, naming it unsigned. +func signedZero(text string, v proven) proven { + if !strings.HasPrefix(text, "-") || v.notOperand != "" || v.lo != 0 { + return v + } + for _, d := range []DataType{DataTypeInteger, DataTypeNumber} { + if v.not[d] == "" { + v.not[d] = fmt.Sprintf("%q is zero written with a sign; write %q", text, text[1:]) + } + } return v } -- 2.52.0 From 08949eb72cbfeacbaa3e1568dab3046ad1ee0e97 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 15 Sep 2026 15:17:33 +0200 Subject: [PATCH 09/10] Tests: CSV quotes a COPY marker, a leading space and a bare CR, and a sub-float literal is no signed zero --- datatype_test.go | 1 + record_test.go | 10 ++++++++++ 2 files changed, 11 insertions(+) diff --git a/datatype_test.go b/datatype_test.go index f2955cf..a537991 100644 --- a/datatype_test.go +++ b/datatype_test.go @@ -111,6 +111,7 @@ func TestDatatypeAcceptsAColumnThatAlwaysParses(t *testing.T) { `{"format":"{calc(x + 1)}","x":"{float(0,9,0)}","datatype":"integer"}`, `{"format":"{float(-1,1,2)}","datatype":"number"}`, `{"format":"{v}","v":["1","2.5","6.022e23"],"datatype":"number"}`, + `{"format":"{v}","v":["-1e-400","0.5"],"datatype":"number"}`, `{"format":"{b}","b":["true","false"],"datatype":"boolean"}`, `{"format":"{calc(net * qty, 2)}","net":["19.99","5.00"],"qty":["3","7"],"datatype":"number"}`, `{"format":"{calc(a + b)}","a":"{int(1,9)}","b":"{int(-9,9)}","datatype":"integer"}`, diff --git a/record_test.go b/record_test.go index c57bdec..0301804 100644 --- a/record_test.go +++ b/record_test.go @@ -131,6 +131,16 @@ func TestRecordCSV(t *testing.T) { if len(seen) != 4 { t.Fatalf("round-tripped %d of the 4 values; the comma, quote and newline shapes must each survive", len(seen)) } + for value, want := range map[string]string{`\.`: `"\."`, " x": `" x"`, " x": "\" x\"", "a\rb": "\"a\rb\""} { + body, _ := json.Marshal(map[string]string{"format": "", "v": value}) + r, err := newGenerator(t, writeData(t, map[string]string{"q": string(body)})).FakeRecord("q") + if err != nil { + t.Fatal(err) + } + if got := r.CSVLine(); got != want { + t.Errorf("CSVLine() of %q = %q, want %q, quoted where encoding/csv quotes", value, got, want) + } + } } func TestRecordCSVEmptyValueStaysARow(t *testing.T) { -- 2.52.0 From 88143153eda09f6e842d56c59879ee18971bf399 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 15 Sep 2026 15:17:38 +0200 Subject: [PATCH 10/10] Read a signed zero off the mantissa, and drop the comment on the unused Pow --- value.go | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/value.go b/value.go index 6d19dc7..d8aa48c 100644 --- a/value.go +++ b/value.go @@ -205,7 +205,7 @@ func magnitude(v proven) float64 { return math.Max(math.Abs(v.lo), math.Abs(v.hi // and within int64, else a number. func printedNumber(token string, v proven, dp int) proven { if dp >= 0 { - half, _ := strconv.ParseFloat("5e-"+strconv.Itoa(dp+1), 64) // math.Pow(10, -dp) can land below the tie and let a printed zero through + half, _ := strconv.ParseFloat("5e-"+strconv.Itoa(dp+1), 64) v = proven{lo: v.lo - half, hi: v.hi + half, nonZero: math.Max(0, v.nonZero-half), integral: v.integral || dp == 0} } if dp != 0 && !(dp < 0 && v.integral) { @@ -271,7 +271,8 @@ func literalValue(text string) proven { // signedZero refuses a zero written with a sign as a typed value, naming it unsigned. func signedZero(text string, v proven) proven { - if !strings.HasPrefix(text, "-") || v.notOperand != "" || v.lo != 0 { + mantissa, _, _ := strings.Cut(strings.ToLower(text), "e") + if !strings.HasPrefix(text, "-") || v.notOperand != "" || strings.Trim(mantissa, "-0.") != "" { return v } for _, d := range []DataType{DataTypeInteger, DataTypeNumber} { -- 2.52.0