diff --git a/README.md b/README.md index ed89bcb..de34fa8 100644 --- a/README.md +++ b/README.md @@ -310,6 +310,12 @@ order.id: datatype integer: {digits(3)} prints text, not an integer order.id: datatype integer: "1{digits(2)}" is not one value; write one literal or one {int()}, {float()}, {seq()} or {calc()}, or read one ``` +A column of one reference alone to another record's column — `"score": "{/src.score}"` +— is that column: it takes the column's datatype and is null where the column is, and +a struct field tagged `src.score` is nil there. A `datatype` of its own types the +column's values where they prove it, and one restating the datatype it takes is refused. +Any other read renders the column's text, a null as `""`. + A typed column's `{calc()}` must be proven to print a number: each operand a number literal, an `{int()}`, `{float()}`, `{seq()}` or `{digits()}` call, a calc, or a read of such values, whose bounds keep every divisor from zero and the result within `1e300`. @@ -330,7 +336,7 @@ renders a null as `""`. The other items' weights skew its odds: ``` `deleted_at` is null every draw, `middle` a name three draws in four. Rejected at -load: `null` anywhere but a column, naming `""`, and a column whose items declare +load: `null` anywhere but a column, naming `""`, and a column whose items hold different datatypes. ### Options and fields @@ -543,10 +549,11 @@ tokens add cost in proportion to the output. tag of one reference alone, `{/users}`, is refused naming the path `users`: both render the same text, and only the path names a record. `IsTemplate` exports the rule, so the CLI, struct tags and any other caller read one. -- **An inline template skips the cycle fence, and only that one.** `New` proves the - loaded tree acyclic, an inline node is a finite tree of its own, and nothing in - the tree can reference it, so no render of it reaches itself. Every other fence - runs over both, from one `checkScope`. +- **An inline template skips the cycle fence.** `New` proves the loaded tree + acyclic, an inline node is a finite tree of its own, and nothing in the tree can + reference it, so no render of it reaches itself. Every other fence runs over both, + from one `checkScope`, except that struct tags leave column agreement to their Go + types. - **An inline template that does not compile is misuse (exit 2), including a reference that resolves to nothing** — the whole argument is the spelling under test, and `NewTemplate` compiles, links and validates as one step. An unknown @@ -621,7 +628,8 @@ tokens add cost in proportion to the output. its columns from a struct instead, because a Go caller has already written that schema: the fields name the columns and their types are the datatypes, so a tag says only what to draw, and a `datatype` in it would be a second spelling of the - type. + type. The Go type is a struct column's one datatype, so its items need not agree on + one among themselves; each must only hold that type. - **A struct's records follow Go's field access, and compile on first use.** The fields an embedded struct promotes are the struct's own — `e.First`, as `encoding/json` and SQL mappers read them — so they are columns of its record and @@ -663,6 +671,11 @@ tokens add cost in proportion to the output. - **A typed column holds one value, not composed text.** Its bounds come from a literal or a call's arguments, so a load error names a real value, a range check is one comparison, and `1{digits(2)}` is a second spelling of `{int(100,199)}`. +- **A column of one reference alone is the column it reads.** `{/src.score}` renders + exactly what `src.score` draws, so it takes that column's datatype and null rather + than restating them, and a `datatype` restating the one it takes is a second + spelling. Any other `datatype` still types the values — the one way to type a column + someone else wrote. - **A typed column's calc is refused unless proven.** Operand bounds must keep each divisor from zero and the result finite; what they cannot show is refused rather than trusted, since a bare `NaN` breaks the JSON and SQL it lands in. diff --git a/datatype.go b/datatype.go index 6ddc484..36ba1ad 100644 --- a/datatype.go +++ b/datatype.go @@ -64,19 +64,78 @@ func datatypeOf(m map[string]any, pos position) (DataType, error) { return 0, fmt.Errorf(`datatype takes "integer", "number" or "boolean", got %q`, name) } -// columnDatatype is the datatype a column's items declare. They must agree, since a +// checkColumns rejects a record column whose items hold different datatypes, checking a column +// after the columns its items read, so a column read is named before its readers. +func checkColumns(s nodeScope) error { + checked := map[node]bool{} + var check func(path, name string, column node) error + check = func(path, name string, column node) error { + if checked[column] { + return nil + } + checked[column] = true + items, _ := columnItems(column) + for _, it := range items { + if r := it.readsColumn; r != nil { + if err := check(r.a.key[1:], r.a.tail[0], r.column); err != nil { + return err + } + } + } + if _, err := columnDatatype(column); err != nil { + return fmt.Errorf("%s: field %q: %w", path, name, err) + } + return nil + } + return s(func(path string, n node) error { + t, ok := n.(*template) + if !ok || !t.record { + return nil + } + for _, name := range recordColumns(t) { + if err := check(path, name, t.fields[name]); err != nil { + return err + } + } + return nil + }) +} + +// columnDatatype is the datatype a column's items hold. They must agree, since a // column holds one; a column only ever null is a string. func columnDatatype(n node) (DataType, error) { items, _ := columnItems(n) if len(items) == 0 { return DataTypeString, nil } + first := itemDatatype(items[0]) for _, t := range items[1:] { - if t.datatype != items[0].datatype { - return items[0].datatype, disagreement(items[0], t) + if d := itemDatatype(t); d != first { + return first, disagreement(items[0], first, t, d) } } - return items[0].datatype, nil + return first, nil +} + +// itemDatatype is the datatype a column item declares, else that of the column it is. +func itemDatatype(t *template) DataType { + if t.datatype != DataTypeString { + return t.datatype + } + return readDatatype(t) +} + +// readDatatype is the datatype of the column t is, reading it by one reference alone; a string +// when t reads none. +func readDatatype(t *template) DataType { + if t.readsColumn == nil { + return DataTypeString + } + items, _ := columnItems(t.readsColumn.column) + if len(items) == 0 { + return DataTypeString + } + return itemDatatype(items[0]) // checkColumns refuses that column where its items disagree } // columnItems is a column's template items, its choices unwrapped, and whether one is null. @@ -98,17 +157,53 @@ func columnItems(n node) (items []*template, nullable bool) { return items, nullable } -// disagreement names the fix for two items of one column declaring different datatypes. -func disagreement(a, b *template) error { - typed, bare := a, b - if typed.datatype == DataTypeString { - typed, bare = b, a +// disagreement names the fix for two items of one column holding different datatypes. +func disagreement(a *template, da DataType, b *template, db DataType) error { + bare, typed, want := b, a, da + if da == DataTypeString { + bare, typed, want = a, b, db } switch { - case bare.datatype != DataTypeString: - return fmt.Errorf("its items declare %s and %s; a column holds one datatype", a.datatype, b.datatype) - case bare.fields == nil: // a JSON string; an object, which may carry a weight, has a fields map - return fmt.Errorf(`item %q declares no datatype, and a column holds one; write it as {"format":%q,"datatype":%q}`, bare.format, bare.format, typed.datatype) + case da != DataTypeString && db != DataTypeString: + return bothTyped(a, da, b, db) + case typed.datatype == DataTypeString && (&valueProof{}).columnItem(bare).not[want] != "": + return fmt.Errorf(`item %q is not %s, the datatype item %q takes from the column it reads; to read that column as text, %s`, bare.format, dataTypeNouns[want], typed.format, asText(typed)) + case bare.fromString: // an object may carry a weight, which this spelling would drop + return fmt.Errorf(`item %q declares no datatype, and a column holds one; write it as {"format":%q,"datatype":%q}`, bare.format, bare.format, want) } - return fmt.Errorf(`an item declares no datatype beside one declaring %s; a column holds one, so give it "datatype": %q`, typed.datatype, typed.datatype) + return fmt.Errorf(`item %q declares no datatype beside one holding %s; a column holds one, so give it "datatype": %q`, bare.format, want, want) +} + +// bothTyped names the fix for two items holding different datatypes: the one taking its datatype +// from the column it reads is typed as the other where its values prove it, else read as text. +func bothTyped(a *template, da DataType, b *template, db DataType) error { + read, other := a, db + if a.datatype != DataTypeString { + read, other = b, da + } + switch { + case read.datatype != DataTypeString: + return fmt.Errorf("its items hold %s and %s; a column holds one datatype", da, db) + case (&valueProof{}).columnItem(read).not[other] == "": + return fmt.Errorf("its items hold %s and %s; a column holds one datatype, so %s", da, db, typedAs(read, other)) + } + return fmt.Errorf("its items hold %s and %s; a column holds one datatype, so to read %q as text, %s", da, db, read.format, asText(read)) +} + +// typedAs names the spelling giving a column-read item datatype d, keeping the other keys an object +// item carries. +func typedAs(t *template, d DataType) string { + if t.fromString { + return fmt.Sprintf(`write %q as {"format":%q,"datatype":%q}`, t.format, t.format, d) + } + return fmt.Sprintf(`give %q "datatype": %q`, t.format, d) +} + +// asText names the spelling that reads the column a column-read item reads as text, keeping the +// other keys an object item carries. +func asText(t *template) string { + if t.fromString { + return fmt.Sprintf(`write {"format":"{text}","text":%q}`, t.format) + } + return fmt.Sprintf(`set its "format" to "{text}" and add "text": %q`, t.format) } diff --git a/datatype_test.go b/datatype_test.go index a537991..9e37149 100644 --- a/datatype_test.go +++ b/datatype_test.go @@ -30,10 +30,7 @@ func TestDatatypeAndNullSitOnlyInAColumn(t *testing.T) { `{"format":"{p}","p":{"format":"{n}","n":{"format":"1","datatype":"integer"}}}`: "datatype only types a record column", `{"format":"{n}","repeat":2,"n":{"format":"1","datatype":"integer"}}`: "datatype only types a record column", `null`: `so write ""`, - `{"format":"{p}","p":{"format":"{x}","x":[null,"a"]}}`: `so write ""`, - `{"format":"","c":[{"format":"1","datatype":"integer"},"x"]}`: `write it as {"format":"x","datatype":"integer"}`, - `{"format":"","c":[{"format":"1","datatype":"integer"},{"format":"2","weight":3}]}`: `give it "datatype": "integer"`, - `{"format":"","c":[{"format":"1","datatype":"integer"},{"format":"true","datatype":"boolean"}]}`: "a column holds one datatype", + `{"format":"{p}","p":{"format":"{x}","x":[null,"a"]}}`: `so write ""`, } { if _, err := compile(parse(t, src)); err == nil || !strings.Contains(err.Error(), want) { t.Errorf("compile(%s) = %v, want an error containing %q", src, err, want) @@ -44,9 +41,22 @@ func TestDatatypeAndNullSitOnlyInAColumn(t *testing.T) { func TestDatatypeRejectsAValueItsTypeRejects(t *testing.T) { tree := map[string]string{ "cat": `[{"format":"{code}","code":"200"},{"format":"{code}","code":"2x"}]`, - "src": `{"format":"","score":[null,{"format":"{int(1,9)}","datatype":"integer"}]}`, + "src": `{"format":"","code":[null,"200","2x"],"flag":{"format":"{b}","b":["true","false"],"datatype":"boolean"},"score":[null,{"format":"{int(1,9)}","datatype":"integer"}]}`, } for _, c := range []struct{ name, column, want string }{ + {"an item beside a typed one", `[{"format":"1","datatype":"integer"},"x"]`, `write it as {"format":"x","datatype":"integer"}`}, + {"a weighted item beside a typed one", `[{"format":"1","datatype":"integer"},{"format":"2","weight":3}]`, `give it "datatype": "integer"`}, + {"items of two datatypes", `[{"format":"1","datatype":"integer"},{"format":"true","datatype":"boolean"}]`, "a column holds one datatype"}, + {"an item beside a typed column it reads", `["{/src.score}","5"]`, `write it as {"format":"5","datatype":"integer"}`}, + {"a typed item beside a string column it reads", `["{/src.code}",{"format":"1","datatype":"integer"}]`, `write it as {"format":"{/src.code}","datatype":"integer"}`}, + {"text beside a typed column it reads", `["{/src.score}","n/a"]`, `to read that column as text, write {"format":"{text}","text":"{/src.score}"}`}, + {"a datatype over a typed column", `{"format":"{/src.score}","datatype":"integer"}`, `{/src.score} takes datatype integer from the column it reads; drop "datatype"`}, + {"a datatype a typed column's values reject", `{"format":"{/src.score}","datatype":"boolean"}`, "{int(1,9)} prints an integer, not a boolean"}, + {"two typed columns it reads", `["{/src.score}","{/src.flag}"]`, `so to read "{/src.score}" as text, write {"format":"{text}","text":"{/src.score}"}`}, + {"a typed column read that holds the datatype declared beside it", `[{"format":"1.5","datatype":"number"},"{/src.score}"]`, `so write "{/src.score}" as {"format":"{/src.score}","datatype":"number"}`}, + {"text beside a weighted typed column read", `[{"format":"{/src.score}","weight":3},"n/a"]`, `to read that column as text, set its "format" to "{text}" and add "text": "{/src.score}"`}, + {"a value of the column it reads", `{"format":"{/src.code}","datatype":"integer"}`, `"2x" is not an integer`}, + {"a null read into text", `{"format":"{x}","x":"{/src.score}","datatype":"integer"}`, "reads a null"}, {"a sample with leading zeros", `{"format":"{digits(3)}","datatype":"integer"}`, "{digits(3)} prints text, not an integer"}, {"a fraction", `{"format":"{v}","v":["1","1.5"],"datatype":"integer"}`, `"1.5" is not an integer`}, {"past int64", `{"format":"9223372036854775808","datatype":"integer"}`, "past the int64 range"}, @@ -55,7 +65,6 @@ func TestDatatypeRejectsAValueItsTypeRejects(t *testing.T) { {"a sign before a sample", `{"format":"-{int(1,9)}","datatype":"integer"}`, "is not one value"}, {"a repeat", `{"format":"{int(1,9)}","repeat":2,"separator":",","datatype":"integer"}`, "carries a repeat"}, {"through a reference", `{"format":"{/cat.code}","datatype":"integer"}`, `"2x" is not an integer`}, - {"a null through a reference", `{"format":"{/src.score}","datatype":"integer"}`, "reads a null"}, {"a bare dot", `{"format":".5","datatype":"number"}`, `".5" is not a number`}, {"a plus sign", `{"format":"+1","datatype":"number"}`, `"+1" is not a number`}, {"a text sample", `{"format":"{hex(4)}","datatype":"number"}`, "{hex(4)} prints text, not a number"}, @@ -93,6 +102,13 @@ func TestDatatypeRejectsAValueItsTypeRejects(t *testing.T) { t.Errorf("%s: NewTemplate = %v, want the inline template refused the same way", c.name, err) } } + _, err := New(WithoutShippedData(), WithDataPath(writeData(t, map[string]string{ + "a": `{"format":"","c":["{/b.x}",{"format":"1","datatype":"integer"}]}`, + "b": `{"format":"","x":["2",{"format":"1","datatype":"integer"}]}`, + }))) + if want := `b: field "x": item "2"`; err == nil || !strings.Contains(err.Error(), want) { + t.Errorf("a column reading one whose items disagree: New = %v, want the column read reported first, containing %q", err, want) + } } var jsonInteger = regexp.MustCompile(`^(0|-?[1-9][0-9]*)$`) @@ -120,9 +136,12 @@ func TestDatatypeAcceptsAColumnThatAlwaysParses(t *testing.T) { `{"format":"{calc(a / (b + 1), 2)}","a":"{int(1,9)}","b":"{digits(2)}","datatype":"number"}`, `{"format":"{calc(a / b, 0)}","a":"{int(1,9)}","b":"{int(1,9)}","datatype":"integer"}`, `{"format":"{calc(sub * 1.25, 2)}","sub":{"format":"{calc(a * b)}","a":"{int(1,9)}","b":"{float(0,5,2)}"},"datatype":"number"}`, + `{"format":"{/src.code}","datatype":"integer"}`, + `{"format":"{/src.n}","datatype":"number"}`, } { row := `{"format":"","col":` + column + `}` - f, err := New(WithoutShippedData(), WithDataPath(writeData(t, map[string]string{"cat": cat, "row": row})), WithSeed(1)) + src := `{"format":"","code":["200","404"],"n":{"format":"{int(1,9)}","datatype":"integer"}}` + f, err := New(WithoutShippedData(), WithDataPath(writeData(t, map[string]string{"cat": cat, "row": row, "src": src})), WithSeed(1)) if err != nil { t.Errorf("%s: New = %v, want it loaded", column, err) continue diff --git a/graph.go b/graph.go index 07fd20f..962a51c 100644 --- a/graph.go +++ b/graph.go @@ -177,11 +177,18 @@ func inlineScope(n node, label string) nodeScope { return func(fn func(path string, m node) error) error { return eachNode(n, label, fn) } } -// checkScope runs the per-node fences over a scope, each over the whole scope +func checkScope(s nodeScope) error { + if err := checkColumns(s); err != nil { + return err + } + return checkRenders(s) +} + +// checkRenders runs the per-node fences over a scope, each over the whole scope // before the next, so which of several broken nodes is reported does not depend on // the walk. It runs after checkNoCycles, whose guarantee is what lets the walks // terminate. -func checkScope(s nodeScope) error { +func checkRenders(s nodeScope) error { mem := reachMemo{} if err := s(func(path string, n node) error { return repeatCheck(path, n, mem) }); err != nil { return err diff --git a/hold.go b/hold.go index 05099df..16b00a8 100644 --- a/hold.go +++ b/hold.go @@ -253,11 +253,17 @@ func checkNoRepeatedRead(format string, c formatOps, refs map[string]refBinding) } // draws is what an expansion has already drawn for its held names: the variant each -// was drawn as, so every path under it reads one row, and the value each read, by -// its one spelling, so the same read written twice reads one value. +// was drawn as, so every path under it reads one row, and the draw each read made, by +// its one spelling, so the same read written twice reads one draw. type draws struct { variant map[string]node - value map[string]string + value map[string]draw +} + +// draw is what one read drew: its text, and whether it landed on a null. +type draw struct { + text string + null bool } // readField renders one arm of a token. An arm's key is a sibling field or a @@ -267,23 +273,18 @@ type draws struct { // gives one value, and a shown operand is the operand computed. Every other name is // drawn afresh, so {word} {word} still draws twice. checkTokens, checkPath and // linkRefs prove every step, so the walk cannot fail. -func readField(s *session, t *template, held, refScope *draws, a arm) string { +func readField(s *session, t *template, held, refScope *draws, a arm) draw { if !t.held[a.key] { if len(a.tail) > 0 { panic(fmt.Sprintf("fejkdata: %q reads a path into %q, which the expansion does not hold", a.name, a.key)) } - return render(s, t.fields[a.key], refScope) + return draw{text: render(s, t.fields[a.key], refScope)} } - // A reference that reads a path reads the caller's scope, so its draw outlives - // this expansion; a sibling, and a reference read whole, stay local to it. - d := held - if isRef(a.key) && refScope != nil && len(a.tail) > 0 { - d = refScope + d := readScope(held, refScope, a) + if r, done := d.value[a.path]; done { + return r } - if v, read := d.value[a.path]; read { - return v - } - var v string + var r draw _ = walkPath(t.fields[a.key], a.tail, pathWalk{ // Hold the draw at every level passed through, so two paths sharing a // prefix share it. @@ -299,10 +300,35 @@ func readField(s *session, t *template, held, refScope *draws, a arm) string { } return []node{n}, nil }, - leaf: func(n node) error { v = render(s, n, refScope); return nil }, + leaf: func(n node) error { r = renderLeaf(s, n, refScope); return nil }, }) - d.value[a.path] = v - return v + d.value[a.path] = r + return r +} + +// readScope is the draws a held read keeps its draw in: for a reference that reads a path, +// refScope, so its draw outlives this expansion; for a sibling, or a reference read whole, held. +func readScope(held, refScope *draws, a arm) *draws { + if isRef(a.key) && refScope != nil && len(a.tail) > 0 { + return refScope + } + return held +} + +// renderLeaf draws and renders what a read lands on: null on a null item, or on a column of one +// reference alone whose read drew null. +func renderLeaf(s *session, n node, scope *draws) draw { + n = drawn(s, n) + if _, isNull := n.(*null); isNull { + return draw{null: true} + } + r := draw{text: render(s, n, scope)} + if t, _ := n.(*template); t != nil && t.readsColumn != nil { + if d := readScope(nil, scope, t.readsColumn.a); d != nil { + r.null = d.value[t.readsColumn.a.path].null + } + } + return r } // drawn resolves a choice to one variant, so a bound head is a concrete node the diff --git a/inline.go b/inline.go index ca693dc..23b592b 100644 --- a/inline.go +++ b/inline.go @@ -32,7 +32,7 @@ func (f *Generator) NewTemplate(input string) (*Template, error) { if err != nil { return nil, fmt.Errorf("fejkdata: %w", err) } - if err := bindInline(n, "template", f.categories); err != nil { + if err := bindInline(n, "template", f.categories, checkScope); err != nil { return nil, fmt.Errorf("fejkdata: %w", err) } return &Template{g: f, n: n}, nil @@ -101,12 +101,8 @@ func loneReference(arg string) (string, bool) { raw = m["format"] } format, isString := raw.(string) - var units []ftoken - if !isString || eachToken(format, func(t ftoken) error { units = append(units, t); return nil }) != nil || len(units) != 1 { - return "", false - } - body := units[0].body - if units[0].kind != 'b' || !isRef(body) || strings.ContainsAny(body, "|(") { + body, lone := loneRef(format) + if !isString || !lone { return "", false } if strings.HasPrefix(body, "/") { @@ -137,14 +133,14 @@ func inputValue(input string) (any, error) { return raw, nil } -// bindInline links an inline node's references against root and runs the fences over it, -// naming its nodes from label. -func bindInline(n node, label string, root map[string]node) error { +// bindInline links an inline node's references against root and runs check over it, naming its +// nodes from label. +func bindInline(n node, label string, root map[string]node, check func(nodeScope) error) error { scope := inlineScope(n, label) if err := linkNodeRefs(scope, root); err != nil { return err } - return checkScope(scope) + return check(scope) } // linkNodeRefs binds the references in an inline node's templates against the diff --git a/node.go b/node.go index 56f84a5..604ea0c 100644 --- a/node.go +++ b/node.go @@ -58,7 +58,10 @@ type template struct { bound map[string]string // held is every name drawn once per expansion: the bound levels above, plus the // siblings a {calc()} reads. nil when the format holds nothing (see expand). - held map[string]bool + held map[string]bool + fromString bool // written as a JSON string rather than an object + readsColumn *columnRead // set when the format is one reference alone reading a record's column + record bool // compiled at the top without a repeat, so its fields are record columns } func (*template) isNode() {} @@ -124,7 +127,7 @@ func compileString(s string) (node, error) { if err := checkTokens(s, nil); err != nil { return nil, err } - t := &template{format: s, repeat: 1} + t := &template{format: s, repeat: 1, fromString: true} if err := t.compileFormat(); err != nil { return nil, err } @@ -231,7 +234,7 @@ func compileTemplate(m map[string]any, pos position) (node, error) { return nil, err } fieldPos := inFormat - if pos == atTop && projectsColumns(o.repeat) { + if pos == atTop && o.repeat == 1 { fieldPos = inColumn } fields, err := compileFields(m, fieldPos) @@ -244,7 +247,7 @@ func compileTemplate(m map[string]any, pos position) (node, error) { if err := checkTokens(o.format, fields); err != nil { return nil, err } - t := &template{format: o.format, fields: fields, repeat: o.repeat, separator: o.separator, datatype: o.datatype} + t := &template{format: o.format, fields: fields, repeat: o.repeat, separator: o.separator, datatype: o.datatype, record: fieldPos == inColumn} if err := t.compileFormat(); err != nil { return nil, err } @@ -307,9 +310,6 @@ func compileFields(m map[string]any, pos position) (map[string]node, error) { return nil, fmt.Errorf("field %w", err) } n, err := compileAt(m[k], pos) - if err == nil && pos == inColumn { - _, err = columnDatatype(n) - } if err != nil { return nil, fmt.Errorf("field %q: %w", k, err) } diff --git a/record.go b/record.go index 84bf8dd..b630526 100644 --- a/record.go +++ b/record.go @@ -205,28 +205,24 @@ func recordOf(n node) (*template, []Column, error) { if !ok { return nil, nil, errors.New("names a choice, not a template; a record is a template whose fields are its columns") } - if !projectsColumns(t.repeat) { - return nil, nil, fmt.Errorf("carries repeat %d, which composes its format into one string; a record projects columns instead — drop the repeat and render the record again for more rows", t.repeat) - } names := recordColumns(t) if len(names) == 0 { return nil, nil, errors.New("has no fields, so no columns") } + if !t.record { + return nil, nil, fmt.Errorf("carries repeat %d, which composes its format into one string; a record projects columns instead — drop the repeat and render the record again for more rows", t.repeat) + } if err := checkColumnRefs(t, names); err != nil { return nil, nil, err } columns := make([]Column, len(names)) for i, name := range names { - datatype, _ := columnDatatype(t.fields[name]) // compile refused a column whose items disagree + datatype, _ := columnDatatype(t.fields[name]) // checkColumns refused items that disagree wherever DataType is read columns[i] = Column{Name: name, DataType: datatype} } return t, columns, nil } -// projectsColumns reports whether a category or inline template with this repeat is a -// record, its fields the columns; a repeat composes the format into one string instead. -func projectsColumns(repeat int) bool { return repeat == 1 } - // checkColumnRefs rejects the reference reads a record's shared draw cannot answer // for: one column rendering a level another reads a path into, and a column // reading the record back through its own path. @@ -297,15 +293,11 @@ func columnRefs(t *template, columns []string) ([]columnRef, error) { // renderRecord draws each column once, in the name order recordOf fixed, over one // reference scope shared across them. func renderRecord(s *session, t *template, columns []Column) *Record { - scope := &draws{variant: map[string]node{}, value: map[string]string{}} + scope := &draws{variant: map[string]node{}, value: map[string]draw{}} r := &Record{columns: append([]Column(nil), columns...)} for i := range r.columns { - n := drawn(s, t.fields[r.columns[i].Name]) - if _, isNull := n.(*null); isNull { - r.columns[i].Null = true - } else { - r.columns[i].Value = render(s, n, scope) - } + column := renderLeaf(s, t.fields[r.columns[i].Name], scope) + r.columns[i].Value, r.columns[i].Null = column.text, column.null } return r } diff --git a/record_test.go b/record_test.go index 0301804..9094172 100644 --- a/record_test.go +++ b/record_test.go @@ -382,6 +382,68 @@ func TestRecordBareReferenceStaysIndependent(t *testing.T) { } } +func TestRecordColumnOfOneReferenceIsTheColumnItReads(t *testing.T) { + dir := writeData(t, map[string]string{ + "mid": `{"format":"","score":"{/src.score}"}`, + "row": `{"format":"","chain":"{/mid.score}","code":{"format":"{/src.code}","datatype":"integer"},"dot":"{.src.score}","label":"n={/src.score}","mixed":[{"format":"{text}","text":"{/src.score}"},"n/a"],"pick":["{/src.score}",{"format":"7","datatype":"integer"}],"same":"{/src.score}","score":"{/src.score}","text":"{/src.code}"}`, + "src": `{"format":"","code":[null,"200","404"],"score":[null,{"format":"{int(1,9)}","datatype":"integer"}]}`, + }) + f := newGenerator(t, dir, WithSeed(1)) + inline, err := f.NewRecordTemplate(`{"format":"","score":"{/src.score}"}`) + if err != nil { + t.Fatal(err) + } + nulls := map[string]int{} + for i := 0; i < 200; i++ { + r, err := f.FakeRecord("row") + if err != nil { + t.Fatal(err) + } + cols := map[string]Column{} + for _, c := range r.Columns() { + cols[c.Name] = c + } + score, code := cols["score"], cols["code"] + agree := func(name string, want Column) bool { + return cols[name].Null == want.Null && cols[name].Value == want.Value + } + switch { + case score.DataType != DataTypeInteger || cols["chain"].DataType != DataTypeInteger || code.DataType != DataTypeInteger || cols["pick"].DataType != DataTypeInteger || cols["label"].DataType != DataTypeString || cols["mixed"].DataType != DataTypeString || cols["text"].DataType != DataTypeString: + t.Fatalf("%s: want score, chain, code and pick integer columns, label, mixed and text string ones", r.JSON()) + case score.Null == (len(score.Value) == 1 && score.Value >= "1" && score.Value <= "9"): + t.Fatalf("score = %+v, want null or a digit from src.score", score) + case code.Null == (code.Value == "200" || code.Value == "404"): + t.Fatalf("code = %+v, want null or a code from src.code", code) + case !agree("same", score) || !agree("chain", score) || !agree("dot", score) || !agree("text", code): + t.Fatalf("%s: want every read of one src column one draw, through mid too", r.JSON()) + case cols["pick"].Value != "7" && !agree("pick", score), cols["mixed"].Null || cols["mixed"].Value != "n/a" && cols["mixed"].Value != score.Value: + t.Fatalf("pick = %+v, mixed = %+v beside score %+v: want pick 7 or the score's draw, mixed n/a or the score's text", cols["pick"], cols["mixed"], score) + case cols["label"].Null || cols["label"].Value != "n="+score.Value: + t.Fatalf("label = %+v beside score %+v, want the text of the read, a null as \"\"", cols["label"], score) + case strings.Contains(r.JSON(), `"score":null`) != score.Null: + t.Fatalf("JSON() = %s, want a null score written null", r.JSON()) + } + ir := inline.Fake() + ic, sqlValue := ir.Columns()[0], ir.Columns()[0].Value + if ic.Null { + sqlValue = "NULL" + } + if ic.DataType != DataTypeInteger || ic.Null == (len(ic.Value) == 1) || ir.SQLInsert("t") != `INSERT INTO "t" ("score") VALUES (`+sqlValue+`);` || ir.CSVLine() != ic.Value { + t.Fatalf("inline score = %+v written %s and %q, want an integer column, NULL and an empty field or a bare digit", ic, ir.SQLInsert("t"), ir.CSVLine()) + } + for name, c := range map[string]Column{"code": code, "inline": ic, "score": score} { + if c.Null { + nulls[name]++ + } + } + } + for _, name := range []string{"code", "inline", "score"} { + if n := nulls[name]; n == 0 || n == 200 { + t.Errorf("%s drew null %d times in 200 records, want both outcomes", name, n) + } + } +} + func TestRecordSQLQuotesIdentifiers(t *testing.T) { dir := writeData(t, map[string]string{ "row": `{"format": "", "postal-code": "1", "street-number": "2"}`, diff --git a/reference.go b/reference.go index ff26770..08f209e 100644 --- a/reference.go +++ b/reference.go @@ -107,9 +107,44 @@ func linkTemplateRefs(folder []string, path string, t *template, root map[string if err := t.compileFormat(); err != nil { return fmt.Errorf("%s: %w", path, err) } + t.readsColumn = columnReadOf(t) return nil } +// columnRead is a record's column read by a format of that one reference alone, which is the +// column: it takes the column's datatype and null. +type columnRead struct { + a arm + column node +} + +func columnReadOf(t *template) *columnRead { + name, lone := loneRef(t.format) + if !lone || t.repeat != 1 { + return nil + } + a := splitArm(name, t.refs) + target, isTemplate := t.fields[a.key].(*template) + if !isTemplate || !target.record || len(a.tail) != 1 { + return nil + } + return &columnRead{a: a, column: target.fields[a.tail[0]]} +} + +// loneRef is the reference a format of one reference token and nothing else reads. +func loneRef(format string) (string, bool) { + units, body := 0, "" + err := eachToken(format, func(t ftoken) error { + units++ + body = t.body + return nil + }) + if err != nil || units != 1 || !isRef(body) || strings.ContainsAny(body, "|(") { + return "", false + } + return body, true +} + // eachTemplate calls fn once per template, with the folder its category sits in // and the dot path reaching it, folders and names in sorted order. func eachTemplate(root map[string]node, fn func(folder []string, path string, t *template) error) error { diff --git a/render.go b/render.go index 330ed0f..ec12628 100644 --- a/render.go +++ b/render.go @@ -103,7 +103,7 @@ func expand(s *session, t *template, refScope *draws) string { if len(t.held) > 0 { held = &draws{ variant: make(map[string]node, len(t.held)), - value: make(map[string]string, len(t.held)), + value: make(map[string]draw, len(t.held)), } } for i := range t.ops { @@ -112,7 +112,7 @@ func expand(s *session, t *template, refScope *draws) string { case 'l': b.WriteString(o.lit) case 'f': - b.WriteString(readField(s, t, held, refScope, o.arms[s.IntN(len(o.arms))])) + b.WriteString(readField(s, t, held, refScope, o.arms[s.IntN(len(o.arms))]).text) case 'b': // Read before the call, so the value a calc computes is the value the // format showed. calcVars fixed the order op.operands holds. @@ -120,7 +120,7 @@ func expand(s *session, t *template, refScope *draws) string { if len(o.operands) > 0 { operands = make([]string, len(o.operands)) for j, a := range o.operands { - operands[j] = readField(s, t, held, refScope, a) + operands[j] = readField(s, t, held, refScope, a).text } } b.WriteString(o.call(s, b.String(), operands)) // b.String() is the output so far diff --git a/struct.go b/struct.go index 727ae75..11d6dad 100644 --- a/struct.go +++ b/struct.go @@ -258,7 +258,7 @@ func (s *structShape) compileRecord(root map[string]node, t reflect.Type, label if err != nil { return fmt.Errorf("%s: %w", label, err) } - if err := bindInline(n, label, root); err != nil { + if err := bindInline(n, label, root, checkRenders); err != nil { return err } record, columns, err := recordOf(n) @@ -269,9 +269,6 @@ func (s *structShape) compileRecord(root map[string]node, t reflect.Type, label s.fields = make([][]int, len(columns)) for i, c := range columns { sf, _ := t.FieldByName(c.Name) - if c.DataType != DataTypeString { - return fmt.Errorf("%s.%s: its Go type %s sets the datatype; drop \"datatype\"", label, c.Name, sf.Type) - } if err := proof.checkField(label+"."+c.Name, sf.Type, record.fields[c.Name]); err != nil { return err } @@ -318,14 +315,19 @@ func (k columnKind) holds(v proven) bool { return v.lo >= k.lo && v.hi <= k.hi } -// checkField rejects a column some render of which a field of Go type ft cannot hold: a null -// outside a pointer, or a value its kind's datatype or range refuses. +// checkField rejects a column a field of Go type ft cannot fill: a datatype, which the Go type +// sets, a null outside a pointer, or a value its kind's datatype or range refuses. func (p *valueProof) checkField(label string, ft reflect.Type, column node) error { - items, nullable := columnItems(column) + items, _ := columnItems(column) + for _, it := range items { + if it.datatype != DataTypeString { + return fmt.Errorf("%s: its Go type %s sets the datatype; drop \"datatype\"", label, ft) + } + } elem := ft if ft.Kind() == reflect.Pointer { elem = ft.Elem() - } else if nullable { + } else if p.column(column).null { return fmt.Errorf("%s: its tag can draw null, which %s cannot hold; make it *%s", label, ft, ft) } kind := columnKinds[elem.Kind()] @@ -333,7 +335,7 @@ func (p *valueProof) checkField(label string, ft reflect.Type, column node) erro return nil } for _, it := range items { - v := p.of(it) + v := p.columnItem(it) if reason := v.not[kind.datatype]; reason != "" { return fmt.Errorf("%s (%s): %s", label, ft, reason) } diff --git a/struct_test.go b/struct_test.go index 7332a32..61f2fcc 100644 --- a/struct_test.go +++ b/struct_test.go @@ -2,6 +2,7 @@ package fejkdata import ( "reflect" + "strconv" "strings" "testing" ) @@ -53,10 +54,68 @@ func structData(t *testing.T) *Generator { return newGenerator(t, writeData(t, map[string]string{ "person": `[{"format":"{first} {last}","first":"Ada","last":"Lovelace"},{"format":"{first} {last}","first":"Bo","last":"Ek"}]`, "place": `[{"format":"{city}","city":"Stockholm","zip":"111 22"},{"format":"{city}","city":"Tranås","zip":"573 31"}]`, + "mid": `{"format":"","score":["{/src.score}",{"format":"5","datatype":"integer"}]}`, + "src": `{"format":"","code":[null,"200","404"],"del":null,"score":[null,{"format":"{int(1,9)}","datatype":"integer"}]}`, "trip": `{"format":"","leg":[{"format":"{to}","to":"Oslo"},{"format":"{to}","to":"Rome"}]}`, }), WithSeed(1)) } +func TestFakeStructFieldTaggedWithAColumnIsThatColumn(t *testing.T) { + f := structData(t) + nils := map[string]int{} + show := func(p any) string { + switch p := p.(type) { + case *string: + if p != nil { + return *p + } + case *int64: + if p != nil { + return strconv.FormatInt(*p, 10) + } + } + return "nil" + } + for i := 0; i < 200; i++ { + var v struct { + Code *string `fake:"src.code"` + Codes *string `fake:"[\"{/src.score}\",\"{/src.code}\"]"` + Del *int64 `fake:"src.del"` + Label string `fake:"n={/src.score}"` + Mixed *string `fake:"[\"{/src.score}\",\"x\"]"` + Pair *int64 `fake:"[\"{/src.score}\",\"5\"]"` + Score *int64 `fake:"src.score"` + } + if err := f.FakeStruct(&v); err != nil { + t.Fatal(err) + } + score, code := show(v.Score), show(v.Code) + switch { + case show(v.Del) != "nil": + t.Fatalf("Del = %s, want nil from a column only ever null", show(v.Del)) + case show(v.Mixed) != "x" && show(v.Mixed) != score, show(v.Pair) != "5" && show(v.Pair) != score, show(v.Codes) != score && show(v.Codes) != code: + t.Fatalf("Mixed %s, Pair %s, Codes %s beside score %s and code %s: want each the literal or the draw of the column it reads", show(v.Mixed), show(v.Pair), show(v.Codes), score, code) + } + switch { + case v.Code == nil: + nils["code"]++ + case *v.Code != "200" && *v.Code != "404": + t.Fatalf("Code = %q, want nil or a code, never a null's \"\"", *v.Code) + } + switch { + case v.Score == nil && v.Label == "n=": + nils["score"]++ + case v.Score == nil || *v.Score < 1 || *v.Score > 9 || v.Label != "n="+strconv.FormatInt(*v.Score, 10): + t.Fatalf("%+v, want Score nil beside Label n=, or one digit in both", v) + } + } + for _, name := range []string{"code", "score"} { + if n := nils[name]; n == 0 || n == 200 { + t.Errorf("%s was nil %d times in 200 fills, want both outcomes", name, n) + } + } +} + func TestFakeStructFillsTaggedFields(t *testing.T) { a, b := structData(t), structData(t) people := map[string]string{"Ada": "Lovelace", "Bo": "Ek"} @@ -166,6 +225,18 @@ func TestFakeStructErrors(t *testing.T) { {&struct { A int `fake:"[null,\"{int(1,9)}\"]"` }{}, "can draw null, which int cannot hold; make it *int"}, + {&struct { + A int64 `fake:"src.score"` + }{}, "can draw null, which int64 cannot hold; make it *int64"}, + {&struct { + A string `fake:"src.code"` + }{}, "can draw null, which string cannot hold; make it *string"}, + {&struct { + A *bool `fake:"src.score"` + }{}, ".A (*bool): {int(1,9)} prints an integer, not a boolean"}, + {&struct { + A int64 `fake:"mid.score"` + }{}, "can draw null, which int64 cannot hold; make it *int64"}, {&struct { A int `fake:"{digits(3)}"` }{}, ".A (int): {digits(3)} prints text, not an integer"}, diff --git a/todo.md b/todo.md index 1e4124b..1b0f4a6 100644 --- a/todo.md +++ b/todo.md @@ -17,11 +17,6 @@ The record API lands first, so the data update can use it. - `code` and `symbol` sibling fields reading `currency`, as `{code} {symbol}` → a matching pair - `{a} & {b}`, each reading `person` → one person, or two when `a` and `b` name different groups - two bare `{/sv_SE.word}` → two words -- Reference inheritance — settle whether a column that is exactly one reference to - another record's column, like `{/src.score}`, takes that column's datatype and - null. Today a null there writes `""`, a `*T` struct field reading it gets `""` - rather than nil, and a typed column reading it is refused. Settle before draw - groups and the data update. ### Data @@ -29,7 +24,9 @@ The record API lands first, so the data update can use it. blocks as columns (`sv_SE.person` → `femalefirst`, `malefirst`; `misc.uuid` → `variant`), and `sv_SE.address` draws its postal code apart from its locality. It also settles what a version promises about shipped data: its paths, its - record columns, and whether a seed renders the same output across versions. + record columns with their datatypes and nulls — a column of one reference alone + takes both, so typing a column or adding a null breaks its readers — and whether + a seed renders the same output across versions. - `email.local` and `username` share their handle lists, while their name variants differ on purpose. Share the lists only if that is a clean win. diff --git a/value.go b/value.go index d8aa48c..cf59cdf 100644 --- a/value.go +++ b/value.go @@ -16,31 +16,60 @@ type proven struct { integral bool notOperand string // why some render reads as no finite number, the way calc reads it not [len(dataTypeNames)]string + null bool // some draw of a column is null } // valueProof proves what typed columns and their calc operands hold, each node once per scope. type valueProof struct { - memo map[node]proven + memo map[node]proven + columns map[node]proven } -// checkDatatype rejects a typed column some render of which is not text of its datatype. +// checkDatatype rejects a typed column item some render of which is not text of its datatype, +// and one restating the datatype of the column it is. func (p *valueProof) checkDatatype(path string, n node) error { t, ok := n.(*template) if !ok || t.datatype == DataTypeString { return nil } - if err := p.prove(t, t.datatype); err != nil { - return fmt.Errorf("%s: %w", path, err) + if d := readDatatype(t); d == t.datatype { + return fmt.Errorf(`%s: %s takes datatype %s from the column it reads; drop "datatype"`, path, t.format, d) + } + if reason := p.columnItem(t).not[t.datatype]; reason != "" { + return fmt.Errorf("%s: datatype %s: %s", path, t.datatype, reason) } return nil } -// prove reports why some render of n is not text of datatype d. -func (p *valueProof) prove(n node, d DataType) error { - if reason := p.of(n).not[d]; reason != "" { - return fmt.Errorf("datatype %s: %s", d, reason) +// columnItem proves a column item: what it renders, or, when it is a column it reads, that column. +func (p *valueProof) columnItem(t *template) proven { + if t.readsColumn == nil { + return p.of(t) } - return nil + return p.column(t.readsColumn.column) +} + +// column proves a column over what its items draw, a null item marking it null rather than +// rendering "". +func (p *valueProof) column(n node) proven { + if v, done := p.columns[n]; done { + return v + } + if p.columns == nil { + p.columns = map[node]proven{} + } + items, nullable := columnItems(n) + var v proven + for i, it := range items { + if w := p.columnItem(it); i == 0 { + v = w + } else { + v = v.or(w) + } + } + v.null = v.null || nullable + p.columns[n] = v + return v } func (p *valueProof) of(n node) proven { @@ -66,16 +95,21 @@ func (p *valueProof) of(n node) proven { func (p *valueProof) unite(nodes []node) proven { v := p.of(nodes[0]) for _, n := range nodes[1:] { - w := p.of(n) - v.lo, v.hi, v.nonZero = min(v.lo, w.lo), max(v.hi, w.hi), min(v.nonZero, w.nonZero) - v.integral = v.integral && w.integral - if v.notOperand == "" { - v.notOperand = w.notOperand - } - for d := range v.not { - if v.not[d] == "" { - v.not[d] = w.not[d] - } + v = v.or(p.of(n)) + } + return v +} + +// or is what a proof knows of a render that is either v or w. +func (v proven) or(w proven) proven { + v.lo, v.hi, v.nonZero = min(v.lo, w.lo), max(v.hi, w.hi), min(v.nonZero, w.nonZero) + v.integral, v.null = v.integral && w.integral, v.null || w.null + if v.notOperand == "" { + v.notOperand = w.notOperand + } + for d := range v.not { + if v.not[d] == "" { + v.not[d] = w.not[d] } } return v