From d4060e041aa8533e86860fa5006534e2e6bf8ec3 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Thu, 17 Sep 2026 11:47:39 +0200 Subject: [PATCH 01/22] Test the table node: rows TSV, key and name selection, linked tables drawn consistently, and the choice-of-rows fence --- cmd/fejkdata/main_test.go | 37 ++- datatype_test.go | 4 +- inline_test.go | 12 +- loading_test.go | 2 +- perf_test.go | 24 ++ readme_test.go | 54 +++- record_test.go | 6 +- reference_test.go | 8 +- shape_test.go | 14 + struct_test.go | 6 +- table_test.go | 526 ++++++++++++++++++++++++++++++++++++++ transform_test.go | 2 +- 12 files changed, 671 insertions(+), 24 deletions(-) create mode 100644 table_test.go diff --git a/cmd/fejkdata/main_test.go b/cmd/fejkdata/main_test.go index 2e17b5c..a195472 100644 --- a/cmd/fejkdata/main_test.go +++ b/cmd/fejkdata/main_test.go @@ -310,8 +310,41 @@ func TestClassify(t *testing.T) { t.Errorf("classify(%q) = %v, %v; want %v", arg, got, err, want) } } - if _, err := classify("[abc]"); err == nil || !strings.Contains(err.Error(), `holds a "["`) { - t.Errorf("classify([abc]) = %v; want it rejected naming the bracket", err) + if _, err := classify("[abc]"); err == nil || !strings.Contains(err.Error(), `"["`) || !strings.Contains(err.Error(), "JSON") { + t.Errorf("classify([abc]) = %v; want it rejected naming the leading bracket", err) + } + if got, err := classify("geo.SE.municipality[St. Louis].name"); err != nil || got != argPath { + t.Errorf("classify(a selecting path) = %v, %v; want a path", got, err) + } +} + +func TestRunSelectsATableRow(t *testing.T) { + dir := t.TempDir() + for name, content := range map[string]string{ + "region.json": `{"format":"{name}","rows":"region.tsv","key":"code","name":"name"}`, + "region.tsv": "code\tname\n01\tStockholms län\n12\tSkåne län\n", + } { + if err := os.WriteFile(filepath.Join(dir, name), []byte(content), 0o644); err != nil { + t.Fatal(err) + } + } + if code, out, errb := runOut("--no-shipped-data", "-d", dir, "region[12]"); code != 0 || out != "Skåne län\n" { + t.Fatalf("region[12] = %d, %q, stderr=%q", code, out, errb) + } + if code, out, _ := runOut("--no-shipped-data", "-d", dir, "region[Skåne län].code"); code != 0 || out != "12\n" { + t.Fatalf("region[Skåne län].code = %d, %q", code, out) + } + if code, out, _ := runOut("--no-shipped-data", "-d", dir, "--format", "sql", "region[12]"); code != 0 || out != `INSERT INTO "region" ("code", "name") VALUES ('12', 'Skåne län');`+"\n" { + t.Fatalf("--format sql region[12] = %d, %q, want the table named without its selector", code, out) + } + if code, _, errb := runOut("--no-shipped-data", "-d", dir, "region[99]"); code != 1 || !strings.Contains(errb, `"99"`) { + t.Fatalf("region[99] = %d, stderr=%q, want a runtime error naming the row", code, errb) + } + if code, _, errb := runOut("--no-shipped-data", "-d", dir, "[Skåne län]"); code != 2 || !strings.Contains(errb, "JSON") { + t.Fatalf("[Skåne län] = %d, stderr=%q, want misuse: a leading bracket that is no JSON names nothing", code, errb) + } + if code, out, _ := runOut("--seed", "1", "--format", "csv", "misc.country[SE]"); code != 0 || !strings.HasPrefix(out, "alpha2,") || !strings.Contains(out, "\nSE,SWE,") { + t.Fatalf("--format csv misc.country[SE] = %d, %q", code, out) } } diff --git a/datatype_test.go b/datatype_test.go index 9e37149..4ccb06c 100644 --- a/datatype_test.go +++ b/datatype_test.go @@ -40,7 +40,7 @@ func TestDatatypeAndNullSitOnlyInAColumn(t *testing.T) { func TestDatatypeRejectsAValueItsTypeRejects(t *testing.T) { tree := map[string]string{ - "cat": `[{"format":"{code}","code":"200"},{"format":"{code}","code":"2x"}]`, + "cat": `[{"format":"{code}","code":"200"},{"format":"{code}","code":"2x","note":"n"}]`, "src": `{"format":"","code":[null,"200","2x"],"flag":{"format":"{b}","b":["true","false"],"datatype":"boolean"},"score":[null,{"format":"{int(1,9)}","datatype":"integer"}]}`, } for _, c := range []struct{ name, column, want string }{ @@ -114,7 +114,7 @@ func TestDatatypeRejectsAValueItsTypeRejects(t *testing.T) { var jsonInteger = regexp.MustCompile(`^(0|-?[1-9][0-9]*)$`) func TestDatatypeAcceptsAColumnThatAlwaysParses(t *testing.T) { - cat := `[{"format":"{code}","code":"200"},{"format":"{code}","code":"404"}]` + cat := `[{"format":"{code}","code":"200"},{"format":"{code}","code":"404","reason":"Not Found"}]` for _, column := range []string{ `{"format":"{int(1,99)}","datatype":"integer"}`, `{"format":"{int(-9,-1)}","datatype":"integer"}`, diff --git a/inline_test.go b/inline_test.go index 8cdb217..3e47467 100644 --- a/inline_test.go +++ b/inline_test.go @@ -55,7 +55,7 @@ func TestFakeTemplateJSONArray(t *testing.T) { func TestFakeTemplateCorrelatedReferences(t *testing.T) { dir := writeData(t, map[string]string{ - "person": `[{"format":"{first} {last}","first":"Ada","last":"Lovelace"},{"format":"{first} {last}","first":"Bo","last":"Ek"}]`, + "person": `[{"format":"{first} {last}","first":"Ada","last":"Lovelace"},{"format":"{first} {last}","first":"Bo","last":"Ek","born":"1990"}]`, }) f := newGenerator(t, dir, WithSeed(1)) for i := 0; i < 100; i++ { @@ -166,15 +166,19 @@ func TestIsTemplate(t *testing.T) { "{uppercase(/a)}": true, "{{/a}}": true, `"{/a} x"`: true, + "x[1]": false, // a row selected by key or name + "x[St. Louis].y": false, } { if got, err := IsTemplate(arg); err != nil || got != want { t.Errorf("IsTemplate(%q) = %v, %v; want %v", arg, got, err, want) } } for arg, want := range map[string]string{ - "[abc]": `holds a "["`, - "[abc].field": `holds a "["`, - "x[1]": `holds a "["`, + "[abc]": `starts with "["`, + "[abc].field": `starts with "["`, + "x[1]y": `"]"`, + "x[[1]]": `"["`, + "x[]": "empty", "a]b": `holds a "]"`, "a}b": `holds a "}"`, `"abc`: `holds a "\""`, diff --git a/loading_test.go b/loading_test.go index bd947af..e335a64 100644 --- a/loading_test.go +++ b/loading_test.go @@ -154,7 +154,7 @@ func TestListedPathsAllRender(t *testing.T) { } } } - for _, p := range []string{"misc.car.maker", "misc.country.alpha2", "misc.currency.symbol", "misc.httpstatus.code", "misc.mimetype.ext"} { + for _, p := range []string{"misc.car.maker", "misc.country.alpha2", "misc.country.numeric", "misc.currency.symbol", "misc.httpstatus.code", "misc.mimetype.ext"} { if !slices.Contains(paths, p) { t.Errorf("List() omits %q, which the README advertises and Fake renders", p) } diff --git a/perf_test.go b/perf_test.go index 75d1e86..5254fb6 100644 --- a/perf_test.go +++ b/perf_test.go @@ -74,6 +74,30 @@ func TestNoReferenceAllocRegression(t *testing.T) { } } +// A table read pins a row in the render's draws and reads its cells in place. +func TestNoTableAllocRegression(t *testing.T) { + f, err := New(WithoutShippedData(), WithDataFS(fstest.MapFS{ + "region.json": {Data: []byte(`{"format":"{name}","rows":"region.tsv","key":"code","weight":"population"}`)}, + "region.tsv": {Data: []byte("code\tname\tpopulation\n01\tStockholms län\t2400000\n12\tSkåne län\t1400000\n14\tVästra Götalands län\t1750000\n")}, + "x.json": {Data: []byte(`"{/region.name}, {/region.code}"`)}, + })) + if err != nil { + t.Fatal(err) + } + for _, s := range []struct { + name, path string + base float64 + }{ + {"a table drawn", "region", 2}, + {"a row selected", "region[12]", 2}, + {"two columns of one draw", "x", 10}, + } { + if allocs := testing.AllocsPerRun(10000, func() { f.Fake(s.path) }); allocs > s.base*1.10 { + t.Errorf("%s: %.1f allocs/op regressed past %.1f (baseline %.1f + 10%%); a row index built per draw is the usual cause", s.name, allocs, s.base*1.10, s.base) + } + } +} + // A record's fences read the compiled tree, so they belong to New, not to a draw. func TestNoRecordAllocRegression(t *testing.T) { for _, s := range []struct{ name, json string }{ diff --git a/readme_test.go b/readme_test.go index c628709..6283ab9 100644 --- a/readme_test.go +++ b/readme_test.go @@ -8,7 +8,26 @@ import ( "testing" ) -var jsonBlock = regexp.MustCompile("(?s)```json\n(.*?)```") +var ( + jsonBlock = regexp.MustCompile("(?s)```json\n(.*?)```") + tsvBlock = regexp.MustCompile("(?s)```tsv\n(.*?)```") + rowsFile = regexp.MustCompile(`"rows":\s*"([^"]+)"`) +) + +// exampleFiles is a README json block as a data directory's files: the category, and +// the rows TSV it names, taken from the nearest tsv block above it. +func exampleFiles(t *testing.T, src string, at int, body string) map[string]string { + t.Helper() + files := map[string]string{"example.json": body} + if m := rowsFile.FindStringSubmatch(body); m != nil { + tsv := tsvBlock.FindAllStringSubmatch(src[:at], -1) + if tsv == nil { + t.Fatalf("README example names %s with no tsv block above it", m[1]) + } + files[m[1]] = tsv[len(tsv)-1][1] + } + return files +} func readme(t *testing.T) string { t.Helper() @@ -21,12 +40,13 @@ func readme(t *testing.T) string { func TestReadmeExamplesLoadAndRender(t *testing.T) { n := 0 - for _, m := range jsonBlock.FindAllStringSubmatch(readme(t), -1) { - body := m[1] + src := readme(t) + for _, at := range jsonBlock.FindAllStringSubmatchIndex(src, -1) { + body := src[at[2]:at[3]] if strings.Contains(body, "…") { continue } - f, err := New(WithDataPath(writeData(t, map[string]string{"example": body})), WithSeed(1)) + f, err := New(WithDataPath(writeFiles(t, exampleFiles(t, src, at[0], body))), WithSeed(1)) if err != nil { t.Errorf("README example does not load: %v\n%s", err, body) continue @@ -128,3 +148,29 @@ func TestReadmeDatatypeExample(t *testing.T) { t.Errorf("README Datatype example shows %v, want an integer, a number and a boolean column", shown) } } + +func TestReadmeTableExample(t *testing.T) { + src := readme(t) + i := strings.Index(src, "### Table") + if i < 0 { + t.Fatal("README lost the Table section") + } + at := jsonBlock.FindStringSubmatchIndex(src[i:]) + if at == nil { + t.Fatal("README Table section has no json block") + } + body := src[i:][at[2]:at[3]] + f, err := New(WithDataPath(writeFiles(t, exampleFiles(t, src, i+at[0], body))), WithSeed(1)) + if err != nil { + t.Fatal(err) + } + for path, want := range map[string]string{"example[SE]": "Sweden (SE)", "example[Norway].alpha2": "NO"} { + if got := fake(t, f, path); got != want { + t.Errorf("README table example: Fake(%q) = %q, want %q", path, got, want) + } + } + r, err := f.FakeRecord("example") + if err != nil || r.CSVHeader() != "alpha2,name,population" { + t.Fatalf("README table example as a record: %q, %v", r.CSVHeader(), err) + } +} diff --git a/record_test.go b/record_test.go index 486ecac..f75d8dd 100644 --- a/record_test.go +++ b/record_test.go @@ -257,7 +257,7 @@ func TestRecordRejectsAColumnReadingItsOwnRecord(t *testing.T) { func TestRecordBareReferenceStaysIndependentAsAnOperand(t *testing.T) { dir := writeData(t, map[string]string{ - "cur": `[{"format":"{code}","code":"aud"},{"format":"{code}","code":"eur"}]`, + "cur": `[{"format":"{code}","code":"aud"},{"format":"{code}","code":"eur","symbol":"€"}]`, "row": `{"format":"","up":"{uppercase(/cur)}","low":"{lowercase(/cur)}"}`, }) f := newGenerator(t, dir, WithSeed(1)) @@ -305,7 +305,7 @@ func TestRecordSQLInsert(t *testing.T) { func TestRecordRejectsFieldDescent(t *testing.T) { dir := writeData(t, map[string]string{ "cat": `{"format":"{sub}","sub":{"format":"{x}","x":"1"}}`, - "row": `[{"format":"{x}","x":"1"},{"format":"{x}","x":"2"}]`, + "row": `[{"format":"{x}","x":"1"},{"format":"{x}","x":"2","y":"b"}]`, }) f := newGenerator(t, dir, WithSeed(1)) for _, path := range []string{"cat.sub", "row.x"} { @@ -317,7 +317,7 @@ func TestRecordRejectsFieldDescent(t *testing.T) { func TestRecordSharesAReferenceAcrossColumns(t *testing.T) { dir := writeData(t, map[string]string{ - "currency": `[{"format":"{code}","code":"AUD","symbol":"$"},{"format":"{code}","code":"EUR","symbol":"€"}]`, + "currency": `[{"format":"{code}","code":"AUD","symbol":"$"},{"format":"{code}","code":"EUR","symbol":"€","name":"Euro"}]`, "price": `{"format":"{code} {symbol}","code":"{/currency.code}","symbol":"{/currency.symbol}"}`, }) f := newGenerator(t, dir, WithSeed(1)) diff --git a/reference_test.go b/reference_test.go index 8219fd9..e4e1705 100644 --- a/reference_test.go +++ b/reference_test.go @@ -220,7 +220,7 @@ func TestNewErrorPathIsCanonical(t *testing.T) { func TestReferencePathIsHeld(t *testing.T) { dir := writeData(t, map[string]string{ - "person": `[{"format":"{first} {last}","first":"Anna","last":"Andersson"},{"format":"{first} {last}","first":"Bo","last":"Berg"}]`, + "person": `[{"format":"{first} {last}","first":"Anna","last":"Andersson"},{"format":"{first} {last}","first":"Bo","last":"Berg","born":"1980"}]`, "card": `"{/person.first} {/person.last}"`, }) f := newGenerator(t, dir, WithSeed(3)) @@ -291,7 +291,7 @@ func TestReferenceOverlapIsRejected(t *testing.T) { } } -const drawPeople = `[{"format":"{first} {last}","first":"Ada","last":"Lovelace"},{"format":"{first} {last}","first":"Bo","last":"Ek"},{"format":"{first} {last}","first":"Cy","last":"Young"}]` +const drawPeople = `[{"format":"{first} {last}","first":"Ada","last":"Lovelace"},{"format":"{first} {last}","first":"Bo","last":"Ek"},{"format":"{first} {last}","first":"Cy","last":"Young","born":"1867"}]` // onePerson reports whether name is the first name and surname of one drawPeople row. func onePerson(name string) bool { @@ -381,7 +381,7 @@ func TestRelativeReferences(t *testing.T) { func TestRelativeAndRootSpellingsBindOneDraw(t *testing.T) { dir := writeData(t, map[string]string{ - "sv_SE/person": `[{"format":"{first} {last}","first":"Anna","last":"Andersson"},{"format":"{first} {last}","first":"Bo","last":"Berg"}]`, + "sv_SE/person": `[{"format":"{first} {last}","first":"Anna","last":"Andersson"},{"format":"{first} {last}","first":"Bo","last":"Berg","born":"1980"}]`, "sv_SE/card": `"{.person.first} {/sv_SE.person.last}"`, }) f := newGenerator(t, dir, WithSeed(3)) @@ -408,7 +408,7 @@ func TestReferenceSigilErrors(t *testing.T) { } func TestSpellingsOfOneReferenceAreOneLevel(t *testing.T) { - person := `[{"format":"{first} {last}","first":"Ada","last":"Byron"},{"format":"{first} {last}","first":"Bo","last":"Ek"}]` + person := `[{"format":"{first} {last}","first":"Ada","last":"Byron"},{"format":"{first} {last}","first":"Bo","last":"Ek","born":"1990"}]` _, err := New(WithoutShippedData(), WithDataPath(writeData(t, map[string]string{ "sv_SE/person": person, "sv_SE/mail": `"{.person} <{/sv_SE.person.first}>"`, diff --git a/shape_test.go b/shape_test.go index ca12de4..72e3010 100644 --- a/shape_test.go +++ b/shape_test.go @@ -112,3 +112,17 @@ func TestShippedShapeNamesReads(t *testing.T) { t.Fatalf("shippedShape =\n%s\nwant\n%s", got, want) } } + +func TestShippedShapeNamesTables(t *testing.T) { + f := newGenerator(t, writeFiles(t, map[string]string{ + "w.json": `"x"`, + "r.json": `{"format":"{name}","rows":"r.tsv","key":"code","name":"name","weight":"w"}`, + "r.tsv": "code\tname\tw\n1\ta\t2\n2\tb\t3\n", + "m.json": `{"format":"{name} {/w}","rows":"m.tsv","key":"code","parent":"r"}`, + "m.tsv": "code\tname\tr\n10\tc\t1\n20\td\t2\n", + })) + want := "m\tformat \"{name} {/w}\"\tkey code\tparent r\treads w\nm.code\tstring\nm.name\tstring\nm.r\tstring\nr\tformat \"{name}\"\tkey code\tname name\tweight w\nr.code\tstring\nr.m\nr.m.code\nr.m.name\nr.m.r\nr.name\tstring\nr.w\tstring\nw\tformat \"x\"\n" + if got := shippedShape(f); got != want { + t.Fatalf("shippedShape =\n%s\nwant\n%s", got, want) + } +} diff --git a/struct_test.go b/struct_test.go index 61f2fcc..89dffb9 100644 --- a/struct_test.go +++ b/struct_test.go @@ -52,8 +52,8 @@ type structLink struct { func structData(t *testing.T) *Generator { t.Helper() return newGenerator(t, writeData(t, map[string]string{ - "person": `[{"format":"{first} {last}","first":"Ada","last":"Lovelace"},{"format":"{first} {last}","first":"Bo","last":"Ek"}]`, - "place": `[{"format":"{city}","city":"Stockholm","zip":"111 22"},{"format":"{city}","city":"Tranås","zip":"573 31"}]`, + "person": `[{"format":"{first} {last}","first":"Ada","last":"Lovelace"},{"format":"{first} {last}","first":"Bo","last":"Ek","born":"1815"}]`, + "place": `[{"format":"{city}","city":"Stockholm","zip":"111 22"},{"format":"{city}","city":"Tranås","zip":"573 31","region":"F"}]`, "mid": `{"format":"","score":["{/src.score}",{"format":"5","datatype":"integer"}]}`, "src": `{"format":"","code":[null,"200","404"],"del":null,"score":[null,{"format":"{int(1,9)}","datatype":"integer"}]}`, "trip": `{"format":"","leg":[{"format":"{to}","to":"Oslo"},{"format":"{to}","to":"Rome"}]}`, @@ -275,7 +275,7 @@ func TestFakeStructErrors(t *testing.T) { }{}, "is empty"}, {&struct { A string `fake:"[abc]"` - }{}, `holds a "["`}, + }{}, `"["`}, {&struct { A string `fake:"{.person.first} x"` }{}, "write {/person.first}"}, diff --git a/table_test.go b/table_test.go new file mode 100644 index 0000000..54392fa --- /dev/null +++ b/table_test.go @@ -0,0 +1,526 @@ +package fejkdata + +import ( + "os" + "path/filepath" + "reflect" + "regexp" + "slices" + "strings" + "testing" +) + +// writeFiles writes a data directory from a map of relative file name, extension +// included, to content. +func writeFiles(t *testing.T, files map[string]string) string { + t.Helper() + dir := t.TempDir() + for name, content := range files { + p := filepath.Join(dir, name) + if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(p, []byte(content), 0o644); err != nil { + t.Fatal(err) + } + } + return dir +} + +// geo is three linked tables: region <- municipality <- locality. +func geo() map[string]string { + return map[string]string{ + "region.json": `{"format":"{name}","rows":"region.tsv","key":"code","name":"name","weight":"population"}`, + "region.tsv": "code\tname\tpopulation\ttimezone\n01\tStockholms län\t2400000\tEurope/Stockholm\n12\tSkåne län\t1400000\tEurope/Stockholm\n14\tVästra Götalands län\t1750000\tEurope/Stockholm\n", + "municipality.json": `{"format":"{name}","rows":"municipality.tsv","key":"code","name":"name","parent":"region","weight":"population"}`, + "municipality.tsv": "code\tname\tregion\tpopulation\n0180\tStockholm\t01\t980000\n0184\tSolna\t01\t85000\n1280\tMalmö\t12\t360000\n1281\tLund\t12\t130000\n1480\tGöteborg\t14\t590000\n", + "locality.json": `{"format":"{name}","rows":"locality.tsv","key":"code","name":"name","parent":"municipality"}`, + "locality.tsv": "code\tname\tmunicipality\nL1\tStockholm\t0180\nL2\tSolna\t0184\nL3\tMalmö\t1280\nL4\tLund\t1281\nL5\tGöteborg\t1480\nL6\tHisingen\t1480\nL7\tSandby\t1281\nL8\tSandby\t0184\n", + } +} + +var ( + municipalityOf = map[string]string{"L1": "0180", "L2": "0184", "L3": "1280", "L4": "1281", "L5": "1480", "L6": "1480", "L7": "1281", "L8": "0184"} + regionOf = map[string]string{"0180": "01", "0184": "01", "1280": "12", "1281": "12", "1480": "14"} +) + +func with(files map[string]string, more map[string]string) map[string]string { + out := map[string]string{} + for k, v := range files { + out[k] = v + } + for k, v := range more { + out[k] = v + } + return out +} + +func TestTableRendersRowsAndColumns(t *testing.T) { + f := newGenerator(t, writeFiles(t, geo()), WithSeed(1)) + names := map[string]bool{"Stockholms län": true, "Skåne län": true, "Västra Götalands län": true} + for i := 0; i < 50; i++ { + if v := fake(t, f, "region"); !names[v] { + t.Fatalf("region = %q, want a row's format", v) + } + if v := fake(t, f, "region.timezone"); v != "Europe/Stockholm" { + t.Fatalf("region.timezone = %q", v) + } + if v := fake(t, f, "region.code"); v != "01" && v != "12" && v != "14" { + t.Fatalf("region.code = %q", v) + } + } + want := []string{ + "locality", "locality.code", "locality.municipality", "locality.name", + "municipality", "municipality.code", "municipality.locality", "municipality.locality.code", "municipality.locality.municipality", "municipality.locality.name", "municipality.name", "municipality.population", "municipality.region", + "region", "region.code", "region.municipality", "region.municipality.code", "region.municipality.locality", "region.municipality.locality.code", "region.municipality.locality.municipality", "region.municipality.locality.name", "region.municipality.name", "region.municipality.population", "region.municipality.region", "region.name", "region.population", "region.timezone", + } + if got := f.List(); !reflect.DeepEqual(got, want) { + t.Fatalf("List() = %v\nwant %v", got, want) + } + r, err := f.FakeRecord("region") + if err != nil { + t.Fatal(err) + } + if r.CSVHeader() != "code,name,population,timezone" { + t.Fatalf("CSVHeader = %q", r.CSVHeader()) + } + line := strings.Split(r.CSVLine(), ",") + if len(line) != 4 || !names[line[1]] || line[3] != "Europe/Stockholm" { + t.Fatalf("CSVLine = %q, want one row's cells", r.CSVLine()) + } + for _, c := range r.Columns() { + if c.DataType != DataTypeString || c.Null { + t.Fatalf("column %q is %s null=%v, want a string", c.Name, c.DataType, c.Null) + } + } +} + +func TestTableWeightSkewsTheDraw(t *testing.T) { + f := newGenerator(t, writeFiles(t, geo()), WithSeed(3)) + count := map[string]int{} + for i := 0; i < 3000; i++ { + count[fake(t, f, "region.code")]++ + } + if count["01"] < 1200 || count["12"] > 900 { + t.Fatalf("region draws %v, want weighted by population (01 ≈ 43%%, 12 ≈ 25%%)", count) + } + count = map[string]int{} + for i := 0; i < 3000; i++ { + count[fake(t, f, "region[01].municipality.code")]++ + } + if count["0180"] < 2500 || count["0184"] == 0 || len(count) != 2 { + t.Fatalf("municipality draws inside 01 %v, want Stockholm ≈ 92%% and Solna the rest", count) + } +} + +func TestTableCellsAreStringNodes(t *testing.T) { + dir := writeFiles(t, map[string]string{ + "w.json": `"x"`, + "place.json": `{"format":"{zip} {name}","rows":"place.tsv","key":"name"}`, + "place.tsv": "name\tzip\ttag\nStockholm\t1{digits(2)} {digits(2)}\t{/w}\nTranås\t573 {digits(2)}\tplain\n", + }) + f := newGenerator(t, dir, WithSeed(1)) + zip := regexp.MustCompile(`^(1\d\d \d\d Stockholm|573 \d\d Tranås)$`) + for i := 0; i < 50; i++ { + if v := fake(t, f, "place"); !zip.MatchString(v) { + t.Fatalf("place = %q, want the cell's tokens expanded", v) + } + } + if v := fake(t, f, "place[Stockholm].tag"); v != "x" { + t.Fatalf("place[Stockholm].tag = %q, want the reference rendered", v) + } + bad := writeFiles(t, map[string]string{ + "place.json": `{"format":"{name}","rows":"place.tsv"}`, + "place.tsv": "name\tzip\nA\t{name}\n", + }) + if _, err := New(WithoutShippedData(), WithDataPath(bad)); err == nil || !strings.Contains(err.Error(), "place.tsv") || !strings.Contains(err.Error(), `no field "name"`) { + t.Fatalf("New = %v, want a cell reading a column refused, naming the file", err) + } +} + +func TestTableSelectsARowByKeyOrName(t *testing.T) { + files := with(geo(), map[string]string{ + "city.json": `{"format":"{name}","rows":"city.tsv","key":"code","name":"name"}`, + "city.tsv": "code\tname\nSpringfield\tSt. Louis\nSTL\tSpringfield\n", + }) + f := newGenerator(t, writeFiles(t, files), WithSeed(1)) + for path, want := range map[string]string{ + "region[12]": "Skåne län", + "region[Skåne län].code": "12", + "municipality[1281].locality[Sandby]": "Sandby", + "municipality[1281].locality[Sandby].code": "L7", + "municipality[0184].locality[Sandby].code": "L8", + "city[St. Louis].code": "Springfield", + "city[Springfield].name": "St. Louis", // a key wins over a name + } { + if got := fake(t, f, path); got != want { + t.Errorf("Fake(%q) = %q, want %q", path, got, want) + } + } + for path, want := range map[string]string{ + "region[99]": `"99"`, + "locality[Sandby]": "L7 L8", + "region[12].code[1]": "not a table", + "region[]": "empty", + "region[12": "]", + "region[12].municipality[1480]": "not inside", + } { + if _, err := f.Fake(path); err == nil || !strings.Contains(err.Error(), want) { + t.Errorf("Fake(%q) = %v, want an error mentioning %s", path, err, want) + } + } + if v := fake(t, f, "municipality[0180].region"); v != "01" { + t.Fatalf("municipality[0180].region = %q, want the link column's cell", v) + } + noKey := writeFiles(t, map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv"}`, "t.tsv": "a\nx\ny\n"}) + if _, err := newGenerator(t, noKey).Fake("t[x]"); err == nil || !strings.Contains(err.Error(), "no key or name") { + t.Fatalf("Fake(t[x]) = %v, want no column to select by", err) + } +} + +func TestTableDescendsToALinkedTable(t *testing.T) { + f := newGenerator(t, writeFiles(t, geo()), WithSeed(2)) + for i := 0; i < 200; i++ { + if m := fake(t, f, "region[12].municipality.code"); regionOf[m] != "12" { + t.Fatalf("region[12].municipality.code = %q, outside 12", m) + } + if l := fake(t, f, "region[12].locality.code"); regionOf[municipalityOf[l]] != "12" { + t.Fatalf("region[12].locality.code = %q, outside 12", l) + } + if l := fake(t, f, "municipality[1480].locality"); l != "Göteborg" && l != "Hisingen" { + t.Fatalf("municipality[1480].locality = %q", l) + } + if l := fake(t, f, "region.municipality.locality.name"); l == "" { + t.Fatal("region.municipality.locality.name rendered empty") + } + } + r, err := f.FakeRecord("region[14].municipality") + if err != nil { + t.Fatal(err) + } + if line := r.CSVLine(); line != "1480,Göteborg,590000,14" { + t.Fatalf("record inside region 14 = %q", line) + } +} + +func TestLinkedTablesDrawConsistently(t *testing.T) { + spellings := map[string]string{ + "leaf first": `{"format":"{l}|{m}|{r}","l":"{/locality.code}","m":"{/municipality.code}","r":"{/region.code}"}`, + "root first": `{"format":"{r}|{m}|{l}","r":"{/region.code}","m":"{/municipality.code}","l":"{/locality.code}"}`, + "skipping one": `{"format":"{r}|{l}","r":"{/region.code}","l":"{/locality.code}"}`, + "nested": `{"format":"{a}|{l}","a":{"format":"{r}|{m}","r":"{/region.code}","m":"{/municipality.code}"},"l":"{/locality.code}"}`, + "a record": `{"format":"","l":"{/locality.code}","m":"{/municipality.code}","r":"{/region.code}"}`, + } + for name, addr := range spellings { + f := newGenerator(t, writeFiles(t, with(geo(), map[string]string{"addr.json": addr})), WithSeed(5)) + seen := map[string]bool{} + for i := 0; i < 300; i++ { + var parts []string + if name == "a record" { + r, err := f.FakeRecord("addr") + if err != nil { + t.Fatal(err) + } + for _, c := range r.Columns() { + parts = append(parts, c.Value) + } + } else { + parts = strings.Split(fake(t, f, "addr"), "|") + } + var l, m, r string + for _, p := range parts { + switch { + case strings.HasPrefix(p, "L"): + l = p + case len(p) == 4: + m = p + default: + r = p + } + } + if m == "" { + m = municipalityOf[l] + } + if municipalityOf[l] != m || regionOf[m] != r { + t.Fatalf("%s: %v is not one consistent draw", name, parts) + } + seen[l] = true + } + if len(seen) < 4 { + t.Fatalf("%s: only %v drawn in 300 renders", name, seen) + } + } +} + +func TestLinkedTablesDrawApartAcrossGroupsAndRepeats(t *testing.T) { + files := with(geo(), map[string]string{ + "groups.json": `{"format":"{a}|{b}","a":{"format":"{/locality.code}","drawGroup":"g"},"b":"{/locality.code}"}`, + "many.json": `{"format":"{x}","x":{"format":"{/locality.code}","repeat":8,"separator":"|"}}`, + }) + f := newGenerator(t, writeFiles(t, files), WithSeed(9)) + for _, path := range []string{"groups", "many"} { + differ := false + for i := 0; i < 100 && !differ; i++ { + parts := strings.Split(fake(t, f, path), "|") + for _, p := range parts[1:] { + differ = differ || p != parts[0] + } + } + if !differ { + t.Fatalf("%s: every draw agreed in 100 renders, want groups and repeat iterations drawn apart", path) + } + } +} + +func TestTableSelectorInAReference(t *testing.T) { + files := with(geo(), map[string]string{ + "x.json": `{"format":"{a} {b} {c}","a":"{/region[12].name}","b":"{/region[12].municipality.code}","c":"{/region[12].locality.code}"}`, + }) + f := newGenerator(t, writeFiles(t, files), WithSeed(1)) + for i := 0; i < 100; i++ { + parts := strings.Fields(fake(t, f, "x")) + m, l := parts[len(parts)-2], parts[len(parts)-1] + if !strings.HasPrefix(fake(t, f, "x"), "Skåne län") || regionOf[m] != "12" || municipalityOf[l] != m { + t.Fatalf("x = %v, want one draw inside region 12", parts) + } + } + for name, c := range map[string]struct{ json, want string }{ + "unknown row": {`"{/region[99].name}"`, `"99"`}, + "ambiguous name": {`"{/locality[Sandby].name}"`, "L7 L8"}, + "not inside": {`"{/region[12].municipality[0180].name}"`, "not inside"}, + "selector on template": {`"{/x[1].a}"`, "not a table"}, + } { + _, err := New(WithoutShippedData(), WithDataPath(writeFiles(t, with(files, map[string]string{"bad.json": c.json})))) + if err == nil || !strings.Contains(err.Error(), c.want) { + t.Errorf("%s: New = %v, want an error mentioning %q", name, err, c.want) + } + } +} + +func TestTableSelectionFences(t *testing.T) { + rejected := map[string]struct{ json, want string }{ + "bare beside a path": {`"{/region} {/region.name}"`, "reads a path into"}, + "bare beside a linked path": {`"{/region} {/locality.name}"`, "drawGroup"}, + "selected beside unselected": {`"{/region[12].municipality.name} {/municipality.code}"`, "{/region[12].municipality.code}"}, + "two selections": {`"{/region[12].name} {/region[14].name}"`, "drawGroup"}, + "selection beside a descendant": {`"{/region[12].name} {/locality.code}"`, "{/region[12].locality.code}"}, + } + for name, c := range rejected { + _, err := New(WithoutShippedData(), WithDataPath(writeFiles(t, with(geo(), map[string]string{"x.json": c.json})))) + if err == nil || !strings.Contains(err.Error(), c.want) { + t.Errorf("%s: New = %v, want it rejected mentioning %q", name, err, c.want) + } + } + accepted := map[string]string{ + "a descent beside a path into it": `"{/region.municipality} {/region.municipality.name}"`, + "selections in two groups": `{"format":"{a} {b}","a":{"format":"{/region[12].name}","drawGroup":"a"},"b":"{/region[14].name}"}`, + "one selection twice": `"{/region[12].name} {/region[12].timezone} {/region[12].locality.name}"`, + "a bare table in another group": `{"format":"{a} {b}","a":{"format":"{/region}","drawGroup":"a"},"b":"{/locality.name}"}`, + } + for name, json := range accepted { + f, err := New(WithoutShippedData(), WithDataPath(writeFiles(t, with(geo(), map[string]string{"x.json": json}))), WithSeed(1)) + if err != nil { + t.Errorf("%s: New = %v, want it accepted", name, err) + continue + } + fake(t, f, "x") + } + f := newGenerator(t, writeFiles(t, with(geo(), map[string]string{"x.json": accepted["a descent beside a path into it"]})), WithSeed(1)) + for i := 0; i < 100; i++ { + if parts := strings.Fields(fake(t, f, "x")); parts[0] != parts[1] { + t.Fatalf("x = %v, want the descended row and the path into it to agree", parts) + } + } +} + +func TestTableFences(t *testing.T) { + base := map[string]string{ + "region.json": `{"format":"{name}","rows":"region.tsv","key":"code","name":"name"}`, + "region.tsv": "code\tname\n01\tA\n12\tB\n", + } + rejected := map[string]struct { + files map[string]string + want string + }{ + "rows names a missing file": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv"}`}, "t.tsv"}, + "a TSV nothing names": {with(base, map[string]string{"stray.tsv": "a\nx\n"}), "stray.tsv"}, + "rows outside its folder": {with(base, map[string]string{"t.json": `{"format":"{code}","rows":"../region.tsv"}`}), "beside"}, + "rows not a tsv": {with(base, map[string]string{"t.json": `{"format":"{code}","rows":"region.txt"}`, "region.txt": "code\n1\n"}), ".tsv"}, + "key names no column": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","key":"b"}`, "t.tsv": "a\nx\ny\n"}, `"b"`}, + "name names no column": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","name":"b"}`, "t.tsv": "a\nx\ny\n"}, `"b"`}, + "weight names no column": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","weight":"b"}`, "t.tsv": "a\nx\ny\n"}, `"b"`}, + "parent names no column": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","parent":"b"}`, "t.tsv": "a\nx\ny\n"}, `"b"`}, + "key equals name": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","key":"a","name":"a"}`, "t.tsv": "a\nx\ny\n"}, "drop"}, + "duplicate key": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","key":"a"}`, "t.tsv": "a\nx\nx\n"}, `"x"`}, + "empty key": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","key":"a"}`, "t.tsv": "a\tb\n\ty\nx\tz\n"}, "empty"}, + "a bracket in a key": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","key":"a"}`, "t.tsv": "a\nx[1]\ny\n"}, `"["`}, + "a brace in a name": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","name":"a"}`, "t.tsv": "a\nx{1}\ny\n"}, `"{"`}, + "weight not a number": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","weight":"w"}`, "t.tsv": "a\tw\nx\tmany\ny\t2\n"}, `"many"`}, + "weight zero": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","weight":"w"}`, "t.tsv": "a\tw\nx\t0\ny\t2\n"}, "0"}, + "weight negative": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","weight":"w"}`, "t.tsv": "a\tw\nx\t-1\ny\t2\n"}, "-1"}, + "reserved column name": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv"}`, "t.tsv": "a\tb.c\nx\ty\n"}, `"b.c"`}, + "duplicate column": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv"}`, "t.tsv": "a\ta\nx\ty\n"}, `"a"`}, + "empty column name": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv"}`, "t.tsv": "a\t\nx\ty\n"}, "empty"}, + "short row": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv"}`, "t.tsv": "a\tb\nx\ty\nz\n"}, "line 3"}, + "no rows": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv"}`, "t.tsv": "a\n"}, "no rows"}, + "one row": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv"}`, "t.tsv": "a\nx\n"}, "one row"}, + "empty file": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv"}`, "t.tsv": ""}, "header"}, + "parent is not a table": {map[string]string{"p.json": `"x"`, "t.json": `{"format":"{a}","rows":"t.tsv","parent":"p"}`, "t.tsv": "a\tp\nx\tx\ny\tx\n"}, "not a table"}, + "parent does not exist": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","parent":"p"}`, "t.tsv": "a\tp\nx\tx\ny\tx\n"}, `"p"`}, + "parent in another folder": {map[string]string{"g/p.json": `{"format":"{k}","rows":"p.tsv","key":"k"}`, "g/p.tsv": "k\nx\ny\n", "t.json": `{"format":"{a}","rows":"t.tsv","parent":"p"}`, "t.tsv": "a\tp\nx\tx\ny\ty\n"}, `"p"`}, + "parent has no key": {map[string]string{"p.json": `{"format":"{k}","rows":"p.tsv"}`, "p.tsv": "k\nx\ny\n", "t.json": `{"format":"{a}","rows":"t.tsv","parent":"p"}`, "t.tsv": "a\tp\nx\tx\ny\ty\n"}, "key"}, + "dangling link": {with(base, map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","parent":"region"}`, "t.tsv": "a\tregion\nx\t01\ny\t99\n"}), `"99"`}, + "childless parent row": {with(base, map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","parent":"region"}`, "t.tsv": "a\tregion\nx\t01\ny\t01\n"}), `"12"`}, + "parent cycle": {map[string]string{"a.json": `{"format":"{k}","rows":"a.tsv","key":"k","parent":"b"}`, "a.tsv": "k\tb\nx\tx\ny\ty\n", "b.json": `{"format":"{k}","rows":"b.tsv","key":"k","parent":"a"}`, "b.tsv": "k\ta\nx\tx\ny\ty\n"}, "cycle"}, + "child named like a column": {with(base, map[string]string{"name.json": `{"format":"{a}","rows":"name.tsv","parent":"region"}`, "name.tsv": "a\tregion\nx\t01\ny\t12\n"}), `"name"`}, + "rows nested in a field": {map[string]string{"t.json": `{"format":"{x}","x":{"format":"{a}","rows":"x.tsv"}}`, "x.tsv": "a\nx\ny\n"}, "category"}, + "rows in a choice item": {map[string]string{"t.json": `[{"format":"{a}","rows":"t.tsv"},"y"]`, "t.tsv": "a\nx\ny\n"}, "category"}, + "unknown table option": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","fields":"a"}`, "t.tsv": "a\nx\ny\n"}, "a table takes"}, + "repeat on a table": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","repeat":2}`, "t.tsv": "a\nx\ny\n"}, "a table takes"}, + "drawGroup on a table": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","drawGroup":"g"}`, "t.tsv": "a\nx\ny\n"}, "a table takes"}, + "option not a string": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv","key":1}`, "t.tsv": "a\nx\ny\n"}, "string"}, + "format names no column": {map[string]string{"t.json": `{"format":"{b}","rows":"t.tsv"}`, "t.tsv": "a\nx\ny\n"}, `no column "b"`}, + "format reads into a column": {map[string]string{"t.json": `{"format":"{a.x}","rows":"t.tsv"}`, "t.tsv": "a\nx\ny\n"}, `"a"`}, + "a reference into a column": {with(base, map[string]string{"t.json": `"{/region.name.x}"`}), "column"}, + "a category referencing itself": {map[string]string{"t.json": `{"format":"{a}","rows":"t.tsv"}`, "t.tsv": "a\n{/t.a}\ny\n"}, "names the category it sits in"}, + } + for name, c := range rejected { + _, err := New(WithoutShippedData(), WithDataPath(writeFiles(t, c.files))) + if err == nil { + t.Errorf("%s: New = nil error, want it rejected at load", name) + continue + } + if !strings.Contains(err.Error(), c.want) { + t.Errorf("%s: New = %v, want it to mention %q", name, err, c.want) + } + } + if _, err := New(WithoutShippedData(), WithDataFS(os.DirFS(writeFiles(t, base)))); err != nil { + t.Fatalf("New(WithDataFS) = %v, want a table loaded from any fs.FS", err) + } + inline := newGenerator(t, writeFiles(t, base)) + if _, err := inline.NewTemplate(`{"format":"{a}","rows":"t.tsv"}`); err == nil || !strings.Contains(err.Error(), "inline") { + t.Fatalf("NewTemplate(rows) = %v, want a table refused inline", err) + } + empty := newGenerator(t, writeFiles(t, map[string]string{"t.json": `{"format":"","rows":"t.tsv","key":"a"}`, "t.tsv": "a\tb\nx\t1\ny\t2\n"}), WithSeed(1)) + if v := fake(t, empty, "t"); v != "" { + t.Fatalf("a record-only table renders %q, want \"\"", v) + } + if v := fake(t, empty, "t[y].b"); v != "2" { + t.Fatalf("t[y].b = %q", v) + } +} + +func TestSameShapedChoiceIsATable(t *testing.T) { + rows := `[{"format":"{name}","name":"Sweden","alpha2":"SE"},{"format":"{name}","name":"Norway","alpha2":"NO"},{"format":"{name}","name":"Denmark","alpha2":"DK"}]` + _, err := New(WithoutShippedData(), WithDataPath(writeData(t, map[string]string{"country": rows}))) + for _, want := range []string{"country.tsv", `"rows"`, "alpha2\tname", "3 "} { + if err == nil || !strings.Contains(err.Error(), want) { + t.Fatalf("New(same-shaped choice) = %v, want it refused mentioning %q", err, want) + } + } + weighted := `[{"format":"{name}","name":"Sweden","weight":2},{"format":"{name}","name":"Norway"}]` + if _, err := New(WithoutShippedData(), WithDataPath(writeData(t, map[string]string{"country": weighted}))); err == nil || !strings.Contains(err.Error(), "weight") { + t.Fatalf("New(weighted same-shaped choice) = %v, want it refused naming a weight column", err) + } + accepted := map[string]string{ + "nested": `{"format":"{c}","c":` + rows + `}`, + "a choice field": `[{"format":"{maker} {model}","maker":"BMW","model":["X3","X5"]},{"format":"{maker} {model}","maker":"Ford","model":["Focus","Fiesta"]}]`, + "different formats": `[{"format":"{a}-{b}","a":"1","b":"2"},{"format":"{b}-{a}","a":"3","b":"4"}]`, + "different fields": `[{"format":"{a}","a":"1","b":"2"},{"format":"{a}","a":"3"}]`, + "a string item": `[{"format":"{a}","a":"1"},"x"]`, + "strings": `["a","b","c"]`, + } + for name, json := range accepted { + if _, err := New(WithoutShippedData(), WithDataPath(writeData(t, map[string]string{"x": json}))); err != nil { + t.Errorf("%s: New = %v, want it accepted", name, err) + } + } + f := newGenerator(t, writeData(t, map[string]string{"w": `"x"`})) + if _, err := f.NewTemplate(rows); err != nil { + t.Fatalf("NewTemplate(same-shaped choice) = %v, want an inline template exempt", err) + } +} + +func TestTableInAStructTag(t *testing.T) { + f := newGenerator(t, writeFiles(t, geo()), WithSeed(1)) + var v struct { + Region string `fake:"region[12].name"` + Locality string `fake:"region[12].locality.code"` + Any string `fake:"{/municipality[Lund].code}"` + } + if err := f.FakeStruct(&v); err != nil { + t.Fatal(err) + } + if v.Region != "Skåne län" || regionOf[municipalityOf[v.Locality]] != "12" || v.Any != "1281" { + t.Fatalf("FakeStruct = %+v, want the selected rows", v) + } +} + +func TestTableOverridesByLayering(t *testing.T) { + mine := writeFiles(t, map[string]string{ + "misc/country.json": `{"format":"{name}","rows":"country.tsv","key":"alpha2"}`, + "misc/country.tsv": "alpha2\tname\nXX\tNowhere\nYY\tElsewhere\n", + }) + f, err := New(WithDataPath(mine), WithSeed(1)) + if err != nil { + t.Fatal(err) + } + if v := fake(t, f, "misc.country[XX]"); v != "Nowhere" { + t.Fatalf("misc.country[XX] = %q, want the layered table", v) + } + if paths := f.List(); slices.Contains(paths, "misc.country.alpha3") { + t.Fatal("List() still offers the shipped table's columns under an overridden category") + } +} + +func TestShippedTables(t *testing.T) { + f := newGenerator(t, "data/misc", WithSeed(1)) + for path, want := range map[string]string{ + "country[SE]": "Sweden", + "country[Sweden].alpha3": "SWE", + "country[SE].numeric": "752", + "country[SE].tld": ".se", + "country[SE].calling-code": "46", + "country[SE].capital": "Stockholm", + "country[SE].currency": "SEK", + "country[SE].flag": "🇸🇪", + "currency[SEK].name": "Swedish Krona", + "currency[SEK].symbol": "kr", + "currency[SEK].numeric": "752", + "currency[SEK].decimals": "2", + "currency[Euro].code": "EUR", + "language[sv]": "Swedish", + "httpstatus[404]": "404 Not Found", + "httpstatus[404].reason": "Not Found", + "mimetype[application/json].ext": ".json", + } { + if got := fake(t, f, path); got != want { + t.Errorf("Fake(%q) = %q, want %q", path, got, want) + } + } + re := map[string]*regexp.Regexp{ + "country.numeric": regexp.MustCompile(`^\d{3}$`), + "country.tld": regexp.MustCompile(`^\.[a-z]{2}$`), + "country.calling-code": regexp.MustCompile(`^\d{1,4}(-\d{3})?$`), + "country.capital": regexp.MustCompile(`\p{L}`), + "country.currency": regexp.MustCompile(`^[A-Z]{3}$`), + "country.flag": regexp.MustCompile(`^[\x{1F1E6}-\x{1F1FF}]{2}$`), + "country.languages": regexp.MustCompile(`^[a-z]{2,3}(-[A-Z]{2})?(,[a-z]{2,3}(-[A-Z]{2})?)*$`), + "currency.numeric": regexp.MustCompile(`^\d{3}$`), + "currency.decimals": regexp.MustCompile(`^[0-4]$`), + } + for i := 0; i < 100; i++ { + for p, rx := range re { + if v := fake(t, f, p); !rx.MatchString(v) { + t.Fatalf("misc %s = %q, want %s", p, v, rx) + } + } + } + count := map[string]bool{} + for i := 0; i < 5000; i++ { + count[fake(t, f, "country.alpha2")] = true + } + if len(count) < 200 { + t.Fatalf("country draws %d distinct rows in 5000, want the full register", len(count)) + } +} diff --git a/transform_test.go b/transform_test.go index 0272856..740212d 100644 --- a/transform_test.go +++ b/transform_test.go @@ -27,7 +27,7 @@ func TestTransformReadsTheHeldDraw(t *testing.T) { func TestTransformOverAReference(t *testing.T) { dir := writeData(t, map[string]string{ - "person": `[{"format":"{first} {last}","first":"Åsa","last":"Öberg"},{"format":"{first} {last}","first":"Bo","last":"Ek"}]`, + "person": `[{"format":"{first} {last}","first":"Åsa","last":"Öberg"},{"format":"{first} {last}","first":"Bo","last":"Ek","born":"1990"}]`, "email": `"{/person.first} {/person.last} <{lowercase(ascii(/person.first))}.{lowercase(ascii(/person.last))}@example.com>"`, }) f := newGenerator(t, dir, WithSeed(5)) -- 2.52.0 From f6b54c8ae60e31475c45efe4ee2abbfb2764c955 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Thu, 17 Sep 2026 12:28:46 +0200 Subject: [PATCH 02/22] Add the table node: rows TSV, key and name selection, linked tables drawn consistently, and the choice-of-rows fence --- CHANGELOG.md | 12 ++ README.md | 187 +++++++++++++++-- cmd/fejkdata/main.go | 26 ++- data.go | 47 ++++- draw.go | 254 ++++++++++++++++++++-- fejkdata.go | 26 ++- graph.go | 29 ++- hold.go | 26 ++- inline.go | 8 +- node.go | 78 ++++++- path.go | 254 +++++++++++++++++++--- path_test.go | 8 +- record.go | 53 ++++- reference.go | 12 +- render.go | 67 +++--- shape_test.go | 24 +++ struct.go | 3 +- table.go | 486 +++++++++++++++++++++++++++++++++++++++++++ template.go | 44 ++-- value.go | 24 +++ 20 files changed, 1517 insertions(+), 151 deletions(-) create mode 100644 table.go diff --git a/CHANGELOG.md b/CHANGELOG.md index b1d7908..997eabd 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,3 +9,15 @@ replacement, and each removed path, column or flag. ### Added - First release: the CLI, the library and the shipped data set. +- Table categories: a category JSON naming a `rows` TSV beside it, with the options + `key`, `name`, `weight` and `parent`; a path selects a row by key or name, + `misc.country[SE]`, and descends to a linked table by name; linked tables draw + consistently within one render and draw group. `rows` is an option, so no + template may carry a field of that name. +- `New` refuses a root choice of templates sharing one format and one set of string + fields, naming the rows TSV to write instead. +- `misc.country`, `misc.currency`, `misc.language`, `misc.httpstatus` and + `misc.mimetype` are tables. `misc.country` is the full ISO 3166 register with the + columns `calling-code`, `capital`, `currency`, `flag`, `languages`, `numeric` and + `tld` added; `misc.currency` the current ISO 4217 list with `decimals` and + `numeric` added, and its symbols from CLDR. `DATA-LICENSES.md` lists each source. diff --git a/README.md b/README.md index 61c3db2..5d86fe1 100644 --- a/README.md +++ b/README.md @@ -17,6 +17,7 @@ fejkdata sv_SE.person.last # Eriksson fejkdata --seed 42 sv_SE.address # the same address every run fejkdata -n 3 --separator ', ' sv_SE.word # nät, barn, sol fejkdata --list # every path the data offers +fejkdata 'misc.country[SE].capital' # Stockholm — a table's row, selected by key or name fejkdata --data-path ./mydata sv_SE.word # layer a directory over the shipped data fejkdata --no-shipped-data -d ./mydata --list # only your data fejkdata 'name: {/sv_SE.person.last}' # name: — an inline template @@ -24,7 +25,8 @@ fejkdata '{"format":"name: {x}","x":["bosse","lina"]}' # name: bosse or name: l ``` A path names a category, or a field inside one: each dot segment descends one -level — folders, then the category (a JSON file), then fields. An argument that is +level — folders, then the category (a JSON file), then fields — and `[SE]` after a +[table](#table) selects its row. An argument that is a JSON object, array or string, or that carries a `{` token, is instead an **inline template**: a format string or a JSON value compiled and rendered on the spot. Its tokens reach the data by reference from the root — @@ -32,8 +34,8 @@ spot. Its tokens reach the data by reference from the root — available. An inline template sits in no folder, so the folder-relative `{.name}` and `{..name}` are rejected naming the root spelling, and one reference alone — `{/sv_SE.person}` — is the path written as a template, rejected naming the path, as is -a path written `/sv_SE.person`. A path never contains a brace, a bracket or a quote, -so the two cannot collide (see [Decisions](#decisions)). +a path written `/sv_SE.person`. A path never contains a brace or a quote, and a +bracket only as a selector after a name, so the two cannot collide (see [Decisions](#decisions)). | Flag | | |------|--| @@ -220,7 +222,10 @@ Each locale carries `address`, `color`, `company`, `date`, `email`, `ip`, `country` (ISO 3166), `creditcard` (Luhn-valid), `currency` (ISO 4217), `emoji`, `httpstatus`, `language` (ISO 639), `mac`, `mimetype`, `objectid`, `timezone` (IANA), `useragent` and `uuid` (v4). Many carry sub-fields — `misc.currency.symbol`, -`misc.country.alpha2`, `misc.httpstatus.code` — which `--list` shows. +`misc.country.alpha2`, `misc.httpstatus.code` — which `--list` shows. `country`, +`currency`, `httpstatus`, `language` and `mimetype` are [tables](#table), so +`misc.country[SE].capital` and `misc.currency[Euro].symbol` select a row; +[`DATA-LICENSES.md`](DATA-LICENSES.md) names each table's source and licence. ## Data format @@ -231,6 +236,7 @@ Every value is a **node**, nestable without limit: | string | `"Malmö"` | its text, with any `{…}` tokens expanded | | choice | `["a", "b", …]` | one item, picked at random | | template | `{"format": "…", …}` | its format, with `{name}` tokens rendering the named fields | +| table | `{"format": "…", "rows": "x.tsv", …}` | its format over one row of the TSV beside it ([Table](#table)) | ### Format string @@ -340,10 +346,105 @@ renders a null as `""`. The other items' weights skew its odds: load: `null` anywhere but a column, naming `""`, and a column whose items hold different datatypes. +### Table + +A table is a category whose rows come from a TSV beside its JSON file: the header +names the columns, each line below it is one row, and the format renders the row +drawn. Save `mydata/country.tsv` and `mydata/country.json`: + +```tsv +alpha2 name population +DK Denmark 5900000 +NO Norway 5500000 +SE Sweden 10500000 +``` + +```json +{ "format": "{name} ({alpha2})", "rows": "country.tsv", "key": "alpha2", "name": "name", "weight": "population" } +``` + +```sh +fejkdata -d ./mydata country # Sweden (SE), about half the time +fejkdata -d ./mydata country.alpha2 # NO +fejkdata -d ./mydata 'country[SE]' # Sweden (SE) +fejkdata -d ./mydata 'country[Norway].alpha2' # NO +fejkdata -d ./mydata --format csv country # alpha2,name,population → SE,Sweden,10500000 +``` + +`rows` names the TSV beside the category file; `key` names the column a path +selects a row by, `name` a column it also selects by, `weight` a column of positive +numbers that skews the draw, and `parent` the table a column links to +([Linked tables](#linked-tables)). The format's `{tokens}` read the columns, and the +columns are the [record](#records)'s columns, so `--format csv` writes the rows and +`--list` shows `country.alpha2`. A cell is a string node: `1{digits(2)} {digits(2)}` +in a cell draws digits and `{/misc.uuid}` reads a reference, while `{name}` in a cell +is refused, since a cell has no sibling. `New` proves the header, the options and every +cell token, and refuses a TSV no category names, a key that is empty or repeats, a +weight that is not a positive number, and a key or name holding `[`, `]`, `{`, `}`, +`"` or `|`, which a selector cannot spell; the rows are indexed on the first draw that +selects one. The table's options are its own — `rows`, `key`, `name`, `weight` and +`parent` — so a column may be named `name`, as one usually is. + +A choice of templates sharing one format and one set of string fields is a table +written by hand, and `New` refuses it in a data file naming the TSV to write; an +inline template has no file beside it, so there it stays a choice. + +### Row selection + +`[key]` or `[name]` after a table's name selects one row: `misc.country[SE]` and +`misc.country[Sweden]` name one row, and `misc.country[SE].capital` reads its column. +A key wins over a name that spells the same, and a name naming several rows is an +error listing their keys, unless a row selected before it settles which +([Linked tables](#linked-tables)). A selector is part of the path, so it works +wherever a path does: `Fake`, `FakeRecord`, a `{/misc.country[SE].capital}` reference +and a struct tag. A dot inside the brackets belongs to the key or name, so +`city[St. Louis]` selects it. A path starts with a name, and `[` still opens a JSON +array at the start of a CLI argument, so `'[SE]'` alone names nothing. + +### Linked tables + +A table's `parent` names a column and, by the same name, the table beside it that the +column links to by key. Save `mydata/city.tsv` and `mydata/city.json` beside the +`country` table above: + +```tsv +name country population +Copenhagen DK 660000 +Göteborg SE 600000 +Oslo NO 710000 +Stockholm SE 990000 +``` + +```json +{ "format": "{name}", "rows": "city.tsv", "key": "name", "parent": "country", "weight": "population" } +``` + +```sh +fejkdata -d ./mydata 'country[SE].city' # Stockholm or Göteborg +fejkdata -d ./mydata country.city.name # a country drawn, then a city inside it +fejkdata -d ./mydata 'city[Oslo].country' # NO — the link column's cell +fejkdata -d ./mydata '{/city.name}, {/country.name}' # Oslo, Norway — one consistent draw +``` + +A path descends from a row to a linked table by name, at any depth, and `--list` +advertises each direct step. Within one render and [draw group](#draw-group), linked +tables agree: the first table a reference path reads pins its ancestors, and a +descendant read after it is drawn inside them, so `{/city.name}` and +`{/country.name}` are a city and its country whichever is read first. A selected row +pins the render the same way, so every reference path into one family of linked +tables in one render and group selects the same rows: one that selects none beside +one that does is refused naming the spelling that does, `{/country[SE].city.name}` +beside `{/country[SE].name}`, and two selecting different rows are refused naming a +`drawGroup` to draw them apart in. A bare `{/city}` beside a path into its family is +refused too, since a bare reference draws each time. `New` also refuses a link cell +that is no key of the parent, a parent row no child links to, a chain of parents that +closes, and a child named like one of its parent's columns. + ### Options and fields `format`, `weight`, `repeat`, `separator`, `datatype` and `drawGroup` are the only options; -**any other key is a field** (see [Decisions](#decisions)). An object that does nothing a +**any other key is a field** (see [Decisions](#decisions)), and `rows` makes a category +a [table](#table), so no template carries a field of that name. An object that does nothing a string can't — only a `format` — is rejected naming the string, as is a one-item choice naming its item. @@ -470,10 +571,11 @@ its groups by name; the unnamed group spans them all. Renders e.g. `Sara Eriksson pays Ebba Lind; signed Eriksson`: the signature reads the payer's draw, while the payee is drawn apart. Rejected at load, each naming nothing: a `drawGroup` of `""` (the default); one naming the draw group its template already draws -in; one on a template that renders no reference path short of a `repeat` or a nested -`drawGroup`, on a `repeat` itself — each iteration renders in no draw group — or on an -inline template's root, which nothing references. So is a path reading into a level that -carries a `drawGroup`. +in; one on a template that renders no reference path — a bare reference to a +[table](#table) counts, since the group answers for the family it draws in — short of a +`repeat` or a nested `drawGroup`, on a `repeat` itself — each iteration renders in no +draw group — or on an inline template's root, which nothing references. So is a path +reading into a level that carries a `drawGroup`. ### Correlated fields @@ -538,7 +640,7 @@ a minor only adds, and a major is the only release that changes what exists. | Surface | Major | Minor | |---------|-------|-------| -| Shipped data | remove or rename a path; change a category's format; remove a value, or change a weight or a repeat; add a reference from one shipped category into another | a path outside a record's columns, a locale, a value in a list | +| Shipped data | remove or rename a path; change a category's format; remove a value, or change a weight or a repeat; add a reference from one shipped category into another; change a table's key, name, weight or parent column, or remove a row | a path outside a record's columns, a locale, a value in a list, a row | | Records | remove, rename, retype or add a column; let a column be null | a record, as a new category | | Data format | a fence: a spelling `New` rejects that it accepted; a template option, since it reserves a field name | a builtin | | CLI | remove or rename a flag, or change its default; change what an exit code means; change the framing a `--format` writes (header, quoting, statement shape), the `--list` layout, or what an error names | a flag, a format | @@ -559,8 +661,8 @@ with no breaking change. From `v2` the module path carries `/vN`, so fences ship batched into as few majors as possible. [`testdata/shipped_shape.txt`](testdata/shipped_shape.txt) pins every path, each -template category's format, the categories each category reads, and each column's -datatype and nullability; a pull request that +template category's format, the categories each category reads, each column's +datatype and nullability, and each table's key, name, weight and parent columns; a pull request that changes it or `data/` adds its `CHANGELOG.md` entry, which CI checks. A removed, renamed or retyped line is a major. @@ -796,6 +898,52 @@ renamed or retyped line is a major. almost always costs an allocation too (a lost pre-size, a per-item map, an extra copy). The benchmark suite (see Development) reports time for a human, not as a pass/fail gate. +- **Rows live in a TSV, the shape in JSON.** `New` allocates once per node, so a + register of thirty thousand rows written as JSON objects would cost it a second; + a TSV is one allocation whose cells are substrings, and the JSON says only how a + row is composed. The TSV sits beside its category file, named by `rows`, so a + data directory stays a directory of categories, and one nothing names is a + load error rather than a file silently ignored. +- **A selector is bracketed, and a dot inside it is literal.** `municipality[0180]` + reads as selection to anyone who has indexed an array, and `[St. Louis]` keeps a + name whole where a colon or a dot-separated spelling could not; zsh needs the + brackets quoted, which the README's examples show. A key wins over a name that + spells the same, since a key names one row by contract and a name may not. +- **A table read into is pinned; a table rendered whole draws afresh.** A path into a + table pins its row for the render and group, as a reference path pins its level, + and a bare `{/city}` draws each time, as a bare reference does; so a bare table + beside a path into its family is refused like a bare reference beside a path into + it. A bare table reference still counts as a read for a `drawGroup`, since the group + is what draws it apart from the family's pins. +- **Every reference path into one family selects the same rows, per render and + group.** A selector pins rows for the render, and a read that draws freely before it + could pin a row the selector contradicts, so accepting both would make the result + depend on which token rendered first. Requiring one selection per family per group + is checkable at load with no data lookup beyond the selectors themselves, and the + error names the rewrite. Two selectors naming one row by key and by name compare + equal, since the load check resolves them. +- **A parent row with no child row is a load error.** A descendant is drawn inside the + nearest pinned ancestor, so every ancestor row must lead to a row at every level + below it, or a render could find nothing to draw. The import script drops or fills + such rows; the alternative, falling back to a free draw, would break the consistency + the link exists for without saying so. +- **The choice-of-rows fence guards a data file's root, and requires string fields.** + A table is a category with a TSV beside its file, so only a root choice has the + spelling the fence names; a nested choice of same-shaped templates and an inline + one keep loading. Fields must all be strings because a cell is a string node: a + choice whose items carry a nested choice, as `misc.car` does, is not one table but + two linked ones, which a later conversion writes. +- **A table is a record of string columns.** Its columns are the CSV header and the + `INSERT` column list, fixed by the TSV header, so a table is a record by + construction; every column is a string until a typed column option earns its place. +- **The key index is built at load, the rest on first draw.** A link is proved + against the parent's keys and a key's uniqueness is a data mistake, so both are + load-time; the name index and the per-parent child lists serve only a draw or a + selection, so they wait for the first one, keeping `New` linear in the bytes read. +- **`List` advertises direct descents only.** `region.municipality.locality` is + listed, and `region.locality` resolves too but is not: the set of every descent + through a chain of five tables is every subsequence of it, and the direct chain is + the one a reader can predict from the tables' parents. ## Development @@ -839,6 +987,15 @@ in its own commit: REPIN=1 docker compose run --rm --user "$(id -u):$(id -g)" test ``` +A shipped table built from a source is rebuilt by its script under +[`data-import/`](data-import), one command per dataset, fetching the source named in +[`DATA-LICENSES.md`](DATA-LICENSES.md): + +```sh +docker compose run --rm --user "$(id -u):$(id -g)" data-import data-import/country.py +docker compose run --rm --user "$(id -u):$(id -g)" data-import data-import/currency.py +``` + To release, head `CHANGELOG.md` with the version's section in place of `Unreleased` and merge: once `main` passes the gate, CI tags that commit `vX.Y.Z` and publishes the Gitea release with the section as its body. A top heading of `[Unreleased]` @@ -849,7 +1006,8 @@ publishes nothing. ``` fejkdata.go Generator, New, options, the embedded data set, List node.go the node model and JSON -> node compilation -path.go the dotted-path walk, and proving a path resolves +table.go tables: the rows TSV, its options and links, row selection and draws +path.go the dotted-path walk with its selectors, and proving a path resolves render.go Fake and the recursive renderer (choices, format strings, expansions) record.go records: Record, the JSON/CSV/SQL serializers, and their entry points struct.go structs: FakeStruct, fake tags, and a field's Go type as its column's datatype @@ -865,7 +1023,8 @@ datatype.go column datatypes: DataType, where datatype and null may sit, a c value.go the value proof: what a typed column or calc operand holds, checked at load data.go data loading: fs.FS folders/files -> namespace tree, multi-source merge cmd/fejkdata/ the fejkdata CLI -data/ shipped data (JSON), embedded at build: locale folders + a misc folder +data/ shipped data (JSON, and a TSV per table), embedded at build: locale folders + a misc folder +data-import/ the scripts that rebuild each sourced table (see DATA-LICENSES.md) release-tooling/ the release CI publishes from the changelog heading testdata/ the pinned shipped shape (see Versioning) ``` diff --git a/cmd/fejkdata/main.go b/cmd/fejkdata/main.go index bb95e71..dd3d0ef 100644 --- a/cmd/fejkdata/main.go +++ b/cmd/fejkdata/main.go @@ -23,16 +23,17 @@ import ( const usage = `Usage: fejkdata [flags] - a category, or a dotted path into one (person, person.last) + a category, or a dotted path into one (person, person.last); + a table's row by key or name: 'misc.country[SE]', 'misc.country[Sweden].tld'