From 9497824350d6ffb568f14be8cd05c73211398f49 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Fri, 18 Sep 2026 20:03:54 +0200 Subject: [PATCH] Split the layout, checksum and transform declarations out of builtins.go --- builtins.go | 337 --------------------------------------------------- checksum.go | 96 +++++++++++++++ layout.go | 146 ++++++++++++++++++++++ transform.go | 94 ++++++++++++++ 4 files changed, 336 insertions(+), 337 deletions(-) create mode 100644 checksum.go create mode 100644 layout.go create mode 100644 transform.go diff --git a/builtins.go b/builtins.go index 61c568c..883f34d 100644 --- a/builtins.go +++ b/builtins.go @@ -7,8 +7,6 @@ import ( "math" "strconv" "strings" - "time" - "unicode" ) // maxLen caps sample output lengths (hex, nanoid, base64, digits, upper, lower) @@ -91,13 +89,11 @@ func derive(f func(emitted string) string) func([]string) callFn { return func(_ *session, emitted string, _ []string) string { return f(emitted) } } } - func sample(f func(rng) string) func([]string) callFn { return func([]string) callFn { return func(s *session, _ string, _ []string) string { return f(s) } } } - func chars(alphabet string) func([]string) callFn { return func(a []string) callFn { n := atoi(a[0]) @@ -116,96 +112,6 @@ func formatFloat(v float64, dp int) string { return s } -// transforms are the builtins that rewrite one operand's value; they nest, so -// {lowercase(ascii(x))} folds then lowers. -var transforms = map[string]func(string) string{ - "ascii": asciiFold, - "lowercase": strings.ToLower, - "uppercase": strings.ToUpper, -} - -// unwrapTransform peels nested transform calls off an operand arg, returning the -// field it finally names and the transforms to apply, innermost last. -func unwrapTransform(arg string) (leaf string, chain []func(string) string, err error) { - for { - name, args, isCall := funcCall(arg) - if !isCall { - return arg, chain, nil - } - fn, isTransform := transforms[name] - if !isTransform { - return "", nil, fmt.Errorf("%s(%s) is not a transform, so it cannot be an operand", name, strings.Join(args, ",")) - } - if len(args) != 1 { - return "", nil, fmt.Errorf("%s takes 1 arg, got %d", name, len(args)) - } - chain = append(chain, fn) - arg = args[0] - } -} - -func transformArg(fields map[string]node, a []string) error { - leaf, _, err := unwrapTransform(a[0]) - if err != nil { - return err - } - if isRef(leaf) { - _, _, err := refShape(leaf) - return err - } - return checkArm(leaf, fields, false) -} - -func transformOperand(a []string) []string { - leaf, _, err := unwrapTransform(a[0]) - if err != nil { - return nil - } - return []string{leaf} -} - -func transformPrep(outer func(string) string) func([]string) callFn { - return func(a []string) callFn { - _, chain, err := unwrapTransform(a[0]) - if err != nil { - panic(fmt.Sprintf("fejkdata: transform arg %q reached prep unvalidated: %v", a[0], err)) - } - return func(_ *session, _ string, operands []string) string { - v := operands[0] - for i := len(chain) - 1; i >= 0; i-- { - v = chain[i](v) - } - return outer(v) - } - } -} - -// asciiFolds maps the Latin letters with diacritics or ligatures to ASCII. -var asciiFolds = map[rune]string{ - 'À': "A", 'Á': "A", 'Â': "A", 'Ã': "A", 'Ä': "A", 'Å': "A", 'Æ': "AE", 'Ç': "C", - 'È': "E", 'É': "E", 'Ê': "E", 'Ë': "E", 'Ì': "I", 'Í': "I", 'Î': "I", 'Ï': "I", - 'Ð': "D", 'Ñ': "N", 'Ò': "O", 'Ó': "O", 'Ô': "O", 'Õ': "O", 'Ö': "O", 'Ø': "O", - 'Ù': "U", 'Ú': "U", 'Û': "U", 'Ü': "U", 'Ý': "Y", 'Þ': "Th", 'ß': "ss", 'Œ': "OE", - 'à': "a", 'á': "a", 'â': "a", 'ã': "a", 'ä': "a", 'å': "a", 'æ': "ae", 'ç': "c", - 'è': "e", 'é': "e", 'ê': "e", 'ë': "e", 'ì': "i", 'í': "i", 'î': "i", 'ï': "i", - 'ð': "d", 'ñ': "n", 'ò': "o", 'ó': "o", 'ô': "o", 'õ': "o", 'ö': "o", 'ø': "o", - 'ù': "u", 'ú': "u", 'û': "u", 'ü': "u", 'ý': "y", 'þ': "th", 'ÿ': "y", 'œ': "oe", -} - -// asciiFold rewrites s to ASCII: folded Latin letters stay, any other non-ASCII -// rune is dropped. -func asciiFold(s string) string { - var b strings.Builder - for _, r := range s { - if r <= unicode.MaxASCII { - b.WriteRune(r) - } else { - b.WriteString(asciiFolds[r]) - } - } - return b.String() -} - // atoi parses an arg a builtin's check already validated. It panics rather than // returning zero, so a check that stops covering its own args is a stack trace and // not a silently wrong length, range or decimal count. @@ -225,7 +131,6 @@ func atof(s string) float64 { } return f } - func randBytes(r rng, n int) []byte { b := make([]byte, n) for i := range b { @@ -233,7 +138,6 @@ func randBytes(r rng, n int) []byte { } return b } - func randChars(r rng, n int, alphabet string) string { b := make([]byte, n) for i := range b { @@ -256,7 +160,6 @@ func plainInt(s string) (int, error) { } return n, nil } - func posIntArg(_ map[string]node, a []string) error { n, err := plainInt(a[0]) if errors.Is(err, strconv.ErrRange) { @@ -273,7 +176,6 @@ func posIntArg(_ map[string]node, a []string) error { } return nil } - func intRangeArgs(_ map[string]node, a []string) error { lo, err := plainInt(a[0]) if err != nil { @@ -294,7 +196,6 @@ func intRangeArgs(_ map[string]node, a []string) error { } return nil } - func floatArgs(_ map[string]node, a []string) error { lo, e1 := strconv.ParseFloat(a[0], 64) hi, e2 := strconv.ParseFloat(a[1], 64) @@ -322,7 +223,6 @@ func floatArgs(_ map[string]node, a []string) error { } return nil } - func seqArg(_ map[string]node, a []string) error { if len(a) > 1 { return fmt.Errorf("seq takes at most one name, got %d args", len(a)) @@ -376,240 +276,3 @@ func ulid(r rng) string { } return string(out) } - -// luhnCheck returns the Luhn check digit (0-9) over the digits of s; non-digit -// runes are skipped. Doubling runs from the rightmost digit, so the result is -// correct whatever the payload length. -func luhnCheck(s string) int { - sum, double := 0, true - for i := len(s) - 1; i >= 0; i-- { - c := s[i] - if c < '0' || c > '9' { - continue - } - d := int(c - '0') - if double { - if d *= 2; d > 9 { - d -= 9 - } - } - double = !double - sum += d - } - return (10 - sum%10) % 10 -} - -// mod11Check returns the weighted mod-11 check character over the digits of s -// (weights 2..7 cycling from the right). A would-be value of 10 emits 'X', as in -// ISBN-10 / ISO 7064; non-digits are skipped. -func mod11Check(s string) string { - sum, w := 0, 2 - for i := len(s) - 1; i >= 0; i-- { - c := s[i] - if c < '0' || c > '9' { - continue - } - sum += int(c-'0') * w - if w++; w > 7 { - w = 2 - } - } - if chk := (11 - sum%11) % 11; chk != 10 { - return string(rune('0' + chk)) - } - return "X" -} - -// eanCheck returns the EAN-13 / UPC-A / ISBN-13 / GTIN check digit over the -// digits of s: weights 3 and 1 alternating from the rightmost digit, mod 10. -func eanCheck(s string) string { - sum, w := 0, 3 - for i := len(s) - 1; i >= 0; i-- { - c := s[i] - if c < '0' || c > '9' { - continue - } - sum += int(c-'0') * w - w = 4 - w // 3 <-> 1 - } - return string(rune('0' + (10-sum%10)%10)) -} - -// ibanLen maps a supported country code to the full IBAN length. The check digits -// sit between the country code and the BBAN, so — unlike luhn/ean — iban can't be -// a left-to-right derivation; it generates the whole value instead. -var ibanLen = map[string]int{"BE": 16, "DE": 22, "DK": 18, "ES": 24, "FI": 18, "NO": 15, "SE": 24} - -func ibanArg(_ map[string]node, a []string) error { - if _, ok := ibanLen[a[0]]; !ok { - return fmt.Errorf("iban(%q): unsupported country code", a[0]) - } - return nil -} - -// iban generates a structurally valid IBAN for cc: a numeric BBAN of the right -// length, then mod-97 check digits. Real bank/branch structure isn't modelled — -// the result passes length and checksum validation, which is what fake data needs. -func iban(r rng, cc string) string { - bban := make([]byte, ibanLen[cc]-4) - for i := range bban { - bban[i] = byte('0' + r.IntN(10)) - } - rem := 0 - feed := func(d int) { rem = (rem*10 + d) % 97 } - for _, c := range bban { - feed(int(c - '0')) - } - for i := 0; i < len(cc); i++ { // letters A-Z -> 10..35, fed as two digits - v := int(cc[i]-'A') + 10 - feed(v / 10) - feed(v % 10) - } - feed(0) - feed(0) - return fmt.Sprintf("%s%02d%s", cc, 98-rem, bban) -} - -const dayLayout = "2006-01-02" - -// The instants a layout is proved against: layoutProbe2 is alike in no field, while -// layoutDay differs from layoutProbe in its date fields alone and layoutClock in its -// clock fields alone, so formatting two of them tells which kind a layout names. -var ( - layoutProbe = time.Date(2001, 2, 3, 4, 5, 6, 0, time.UTC) - layoutProbe2 = time.Date(2010, 11, 12, 13, 14, 15, 0, time.UTC) - layoutDay = time.Date(2010, 11, 12, 4, 5, 6, 0, time.UTC) - layoutClock = time.Date(2001, 2, 3, 13, 14, 15, 0, time.UTC) -) - -func namesAField(layout string) bool { - return layoutProbe.Format(layout) != layoutProbe2.Format(layout) -} - -func namesADateField(layout string) bool { - return layoutProbe.Format(layout) != layoutDay.Format(layout) -} - -func namesAClockField(layout string) bool { - return layoutProbe.Format(layout) != layoutClock.Format(layout) -} - -// quotedLayout reports whether an arg carries the single quotes a layout is written in. -func quotedLayout(a string) bool { - return len(a) >= 2 && a[0] == '\'' && a[len(a)-1] == '\'' -} - -// layoutArg is the Go layout a quoted arg holds, refused when unquoted or constant. -func layoutArg(a string) (string, error) { - if !quotedLayout(a) { - bare := strings.Trim(a, `'"`) - if strings.HasPrefix(a, `"`) || strings.HasSuffix(a, `"`) { - return "", fmt.Errorf("layout %s is double-quoted; write '%s'", a, bare) - } - return "", fmt.Errorf("layout %s is not quoted; write '%s'", a, bare) - } - layout := a[1 : len(a)-1] - if !namesAField(layout) { - return "", fmt.Errorf("layout '%s' names no field, so it is the constant %q; write it as text", layout, layout) - } - return layout, nil -} - -// layoutOf is layoutArg for an arg a check already validated. -func layoutOf(a string) string { - layout, err := layoutArg(a) - if err != nil { - panic(fmt.Sprintf("fejkdata: builtin arg %q reached prep unvalidated: %v", a, err)) - } - return layout -} - -// layoutArity checks a call ending in a layout takes n args, naming the quoted -// layout where an unquoted one split into more. -func layoutArity(name string, n int, a []string) error { - if len(a) == n { - return nil - } - hint := "" - if len(a) > n && !holdsQuotedLayout(a[n-1:]) { - hint = fmt.Sprintf("; a layout holding a comma is quoted: '%s'", strings.Join(a[n-1:], ", ")) - } - return fmt.Errorf("%s takes %d argument%s, got %d%s", name, n, plural(n), len(a), hint) -} - -// holdsQuotedLayout reports whether the surplus args already carry a quoted layout, -// which no comma split apart. -func holdsQuotedLayout(a []string) bool { - for _, arg := range a { - if quotedLayout(arg) { - return true - } - } - return false -} - -func dateArgs(_ map[string]node, a []string) error { - if err := layoutArity("date", 3, a); err != nil { - return err - } - from, err := time.Parse(dayLayout, a[0]) - if err != nil { - return fmt.Errorf("date(from,to,layout): from %q is not a YYYY-MM-DD date", a[0]) - } - to, err := time.Parse(dayLayout, a[1]) - if err != nil { - return fmt.Errorf("date(from,to,layout): to %q is not a YYYY-MM-DD date", a[1]) - } - if to.Before(from) { - return fmt.Errorf("date(from,to,layout): from %s is after to %s", a[0], a[1]) - } - layout, err := layoutArg(a[2]) - if err != nil { - return err - } - if !namesADateField(layout) { - return fmt.Errorf("date(from,to,layout): '%s' names no date field; write time('%s')", layout, layout) - } - if from.Equal(to) && !namesAClockField(layout) { - return fmt.Errorf("date(%s,%s,'%s') is the constant %q; write it as text", a[0], a[1], layout, from.Format(layout)) - } - return nil -} - -func timeArg(_ map[string]node, a []string) error { - if err := layoutArity("time", 1, a); err != nil { - return err - } - layout, err := layoutArg(a[0]) - if err != nil { - return err - } - if namesADateField(layout) { - return fmt.Errorf("time(layout): '%s' names a date field; write date(from,to,layout)", layout) - } - return nil -} - -// datePrep draws a second in [from 00:00:00, to 23:59:59] UTC; the span is counted -// in seconds, since a Duration overflows past 292 years. -func datePrep(a []string) callFn { - from, err := time.Parse(dayLayout, a[0]) - if err != nil { - panic(fmt.Sprintf("fejkdata: builtin arg %q reached prep unvalidated: %v", a[0], err)) - } - to, err := time.Parse(dayLayout, a[1]) - if err != nil { - panic(fmt.Sprintf("fejkdata: builtin arg %q reached prep unvalidated: %v", a[1], err)) - } - layout, start, span := layoutOf(a[2]), from.Unix(), int(to.Unix()-from.Unix())+86400 - return func(s *session, _ string, _ []string) string { - return time.Unix(start+int64(s.IntN(span)), 0).UTC().Format(layout) - } -} - -func timePrep(a []string) callFn { - layout := layoutOf(a[0]) - return func(s *session, _ string, _ []string) string { - return time.Unix(int64(s.IntN(86400)), 0).UTC().Format(layout) - } -} diff --git a/checksum.go b/checksum.go new file mode 100644 index 0000000..8029f75 --- /dev/null +++ b/checksum.go @@ -0,0 +1,96 @@ +package fejkdata + +import "fmt" + +// luhnCheck returns the Luhn check digit (0-9) over the digits of s; non-digit +// runes are skipped. Doubling runs from the rightmost digit, so the result is +// correct whatever the payload length. +func luhnCheck(s string) int { + sum, double := 0, true + for i := len(s) - 1; i >= 0; i-- { + c := s[i] + if c < '0' || c > '9' { + continue + } + d := int(c - '0') + if double { + if d *= 2; d > 9 { + d -= 9 + } + } + double = !double + sum += d + } + return (10 - sum%10) % 10 +} + +// mod11Check returns the weighted mod-11 check character over the digits of s +// (weights 2..7 cycling from the right). A would-be value of 10 emits 'X', as in +// ISBN-10 / ISO 7064; non-digits are skipped. +func mod11Check(s string) string { + sum, w := 0, 2 + for i := len(s) - 1; i >= 0; i-- { + c := s[i] + if c < '0' || c > '9' { + continue + } + sum += int(c-'0') * w + if w++; w > 7 { + w = 2 + } + } + if chk := (11 - sum%11) % 11; chk != 10 { + return string(rune('0' + chk)) + } + return "X" +} + +// eanCheck returns the EAN-13 / UPC-A / ISBN-13 / GTIN check digit over the +// digits of s: weights 3 and 1 alternating from the rightmost digit, mod 10. +func eanCheck(s string) string { + sum, w := 0, 3 + for i := len(s) - 1; i >= 0; i-- { + c := s[i] + if c < '0' || c > '9' { + continue + } + sum += int(c-'0') * w + w = 4 - w // 3 <-> 1 + } + return string(rune('0' + (10-sum%10)%10)) +} + +// ibanLen maps a supported country code to the full IBAN length. The check digits +// sit between the country code and the BBAN, so — unlike luhn/ean — iban can't be +// a left-to-right derivation; it generates the whole value instead. +var ibanLen = map[string]int{"BE": 16, "DE": 22, "DK": 18, "ES": 24, "FI": 18, "NO": 15, "SE": 24} + +func ibanArg(_ map[string]node, a []string) error { + if _, ok := ibanLen[a[0]]; !ok { + return fmt.Errorf("iban(%q): unsupported country code", a[0]) + } + return nil +} + +// iban generates a structurally valid IBAN for cc: a numeric BBAN of the right +// length, then mod-97 check digits. Real bank/branch structure isn't modelled — +// the result passes length and checksum validation, which is what fake data needs. +func iban(r rng, cc string) string { + bban := make([]byte, ibanLen[cc]-4) + for i := range bban { + bban[i] = byte('0' + r.IntN(10)) + } + rem := 0 + feed := func(d int) { rem = (rem*10 + d) % 97 } + for _, c := range bban { + feed(int(c - '0')) + } + for i := 0; i < len(cc); i++ { // letters A-Z -> 10..35, fed as two digits + v := int(cc[i]-'A') + 10 + feed(v / 10) + feed(v % 10) + } + feed(0) + feed(0) + return fmt.Sprintf("%s%02d%s", cc, 98-rem, bban) +} diff --git a/layout.go b/layout.go new file mode 100644 index 0000000..a8a03c2 --- /dev/null +++ b/layout.go @@ -0,0 +1,146 @@ +package fejkdata + +import ( + "fmt" + "strings" + "time" +) + +const dayLayout = "2006-01-02" + +// The instants a layout is proved against: layoutProbe2 is alike in no field, while +// layoutDay differs from layoutProbe in its date fields alone and layoutClock in its +// clock fields alone, so formatting two of them tells which kind a layout names. +var ( + layoutProbe = time.Date(2001, 2, 3, 4, 5, 6, 0, time.UTC) + layoutProbe2 = time.Date(2010, 11, 12, 13, 14, 15, 0, time.UTC) + layoutDay = time.Date(2010, 11, 12, 4, 5, 6, 0, time.UTC) + layoutClock = time.Date(2001, 2, 3, 13, 14, 15, 0, time.UTC) +) + +func namesAField(layout string) bool { + return layoutProbe.Format(layout) != layoutProbe2.Format(layout) +} +func namesADateField(layout string) bool { + return layoutProbe.Format(layout) != layoutDay.Format(layout) +} +func namesAClockField(layout string) bool { + return layoutProbe.Format(layout) != layoutClock.Format(layout) +} + +// quotedLayout reports whether an arg carries the single quotes a layout is written in. +func quotedLayout(a string) bool { + return len(a) >= 2 && a[0] == '\'' && a[len(a)-1] == '\'' +} + +// layoutArg is the Go layout a quoted arg holds, refused when unquoted or constant. +func layoutArg(a string) (string, error) { + if !quotedLayout(a) { + bare := strings.Trim(a, `'"`) + if strings.HasPrefix(a, `"`) || strings.HasSuffix(a, `"`) { + return "", fmt.Errorf("layout %s is double-quoted; write '%s'", a, bare) + } + return "", fmt.Errorf("layout %s is not quoted; write '%s'", a, bare) + } + layout := a[1 : len(a)-1] + if !namesAField(layout) { + return "", fmt.Errorf("layout '%s' names no field, so it is the constant %q; write it as text", layout, layout) + } + return layout, nil +} + +// layoutOf is layoutArg for an arg a check already validated. +func layoutOf(a string) string { + layout, err := layoutArg(a) + if err != nil { + panic(fmt.Sprintf("fejkdata: builtin arg %q reached prep unvalidated: %v", a, err)) + } + return layout +} + +// layoutArity checks a call ending in a layout takes n args, naming the quoted +// layout where an unquoted one split into more. +func layoutArity(name string, n int, a []string) error { + if len(a) == n { + return nil + } + hint := "" + if len(a) > n && !holdsQuotedLayout(a[n-1:]) { + hint = fmt.Sprintf("; a layout holding a comma is quoted: '%s'", strings.Join(a[n-1:], ", ")) + } + return fmt.Errorf("%s takes %d argument%s, got %d%s", name, n, plural(n), len(a), hint) +} + +// holdsQuotedLayout reports whether the surplus args already carry a quoted layout, +// which no comma split apart. +func holdsQuotedLayout(a []string) bool { + for _, arg := range a { + if quotedLayout(arg) { + return true + } + } + return false +} +func dateArgs(_ map[string]node, a []string) error { + if err := layoutArity("date", 3, a); err != nil { + return err + } + from, err := time.Parse(dayLayout, a[0]) + if err != nil { + return fmt.Errorf("date(from,to,layout): from %q is not a YYYY-MM-DD date", a[0]) + } + to, err := time.Parse(dayLayout, a[1]) + if err != nil { + return fmt.Errorf("date(from,to,layout): to %q is not a YYYY-MM-DD date", a[1]) + } + if to.Before(from) { + return fmt.Errorf("date(from,to,layout): from %s is after to %s", a[0], a[1]) + } + layout, err := layoutArg(a[2]) + if err != nil { + return err + } + if !namesADateField(layout) { + return fmt.Errorf("date(from,to,layout): '%s' names no date field; write time('%s')", layout, layout) + } + if from.Equal(to) && !namesAClockField(layout) { + return fmt.Errorf("date(%s,%s,'%s') is the constant %q; write it as text", a[0], a[1], layout, from.Format(layout)) + } + return nil +} +func timeArg(_ map[string]node, a []string) error { + if err := layoutArity("time", 1, a); err != nil { + return err + } + layout, err := layoutArg(a[0]) + if err != nil { + return err + } + if namesADateField(layout) { + return fmt.Errorf("time(layout): '%s' names a date field; write date(from,to,layout)", layout) + } + return nil +} + +// datePrep draws a second in [from 00:00:00, to 23:59:59] UTC; the span is counted +// in seconds, since a Duration overflows past 292 years. +func datePrep(a []string) callFn { + from, err := time.Parse(dayLayout, a[0]) + if err != nil { + panic(fmt.Sprintf("fejkdata: builtin arg %q reached prep unvalidated: %v", a[0], err)) + } + to, err := time.Parse(dayLayout, a[1]) + if err != nil { + panic(fmt.Sprintf("fejkdata: builtin arg %q reached prep unvalidated: %v", a[1], err)) + } + layout, start, span := layoutOf(a[2]), from.Unix(), int(to.Unix()-from.Unix())+86400 + return func(s *session, _ string, _ []string) string { + return time.Unix(start+int64(s.IntN(span)), 0).UTC().Format(layout) + } +} +func timePrep(a []string) callFn { + layout := layoutOf(a[0]) + return func(s *session, _ string, _ []string) string { + return time.Unix(int64(s.IntN(86400)), 0).UTC().Format(layout) + } +} diff --git a/transform.go b/transform.go new file mode 100644 index 0000000..872917b --- /dev/null +++ b/transform.go @@ -0,0 +1,94 @@ +package fejkdata + +import ( + "fmt" + "strings" + "unicode" +) + +// transforms are the builtins that rewrite one operand's value; they nest, so +// {lowercase(ascii(x))} folds then lowers. +var transforms = map[string]func(string) string{ + "ascii": asciiFold, + "lowercase": strings.ToLower, + "uppercase": strings.ToUpper, +} + +// unwrapTransform peels nested transform calls off an operand arg, returning the +// field it finally names and the transforms to apply, innermost last. +func unwrapTransform(arg string) (leaf string, chain []func(string) string, err error) { + for { + name, args, isCall := funcCall(arg) + if !isCall { + return arg, chain, nil + } + fn, isTransform := transforms[name] + if !isTransform { + return "", nil, fmt.Errorf("%s(%s) is not a transform, so it cannot be an operand", name, strings.Join(args, ",")) + } + if len(args) != 1 { + return "", nil, fmt.Errorf("%s takes 1 arg, got %d", name, len(args)) + } + chain = append(chain, fn) + arg = args[0] + } +} +func transformArg(fields map[string]node, a []string) error { + leaf, _, err := unwrapTransform(a[0]) + if err != nil { + return err + } + if isRef(leaf) { + _, _, err := refShape(leaf) + return err + } + return checkArm(leaf, fields, false) +} +func transformOperand(a []string) []string { + leaf, _, err := unwrapTransform(a[0]) + if err != nil { + return nil + } + return []string{leaf} +} +func transformPrep(outer func(string) string) func([]string) callFn { + return func(a []string) callFn { + _, chain, err := unwrapTransform(a[0]) + if err != nil { + panic(fmt.Sprintf("fejkdata: transform arg %q reached prep unvalidated: %v", a[0], err)) + } + return func(_ *session, _ string, operands []string) string { + v := operands[0] + for i := len(chain) - 1; i >= 0; i-- { + v = chain[i](v) + } + return outer(v) + } + } +} + +// asciiFolds maps the Latin letters with diacritics or ligatures to ASCII. +var asciiFolds = map[rune]string{ + 'À': "A", 'Á': "A", 'Â': "A", 'Ã': "A", 'Ä': "A", 'Å': "A", 'Æ': "AE", 'Ç': "C", + 'È': "E", 'É': "E", 'Ê': "E", 'Ë': "E", 'Ì': "I", 'Í': "I", 'Î': "I", 'Ï': "I", + 'Ð': "D", 'Ñ': "N", 'Ò': "O", 'Ó': "O", 'Ô': "O", 'Õ': "O", 'Ö': "O", 'Ø': "O", + 'Ù': "U", 'Ú': "U", 'Û': "U", 'Ü': "U", 'Ý': "Y", 'Þ': "Th", 'ß': "ss", 'Œ': "OE", + 'à': "a", 'á': "a", 'â': "a", 'ã': "a", 'ä': "a", 'å': "a", 'æ': "ae", 'ç': "c", + 'è': "e", 'é': "e", 'ê': "e", 'ë': "e", 'ì': "i", 'í': "i", 'î': "i", 'ï': "i", + 'ð': "d", 'ñ': "n", 'ò': "o", 'ó': "o", 'ô': "o", 'õ': "o", 'ö': "o", 'ø': "o", + 'ù': "u", 'ú': "u", 'û': "u", 'ü': "u", 'ý': "y", 'þ': "th", 'ÿ': "y", 'œ': "oe", +} + +// asciiFold rewrites s to ASCII: folded Latin letters stay, any other non-ASCII +// rune is dropped. +func asciiFold(s string) string { + var b strings.Builder + for _, r := range s { + if r <= unicode.MaxASCII { + b.WriteRune(r) + } else { + b.WriteString(asciiFolds[r]) + } + } + return b.String() +}