Weighted person names, valid personal ids and date() in both locales #21

Merged
lilleman merged 38 commits from person-ids-date into main 2026-09-18 20:45:51 +02:00
4 changed files with 336 additions and 337 deletions
Showing only changes of commit 9497824350 - Show all commits
-337
View File
@@ -7,8 +7,6 @@ import (
"math" "math"
"strconv" "strconv"
"strings" "strings"
"time"
"unicode"
) )
// maxLen caps sample output lengths (hex, nanoid, base64, digits, upper, lower) // maxLen caps sample output lengths (hex, nanoid, base64, digits, upper, lower)
@@ -91,13 +89,11 @@ func derive(f func(emitted string) string) func([]string) callFn {
return func(_ *session, emitted string, _ []string) string { return f(emitted) } return func(_ *session, emitted string, _ []string) string { return f(emitted) }
} }
} }
func sample(f func(rng) string) func([]string) callFn { func sample(f func(rng) string) func([]string) callFn {
return func([]string) callFn { return func([]string) callFn {
return func(s *session, _ string, _ []string) string { return f(s) } return func(s *session, _ string, _ []string) string { return f(s) }
} }
} }
func chars(alphabet string) func([]string) callFn { func chars(alphabet string) func([]string) callFn {
return func(a []string) callFn { return func(a []string) callFn {
n := atoi(a[0]) n := atoi(a[0])
@@ -116,96 +112,6 @@ func formatFloat(v float64, dp int) string {
return s return s
} }
// transforms are the builtins that rewrite one operand's value; they nest, so
// {lowercase(ascii(x))} folds then lowers.
var transforms = map[string]func(string) string{
"ascii": asciiFold,
"lowercase": strings.ToLower,
"uppercase": strings.ToUpper,
}
// unwrapTransform peels nested transform calls off an operand arg, returning the
// field it finally names and the transforms to apply, innermost last.
func unwrapTransform(arg string) (leaf string, chain []func(string) string, err error) {
for {
name, args, isCall := funcCall(arg)
if !isCall {
return arg, chain, nil
}
fn, isTransform := transforms[name]
if !isTransform {
return "", nil, fmt.Errorf("%s(%s) is not a transform, so it cannot be an operand", name, strings.Join(args, ","))
}
if len(args) != 1 {
return "", nil, fmt.Errorf("%s takes 1 arg, got %d", name, len(args))
}
chain = append(chain, fn)
arg = args[0]
}
}
func transformArg(fields map[string]node, a []string) error {
leaf, _, err := unwrapTransform(a[0])
if err != nil {
return err
}
if isRef(leaf) {
_, _, err := refShape(leaf)
return err
}
return checkArm(leaf, fields, false)
}
func transformOperand(a []string) []string {
leaf, _, err := unwrapTransform(a[0])
if err != nil {
return nil
}
return []string{leaf}
}
func transformPrep(outer func(string) string) func([]string) callFn {
return func(a []string) callFn {
_, chain, err := unwrapTransform(a[0])
if err != nil {
panic(fmt.Sprintf("fejkdata: transform arg %q reached prep unvalidated: %v", a[0], err))
}
return func(_ *session, _ string, operands []string) string {
v := operands[0]
for i := len(chain) - 1; i >= 0; i-- {
v = chain[i](v)
}
return outer(v)
}
}
}
// asciiFolds maps the Latin letters with diacritics or ligatures to ASCII.
var asciiFolds = map[rune]string{
'À': "A", 'Á': "A", 'Â': "A", 'Ã': "A", 'Ä': "A", 'Å': "A", 'Æ': "AE", 'Ç': "C",
'È': "E", 'É': "E", 'Ê': "E", 'Ë': "E", 'Ì': "I", 'Í': "I", 'Î': "I", 'Ï': "I",
'Ð': "D", 'Ñ': "N", 'Ò': "O", 'Ó': "O", 'Ô': "O", 'Õ': "O", 'Ö': "O", 'Ø': "O",
'Ù': "U", 'Ú': "U", 'Û': "U", 'Ü': "U", 'Ý': "Y", 'Þ': "Th", 'ß': "ss", 'Œ': "OE",
'à': "a", 'á': "a", 'â': "a", 'ã': "a", 'ä': "a", 'å': "a", 'æ': "ae", 'ç': "c",
'è': "e", 'é': "e", 'ê': "e", 'ë': "e", 'ì': "i", 'í': "i", 'î': "i", 'ï': "i",
'ð': "d", 'ñ': "n", 'ò': "o", 'ó': "o", 'ô': "o", 'õ': "o", 'ö': "o", 'ø': "o",
'ù': "u", 'ú': "u", 'û': "u", 'ü': "u", 'ý': "y", 'þ': "th", 'ÿ': "y", 'œ': "oe",
}
// asciiFold rewrites s to ASCII: folded Latin letters stay, any other non-ASCII
// rune is dropped.
func asciiFold(s string) string {
var b strings.Builder
for _, r := range s {
if r <= unicode.MaxASCII {
b.WriteRune(r)
} else {
b.WriteString(asciiFolds[r])
}
}
return b.String()
}
// atoi parses an arg a builtin's check already validated. It panics rather than // atoi parses an arg a builtin's check already validated. It panics rather than
// returning zero, so a check that stops covering its own args is a stack trace and // returning zero, so a check that stops covering its own args is a stack trace and
// not a silently wrong length, range or decimal count. // not a silently wrong length, range or decimal count.
@@ -225,7 +131,6 @@ func atof(s string) float64 {
} }
return f return f
} }
func randBytes(r rng, n int) []byte { func randBytes(r rng, n int) []byte {
b := make([]byte, n) b := make([]byte, n)
for i := range b { for i := range b {
@@ -233,7 +138,6 @@ func randBytes(r rng, n int) []byte {
} }
return b return b
} }
func randChars(r rng, n int, alphabet string) string { func randChars(r rng, n int, alphabet string) string {
b := make([]byte, n) b := make([]byte, n)
for i := range b { for i := range b {
@@ -256,7 +160,6 @@ func plainInt(s string) (int, error) {
} }
return n, nil return n, nil
} }
func posIntArg(_ map[string]node, a []string) error { func posIntArg(_ map[string]node, a []string) error {
n, err := plainInt(a[0]) n, err := plainInt(a[0])
if errors.Is(err, strconv.ErrRange) { if errors.Is(err, strconv.ErrRange) {
@@ -273,7 +176,6 @@ func posIntArg(_ map[string]node, a []string) error {
} }
return nil return nil
} }
func intRangeArgs(_ map[string]node, a []string) error { func intRangeArgs(_ map[string]node, a []string) error {
lo, err := plainInt(a[0]) lo, err := plainInt(a[0])
if err != nil { if err != nil {
@@ -294,7 +196,6 @@ func intRangeArgs(_ map[string]node, a []string) error {
} }
return nil return nil
} }
func floatArgs(_ map[string]node, a []string) error { func floatArgs(_ map[string]node, a []string) error {
lo, e1 := strconv.ParseFloat(a[0], 64) lo, e1 := strconv.ParseFloat(a[0], 64)
hi, e2 := strconv.ParseFloat(a[1], 64) hi, e2 := strconv.ParseFloat(a[1], 64)
@@ -322,7 +223,6 @@ func floatArgs(_ map[string]node, a []string) error {
} }
return nil return nil
} }
func seqArg(_ map[string]node, a []string) error { func seqArg(_ map[string]node, a []string) error {
if len(a) > 1 { if len(a) > 1 {
return fmt.Errorf("seq takes at most one name, got %d args", len(a)) return fmt.Errorf("seq takes at most one name, got %d args", len(a))
@@ -376,240 +276,3 @@ func ulid(r rng) string {
} }
return string(out) return string(out)
} }
// luhnCheck returns the Luhn check digit (0-9) over the digits of s; non-digit
// runes are skipped. Doubling runs from the rightmost digit, so the result is
// correct whatever the payload length.
func luhnCheck(s string) int {
sum, double := 0, true
for i := len(s) - 1; i >= 0; i-- {
c := s[i]
if c < '0' || c > '9' {
continue
}
d := int(c - '0')
if double {
if d *= 2; d > 9 {
d -= 9
}
}
double = !double
sum += d
}
return (10 - sum%10) % 10
}
// mod11Check returns the weighted mod-11 check character over the digits of s
// (weights 2..7 cycling from the right). A would-be value of 10 emits 'X', as in
// ISBN-10 / ISO 7064; non-digits are skipped.
func mod11Check(s string) string {
sum, w := 0, 2
for i := len(s) - 1; i >= 0; i-- {
c := s[i]
if c < '0' || c > '9' {
continue
}
sum += int(c-'0') * w
if w++; w > 7 {
w = 2
}
}
if chk := (11 - sum%11) % 11; chk != 10 {
return string(rune('0' + chk))
}
return "X"
}
// eanCheck returns the EAN-13 / UPC-A / ISBN-13 / GTIN check digit over the
// digits of s: weights 3 and 1 alternating from the rightmost digit, mod 10.
func eanCheck(s string) string {
sum, w := 0, 3
for i := len(s) - 1; i >= 0; i-- {
c := s[i]
if c < '0' || c > '9' {
continue
}
sum += int(c-'0') * w
w = 4 - w // 3 <-> 1
}
return string(rune('0' + (10-sum%10)%10))
}
// ibanLen maps a supported country code to the full IBAN length. The check digits
// sit between the country code and the BBAN, so — unlike luhn/ean — iban can't be
// a left-to-right derivation; it generates the whole value instead.
var ibanLen = map[string]int{"BE": 16, "DE": 22, "DK": 18, "ES": 24, "FI": 18, "NO": 15, "SE": 24}
func ibanArg(_ map[string]node, a []string) error {
if _, ok := ibanLen[a[0]]; !ok {
return fmt.Errorf("iban(%q): unsupported country code", a[0])
}
return nil
}
// iban generates a structurally valid IBAN for cc: a numeric BBAN of the right
// length, then mod-97 check digits. Real bank/branch structure isn't modelled —
// the result passes length and checksum validation, which is what fake data needs.
func iban(r rng, cc string) string {
bban := make([]byte, ibanLen[cc]-4)
for i := range bban {
bban[i] = byte('0' + r.IntN(10))
}
rem := 0
feed := func(d int) { rem = (rem*10 + d) % 97 }
for _, c := range bban {
feed(int(c - '0'))
}
for i := 0; i < len(cc); i++ { // letters A-Z -> 10..35, fed as two digits
v := int(cc[i]-'A') + 10
feed(v / 10)
feed(v % 10)
}
feed(0)
feed(0)
return fmt.Sprintf("%s%02d%s", cc, 98-rem, bban)
}
const dayLayout = "2006-01-02"
// The instants a layout is proved against: layoutProbe2 is alike in no field, while
// layoutDay differs from layoutProbe in its date fields alone and layoutClock in its
// clock fields alone, so formatting two of them tells which kind a layout names.
var (
layoutProbe = time.Date(2001, 2, 3, 4, 5, 6, 0, time.UTC)
layoutProbe2 = time.Date(2010, 11, 12, 13, 14, 15, 0, time.UTC)
layoutDay = time.Date(2010, 11, 12, 4, 5, 6, 0, time.UTC)
layoutClock = time.Date(2001, 2, 3, 13, 14, 15, 0, time.UTC)
)
func namesAField(layout string) bool {
return layoutProbe.Format(layout) != layoutProbe2.Format(layout)
}
func namesADateField(layout string) bool {
return layoutProbe.Format(layout) != layoutDay.Format(layout)
}
func namesAClockField(layout string) bool {
return layoutProbe.Format(layout) != layoutClock.Format(layout)
}
// quotedLayout reports whether an arg carries the single quotes a layout is written in.
func quotedLayout(a string) bool {
return len(a) >= 2 && a[0] == '\'' && a[len(a)-1] == '\''
}
// layoutArg is the Go layout a quoted arg holds, refused when unquoted or constant.
func layoutArg(a string) (string, error) {
if !quotedLayout(a) {
bare := strings.Trim(a, `'"`)
if strings.HasPrefix(a, `"`) || strings.HasSuffix(a, `"`) {
return "", fmt.Errorf("layout %s is double-quoted; write '%s'", a, bare)
}
return "", fmt.Errorf("layout %s is not quoted; write '%s'", a, bare)
}
layout := a[1 : len(a)-1]
if !namesAField(layout) {
return "", fmt.Errorf("layout '%s' names no field, so it is the constant %q; write it as text", layout, layout)
}
return layout, nil
}
// layoutOf is layoutArg for an arg a check already validated.
func layoutOf(a string) string {
layout, err := layoutArg(a)
if err != nil {
panic(fmt.Sprintf("fejkdata: builtin arg %q reached prep unvalidated: %v", a, err))
}
return layout
}
// layoutArity checks a call ending in a layout takes n args, naming the quoted
// layout where an unquoted one split into more.
func layoutArity(name string, n int, a []string) error {
if len(a) == n {
return nil
}
hint := ""
if len(a) > n && !holdsQuotedLayout(a[n-1:]) {
hint = fmt.Sprintf("; a layout holding a comma is quoted: '%s'", strings.Join(a[n-1:], ", "))
}
return fmt.Errorf("%s takes %d argument%s, got %d%s", name, n, plural(n), len(a), hint)
}
// holdsQuotedLayout reports whether the surplus args already carry a quoted layout,
// which no comma split apart.
func holdsQuotedLayout(a []string) bool {
for _, arg := range a {
if quotedLayout(arg) {
return true
}
}
return false
}
func dateArgs(_ map[string]node, a []string) error {
if err := layoutArity("date", 3, a); err != nil {
return err
}
from, err := time.Parse(dayLayout, a[0])
if err != nil {
return fmt.Errorf("date(from,to,layout): from %q is not a YYYY-MM-DD date", a[0])
}
to, err := time.Parse(dayLayout, a[1])
if err != nil {
return fmt.Errorf("date(from,to,layout): to %q is not a YYYY-MM-DD date", a[1])
}
if to.Before(from) {
return fmt.Errorf("date(from,to,layout): from %s is after to %s", a[0], a[1])
}
layout, err := layoutArg(a[2])
if err != nil {
return err
}
if !namesADateField(layout) {
return fmt.Errorf("date(from,to,layout): '%s' names no date field; write time('%s')", layout, layout)
}
if from.Equal(to) && !namesAClockField(layout) {
return fmt.Errorf("date(%s,%s,'%s') is the constant %q; write it as text", a[0], a[1], layout, from.Format(layout))
}
return nil
}
func timeArg(_ map[string]node, a []string) error {
if err := layoutArity("time", 1, a); err != nil {
return err
}
layout, err := layoutArg(a[0])
if err != nil {
return err
}
if namesADateField(layout) {
return fmt.Errorf("time(layout): '%s' names a date field; write date(from,to,layout)", layout)
}
return nil
}
// datePrep draws a second in [from 00:00:00, to 23:59:59] UTC; the span is counted
// in seconds, since a Duration overflows past 292 years.
func datePrep(a []string) callFn {
from, err := time.Parse(dayLayout, a[0])
if err != nil {
panic(fmt.Sprintf("fejkdata: builtin arg %q reached prep unvalidated: %v", a[0], err))
}
to, err := time.Parse(dayLayout, a[1])
if err != nil {
panic(fmt.Sprintf("fejkdata: builtin arg %q reached prep unvalidated: %v", a[1], err))
}
layout, start, span := layoutOf(a[2]), from.Unix(), int(to.Unix()-from.Unix())+86400
return func(s *session, _ string, _ []string) string {
return time.Unix(start+int64(s.IntN(span)), 0).UTC().Format(layout)
}
}
func timePrep(a []string) callFn {
layout := layoutOf(a[0])
return func(s *session, _ string, _ []string) string {
return time.Unix(int64(s.IntN(86400)), 0).UTC().Format(layout)
}
}
+96
View File
@@ -0,0 +1,96 @@
package fejkdata
import "fmt"
// luhnCheck returns the Luhn check digit (0-9) over the digits of s; non-digit
// runes are skipped. Doubling runs from the rightmost digit, so the result is
// correct whatever the payload length.
func luhnCheck(s string) int {
sum, double := 0, true
for i := len(s) - 1; i >= 0; i-- {
c := s[i]
if c < '0' || c > '9' {
continue
}
d := int(c - '0')
if double {
if d *= 2; d > 9 {
d -= 9
}
}
double = !double
sum += d
}
return (10 - sum%10) % 10
}
// mod11Check returns the weighted mod-11 check character over the digits of s
// (weights 2..7 cycling from the right). A would-be value of 10 emits 'X', as in
// ISBN-10 / ISO 7064; non-digits are skipped.
func mod11Check(s string) string {
sum, w := 0, 2
for i := len(s) - 1; i >= 0; i-- {
c := s[i]
if c < '0' || c > '9' {
continue
}
sum += int(c-'0') * w
if w++; w > 7 {
w = 2
}
}
if chk := (11 - sum%11) % 11; chk != 10 {
return string(rune('0' + chk))
}
return "X"
}
// eanCheck returns the EAN-13 / UPC-A / ISBN-13 / GTIN check digit over the
// digits of s: weights 3 and 1 alternating from the rightmost digit, mod 10.
func eanCheck(s string) string {
sum, w := 0, 3
for i := len(s) - 1; i >= 0; i-- {
c := s[i]
if c < '0' || c > '9' {
continue
}
sum += int(c-'0') * w
w = 4 - w // 3 <-> 1
}
return string(rune('0' + (10-sum%10)%10))
}
// ibanLen maps a supported country code to the full IBAN length. The check digits
// sit between the country code and the BBAN, so — unlike luhn/ean — iban can't be
// a left-to-right derivation; it generates the whole value instead.
var ibanLen = map[string]int{"BE": 16, "DE": 22, "DK": 18, "ES": 24, "FI": 18, "NO": 15, "SE": 24}
func ibanArg(_ map[string]node, a []string) error {
if _, ok := ibanLen[a[0]]; !ok {
return fmt.Errorf("iban(%q): unsupported country code", a[0])
}
return nil
}
// iban generates a structurally valid IBAN for cc: a numeric BBAN of the right
// length, then mod-97 check digits. Real bank/branch structure isn't modelled —
// the result passes length and checksum validation, which is what fake data needs.
func iban(r rng, cc string) string {
bban := make([]byte, ibanLen[cc]-4)
for i := range bban {
bban[i] = byte('0' + r.IntN(10))
}
rem := 0
feed := func(d int) { rem = (rem*10 + d) % 97 }
for _, c := range bban {
feed(int(c - '0'))
}
for i := 0; i < len(cc); i++ { // letters A-Z -> 10..35, fed as two digits
v := int(cc[i]-'A') + 10
feed(v / 10)
feed(v % 10)
}
feed(0)
feed(0)
return fmt.Sprintf("%s%02d%s", cc, 98-rem, bban)
}
+146
View File
@@ -0,0 +1,146 @@
package fejkdata
import (
"fmt"
"strings"
"time"
)
const dayLayout = "2006-01-02"
// The instants a layout is proved against: layoutProbe2 is alike in no field, while
// layoutDay differs from layoutProbe in its date fields alone and layoutClock in its
// clock fields alone, so formatting two of them tells which kind a layout names.
var (
layoutProbe = time.Date(2001, 2, 3, 4, 5, 6, 0, time.UTC)
layoutProbe2 = time.Date(2010, 11, 12, 13, 14, 15, 0, time.UTC)
layoutDay = time.Date(2010, 11, 12, 4, 5, 6, 0, time.UTC)
layoutClock = time.Date(2001, 2, 3, 13, 14, 15, 0, time.UTC)
)
func namesAField(layout string) bool {
return layoutProbe.Format(layout) != layoutProbe2.Format(layout)
}
func namesADateField(layout string) bool {
return layoutProbe.Format(layout) != layoutDay.Format(layout)
}
func namesAClockField(layout string) bool {
return layoutProbe.Format(layout) != layoutClock.Format(layout)
}
// quotedLayout reports whether an arg carries the single quotes a layout is written in.
func quotedLayout(a string) bool {
return len(a) >= 2 && a[0] == '\'' && a[len(a)-1] == '\''
}
// layoutArg is the Go layout a quoted arg holds, refused when unquoted or constant.
func layoutArg(a string) (string, error) {
if !quotedLayout(a) {
bare := strings.Trim(a, `'"`)
if strings.HasPrefix(a, `"`) || strings.HasSuffix(a, `"`) {
return "", fmt.Errorf("layout %s is double-quoted; write '%s'", a, bare)
}
return "", fmt.Errorf("layout %s is not quoted; write '%s'", a, bare)
}
layout := a[1 : len(a)-1]
if !namesAField(layout) {
return "", fmt.Errorf("layout '%s' names no field, so it is the constant %q; write it as text", layout, layout)
}
return layout, nil
}
// layoutOf is layoutArg for an arg a check already validated.
func layoutOf(a string) string {
layout, err := layoutArg(a)
if err != nil {
panic(fmt.Sprintf("fejkdata: builtin arg %q reached prep unvalidated: %v", a, err))
}
return layout
}
// layoutArity checks a call ending in a layout takes n args, naming the quoted
// layout where an unquoted one split into more.
func layoutArity(name string, n int, a []string) error {
if len(a) == n {
return nil
}
hint := ""
if len(a) > n && !holdsQuotedLayout(a[n-1:]) {
hint = fmt.Sprintf("; a layout holding a comma is quoted: '%s'", strings.Join(a[n-1:], ", "))
}
return fmt.Errorf("%s takes %d argument%s, got %d%s", name, n, plural(n), len(a), hint)
}
// holdsQuotedLayout reports whether the surplus args already carry a quoted layout,
// which no comma split apart.
func holdsQuotedLayout(a []string) bool {
for _, arg := range a {
if quotedLayout(arg) {
return true
}
}
return false
}
func dateArgs(_ map[string]node, a []string) error {
if err := layoutArity("date", 3, a); err != nil {
return err
}
from, err := time.Parse(dayLayout, a[0])
if err != nil {
return fmt.Errorf("date(from,to,layout): from %q is not a YYYY-MM-DD date", a[0])
}
to, err := time.Parse(dayLayout, a[1])
if err != nil {
return fmt.Errorf("date(from,to,layout): to %q is not a YYYY-MM-DD date", a[1])
}
if to.Before(from) {
return fmt.Errorf("date(from,to,layout): from %s is after to %s", a[0], a[1])
}
layout, err := layoutArg(a[2])
if err != nil {
return err
}
if !namesADateField(layout) {
return fmt.Errorf("date(from,to,layout): '%s' names no date field; write time('%s')", layout, layout)
}
if from.Equal(to) && !namesAClockField(layout) {
return fmt.Errorf("date(%s,%s,'%s') is the constant %q; write it as text", a[0], a[1], layout, from.Format(layout))
}
return nil
}
func timeArg(_ map[string]node, a []string) error {
if err := layoutArity("time", 1, a); err != nil {
return err
}
layout, err := layoutArg(a[0])
if err != nil {
return err
}
if namesADateField(layout) {
return fmt.Errorf("time(layout): '%s' names a date field; write date(from,to,layout)", layout)
}
return nil
}
// datePrep draws a second in [from 00:00:00, to 23:59:59] UTC; the span is counted
// in seconds, since a Duration overflows past 292 years.
func datePrep(a []string) callFn {
from, err := time.Parse(dayLayout, a[0])
if err != nil {
panic(fmt.Sprintf("fejkdata: builtin arg %q reached prep unvalidated: %v", a[0], err))
}
to, err := time.Parse(dayLayout, a[1])
if err != nil {
panic(fmt.Sprintf("fejkdata: builtin arg %q reached prep unvalidated: %v", a[1], err))
}
layout, start, span := layoutOf(a[2]), from.Unix(), int(to.Unix()-from.Unix())+86400
return func(s *session, _ string, _ []string) string {
return time.Unix(start+int64(s.IntN(span)), 0).UTC().Format(layout)
}
}
func timePrep(a []string) callFn {
layout := layoutOf(a[0])
return func(s *session, _ string, _ []string) string {
return time.Unix(int64(s.IntN(86400)), 0).UTC().Format(layout)
}
}
+94
View File
@@ -0,0 +1,94 @@
package fejkdata
import (
"fmt"
"strings"
"unicode"
)
// transforms are the builtins that rewrite one operand's value; they nest, so
// {lowercase(ascii(x))} folds then lowers.
var transforms = map[string]func(string) string{
"ascii": asciiFold,
"lowercase": strings.ToLower,
"uppercase": strings.ToUpper,
}
// unwrapTransform peels nested transform calls off an operand arg, returning the
// field it finally names and the transforms to apply, innermost last.
func unwrapTransform(arg string) (leaf string, chain []func(string) string, err error) {
for {
name, args, isCall := funcCall(arg)
if !isCall {
return arg, chain, nil
}
fn, isTransform := transforms[name]
if !isTransform {
return "", nil, fmt.Errorf("%s(%s) is not a transform, so it cannot be an operand", name, strings.Join(args, ","))
}
if len(args) != 1 {
return "", nil, fmt.Errorf("%s takes 1 arg, got %d", name, len(args))
}
chain = append(chain, fn)
arg = args[0]
}
}
func transformArg(fields map[string]node, a []string) error {
leaf, _, err := unwrapTransform(a[0])
if err != nil {
return err
}
if isRef(leaf) {
_, _, err := refShape(leaf)
return err
}
return checkArm(leaf, fields, false)
}
func transformOperand(a []string) []string {
leaf, _, err := unwrapTransform(a[0])
if err != nil {
return nil
}
return []string{leaf}
}
func transformPrep(outer func(string) string) func([]string) callFn {
return func(a []string) callFn {
_, chain, err := unwrapTransform(a[0])
if err != nil {
panic(fmt.Sprintf("fejkdata: transform arg %q reached prep unvalidated: %v", a[0], err))
}
return func(_ *session, _ string, operands []string) string {
v := operands[0]
for i := len(chain) - 1; i >= 0; i-- {
v = chain[i](v)
}
return outer(v)
}
}
}
// asciiFolds maps the Latin letters with diacritics or ligatures to ASCII.
var asciiFolds = map[rune]string{
'À': "A", 'Á': "A", 'Â': "A", 'Ã': "A", 'Ä': "A", 'Å': "A", 'Æ': "AE", 'Ç': "C",
'È': "E", 'É': "E", 'Ê': "E", 'Ë': "E", 'Ì': "I", 'Í': "I", 'Î': "I", 'Ï': "I",
'Ð': "D", 'Ñ': "N", 'Ò': "O", 'Ó': "O", 'Ô': "O", 'Õ': "O", 'Ö': "O", 'Ø': "O",
'Ù': "U", 'Ú': "U", 'Û': "U", 'Ü': "U", 'Ý': "Y", 'Þ': "Th", 'ß': "ss", 'Œ': "OE",
'à': "a", 'á': "a", 'â': "a", 'ã': "a", 'ä': "a", 'å': "a", 'æ': "ae", 'ç': "c",
'è': "e", 'é': "e", 'ê': "e", 'ë': "e", 'ì': "i", 'í': "i", 'î': "i", 'ï': "i",
'ð': "d", 'ñ': "n", 'ò': "o", 'ó': "o", 'ô': "o", 'õ': "o", 'ö': "o", 'ø': "o",
'ù': "u", 'ú': "u", 'û': "u", 'ü': "u", 'ý': "y", 'þ': "th", 'ÿ': "y", 'œ': "oe",
}
// asciiFold rewrites s to ASCII: folded Latin letters stay, any other non-ASCII
// rune is dropped.
func asciiFold(s string) string {
var b strings.Builder
for _, r := range s {
if r <= unicode.MaxASCII {
b.WriteRune(r)
} else {
b.WriteString(asciiFolds[r])
}
}
return b.String()
}