package fakes import ( "encoding/base64" "fmt" "math" "strconv" "strings" ) // maxLen caps generator output lengths (hex, nanoid, base64) and maxDecimals caps // float/calc decimal places, so a fat-fingered or overflowing argument fails at // New instead of trying to allocate gigabytes — or panicking — at render. const ( maxLen = 1 << 20 // 1,048,576 chars/bytes maxDecimals = 1024 ) // builtins is the registry of {name(args)} functions. Two kinds: derivations read // the digits emitted so far in the current expansion (luhn, mod11, ean — place // them after their payload); generators read only the rng (uuid, ulid, ...). All // must stay pure over (rng, emitted, args) so a seeded faker is reproducible — a // time-based id (uuid v7, ulid) draws its timestamp from the rng, not the wall // clock. Add a builtin only for what data can't express: a random v4 UUID and a // 24-hex ObjectID both ship as data, so the uuid builtin is v7. var builtins = map[string]builtin{ "luhn": {arity: 0, prep: derive(func(e string) string { return string(rune('0' + luhnCheck(e))) })}, "mod11": {arity: 0, prep: derive(mod11Check)}, "ean": {arity: 0, prep: derive(eanCheck)}, "uuid": {arity: 0, prep: generate(uuidV7)}, "ulid": {arity: 0, prep: generate(ulid)}, "nanoid": {arity: 1, check: posIntArg, prep: func(a []string) callFn { n := atoi(a[0]) return func(s *session, _ string, _ map[string]node) string { return nanoid(s, n) } }}, "hex": {arity: 1, check: posIntArg, prep: func(a []string) callFn { n := atoi(a[0]) return func(s *session, _ string, _ map[string]node) string { return randHex(s, n) } }}, "base64": {arity: 1, check: posIntArg, prep: func(a []string) callFn { n := atoi(a[0]) return func(s *session, _ string, _ map[string]node) string { return base64.StdEncoding.EncodeToString(randBytes(s, n)) } }}, "int": {arity: 2, check: intRangeArgs, prep: func(a []string) callFn { lo, span := atoi(a[0]), atoi(a[1])-atoi(a[0])+1 return func(s *session, _ string, _ map[string]node) string { return strconv.Itoa(lo + s.IntN(span)) } }}, "float": {arity: 3, check: floatArgs, prep: func(a []string) callFn { lo, _ := strconv.ParseFloat(a[0], 64) hi, _ := strconv.ParseFloat(a[1], 64) dp := atoi(a[2]) return func(s *session, _ string, _ map[string]node) string { return strconv.FormatFloat(lo+s.Float64()*(hi-lo), 'f', dp, 64) } }}, "iban": {arity: 1, check: ibanArg, prep: func(a []string) callFn { cc := a[0] return func(s *session, _ string, _ map[string]node) string { return iban(s, cc) } }}, // calc is the one builtin that reads the sibling fields (to render its operands); // every other ignores them. 1 or 2 args: the expression and an optional decimals. "calc": {arity: -1, check: checkCalc, prep: calcPrep}, // seq is the one stateful builtin: a per-session counter from 1, advancing on // each call. An optional name selects an independent counter; no name uses the // default one. Deterministic by construction, so a seeded faker stays stable. "seq": {arity: -1, check: seqArg, prep: func(a []string) callFn { key := "" if len(a) == 1 { key = a[0] } return func(s *session, _ string, _ map[string]node) string { return strconv.FormatUint(s.next(key), 10) } }}, } // derive and generate are the two argument-free builtin shapes the README names: a // derivation reads the output emitted so far, a generator reads only the rng. Each // lifts that one function into the prep every registry entry supplies. func derive(f func(emitted string) string) func([]string) callFn { return func([]string) callFn { return func(_ *session, emitted string, _ map[string]node) string { return f(emitted) } } } func generate(f func(rng) string) func([]string) callFn { return func([]string) callFn { return func(s *session, _ string, _ map[string]node) string { return f(s) } } } const hexDigits = "0123456789abcdef" // atoi parses an arg already validated by a builtin's check, so it cannot fail. func atoi(s string) int { n, _ := strconv.Atoi(s); return n } func randBytes(r rng, n int) []byte { b := make([]byte, n) for i := range b { b[i] = byte(r.IntN(256)) } return b } func randHex(r rng, n int) string { b := make([]byte, n) for i := range b { b[i] = hexDigits[r.IntN(16)] } return string(b) } func posIntArg(_ map[string]node, a []string) error { n, err := strconv.Atoi(a[0]) if err != nil || n < 1 { return fmt.Errorf("count %q must be a positive integer", a[0]) } if n > maxLen { return fmt.Errorf("count %d exceeds the maximum %d", n, maxLen) } return nil } func intRangeArgs(_ map[string]node, a []string) error { lo, e1 := strconv.Atoi(a[0]) hi, e2 := strconv.Atoi(a[1]) if e1 != nil || e2 != nil { return fmt.Errorf("int(min,max) needs integer args, got %q,%q", a[0], a[1]) } if lo > hi { return fmt.Errorf("int(min,max): min %d > max %d", lo, hi) } if uint64(hi)-uint64(lo) >= uint64(math.MaxInt64) { // span hi-lo+1 would overflow int -> IntN panic return fmt.Errorf("int(min,max): range %d..%d is too wide", lo, hi) } return nil } func floatArgs(_ map[string]node, a []string) error { lo, e1 := strconv.ParseFloat(a[0], 64) hi, e2 := strconv.ParseFloat(a[1], 64) dp, e3 := strconv.Atoi(a[2]) if e1 != nil || e2 != nil || e3 != nil { return fmt.Errorf("float(min,max,dp) needs numeric args, got %q,%q,%q", a[0], a[1], a[2]) } if lo > hi { return fmt.Errorf("float(min,max,dp): min %v > max %v", lo, hi) } if math.IsInf(hi-lo, 0) { // an overflowing span would render as "+Inf" return fmt.Errorf("float(min,max,dp): range %v..%v is too wide", lo, hi) } if dp < 0 || dp > maxDecimals { return fmt.Errorf("float(min,max,dp): decimals %d out of range 0..%d", dp, maxDecimals) } return nil } func seqArg(_ map[string]node, a []string) error { if len(a) > 1 { return fmt.Errorf("seq takes at most one name, got %d args", len(a)) } if len(a) == 1 && a[0] == "" { return fmt.Errorf("seq name must not be empty") } return nil } // nanoidAlphabet is the 64-char URL-safe set Nano IDs use (order is irrelevant // to the uniform pick). const nanoidAlphabet = "_-0123456789abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ" func nanoid(r rng, n int) string { b := make([]byte, n) for i := range b { b[i] = nanoidAlphabet[r.IntN(len(nanoidAlphabet))] } return string(b) } // uuidV7 builds an RFC 9562 v7 UUID. The 48-bit timestamp field is drawn from // the rng (not the clock) to stay reproducible, then the version (7) and variant // (10) bits are forced; the rest is random. func uuidV7(r rng) string { b := randBytes(r, 16) b[6] = b[6]&0x0f | 0x70 b[8] = b[8]&0x3f | 0x80 var sb strings.Builder for i, x := range b { if i == 4 || i == 6 || i == 8 || i == 10 { sb.WriteByte('-') } sb.WriteByte(hexDigits[x>>4]) sb.WriteByte(hexDigits[x&0x0f]) } return sb.String() } // crockford is the ULID/Crockford base32 alphabet (no I, L, O, U). const crockford = "0123456789ABCDEFGHJKMNPQRSTVWXYZ" // ulid builds a 26-char ULID: 128 random bits (timestamp from the rng) encoded // big-endian into base32, the 130-bit stream left-padded with two zero bits. func ulid(r rng) string { b := randBytes(r, 16) out := make([]byte, 26) for i := range out { v := 0 for j := 0; j < 5; j++ { // 5 bits per char; the two leading pad bits are zero real := 5*i + j - 2 bit := 0 if real >= 0 { bit = int(b[real/8]>>(7-uint(real%8))) & 1 } v = v<<1 | bit } out[i] = crockford[v] } return string(out) } // luhnCheck returns the Luhn check digit (0-9) over the digits of s; non-digit // runes are skipped. Doubling runs from the rightmost digit, so the result is // correct whatever the payload length. func luhnCheck(s string) int { sum, double := 0, true for i := len(s) - 1; i >= 0; i-- { c := s[i] if c < '0' || c > '9' { continue } d := int(c - '0') if double { if d *= 2; d > 9 { d -= 9 } } double = !double sum += d } return (10 - sum%10) % 10 } // mod11Check returns the weighted mod-11 check character over the digits of s // (weights 2..7 cycling from the right). A would-be value of 10 emits 'X', as in // ISBN-10 / ISO 7064; non-digits are skipped. func mod11Check(s string) string { sum, w := 0, 2 for i := len(s) - 1; i >= 0; i-- { c := s[i] if c < '0' || c > '9' { continue } sum += int(c-'0') * w if w++; w > 7 { w = 2 } } if chk := (11 - sum%11) % 11; chk != 10 { return string(rune('0' + chk)) } return "X" } // eanCheck returns the EAN-13 / UPC-A / ISBN-13 / GTIN check digit over the // digits of s: weights 3 and 1 alternating from the rightmost digit, mod 10. func eanCheck(s string) string { sum, w := 0, 3 for i := len(s) - 1; i >= 0; i-- { c := s[i] if c < '0' || c > '9' { continue } sum += int(c-'0') * w w = 4 - w // 3 <-> 1 } return string(rune('0' + (10-sum%10)%10)) } // ibanLen maps a supported country code to the full IBAN length. The check digits // sit between the country code and the BBAN, so — unlike luhn/ean — iban can't be // a left-to-right derivation; it generates the whole value instead. var ibanLen = map[string]int{"BE": 16, "DE": 22, "DK": 18, "ES": 24, "FI": 18, "NO": 15, "SE": 24} func ibanArg(_ map[string]node, a []string) error { if _, ok := ibanLen[a[0]]; !ok { return fmt.Errorf("iban(%q): unsupported country code", a[0]) } return nil } // iban generates a structurally valid IBAN for cc: a numeric BBAN of the right // length, then mod-97 check digits. Real bank/branch structure isn't modelled — // the result passes length and checksum validation, which is what fake data needs. func iban(r rng, cc string) string { bban := make([]byte, ibanLen[cc]-4) for i := range bban { bban[i] = byte('0' + r.IntN(10)) } rem := 0 feed := func(d int) { rem = (rem*10 + d) % 97 } for _, c := range bban { feed(int(c - '0')) } for i := 0; i < len(cc); i++ { // letters A-Z -> 10..35, fed as two digits v := int(cc[i]-'A') + 10 feed(v / 10) feed(v % 10) } feed(0) feed(0) return fmt.Sprintf("%s%02d%s", cc, 98-rem, bban) }