Skip to content

Mock Data Generator — Python source

Generate deterministic fake data (names, emails, numbers, dates, booleans, UUIDs, pick-from-list) from a schema and a seed. Reproducible output.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""mock-data-generator — Python polyglot showcase port.

CosmoDev polyglot showcase port of mock-data-generator.
Ported from src/lib/mockData.ts (the canonical TypeScript reference).

Pure mock-data generation — deterministic and seed-stable. Identical
(schema, count, seed) always yields byte-identical output, so a seed reproduces
a fixture exactly. There is no `random`/`time` side-effect: the only source of
"randomness" is a seeded mulberry32 PRNG, which is fast but NOT
cryptographically secure — mock data is not a secret, so speed wins.

Python integers are arbitrary precision, so the 32-bit semantics of the
reference JS (`|0`, `>>>`, `Math.imul`) are emulated here with explicit masking
to keep the PRNG stream bit-identical to the other ports.

Display source — part of CosmoDev's polyglot tool pages.

Usage::

    from python import generate_mock
    rows = generate_mock(
        [{"name": "id", "type": "index"}, {"name": "email", "type": "email"}],
        count=5,
        seed=1234,
    )

Requires Python 3.10+ (uses ``match``/``case`` for field dispatch).
"""

from __future__ import annotations

import datetime
import math
import re
from typing import Any, Callable, List, TypedDict

_U32 = 0xFFFFFFFF
"""Mask for the low 32 bits — the width of the reference PRNG's arithmetic."""


FIRST_NAMES: List[str] = [
    "Ava", "Liam", "Noah", "Emma", "Olivia", "Aiden", "Sophia", "Mason", "Isabella",
    "Lucas", "Mia", "Ethan", "Amelia", "Leo", "Harper", "Ezra", "Ella", "Owen",
    "Luna", "Finn", "Zoe", "Jude", "Nora", "Kai", "Ruby", "Theo", "Ivy", "Max",
]

LAST_NAMES: List[str] = [
    "Smith", "Johnson", "Williams", "Brown", "Jones", "Garcia", "Miller", "Davis",
    "Rodriguez", "Martinez", "Hernandez", "Lopez", "Gonzalez", "Wilson", "Anderson",
    "Thomas", "Taylor", "Moore", "Jackson", "Martin", "Lee", "Perez", "Thompson",
    "White", "Harris", "Sanchez", "Clark", "Ramirez",
]


# `from` is a Python keyword, so the TypedDict is built via the functional form
# which allows arbitrary string keys.
FieldSpec = TypedDict(
    "FieldSpec",
    {
        "name": str,
        "type": str,            # one of the FieldType literals
        "min": float,
        "max": float,
        "from": str,            # ISO YYYY-MM-DD
        "to": str,
        "options": List[str],
        "length": int,
    },
    total=False,
)


def _u32(x: int) -> int:
    """Truncate to an unsigned 32-bit value (emulates JS ``x >>> 0``)."""
    return x & _U32


def _imul(a: int, b: int) -> int:
    """Low 32 bits of a product (emulates JS ``Math.imul``).

    The low-32-bit truncation is identical whether the operands are read as
    signed or unsigned, so the unsigned mask alone reproduces the reference.
    """
    return _u32(a * b)


def mulberry32(seed: int) -> Callable[[], float]:
    """Return a deterministic PRNG closure yielding floats in [0, 1).

    The state is captured by closure and advanced one draw per call, mirroring
    the JS reference's consumption order (the thing that pins reproducibility).
    """
    a = _u32(int(seed))

    def rand() -> float:
        nonlocal a
        a = _u32(a + 0x6D2B79F5)
        t = _imul(a ^ (a >> 15), 1 | a)
        t = _u32(_u32(t + _imul(t ^ (t >> 7), 61 | t)) ^ t)
        return _u32(t ^ (t >> 14)) / 4294967296.0

    return rand


def _is_real_number(value: Any) -> bool:
    """True for finite ints/floats only — mirrors JS ``Number.isFinite``."""
    if isinstance(value, bool):
        # bool is an int subclass in Python; treat it as non-numeric here so a
        # stray True/False never masquerades as a bound.
        return False
    return isinstance(value, (int, float)) and math.isfinite(value)


def _clamp_int(min_val: Any, max_val: Any, def_lo: int, def_hi: int) -> tuple[int, int]:
    """Resolve [min, max] into an ordered integer range with given defaults.

    Missing or non-finite bounds collapse to ``def_lo``/``def_hi`` (0/100 for
    integers, 1/99 for usernames). The result is always lo <= hi so the caller's
    span math never inverts.
    """
    lo = math.floor(min_val) if _is_real_number(min_val) else def_lo
    hi = math.floor(max_val) if _is_real_number(max_val) else def_hi
    return (min(lo, hi), max(lo, hi))


def _uuid_from_rng(rng: Callable[[], float]) -> str:
    """Build a v4-shaped UUID from the PRNG stream.

    Fixes the version nibble (position 12 -> '4') and a variant nibble
    (position 16 -> 8/9/a/b) so the string parses as a legal RFC 4122 v4 UUID,
    even though the bytes are deterministic, not secret.
    """
    hex_chars = [format(math.floor(rng() * 16), "x") for _ in range(32)]
    hex_chars[12] = "4"
    hex_chars[16] = ("8", "9", "a", "b")[math.floor(rng() * 4)]
    return (
        "".join(hex_chars[0:8])
        + "-"
        + "".join(hex_chars[8:12])
        + "-"
        + "".join(hex_chars[12:16])
        + "-"
        + "".join(hex_chars[16:20])
        + "-"
        + "".join(hex_chars[20:32])
    )


def _sanitize_field_name(name: Any, idx: int) -> str:
    """Coerce a column name into an identifier-safe key.

    Strips everything outside [A-Za-z0-9_$]; an all-stripped name becomes
    ``fieldN`` so no row ever loses a key.
    """
    if not isinstance(name, str):
        name = str(name)
    cleaned = re.sub(r"[^A-Za-z0-9_$]", "", name)
    return cleaned if cleaned else f"field{idx}"


def _parse_date_ms(value: Any, fallback: str = "2000-01-01") -> int:
    """Parse an ISO YYYY-MM-DD string to UTC milliseconds.

    Mirrors JS ``Date.parse`` of a date-only ISO string (which is interpreted as
    UTC midnight). Empty/unparseable input falls back gracefully so a bad date
    never corrupts the stream.
    """
    text = value if isinstance(value, str) and value else fallback
    try:
        dt = datetime.datetime.strptime(text, "%Y-%m-%d").replace(
            tzinfo=datetime.timezone.utc
        )
    except ValueError:
        dt = datetime.datetime.strptime(fallback, "%Y-%m-%d").replace(
            tzinfo=datetime.timezone.utc
        )
    return int(dt.timestamp() * 1000)


def generate_mock(
    schema: List[dict], count: int, seed: int
) -> List[dict[str, Any]]:
    """Generate ``count`` rows of mock data from ``schema``.

    The PRNG is seeded once and consumed left-to-right, row by row and field by
    field — that consumption order is what makes the output reproducible:
    inserting or reordering a field shifts every later value. Negative or NaN
    counts collapse to zero rows.
    """
    rng = mulberry32(seed)
    n = max(0, math.floor(count))
    rows: List[dict[str, Any]] = []

    for i in range(n):
        row: dict[str, Any] = {}
        for f, spec in enumerate(schema):
            key = _sanitize_field_name(spec.get("name"), f)
            row[key] = _generate_field(spec, i, rng)
        rows.append(row)
    return rows


def _generate_field(spec: dict, index: int, rng: Callable[[], float]) -> Any:
    """Produce one value for one field.

    Each branch consumes a fixed number of PRNG draws, keeping the stream
    aligned across rows and identical to the reference implementation.
    """
    field_type = spec.get("type")

    match field_type:
        case "index":
            return index

        case "firstName":
            return FIRST_NAMES[int(rng() * len(FIRST_NAMES))]

        case "lastName":
            return LAST_NAMES[int(rng() * len(LAST_NAMES))]

        case "fullName":
            first = FIRST_NAMES[int(rng() * len(FIRST_NAMES))]
            last = LAST_NAMES[int(rng() * len(LAST_NAMES))]
            return f"{first} {last}"

        case "username":
            first = FIRST_NAMES[int(rng() * len(FIRST_NAMES))].lower()
            lo, hi = _clamp_int(spec.get("min"), spec.get("max"), 1, 99)
            num = lo + int(rng() * (hi - lo + 1))
            return f"{first}{num}"

        case "email":
            first = FIRST_NAMES[int(rng() * len(FIRST_NAMES))].lower()
            last = LAST_NAMES[int(rng() * len(LAST_NAMES))].lower()
            return f"{first}.{last}@example.com"

        case "integer":
            lo, hi = _clamp_int(spec.get("min"), spec.get("max"), 0, 100)
            return lo + int(rng() * (hi - lo + 1))

        case "number":
            min_v = spec.get("min") if _is_real_number(spec.get("min")) else 0
            max_v = spec.get("max") if _is_real_number(spec.get("max")) else 1
            lo, hi = min(min_v, max_v), max(min_v, max_v)
            value = lo + rng() * (hi - lo)
            # Round to 4 decimals; floor(x + 0.5) reproduces JS Math.round
            # (round-half-up toward +inf) across the whole real line.
            return math.floor(value * 10000 + 0.5) / 10000

        case "boolean":
            return rng() < 0.5

        case "uuid":
            return _uuid_from_rng(rng)

        case "date":
            from_ms = _parse_date_ms(spec.get("from"), "2000-01-01")
            to_ms = _parse_date_ms(spec.get("to"), "2025-12-31")
            lo, hi = min(from_ms, to_ms), max(from_ms, to_ms)
            ms = lo + math.floor(rng() * (hi - lo))
            return datetime.datetime.fromtimestamp(
                ms / 1000, tz=datetime.timezone.utc
            ).strftime("%Y-%m-%d")

        case "pick":
            options = spec.get("options") or []
            if not options:
                return None
            return options[int(rng() * len(options))]

        case "string":
            raw_len = spec.get("length")
            if _is_real_number(raw_len):
                length = math.floor(raw_len)
            else:
                length = 8
            length = max(1, length)
            chars = "abcdefghijklmnopqrstuvwxyz"
            return "".join(chars[int(rng() * len(chars))] for _ in range(length))

        case _:
            return None


if __name__ == "__main__":
    # Tiny smoke demo so the file is runnable standalone.
    import json

    demo_schema = [
        {"name": "id", "type": "index"},
        {"name": "email", "type": "email"},
        {"name": "age", "type": "integer", "min": 18, "max": 65},
    ]
    print(json.dumps(generate_mock(demo_schema, 3, 1234), indent=2))

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →