Skip to content

Mock LLM Responder — Python source

Generate deterministic mock LLM API responses - chat completion JSON, SSE event streams with chunk timing, and a replay curl - for testing clients without an API key.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

# mock-llm-responder — deterministic mock LLM API responses (Python port).
#
# Polyglot showcase port of CosmoDev's mock-llm-responder, mirrored from the
# canonical TypeScript lib (src/lib/mockLlmResponder.ts). Every output — the
# OpenAI-style chat completion JSON, the SSE event stream, and the replay
# curl — is a pure function of the spec: no clock, no unseeded randomness.
# The only "randomness" is a mulberry32 PRNG seeded from an FNV-1a hash of
# the spec, so the same spec always produces the same bytes. That is what
# makes a client-side test suite reproducible.
#
# Lengths are counted in characters (len(str)), which matches the reference
# lib's code units for ASCII content.

from __future__ import annotations

import json
import math
from dataclasses import dataclass
from typing import Any, Callable

SCENARIOS = [
    "echo",
    "canned-answer",
    "streamed-lorem",
    "error-429",
    "error-500",
    "slow-chunks",
]

DEFAULT_MODEL = "mock-gpt-4o-mini"
DEFAULT_PROMPT = "Hello, mock model!"
DEFAULT_MAX_TOKENS = 64
MIN_MAX_TOKENS = 1
MAX_MAX_TOKENS = 4096

MOCK_EPOCH = 1735689600  # 2025-01-01T00:00:00Z, fixed for every mock response

CANNED_ANSWER = (
    "This is a canned response. A mock model returns the same answer for "
    "every request, which keeps client tests deterministic."
)

POEM_WORDS = [
    "cosmos", "nebula", "quantum", "signal", "photon", "drift",
    "orbit", "vector", "cipher", "lumen", "aurora", "echo",
    "helix", "nova", "pulse", "tide", "vertex", "zenith",
    "quasar", "ion", "halo", "flux", "prism", "comet",
]

_MASK32 = 0xFFFFFFFF


def _hash_string(s: str) -> int:
    """FNV-1a 32-bit hash — turns the spec into a deterministic seed / id."""
    h = 0x811C9DC5
    for ch in s:
        h = ((h ^ ord(ch)) * 0x01000193) & _MASK32
    return h


def _mulberry32(seed: int) -> Callable[[], float]:
    """Tiny seeded PRNG; same seed, same sequence, forever."""
    a = seed & _MASK32

    def rng() -> float:
        nonlocal a
        a = (a + 0x6D2B79F5) & _MASK32
        t = a
        t = ((t ^ (t >> 15)) * (t | 1)) & _MASK32
        t = (t ^ (t + (((t ^ (t >> 7)) * (t | 61)) & _MASK32))) & _MASK32
        return ((t ^ (t >> 14)) & _MASK32) / 4294967296

    return rng


def token_count(text: str) -> int:
    """~4 chars per token, floor of 1 — deterministic, no tokenizer needed."""
    if text == "":
        return 0
    return max(1, math.ceil(len(text) / 4))


def truncate_to_tokens(text: str, max_tokens: int) -> str:
    """Cut text so it fits in `max_tokens` tokens (4 chars each)."""
    if token_count(text) <= max_tokens:
        return text
    return text[: max_tokens * 4].rstrip()


@dataclass
class MockSpec:
    scenario: str
    model: str
    max_tokens: int
    prompt: str


def normalize_spec(raw: dict[str, Any] | None) -> MockSpec:
    """Validate + default a raw spec. Unknown scenarios raise ValueError."""
    scenario = (raw or {}).get("scenario")
    if not isinstance(scenario, str) or scenario not in SCENARIOS:
        raise ValueError(f"Unknown scenario: {json.dumps(scenario)}")
    model = (raw or {}).get("model")
    if isinstance(model, str) and model.strip() != "":
        model = model.strip()
    else:
        model = DEFAULT_MODEL
    t = (raw or {}).get("maxTokens")
    if isinstance(t, (int, float)) and not isinstance(t, bool) and math.isfinite(t):
        max_tokens = min(MAX_MAX_TOKENS, max(MIN_MAX_TOKENS, math.floor(t)))
    else:
        max_tokens = DEFAULT_MAX_TOKENS
    prompt = (raw or {}).get("prompt")
    if not isinstance(prompt, str) or prompt == "":
        prompt = DEFAULT_PROMPT
    return MockSpec(scenario, model, max_tokens, prompt)


def _spec_seed(spec: MockSpec, salt: str) -> int:
    return _hash_string(f"{spec.scenario}|{spec.model}|{spec.max_tokens}|{salt}")


def build_id(spec: MockSpec) -> str:
    return f"chatcmpl-mock-{_spec_seed(spec, 'id'):08x}"


def _make_line(rng: Callable[[], float]) -> str:
    """One poem line of 5-7 vocabulary words."""
    n = 5 + int(math.floor(rng() * 3))
    words = [POEM_WORDS[int(math.floor(rng() * len(POEM_WORDS)))] for _ in range(n)]
    return " ".join(words)


def _build_poem(spec: MockSpec) -> str:
    """Poem-ish lorem, grown line by line until the token budget is full."""
    rng = _mulberry32(_spec_seed(spec, "poem"))
    text = ""
    while True:
        line = _make_line(rng)
        candidate = line if text == "" else f"{text}\n{line}"
        if text != "" and token_count(candidate) > spec.max_tokens:
            break
        text = candidate
    return truncate_to_tokens(text, spec.max_tokens)


def build_content(spec: MockSpec) -> str:
    """The assistant content a scenario produces ('' for the error scenarios)."""
    if spec.scenario == "echo":
        return truncate_to_tokens(spec.prompt, spec.max_tokens)
    if spec.scenario == "canned-answer":
        return truncate_to_tokens(CANNED_ANSWER, spec.max_tokens)
    if spec.scenario in ("streamed-lorem", "slow-chunks"):
        return _build_poem(spec)
    return ""


def build_completion(spec: MockSpec) -> dict[str, Any]:
    """OpenAI-style chat completion (or error envelope) for the spec."""
    if spec.scenario == "error-429":
        return {
            "ok": False,
            "status": 429,
            "body": {
                "error": {
                    "message": "Rate limit reached for the mock model. Please retry after 1 second.",
                    "type": "rate_limit_error",
                    "code": "rate_limit_exceeded",
                }
            },
        }
    if spec.scenario == "error-500":
        return {
            "ok": False,
            "status": 500,
            "body": {
                "error": {
                    "message": "The mock server had an error while processing your request.",
                    "type": "server_error",
                    "code": "internal_server_error",
                }
            },
        }
    content = build_content(spec)
    completion_tokens = token_count(content)
    return {
        "ok": True,
        "status": 200,
        "body": {
            "id": build_id(spec),
            "object": "chat.completion",
            "created": MOCK_EPOCH,
            "model": spec.model,
            "choices": [
                {
                    "index": 0,
                    "message": {"role": "assistant", "content": content},
                    "finish_reason": "length" if completion_tokens >= spec.max_tokens else "stop",
                }
            ],
            "usage": {
                "prompt_tokens": token_count(spec.prompt),
                "completion_tokens": completion_tokens,
                "total_tokens": token_count(spec.prompt) + completion_tokens,
            },
        },
    }


def is_stream_scenario(scenario: str) -> bool:
    """True for the scenarios meant to be consumed as an SSE stream."""
    return scenario in ("streamed-lorem", "slow-chunks")


def _word_tokens(s: str) -> list[str]:
    """Word tokens that keep their trailing whitespace, so any grouping
    reassembles to the original content byte for byte."""
    tokens = []
    i, n = 0, len(s)
    while i < n:
        if s[i].isspace():
            i += 1
            continue
        start = i
        while i < n and not s[i].isspace():
            i += 1
        while i < n and s[i].isspace():
            i += 1
        tokens.append(s[start:i])
    return tokens


def chunk_content(spec: MockSpec) -> list[str]:
    """Split content into stream chunks. Chunks reassemble to the exact content."""
    if spec.scenario in ("error-429", "error-500"):
        return []
    tokens = _word_tokens(build_content(spec))
    if spec.scenario == "streamed-lorem":
        per_chunk = 4
    elif spec.scenario == "slow-chunks":
        per_chunk = 2
    else:  # echo / canned-answer: one single chunk
        per_chunk = max(1, len(tokens))
    return ["".join(tokens[i : i + per_chunk]) for i in range(0, len(tokens), per_chunk)]


def chunk_delay_ms(spec: MockSpec) -> int:
    """Inter-chunk delay the stub should sleep between chunks, in ms."""
    return {
        "echo": 25,
        "canned-answer": 120,
        "streamed-lorem": 40,
        "slow-chunks": 600,
    }.get(spec.scenario, 0)


def first_byte_ms(spec: MockSpec) -> int:
    """Time-to-first-byte the stub should sleep before the first event, in ms."""
    return {
        "echo": 20,
        "canned-answer": 350,
        "streamed-lorem": 60,
        "slow-chunks": 900,
    }.get(spec.scenario, 0)


def _dumps(x: Any) -> str:
    return json.dumps(x, separators=(",", ":"), ensure_ascii=False)


def build_sse(spec: MockSpec) -> str:
    """The SSE event stream: `data:` lines, timing comment markers, [DONE]."""
    res = build_completion(spec)
    lines = [
        f": mock scenario={spec.scenario} first-byte={first_byte_ms(spec)}ms"
        f" inter-chunk={chunk_delay_ms(spec)}ms",
        "",
    ]
    if not res["ok"]:
        lines += [f"data: {_dumps(res['body'])}", ""]
    else:
        for i, chunk in enumerate(chunk_content(spec)):
            delta = (
                {"role": "assistant", "content": chunk} if i == 0 else {"content": chunk}
            )
            lines += [
                "data: "
                + _dumps(
                    {
                        "id": build_id(spec),
                        "object": "chat.completion.chunk",
                        "created": MOCK_EPOCH,
                        "model": spec.model,
                        "choices": [{"index": 0, "delta": delta, "finish_reason": None}],
                    }
                ),
                "",
            ]
        body = res["body"]
        lines += [
            "data: "
            + _dumps(
                {
                    "id": build_id(spec),
                    "object": "chat.completion.chunk",
                    "created": MOCK_EPOCH,
                    "model": spec.model,
                    "choices": [
                        {
                            "index": 0,
                            "delta": {},
                            "finish_reason": body["choices"][0]["finish_reason"],
                        }
                    ],
                    "usage": body["usage"],
                }
            ),
            "",
        ]
    lines += ["data: [DONE]", ""]
    return "\n".join(lines)


def build_curl(spec: MockSpec) -> str:
    """A curl command that replays the request against a local stub on :8080."""
    status = build_completion(spec)["status"]
    body: dict[str, Any] = {
        "model": spec.model,
        "messages": [{"role": "user", "content": spec.prompt}],
        "max_tokens": spec.max_tokens,
    }
    stream = is_stream_scenario(spec.scenario)
    if stream:
        body["stream"] = True
    return "\n".join(
        [
            f"# Local stub: reply {status} with the body shown in the JSON tab.",
            f"curl {'-N -s' if stream else '-s'}"
            " http://localhost:8080/v1/chat/completions \\",
            "  -H 'Content-Type: application/json' \\",
            f"  -d '{_dumps(body)}'",
        ]
    )


def _demo_line(spec: MockSpec) -> str:
    res = build_completion(spec)
    out: dict[str, Any] = {
        "spec": f"{spec.scenario}|{spec.model}|{spec.max_tokens}|{spec.prompt}",
        "status": res["status"],
        "id": "",
        "finish": "",
        "pt": 0,
        "ct": 0,
        "tt": 0,
        "chunks": 0,
        "content": "",
        "err": "",
        "curl": build_curl(spec),
        "sse": build_sse(spec),
    }
    if res["ok"]:
        choice = res["body"]["choices"][0]
        out.update(
            id=res["body"]["id"],
            finish=choice["finish_reason"],
            pt=res["body"]["usage"]["prompt_tokens"],
            ct=res["body"]["usage"]["completion_tokens"],
            tt=res["body"]["usage"]["total_tokens"],
            content=choice["message"]["content"],
            chunks=len(chunk_content(spec)),
        )
    else:
        out["err"] = res["body"]["error"]["code"]
    return _dumps(out)


if __name__ == "__main__":
    # Demo: prints one canonical JSON line per scenario — handy for diffing
    # against the other ports (node javascript.js, php php.php, rustc rust.rs).
    for raw in [
        {"scenario": "echo"},
        {"scenario": "canned-answer", "maxTokens": 10},
        {"scenario": "streamed-lorem", "maxTokens": 40},
        {"scenario": "slow-chunks", "maxTokens": 30},
        {"scenario": "error-429"},
        {"scenario": "error-500"},
        {"scenario": "echo", "model": 'my-model "x"', "maxTokens": 8, "prompt": 'Say "hi"\nline'},
    ]:
        print(_demo_line(normalize_spec(raw)))

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →