Mock LLM Responder — Python source
Generate deterministic mock LLM API responses - chat completion JSON, SSE event streams with chunk timing, and a replay curl - for testing clients without an API key.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
# mock-llm-responder — deterministic mock LLM API responses (Python port).
#
# Polyglot showcase port of CosmoDev's mock-llm-responder, mirrored from the
# canonical TypeScript lib (src/lib/mockLlmResponder.ts). Every output — the
# OpenAI-style chat completion JSON, the SSE event stream, and the replay
# curl — is a pure function of the spec: no clock, no unseeded randomness.
# The only "randomness" is a mulberry32 PRNG seeded from an FNV-1a hash of
# the spec, so the same spec always produces the same bytes. That is what
# makes a client-side test suite reproducible.
#
# Lengths are counted in characters (len(str)), which matches the reference
# lib's code units for ASCII content.
from __future__ import annotations
import json
import math
from dataclasses import dataclass
from typing import Any, Callable
SCENARIOS = [
"echo",
"canned-answer",
"streamed-lorem",
"error-429",
"error-500",
"slow-chunks",
]
DEFAULT_MODEL = "mock-gpt-4o-mini"
DEFAULT_PROMPT = "Hello, mock model!"
DEFAULT_MAX_TOKENS = 64
MIN_MAX_TOKENS = 1
MAX_MAX_TOKENS = 4096
MOCK_EPOCH = 1735689600 # 2025-01-01T00:00:00Z, fixed for every mock response
CANNED_ANSWER = (
"This is a canned response. A mock model returns the same answer for "
"every request, which keeps client tests deterministic."
)
POEM_WORDS = [
"cosmos", "nebula", "quantum", "signal", "photon", "drift",
"orbit", "vector", "cipher", "lumen", "aurora", "echo",
"helix", "nova", "pulse", "tide", "vertex", "zenith",
"quasar", "ion", "halo", "flux", "prism", "comet",
]
_MASK32 = 0xFFFFFFFF
def _hash_string(s: str) -> int:
"""FNV-1a 32-bit hash — turns the spec into a deterministic seed / id."""
h = 0x811C9DC5
for ch in s:
h = ((h ^ ord(ch)) * 0x01000193) & _MASK32
return h
def _mulberry32(seed: int) -> Callable[[], float]:
"""Tiny seeded PRNG; same seed, same sequence, forever."""
a = seed & _MASK32
def rng() -> float:
nonlocal a
a = (a + 0x6D2B79F5) & _MASK32
t = a
t = ((t ^ (t >> 15)) * (t | 1)) & _MASK32
t = (t ^ (t + (((t ^ (t >> 7)) * (t | 61)) & _MASK32))) & _MASK32
return ((t ^ (t >> 14)) & _MASK32) / 4294967296
return rng
def token_count(text: str) -> int:
"""~4 chars per token, floor of 1 — deterministic, no tokenizer needed."""
if text == "":
return 0
return max(1, math.ceil(len(text) / 4))
def truncate_to_tokens(text: str, max_tokens: int) -> str:
"""Cut text so it fits in `max_tokens` tokens (4 chars each)."""
if token_count(text) <= max_tokens:
return text
return text[: max_tokens * 4].rstrip()
@dataclass
class MockSpec:
scenario: str
model: str
max_tokens: int
prompt: str
def normalize_spec(raw: dict[str, Any] | None) -> MockSpec:
"""Validate + default a raw spec. Unknown scenarios raise ValueError."""
scenario = (raw or {}).get("scenario")
if not isinstance(scenario, str) or scenario not in SCENARIOS:
raise ValueError(f"Unknown scenario: {json.dumps(scenario)}")
model = (raw or {}).get("model")
if isinstance(model, str) and model.strip() != "":
model = model.strip()
else:
model = DEFAULT_MODEL
t = (raw or {}).get("maxTokens")
if isinstance(t, (int, float)) and not isinstance(t, bool) and math.isfinite(t):
max_tokens = min(MAX_MAX_TOKENS, max(MIN_MAX_TOKENS, math.floor(t)))
else:
max_tokens = DEFAULT_MAX_TOKENS
prompt = (raw or {}).get("prompt")
if not isinstance(prompt, str) or prompt == "":
prompt = DEFAULT_PROMPT
return MockSpec(scenario, model, max_tokens, prompt)
def _spec_seed(spec: MockSpec, salt: str) -> int:
return _hash_string(f"{spec.scenario}|{spec.model}|{spec.max_tokens}|{salt}")
def build_id(spec: MockSpec) -> str:
return f"chatcmpl-mock-{_spec_seed(spec, 'id'):08x}"
def _make_line(rng: Callable[[], float]) -> str:
"""One poem line of 5-7 vocabulary words."""
n = 5 + int(math.floor(rng() * 3))
words = [POEM_WORDS[int(math.floor(rng() * len(POEM_WORDS)))] for _ in range(n)]
return " ".join(words)
def _build_poem(spec: MockSpec) -> str:
"""Poem-ish lorem, grown line by line until the token budget is full."""
rng = _mulberry32(_spec_seed(spec, "poem"))
text = ""
while True:
line = _make_line(rng)
candidate = line if text == "" else f"{text}\n{line}"
if text != "" and token_count(candidate) > spec.max_tokens:
break
text = candidate
return truncate_to_tokens(text, spec.max_tokens)
def build_content(spec: MockSpec) -> str:
"""The assistant content a scenario produces ('' for the error scenarios)."""
if spec.scenario == "echo":
return truncate_to_tokens(spec.prompt, spec.max_tokens)
if spec.scenario == "canned-answer":
return truncate_to_tokens(CANNED_ANSWER, spec.max_tokens)
if spec.scenario in ("streamed-lorem", "slow-chunks"):
return _build_poem(spec)
return ""
def build_completion(spec: MockSpec) -> dict[str, Any]:
"""OpenAI-style chat completion (or error envelope) for the spec."""
if spec.scenario == "error-429":
return {
"ok": False,
"status": 429,
"body": {
"error": {
"message": "Rate limit reached for the mock model. Please retry after 1 second.",
"type": "rate_limit_error",
"code": "rate_limit_exceeded",
}
},
}
if spec.scenario == "error-500":
return {
"ok": False,
"status": 500,
"body": {
"error": {
"message": "The mock server had an error while processing your request.",
"type": "server_error",
"code": "internal_server_error",
}
},
}
content = build_content(spec)
completion_tokens = token_count(content)
return {
"ok": True,
"status": 200,
"body": {
"id": build_id(spec),
"object": "chat.completion",
"created": MOCK_EPOCH,
"model": spec.model,
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": content},
"finish_reason": "length" if completion_tokens >= spec.max_tokens else "stop",
}
],
"usage": {
"prompt_tokens": token_count(spec.prompt),
"completion_tokens": completion_tokens,
"total_tokens": token_count(spec.prompt) + completion_tokens,
},
},
}
def is_stream_scenario(scenario: str) -> bool:
"""True for the scenarios meant to be consumed as an SSE stream."""
return scenario in ("streamed-lorem", "slow-chunks")
def _word_tokens(s: str) -> list[str]:
"""Word tokens that keep their trailing whitespace, so any grouping
reassembles to the original content byte for byte."""
tokens = []
i, n = 0, len(s)
while i < n:
if s[i].isspace():
i += 1
continue
start = i
while i < n and not s[i].isspace():
i += 1
while i < n and s[i].isspace():
i += 1
tokens.append(s[start:i])
return tokens
def chunk_content(spec: MockSpec) -> list[str]:
"""Split content into stream chunks. Chunks reassemble to the exact content."""
if spec.scenario in ("error-429", "error-500"):
return []
tokens = _word_tokens(build_content(spec))
if spec.scenario == "streamed-lorem":
per_chunk = 4
elif spec.scenario == "slow-chunks":
per_chunk = 2
else: # echo / canned-answer: one single chunk
per_chunk = max(1, len(tokens))
return ["".join(tokens[i : i + per_chunk]) for i in range(0, len(tokens), per_chunk)]
def chunk_delay_ms(spec: MockSpec) -> int:
"""Inter-chunk delay the stub should sleep between chunks, in ms."""
return {
"echo": 25,
"canned-answer": 120,
"streamed-lorem": 40,
"slow-chunks": 600,
}.get(spec.scenario, 0)
def first_byte_ms(spec: MockSpec) -> int:
"""Time-to-first-byte the stub should sleep before the first event, in ms."""
return {
"echo": 20,
"canned-answer": 350,
"streamed-lorem": 60,
"slow-chunks": 900,
}.get(spec.scenario, 0)
def _dumps(x: Any) -> str:
return json.dumps(x, separators=(",", ":"), ensure_ascii=False)
def build_sse(spec: MockSpec) -> str:
"""The SSE event stream: `data:` lines, timing comment markers, [DONE]."""
res = build_completion(spec)
lines = [
f": mock scenario={spec.scenario} first-byte={first_byte_ms(spec)}ms"
f" inter-chunk={chunk_delay_ms(spec)}ms",
"",
]
if not res["ok"]:
lines += [f"data: {_dumps(res['body'])}", ""]
else:
for i, chunk in enumerate(chunk_content(spec)):
delta = (
{"role": "assistant", "content": chunk} if i == 0 else {"content": chunk}
)
lines += [
"data: "
+ _dumps(
{
"id": build_id(spec),
"object": "chat.completion.chunk",
"created": MOCK_EPOCH,
"model": spec.model,
"choices": [{"index": 0, "delta": delta, "finish_reason": None}],
}
),
"",
]
body = res["body"]
lines += [
"data: "
+ _dumps(
{
"id": build_id(spec),
"object": "chat.completion.chunk",
"created": MOCK_EPOCH,
"model": spec.model,
"choices": [
{
"index": 0,
"delta": {},
"finish_reason": body["choices"][0]["finish_reason"],
}
],
"usage": body["usage"],
}
),
"",
]
lines += ["data: [DONE]", ""]
return "\n".join(lines)
def build_curl(spec: MockSpec) -> str:
"""A curl command that replays the request against a local stub on :8080."""
status = build_completion(spec)["status"]
body: dict[str, Any] = {
"model": spec.model,
"messages": [{"role": "user", "content": spec.prompt}],
"max_tokens": spec.max_tokens,
}
stream = is_stream_scenario(spec.scenario)
if stream:
body["stream"] = True
return "\n".join(
[
f"# Local stub: reply {status} with the body shown in the JSON tab.",
f"curl {'-N -s' if stream else '-s'}"
" http://localhost:8080/v1/chat/completions \\",
" -H 'Content-Type: application/json' \\",
f" -d '{_dumps(body)}'",
]
)
def _demo_line(spec: MockSpec) -> str:
res = build_completion(spec)
out: dict[str, Any] = {
"spec": f"{spec.scenario}|{spec.model}|{spec.max_tokens}|{spec.prompt}",
"status": res["status"],
"id": "",
"finish": "",
"pt": 0,
"ct": 0,
"tt": 0,
"chunks": 0,
"content": "",
"err": "",
"curl": build_curl(spec),
"sse": build_sse(spec),
}
if res["ok"]:
choice = res["body"]["choices"][0]
out.update(
id=res["body"]["id"],
finish=choice["finish_reason"],
pt=res["body"]["usage"]["prompt_tokens"],
ct=res["body"]["usage"]["completion_tokens"],
tt=res["body"]["usage"]["total_tokens"],
content=choice["message"]["content"],
chunks=len(chunk_content(spec)),
)
else:
out["err"] = res["body"]["error"]["code"]
return _dumps(out)
if __name__ == "__main__":
# Demo: prints one canonical JSON line per scenario — handy for diffing
# against the other ports (node javascript.js, php php.php, rustc rust.rs).
for raw in [
{"scenario": "echo"},
{"scenario": "canned-answer", "maxTokens": 10},
{"scenario": "streamed-lorem", "maxTokens": 40},
{"scenario": "slow-chunks", "maxTokens": 30},
{"scenario": "error-429"},
{"scenario": "error-500"},
{"scenario": "echo", "model": 'my-model "x"', "maxTokens": 8, "prompt": 'Say "hi"\nline'},
]:
print(_demo_line(normalize_spec(raw)))
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →