Mock LLM Responder — Rust source
Generate deterministic mock LLM API responses - chat completion JSON, SSE event streams with chunk timing, and a replay curl - for testing clients without an API key.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
// mock-llm-responder — deterministic mock LLM API responses (Rust port).
//
// Polyglot showcase port of CosmoDev's mock-llm-responder, mirrored from the
// canonical TypeScript lib (src/lib/mockLlmResponder.ts). Every output — the
// OpenAI-style chat completion JSON, the SSE event stream, and the replay
// curl — is a pure function of the spec: no clock, no unseeded randomness.
// The only "randomness" is a mulberry32 PRNG seeded from an FNV-1a hash of
// the spec, so the same spec always produces the same bytes. That is what
// makes a client-side test suite reproducible.
//
// Stdlib only: no serde, no rand. JSON is hand-rolled with a tiny escaper so
// field order (and therefore bytes) matches the reference exactly. Lengths
// are counted in chars, which matches the reference's units for ASCII
// content.
const SCENARIOS: [&str; 6] = [
"echo",
"canned-answer",
"streamed-lorem",
"error-429",
"error-500",
"slow-chunks",
];
const DEFAULT_MODEL: &str = "mock-gpt-4o-mini";
const DEFAULT_PROMPT: &str = "Hello, mock model!";
const DEFAULT_MAX_TOKENS: i64 = 64;
const MIN_MAX_TOKENS: i64 = 1;
const MAX_MAX_TOKENS: i64 = 4096;
/// Fixed timestamp for every mock response (2025-01-01T00:00:00Z).
const MOCK_EPOCH: i64 = 1735689600;
/// The canned-answer scenario always returns this text.
const CANNED_ANSWER: &str = "This is a canned response. A mock model returns the same answer for every request, which keeps client tests deterministic.";
/// Vocabulary for the poem-ish lorem scenarios.
const POEM_WORDS: [&str; 24] = [
"cosmos", "nebula", "quantum", "signal", "photon", "drift",
"orbit", "vector", "cipher", "lumen", "aurora", "echo",
"helix", "nova", "pulse", "tide", "vertex", "zenith",
"quasar", "ion", "halo", "flux", "prism", "comet",
];
/// A fully validated mock request spec.
#[derive(Debug, Clone)]
pub struct MockSpec {
pub scenario: String,
pub model: String,
pub max_tokens: i64,
pub prompt: String,
}
/// Validate + default input: `None` fields mean "use default".
#[derive(Debug, Clone)]
pub struct RawSpec {
pub scenario: String,
pub model: Option<String>,
pub max_tokens: Option<f64>,
pub prompt: Option<String>,
}
/// Either a 200 completion or a 429/500 error envelope.
#[derive(Debug, Clone)]
pub enum MockResponse {
Ok {
id: String,
finish_reason: &'static str,
prompt_tokens: i64,
completion_tokens: i64,
content: String,
},
Err {
status: u16,
message: &'static str,
type_: &'static str,
code: &'static str,
},
}
/// FNV-1a 32-bit hash — turns the spec into a deterministic seed / id.
fn hash_string(s: &str) -> u32 {
let mut h: u32 = 0x811c_9dc5;
for b in s.as_bytes() {
h ^= *b as u32;
h = h.wrapping_mul(0x0100_0193);
}
h
}
/// mulberry32 — tiny seeded PRNG; same seed, same sequence, forever.
fn mulberry32(seed: u32) -> impl FnMut() -> f64 {
let mut a = seed;
move || {
a = a.wrapping_add(0x6d2b_79f5);
let mut t = a;
t = (t ^ (t >> 15)).wrapping_mul(t | 1);
t ^= t.wrapping_add((t ^ (t >> 7)).wrapping_mul(t | 61));
((t ^ (t >> 14)) as f64) / 4_294_967_296.0
}
}
/// ~4 chars per token, floor of 1 — deterministic, no tokenizer needed.
pub fn token_count(text: &str) -> i64 {
let n = text.chars().count() as i64;
if n == 0 {
return 0;
}
std::cmp::max(1, (n + 3) / 4)
}
/// Cut text so it fits in `max_tokens` tokens (4 chars each).
pub fn truncate_to_tokens(text: &str, max_tokens: i64) -> String {
if token_count(text) <= max_tokens {
return text.to_string();
}
text.chars()
.take((max_tokens * 4) as usize)
.collect::<String>()
.trim_end()
.to_string()
}
/// Validate + default a raw spec. Unknown scenarios error.
pub fn normalize_spec(raw: RawSpec) -> Result<MockSpec, String> {
if !SCENARIOS.contains(&raw.scenario.as_str()) {
return Err(format!("Unknown scenario: {:?}", raw.scenario));
}
let model = match raw.model {
Some(m) if !m.trim().is_empty() => m.trim().to_string(),
_ => DEFAULT_MODEL.to_string(),
};
let max_tokens = match raw.max_tokens {
Some(t) if t.is_finite() => MAX_MAX_TOKENS.min(MIN_MAX_TOKENS.max(t.floor() as i64)),
_ => DEFAULT_MAX_TOKENS,
};
let prompt = match raw.prompt {
Some(p) if !p.is_empty() => p,
_ => DEFAULT_PROMPT.to_string(),
};
Ok(MockSpec { scenario: raw.scenario, model, max_tokens, prompt })
}
fn spec_seed(spec: &MockSpec, salt: &str) -> u32 {
hash_string(&format!("{}|{}|{}|{}", spec.scenario, spec.model, spec.max_tokens, salt))
}
/// The deterministic mock completion id.
pub fn build_id(spec: &MockSpec) -> String {
format!("chatcmpl-mock-{:08x}", spec_seed(spec, "id"))
}
/// One poem line of 5-7 vocabulary words.
fn make_line(rng: &mut dyn FnMut() -> f64) -> String {
let n = 5 + (rng() * 3.0).floor() as usize;
let words: Vec<&str> = (0..n)
.map(|_| POEM_WORDS[(rng() * POEM_WORDS.len() as f64).floor() as usize])
.collect();
words.join(" ")
}
/// Poem-ish lorem, grown line by line until the token budget is full.
fn build_poem(spec: &MockSpec) -> String {
let mut rng = mulberry32(spec_seed(spec, "poem"));
let mut text = String::new();
loop {
let line = make_line(&mut rng);
let candidate = if text.is_empty() { line } else { format!("{}\n{}", text, line) };
if !text.is_empty() && token_count(&candidate) > spec.max_tokens {
break;
}
text = candidate;
}
truncate_to_tokens(&text, spec.max_tokens)
}
/// The assistant content a scenario produces ("" for the error scenarios).
pub fn build_content(spec: &MockSpec) -> String {
match spec.scenario.as_str() {
"echo" => truncate_to_tokens(&spec.prompt, spec.max_tokens),
"canned-answer" => truncate_to_tokens(CANNED_ANSWER, spec.max_tokens),
"streamed-lorem" | "slow-chunks" => build_poem(spec),
_ => String::new(), // error-429, error-500
}
}
/// OpenAI-style chat completion (or error envelope) for the spec.
pub fn build_completion(spec: &MockSpec) -> MockResponse {
match spec.scenario.as_str() {
"error-429" => MockResponse::Err {
status: 429,
message: "Rate limit reached for the mock model. Please retry after 1 second.",
type_: "rate_limit_error",
code: "rate_limit_exceeded",
},
"error-500" => MockResponse::Err {
status: 500,
message: "The mock server had an error while processing your request.",
type_: "server_error",
code: "internal_server_error",
},
_ => {
let content = build_content(spec);
let completion_tokens = token_count(&content);
MockResponse::Ok {
id: build_id(spec),
finish_reason: if completion_tokens >= spec.max_tokens { "length" } else { "stop" },
prompt_tokens: token_count(&spec.prompt),
completion_tokens,
content,
}
}
}
}
/// True for the scenarios meant to be consumed as an SSE stream.
pub fn is_stream_scenario(scenario: &str) -> bool {
scenario == "streamed-lorem" || scenario == "slow-chunks"
}
/// Word tokens that keep their trailing whitespace, so any grouping
/// reassembles the original content byte for byte (the TS lib's /\S+\s*/g).
fn word_tokens(s: &str) -> Vec<String> {
let mut tokens = Vec::new();
let rs: Vec<char> = s.chars().collect();
let mut i = 0;
while i < rs.len() {
if rs[i].is_whitespace() {
i += 1;
continue;
}
let start = i;
while i < rs.len() && !rs[i].is_whitespace() {
i += 1;
}
while i < rs.len() && rs[i].is_whitespace() {
i += 1;
}
tokens.push(rs[start..i].iter().collect());
}
tokens
}
/// Split content into stream chunks. Chunks reassemble to the exact content.
pub fn chunk_content(spec: &MockSpec) -> Vec<String> {
if spec.scenario == "error-429" || spec.scenario == "error-500" {
return Vec::new();
}
let per_chunk = match spec.scenario.as_str() {
"streamed-lorem" => 4usize,
"slow-chunks" => 2usize,
_ => usize::MAX, // echo / canned-answer: one single chunk
};
let tokens = word_tokens(&build_content(spec));
let mut chunks = Vec::new();
let mut i = 0;
while i < tokens.len() {
let end = std::cmp::min(i + per_chunk, tokens.len());
chunks.push(tokens[i..end].concat());
i += per_chunk;
}
chunks
}
/// Inter-chunk delay the stub should sleep between chunks, in ms.
pub fn chunk_delay_ms(spec: &MockSpec) -> i64 {
match spec.scenario.as_str() {
"echo" => 25,
"canned-answer" => 120,
"streamed-lorem" => 40,
"slow-chunks" => 600,
_ => 0,
}
}
/// Time-to-first-byte the stub should sleep before the first event, in ms.
pub fn first_byte_ms(spec: &MockSpec) -> i64 {
match spec.scenario.as_str() {
"echo" => 20,
"canned-answer" => 350,
"streamed-lorem" => 60,
"slow-chunks" => 900,
_ => 0,
}
}
/// JSON string escaper with JSON.stringify semantics.
fn json_escape(s: &str) -> String {
let mut out = String::with_capacity(s.len());
for c in s.chars() {
match c {
'"' => out.push_str("\\\""),
'\\' => out.push_str("\\\\"),
'\n' => out.push_str("\\n"),
'\r' => out.push_str("\\r"),
'\t' => out.push_str("\\t"),
c if (c as u32) < 0x20 => out.push_str(&format!("\\u{:04x}", c as u32)),
c => out.push(c),
}
}
out
}
/// The SSE event stream: `data:` lines, timing comment markers, [DONE].
pub fn build_sse(spec: &MockSpec) -> String {
let mut lines: Vec<String> = vec![
format!(
": mock scenario={} first-byte={}ms inter-chunk={}ms",
spec.scenario,
first_byte_ms(spec),
chunk_delay_ms(spec)
),
String::new(),
];
match build_completion(spec) {
MockResponse::Err { message, type_, code, .. } => {
lines.push(format!(
"data: {{\"error\":{{\"message\":\"{}\",\"type\":\"{}\",\"code\":\"{}\"}}}}",
json_escape(message),
type_,
code
));
lines.push(String::new());
}
MockResponse::Ok { finish_reason, prompt_tokens, completion_tokens, .. } => {
for (i, chunk) in chunk_content(spec).iter().enumerate() {
let delta = if i == 0 {
format!("{{\"role\":\"assistant\",\"content\":\"{}\"}}", json_escape(chunk))
} else {
format!("{{\"content\":\"{}\"}}", json_escape(chunk))
};
lines.push(format!(
"data: {{\"id\":\"{}\",\"object\":\"chat.completion.chunk\",\"created\":{},\"model\":\"{}\",\"choices\":[{{\"index\":0,\"delta\":{},\"finish_reason\":null}}]}}",
build_id(spec),
MOCK_EPOCH,
json_escape(&spec.model),
delta
));
lines.push(String::new());
}
lines.push(format!(
"data: {{\"id\":\"{}\",\"object\":\"chat.completion.chunk\",\"created\":{},\"model\":\"{}\",\"choices\":[{{\"index\":0,\"delta\":{{}},\"finish_reason\":\"{}\"}}],\"usage\":{{\"prompt_tokens\":{},\"completion_tokens\":{},\"total_tokens\":{}}}}}",
build_id(spec),
MOCK_EPOCH,
json_escape(&spec.model),
finish_reason,
prompt_tokens,
completion_tokens,
prompt_tokens + completion_tokens
));
lines.push(String::new());
}
}
lines.push("data: [DONE]".to_string());
lines.push(String::new());
lines.join("\n")
}
/// A curl command that replays the request against a local stub on :8080.
pub fn build_curl(spec: &MockSpec) -> String {
let status = match build_completion(spec) {
MockResponse::Err { status, .. } => status,
MockResponse::Ok { .. } => 200,
};
let stream = is_stream_scenario(&spec.scenario);
let mut body = format!(
"{{\"model\":\"{}\",\"messages\":[{{\"role\":\"user\",\"content\":\"{}\"}}],\"max_tokens\":{}",
json_escape(&spec.model),
json_escape(&spec.prompt),
spec.max_tokens
);
if stream {
body.push_str(",\"stream\":true");
}
body.push('}');
[
format!("# Local stub: reply {} with the body shown in the JSON tab.", status),
format!(
"curl {} http://localhost:8080/v1/chat/completions \\",
if stream { "-N -s" } else { "-s" }
),
" -H 'Content-Type: application/json' \\".to_string(),
format!(" -d '{}'", body),
]
.join("\n")
}
/// One canonical demo line — handy for diffing against the other ports.
fn demo_line(spec: &MockSpec) -> String {
let spec_str = format!("{}|{}|{}|{}", spec.scenario, spec.model, spec.max_tokens, spec.prompt);
let (status, id, finish, pt, ct, content, err) = match build_completion(spec) {
MockResponse::Err { status, code, .. } => (
status as i64,
String::new(),
String::new(),
0,
0,
String::new(),
code.to_string(),
),
MockResponse::Ok { id, finish_reason, prompt_tokens, completion_tokens, content, .. } => (
200,
id,
finish_reason.to_string(),
prompt_tokens,
completion_tokens,
content,
String::new(),
),
};
let tt = pt + ct;
let chunks = chunk_content(spec).len();
format!(
"{{\"spec\":\"{}\",\"status\":{},\"id\":\"{}\",\"finish\":\"{}\",\"pt\":{},\"ct\":{},\"tt\":{},\"chunks\":{},\"content\":\"{}\",\"err\":\"{}\",\"curl\":\"{}\",\"sse\":\"{}\"}}",
json_escape(&spec_str),
status,
id,
finish,
pt,
ct,
tt,
chunks,
json_escape(&content),
err,
json_escape(&build_curl(spec)),
json_escape(&build_sse(spec))
)
}
fn main() {
// Demo: prints one canonical JSON line per scenario — handy for diffing
// against the other ports (node javascript.js, python python.py, php php.php).
let demos = vec![
RawSpec { scenario: "echo".into(), model: None, max_tokens: None, prompt: None },
RawSpec { scenario: "canned-answer".into(), model: None, max_tokens: Some(10.0), prompt: None },
RawSpec { scenario: "streamed-lorem".into(), model: None, max_tokens: Some(40.0), prompt: None },
RawSpec { scenario: "slow-chunks".into(), model: None, max_tokens: Some(30.0), prompt: None },
RawSpec { scenario: "error-429".into(), model: None, max_tokens: None, prompt: None },
RawSpec { scenario: "error-500".into(), model: None, max_tokens: None, prompt: None },
RawSpec {
scenario: "echo".into(),
model: Some("my-model \"x\"".into()),
max_tokens: Some(8.0),
prompt: Some("Say \"hi\"\nline".into()),
},
];
for raw in demos {
let spec = normalize_spec(raw).expect("demo specs are valid");
println!("{}", demo_line(&spec));
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →