Skip to content

Mock LLM Responder — Rust source

Generate deterministic mock LLM API responses - chat completion JSON, SSE event streams with chunk timing, and a replay curl - for testing clients without an API key.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

// mock-llm-responder — deterministic mock LLM API responses (Rust port).
//
// Polyglot showcase port of CosmoDev's mock-llm-responder, mirrored from the
// canonical TypeScript lib (src/lib/mockLlmResponder.ts). Every output — the
// OpenAI-style chat completion JSON, the SSE event stream, and the replay
// curl — is a pure function of the spec: no clock, no unseeded randomness.
// The only "randomness" is a mulberry32 PRNG seeded from an FNV-1a hash of
// the spec, so the same spec always produces the same bytes. That is what
// makes a client-side test suite reproducible.
//
// Stdlib only: no serde, no rand. JSON is hand-rolled with a tiny escaper so
// field order (and therefore bytes) matches the reference exactly. Lengths
// are counted in chars, which matches the reference's units for ASCII
// content.

const SCENARIOS: [&str; 6] = [
    "echo",
    "canned-answer",
    "streamed-lorem",
    "error-429",
    "error-500",
    "slow-chunks",
];

const DEFAULT_MODEL: &str = "mock-gpt-4o-mini";
const DEFAULT_PROMPT: &str = "Hello, mock model!";
const DEFAULT_MAX_TOKENS: i64 = 64;
const MIN_MAX_TOKENS: i64 = 1;
const MAX_MAX_TOKENS: i64 = 4096;

/// Fixed timestamp for every mock response (2025-01-01T00:00:00Z).
const MOCK_EPOCH: i64 = 1735689600;

/// The canned-answer scenario always returns this text.
const CANNED_ANSWER: &str = "This is a canned response. A mock model returns the same answer for every request, which keeps client tests deterministic.";

/// Vocabulary for the poem-ish lorem scenarios.
const POEM_WORDS: [&str; 24] = [
    "cosmos", "nebula", "quantum", "signal", "photon", "drift",
    "orbit", "vector", "cipher", "lumen", "aurora", "echo",
    "helix", "nova", "pulse", "tide", "vertex", "zenith",
    "quasar", "ion", "halo", "flux", "prism", "comet",
];

/// A fully validated mock request spec.
#[derive(Debug, Clone)]
pub struct MockSpec {
    pub scenario: String,
    pub model: String,
    pub max_tokens: i64,
    pub prompt: String,
}

/// Validate + default input: `None` fields mean "use default".
#[derive(Debug, Clone)]
pub struct RawSpec {
    pub scenario: String,
    pub model: Option<String>,
    pub max_tokens: Option<f64>,
    pub prompt: Option<String>,
}

/// Either a 200 completion or a 429/500 error envelope.
#[derive(Debug, Clone)]
pub enum MockResponse {
    Ok {
        id: String,
        finish_reason: &'static str,
        prompt_tokens: i64,
        completion_tokens: i64,
        content: String,
    },
    Err {
        status: u16,
        message: &'static str,
        type_: &'static str,
        code: &'static str,
    },
}

/// FNV-1a 32-bit hash — turns the spec into a deterministic seed / id.
fn hash_string(s: &str) -> u32 {
    let mut h: u32 = 0x811c_9dc5;
    for b in s.as_bytes() {
        h ^= *b as u32;
        h = h.wrapping_mul(0x0100_0193);
    }
    h
}

/// mulberry32 — tiny seeded PRNG; same seed, same sequence, forever.
fn mulberry32(seed: u32) -> impl FnMut() -> f64 {
    let mut a = seed;
    move || {
        a = a.wrapping_add(0x6d2b_79f5);
        let mut t = a;
        t = (t ^ (t >> 15)).wrapping_mul(t | 1);
        t ^= t.wrapping_add((t ^ (t >> 7)).wrapping_mul(t | 61));
        ((t ^ (t >> 14)) as f64) / 4_294_967_296.0
    }
}

/// ~4 chars per token, floor of 1 — deterministic, no tokenizer needed.
pub fn token_count(text: &str) -> i64 {
    let n = text.chars().count() as i64;
    if n == 0 {
        return 0;
    }
    std::cmp::max(1, (n + 3) / 4)
}

/// Cut text so it fits in `max_tokens` tokens (4 chars each).
pub fn truncate_to_tokens(text: &str, max_tokens: i64) -> String {
    if token_count(text) <= max_tokens {
        return text.to_string();
    }
    text.chars()
        .take((max_tokens * 4) as usize)
        .collect::<String>()
        .trim_end()
        .to_string()
}

/// Validate + default a raw spec. Unknown scenarios error.
pub fn normalize_spec(raw: RawSpec) -> Result<MockSpec, String> {
    if !SCENARIOS.contains(&raw.scenario.as_str()) {
        return Err(format!("Unknown scenario: {:?}", raw.scenario));
    }
    let model = match raw.model {
        Some(m) if !m.trim().is_empty() => m.trim().to_string(),
        _ => DEFAULT_MODEL.to_string(),
    };
    let max_tokens = match raw.max_tokens {
        Some(t) if t.is_finite() => MAX_MAX_TOKENS.min(MIN_MAX_TOKENS.max(t.floor() as i64)),
        _ => DEFAULT_MAX_TOKENS,
    };
    let prompt = match raw.prompt {
        Some(p) if !p.is_empty() => p,
        _ => DEFAULT_PROMPT.to_string(),
    };
    Ok(MockSpec { scenario: raw.scenario, model, max_tokens, prompt })
}

fn spec_seed(spec: &MockSpec, salt: &str) -> u32 {
    hash_string(&format!("{}|{}|{}|{}", spec.scenario, spec.model, spec.max_tokens, salt))
}

/// The deterministic mock completion id.
pub fn build_id(spec: &MockSpec) -> String {
    format!("chatcmpl-mock-{:08x}", spec_seed(spec, "id"))
}

/// One poem line of 5-7 vocabulary words.
fn make_line(rng: &mut dyn FnMut() -> f64) -> String {
    let n = 5 + (rng() * 3.0).floor() as usize;
    let words: Vec<&str> = (0..n)
        .map(|_| POEM_WORDS[(rng() * POEM_WORDS.len() as f64).floor() as usize])
        .collect();
    words.join(" ")
}

/// Poem-ish lorem, grown line by line until the token budget is full.
fn build_poem(spec: &MockSpec) -> String {
    let mut rng = mulberry32(spec_seed(spec, "poem"));
    let mut text = String::new();
    loop {
        let line = make_line(&mut rng);
        let candidate = if text.is_empty() { line } else { format!("{}\n{}", text, line) };
        if !text.is_empty() && token_count(&candidate) > spec.max_tokens {
            break;
        }
        text = candidate;
    }
    truncate_to_tokens(&text, spec.max_tokens)
}

/// The assistant content a scenario produces ("" for the error scenarios).
pub fn build_content(spec: &MockSpec) -> String {
    match spec.scenario.as_str() {
        "echo" => truncate_to_tokens(&spec.prompt, spec.max_tokens),
        "canned-answer" => truncate_to_tokens(CANNED_ANSWER, spec.max_tokens),
        "streamed-lorem" | "slow-chunks" => build_poem(spec),
        _ => String::new(), // error-429, error-500
    }
}

/// OpenAI-style chat completion (or error envelope) for the spec.
pub fn build_completion(spec: &MockSpec) -> MockResponse {
    match spec.scenario.as_str() {
        "error-429" => MockResponse::Err {
            status: 429,
            message: "Rate limit reached for the mock model. Please retry after 1 second.",
            type_: "rate_limit_error",
            code: "rate_limit_exceeded",
        },
        "error-500" => MockResponse::Err {
            status: 500,
            message: "The mock server had an error while processing your request.",
            type_: "server_error",
            code: "internal_server_error",
        },
        _ => {
            let content = build_content(spec);
            let completion_tokens = token_count(&content);
            MockResponse::Ok {
                id: build_id(spec),
                finish_reason: if completion_tokens >= spec.max_tokens { "length" } else { "stop" },
                prompt_tokens: token_count(&spec.prompt),
                completion_tokens,
                content,
            }
        }
    }
}

/// True for the scenarios meant to be consumed as an SSE stream.
pub fn is_stream_scenario(scenario: &str) -> bool {
    scenario == "streamed-lorem" || scenario == "slow-chunks"
}

/// Word tokens that keep their trailing whitespace, so any grouping
/// reassembles the original content byte for byte (the TS lib's /\S+\s*/g).
fn word_tokens(s: &str) -> Vec<String> {
    let mut tokens = Vec::new();
    let rs: Vec<char> = s.chars().collect();
    let mut i = 0;
    while i < rs.len() {
        if rs[i].is_whitespace() {
            i += 1;
            continue;
        }
        let start = i;
        while i < rs.len() && !rs[i].is_whitespace() {
            i += 1;
        }
        while i < rs.len() && rs[i].is_whitespace() {
            i += 1;
        }
        tokens.push(rs[start..i].iter().collect());
    }
    tokens
}

/// Split content into stream chunks. Chunks reassemble to the exact content.
pub fn chunk_content(spec: &MockSpec) -> Vec<String> {
    if spec.scenario == "error-429" || spec.scenario == "error-500" {
        return Vec::new();
    }
    let per_chunk = match spec.scenario.as_str() {
        "streamed-lorem" => 4usize,
        "slow-chunks" => 2usize,
        _ => usize::MAX, // echo / canned-answer: one single chunk
    };
    let tokens = word_tokens(&build_content(spec));
    let mut chunks = Vec::new();
    let mut i = 0;
    while i < tokens.len() {
        let end = std::cmp::min(i + per_chunk, tokens.len());
        chunks.push(tokens[i..end].concat());
        i += per_chunk;
    }
    chunks
}

/// Inter-chunk delay the stub should sleep between chunks, in ms.
pub fn chunk_delay_ms(spec: &MockSpec) -> i64 {
    match spec.scenario.as_str() {
        "echo" => 25,
        "canned-answer" => 120,
        "streamed-lorem" => 40,
        "slow-chunks" => 600,
        _ => 0,
    }
}

/// Time-to-first-byte the stub should sleep before the first event, in ms.
pub fn first_byte_ms(spec: &MockSpec) -> i64 {
    match spec.scenario.as_str() {
        "echo" => 20,
        "canned-answer" => 350,
        "streamed-lorem" => 60,
        "slow-chunks" => 900,
        _ => 0,
    }
}

/// JSON string escaper with JSON.stringify semantics.
fn json_escape(s: &str) -> String {
    let mut out = String::with_capacity(s.len());
    for c in s.chars() {
        match c {
            '"' => out.push_str("\\\""),
            '\\' => out.push_str("\\\\"),
            '\n' => out.push_str("\\n"),
            '\r' => out.push_str("\\r"),
            '\t' => out.push_str("\\t"),
            c if (c as u32) < 0x20 => out.push_str(&format!("\\u{:04x}", c as u32)),
            c => out.push(c),
        }
    }
    out
}

/// The SSE event stream: `data:` lines, timing comment markers, [DONE].
pub fn build_sse(spec: &MockSpec) -> String {
    let mut lines: Vec<String> = vec![
        format!(
            ": mock scenario={} first-byte={}ms inter-chunk={}ms",
            spec.scenario,
            first_byte_ms(spec),
            chunk_delay_ms(spec)
        ),
        String::new(),
    ];
    match build_completion(spec) {
        MockResponse::Err { message, type_, code, .. } => {
            lines.push(format!(
                "data: {{\"error\":{{\"message\":\"{}\",\"type\":\"{}\",\"code\":\"{}\"}}}}",
                json_escape(message),
                type_,
                code
            ));
            lines.push(String::new());
        }
        MockResponse::Ok { finish_reason, prompt_tokens, completion_tokens, .. } => {
            for (i, chunk) in chunk_content(spec).iter().enumerate() {
                let delta = if i == 0 {
                    format!("{{\"role\":\"assistant\",\"content\":\"{}\"}}", json_escape(chunk))
                } else {
                    format!("{{\"content\":\"{}\"}}", json_escape(chunk))
                };
                lines.push(format!(
                    "data: {{\"id\":\"{}\",\"object\":\"chat.completion.chunk\",\"created\":{},\"model\":\"{}\",\"choices\":[{{\"index\":0,\"delta\":{},\"finish_reason\":null}}]}}",
                    build_id(spec),
                    MOCK_EPOCH,
                    json_escape(&spec.model),
                    delta
                ));
                lines.push(String::new());
            }
            lines.push(format!(
                "data: {{\"id\":\"{}\",\"object\":\"chat.completion.chunk\",\"created\":{},\"model\":\"{}\",\"choices\":[{{\"index\":0,\"delta\":{{}},\"finish_reason\":\"{}\"}}],\"usage\":{{\"prompt_tokens\":{},\"completion_tokens\":{},\"total_tokens\":{}}}}}",
                build_id(spec),
                MOCK_EPOCH,
                json_escape(&spec.model),
                finish_reason,
                prompt_tokens,
                completion_tokens,
                prompt_tokens + completion_tokens
            ));
            lines.push(String::new());
        }
    }
    lines.push("data: [DONE]".to_string());
    lines.push(String::new());
    lines.join("\n")
}

/// A curl command that replays the request against a local stub on :8080.
pub fn build_curl(spec: &MockSpec) -> String {
    let status = match build_completion(spec) {
        MockResponse::Err { status, .. } => status,
        MockResponse::Ok { .. } => 200,
    };
    let stream = is_stream_scenario(&spec.scenario);
    let mut body = format!(
        "{{\"model\":\"{}\",\"messages\":[{{\"role\":\"user\",\"content\":\"{}\"}}],\"max_tokens\":{}",
        json_escape(&spec.model),
        json_escape(&spec.prompt),
        spec.max_tokens
    );
    if stream {
        body.push_str(",\"stream\":true");
    }
    body.push('}');
    [
        format!("# Local stub: reply {} with the body shown in the JSON tab.", status),
        format!(
            "curl {} http://localhost:8080/v1/chat/completions \\",
            if stream { "-N -s" } else { "-s" }
        ),
        "  -H 'Content-Type: application/json' \\".to_string(),
        format!("  -d '{}'", body),
    ]
    .join("\n")
}

/// One canonical demo line — handy for diffing against the other ports.
fn demo_line(spec: &MockSpec) -> String {
    let spec_str = format!("{}|{}|{}|{}", spec.scenario, spec.model, spec.max_tokens, spec.prompt);
    let (status, id, finish, pt, ct, content, err) = match build_completion(spec) {
        MockResponse::Err { status, code, .. } => (
            status as i64,
            String::new(),
            String::new(),
            0,
            0,
            String::new(),
            code.to_string(),
        ),
        MockResponse::Ok { id, finish_reason, prompt_tokens, completion_tokens, content, .. } => (
            200,
            id,
            finish_reason.to_string(),
            prompt_tokens,
            completion_tokens,
            content,
            String::new(),
        ),
    };
    let tt = pt + ct;
    let chunks = chunk_content(spec).len();
    format!(
        "{{\"spec\":\"{}\",\"status\":{},\"id\":\"{}\",\"finish\":\"{}\",\"pt\":{},\"ct\":{},\"tt\":{},\"chunks\":{},\"content\":\"{}\",\"err\":\"{}\",\"curl\":\"{}\",\"sse\":\"{}\"}}",
        json_escape(&spec_str),
        status,
        id,
        finish,
        pt,
        ct,
        tt,
        chunks,
        json_escape(&content),
        err,
        json_escape(&build_curl(spec)),
        json_escape(&build_sse(spec))
    )
}

fn main() {
    // Demo: prints one canonical JSON line per scenario — handy for diffing
    // against the other ports (node javascript.js, python python.py, php php.php).
    let demos = vec![
        RawSpec { scenario: "echo".into(), model: None, max_tokens: None, prompt: None },
        RawSpec { scenario: "canned-answer".into(), model: None, max_tokens: Some(10.0), prompt: None },
        RawSpec { scenario: "streamed-lorem".into(), model: None, max_tokens: Some(40.0), prompt: None },
        RawSpec { scenario: "slow-chunks".into(), model: None, max_tokens: Some(30.0), prompt: None },
        RawSpec { scenario: "error-429".into(), model: None, max_tokens: None, prompt: None },
        RawSpec { scenario: "error-500".into(), model: None, max_tokens: None, prompt: None },
        RawSpec {
            scenario: "echo".into(),
            model: Some("my-model \"x\"".into()),
            max_tokens: Some(8.0),
            prompt: Some("Say \"hi\"\nline".into()),
        },
    ];
    for raw in demos {
        let spec = normalize_spec(raw).expect("demo specs are valid");
        println!("{}", demo_line(&spec));
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →