Mock LLM Responder — TypeScript source
Generate deterministic mock LLM API responses - chat completion JSON, SSE event streams with chunk timing, and a replay curl - for testing clients without an API key.
This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
/**
* Mock LLM responder - deterministic mock LLM API responses for testing
* clients. Every output (chat completion JSON, SSE event stream, replay curl)
* is a pure function of the spec: no clock, no RNG that is not seeded from
* the spec itself. Same spec in, same bytes out - which is exactly what makes
* a client-side test suite reproducible.
*/
/** The six mock scenarios. */
export type Scenario =
| 'echo'
| 'canned-answer'
| 'streamed-lorem'
| 'error-429'
| 'error-500'
| 'slow-chunks';
export const SCENARIOS: readonly Scenario[] = [
'echo',
'canned-answer',
'streamed-lorem',
'error-429',
'error-500',
'slow-chunks',
] as const;
export const DEFAULT_MODEL = 'mock-gpt-4o-mini';
export const DEFAULT_PROMPT = 'Hello, mock model!';
export const DEFAULT_MAX_TOKENS = 64;
export const MIN_MAX_TOKENS = 1;
export const MAX_MAX_TOKENS = 4096;
/** Fixed timestamp for every mock response (2025-01-01T00:00:00Z). */
export const MOCK_EPOCH = 1735689600;
/** The canned-answer scenario always returns this text. */
export const CANNED_ANSWER =
'This is a canned response. A mock model returns the same answer for every request, which keeps client tests deterministic.';
/** Vocabulary for the poem-ish lorem scenarios. */
export const POEM_WORDS = [
'cosmos', 'nebula', 'quantum', 'signal', 'photon', 'drift',
'orbit', 'vector', 'cipher', 'lumen', 'aurora', 'echo',
'helix', 'nova', 'pulse', 'tide', 'vertex', 'zenith',
'quasar', 'ion', 'halo', 'flux', 'prism', 'comet',
] as const;
/** Fully validated mock request spec. */
export interface MockSpec {
scenario: Scenario;
model: string;
maxTokens: number;
prompt: string;
}
export interface ChatMessage {
role: 'system' | 'user' | 'assistant';
content: string;
}
export interface Usage {
prompt_tokens: number;
completion_tokens: number;
total_tokens: number;
}
export interface ChatCompletion {
id: string;
object: 'chat.completion';
created: number;
model: string;
choices: Array<{
index: number;
message: ChatMessage;
finish_reason: 'stop' | 'length';
}>;
usage: Usage;
}
export interface ApiErrorBody {
error: { message: string; type: string; code: string };
}
export type MockResponse =
| { ok: true; status: 200; body: ChatCompletion }
| { ok: false; status: 429 | 500; body: ApiErrorBody };
/** FNV-1a 32-bit hash - turns the spec into a deterministic seed / id. */
function hashString(s: string): number {
let h = 0x811c9dc5;
for (let i = 0; i < s.length; i++) {
h ^= s.charCodeAt(i);
h = Math.imul(h, 0x01000193);
}
return h >>> 0;
}
/** mulberry32 - tiny seeded PRNG; same seed, same sequence, forever. */
function mulberry32(seed: number): () => number {
let a = seed >>> 0;
return () => {
a = (a + 0x6d2b79f5) | 0;
let t = Math.imul(a ^ (a >>> 15), 1 | a);
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
};
}
/** ~4 chars per token, floor of 1 - deterministic, no tokenizer needed. */
export function tokenCount(text: string): number {
if (text === '') return 0;
return Math.max(1, Math.ceil(text.length / 4));
}
/** Cut text so it fits in `maxTokens` tokens (4 chars each). */
export function truncateToTokens(text: string, maxTokens: number): string {
if (tokenCount(text) <= maxTokens) return text;
return text.slice(0, maxTokens * 4).trimEnd();
}
/** Validate + default a raw spec. Unknown scenarios throw. */
export function normalizeSpec(raw: Partial<MockSpec> | undefined | null): MockSpec {
const scenario = raw?.scenario;
if (typeof scenario !== 'string' || !SCENARIOS.includes(scenario as Scenario)) {
throw new Error(`Unknown scenario: ${JSON.stringify(scenario)}`);
}
const model =
typeof raw?.model === 'string' && raw.model.trim() !== '' ? raw.model.trim() : DEFAULT_MODEL;
const t = raw?.maxTokens;
const maxTokens =
typeof t === 'number' && Number.isFinite(t)
? Math.min(MAX_MAX_TOKENS, Math.max(MIN_MAX_TOKENS, Math.floor(t)))
: DEFAULT_MAX_TOKENS;
const prompt = typeof raw?.prompt === 'string' && raw.prompt !== '' ? raw.prompt : DEFAULT_PROMPT;
return { scenario: scenario as Scenario, model, maxTokens, prompt };
}
function specSeed(spec: MockSpec, salt: string): number {
return hashString(`${spec.scenario}|${spec.model}|${spec.maxTokens}|${salt}`);
}
function buildId(spec: MockSpec): string {
return `chatcmpl-mock-${specSeed(spec, 'id').toString(16).padStart(8, '0')}`;
}
/** One poem line of 5-7 vocabulary words. */
function makeLine(rng: () => number): string {
const n = 5 + Math.floor(rng() * 3);
const words: string[] = [];
for (let i = 0; i < n; i++) words.push(POEM_WORDS[Math.floor(rng() * POEM_WORDS.length)]);
return words.join(' ');
}
/** Poem-ish lorem, grown line by line until the token budget is full. */
function buildPoem(spec: MockSpec): string {
const rng = mulberry32(specSeed(spec, 'poem'));
let text = '';
for (;;) {
const line = makeLine(rng);
const candidate = text === '' ? line : `${text}\n${line}`;
if (text !== '' && tokenCount(candidate) > spec.maxTokens) break;
text = candidate;
}
return truncateToTokens(text, spec.maxTokens);
}
/** The assistant content a scenario produces ('' for the error scenarios). */
export function buildContent(spec: MockSpec): string {
switch (spec.scenario) {
case 'echo':
return truncateToTokens(spec.prompt, spec.maxTokens);
case 'canned-answer':
return truncateToTokens(CANNED_ANSWER, spec.maxTokens);
case 'streamed-lorem':
case 'slow-chunks':
return buildPoem(spec);
case 'error-429':
case 'error-500':
return '';
}
}
/** OpenAI-style chat completion (or error envelope) for the spec. */
export function buildCompletion(spec: MockSpec): MockResponse {
if (spec.scenario === 'error-429') {
return {
ok: false,
status: 429,
body: {
error: {
message: 'Rate limit reached for the mock model. Please retry after 1 second.',
type: 'rate_limit_error',
code: 'rate_limit_exceeded',
},
},
};
}
if (spec.scenario === 'error-500') {
return {
ok: false,
status: 500,
body: {
error: {
message: 'The mock server had an error while processing your request.',
type: 'server_error',
code: 'internal_server_error',
},
},
};
}
const content = buildContent(spec);
const completionTokens = tokenCount(content);
return {
ok: true,
status: 200,
body: {
id: buildId(spec),
object: 'chat.completion',
created: MOCK_EPOCH,
model: spec.model,
choices: [
{
index: 0,
message: { role: 'assistant', content },
finish_reason: completionTokens >= spec.maxTokens ? 'length' : 'stop',
},
],
usage: {
prompt_tokens: tokenCount(spec.prompt),
completion_tokens: completionTokens,
total_tokens: tokenCount(spec.prompt) + completionTokens,
},
},
};
}
/** True for the scenarios meant to be consumed as an SSE stream. */
export function isStreamScenario(scenario: Scenario): boolean {
return scenario === 'streamed-lorem' || scenario === 'slow-chunks';
}
/** Split content into stream chunks. Chunks reassemble to the exact content. */
export function chunkContent(spec: MockSpec): string[] {
if (spec.scenario === 'error-429' || spec.scenario === 'error-500') return [];
const perChunk =
spec.scenario === 'streamed-lorem' ? 4 : spec.scenario === 'slow-chunks' ? 2 : Infinity;
// Word tokens keep their trailing whitespace, so any grouping reassembles
// to the original content byte for byte.
const tokens = buildContent(spec).match(/\S+\s*/g) ?? [];
const chunks: string[] = [];
for (let i = 0; i < tokens.length; i += perChunk) {
chunks.push(tokens.slice(i, i + perChunk).join(''));
}
return chunks;
}
/** Inter-chunk delay the stub should sleep between chunks, in ms. */
export function chunkDelayMs(spec: MockSpec): number {
switch (spec.scenario) {
case 'echo':
return 25;
case 'canned-answer':
return 120;
case 'streamed-lorem':
return 40;
case 'slow-chunks':
return 600;
case 'error-429':
case 'error-500':
return 0;
}
}
/** Time-to-first-byte the stub should sleep before the first event, in ms. */
export function firstByteMs(spec: MockSpec): number {
switch (spec.scenario) {
case 'echo':
return 20;
case 'canned-answer':
return 350;
case 'streamed-lorem':
return 60;
case 'slow-chunks':
return 900;
case 'error-429':
case 'error-500':
return 0;
}
}
/** The SSE event stream: `data:` lines, timing comment markers, [DONE]. */
export function buildSse(spec: MockSpec): string {
const res = buildCompletion(spec);
const lines: string[] = [
`: mock scenario=${spec.scenario} first-byte=${firstByteMs(spec)}ms inter-chunk=${chunkDelayMs(spec)}ms`,
'',
];
if (!res.ok) {
lines.push(`data: ${JSON.stringify(res.body)}`, '');
} else {
const chunks = chunkContent(spec);
chunks.forEach((chunk, i) => {
const delta = i === 0 ? { role: 'assistant' as const, content: chunk } : { content: chunk };
lines.push(
`data: ${JSON.stringify({
id: buildId(spec),
object: 'chat.completion.chunk',
created: MOCK_EPOCH,
model: spec.model,
choices: [{ index: 0, delta, finish_reason: null }],
})}`,
'',
);
});
const { body } = res;
lines.push(
`data: ${JSON.stringify({
id: buildId(spec),
object: 'chat.completion.chunk',
created: MOCK_EPOCH,
model: spec.model,
choices: [{ index: 0, delta: {}, finish_reason: body.choices[0].finish_reason }],
usage: body.usage,
})}`,
'',
);
}
lines.push('data: [DONE]', '');
return lines.join('\n');
}
/** A curl command that replays the request against a local stub on :8080. */
export function buildCurl(spec: MockSpec): string {
const status = buildCompletion(spec).status;
const body: Record<string, unknown> = {
model: spec.model,
messages: [{ role: 'user', content: spec.prompt }],
max_tokens: spec.maxTokens,
};
const stream = isStreamScenario(spec.scenario);
if (stream) body.stream = true;
return [
`# Local stub: reply ${status} with the body shown in the JSON tab.`,
`curl ${stream ? '-N -s' : '-s'} http://localhost:8080/v1/chat/completions \\`,
` -H 'Content-Type: application/json' \\`,
` -d '${JSON.stringify(body)}'`,
].join('\n');
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →