Skip to content

Mock LLM Responder — TypeScript source

Generate deterministic mock LLM API responses - chat completion JSON, SSE event streams with chunk timing, and a replay curl - for testing clients without an API key.

This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

/**
 * Mock LLM responder - deterministic mock LLM API responses for testing
 * clients. Every output (chat completion JSON, SSE event stream, replay curl)
 * is a pure function of the spec: no clock, no RNG that is not seeded from
 * the spec itself. Same spec in, same bytes out - which is exactly what makes
 * a client-side test suite reproducible.
 */

/** The six mock scenarios. */
export type Scenario =
  | 'echo'
  | 'canned-answer'
  | 'streamed-lorem'
  | 'error-429'
  | 'error-500'
  | 'slow-chunks';

export const SCENARIOS: readonly Scenario[] = [
  'echo',
  'canned-answer',
  'streamed-lorem',
  'error-429',
  'error-500',
  'slow-chunks',
] as const;

export const DEFAULT_MODEL = 'mock-gpt-4o-mini';
export const DEFAULT_PROMPT = 'Hello, mock model!';
export const DEFAULT_MAX_TOKENS = 64;
export const MIN_MAX_TOKENS = 1;
export const MAX_MAX_TOKENS = 4096;

/** Fixed timestamp for every mock response (2025-01-01T00:00:00Z). */
export const MOCK_EPOCH = 1735689600;

/** The canned-answer scenario always returns this text. */
export const CANNED_ANSWER =
  'This is a canned response. A mock model returns the same answer for every request, which keeps client tests deterministic.';

/** Vocabulary for the poem-ish lorem scenarios. */
export const POEM_WORDS = [
  'cosmos', 'nebula', 'quantum', 'signal', 'photon', 'drift',
  'orbit', 'vector', 'cipher', 'lumen', 'aurora', 'echo',
  'helix', 'nova', 'pulse', 'tide', 'vertex', 'zenith',
  'quasar', 'ion', 'halo', 'flux', 'prism', 'comet',
] as const;

/** Fully validated mock request spec. */
export interface MockSpec {
  scenario: Scenario;
  model: string;
  maxTokens: number;
  prompt: string;
}

export interface ChatMessage {
  role: 'system' | 'user' | 'assistant';
  content: string;
}

export interface Usage {
  prompt_tokens: number;
  completion_tokens: number;
  total_tokens: number;
}

export interface ChatCompletion {
  id: string;
  object: 'chat.completion';
  created: number;
  model: string;
  choices: Array<{
    index: number;
    message: ChatMessage;
    finish_reason: 'stop' | 'length';
  }>;
  usage: Usage;
}

export interface ApiErrorBody {
  error: { message: string; type: string; code: string };
}

export type MockResponse =
  | { ok: true; status: 200; body: ChatCompletion }
  | { ok: false; status: 429 | 500; body: ApiErrorBody };

/** FNV-1a 32-bit hash - turns the spec into a deterministic seed / id. */
function hashString(s: string): number {
  let h = 0x811c9dc5;
  for (let i = 0; i < s.length; i++) {
    h ^= s.charCodeAt(i);
    h = Math.imul(h, 0x01000193);
  }
  return h >>> 0;
}

/** mulberry32 - tiny seeded PRNG; same seed, same sequence, forever. */
function mulberry32(seed: number): () => number {
  let a = seed >>> 0;
  return () => {
    a = (a + 0x6d2b79f5) | 0;
    let t = Math.imul(a ^ (a >>> 15), 1 | a);
    t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
    return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
  };
}

/** ~4 chars per token, floor of 1 - deterministic, no tokenizer needed. */
export function tokenCount(text: string): number {
  if (text === '') return 0;
  return Math.max(1, Math.ceil(text.length / 4));
}

/** Cut text so it fits in `maxTokens` tokens (4 chars each). */
export function truncateToTokens(text: string, maxTokens: number): string {
  if (tokenCount(text) <= maxTokens) return text;
  return text.slice(0, maxTokens * 4).trimEnd();
}

/** Validate + default a raw spec. Unknown scenarios throw. */
export function normalizeSpec(raw: Partial<MockSpec> | undefined | null): MockSpec {
  const scenario = raw?.scenario;
  if (typeof scenario !== 'string' || !SCENARIOS.includes(scenario as Scenario)) {
    throw new Error(`Unknown scenario: ${JSON.stringify(scenario)}`);
  }
  const model =
    typeof raw?.model === 'string' && raw.model.trim() !== '' ? raw.model.trim() : DEFAULT_MODEL;
  const t = raw?.maxTokens;
  const maxTokens =
    typeof t === 'number' && Number.isFinite(t)
      ? Math.min(MAX_MAX_TOKENS, Math.max(MIN_MAX_TOKENS, Math.floor(t)))
      : DEFAULT_MAX_TOKENS;
  const prompt = typeof raw?.prompt === 'string' && raw.prompt !== '' ? raw.prompt : DEFAULT_PROMPT;
  return { scenario: scenario as Scenario, model, maxTokens, prompt };
}

function specSeed(spec: MockSpec, salt: string): number {
  return hashString(`${spec.scenario}|${spec.model}|${spec.maxTokens}|${salt}`);
}

function buildId(spec: MockSpec): string {
  return `chatcmpl-mock-${specSeed(spec, 'id').toString(16).padStart(8, '0')}`;
}

/** One poem line of 5-7 vocabulary words. */
function makeLine(rng: () => number): string {
  const n = 5 + Math.floor(rng() * 3);
  const words: string[] = [];
  for (let i = 0; i < n; i++) words.push(POEM_WORDS[Math.floor(rng() * POEM_WORDS.length)]);
  return words.join(' ');
}

/** Poem-ish lorem, grown line by line until the token budget is full. */
function buildPoem(spec: MockSpec): string {
  const rng = mulberry32(specSeed(spec, 'poem'));
  let text = '';
  for (;;) {
    const line = makeLine(rng);
    const candidate = text === '' ? line : `${text}\n${line}`;
    if (text !== '' && tokenCount(candidate) > spec.maxTokens) break;
    text = candidate;
  }
  return truncateToTokens(text, spec.maxTokens);
}

/** The assistant content a scenario produces ('' for the error scenarios). */
export function buildContent(spec: MockSpec): string {
  switch (spec.scenario) {
    case 'echo':
      return truncateToTokens(spec.prompt, spec.maxTokens);
    case 'canned-answer':
      return truncateToTokens(CANNED_ANSWER, spec.maxTokens);
    case 'streamed-lorem':
    case 'slow-chunks':
      return buildPoem(spec);
    case 'error-429':
    case 'error-500':
      return '';
  }
}

/** OpenAI-style chat completion (or error envelope) for the spec. */
export function buildCompletion(spec: MockSpec): MockResponse {
  if (spec.scenario === 'error-429') {
    return {
      ok: false,
      status: 429,
      body: {
        error: {
          message: 'Rate limit reached for the mock model. Please retry after 1 second.',
          type: 'rate_limit_error',
          code: 'rate_limit_exceeded',
        },
      },
    };
  }
  if (spec.scenario === 'error-500') {
    return {
      ok: false,
      status: 500,
      body: {
        error: {
          message: 'The mock server had an error while processing your request.',
          type: 'server_error',
          code: 'internal_server_error',
        },
      },
    };
  }
  const content = buildContent(spec);
  const completionTokens = tokenCount(content);
  return {
    ok: true,
    status: 200,
    body: {
      id: buildId(spec),
      object: 'chat.completion',
      created: MOCK_EPOCH,
      model: spec.model,
      choices: [
        {
          index: 0,
          message: { role: 'assistant', content },
          finish_reason: completionTokens >= spec.maxTokens ? 'length' : 'stop',
        },
      ],
      usage: {
        prompt_tokens: tokenCount(spec.prompt),
        completion_tokens: completionTokens,
        total_tokens: tokenCount(spec.prompt) + completionTokens,
      },
    },
  };
}

/** True for the scenarios meant to be consumed as an SSE stream. */
export function isStreamScenario(scenario: Scenario): boolean {
  return scenario === 'streamed-lorem' || scenario === 'slow-chunks';
}

/** Split content into stream chunks. Chunks reassemble to the exact content. */
export function chunkContent(spec: MockSpec): string[] {
  if (spec.scenario === 'error-429' || spec.scenario === 'error-500') return [];
  const perChunk =
    spec.scenario === 'streamed-lorem' ? 4 : spec.scenario === 'slow-chunks' ? 2 : Infinity;
  // Word tokens keep their trailing whitespace, so any grouping reassembles
  // to the original content byte for byte.
  const tokens = buildContent(spec).match(/\S+\s*/g) ?? [];
  const chunks: string[] = [];
  for (let i = 0; i < tokens.length; i += perChunk) {
    chunks.push(tokens.slice(i, i + perChunk).join(''));
  }
  return chunks;
}

/** Inter-chunk delay the stub should sleep between chunks, in ms. */
export function chunkDelayMs(spec: MockSpec): number {
  switch (spec.scenario) {
    case 'echo':
      return 25;
    case 'canned-answer':
      return 120;
    case 'streamed-lorem':
      return 40;
    case 'slow-chunks':
      return 600;
    case 'error-429':
    case 'error-500':
      return 0;
  }
}

/** Time-to-first-byte the stub should sleep before the first event, in ms. */
export function firstByteMs(spec: MockSpec): number {
  switch (spec.scenario) {
    case 'echo':
      return 20;
    case 'canned-answer':
      return 350;
    case 'streamed-lorem':
      return 60;
    case 'slow-chunks':
      return 900;
    case 'error-429':
    case 'error-500':
      return 0;
  }
}

/** The SSE event stream: `data:` lines, timing comment markers, [DONE]. */
export function buildSse(spec: MockSpec): string {
  const res = buildCompletion(spec);
  const lines: string[] = [
    `: mock scenario=${spec.scenario} first-byte=${firstByteMs(spec)}ms inter-chunk=${chunkDelayMs(spec)}ms`,
    '',
  ];
  if (!res.ok) {
    lines.push(`data: ${JSON.stringify(res.body)}`, '');
  } else {
    const chunks = chunkContent(spec);
    chunks.forEach((chunk, i) => {
      const delta = i === 0 ? { role: 'assistant' as const, content: chunk } : { content: chunk };
      lines.push(
        `data: ${JSON.stringify({
          id: buildId(spec),
          object: 'chat.completion.chunk',
          created: MOCK_EPOCH,
          model: spec.model,
          choices: [{ index: 0, delta, finish_reason: null }],
        })}`,
        '',
      );
    });
    const { body } = res;
    lines.push(
      `data: ${JSON.stringify({
        id: buildId(spec),
        object: 'chat.completion.chunk',
        created: MOCK_EPOCH,
        model: spec.model,
        choices: [{ index: 0, delta: {}, finish_reason: body.choices[0].finish_reason }],
        usage: body.usage,
      })}`,
      '',
    );
  }
  lines.push('data: [DONE]', '');
  return lines.join('\n');
}

/** A curl command that replays the request against a local stub on :8080. */
export function buildCurl(spec: MockSpec): string {
  const status = buildCompletion(spec).status;
  const body: Record<string, unknown> = {
    model: spec.model,
    messages: [{ role: 'user', content: spec.prompt }],
    max_tokens: spec.maxTokens,
  };
  const stream = isStreamScenario(spec.scenario);
  if (stream) body.stream = true;
  return [
    `# Local stub: reply ${status} with the body shown in the JSON tab.`,
    `curl ${stream ? '-N -s' : '-s'} http://localhost:8080/v1/chat/completions \\`,
    `  -H 'Content-Type: application/json' \\`,
    `  -d '${JSON.stringify(body)}'`,
  ].join('\n');
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →