Skip to content

Mock LLM Responder — JavaScript source

Generate deterministic mock LLM API responses - chat completion JSON, SSE event streams with chunk timing, and a replay curl - for testing clients without an API key.

This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

// mock-llm-responder — deterministic mock LLM API responses (JavaScript port).
//
// Polyglot showcase port of CosmoDev's mock-llm-responder, mirrored from the
// canonical TypeScript lib (src/lib/mockLlmResponder.ts). Every output — the
// OpenAI-style chat completion JSON, the SSE event stream, and the replay
// curl — is a pure function of the spec: no clock, no unseeded randomness.
// The only "randomness" is a mulberry32 PRNG seeded from an FNV-1a hash of
// the spec, so the same spec always produces the same bytes. That is what
// makes a client-side test suite reproducible.
//
// Lengths are counted in UTF-16 code units (String#length), exactly like the
// reference lib; identical to bytes and code points for ASCII content.

export const SCENARIOS = [
  'echo',
  'canned-answer',
  'streamed-lorem',
  'error-429',
  'error-500',
  'slow-chunks',
];

export const DEFAULT_MODEL = 'mock-gpt-4o-mini';
export const DEFAULT_PROMPT = 'Hello, mock model!';
export const DEFAULT_MAX_TOKENS = 64;
const MIN_MAX_TOKENS = 1;
const MAX_MAX_TOKENS = 4096;

/** Fixed timestamp for every mock response (2025-01-01T00:00:00Z). */
export const MOCK_EPOCH = 1735689600;

/** The canned-answer scenario always returns this text. */
export const CANNED_ANSWER =
  'This is a canned response. A mock model returns the same answer for every request, which keeps client tests deterministic.';

/** Vocabulary for the poem-ish lorem scenarios. */
export const POEM_WORDS = [
  'cosmos', 'nebula', 'quantum', 'signal', 'photon', 'drift',
  'orbit', 'vector', 'cipher', 'lumen', 'aurora', 'echo',
  'helix', 'nova', 'pulse', 'tide', 'vertex', 'zenith',
  'quasar', 'ion', 'halo', 'flux', 'prism', 'comet',
];

/** FNV-1a 32-bit hash — turns the spec into a deterministic seed / id. */
function hashString(s) {
  let h = 0x811c9dc5;
  for (let i = 0; i < s.length; i++) {
    h ^= s.charCodeAt(i);
    h = Math.imul(h, 0x01000193);
  }
  return h >>> 0;
}

/** mulberry32 — tiny seeded PRNG; same seed, same sequence, forever. */
function mulberry32(seed) {
  let a = seed >>> 0;
  return () => {
    a = (a + 0x6d2b79f5) | 0;
    let t = Math.imul(a ^ (a >>> 15), 1 | a);
    t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
    return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
  };
}

/** ~4 chars per token, floor of 1 — deterministic, no tokenizer needed. */
export function tokenCount(text) {
  if (text === '') return 0;
  return Math.max(1, Math.ceil(text.length / 4));
}

/** Cut text so it fits in `maxTokens` tokens (4 chars each). */
export function truncateToTokens(text, maxTokens) {
  if (tokenCount(text) <= maxTokens) return text;
  return text.slice(0, maxTokens * 4).trimEnd();
}

/** Validate + default a raw spec. Unknown scenarios throw. */
export function normalizeSpec(raw) {
  const scenario = raw && raw.scenario;
  if (typeof scenario !== 'string' || !SCENARIOS.includes(scenario)) {
    throw new Error(`Unknown scenario: ${JSON.stringify(scenario)}`);
  }
  const model =
    typeof raw.model === 'string' && raw.model.trim() !== '' ? raw.model.trim() : DEFAULT_MODEL;
  const t = raw.maxTokens;
  const maxTokens =
    typeof t === 'number' && Number.isFinite(t)
      ? Math.min(MAX_MAX_TOKENS, Math.max(MIN_MAX_TOKENS, Math.floor(t)))
      : DEFAULT_MAX_TOKENS;
  const prompt = typeof raw.prompt === 'string' && raw.prompt !== '' ? raw.prompt : DEFAULT_PROMPT;
  return { scenario, model, maxTokens, prompt };
}

function specSeed(spec, salt) {
  return hashString(`${spec.scenario}|${spec.model}|${spec.maxTokens}|${salt}`);
}

function buildId(spec) {
  return `chatcmpl-mock-${specSeed(spec, 'id').toString(16).padStart(8, '0')}`;
}

/** One poem line of 5-7 vocabulary words. */
function makeLine(rng) {
  const n = 5 + Math.floor(rng() * 3);
  const words = [];
  for (let i = 0; i < n; i++) words.push(POEM_WORDS[Math.floor(rng() * POEM_WORDS.length)]);
  return words.join(' ');
}

/** Poem-ish lorem, grown line by line until the token budget is full. */
function buildPoem(spec) {
  const rng = mulberry32(specSeed(spec, 'poem'));
  let text = '';
  for (;;) {
    const line = makeLine(rng);
    const candidate = text === '' ? line : `${text}\n${line}`;
    if (text !== '' && tokenCount(candidate) > spec.maxTokens) break;
    text = candidate;
  }
  return truncateToTokens(text, spec.maxTokens);
}

/** The assistant content a scenario produces ('' for the error scenarios). */
export function buildContent(spec) {
  switch (spec.scenario) {
    case 'echo':
      return truncateToTokens(spec.prompt, spec.maxTokens);
    case 'canned-answer':
      return truncateToTokens(CANNED_ANSWER, spec.maxTokens);
    case 'streamed-lorem':
    case 'slow-chunks':
      return buildPoem(spec);
    case 'error-429':
    case 'error-500':
      return '';
  }
}

/** OpenAI-style chat completion (or error envelope) for the spec. */
export function buildCompletion(spec) {
  if (spec.scenario === 'error-429') {
    return {
      ok: false,
      status: 429,
      body: {
        error: {
          message: 'Rate limit reached for the mock model. Please retry after 1 second.',
          type: 'rate_limit_error',
          code: 'rate_limit_exceeded',
        },
      },
    };
  }
  if (spec.scenario === 'error-500') {
    return {
      ok: false,
      status: 500,
      body: {
        error: {
          message: 'The mock server had an error while processing your request.',
          type: 'server_error',
          code: 'internal_server_error',
        },
      },
    };
  }
  const content = buildContent(spec);
  const completionTokens = tokenCount(content);
  return {
    ok: true,
    status: 200,
    body: {
      id: buildId(spec),
      object: 'chat.completion',
      created: MOCK_EPOCH,
      model: spec.model,
      choices: [
        {
          index: 0,
          message: { role: 'assistant', content },
          finish_reason: completionTokens >= spec.maxTokens ? 'length' : 'stop',
        },
      ],
      usage: {
        prompt_tokens: tokenCount(spec.prompt),
        completion_tokens: completionTokens,
        total_tokens: tokenCount(spec.prompt) + completionTokens,
      },
    },
  };
}

/** True for the scenarios meant to be consumed as an SSE stream. */
export function isStreamScenario(scenario) {
  return scenario === 'streamed-lorem' || scenario === 'slow-chunks';
}

/** Split content into stream chunks. Chunks reassemble to the exact content. */
export function chunkContent(spec) {
  if (spec.scenario === 'error-429' || spec.scenario === 'error-500') return [];
  const perChunk =
    spec.scenario === 'streamed-lorem' ? 4 : spec.scenario === 'slow-chunks' ? 2 : Infinity;
  // Word tokens keep their trailing whitespace, so any grouping reassembles
  // to the original content byte for byte.
  const tokens = buildContent(spec).match(/\S+\s*/g) ?? [];
  const chunks = [];
  for (let i = 0; i < tokens.length; i += perChunk) {
    chunks.push(tokens.slice(i, i + perChunk).join(''));
  }
  return chunks;
}

/** Inter-chunk delay the stub should sleep between chunks, in ms. */
export function chunkDelayMs(spec) {
  switch (spec.scenario) {
    case 'echo': return 25;
    case 'canned-answer': return 120;
    case 'streamed-lorem': return 40;
    case 'slow-chunks': return 600;
    case 'error-429':
    case 'error-500': return 0;
  }
}

/** Time-to-first-byte the stub should sleep before the first event, in ms. */
export function firstByteMs(spec) {
  switch (spec.scenario) {
    case 'echo': return 20;
    case 'canned-answer': return 350;
    case 'streamed-lorem': return 60;
    case 'slow-chunks': return 900;
    case 'error-429':
    case 'error-500': return 0;
  }
}

/** The SSE event stream: `data:` lines, timing comment markers, [DONE]. */
export function buildSse(spec) {
  const res = buildCompletion(spec);
  const lines = [
    `: mock scenario=${spec.scenario} first-byte=${firstByteMs(spec)}ms inter-chunk=${chunkDelayMs(spec)}ms`,
    '',
  ];
  if (!res.ok) {
    lines.push(`data: ${JSON.stringify(res.body)}`, '');
  } else {
    const chunks = chunkContent(spec);
    chunks.forEach((chunk, i) => {
      const delta = i === 0 ? { role: 'assistant', content: chunk } : { content: chunk };
      lines.push(
        `data: ${JSON.stringify({
          id: buildId(spec),
          object: 'chat.completion.chunk',
          created: MOCK_EPOCH,
          model: spec.model,
          choices: [{ index: 0, delta, finish_reason: null }],
        })}`,
        '',
      );
    });
    const { body } = res;
    lines.push(
      `data: ${JSON.stringify({
        id: buildId(spec),
        object: 'chat.completion.chunk',
        created: MOCK_EPOCH,
        model: spec.model,
        choices: [{ index: 0, delta: {}, finish_reason: body.choices[0].finish_reason }],
        usage: body.usage,
      })}`,
      '',
    );
  }
  lines.push('data: [DONE]', '');
  return lines.join('\n');
}

/** A curl command that replays the request against a local stub on :8080. */
export function buildCurl(spec) {
  const status = buildCompletion(spec).status;
  const body = {
    model: spec.model,
    messages: [{ role: 'user', content: spec.prompt }],
    max_tokens: spec.maxTokens,
  };
  const stream = isStreamScenario(spec.scenario);
  if (stream) body.stream = true;
  return [
    `# Local stub: reply ${status} with the body shown in the JSON tab.`,
    `curl ${stream ? '-N -s' : '-s'} http://localhost:8080/v1/chat/completions \\`,
    `  -H 'Content-Type: application/json' \\`,
    `  -d '${JSON.stringify(body)}'`,
  ].join('\n');
}

// Demo: run this file directly (`node javascript.js`) to print one canonical
// JSON line per scenario — handy for diffing against the other ports.
import { realpathSync } from 'node:fs';
import { pathToFileURL } from 'node:url';

function demoLine(spec) {
  const res = buildCompletion(spec);
  const out = {
    spec: `${spec.scenario}|${spec.model}|${spec.maxTokens}|${spec.prompt}`,
    status: res.status,
    id: '',
    finish: '',
    pt: 0,
    ct: 0,
    tt: 0,
    chunks: 0,
    content: '',
    err: '',
    curl: buildCurl(spec),
    sse: buildSse(spec),
  };
  if (res.ok) {
    const choice = res.body.choices[0];
    out.id = res.body.id;
    out.finish = choice.finish_reason;
    out.pt = res.body.usage.prompt_tokens;
    out.ct = res.body.usage.completion_tokens;
    out.tt = res.body.usage.total_tokens;
    out.content = choice.message.content;
    out.chunks = chunkContent(spec).length;
  } else {
    out.err = res.body.error.code;
  }
  return JSON.stringify(out);
}

if (process.argv[1] && import.meta.url === pathToFileURL(realpathSync(process.argv[1])).href) {
  for (const raw of [
    { scenario: 'echo' },
    { scenario: 'canned-answer', maxTokens: 10 },
    { scenario: 'streamed-lorem', maxTokens: 40 },
    { scenario: 'slow-chunks', maxTokens: 30 },
    { scenario: 'error-429' },
    { scenario: 'error-500' },
    { scenario: 'echo', model: 'my-model "x"', maxTokens: 8, prompt: 'Say "hi"\nline' },
  ]) {
    console.log(demoLine(normalizeSpec(raw)));
  }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →