From e91f3c08f36445bf88729428691c11458272cde3 Mon Sep 17 00:00:00 2001 From: Brian Genisio Date: Mon, 14 Sep 2026 07:03:02 -0400 Subject: [PATCH 1/2] Run independent evals with bounded concurrency. Cap in-flight LLM calls via session.config.json maxConcurrency, and show how long each evaluation took in the results summary. Co-authored-by: Cursor --- README.md | 1 + lib/concurrency.js | 51 +++++++++++++++++++++++++ lib/eval-compare.js | 71 ++++++++++++++++++++++------------- lib/eval-report.js | 5 ++- lib/eval-run.js | 31 +++++++++------ lib/format-duration.js | 26 +++++++++++++ lib/session-config.js | 3 ++ public/app.js | 13 +++++-- server.js | 5 ++- session.config.example.json | 1 + tests/concurrency.test.js | 60 +++++++++++++++++++++++++++++ tests/eval-compare.test.js | 44 ++++++++++++++++++++++ tests/eval-report.test.js | 4 +- tests/eval-run.test.js | 33 +++++++++++++++- tests/format-duration.test.js | 27 +++++++++++++ tests/server.test.js | 27 +++++++++++++ tests/session-config.test.js | 10 +++++ 17 files changed, 364 insertions(+), 48 deletions(-) create mode 100644 lib/concurrency.js create mode 100644 lib/format-duration.js create mode 100644 tests/concurrency.test.js create mode 100644 tests/format-duration.test.js diff --git a/README.md b/README.md index f7518b5..90f42f5 100644 --- a/README.md +++ b/README.md @@ -34,6 +34,7 @@ Fill in `.env` with the API key (and optional `*_BASE_URL`) for the provider you - `model` (optional) — `provider/model-id` (default `anthropic/claude-sonnet-4-6`); must be listed in `allowedModels` - `allowedModels` (optional) — picker list of `provider/model-id` refs (defaults to Anthropic, OpenAI, Gemini, and DeepSeek examples above) - `allowUserModelSelection` (optional) — when `true`, show a model picker and let the saved eval session override `model` with an entry from `allowedModels` (default `false`) +- `maxConcurrency` (optional) — max in-flight LLM calls during an evaluation (default `4`, range 1–50). Set to `1` for serial. - `defaults` (optional) — `minRuns`, `maxRuns`, `minCases`, `maxCases` (each 1–5) - `initialSession` (optional) — `promptA`, `promptB`, and `cases` (`input` / `expectedAnswer`) diff --git a/lib/concurrency.js b/lib/concurrency.js new file mode 100644 index 0000000..0b23bb1 --- /dev/null +++ b/lib/concurrency.js @@ -0,0 +1,51 @@ +/** + * Bounded in-flight pool for independent async work (eval LLM calls). + */ + +export const MIN_CONCURRENCY = 1; +export const MAX_CONCURRENCY = 50; +export const DEFAULT_CONCURRENCY = 4; + +/** + * @param {unknown} value + * @returns {number} + */ +export function normalizeConcurrency(value) { + const n = Number.parseInt(String(value ?? ''), 10); + if (!Number.isFinite(n)) return DEFAULT_CONCURRENCY; + return Math.min(MAX_CONCURRENCY, Math.max(MIN_CONCURRENCY, n)); +} + +/** + * @param {unknown} max + * @returns {{ concurrency: number, schedule: (fn: () => T | Promise) => Promise }} + */ +export function createConcurrencyLimiter(max) { + const concurrency = normalizeConcurrency(max); + let active = 0; + /** @type {Array<() => void>} */ + const queue = []; + + /** + * @template T + * @param {() => T | Promise} fn + * @returns {Promise} + */ + async function schedule(fn) { + while (active >= concurrency) { + await new Promise((resolve) => { + queue.push(resolve); + }); + } + active += 1; + try { + return await fn(); + } finally { + active -= 1; + const next = queue.shift(); + if (next) next(); + } + } + + return { concurrency, schedule }; +} diff --git a/lib/eval-compare.js b/lib/eval-compare.js index d08b16b..d1db937 100644 --- a/lib/eval-compare.js +++ b/lib/eval-compare.js @@ -4,6 +4,7 @@ * so every run is an independent LLM complete() call. */ +import { createConcurrencyLimiter, normalizeConcurrency } from './concurrency.js'; import { DEFAULT_METRIC_ID, summarizeScores } from './metrics/index.js'; import { normalizeRunCount, runEvalBatch } from './eval-run.js'; @@ -149,6 +150,7 @@ function validatePrompts(prompts) { * expectedAnswer?: string, * metricId?: string, * runs?: number, + * maxConcurrency?: number, * sessionInput?: Record, * runBatch?: typeof runEvalBatch, * }} opts @@ -160,54 +162,66 @@ export async function runPromptComparison(deps, opts) { const cases = normalizeCases(opts); const metricId = opts.metricId ?? DEFAULT_METRIC_ID; const runs = normalizeRunCount(opts.runs); + const maxConcurrency = normalizeConcurrency(opts.maxConcurrency); + const { schedule } = createConcurrencyLimiter(maxConcurrency); const runBatch = opts.runBatch ?? runEvalBatch; const anyExpected = cases.some((c) => c.expectedAnswer.trim() !== ''); + const startedAt = Date.now(); const conditions = { metricId: anyExpected ? metricId : null, runs, caseCount: cases.length, + maxConcurrency, }; - /** @type {Array} */ - const caseResults = []; /** @type {Record} */ const overallScoresByPrompt = Object.fromEntries(prompts.map((p) => [p.id, []])); - for (const testCase of cases) { - /** @type {Array} */ - const promptResults = []; - for (const p of prompts) { - const batch = await runBatch(deps, { - promptTemplate: p.promptTemplate, - input: testCase.input, - expectedAnswer: testCase.expectedAnswer, - metricId, - runs, - sessionInput: opts.sessionInput, - }); - promptResults.push({ - id: p.id, - label: p.label || p.id, - promptTemplate: p.promptTemplate, - ...batch, - }); - for (const r of batch.results) { + const casePromptResults = await Promise.all( + cases.map((testCase) => + Promise.all( + prompts.map(async (p) => { + const batch = await runBatch(deps, { + promptTemplate: p.promptTemplate, + input: testCase.input, + expectedAnswer: testCase.expectedAnswer, + metricId, + runs, + sessionInput: opts.sessionInput, + maxConcurrency, + schedule, + }); + return { + id: p.id, + label: p.label || p.id, + promptTemplate: p.promptTemplate, + ...batch, + }; + }), + ), + ), + ); + + /** @type {Array} */ + const caseResults = cases.map((testCase, i) => { + const promptResults = casePromptResults[i]; + for (const promptResult of promptResults) { + for (const r of promptResult.results ?? []) { if (typeof r.score === 'number' && Number.isFinite(r.score)) { - overallScoresByPrompt[p.id].push(r.score); + overallScoresByPrompt[promptResult.id].push(r.score); } } } - - caseResults.push({ + return { id: testCase.id, label: testCase.label, input: testCase.input, expectedAnswer: testCase.expectedAnswer.trim() !== '' ? testCase.expectedAnswer : null, prompts: promptResults, comparison: compareByMean(promptResults), - }); - } + }; + }); const overallPrompts = prompts.map((p) => { const aggregate = summarizeScores(overallScoresByPrompt[p.id]); @@ -230,7 +244,10 @@ export async function runPromptComparison(deps, opts) { }); return { - conditions, + conditions: { + ...conditions, + durationMs: Math.max(0, Date.now() - startedAt), + }, cases: caseResults, prompts: overallPrompts, comparison: compareByMean(overallPrompts), diff --git a/lib/eval-report.js b/lib/eval-report.js index fd86bbf..f838427 100644 --- a/lib/eval-report.js +++ b/lib/eval-report.js @@ -4,6 +4,7 @@ import fs from 'fs/promises'; import path from 'path'; +import { formatDuration } from './format-duration.js'; /** * @param {unknown} value @@ -45,7 +46,7 @@ function clip(text, max = 240) { * promptA?: string, * promptB?: string, * result: { - * conditions?: { metricId?: string | null, runs?: number, caseCount?: number }, + * conditions?: { metricId?: string | null, runs?: number, caseCount?: number, maxConcurrency?: number, durationMs?: number }, * comparison?: { outcome?: string, winnerId?: string | null, means?: Record }, * prompts?: Array<{ id: string, label?: string, aggregate?: { mean?: number, min?: number, max?: number, count?: number } | null }>, * cases?: Array<{ @@ -88,7 +89,9 @@ export function buildEvalReportMarkdown(opts) { `| Provider | ${escapeCell(opts.provider || '—')} |`, `| Metric | ${escapeCell(conditions.metricId ?? 'none (unscored)')} |`, `| Runs each | ${escapeCell(conditions.runs ?? '—')} |`, + `| Max concurrency | ${escapeCell(conditions.maxConcurrency ?? '—')} |`, `| Cases | ${escapeCell(conditions.caseCount ?? cases.length)} |`, + `| Duration | ${escapeCell(formatDuration(conditions.durationMs) || '—')} |`, `| Mode | ${multi ? 'Prompt A vs Prompt B' : 'Single prompt'} |`, '', '## Prompts', diff --git a/lib/eval-run.js b/lib/eval-run.js index 362c666..5a05195 100644 --- a/lib/eval-run.js +++ b/lib/eval-run.js @@ -10,6 +10,7 @@ import { randomUUID } from 'node:crypto'; import fs from 'node:fs/promises'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; +import { createConcurrencyLimiter } from './concurrency.js'; import { renderPromptTemplate } from './prompt-render.js'; import { DEFAULT_METRIC_ID, @@ -115,7 +116,7 @@ export async function runSingleEval(deps, opts) { } /** - * Render a template and execute N independent LLM runs sequentially. + * Render a template and execute N independent LLM runs with bounded concurrency. * * @param {EvalRunDeps} deps * @param {{ @@ -124,6 +125,8 @@ export async function runSingleEval(deps, opts) { * runs?: number, * expectedAnswer?: string, * metricId?: string, + * maxConcurrency?: number, + * schedule?: (fn: () => T | Promise) => Promise, * }} opts */ export async function runEvalBatch(deps, opts) { @@ -133,6 +136,9 @@ export async function runEvalBatch(deps, opts) { const metricId = opts.metricId ?? DEFAULT_METRIC_ID; const runs = normalizeRunCount(opts.runs); const renderedPrompt = renderPromptTemplate(promptTemplate, input); + const schedule = typeof opts.schedule === 'function' + ? opts.schedule + : createConcurrencyLimiter(opts.maxConcurrency).schedule; if (!renderedPrompt.trim()) { const err = new Error('promptTemplate (or input) must produce a non-empty prompt'); @@ -141,17 +147,18 @@ export async function runEvalBatch(deps, opts) { } /** @type {EvalResult[]} */ - const results = []; - for (let run = 1; run <= runs; run += 1) { - results.push( - await runSingleEval(deps, { - renderedPrompt, - run, - expectedAnswer, - metricId, - }), - ); - } + const results = await Promise.all( + Array.from({ length: runs }, (_, i) => + schedule(() => + runSingleEval(deps, { + renderedPrompt, + run: i + 1, + expectedAnswer, + metricId, + }), + ), + ), + ); const scoringEnabled = expectedAnswer.trim() !== ''; const aggregate = scoringEnabled diff --git a/lib/format-duration.js b/lib/format-duration.js new file mode 100644 index 0000000..4619075 --- /dev/null +++ b/lib/format-duration.js @@ -0,0 +1,26 @@ +/** + * Compact wall-clock duration for eval summaries. + */ + +/** + * @param {unknown} ms + * @returns {string} + */ +export function formatDuration(ms) { + const n = Number(ms); + if (!Number.isFinite(n) || n < 0) return ''; + if (n < 1000) return `${Math.round(n)}ms`; + + const totalSeconds = n / 1000; + if (totalSeconds < 60) { + const rounded = totalSeconds < 10 + ? Math.round(totalSeconds * 10) / 10 + : Math.round(totalSeconds); + return Number.isInteger(rounded) ? `${rounded}s` : `${rounded.toFixed(1)}s`; + } + + const minutes = Math.floor(totalSeconds / 60); + const seconds = Math.round(totalSeconds % 60); + if (seconds === 60) return `${minutes + 1}m`; + return seconds === 0 ? `${minutes}m` : `${minutes}m ${seconds}s`; +} diff --git a/lib/session-config.js b/lib/session-config.js index 01b6a40..cd4d7b7 100644 --- a/lib/session-config.js +++ b/lib/session-config.js @@ -7,6 +7,7 @@ * This module is browser-safe. */ +import { normalizeConcurrency } from './concurrency.js'; import { DEFAULT_MODEL_REF, parseModelRef } from './llm/model-ref.js'; export { DEFAULT_MODEL_REF }; @@ -173,6 +174,7 @@ function resolveConfiguredModel(rawModel, allowedModels) { * model: string, * allowedModels: string[], * allowUserModelSelection: boolean, + * maxConcurrency: number, * defaults: { minRuns: number, maxRuns: number, minCases: number, maxCases: number }, * initialSession: { promptA: string, promptB: string, cases: { input: string, expectedAnswer: string }[] }, * }} @@ -205,6 +207,7 @@ export function normalizeSessionConfig(raw) { model, allowedModels, allowUserModelSelection: src.allowUserModelSelection === true, + maxConcurrency: normalizeConcurrency(src.maxConcurrency), defaults: { minRuns: runs.min, maxRuns: runs.max, diff --git a/public/app.js b/public/app.js index 98b6a30..b9ef2f4 100644 --- a/public/app.js +++ b/public/app.js @@ -4,6 +4,7 @@ */ import { isRenderableResult, normalizeEvalSession } from '../lib/eval-session.js'; +import { formatDuration } from '../lib/format-duration.js'; import { collectPromptScoresByCase } from '../lib/score-distribution.js'; import { enqueueSessionsWrite } from '../lib/sessions-file.js'; import { @@ -431,10 +432,14 @@ function renderComparison(data) { caseDetailsPanel.open = false; const multi = isMultiPrompt(data); - const { runs, caseCount } = data.conditions; - resultsMeta.textContent = - `${caseCount} case${caseCount === 1 ? '' : 's'} · ${runs} run${runs === 1 ? '' : 's'} each` - + (multi ? ' · A vs B' : ''); + const { runs, caseCount, durationMs } = data.conditions; + const duration = formatDuration(durationMs); + resultsMeta.textContent = [ + `${caseCount} case${caseCount === 1 ? '' : 's'}`, + `${runs} run${runs === 1 ? '' : 's'} each`, + duration, + multi ? 'A vs B' : '', + ].filter(Boolean).join(' · '); renderVerdict(data); renderOverallCards(data); diff --git a/server.js b/server.js index f2cabf3..3cff8cc 100644 --- a/server.js +++ b/server.js @@ -120,8 +120,8 @@ app.use(express.json()); app.use('/design-system', express.static(path.join(__dirname, 'design-system'))); app.use(express.static(path.join(__dirname, 'public'))); -// Local eval session defaults (model, prompts, cases, UI min/max). Not secrets — -// those stay in .env. Missing file → empty initial session + built-in limits. +// Local eval session defaults (model, concurrency, prompts, cases, UI min/max). +// Not secrets — those stay in .env. Missing file → empty initial session + built-in limits. app.get('/api/session-config', async (_req, res) => { const raw = await readJsonFile(SESSION_CONFIG_FILE, {}); res.json(normalizeSessionConfig(raw)); @@ -238,6 +238,7 @@ app.post('/api/eval/compare', async (req, res) => { expectedAnswer: typeof expectedAnswer === 'string' ? expectedAnswer : '', metricId: typeof metricId === 'string' && metricId ? metricId : DEFAULT_METRIC_ID, runs, + maxConcurrency: normalizeSessionConfig(await readJsonFile(SESSION_CONFIG_FILE, {})).maxConcurrency, }, ); await persistEvalReport({ diff --git a/session.config.example.json b/session.config.example.json index faf141a..2dfd22a 100644 --- a/session.config.example.json +++ b/session.config.example.json @@ -7,6 +7,7 @@ "~deepseek/deepseek-v4-flash-latest" ], "allowUserModelSelection": true, + "maxConcurrency": 4, "defaults": { "minRuns": 1, "maxRuns": 5, diff --git a/tests/concurrency.test.js b/tests/concurrency.test.js new file mode 100644 index 0000000..17aa695 --- /dev/null +++ b/tests/concurrency.test.js @@ -0,0 +1,60 @@ +import { describe, it, expect, vi } from 'vitest'; +import { + DEFAULT_CONCURRENCY, + MAX_CONCURRENCY, + MIN_CONCURRENCY, + createConcurrencyLimiter, + normalizeConcurrency, +} from '../lib/concurrency.js'; + +describe('normalizeConcurrency', () => { + it('defaults invalid values and clamps to 1–50', () => { + expect(normalizeConcurrency(undefined)).toBe(DEFAULT_CONCURRENCY); + expect(normalizeConcurrency('')).toBe(DEFAULT_CONCURRENCY); + expect(normalizeConcurrency('nope')).toBe(DEFAULT_CONCURRENCY); + expect(normalizeConcurrency(0)).toBe(MIN_CONCURRENCY); + expect(normalizeConcurrency(99)).toBe(MAX_CONCURRENCY); + expect(normalizeConcurrency('3')).toBe(3); + }); +}); + +describe('createConcurrencyLimiter', () => { + function deferred() { + /** @type {() => void} */ + let resolve = () => {}; + const promise = new Promise((r) => { + resolve = r; + }); + return { promise, resolve }; + } + + it('never runs more than max tasks at once and preserves result order', async () => { + const { schedule } = createConcurrencyLimiter(2); + let inFlight = 0; + let peak = 0; + const gates = [deferred(), deferred(), deferred()]; + let started = 0; + + const pending = [0, 1, 2].map((i) => + schedule(async () => { + started += 1; + inFlight += 1; + peak = Math.max(peak, inFlight); + await gates[i].promise; + inFlight -= 1; + return i; + }), + ); + + await vi.waitFor(() => expect(started).toBe(2)); + expect(peak).toBe(2); + + gates[0].resolve(); + await vi.waitFor(() => expect(started).toBe(3)); + gates[1].resolve(); + gates[2].resolve(); + + await expect(Promise.all(pending)).resolves.toEqual([0, 1, 2]); + expect(peak).toBe(2); + }); +}); diff --git a/tests/eval-compare.test.js b/tests/eval-compare.test.js index f508891..ebda9ca 100644 --- a/tests/eval-compare.test.js +++ b/tests/eval-compare.test.js @@ -114,7 +114,10 @@ describe('runPromptComparison', () => { metricId: 'exact-match', runs: 1, caseCount: 1, + maxConcurrency: 4, + durationMs: expect.any(Number), }); + expect(result.conditions.durationMs).toBeGreaterThanOrEqual(0); expect(result.cases).toHaveLength(1); expect(result.comparison).toEqual({ outcome: 'winner', @@ -182,6 +185,47 @@ describe('runPromptComparison', () => { expect(result.comparison.outcome).toBe('unscored'); }); + it('shares maxConcurrency across prompt/case runs', async () => { + let started = 0; + let release = () => {}; + const hang = new Promise((resolve) => { + release = resolve; + }); + const complete = vi.fn().mockImplementation(async () => { + started += 1; + await hang; + return { text: 'Paris' }; + }); + + const pending = runPromptComparison( + { + llm: { name: 'anthropic', model: 'claude-sonnet-4-6', complete }, + systemPrompt: 'You are being evaluated.', + }, + { + prompts: [ + { id: 'A', label: 'Prompt A', promptTemplate: 'A {{input}}' }, + { id: 'B', label: 'Prompt B', promptTemplate: 'B {{input}}' }, + ], + cases: [ + { input: 'France', expectedAnswer: 'Paris' }, + { input: 'Spain', expectedAnswer: 'Madrid' }, + ], + metricId: 'exact-match', + runs: 2, + maxConcurrency: 3, + }, + ); + + await vi.waitFor(() => expect(started).toBe(3)); + release(); + const result = await pending; + expect(complete).toHaveBeenCalledTimes(8); + expect(result.conditions.maxConcurrency).toBe(3); + expect(result.cases).toHaveLength(2); + expect(result.cases[0].prompts.map((p) => p.id)).toEqual(['A', 'B']); + }); + it('rejects an empty prompts list', async () => { await expect( runPromptComparison( diff --git a/tests/eval-report.test.js b/tests/eval-report.test.js index 7b06b82..28d7c70 100644 --- a/tests/eval-report.test.js +++ b/tests/eval-report.test.js @@ -9,7 +9,7 @@ describe('buildEvalReportMarkdown', () => { promptA: 'Answer with only the capital.', generatedAt: '2026-09-08T12:00:00.000Z', result: { - conditions: { metricId: 'exact-match', runs: 2, caseCount: 1 }, + conditions: { metricId: 'exact-match', runs: 2, caseCount: 1, durationMs: 4200 }, comparison: { outcome: 'unscored', winnerId: null, means: { A: 1 } }, prompts: [ { id: 'A', label: 'Prompt', aggregate: { mean: 1, min: 1, max: 1, count: 2 } }, @@ -41,6 +41,8 @@ describe('buildEvalReportMarkdown', () => { expect(md).toContain('Single prompt'); expect(md).toContain('anthropic/claude-haiku-4-5-20251001'); expect(md).toContain('exact-match'); + expect(md).toContain('| Runs each | 2 |'); + expect(md).toContain('| Duration | 4.2s |'); expect(md).toContain('France'); expect(md).toContain('Paris'); expect(md).toContain('| 1 | ok | 1.00 | Paris |'); diff --git a/tests/eval-run.test.js b/tests/eval-run.test.js index 14a8a4b..9d55d12 100644 --- a/tests/eval-run.test.js +++ b/tests/eval-run.test.js @@ -52,7 +52,7 @@ describe('runSingleEval / runEvalBatch', () => { expect(result.sessionId.length).toBeGreaterThan(0); }); - it('runs N independent completions sequentially', async () => { + it('runs N independent completions', async () => { const { deps, complete } = makeDeps(); const batch = await runEvalBatch(deps, { promptTemplate: 'Echo: {{input}}', @@ -67,6 +67,37 @@ describe('runSingleEval / runEvalBatch', () => { expect(batch.metricId).toBeNull(); expect(complete).toHaveBeenCalledTimes(3); expect(new Set(batch.results.map((r) => r.sessionId)).size).toBe(3); + expect(batch.results.map((r) => r.run)).toEqual([1, 2, 3]); + }); + + it('caps in-flight complete() calls at maxConcurrency', async () => { + let started = 0; + let release = () => {}; + const hang = new Promise((resolve) => { + release = resolve; + }); + const complete = vi.fn().mockImplementation(async () => { + started += 1; + await hang; + return { text: 'ok' }; + }); + const deps = { + llm: { name: 'anthropic', model: 'claude-sonnet-4-6', complete }, + systemPrompt: 'You are being evaluated.', + }; + + const pending = runEvalBatch(deps, { + promptTemplate: 'Echo: {{input}}', + input: 'ping', + runs: 3, + maxConcurrency: 2, + }); + + await vi.waitFor(() => expect(started).toBe(2)); + release(); + const batch = await pending; + expect(complete).toHaveBeenCalledTimes(3); + expect(batch.results.map((r) => r.run)).toEqual([1, 2, 3]); }); it('scores outputs when expectedAnswer is provided', async () => { diff --git a/tests/format-duration.test.js b/tests/format-duration.test.js new file mode 100644 index 0000000..a428e20 --- /dev/null +++ b/tests/format-duration.test.js @@ -0,0 +1,27 @@ +import { describe, it, expect } from 'vitest'; +import { formatDuration } from '../lib/format-duration.js'; + +describe('formatDuration', () => { + it('returns empty for missing or invalid values', () => { + expect(formatDuration(undefined)).toBe(''); + expect(formatDuration(-1)).toBe(''); + expect(formatDuration('nope')).toBe(''); + }); + + it('shows milliseconds under one second', () => { + expect(formatDuration(0)).toBe('0ms'); + expect(formatDuration(847)).toBe('847ms'); + }); + + it('shows compact seconds under a minute', () => { + expect(formatDuration(1200)).toBe('1.2s'); + expect(formatDuration(4200)).toBe('4.2s'); + expect(formatDuration(10000)).toBe('10s'); + expect(formatDuration(12400)).toBe('12s'); + }); + + it('shows minutes for longer runs', () => { + expect(formatDuration(60000)).toBe('1m'); + expect(formatDuration(65000)).toBe('1m 5s'); + }); +}); diff --git a/tests/server.test.js b/tests/server.test.js index b2eb420..d46d3d4 100644 --- a/tests/server.test.js +++ b/tests/server.test.js @@ -185,6 +185,7 @@ describe('GET /api/session-config', () => { '~deepseek/deepseek-v4-flash-latest', ]); expect(res.body.allowUserModelSelection).toBe(false); + expect(res.body.maxConcurrency).toBe(4); expect(res.body.defaults).toEqual({ minRuns: 1, maxRuns: 5, @@ -208,6 +209,7 @@ describe('GET /api/session-config', () => { 'google/gemini-3.6-flash', ], allowUserModelSelection: true, + maxConcurrency: 2, defaults: { minRuns: 2, maxRuns: 4 }, initialSession: { promptA: 'Prompt A', @@ -226,6 +228,7 @@ describe('GET /api/session-config', () => { 'google/gemini-3.6-flash', ]); expect(res.body.allowUserModelSelection).toBe(true); + expect(res.body.maxConcurrency).toBe(2); expect(res.body.defaults.minRuns).toBe(2); expect(res.body.defaults.maxRuns).toBe(4); expect(res.body.initialSession.promptA).toBe('Prompt A'); @@ -559,6 +562,30 @@ describe('POST /api/eval/compare', () => { expect(opts.prompts.map((p) => p.id)).toEqual(['A', 'B']); expect(opts.cases).toHaveLength(2); expect(opts.runs).toBe(2); + expect(opts.maxConcurrency).toBe(4); + }); + + it('forwards maxConcurrency from session.config.json', async () => { + fs.readFile.mockImplementation(async (p) => { + if (String(p).includes('session.config.json')) { + return JSON.stringify({ maxConcurrency: 8 }); + } + throw new Error('ENOENT'); + }); + runPromptComparison.mockResolvedValue({ + conditions: { metricId: 'exact-match', runs: 1, caseCount: 1, maxConcurrency: 8 }, + cases: [], + prompts: [], + comparison: { outcome: 'unscored', winnerId: null, means: {} }, + }); + + const res = await request(app) + .post('/api/eval/compare') + .send({ promptA: 'Hi', input: 'x', runs: 1 }); + + expect(res.status).toBe(200); + const [, opts] = runPromptComparison.mock.calls[0]; + expect(opts.maxConcurrency).toBe(8); }); it('accepts a single prompt when promptB is omitted', async () => { diff --git a/tests/session-config.test.js b/tests/session-config.test.js index 586fa4c..b6c47d6 100644 --- a/tests/session-config.test.js +++ b/tests/session-config.test.js @@ -1,4 +1,5 @@ import { describe, it, expect } from 'vitest'; +import { DEFAULT_CONCURRENCY } from '../lib/concurrency.js'; import { DEFAULT_ALLOWED_MODELS, DEFAULT_MODEL_REF, @@ -14,6 +15,7 @@ describe('normalizeSessionConfig', () => { model: DEFAULT_MODEL_REF, allowedModels: [...DEFAULT_ALLOWED_MODELS], allowUserModelSelection: false, + maxConcurrency: DEFAULT_CONCURRENCY, defaults: { ...FALLBACK_DEFAULTS }, initialSession: { promptA: '', promptB: '', cases: [] }, }); @@ -21,6 +23,7 @@ describe('normalizeSessionConfig', () => { model: DEFAULT_MODEL_REF, allowedModels: [...DEFAULT_ALLOWED_MODELS], allowUserModelSelection: false, + maxConcurrency: DEFAULT_CONCURRENCY, defaults: { ...FALLBACK_DEFAULTS }, initialSession: { promptA: '', promptB: '', cases: [] }, }); @@ -71,6 +74,13 @@ describe('normalizeSessionConfig', () => { }); }); + it('normalizes maxConcurrency and defaults when omitted', () => { + expect(normalizeSessionConfig({ maxConcurrency: 2 }).maxConcurrency).toBe(2); + expect(normalizeSessionConfig({ maxConcurrency: 0 }).maxConcurrency).toBe(1); + expect(normalizeSessionConfig({ maxConcurrency: 99 }).maxConcurrency).toBe(50); + expect(normalizeSessionConfig({}).maxConcurrency).toBe(DEFAULT_CONCURRENCY); + }); + it('clamps bounds to the fallback range and ignores inverted pairs', () => { expect(normalizeSessionConfig({ defaults: { minRuns: 0, maxRuns: 99 }, From 178a5558a389d7836af99ac325d607f11c795ee4 Mon Sep 17 00:00:00 2001 From: Brian Genisio Date: Mon, 14 Sep 2026 08:49:55 -0400 Subject: [PATCH 2/2] Validate prompts before eval and fix duration rounding. Reject empty rendered prompts before any batch LLM calls, and format 59999ms as 1m instead of 60s. Co-authored-by: Cursor --- lib/eval-compare.js | 20 ++++++++++++++++++++ lib/format-duration.js | 1 + tests/eval-compare.test.js | 25 +++++++++++++++++++++++++ tests/format-duration.test.js | 1 + 4 files changed, 47 insertions(+) diff --git a/lib/eval-compare.js b/lib/eval-compare.js index d1db937..ca044b6 100644 --- a/lib/eval-compare.js +++ b/lib/eval-compare.js @@ -7,6 +7,7 @@ import { createConcurrencyLimiter, normalizeConcurrency } from './concurrency.js'; import { DEFAULT_METRIC_ID, summarizeScores } from './metrics/index.js'; import { normalizeRunCount, runEvalBatch } from './eval-run.js'; +import { renderPromptTemplate } from './prompt-render.js'; export const MIN_EVAL_CASES = 1; export const MAX_EVAL_CASES = 5; @@ -139,6 +140,24 @@ function validatePrompts(prompts) { } } +/** + * Render every prompt × case and reject before any batch work if one is empty. + * @param {PromptVariant[]} prompts + * @param {Array<{ input: string }>} cases + */ +function assertRenderedPrompts(prompts, cases) { + for (const testCase of cases) { + for (const p of prompts) { + const renderedPrompt = renderPromptTemplate(p.promptTemplate, testCase.input); + if (!renderedPrompt.trim()) { + const err = new Error('promptTemplate (or input) must produce a non-empty prompt'); + err.code = 'EMPTY_PROMPT'; + throw err; + } + } + } +} + /** * Run prompts across one or more test cases under shared metric/runs. * @@ -167,6 +186,7 @@ export async function runPromptComparison(deps, opts) { const runBatch = opts.runBatch ?? runEvalBatch; const anyExpected = cases.some((c) => c.expectedAnswer.trim() !== ''); + assertRenderedPrompts(prompts, cases); const startedAt = Date.now(); const conditions = { metricId: anyExpected ? metricId : null, diff --git a/lib/format-duration.js b/lib/format-duration.js index 4619075..8da1032 100644 --- a/lib/format-duration.js +++ b/lib/format-duration.js @@ -16,6 +16,7 @@ export function formatDuration(ms) { const rounded = totalSeconds < 10 ? Math.round(totalSeconds * 10) / 10 : Math.round(totalSeconds); + if (rounded >= 60) return '1m'; return Number.isInteger(rounded) ? `${rounded}s` : `${rounded.toFixed(1)}s`; } diff --git a/tests/eval-compare.test.js b/tests/eval-compare.test.js index ebda9ca..1c893c0 100644 --- a/tests/eval-compare.test.js +++ b/tests/eval-compare.test.js @@ -226,6 +226,31 @@ describe('runPromptComparison', () => { expect(result.cases[0].prompts.map((p) => p.id)).toEqual(['A', 'B']); }); + it('rejects an empty rendered prompt before any runBatch call', async () => { + const runBatch = vi.fn(); + const complete = vi.fn(); + + await expect( + runPromptComparison( + { llm: { complete } }, + { + prompts: [ + { id: 'A', label: 'Prompt A', promptTemplate: 'Capital of {{input}}' }, + { id: 'B', label: 'Prompt B', promptTemplate: '{{input}}' }, + ], + cases: [ + { input: 'France', expectedAnswer: 'Paris' }, + { input: ' ', expectedAnswer: 'Madrid' }, + ], + runBatch, + }, + ), + ).rejects.toMatchObject({ code: 'EMPTY_PROMPT' }); + + expect(runBatch).not.toHaveBeenCalled(); + expect(complete).not.toHaveBeenCalled(); + }); + it('rejects an empty prompts list', async () => { await expect( runPromptComparison( diff --git a/tests/format-duration.test.js b/tests/format-duration.test.js index a428e20..069ad43 100644 --- a/tests/format-duration.test.js +++ b/tests/format-duration.test.js @@ -18,6 +18,7 @@ describe('formatDuration', () => { expect(formatDuration(4200)).toBe('4.2s'); expect(formatDuration(10000)).toBe('10s'); expect(formatDuration(12400)).toBe('12s'); + expect(formatDuration(59999)).toBe('1m'); }); it('shows minutes for longer runs', () => {