Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,7 @@ Fill in `.env` with the API key (and optional `*_BASE_URL`) for the provider you
- `model` (optional) — `provider/model-id` (default `anthropic/claude-sonnet-4-6`); must be listed in `allowedModels`
- `allowedModels` (optional) — picker list of `provider/model-id` refs (defaults to Anthropic, OpenAI, Gemini, and DeepSeek examples above)
- `allowUserModelSelection` (optional) — when `true`, show a model picker and let the saved eval session override `model` with an entry from `allowedModels` (default `false`)
- `maxConcurrency` (optional) — max in-flight LLM calls during an evaluation (default `4`, range 1–50). Set to `1` for serial.
- `defaults` (optional) — `minRuns`, `maxRuns`, `minCases`, `maxCases` (each 1–5)
- `initialSession` (optional) — `promptA`, `promptB`, and `cases` (`input` / `expectedAnswer`)

Expand Down
51 changes: 51 additions & 0 deletions lib/concurrency.js
Original file line number Diff line number Diff line change
@@ -0,0 +1,51 @@
/**
* Bounded in-flight pool for independent async work (eval LLM calls).
*/

export const MIN_CONCURRENCY = 1;
export const MAX_CONCURRENCY = 50;
export const DEFAULT_CONCURRENCY = 4;

/**
* @param {unknown} value
* @returns {number}
*/
export function normalizeConcurrency(value) {
const n = Number.parseInt(String(value ?? ''), 10);
if (!Number.isFinite(n)) return DEFAULT_CONCURRENCY;
return Math.min(MAX_CONCURRENCY, Math.max(MIN_CONCURRENCY, n));
}

/**
* @param {unknown} max
* @returns {{ concurrency: number, schedule: <T>(fn: () => T | Promise<T>) => Promise<T> }}
*/
export function createConcurrencyLimiter(max) {
const concurrency = normalizeConcurrency(max);
let active = 0;
/** @type {Array<() => void>} */
const queue = [];

/**
* @template T
* @param {() => T | Promise<T>} fn
* @returns {Promise<T>}
*/
async function schedule(fn) {
while (active >= concurrency) {
await new Promise((resolve) => {
queue.push(resolve);
});
}
active += 1;
try {
return await fn();
} finally {
active -= 1;
const next = queue.shift();
if (next) next();
}
}

return { concurrency, schedule };
}
91 changes: 64 additions & 27 deletions lib/eval-compare.js
Original file line number Diff line number Diff line change
Expand Up @@ -4,8 +4,10 @@
* so every run is an independent LLM complete() call.
*/

import { createConcurrencyLimiter, normalizeConcurrency } from './concurrency.js';
import { DEFAULT_METRIC_ID, summarizeScores } from './metrics/index.js';
import { normalizeRunCount, runEvalBatch } from './eval-run.js';
import { renderPromptTemplate } from './prompt-render.js';

export const MIN_EVAL_CASES = 1;
export const MAX_EVAL_CASES = 5;
Expand Down Expand Up @@ -138,6 +140,24 @@ function validatePrompts(prompts) {
}
}

/**
* Render every prompt × case and reject before any batch work if one is empty.
* @param {PromptVariant[]} prompts
* @param {Array<{ input: string }>} cases
*/
function assertRenderedPrompts(prompts, cases) {
for (const testCase of cases) {
for (const p of prompts) {
const renderedPrompt = renderPromptTemplate(p.promptTemplate, testCase.input);
if (!renderedPrompt.trim()) {
const err = new Error('promptTemplate (or input) must produce a non-empty prompt');
err.code = 'EMPTY_PROMPT';
throw err;
}
}
}
}

/**
* Run prompts across one or more test cases under shared metric/runs.
*
Expand All @@ -149,6 +169,7 @@ function validatePrompts(prompts) {
* expectedAnswer?: string,
* metricId?: string,
* runs?: number,
* maxConcurrency?: number,
* sessionInput?: Record<string, unknown>,
* runBatch?: typeof runEvalBatch,
* }} opts
Expand All @@ -160,54 +181,67 @@ export async function runPromptComparison(deps, opts) {
const cases = normalizeCases(opts);
const metricId = opts.metricId ?? DEFAULT_METRIC_ID;
const runs = normalizeRunCount(opts.runs);
const maxConcurrency = normalizeConcurrency(opts.maxConcurrency);
const { schedule } = createConcurrencyLimiter(maxConcurrency);
const runBatch = opts.runBatch ?? runEvalBatch;

const anyExpected = cases.some((c) => c.expectedAnswer.trim() !== '');
assertRenderedPrompts(prompts, cases);
const startedAt = Date.now();
const conditions = {
metricId: anyExpected ? metricId : null,
runs,
caseCount: cases.length,
maxConcurrency,
};

/** @type {Array<object>} */
const caseResults = [];
/** @type {Record<string, number[]>} */
const overallScoresByPrompt = Object.fromEntries(prompts.map((p) => [p.id, []]));

for (const testCase of cases) {
/** @type {Array<object>} */
const promptResults = [];
for (const p of prompts) {
const batch = await runBatch(deps, {
promptTemplate: p.promptTemplate,
input: testCase.input,
expectedAnswer: testCase.expectedAnswer,
metricId,
runs,
sessionInput: opts.sessionInput,
});
promptResults.push({
id: p.id,
label: p.label || p.id,
promptTemplate: p.promptTemplate,
...batch,
});
for (const r of batch.results) {
const casePromptResults = await Promise.all(
cases.map((testCase) =>
Promise.all(
prompts.map(async (p) => {
Comment thread
coderabbitai[bot] marked this conversation as resolved.
const batch = await runBatch(deps, {
promptTemplate: p.promptTemplate,
input: testCase.input,
expectedAnswer: testCase.expectedAnswer,
metricId,
runs,
sessionInput: opts.sessionInput,
maxConcurrency,
schedule,
});
return {
id: p.id,
label: p.label || p.id,
promptTemplate: p.promptTemplate,
...batch,
};
}),
),
),
);

/** @type {Array<object>} */
const caseResults = cases.map((testCase, i) => {
const promptResults = casePromptResults[i];
for (const promptResult of promptResults) {
for (const r of promptResult.results ?? []) {
if (typeof r.score === 'number' && Number.isFinite(r.score)) {
overallScoresByPrompt[p.id].push(r.score);
overallScoresByPrompt[promptResult.id].push(r.score);
}
}
}

caseResults.push({
return {
id: testCase.id,
label: testCase.label,
input: testCase.input,
expectedAnswer: testCase.expectedAnswer.trim() !== '' ? testCase.expectedAnswer : null,
prompts: promptResults,
comparison: compareByMean(promptResults),
});
}
};
});

const overallPrompts = prompts.map((p) => {
const aggregate = summarizeScores(overallScoresByPrompt[p.id]);
Expand All @@ -230,7 +264,10 @@ export async function runPromptComparison(deps, opts) {
});

return {
conditions,
conditions: {
...conditions,
durationMs: Math.max(0, Date.now() - startedAt),
},
cases: caseResults,
prompts: overallPrompts,
comparison: compareByMean(overallPrompts),
Expand Down
5 changes: 4 additions & 1 deletion lib/eval-report.js
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@

import fs from 'fs/promises';
import path from 'path';
import { formatDuration } from './format-duration.js';

/**
* @param {unknown} value
Expand Down Expand Up @@ -45,7 +46,7 @@ function clip(text, max = 240) {
* promptA?: string,
* promptB?: string,
* result: {
* conditions?: { metricId?: string | null, runs?: number, caseCount?: number },
* conditions?: { metricId?: string | null, runs?: number, caseCount?: number, maxConcurrency?: number, durationMs?: number },
* comparison?: { outcome?: string, winnerId?: string | null, means?: Record<string, number | null> },
* prompts?: Array<{ id: string, label?: string, aggregate?: { mean?: number, min?: number, max?: number, count?: number } | null }>,
* cases?: Array<{
Expand Down Expand Up @@ -88,7 +89,9 @@ export function buildEvalReportMarkdown(opts) {
`| Provider | ${escapeCell(opts.provider || '—')} |`,
`| Metric | ${escapeCell(conditions.metricId ?? 'none (unscored)')} |`,
`| Runs each | ${escapeCell(conditions.runs ?? '—')} |`,
`| Max concurrency | ${escapeCell(conditions.maxConcurrency ?? '—')} |`,
`| Cases | ${escapeCell(conditions.caseCount ?? cases.length)} |`,
`| Duration | ${escapeCell(formatDuration(conditions.durationMs) || '—')} |`,
`| Mode | ${multi ? 'Prompt A vs Prompt B' : 'Single prompt'} |`,
'',
'## Prompts',
Expand Down
31 changes: 19 additions & 12 deletions lib/eval-run.js
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@ import { randomUUID } from 'node:crypto';
import fs from 'node:fs/promises';
import path from 'node:path';
import { fileURLToPath } from 'node:url';
import { createConcurrencyLimiter } from './concurrency.js';
import { renderPromptTemplate } from './prompt-render.js';
import {
DEFAULT_METRIC_ID,
Expand Down Expand Up @@ -115,7 +116,7 @@ export async function runSingleEval(deps, opts) {
}

/**
* Render a template and execute N independent LLM runs sequentially.
* Render a template and execute N independent LLM runs with bounded concurrency.
*
* @param {EvalRunDeps} deps
* @param {{
Expand All @@ -124,6 +125,8 @@ export async function runSingleEval(deps, opts) {
* runs?: number,
* expectedAnswer?: string,
* metricId?: string,
* maxConcurrency?: number,
* schedule?: <T>(fn: () => T | Promise<T>) => Promise<T>,
* }} opts
*/
export async function runEvalBatch(deps, opts) {
Expand All @@ -133,6 +136,9 @@ export async function runEvalBatch(deps, opts) {
const metricId = opts.metricId ?? DEFAULT_METRIC_ID;
const runs = normalizeRunCount(opts.runs);
const renderedPrompt = renderPromptTemplate(promptTemplate, input);
const schedule = typeof opts.schedule === 'function'
? opts.schedule
: createConcurrencyLimiter(opts.maxConcurrency).schedule;

if (!renderedPrompt.trim()) {
const err = new Error('promptTemplate (or input) must produce a non-empty prompt');
Expand All @@ -141,17 +147,18 @@ export async function runEvalBatch(deps, opts) {
}

/** @type {EvalResult[]} */
const results = [];
for (let run = 1; run <= runs; run += 1) {
results.push(
await runSingleEval(deps, {
renderedPrompt,
run,
expectedAnswer,
metricId,
}),
);
}
const results = await Promise.all(
Array.from({ length: runs }, (_, i) =>
schedule(() =>
runSingleEval(deps, {
renderedPrompt,
run: i + 1,
expectedAnswer,
metricId,
}),
),
),
);

const scoringEnabled = expectedAnswer.trim() !== '';
const aggregate = scoringEnabled
Expand Down
27 changes: 27 additions & 0 deletions lib/format-duration.js
Original file line number Diff line number Diff line change
@@ -0,0 +1,27 @@
/**
* Compact wall-clock duration for eval summaries.
*/

/**
* @param {unknown} ms
* @returns {string}
*/
export function formatDuration(ms) {
const n = Number(ms);
if (!Number.isFinite(n) || n < 0) return '';
if (n < 1000) return `${Math.round(n)}ms`;

const totalSeconds = n / 1000;
if (totalSeconds < 60) {
const rounded = totalSeconds < 10
? Math.round(totalSeconds * 10) / 10
: Math.round(totalSeconds);
if (rounded >= 60) return '1m';
return Number.isInteger(rounded) ? `${rounded}s` : `${rounded.toFixed(1)}s`;
Comment thread
coderabbitai[bot] marked this conversation as resolved.
}

const minutes = Math.floor(totalSeconds / 60);
const seconds = Math.round(totalSeconds % 60);
if (seconds === 60) return `${minutes + 1}m`;
return seconds === 0 ? `${minutes}m` : `${minutes}m ${seconds}s`;
}
3 changes: 3 additions & 0 deletions lib/session-config.js
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
* This module is browser-safe.
*/

import { normalizeConcurrency } from './concurrency.js';
import { DEFAULT_MODEL_REF, parseModelRef } from './llm/model-ref.js';

export { DEFAULT_MODEL_REF };
Expand Down Expand Up @@ -173,6 +174,7 @@ function resolveConfiguredModel(rawModel, allowedModels) {
* model: string,
* allowedModels: string[],
* allowUserModelSelection: boolean,
* maxConcurrency: number,
* defaults: { minRuns: number, maxRuns: number, minCases: number, maxCases: number },
* initialSession: { promptA: string, promptB: string, cases: { input: string, expectedAnswer: string }[] },
* }}
Expand Down Expand Up @@ -205,6 +207,7 @@ export function normalizeSessionConfig(raw) {
model,
allowedModels,
allowUserModelSelection: src.allowUserModelSelection === true,
maxConcurrency: normalizeConcurrency(src.maxConcurrency),
defaults: {
minRuns: runs.min,
maxRuns: runs.max,
Expand Down
13 changes: 9 additions & 4 deletions public/app.js
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@
*/

import { isRenderableResult, normalizeEvalSession } from '../lib/eval-session.js';
import { formatDuration } from '../lib/format-duration.js';
import { collectPromptScoresByCase } from '../lib/score-distribution.js';
import { enqueueSessionsWrite } from '../lib/sessions-file.js';
import {
Expand Down Expand Up @@ -431,10 +432,14 @@ function renderComparison(data) {
caseDetailsPanel.open = false;

const multi = isMultiPrompt(data);
const { runs, caseCount } = data.conditions;
resultsMeta.textContent =
`${caseCount} case${caseCount === 1 ? '' : 's'} · ${runs} run${runs === 1 ? '' : 's'} each`
+ (multi ? ' · A vs B' : '');
const { runs, caseCount, durationMs } = data.conditions;
const duration = formatDuration(durationMs);
resultsMeta.textContent = [
`${caseCount} case${caseCount === 1 ? '' : 's'}`,
`${runs} run${runs === 1 ? '' : 's'} each`,
duration,
multi ? 'A vs B' : '',
].filter(Boolean).join(' · ');

renderVerdict(data);
renderOverallCards(data);
Expand Down
Loading