From 54568d3a687a6c19378c898e527fa09a81e68cb0 Mon Sep 17 00:00:00 2001 From: Dylan O'Neill Date: Tue, 29 Sep 2026 20:50:00 -0700 Subject: [PATCH 1/2] feat(evaluations)!: collapse handler and judge_handlers into one handlers list run() no longer takes separate handler and judge_handlers arguments. Pass every handler a run needs in one handlers list. run() picks the handler for each call: the sole entry if there is only one, otherwise the entry tagged for the provider that call needs, and for a judge, also for its mode. A caller no longer decides ahead of time which handler serves generation and which serves each judge's provider. run() also rejects two handlers tagged for the identical provider and mode, and a generation provider that cannot be resolved to exactly one handler, before any network request. This also drops the wildcard ("*", mode) fallback and the priority order between an exact-provider judge handler and a wildcard adapter: _select_handler compares a provider name literally. An agent-mode handler still serves a messages-mode judge for the same provider, with its messages collapsed into one instructions block. handler/judge_handlers shipped in launchdarkly-ai-server 0.2.3 on PyPI, so this is a second breaking change on top of an already released one, not a pre-release correction. Implements the design in https://github.com/launchdarkly/ai-sdks-monorepo/pull/30. Co-Authored-By: Claude Sonnet 5 --- packages/ai/README.md | 4 +- packages/client/README.md | 12 +- .../evaluations/module.py | 73 ++-- .../evaluations/runner.py | 151 ++++---- packages/client/tests/test_evaluations_run.py | 365 +++++++++++++----- 5 files changed, 372 insertions(+), 233 deletions(-) diff --git a/packages/ai/README.md b/packages/ai/README.md index 829bb567..a2d43355 100644 --- a/packages/ai/README.md +++ b/packages/ai/README.md @@ -64,7 +64,7 @@ result = await evals.run( project_key="my-project", key="unique-evaluation-key", dataset="golden-dataset", - handler=my_handler, + handlers=[my_handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[ Judge(key="accuracy-judge"), @@ -73,7 +73,7 @@ result = await evals.run( ) ``` -`LD_API_TOKEN` is required. Configure `LD_SDK_KEY` — or initialize your own client with `init_client(client=...)` — to emit one `$ld:ai:offline-evals:generation` event per generated row, plus one `$ld:ai:offline-evals:criterion` event per `(row, criterion)` when `criteria` are supplied, through the standard SDK event transport. The SDK reports scores; LaunchDarkly rules on them at ingest. A judge served by a different provider than `generation` needs a handler for it in `judge_handlers`. Each row's tool calls are recorded during generation and rendered into the judge's `{{message_history}}`, between the row input and the generated output, so a rubric can grade the tool trajectory as well as the final answer. Use `LD_API_BASE_URI` for staging or local management API traffic; it is separate from the SDK delivery setting `LD_BASE_URI`. Evaluation-run links use the explicit `ui_base_uri` option or `LD_UI_BASE_URI`, defaulting to `https://app.launchdarkly.com`; set it when the project is not in production, or a run created elsewhere still links to the production app. See the [core evaluations guide](../client/README.md#run-an-evaluation-from-code). +`LD_API_TOKEN` is required. Configure `LD_SDK_KEY` — or initialize your own client with `init_client(client=...)` — to emit one `$ld:ai:offline-evals:generation` event per generated row, plus one `$ld:ai:offline-evals:criterion` event per `(row, criterion)` when `criteria` are supplied, through the standard SDK event transport. The SDK reports scores; LaunchDarkly rules on them at ingest. A judge served by a different provider than `generation` needs its own tagged handler in `handlers`. Each row's tool calls are recorded during generation and rendered into the judge's `{{message_history}}`, between the row input and the generated output, so a rubric can grade the tool trajectory as well as the final answer. Use `LD_API_BASE_URI` for staging or local management API traffic; it is separate from the SDK delivery setting `LD_BASE_URI`. Evaluation-run links use the explicit `ui_base_uri` option or `LD_UI_BASE_URI`, defaulting to `https://app.launchdarkly.com`; set it when the project is not in production, or a run created elsewhere still links to the production app. See the [core evaluations guide](../client/README.md#run-an-evaluation-from-code). --- diff --git a/packages/client/README.md b/packages/client/README.md index b2607d4e..60aa58c0 100644 --- a/packages/client/README.md +++ b/packages/client/README.md @@ -64,7 +64,7 @@ async def main() -> int: project_key="my-project", key="support-qa-2026-08-20", dataset="support-golden", - handler=create_openai_messages_handler(), + handlers=[create_openai_messages_handler()], generation={ "provider": "OpenAI", "model": "gpt-4o", @@ -100,15 +100,15 @@ result = await init_evaluations().run( project_key="my-project", key="support-qa-2026-08-20", dataset="support-golden", - handler=create_openai_messages_handler(), + # One handler per provider a call can hit. The accuracy-judge criterion + # below is served by a different provider than generation, so both + # handlers are listed here. + handlers=[create_openai_messages_handler(), create_claude_messages_handler()], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[ Judge(key="accuracy-judge", threshold=0.8), Scorer(name="mentions-policy", fn=mentions_policy), ], - # Needed only because this judge is served by a different provider than - # the generation config above. - judge_handlers=[create_claude_messages_handler()], ) ``` @@ -161,7 +161,7 @@ Two limits keep a trajectory from spending the judge's context window: at most 5 A tool result is now judge-prompt input. It stays literal for the same reason the generated output does: the judge config is handed to the handler unrendered and the handler makes exactly one template pass, so a `{{...}}` sequence coming back from a tool is never expanded into the judge prompt. -**Judges are independent AI Configs, so handlers are routed per judge.** A judge may resolve to a different provider or mode than `generation`, and a handler built for one provider cannot execute another's config. `handler` runs a judge when it provides for that judge's provider; pass handlers for any other providers in `judge_handlers`. Selection prefers a handler naming the judge's provider outright over a wildcard multi-provider adapter, and an agent-mode handler can serve a messages-mode judge with its messages collapsed into one instructions block. A plain callable that declares no `provides_for` routes itself, exactly as it already does for the generation config. +**Judges are independent AI Configs, so handlers are routed per judge.** A judge may resolve to a different provider or mode than `generation`, and a handler built for one provider cannot execute another's config. Pass every handler your run needs in `handlers`. With one handler, it runs generation and every judge. With more than one, each is tagged with `provides_for` (via `create_handler()` or a provider package's `create_*_handler()`), and `run()` picks the entry tagged for the provider a call needs — for a judge, its exact `(provider, mode)` pair when that disambiguates, otherwise the sole handler left for that provider. An agent-mode handler can still serve a messages-mode judge for the same provider, with its messages collapsed into one instructions block. `run()` raises before any record is created if a call's provider matches no handler, or matches more than one with no way to disambiguate. Judges are resolved through flag delivery, and handlers are matched to them, **before** any evaluation records are created — a missing judge or one no handler covers fails the run up front rather than after the generation spend. After that point a criterion failure never aborts the run: an unparseable judge response, an out-of-range score, a raising handler or scorer, and a row whose generation errored each become a per-criterion `ERROR` event with a cause code (`invalid_judge_output`, `invalid_score`, `handler_raised`, `scorer_raised`, `generation_incomplete`) and a top-level `errorMessage`. Event *delivery* is different: the backend needs one result per `(row, criterion)` to finish row accounting, so if tracking a criterion event fails, every remaining result is still attempted and flushed and then `run()` raises — rather than polling to its timeout with the cause hidden. diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/module.py b/packages/client/src/launchdarkly_ai_server/evaluations/module.py index 4cc684aa..8266649b 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/module.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/module.py @@ -24,6 +24,7 @@ ToolImplementation, _provides_for, _segment, + _select_handler, ) from .types import AIConfig, EvalRunResult, GenerationConfig, RunSummary @@ -126,12 +127,11 @@ async def run( project_key: str, key: str, dataset: str, - handler: EvalHandler, + handlers: list[EvalHandler], generation: GenerationConfig | None = None, ai_config: AIConfig | None = None, tools: Mapping[str, ToolImplementation] | None = None, criteria: list[Criterion] | None = None, - judge_handlers: list[EvalHandler] | None = None, concurrency: int = 10, poll_interval_seconds: float | None = None, poll_timeout_seconds: float | None = None, @@ -139,17 +139,21 @@ async def run( """ Create and run an evaluation in the caller's process. - Each dataset row is generated with ``handler``; every entry in - ``criteria`` — LaunchDarkly :class:`Judge` references and local - deterministic :class:`Scorer` functions — is then run against each - generated row, and one evaluation event is emitted per + run() generates each dataset row with a handler from ``handlers``. + Every entry in ``criteria`` (a LaunchDarkly :class:`Judge` reference + or a local :class:`Scorer` function) then runs against each + generated row. run() emits one evaluation event per ``(row, criterion)`` result. - A :class:`Judge` is an independent AI Config and may be served by a - different provider or mode than ``generation``. ``handler`` runs a judge - only when it provides for that judge's provider; pass handlers for any - other providers your judges use in ``judge_handlers``. A judge no - handler covers fails the run before any records are created. + Pass one handler in ``handlers`` when generation and every judge use + the same provider. Pass more than one handler otherwise. Build each + handler with ``create_handler()`` (or a provider package's + ``create_*_handler()``) and give it a ``provides_for`` tag for its + provider and mode. With more than one handler, run() picks the + handler tagged for the provider a call needs, and for a judge, also + for its mode. run() raises an error before any record exists if no + handler matches a call, or if more than one handler could match and + the call has no mode to break the tie. The returned pass/fail result is derived from LaunchDarkly's run summary. A CI script can exit with ``0 if result.passed else 1`` after awaiting @@ -174,7 +178,7 @@ async def run( project_key=project_key, key=key, dataset=dataset, - handler=handler, + handlers=handlers, concurrency=concurrency, poll_interval_seconds=poll_interval_seconds, poll_timeout_seconds=poll_timeout_seconds, @@ -206,11 +210,12 @@ async def run( Judge(key=judge_key) for judge_key in ai_config_variation.judge_keys ] generation = self._validate_generation(generation) + # Checkable as soon as generation is known, and always before any + # network request the run still needs to make. + generation_handler = _select_handler(generation["provider"], handlers) run_tools = dict(tools or {}) run_criteria = list(criteria or []) - run_judge_handlers = list(judge_handlers or []) self._validate_criteria(run_criteria) - self._validate_judge_handlers(run_judge_handlers) ld_judges = [ criterion for criterion in run_criteria if isinstance(criterion, Judge) ] @@ -236,7 +241,7 @@ async def run( resolved_tool.version, ) resolved_judges = await self._runner._resolve_judges( - project_key, ld_judges, handler, run_judge_handlers + project_key, ld_judges, handlers ) dataset_ref = await asyncio.to_thread( self._runner._fetch_dataset, project_key, dataset @@ -261,7 +266,7 @@ async def run( config = self._runner._build_handler_config(generation, resolved_tools) results = await self._runner._run_rows( rows, - handler, + generation_handler, config, run_tools, concurrency, @@ -407,24 +412,29 @@ def _validate_criteria(criteria: list[Criterion]) -> None: ) @staticmethod - def _validate_judge_handlers(judge_handlers: list[EvalHandler]) -> None: - """Reject judge handlers that cannot be routed by provider and mode. + def _validate_handlers(handlers: list[EvalHandler]) -> None: + """Reject an empty or invalid ``handlers`` list before any request. - A judge handler is only ever chosen by matching its ``provides_for`` - against the judge's resolved provider and mode. One without that - metadata could never be selected, so it would silently fall through to - the generation handler instead of running the judge it was passed for. + Two entries tagged for the identical provider and mode could never be + told apart later, so that check also runs here rather than at + selection time. """ - for index, candidate in enumerate(judge_handlers): + if not handlers: + raise EvaluationsError("handlers must not be empty") + seen: dict[tuple[str, str], int] = {} + for index, candidate in enumerate(handlers): if not callable(candidate): - raise EvaluationsError(f"judge_handlers[{index}] must be callable") - if _provides_for(candidate) is None: + raise EvaluationsError(f"handlers[{index}] must be callable") + tag = _provides_for(candidate) + if tag is None: + continue + if tag in seen: raise EvaluationsError( - f"judge_handlers[{index}] does not declare provides_for. " - "Build judge handlers with create_handler() (or a provider " - "package's create_*_handler()) so they can be matched to a " - "judge's provider and mode." + f"handlers[{seen[tag]}] and handlers[{index}] both declare " + f"provides_for {tag[0]!r} in {tag[1]!r} mode. Only one " + "handler may serve a given provider and mode." ) + seen[tag] = index @staticmethod def _validate_run_args( @@ -432,7 +442,7 @@ def _validate_run_args( project_key: str, key: str, dataset: str, - handler: EvalHandler, + handlers: list[EvalHandler], concurrency: int, poll_interval_seconds: float, poll_timeout_seconds: float, @@ -444,8 +454,7 @@ def _validate_run_args( ): if not value.strip(): raise EvaluationsError(f"{name} must not be blank") - if not callable(handler): - raise EvaluationsError("handler must be callable") + EvaluationsModule._validate_handlers(handlers) if concurrency < 1: raise EvaluationsError("concurrency must be at least 1") for name, seconds in ( diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/runner.py b/packages/client/src/launchdarkly_ai_server/evaluations/runner.py index 6240e980..1063f34e 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/runner.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/runner.py @@ -85,83 +85,52 @@ def _provides_for( return None -def _covers_provider( - provides_for: tuple[str, Literal["agent", "messages"]], +def _select_handler( provider: str | None, -) -> bool: - return provides_for[0] == provider or provides_for[0] == "*" - - -def _find_judge_handler( - judge_handlers: list[EvalHandler], - provider: str | None, - mode: Literal["agent", "messages"], -) -> EvalHandler | None: - """Find a handler for ``provider`` in ``mode``, exact match before wildcard. - - A wildcard handler is a fallback for multi-provider adapters, so it is only - chosen when no handler names the provider outright -- the priority - ``config()`` already applies to a generation config. Searching in one pass - would instead let the order the caller happened to list its handlers in - decide, sending an OpenAI judge through a LangChain adapter that was merely - listed first. - """ - for exact in (True, False): - for candidate in judge_handlers: - provides_for = _provides_for(candidate) - if provides_for is None or provides_for[1] != mode: - continue - if exact: - if provides_for[0] == provider: - return candidate - elif provides_for[0] == "*": - return candidate - return None - - -def _select_judge_handler( - resolved: ResolvedJudge, - handler: EvalHandler, - judge_handlers: list[EvalHandler], -) -> JudgeExecution | None: - """Pick the handler that can run this judge's config, or ``None``. - - A judge is an independent AI Config: it may resolve to a different provider - and mode than the evaluation's generation config, and a handler built for - one provider cannot execute another's config. The priority mirrors the - online path (``judges.run_judges``): - - 1. a judge handler in the judge's mode, naming its provider outright - before any wildcard adapter; - 2. an agent-mode judge handler for a messages-mode judge, whose messages - are collapsed into a single instructions block; - 3. the generation handler, when it covers the judge's provider. - - A handler that declares no ``provides_for`` is a plain callable doing its - own routing -- the same contract it already honours for the generation - config -- so it is treated as covering every judge. + handlers: list[EvalHandler], + mode: Literal["agent", "messages"] | None = None, +) -> EvalHandler: + """Pick the handler in ``handlers`` that serves ``provider``. + + A single handler always runs, whatever it declares. With more than one + handler, only entries whose ``provides_for`` names ``provider`` are + candidates. ``mode`` is a judge's resolved mode (generation has none); an + exact ``(provider, mode)`` match wins over any other candidate for that + provider. + + Raises :class:`EvaluationsError` when no candidate serves ``provider``, + or when more than one does and ``mode`` cannot break the tie. """ - match = _find_judge_handler(judge_handlers, resolved.provider, resolved.mode) - if match is not None: - return JudgeExecution(resolved=resolved, handler=match) - if resolved.mode == "messages": - agent_fallback = _find_judge_handler(judge_handlers, resolved.provider, "agent") - if agent_fallback is not None: - return JudgeExecution( - resolved=resolved, handler=agent_fallback, collapse_messages=True - ) - generation_provides_for = _provides_for(handler) - if generation_provides_for is None: - return JudgeExecution(resolved=resolved, handler=handler) - if _covers_provider(generation_provides_for, resolved.provider): - return JudgeExecution( - resolved=resolved, - handler=handler, - collapse_messages=( - generation_provides_for[1] == "agent" and resolved.mode == "messages" - ), + if len(handlers) == 1: + return handlers[0] + + candidates = [ + handler + for handler in handlers + if (tag := _provides_for(handler)) is not None and tag[0] == provider + ] + if mode is not None: + exact = [ + handler + for handler in candidates + if (tag := _provides_for(handler)) is not None and tag[1] == mode + ] + if len(exact) == 1: + return exact[0] + + if not candidates: + raise EvaluationsError(f"no handler registered for provider {provider!r}") + if len(candidates) > 1: + modes = sorted( + tag[1] + for handler in candidates + if (tag := _provides_for(handler)) is not None ) - return None + raise EvaluationsError( + f"{len(candidates)} handlers registered for provider {provider!r} " + f"(modes: {modes}); can't disambiguate for this call." + ) + return candidates[0] def _segment(value: str) -> str: @@ -342,8 +311,7 @@ async def _resolve_judges( self, project_key: str, judges: list[Judge], - handler: EvalHandler, - judge_handlers: list[EvalHandler] | None = None, + handlers: list[EvalHandler], ) -> dict[str, JudgeExecution]: """Resolve LD Judge configs before any evaluation records are created. @@ -351,7 +319,6 @@ async def _resolve_judges( rather than at scoring time, so a judge no handler covers fails the run before any records exist or any generation spend happens. """ - available_judge_handlers = list(judge_handlers or []) resolved: dict[str, JudgeExecution] = {} # variation() rejects a context without kind and key; use the same # context shape the emitted evaluation events are attributed to. @@ -392,18 +359,30 @@ async def _resolve_judges( meta.get("mode") if isinstance(meta.get("mode"), str) else None ), ) - execution = _select_judge_handler( - resolved_judge, handler, available_judge_handlers - ) - if execution is None: + try: + selected = _select_handler( + resolved_judge.provider, handlers, mode=resolved_judge.mode + ) + except EvaluationsError as error: raise EvaluationsError( f"No handler can run LaunchDarkly judge {judge.key!r}: its " f"config is served by provider {resolved_judge.provider!r} in " - f"{resolved_judge.mode!r} mode, which neither the generation " - "handler nor any judge_handlers entry provides for. Pass a " - "handler for that provider to run(judge_handlers=[...])." - ) - resolved[judge.key] = execution + f"{resolved_judge.mode!r} mode. {error} Pass a handler for " + "that provider to run(handlers=[...])." + ) from error + selected_tag = _provides_for(selected) + # A messages-mode judge served by an agent-only handler needs its + # messages folded into one instructions block first. + collapse_messages = bool( + selected_tag is not None + and selected_tag[1] == "agent" + and resolved_judge.mode == "messages" + ) + resolved[judge.key] = JudgeExecution( + resolved=resolved_judge, + handler=selected, + collapse_messages=collapse_messages, + ) return resolved def _fetch_dataset(self, project_key: str, dataset_key: str) -> DatasetRef: diff --git a/packages/client/tests/test_evaluations_run.py b/packages/client/tests/test_evaluations_run.py index 4a4b8571..682661c6 100644 --- a/packages/client/tests/test_evaluations_run.py +++ b/packages/client/tests/test_evaluations_run.py @@ -234,7 +234,7 @@ async def test_complete_run_with_zero_failed_and_error_rows_passes( project_key="proj", key="support-qa-unique", dataset="golden", - handler=successful_handler, + handlers=[successful_handler], tools={"lookup_order": lookup_order}, generation={ "provider": "OpenAI", @@ -410,7 +410,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, ) @@ -473,7 +473,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, ) @@ -535,7 +535,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, ) @@ -594,7 +594,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, ) @@ -648,7 +648,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, ) @@ -693,7 +693,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, poll_interval_seconds=0, poll_timeout_seconds=600, @@ -710,7 +710,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, poll_timeout_seconds=-1, ) @@ -735,7 +735,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, poll_interval_seconds=poll_interval_seconds, poll_timeout_seconds=poll_timeout_seconds, @@ -787,7 +787,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, ) @@ -819,7 +819,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, ) @@ -872,7 +872,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, ) @@ -890,7 +890,7 @@ async def test_run_rejects_instructions_and_messages_before_network_io() -> None project_key="proj", key="eval-key", dataset="golden", - handler=successful_handler, + handlers=[successful_handler], generation={ "provider": "OpenAI", "model": "gpt-4o", @@ -914,7 +914,7 @@ async def test_missing_tool_aborts_before_any_mutating_request() -> None: project_key="proj", key="eval-key", dataset="golden", - handler=successful_handler, + handlers=[successful_handler], tools={"missing_tool": lookup_order}, generation={"provider": "OpenAI", "model": "gpt-4o"}, ) @@ -943,7 +943,7 @@ async def test_empty_dataset_fails_before_evaluation_or_run_creation() -> None: project_key="proj", key="eval-key", dataset="golden", - handler=successful_handler, + handlers=[successful_handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, ) @@ -1040,7 +1040,7 @@ async def fake_init_client(options: dict[str, Any]) -> MagicMock: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, ) @@ -1159,7 +1159,7 @@ async def handler( project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -1290,7 +1290,7 @@ async def handler( project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy", threshold=threshold)], ) @@ -1379,7 +1379,7 @@ async def handler( project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -1450,7 +1450,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="security-judge")], ) @@ -1508,7 +1508,7 @@ def check_refund(row: DatasetRow, output: Any) -> bool: project_key="proj", key="support-qa", dataset="support-golden-v3", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Scorer(name="refund-exists", fn=check_refund)], ) @@ -1637,7 +1637,7 @@ async def handler( project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -1681,7 +1681,7 @@ async def handler( project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -1735,7 +1735,7 @@ async def handler( project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -1755,7 +1755,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[ Judge(key="accuracy"), @@ -1783,7 +1783,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[ Judge(key="Accuracy"), @@ -1819,7 +1819,7 @@ async def handler( project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -1871,7 +1871,7 @@ async def handler( project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[ Judge(key="$ld:ai:judge:accuracy"), @@ -1944,14 +1944,12 @@ async def _generation_only( @pytest.mark.asyncio -async def test_judge_on_another_provider_fails_before_any_records_are_created( +async def test_judge_on_an_uncovered_provider_fails_before_any_records_are_created( monkeypatch: pytest.MonkeyPatch, ) -> None: - """A provider handler cannot execute another provider's judge config. - - Passing it anyway spent the generation budget and then recorded every row - as handler_raised, so the mismatch is caught while it is still only a - configuration error: before the dataset is read or any record is created. + """A judge's provider only becomes known after the judge-resolution GET, + so this check runs there -- but still before the dataset is read or any + evaluation record is created. """ transport = judge_run_transport() judge_variation(monkeypatch, provider="Anthropic") @@ -1962,7 +1960,10 @@ async def test_judge_on_another_provider_fails_before_any_records_are_created( project_key="proj", key="support-qa", dataset="golden", - handler=create_handler(("OpenAI", "messages"), _generation_only), + handlers=[ + create_handler(("OpenAI", "messages"), _generation_only), + create_handler(("Gemini", "messages"), _generation_only), + ], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -1972,6 +1973,36 @@ async def test_judge_on_another_provider_fails_before_any_records_are_created( assert transport.requests == [] +@pytest.mark.asyncio +async def test_single_handler_serves_generation_and_every_judge_regardless_of_provider( + monkeypatch: pytest.MonkeyPatch, + stub_sdk_client: MagicMock, +) -> None: + """A single handler in ``handlers`` always runs, whatever it declares. + + This keeps a single-provider eval script working with one plain handler: + it never needs a ``provides_for`` tag, and it serves a judge on any + provider too. + """ + transport = judge_run_transport() + judge_variation(monkeypatch, provider="Anthropic") + evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + + async def handler(*args: object) -> dict[str, Any]: + return {"output": '{"score": 1, "reasoning": "ok"}'} + + result = await evals.run( + project_key="proj", + key="support-qa", + dataset="golden", + handlers=[handler], + generation={"provider": "OpenAI", "model": "gpt-4o"}, + criteria=[Judge(key="$ld:ai:judge:accuracy")], + ) + + assert result.passed is True + + @pytest.mark.asyncio async def test_judge_handlers_route_a_judge_to_its_own_provider( monkeypatch: pytest.MonkeyPatch, @@ -1996,10 +2027,12 @@ async def anthropic_judge( project_key="proj", key="support-qa", dataset="golden", - handler=create_handler(("OpenAI", "messages"), _generation_only), + handlers=[ + create_handler(("OpenAI", "messages"), _generation_only), + create_handler(("Anthropic", "messages"), anthropic_judge), + ], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], - judge_handlers=[create_handler(("Anthropic", "messages"), anthropic_judge)], ) assert result.passed is True @@ -2011,19 +2044,20 @@ async def anthropic_judge( @pytest.mark.asyncio -@pytest.mark.parametrize("wildcard_first", [True, False]) -async def test_exact_provider_judge_handler_beats_a_wildcard_adapter( +@pytest.mark.parametrize("agent_handler_first", [True, False]) +async def test_exact_mode_judge_handler_disambiguates_same_provider_handlers( monkeypatch: pytest.MonkeyPatch, stub_sdk_client: MagicMock, - wildcard_first: bool, + agent_handler_first: bool, ) -> None: - """A wildcard is a fallback, so the order handlers are listed in cannot decide. + """Two handlers may serve the same provider in different modes. - Taking the first provider-or-wildcard match would send an Anthropic judge - through a multi-provider adapter that merely happened to be listed first. + A judge carries its own resolved mode, so it picks the handler tagged for + that exact ``(provider, mode)`` pair, whichever order ``handlers`` lists + them in. """ transport = judge_run_transport() - judge_variation(monkeypatch, provider="Anthropic") + judge_variation(monkeypatch, provider="Anthropic", mode="messages") evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) chosen: list[str] = [] @@ -2040,55 +2074,29 @@ async def run( return run - wildcard = create_handler(("*", "messages"), judge_handler("wildcard")) - exact = create_handler(("Anthropic", "messages"), judge_handler("exact")) - - result = await evals.run( - project_key="proj", - key="support-qa", - dataset="golden", - handler=create_handler(("OpenAI", "messages"), _generation_only), - generation={"provider": "OpenAI", "model": "gpt-4o"}, - criteria=[Judge(key="$ld:ai:judge:accuracy")], - judge_handlers=[wildcard, exact] if wildcard_first else [exact, wildcard], + agent_handler = create_handler(("Anthropic", "agent"), judge_handler("agent")) + messages_handler = create_handler( + ("Anthropic", "messages"), judge_handler("messages") ) - assert result.passed is True - assert chosen == ["exact"] - - -@pytest.mark.asyncio -async def test_wildcard_judge_handler_runs_a_judge_no_handler_names( - monkeypatch: pytest.MonkeyPatch, - stub_sdk_client: MagicMock, -) -> None: - transport = judge_run_transport() - judge_variation(monkeypatch, provider="Anthropic") - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) - judged: list[dict[str, Any]] = [] - - async def wildcard_judge( - config: dict[str, Any], - user_input: str | None = None, - tool_handlers: dict[str, Callable[..., Any]] | None = None, - variables: dict[str, Any] | None = None, - history: list[dict[str, Any]] | None = None, - ) -> dict[str, Any]: - judged.append(config) - return {"output": '{"score": 1, "reasoning": "ok"}'} - result = await evals.run( project_key="proj", key="support-qa", dataset="golden", - handler=create_handler(("OpenAI", "messages"), _generation_only), + handlers=[ + create_handler(("OpenAI", "messages"), _generation_only), + *( + [agent_handler, messages_handler] + if agent_handler_first + else [messages_handler, agent_handler] + ), + ], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], - judge_handlers=[create_handler(("*", "messages"), wildcard_judge)], ) assert result.passed is True - assert [config["provider"]["name"] for config in judged] == ["Anthropic"] + assert chosen == ["messages"] @pytest.mark.asyncio @@ -2096,7 +2104,8 @@ async def test_agent_handler_runs_a_messages_mode_judge_with_collapsed_messages( monkeypatch: pytest.MonkeyPatch, stub_sdk_client: MagicMock, ) -> None: - """Mirrors the online path's agent-mode fallback for a messages-mode judge.""" + """A messages-mode judge served only by an agent-mode handler for its + provider gets its messages collapsed into a single instructions block.""" transport = judge_run_transport() judge_variation( monkeypatch, @@ -2126,10 +2135,12 @@ async def anthropic_agent_judge( project_key="proj", key="support-qa", dataset="golden", - handler=create_handler(("OpenAI", "messages"), _generation_only), + handlers=[ + create_handler(("OpenAI", "messages"), _generation_only), + create_handler(("Anthropic", "agent"), anthropic_agent_judge), + ], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], - judge_handlers=[create_handler(("Anthropic", "agent"), anthropic_agent_judge)], ) assert result.passed is True @@ -2165,7 +2176,7 @@ async def openai_handler( project_key="proj", key="support-qa", dataset="golden", - handler=create_handler(("OpenAI", "messages"), openai_handler), + handlers=[create_handler(("OpenAI", "messages"), openai_handler)], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -2175,28 +2186,168 @@ async def openai_handler( @pytest.mark.asyncio -async def test_judge_handlers_must_declare_the_provider_they_serve( +async def test_an_untagged_handler_cannot_serve_a_judge_when_others_are_present( monkeypatch: pytest.MonkeyPatch, ) -> None: - """An unrouted judge handler would silently never be selected.""" + """With more than one handler, only a ``provides_for`` tag makes a handler + a candidate. An untagged handler passed alongside a tagged one is never + selected, even though it is present in ``handlers``. + """ transport = judge_run_transport() - accuracy_judge_variation(monkeypatch) + judge_variation(monkeypatch, provider="Anthropic") evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) - with pytest.raises(EvaluationsError, match="does not declare provides_for"): + with pytest.raises( + EvaluationsError, match="no handler registered for provider 'Anthropic'" + ): await evals.run( project_key="proj", key="support-qa", dataset="golden", - handler=_generation_only, + handlers=[ + create_handler(("OpenAI", "messages"), _generation_only), + _generation_only, + ], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], - judge_handlers=[_generation_only], ) assert transport.requests == [] +@pytest.mark.asyncio +async def test_provider_tag_is_matched_literally_not_as_a_wildcard( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """A ``("*", mode)`` tag is not treated specially: it is compared to a + provider name literally, the same as any other tag, so it never matches + a real provider by name. + """ + transport = judge_run_transport() + judge_variation(monkeypatch, provider="Anthropic") + evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + + async def star_handler(*args: object) -> dict[str, Any]: + return {"output": '{"score": 1, "reasoning": "ok"}'} + + with pytest.raises( + EvaluationsError, match="no handler registered for provider 'Anthropic'" + ): + await evals.run( + project_key="proj", + key="support-qa", + dataset="golden", + handlers=[ + create_handler(("OpenAI", "messages"), _generation_only), + create_handler(("*", "messages"), star_handler), + ], + generation={"provider": "OpenAI", "model": "gpt-4o"}, + criteria=[Judge(key="$ld:ai:judge:accuracy")], + ) + + assert transport.requests == [] + + +@pytest.mark.asyncio +async def test_duplicate_provides_for_tags_are_rejected_before_any_network_io() -> None: + """Two handlers tagged for the identical provider and mode could never be + told apart later, so this is checked eagerly, before the dataset or any + judge is even read.""" + transport = SequencedTransport([]) + evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + + async def first(*args: object) -> dict[str, Any]: + return {"output": "generated"} + + async def second(*args: object) -> dict[str, Any]: + return {"output": "generated"} + + with pytest.raises(EvaluationsError, match=r"'Anthropic'.*'messages'"): + await evals.run( + project_key="proj", + key="support-qa", + dataset="golden", + handlers=[ + create_handler(("Anthropic", "messages"), first), + create_handler(("Anthropic", "messages"), second), + ], + generation={"provider": "Anthropic", "model": "claude"}, + ) + + assert transport.requests == [] + + +@pytest.mark.asyncio +async def test_generation_provider_must_resolve_unambiguously_before_any_network_io() -> ( + None +): + """Generation has no mode, so it can never break a tie between two + handlers that both declare the same provider. This is checked as soon as + ``generation`` is known, before any request the run would otherwise make. + """ + transport = SequencedTransport([]) + evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + + async def agent_handler(*args: object) -> dict[str, Any]: + return {"output": "generated"} + + async def messages_handler(*args: object) -> dict[str, Any]: + return {"output": "generated"} + + with pytest.raises( + EvaluationsError, + match=r"2 handlers registered for provider 'Anthropic'", + ): + await evals.run( + project_key="proj", + key="support-qa", + dataset="golden", + handlers=[ + create_handler(("Anthropic", "agent"), agent_handler), + create_handler(("Anthropic", "messages"), messages_handler), + ], + generation={"provider": "Anthropic", "model": "claude"}, + ) + + assert transport.requests == [] + + +@pytest.mark.asyncio +async def test_generation_matches_the_handler_tagged_for_its_provider( + monkeypatch: pytest.MonkeyPatch, + stub_sdk_client: MagicMock, +) -> None: + """With more than one handler, generation is routed by provider name + alone -- it has no mode to further disambiguate with.""" + transport = judge_run_transport() + judge_variation(monkeypatch, provider="OpenAI") + evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + calls: list[str] = [] + + async def openai_handler(*args: object) -> dict[str, Any]: + calls.append("OpenAI") + return {"output": '{"score": 1, "reasoning": "ok"}'} + + async def anthropic_handler(*args: object) -> dict[str, Any]: + calls.append("Anthropic") + return {"output": '{"score": 1, "reasoning": "ok"}'} + + result = await evals.run( + project_key="proj", + key="support-qa", + dataset="golden", + handlers=[ + create_handler(("OpenAI", "messages"), openai_handler), + create_handler(("Anthropic", "messages"), anthropic_handler), + ], + generation={"provider": "OpenAI", "model": "gpt-4o"}, + criteria=[Judge(key="$ld:ai:judge:accuracy")], + ) + + assert result.passed is True + assert calls == ["OpenAI", "OpenAI"] + + @pytest.mark.asyncio async def test_criteria_run_concurrently_within_the_concurrency_bound( monkeypatch: pytest.MonkeyPatch, @@ -2253,7 +2404,7 @@ async def handler( project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], concurrency=2, @@ -2344,7 +2495,7 @@ async def handler( project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], tools={"lookup_order": lookup_order}, generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], @@ -2406,7 +2557,7 @@ async def handler( project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], tools={"lookup_order": lookup_order}, generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], @@ -2446,7 +2597,7 @@ async def handler( project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], tools={"lookup_order": lambda args: "unused"}, generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], @@ -2488,7 +2639,7 @@ async def handler( project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -2548,7 +2699,7 @@ async def handler( project_key="proj", key="support-qa", dataset="golden", - handler=handler, + handlers=[handler], tools={"lookup_order": lambda args: "{{expected_output}} leaked?"}, generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], @@ -2679,7 +2830,7 @@ async def handler(config: dict[str, Any], *args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], ai_config=AIConfig(key="support-agent", variation="control"), ) @@ -2716,7 +2867,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], ai_config=AIConfig(key="support-agent", variation="control"), generation={ "model": "gpt-4o-mini", @@ -2754,7 +2905,7 @@ async def test_variation_tools_without_implementations_fail_before_mutating_requ project_key="proj", key="eval-key", dataset="golden", - handler=successful_handler, + handlers=[successful_handler], ai_config=AIConfig(key="support-agent", variation="control"), ) @@ -2801,7 +2952,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], ai_config=AIConfig(key="support-agent", variation="control"), ) @@ -2820,7 +2971,7 @@ async def test_unknown_variation_fails_before_any_records_are_created() -> None: project_key="proj", key="eval-key", dataset="golden", - handler=successful_handler, + handlers=[successful_handler], ai_config=AIConfig(key="support-agent", variation="missing"), ) @@ -2853,7 +3004,7 @@ async def test_config_source_is_validated_before_network_io( project_key="proj", key="eval-key", dataset="golden", - handler=successful_handler, + handlers=[successful_handler], **source, # type: ignore[arg-type] ) @@ -2885,7 +3036,7 @@ async def handler(*args: object) -> dict[str, Any]: project_key="proj", key="eval-key", dataset="golden", - handler=handler, + handlers=[handler], ai_config=AIConfig(key="support-agent", variation="control"), tools={"lookup_order": lookup_order}, ) @@ -2912,7 +3063,7 @@ async def test_non_string_model_config_key_fails_loudly( project_key="proj", key="eval-key", dataset="golden", - handler=successful_handler, + handlers=[successful_handler], ai_config=AIConfig(key="support-agent", variation="control"), ) @@ -2935,7 +3086,7 @@ async def test_variation_without_a_model_config_needs_an_explicit_provider( project_key="proj", key="eval-key", dataset="golden", - handler=successful_handler, + handlers=[successful_handler], ai_config=AIConfig(key="support-agent", variation="control"), ) From 4cd1e0ee4a2b34d0a06174d31744185d13ac4899 Mon Sep 17 00:00:00 2001 From: Dylan O'Neill Date: Tue, 29 Sep 2026 20:58:15 -0700 Subject: [PATCH 2/2] fix(evaluations): restore the wildcard-provider handler fallback in _select_handler _select_handler compared provides_for's provider literally, so a wildcard ("*", mode) handler was never selected. config() and graph() keep the wildcard as a fallback for a multi-provider adapter, so evals now matches: exact-provider handlers are tried first, and a wildcard handler is only a candidate when none names the call's provider outright. Restores the two tests this dropped, adapted to the current single handlers list, and updates packages/client/README.md's judge-routing paragraph to describe the fallback again. Co-Authored-By: Claude Sonnet 5 --- packages/client/README.md | 2 +- .../evaluations/runner.py | 17 +++- packages/client/tests/test_evaluations_run.py | 95 +++++++++++++++---- 3 files changed, 88 insertions(+), 26 deletions(-) diff --git a/packages/client/README.md b/packages/client/README.md index 60aa58c0..a4b0c14d 100644 --- a/packages/client/README.md +++ b/packages/client/README.md @@ -161,7 +161,7 @@ Two limits keep a trajectory from spending the judge's context window: at most 5 A tool result is now judge-prompt input. It stays literal for the same reason the generated output does: the judge config is handed to the handler unrendered and the handler makes exactly one template pass, so a `{{...}}` sequence coming back from a tool is never expanded into the judge prompt. -**Judges are independent AI Configs, so handlers are routed per judge.** A judge may resolve to a different provider or mode than `generation`, and a handler built for one provider cannot execute another's config. Pass every handler your run needs in `handlers`. With one handler, it runs generation and every judge. With more than one, each is tagged with `provides_for` (via `create_handler()` or a provider package's `create_*_handler()`), and `run()` picks the entry tagged for the provider a call needs — for a judge, its exact `(provider, mode)` pair when that disambiguates, otherwise the sole handler left for that provider. An agent-mode handler can still serve a messages-mode judge for the same provider, with its messages collapsed into one instructions block. `run()` raises before any record is created if a call's provider matches no handler, or matches more than one with no way to disambiguate. +**Judges are independent AI Configs, so handlers are routed per judge.** A judge may resolve to a different provider or mode than `generation`, and a handler built for one provider cannot execute another's config. Pass every handler your run needs in `handlers`. With one handler, it runs generation and every judge. With more than one, each is tagged with `provides_for` (via `create_handler()` or a provider package's `create_*_handler()`), and `run()` picks the entry tagged for the provider a call needs, falling back to a wildcard `("*", mode)` entry when no handler names that provider exactly — the same fallback `config()` and `graph()` give a multi-provider adapter such as LangChain. For a judge, an exact `(provider, mode)` match wins when that disambiguates, otherwise the sole handler left for that provider. An agent-mode handler can still serve a messages-mode judge for the same provider, with its messages collapsed into one instructions block. `run()` raises before any record is created if a call's provider matches no handler, or matches more than one with no way to disambiguate. Judges are resolved through flag delivery, and handlers are matched to them, **before** any evaluation records are created — a missing judge or one no handler covers fails the run up front rather than after the generation spend. After that point a criterion failure never aborts the run: an unparseable judge response, an out-of-range score, a raising handler or scorer, and a row whose generation errored each become a per-criterion `ERROR` event with a cause code (`invalid_judge_output`, `invalid_score`, `handler_raised`, `scorer_raised`, `generation_incomplete`) and a top-level `errorMessage`. Event *delivery* is different: the backend needs one result per `(row, criterion)` to finish row accounting, so if tracking a criterion event fails, every remaining result is still attempted and flushed and then `run()` raises — rather than polling to its timeout with the cause hidden. diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/runner.py b/packages/client/src/launchdarkly_ai_server/evaluations/runner.py index 1063f34e..a4cef0bd 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/runner.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/runner.py @@ -93,10 +93,12 @@ def _select_handler( """Pick the handler in ``handlers`` that serves ``provider``. A single handler always runs, whatever it declares. With more than one - handler, only entries whose ``provides_for`` names ``provider`` are - candidates. ``mode`` is a judge's resolved mode (generation has none); an - exact ``(provider, mode)`` match wins over any other candidate for that - provider. + handler, an entry whose ``provides_for`` names ``provider`` exactly is a + candidate. A wildcard entry tagged ``("*", mode)`` is a candidate only + when no entry names ``provider`` exactly; this mirrors the fallback + ``config()`` and ``graph()`` give a multi-provider adapter. ``mode`` is a + judge's resolved mode (generation has none); an exact ``(provider, + mode)`` match wins over any other candidate for that provider. Raises :class:`EvaluationsError` when no candidate serves ``provider``, or when more than one does and ``mode`` cannot break the tie. @@ -104,11 +106,16 @@ def _select_handler( if len(handlers) == 1: return handlers[0] - candidates = [ + exact_provider = [ handler for handler in handlers if (tag := _provides_for(handler)) is not None and tag[0] == provider ] + candidates = exact_provider or [ + handler + for handler in handlers + if (tag := _provides_for(handler)) is not None and tag[0] == "*" + ] if mode is not None: exact = [ handler diff --git a/packages/client/tests/test_evaluations_run.py b/packages/client/tests/test_evaluations_run.py index 682661c6..9c6601ae 100644 --- a/packages/client/tests/test_evaluations_run.py +++ b/packages/client/tests/test_evaluations_run.py @@ -2216,36 +2216,91 @@ async def test_an_untagged_handler_cannot_serve_a_judge_when_others_are_present( @pytest.mark.asyncio -async def test_provider_tag_is_matched_literally_not_as_a_wildcard( +@pytest.mark.parametrize("wildcard_first", [True, False]) +async def test_exact_provider_judge_handler_beats_a_wildcard_adapter( monkeypatch: pytest.MonkeyPatch, + stub_sdk_client: MagicMock, + wildcard_first: bool, ) -> None: - """A ``("*", mode)`` tag is not treated specially: it is compared to a - provider name literally, the same as any other tag, so it never matches - a real provider by name. + """A wildcard is a fallback, so the order handlers are listed in cannot decide. + + Taking the first provider-or-wildcard match would send an Anthropic judge + through a multi-provider adapter that merely happened to be listed first. """ transport = judge_run_transport() judge_variation(monkeypatch, provider="Anthropic") evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + chosen: list[str] = [] + + def judge_handler(name: str) -> Any: + async def run( + config: dict[str, Any], + user_input: str | None = None, + tool_handlers: dict[str, Callable[..., Any]] | None = None, + variables: dict[str, Any] | None = None, + history: list[dict[str, Any]] | None = None, + ) -> dict[str, Any]: + chosen.append(name) + return {"output": '{"score": 1, "reasoning": "ok"}'} + + return run + + wildcard = create_handler(("*", "messages"), judge_handler("wildcard")) + exact = create_handler(("Anthropic", "messages"), judge_handler("exact")) + generation = create_handler(("OpenAI", "messages"), _generation_only) + + result = await evals.run( + project_key="proj", + key="support-qa", + dataset="golden", + handlers=[wildcard, exact, generation] + if wildcard_first + else [exact, wildcard, generation], + generation={"provider": "OpenAI", "model": "gpt-4o"}, + criteria=[Judge(key="$ld:ai:judge:accuracy")], + ) + + assert result.passed is True + assert chosen == ["exact"] - async def star_handler(*args: object) -> dict[str, Any]: + +@pytest.mark.asyncio +async def test_wildcard_judge_handler_runs_a_judge_no_handler_names( + monkeypatch: pytest.MonkeyPatch, + stub_sdk_client: MagicMock, +) -> None: + """A wildcard-tagged handler is a fallback: it is selected for a judge + when no handler in ``handlers`` names that judge's provider outright. + """ + transport = judge_run_transport() + judge_variation(monkeypatch, provider="Anthropic") + evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + judged: list[dict[str, Any]] = [] + + async def wildcard_judge( + config: dict[str, Any], + user_input: str | None = None, + tool_handlers: dict[str, Callable[..., Any]] | None = None, + variables: dict[str, Any] | None = None, + history: list[dict[str, Any]] | None = None, + ) -> dict[str, Any]: + judged.append(config) return {"output": '{"score": 1, "reasoning": "ok"}'} - with pytest.raises( - EvaluationsError, match="no handler registered for provider 'Anthropic'" - ): - await evals.run( - project_key="proj", - key="support-qa", - dataset="golden", - handlers=[ - create_handler(("OpenAI", "messages"), _generation_only), - create_handler(("*", "messages"), star_handler), - ], - generation={"provider": "OpenAI", "model": "gpt-4o"}, - criteria=[Judge(key="$ld:ai:judge:accuracy")], - ) + result = await evals.run( + project_key="proj", + key="support-qa", + dataset="golden", + handlers=[ + create_handler(("OpenAI", "messages"), _generation_only), + create_handler(("*", "messages"), wildcard_judge), + ], + generation={"provider": "OpenAI", "model": "gpt-4o"}, + criteria=[Judge(key="$ld:ai:judge:accuracy")], + ) - assert transport.requests == [] + assert result.passed is True + assert [config["provider"]["name"] for config in judged] == ["Anthropic"] @pytest.mark.asyncio