From 26eafe03ee29d80530434d455046ca5dd9dd23fb Mon Sep 17 00:00:00 2001 From: Dylan O'Neill Date: Fri, 9 Oct 2026 09:58:30 -0700 Subject: [PATCH] feat(evaluations)!: route handlers by provider and mode MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Implement the handler routing rules of TESTING.md §8 in launchdarkly/ai-sdks-monorepo#52. - run() takes `handler` or `handlers`, not both. `judge_handlers` is removed. Every handler must declare provides_for, and two handlers cannot declare the same provider and mode. - Generation and every judge select a handler with the config() rule (utils.select_handler). There is no mode fallback, so a messages-mode judge never runs on an agent handler. - generation.mode is required. The values are "completion" and "agent". GenerationConfig stays a TypedDict with Required fields. GenerationOverrides is the type for overrides with ai_config. - The prompt must match the handler mode. A messages handler needs messages and an agent handler needs instructions. The harness does not convert one into the other. - Judges get no tools. - ai_config reads the AI Config mode first, then the variation, then the model config. Each check runs right after the read that supplies its input. The mode selects the prompt field of the variation. Co-Authored-By: Claude Opus 5.5 --- packages/ai/README.md | 6 +- packages/client/README.md | 30 +- packages/client/agents.md | 2 +- .../evaluations/__init__.py | 4 + .../evaluations/module.py | 296 ++++-- .../evaluations/runner.py | 239 ++--- .../evaluations/types.py | 56 +- packages/client/tests/test_evaluations_run.py | 888 ++++++++++++++---- 8 files changed, 1125 insertions(+), 396 deletions(-) diff --git a/packages/ai/README.md b/packages/ai/README.md index 5147a215..b1b9ecad 100644 --- a/packages/ai/README.md +++ b/packages/ai/README.md @@ -63,8 +63,8 @@ evals = init_evaluations(project_key="my-project") result = await evals.run( key="unique-evaluation-key", dataset="golden-dataset", - handler=my_handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=my_handler, # built with create_handler(("OpenAI", "messages"), fn) + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[ Judge(key="accuracy-judge"), Scorer(name="mentions-policy", fn=lambda row, output: "policy" in (output or "")), @@ -72,7 +72,7 @@ result = await evals.run( ) ``` -`LD_API_TOKEN` is required. Configure `LD_SDK_KEY` — or initialize your own client with `init_client(client=...)` — to emit one `$ld:ai:offline-evals:generation` event per generated row, plus one `$ld:ai:offline-evals:criterion` event per `(row, criterion)` when `criteria` are supplied, through the standard SDK event transport. The SDK reports scores; LaunchDarkly rules on them at ingest. A judge served by a different provider than `generation` needs a handler for it in `judge_handlers`. Each row's tool calls are recorded during generation and rendered into the judge's `{{message_history}}`, between the row input and the generated output, so a rubric can grade the tool trajectory as well as the final answer. Use `LD_API_BASE_URI` for staging or local management API traffic; it is separate from the SDK delivery setting `LD_BASE_URI`. Evaluation-run links use the explicit `ui_base_uri` option or `LD_UI_BASE_URI`, defaulting to `https://app.launchdarkly.com`; set it when the project is not in production, or a run created elsewhere still links to the production app. `tools` is a list of `EvalTool`. Construct one to define a tool in code, or await `evals.tools.get()` for a tool that already exists in LaunchDarkly. See the [core evaluations guide](https://github.com/launchdarkly/python-ai-sdk/blob/main/packages/client/README.md#run-an-evaluation-from-code). +`LD_API_TOKEN` is required. Configure `LD_SDK_KEY` — or initialize your own client with `init_client(client=...)` — to emit one `$ld:ai:offline-evals:generation` event per generated row, plus one `$ld:ai:offline-evals:criterion` event per `(row, criterion)` when `criteria` are supplied, through the standard SDK event transport. The SDK reports scores; LaunchDarkly rules on them at ingest. Pass one handler as `handler` or a list as `handlers`. The generation config and each judge select a handler by provider and mode, so a judge served by a different provider than `generation` needs its own handler in `handlers`. `generation.mode` is required: `"completion"` for a messages handler and a `messages` prompt, `"agent"` for an agent handler and an `instructions` prompt. Each row's tool calls are recorded during generation and rendered into the judge's `{{message_history}}`, between the row input and the generated output, so a rubric can grade the tool trajectory as well as the final answer. Use `LD_API_BASE_URI` for staging or local management API traffic; it is separate from the SDK delivery setting `LD_BASE_URI`. Evaluation-run links use the explicit `ui_base_uri` option or `LD_UI_BASE_URI`, defaulting to `https://app.launchdarkly.com`; set it when the project is not in production, or a run created elsewhere still links to the production app. `tools` is a list of `EvalTool`. Construct one to define a tool in code, or await `evals.tools.get()` for a tool that already exists in LaunchDarkly. See the [core evaluations guide](https://github.com/launchdarkly/python-ai-sdk/blob/main/packages/client/README.md#run-an-evaluation-from-code). --- diff --git a/packages/client/README.md b/packages/client/README.md index 23ecdd06..03a1aabd 100644 --- a/packages/client/README.md +++ b/packages/client/README.md @@ -67,7 +67,8 @@ async def main() -> int: generation={ "provider": "OpenAI", "model": "gpt-4o", - "instructions": "You are a support agent.", + "mode": "completion", + "messages": [{"role": "system", "content": "You are a support agent."}], }, ) print(result.url, result.summary) @@ -77,7 +78,15 @@ async def main() -> int: sys.exit(asyncio.run(main())) ``` -`project_key` is supplied during initialization, either as an argument to `init_evaluations()` or through `LD_PROJECT_KEY`. `generation.instructions` is shorthand for one system message; use `generation.messages` instead for a full message list, but do not supply both. The harness never retries a handler invocation because doing so could repeat tool side effects. Its retries apply only to LaunchDarkly management API requests. +`project_key` is supplied during initialization, either as an argument to `init_evaluations()` or through `LD_PROJECT_KEY`. `generation` requires `provider`, `model`, and `mode`. `generation` is a `TypedDict`, so you pass a plain `dict` and import no type. + +**Choose the mode.** Use `"completion"` when the prompt is a list of `messages` and the handler sends one request to a chat or messages API, for example `create_openai_messages_handler()`. Use `"agent"` when the prompt is one `instructions` string and an agent framework runs the model, for example OpenAI Agents, Claude Agents, or LangChain agents. If you are not sure, use the mode of the AI Config in LaunchDarkly. The mode has no default. + +**The prompt must match the handler.** A messages handler needs `messages`, and an agent handler needs `instructions`. The harness does not convert one into the other: a mismatch raises before any network I/O, and the error names the field the handler needs. + +**Pass one handler as `handler` or a list as `handlers`, not both.** Each handler must declare `provides_for`; the provider packages' `create_*_handler()` factories and `create_handler()` set it. The generation config and every judge select a handler from the list by provider and mode, with the same rule as `config()`: a handler that names the provider wins over a `"*"` handler, and the mode must match exactly. To evaluate your own application code, wrap it with `create_handler((provider, mode), fn)`. + +**Start from an AI Config.** Pass `ai_config=AIConfig(key=..., variation=...)` instead of a full `generation`. The harness reads the AI Config's mode first, then the variation, then the model config it links, and runs each check as soon as the read that supplies its input returns. The mode selects the prompt field: `instructions` in agent mode, `messages` in completion and judge mode. `generation` can then override single fields, except `mode`. The harness never retries a handler invocation because doing so could repeat tool side effects. Its retries apply only to LaunchDarkly management API requests. Generation and criterion events are the only path by which row results reach LaunchDarkly, so `init_evaluations()` raises rather than creating a run that can never complete unless it can resolve an event transport: either an SDK key (`sdk_key` or `LD_SDK_KEY`) or a client already initialized through `init_client(client=...)`. Bringing your own client lets a process emit evaluation events without an SDK key in scope. Every generated row is emitted and flushed unconditionally; no feature flag gates event publishing. The harness then polls the summary endpoint until row accounting shows processing is complete. @@ -94,7 +103,7 @@ result = await evals.run( {"input": "Where is order {{order_id}}?", "variables": {"order_id": "A-17"}}, ], handler=create_openai_messages_handler(), - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) ``` @@ -117,15 +126,14 @@ def mentions_policy(row: DatasetRow, output: str | None) -> bool: result = await init_evaluations(project_key="my-project").run( key="support-qa-2026-08-20", dataset="support-golden", - handler=create_openai_messages_handler(), - generation={"provider": "OpenAI", "model": "gpt-4o"}, + # The generation config uses OpenAI and the judge uses Anthropic, so the + # list holds a handler for each. + handlers=[create_openai_messages_handler(), create_claude_messages_handler()], + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[ Judge(key="accuracy-judge", threshold=0.8), Scorer(name="mentions-policy", fn=mentions_policy), ], - # Needed only because this judge is served by a different provider than - # the generation config above. - judge_handlers=[create_claude_messages_handler()], ) ``` @@ -178,9 +186,9 @@ Two limits keep a trajectory from spending the judge's context window: at most 5 A tool result is now judge-prompt input. It stays literal for the same reason the generated output does: the judge config is handed to the handler unrendered and the handler makes exactly one template pass, so a `{{...}}` sequence coming back from a tool is never expanded into the judge prompt. -**Judges are independent AI Configs, so handlers are routed per judge.** A judge may resolve to a different provider or mode than `generation`, and a handler built for one provider cannot execute another's config. `handler` runs a judge when it provides for that judge's provider; pass handlers for any other providers in `judge_handlers`. Selection prefers a handler naming the judge's provider outright over a wildcard multi-provider adapter, and an agent-mode handler can serve a messages-mode judge with its messages collapsed into one instructions block. A plain callable that declares no `provides_for` routes itself, exactly as it already does for the generation config. +**Judges are independent AI Configs, so handlers are routed per judge.** A judge may resolve to a different provider or mode than `generation`. Each judge selects its handler from `handler` or `handlers` by its provider and mode, with the same rule as generation. There is no mode fallback: a messages-mode judge never runs on an agent handler. A judge's prompt must also match its handler's mode. Judges get no tools: the harness passes an empty tool map and removes `tools` from the judge config. -Judges are resolved through flag delivery, and handlers are matched to them, **before** any evaluation records are created — a missing judge or one no handler covers fails the run up front rather than after the generation spend. After that point a criterion failure never aborts the run: an unparseable judge response, an out-of-range score, a raising handler or scorer, and a row whose generation errored each become a per-criterion `ERROR` event with a cause code (`invalid_judge_output`, `invalid_score`, `handler_raised`, `scorer_raised`, `generation_incomplete`) and a top-level `errorMessage`. Event *delivery* is different: the backend needs one result per `(row, criterion)` to finish row accounting, so if tracking a criterion event fails, every remaining result is still attempted and flushed and then `run()` raises — rather than polling to its timeout with the cause hidden. +Judges are resolved through flag delivery, and handlers are matched to them, **before** any evaluation records are created — a missing judge, a judge that no handler covers, or a judge whose prompt does not match its handler fails the run up front rather than after the generation spend. One error lists every judge with a problem. After that point a criterion failure never aborts the run: an unparseable judge response, an out-of-range score, a raising handler or scorer, and a row whose generation errored each become a per-criterion `ERROR` event with a cause code (`invalid_judge_output`, `invalid_score`, `handler_raised`, `scorer_raised`, `generation_incomplete`) and a top-level `errorMessage`. Event *delivery* is different: the backend needs one result per `(row, criterion)` to finish row accounting, so if tracking a criterion event fails, every remaining result is still attempted and flushed and then `run()` raises — rather than polling to its timeout with the cause hidden. The client uses **lazy initialization**: importing the package does not connect to LaunchDarkly. The singleton is created automatically on the first API call that needs it (`config().invoke()`, `graph().invoke()`, `resolve_graph()`, etc.), as long as `LD_SDK_KEY` is set in the environment. @@ -250,7 +258,7 @@ result = await evals.run( key="support-qa-2026-08-20", dataset="support-golden", handler=create_openai_messages_handler(), - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, tools=[search_docs_tool, lookup_order_tool], ) ``` diff --git a/packages/client/agents.md b/packages/client/agents.md index a134b278..081fe2ea 100644 --- a/packages/client/agents.md +++ b/packages/client/agents.md @@ -131,7 +131,7 @@ Handlers may return any of these — the client normalizes them before emitting `init_evaluations()` creates an evaluations harness using `LD_API_TOKEN` and the management API host `LD_API_BASE_URI`. Do not reuse `LD_BASE_URI`: that variable configures SDK delivery and may point at a relay proxy. Evaluation-run links use the separate `ui_base_uri` option, then `LD_UI_BASE_URI`, then `https://app.launchdarkly.com`; do not derive their host from `LD_API_BASE_URI`. An event transport is resolved in `init_evaluations()`, which raises before any network I/O when it finds neither an SDK key (`sdk_key` or `LD_SDK_KEY`) nor an already-initialized event-capable client: generation events are the only ingest path for row results, so a run without a transport could never complete. The lifecycle module's bring-your-own-client path (`init_client(client=...)`) therefore satisfies the check on its own, and `run()` reuses that singleton through `_resolve_client`; `run()` raises if the client disappears before it emits. Both polling arguments reject NaN, which would otherwise never compare past a deadline and hang the run. The harness always queues one `$ld:ai:offline-evals:generation` custom event per row through the standard SDK event transport and flushes before returning. No feature flag gates event emission. The harness polls the run summary endpoint until a nonzero `total_rows` has `pending_rows == 0` and `passed + failed + error` rows accounting for the total, polling every `poll_interval_seconds` (default 2s) until `poll_timeout_seconds` (default 180s); both are `run()` arguments so large datasets can widen them. The summary endpoint does not return run state, so `RunSummary` exposes row counts only. -`init_evaluations()` takes `project_key`; `run()` does not. Dataset lookup/row pagination, evaluation creation, and run creation are private helpers; `run()` and `tools.get()` are the public surface. `tools` is a `list[EvalTool]`. `await evals.tools.get(key, implementation=...)` issues `GET projects/

/ai-tools/`, pins the returned version, and builds a library tool through `EvalTool._library`. `get` is a coroutine that reads in a worker thread. `EvalTool` is frozen with `schema` required, so a constructed tool is always inline. `_validate_tools` also rejects a library tool that has no `project_key` or one from another project. `run()` issues no tool request at all: a library tool was already read by `tools.get()`. Library tools are sent as `{key, version, source: "library"}` and inline tools as `{key, schema, description, source: "inline"}`. `_validate_tools` runs in `run()` before any I/O, so a bad list — a non-`EvalTool` entry, a blank or uppercase key, a non-object or non-serializable `schema` (`allow_nan=False`), a non-callable implementation, a repeated key (compared without case), or a `NativeTool` on an inline tool — fails with zero requests recorded. `_validate_tool_key` is shared by `tools.get()` and `_validate_tools`, so the key rules are identical on both paths. Handler config synthesis is unforked: `config["tools"][tool.key] = {"description", "parameters"}` is fed from whichever kind the tool is, so handlers cannot tell them apart, and `_tool_handlers` maps each tool to its executable. Each call creates a new evaluation with `POST` and a run with `source="api"`, so its key must be unique. The harness directly invokes the supplied handler once per row and never retries it — event delivery is never a reason to rerun a handler because that would repeat tool side effects; retries apply only to management API requests. A 429 is replayed for any method, but 5xx responses and transport failures are replayed only for `GET`/`HEAD`, so an evaluation or run `POST` that the server may already have applied is never duplicated. Management API calls run in a worker thread (`asyncio.to_thread`) because the client is synchronous; the caller's event loop stays free. Generation events go through the already-initialized SDK client when the application has one — `init_client` is idempotent, so an existing singleton wins and the evaluations SDK key is ignored with a warning. Dataset-owned `input`, `expected_output`, `metadata`, and `variables` are deliberately excluded from the event payload. The harness flushes events, polls the run summary endpoint until row accounting is complete (`total_rows > 0`, `pending_rows == 0`, and `passed + failed + error == total_rows`), and raises a timeout once `poll_timeout_seconds` elapses if the backend never reaches one. `RunSummary` includes row counts only, and `EvalRunResult.passed` is true only when error and pending row counts are both zero. +`init_evaluations()` takes `project_key`; `run()` does not. Dataset lookup/row pagination, evaluation creation, and run creation are private helpers; `run()` and `tools.get()` are the public surface. `tools` is a `list[EvalTool]`. `await evals.tools.get(key, implementation=...)` issues `GET projects/

/ai-tools/`, pins the returned version, and builds a library tool through `EvalTool._library`. `get` is a coroutine that reads in a worker thread. `EvalTool` is frozen with `schema` required, so a constructed tool is always inline. `_validate_tools` also rejects a library tool that has no `project_key` or one from another project. `run()` issues no tool request at all: a library tool was already read by `tools.get()`. Library tools are sent as `{key, version, source: "library"}` and inline tools as `{key, schema, description, source: "inline"}`. `_validate_tools` runs in `run()` before any I/O, so a bad list — a non-`EvalTool` entry, a blank or uppercase key, a non-object or non-serializable `schema` (`allow_nan=False`), a non-callable implementation, a repeated key (compared without case), or a `NativeTool` on an inline tool — fails with zero requests recorded. `_validate_tool_key` is shared by `tools.get()` and `_validate_tools`, so the key rules are identical on both paths. Handler config synthesis is unforked: `config["tools"][tool.key] = {"description", "parameters"}` is fed from whichever kind the tool is, so handlers cannot tell them apart, and `_tool_handlers` maps each tool to its executable. Each call creates a new evaluation with `POST` and a run with `source="api"`, so its key must be unique. `run()` takes `handler` or `handlers`, never both, and every handler must declare `provides_for`. The generation config and each judge select a handler with `utils.select_handler`, the `config()` rule, with no mode fallback. The prompt must match the handler mode (`prompt_mode_error`), and judges get no tools. With `ai_config`, `_read_ai_config` reads the AI Config mode, the variation, and the model config in that order, and runs each check right after the read that supplies its input. The harness directly invokes the selected handler once per row and never retries it — event delivery is never a reason to rerun a handler because that would repeat tool side effects; retries apply only to management API requests. A 429 is replayed for any method, but 5xx responses and transport failures are replayed only for `GET`/`HEAD`, so an evaluation or run `POST` that the server may already have applied is never duplicated. Management API calls run in a worker thread (`asyncio.to_thread`) because the client is synchronous; the caller's event loop stays free. Generation events go through the already-initialized SDK client when the application has one — `init_client` is idempotent, so an existing singleton wins and the evaluations SDK key is ignored with a warning. Dataset-owned `input`, `expected_output`, `metadata`, and `variables` are deliberately excluded from the event payload. The harness flushes events, polls the run summary endpoint until row accounting is complete (`total_rows > 0`, `pending_rows == 0`, and `passed + failed + error == total_rows`), and raises a timeout once `poll_timeout_seconds` elapses if the backend never reaches one. `RunSummary` includes row counts only, and `EvalRunResult.passed` is true only when error and pending row counts are both zero. --- diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/__init__.py b/packages/client/src/launchdarkly_ai_server/evaluations/__init__.py index 5ba1535f..40d01469 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/__init__.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/__init__.py @@ -17,6 +17,8 @@ DatasetRow, EvalRunResult, GenerationConfig, + GenerationMode, + GenerationOverrides, RunSummary, Usage, ) @@ -31,6 +33,8 @@ "EvaluationsError", "EvaluationsModule", "GenerationConfig", + "GenerationMode", + "GenerationOverrides", "HttpResponse", "Judge", "LDApiClient", diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/module.py b/packages/client/src/launchdarkly_ai_server/evaluations/module.py index 2348690f..eb305ec7 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/module.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/module.py @@ -8,9 +8,10 @@ import os import time from collections.abc import Mapping, Sequence -from typing import Any, cast +from typing import Any, Literal, cast, overload from ..lifecycle import get_client, init_client +from ..utils import normalize_mode from .api import ( DEFAULT_BASE_URI, EvaluationsError, @@ -24,15 +25,20 @@ EvalHandler, EvaluationsRunner, _provides_for, + describe_handlers, + prompt_mode_error, render_row, + select_eval_handler, ) from .tools import EvalTool, ToolsClient, tool_handlers, validate_tools from .types import ( AIConfig, + AIConfigVariation, DatasetRef, DatasetRow, EvalRunResult, GenerationConfig, + GenerationOverrides, InlineDatasetRow, RunSummary, ) @@ -43,6 +49,7 @@ SUMMARY_POLL_INTERVAL_SECONDS = 2.0 SUMMARY_POLL_TIMEOUT_SECONDS = 180.0 INLINE_ROW_FIELDS = ("rowIdx", "input", "expectedOutput", "variables", "metadata") +GENERATION_MODES = ("completion", "agent") def _render_inline_row(row: DatasetRow) -> DatasetRow: @@ -168,9 +175,14 @@ def _is_terminal_summary(summary: RunSummary) -> bool: ) +def _handler_mode(mode: str) -> Literal["agent", "messages"]: + """Return the handler mode for a validated generation mode.""" + return "agent" if mode == "agent" else "messages" + + def _merge_generation( - base: GenerationConfig, override: GenerationConfig | None -) -> GenerationConfig: + base: GenerationOverrides, override: GenerationOverrides | None +) -> dict[str, Any]: """Layer a caller's generation settings over a fetched variation's. Keys the caller sets replace the fetched ones, except ``parameters``, which @@ -181,7 +193,7 @@ def _merge_generation( """ merged: dict[str, Any] = dict(base) if not override: - return cast(GenerationConfig, merged) + return merged if "instructions" in override or "messages" in override: merged.pop("instructions", None) merged.pop("messages", None) @@ -194,7 +206,7 @@ def _merge_generation( } else: merged[field_name] = value - return cast(GenerationConfig, merged) + return merged class EvaluationsModule: @@ -238,17 +250,51 @@ def ui_base_uri(self) -> str: """LaunchDarkly application host used for evaluation-run links.""" return self._ui_base_uri + @overload + async def run( + self, + *, + key: str, + dataset: str | Sequence[InlineDatasetRow], + generation: GenerationConfig, + handler: EvalHandler | None = None, + handlers: Sequence[EvalHandler] | None = None, + ai_config: None = None, + tools: Sequence[EvalTool] | None = None, + criteria: list[Criterion] | None = None, + concurrency: int = 10, + poll_interval_seconds: float | None = None, + poll_timeout_seconds: float | None = None, + ) -> EvalRunResult: ... + + @overload + async def run( + self, + *, + key: str, + dataset: str | Sequence[InlineDatasetRow], + ai_config: AIConfig, + generation: GenerationOverrides | None = None, + handler: EvalHandler | None = None, + handlers: Sequence[EvalHandler] | None = None, + tools: Sequence[EvalTool] | None = None, + criteria: list[Criterion] | None = None, + concurrency: int = 10, + poll_interval_seconds: float | None = None, + poll_timeout_seconds: float | None = None, + ) -> EvalRunResult: ... + async def run( self, *, key: str, dataset: str | Sequence[InlineDatasetRow], - handler: EvalHandler, - generation: GenerationConfig | None = None, + handler: EvalHandler | None = None, + handlers: Sequence[EvalHandler] | None = None, + generation: GenerationConfig | GenerationOverrides | None = None, ai_config: AIConfig | None = None, tools: Sequence[EvalTool] | None = None, criteria: list[Criterion] | None = None, - judge_handlers: list[EvalHandler] | None = None, concurrency: int = 10, poll_interval_seconds: float | None = None, poll_timeout_seconds: float | None = None, @@ -256,12 +302,21 @@ async def run( """ Create and run an evaluation in the caller's process. - Each dataset row is generated with ``handler``; every entry in - ``criteria`` — LaunchDarkly :class:`Judge` references and local - deterministic :class:`Scorer` functions — is then run against each + Each dataset row is generated by a handler; every entry in + ``criteria`` (LaunchDarkly :class:`Judge` references and local + deterministic :class:`Scorer` functions) is then run against each generated row, and one evaluation event is emitted per ``(row, criterion)`` result. + Pass one handler as ``handler`` or a list as ``handlers``, not both. + Each handler must declare ``provides_for``: build it with + ``create_handler()`` or a provider package's ``create_*_handler()``. + The generation config and each judge config select their handler by + provider and mode, so one list can hold an OpenAI handler for + generation and an Anthropic handler for a judge. The mode must match + exactly. A messages handler needs a ``messages`` prompt and an agent + handler needs an ``instructions`` prompt. Judges get no tools. + ``tools`` is a list of :class:`EvalTool`. Construct one to define a tool in code. Call ``evals.tools.get(key, implementation=...)`` to use a tool from the LaunchDarkly tool library, which reads the tool and pins @@ -277,26 +332,22 @@ async def run( uploaded to the run before any generation starts, and templates in them render exactly as a stored dataset's do. - A :class:`Judge` is an independent AI Config and may be served by a - different provider or mode than ``generation``. ``handler`` runs a judge - only when it provides for that judge's provider; pass handlers for any - other providers your judges use in ``judge_handlers``. A judge no - handler covers fails the run before any records are created. - The returned pass/fail result is derived from LaunchDarkly's run summary. A CI script can exit with ``0 if result.passed else 1`` after awaiting this method. Large datasets may need a longer ``poll_timeout_seconds`` and a wider ``poll_interval_seconds``; both default to ``SUMMARY_POLL_TIMEOUT_SECONDS`` / ``SUMMARY_POLL_INTERVAL_SECONDS``. - Pass ``ai_config`` to start from an existing AI Config - variation instead of a hand-built ``generation``. Its model, provider, - parameters, prompt and output format become the defaults, and anything - set in ``generation`` overrides them field by field (``parameters`` - merge key by key). When ``tools`` is omitted the variation's tools are - used, so each needs an implementation; pass ``tools`` to replace the - set. When ``criteria`` is omitted the variation's attached judges run; - pass ``criteria`` (even ``[]``) to replace them. + ``generation`` must set ``provider``, ``model`` and ``mode`` + (``"completion"`` or ``"agent"``). Pass ``ai_config`` instead to start + from an existing AI Config variation. The AI Config supplies the mode. + The variation's model, provider, parameters, prompt and output format + become the defaults, and ``generation`` can override them field by + field (``parameters`` merge key by key), except ``mode``. When + ``tools`` is omitted the variation's tools are used, so each needs an + implementation; pass ``tools`` to replace the set. When ``criteria`` is + omitted the variation's attached judges run; pass ``criteria`` (even + ``[]``) to replace them. """ if poll_interval_seconds is None: poll_interval_seconds = SUMMARY_POLL_INTERVAL_SECONDS @@ -304,11 +355,11 @@ async def run( poll_timeout_seconds = SUMMARY_POLL_TIMEOUT_SECONDS self._validate_run_args( key=key, - handler=handler, concurrency=concurrency, poll_interval_seconds=poll_interval_seconds, poll_timeout_seconds=poll_timeout_seconds, ) + pool = self._validate_handlers(handler, handlers) inline_rows = self._validate_dataset_source(dataset) self._validate_config_source(generation=generation, ai_config=ai_config) run_tools = list(tools or []) @@ -316,15 +367,22 @@ async def run( run_tool_handlers = tool_handlers(run_tools) pinned_tool_versions: dict[str, int] = {} config_label = "" - if ai_config is not None: + if ai_config is None: + run_generation = self._validate_generation( + cast(dict[str, Any], dict(generation or {})) + ) + generation_handler = self._select_generation_handler( + pool, run_generation, "generation" + ) + else: config_label = f"{ai_config.key!r}/{ai_config.variation!r}" - ai_config_variation = await asyncio.to_thread( - self._runner._fetch_config_variation, - self._project_key, - ai_config.key, - ai_config.variation, + ( + run_generation, + generation_handler, + ai_config_variation, + ) = await self._read_ai_config( + ai_config, cast(GenerationOverrides | None, generation), pool ) - generation = _merge_generation(ai_config_variation.generation, generation) if tools is None and ai_config_variation.tool_versions: raise EvaluationsError( f"AI Config variation {config_label} uses tools " @@ -350,11 +408,8 @@ async def run( criteria = [ Judge(key=judge_key) for judge_key in ai_config_variation.judge_keys ] - generation = self._validate_generation(generation) run_criteria = list(criteria or []) - run_judge_handlers = list(judge_handlers or []) self._validate_criteria(run_criteria) - self._validate_judge_handlers(run_judge_handlers) ld_judges = [ criterion for criterion in run_criteria if isinstance(criterion, Judge) ] @@ -379,7 +434,7 @@ async def run( tool.version, ) resolved_judges = await self._runner._resolve_judges( - self._project_key, ld_judges, handler, run_judge_handlers + self._project_key, ld_judges, pool ) if isinstance(dataset, str): dataset_ref = await asyncio.to_thread( @@ -395,7 +450,7 @@ async def run( self._runner._create_evaluation, self._project_key, key, - generation, + run_generation, run_tools, run_criteria, ) @@ -425,10 +480,10 @@ async def run( self._project_key, evaluation.id, evaluation_run.id ) raise - config = self._runner._build_handler_config(generation, run_tools) + config = self._runner._build_handler_config(run_generation, run_tools) results = await self._runner._run_rows( dataset_rows, - handler, + generation_handler, config, run_tool_handlers, concurrency, @@ -445,7 +500,6 @@ async def run( if run_criteria: criterion_results = await self._runner._run_criteria_for_results( results, - run_tool_handlers, run_criteria, resolved_judges, concurrency, @@ -572,30 +626,133 @@ def _validate_criteria(criteria: list[Criterion]) -> None: ) @staticmethod - def _validate_judge_handlers(judge_handlers: list[EvalHandler]) -> None: - """Reject judge handlers that cannot be routed by provider and mode. - - A judge handler is only ever chosen by matching its ``provides_for`` - against the judge's resolved provider and mode. One without that - metadata could never be selected, so it would silently fall through to - the generation handler instead of running the judge it was passed for. + def _validate_handlers( + handler: EvalHandler | None, + handlers: Sequence[EvalHandler] | None, + ) -> list[EvalHandler]: + """Return the handler pool, or raise when the arguments are invalid. + + Exactly one of ``handler`` and ``handlers`` is required. Every handler + must be callable and declare ``provides_for``, and no two handlers may + declare the same provider and mode. """ - for index, candidate in enumerate(judge_handlers): + if handler is not None and handlers is not None: + raise EvaluationsError("Pass handler or handlers, not both") + if handler is None and handlers is None: + raise EvaluationsError( + "Pass handler (one handler) or handlers (a list of handlers)" + ) + named: list[tuple[str, EvalHandler]] + if handler is not None: + named = [("handler", handler)] + else: + if not isinstance(handlers, list | tuple): + raise EvaluationsError("handlers must be a list of handlers") + if not handlers: + raise EvaluationsError("handlers must not be empty") + named = [ + (f"handlers[{index}]", item) for index, item in enumerate(handlers) + ] + seen: dict[tuple[str, str], str] = {} + for name, candidate in named: if not callable(candidate): - raise EvaluationsError(f"judge_handlers[{index}] must be callable") - if _provides_for(candidate) is None: + raise EvaluationsError(f"{name} must be callable") + provides_for = _provides_for(candidate) + if provides_for is None: + raise EvaluationsError( + f"{name} does not declare provides_for. Build it with " + "create_handler() or a provider package's create_*_handler() " + "so that it can be matched to a config's provider and mode." + ) + if provides_for in seen: raise EvaluationsError( - f"judge_handlers[{index}] does not declare provides_for. " - "Build judge handlers with create_handler() (or a provider " - "package's create_*_handler()) so they can be matched to a " - "judge's provider and mode." + f"{seen[provides_for]} and {name} both provide for " + f"{provides_for!r}. Pass one handler for each provider and mode." ) + seen[provides_for] = name + return [candidate for _, candidate in named] + + @staticmethod + def _select_generation_handler( + pool: list[EvalHandler], + generation: GenerationConfig, + label: str, + ) -> EvalHandler: + """Select the generation handler, or raise when none matches. + + Also raises when the prompt does not match the handler's mode. + """ + mode = _handler_mode(generation["mode"]) + selected = select_eval_handler(pool, generation["provider"], mode) + if selected is None: + raise EvaluationsError( + f"No handler can run {label}: it needs provider " + f"{generation['provider']!r} in {generation['mode']!r} mode " + f"({mode!r} handlers). The handlers provide for: " + f"{describe_handlers(pool)}." + ) + prompt_error = prompt_mode_error(generation, mode) + if prompt_error is not None: + raise EvaluationsError( + f"The {label} {prompt_error}. Handler: {_provides_for(selected)!r}." + ) + return selected + + async def _read_ai_config( + self, + ai_config: AIConfig, + overrides: GenerationOverrides | None, + pool: list[EvalHandler], + ) -> tuple[GenerationConfig, EvalHandler, AIConfigVariation]: + """Read an AI Config variation and select its generation handler. + + Each check runs as soon as the read that supplies its input returns: + the mode after the AI Config read, the prompt after the variation + read, and the provider after the model config read. + """ + label = f"AI Config {ai_config.key!r} variation {ai_config.variation!r}" + config_mode = await asyncio.to_thread( + self._runner._fetch_ai_config_mode, self._project_key, ai_config.key + ) + mode = normalize_mode(config_mode) + if not any( + (provides_for := _provides_for(candidate)) is not None + and provides_for[1] == mode + for candidate in pool + ): + raise EvaluationsError( + f"AI Config {ai_config.key!r} is in {config_mode!r} mode, which " + f"needs a handler in {mode!r} mode. The handlers provide for: " + f"{describe_handlers(pool)}." + ) + latest = await asyncio.to_thread( + self._runner._fetch_config_variation, + self._project_key, + ai_config.key, + ai_config.variation, + ) + prompt = _merge_generation( + AIConfigVariation.from_api(latest, None, mode).generation, overrides + ) + prompt_error = prompt_mode_error(prompt, mode) + if prompt_error is not None: + raise EvaluationsError( + f"{label} is in {config_mode!r} mode: the {prompt_error}." + ) + model_config = await asyncio.to_thread( + self._runner._fetch_linked_model_config, self._project_key, latest + ) + variation = AIConfigVariation.from_api(latest, model_config, mode) + merged = _merge_generation(variation.generation, overrides) + merged["mode"] = "agent" if mode == "agent" else "completion" + generation = self._validate_generation(merged) + handler = self._select_generation_handler(pool, generation, label) + return generation, handler, variation @staticmethod def _validate_run_args( *, key: str, - handler: EvalHandler, concurrency: int, poll_interval_seconds: float, poll_timeout_seconds: float, @@ -603,8 +760,6 @@ def _validate_run_args( for name, value in (("key", key),): if not value.strip(): raise EvaluationsError(f"{name} must not be blank") - if not callable(handler): - raise EvaluationsError("handler must be callable") if concurrency < 1: raise EvaluationsError("concurrency must be at least 1") for name, seconds in ( @@ -659,7 +814,7 @@ def _validate_dataset_source( @staticmethod def _validate_config_source( *, - generation: GenerationConfig | None, + generation: GenerationConfig | GenerationOverrides | None, ai_config: AIConfig | None, ) -> None: """Require a generation source before any request is made.""" @@ -670,6 +825,11 @@ def _validate_config_source( "Config variation" ) return + if generation is not None and "mode" in generation: + raise EvaluationsError( + "generation.mode cannot be set with ai_config: the AI Config " + "supplies the mode" + ) for name, value in ( ("ai_config.key", ai_config.key), ("ai_config.variation", ai_config.variation), @@ -678,24 +838,28 @@ def _validate_config_source( raise EvaluationsError(f"{name} must not be blank") @staticmethod - def _validate_generation(generation: GenerationConfig | None) -> GenerationConfig: + def _validate_generation(generation: Mapping[str, Any]) -> GenerationConfig: """Check the final generation settings, after any fetched variation is merged.""" - if generation is None: - raise EvaluationsError( - "Pass generation, or ai_config to evaluate an existing AI " - "Config variation" - ) provider = generation.get("provider") model = generation.get("model") if not isinstance(provider, str) or not provider.strip(): raise EvaluationsError("generation.provider is required") if not isinstance(model, str) or not model.strip(): raise EvaluationsError("generation.model is required") + mode = generation.get("mode") + if mode is None: + raise EvaluationsError( + "generation.mode is required. Set it to 'completion' or 'agent'." + ) + if mode not in GENERATION_MODES: + raise EvaluationsError( + f"generation.mode must be 'completion' or 'agent', got {mode!r}" + ) if "instructions" in generation and "messages" in generation: raise EvaluationsError( "generation.instructions and generation.messages are mutually exclusive" ) - return generation + return cast(GenerationConfig, dict(generation)) def init_evaluations( diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/runner.py b/packages/client/src/launchdarkly_ai_server/evaluations/runner.py index b12bab91..acc9a065 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/runner.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/runner.py @@ -10,7 +10,7 @@ from collections.abc import Awaitable, Callable, Mapping, Sequence from dataclasses import dataclass from datetime import UTC, datetime -from typing import Any, Literal +from typing import Any, Literal, cast from ..judge_scoring import ( FORMATTING_INSTRUCTIONS, @@ -25,10 +25,10 @@ row_fields, ) from ..utils import ( - collapse_messages_to_instructions, normalize_mode, parse_template, parse_usage, + select_handler, to_ld_context, ) from .api import ( @@ -54,7 +54,6 @@ handler_config_tools, ) from .types import ( - AIConfigVariation, DatasetRef, DatasetRow, EvaluationRef, @@ -74,13 +73,15 @@ EvalHandler = Callable[..., Awaitable[dict[str, Any]]] +AI_CONFIG_MODES = ("agent", "completion", "judge") + + @dataclass(frozen=True) class JudgeExecution: """A resolved judge paired with the handler selected to run its config.""" resolved: ResolvedJudge handler: EvalHandler - collapse_messages: bool = False def _provides_for( @@ -96,81 +97,52 @@ def _provides_for( return None -def _covers_provider( - provides_for: tuple[str, Literal["agent", "messages"]], - provider: str | None, -) -> bool: - return provides_for[0] == provider or provides_for[0] == "*" +def describe_handlers(handlers: Sequence[EvalHandler]) -> str: + """Return the ``provides_for`` of each handler, for an error message.""" + return ", ".join(repr(_provides_for(handler)) for handler in handlers) -def _find_judge_handler( - judge_handlers: list[EvalHandler], +def select_eval_handler( + handlers: Sequence[EvalHandler], provider: str | None, mode: Literal["agent", "messages"], ) -> EvalHandler | None: - """Find a handler for ``provider`` in ``mode``, exact match before wildcard. - - A wildcard handler is a fallback for multi-provider adapters, so it is only - chosen when no handler names the provider outright -- the priority - ``config()`` already applies to a generation config. Searching in one pass - would instead let the order the caller happened to list its handlers in - decide, sending an OpenAI judge through a LangChain adapter that was merely - listed first. + """Select the handler for ``provider`` and ``mode`` with the ``config()`` rule. + + Returns ``None`` when no handler matches. There is no mode fallback. """ - for exact in (True, False): - for candidate in judge_handlers: - provides_for = _provides_for(candidate) - if provides_for is None or provides_for[1] != mode: - continue - if exact: - if provides_for[0] == provider: - return candidate - elif provides_for[0] == "*": - return candidate - return None + if not provider: + return None + try: + selected = select_handler( + {"provider": {"name": provider}}, + {"mode": mode}, + list(handlers), # type: ignore[arg-type] + ) + except ValueError: + return None + return cast(EvalHandler, selected) -def _select_judge_handler( - resolved: ResolvedJudge, - handler: EvalHandler, - judge_handlers: list[EvalHandler], -) -> JudgeExecution | None: - """Pick the handler that can run this judge's config, or ``None``. - - A judge is an independent AI Config: it may resolve to a different provider - and mode than the evaluation's generation config, and a handler built for - one provider cannot execute another's config. The priority mirrors the - online path (``judges.run_judges``): - - 1. a judge handler in the judge's mode, naming its provider outright - before any wildcard adapter; - 2. an agent-mode judge handler for a messages-mode judge, whose messages - are collapsed into a single instructions block; - 3. the generation handler, when it covers the judge's provider. - - A handler that declares no ``provides_for`` is a plain callable doing its - own routing -- the same contract it already honours for the generation - config -- so it is treated as covering every judge. +def prompt_mode_error( + config: Mapping[str, Any], + mode: Literal["agent", "messages"], +) -> str | None: + """Return an error when the prompt field does not match the handler mode. + + A messages handler needs ``messages`` and an agent handler needs + ``instructions``. A config with neither field is valid. Returns ``None`` + when the prompt matches. """ - match = _find_judge_handler(judge_handlers, resolved.provider, resolved.mode) - if match is not None: - return JudgeExecution(resolved=resolved, handler=match) - if resolved.mode == "messages": - agent_fallback = _find_judge_handler(judge_handlers, resolved.provider, "agent") - if agent_fallback is not None: - return JudgeExecution( - resolved=resolved, handler=agent_fallback, collapse_messages=True - ) - generation_provides_for = _provides_for(handler) - if generation_provides_for is None: - return JudgeExecution(resolved=resolved, handler=handler) - if _covers_provider(generation_provides_for, resolved.provider): - return JudgeExecution( - resolved=resolved, - handler=handler, - collapse_messages=( - generation_provides_for[1] == "agent" and resolved.mode == "messages" - ), + if mode == "messages" and config.get("instructions"): + return ( + "prompt needs 'messages' because the handler is a messages " + "handler, but it has 'instructions'" + ) + if mode == "agent" and config.get("messages"): + return ( + "prompt needs 'instructions' because the handler is an agent " + "handler, but it has 'messages'" ) return None @@ -248,19 +220,43 @@ class EvaluationsRunner: def __init__(self, api: LDApiClient) -> None: self._api = api + def _fetch_ai_config_mode(self, project_key: str, config_key: str) -> str: + """Read the mode of an AI Config from the management API. + + Returns ``"agent"``, ``"completion"`` or ``"judge"``. An absent mode is + ``"completion"``. Raises :class:`EvaluationsError` for any other value, + and when the AI Config does not exist. + """ + description = f"AI Config {config_key!r}" + path = f"projects/{segment(project_key)}/ai-configs/{segment(config_key)}" + try: + raw = require_mapping(self._api.get(path), description=description) + except LDApiError as error: + if error.status == 404: + raise EvaluationsError( + f"LaunchDarkly {description} was not found in project {project_key!r}" + ) from error + raise + mode = raw.get("mode", "completion") + if mode is None: + mode = "completion" + if not isinstance(mode, str) or mode not in AI_CONFIG_MODES: + raise EvaluationsError( + f"LaunchDarkly {description} has an unknown mode {mode!r}. " + "Expected one of: " + ", ".join(repr(m) for m in AI_CONFIG_MODES) + ) + return mode + def _fetch_config_variation( self, project_key: str, config_key: str, variation_key: str, - ) -> AIConfigVariation: - """Read an AI Config variation by key from the management API. - - Flag delivery cannot select a variation by key -- it serves whichever - variation targeting picks for a context -- so this reads the variation - definition directly. Provider and base model parameters live on the - linked model config, and are layered the way the served flag payload - layers them: model-config parameters first, variation parameters over. + ) -> Mapping[str, Any]: + """Read the latest version of an AI Config variation by key. + + Raises :class:`EvaluationsError` when the variation does not exist, + has no versions, or has a ``modelConfigKey`` that is not a string. """ description = f"AI Config variation {config_key!r}/{variation_key!r}" path = ( @@ -285,22 +281,25 @@ def _fetch_config_variation( if not versions: raise EvaluationsError(f"LaunchDarkly {description} has no versions") latest = max(versions, key=lambda item: int(item["version"])) - - # Absent or empty means the variation links no model config, so it has - # no provider -- flag delivery serves an empty provider name for it too. - # Anything other than a string is a response we do not understand. - model_config: Mapping[str, Any] | None = None + # Absent or empty means that the variation links no model config. model_config_key = latest.get("modelConfigKey") if model_config_key is not None and not isinstance(model_config_key, str): raise EvaluationsError( f"LaunchDarkly {description} has a non-string modelConfigKey: " f"{model_config_key!r}" ) - if model_config_key: - model_config = self._fetch_model_config( - project_key, model_config_key, latest.get("modelConfigVersion") - ) - return AIConfigVariation.from_api(latest, model_config) + return latest + + def _fetch_linked_model_config( + self, project_key: str, variation: Mapping[str, Any] + ) -> Mapping[str, Any] | None: + """Read the model config a variation links, or return ``None``.""" + model_config_key = variation.get("modelConfigKey") + if not isinstance(model_config_key, str) or not model_config_key: + return None + return self._fetch_model_config( + project_key, model_config_key, variation.get("modelConfigVersion") + ) def _fetch_model_config( self, @@ -331,17 +330,16 @@ async def _resolve_judges( self, project_key: str, judges: list[Judge], - handler: EvalHandler, - judge_handlers: list[EvalHandler] | None = None, + handlers: Sequence[EvalHandler], ) -> dict[str, JudgeExecution]: - """Resolve LD Judge configs before any evaluation records are created. + """Resolve LD Judge configs and select a handler for each one. - Each judge is paired with the handler that can execute its config here, - rather than at scoring time, so a judge no handler covers fails the run - before any records exist or any generation spend happens. + Runs before any evaluation record is created. Raises one + :class:`EvaluationsError` that lists every judge with no handler and + every judge whose prompt does not match its handler's mode. """ - available_judge_handlers = list(judge_handlers or []) resolved: dict[str, JudgeExecution] = {} + problems: list[str] = [] # variation() rejects a context without kind and key; use the same # context shape the emitted evaluation events are attributed to. context: dict[str, Any] = {"kind": "evaluation", "key": project_key} @@ -369,9 +367,13 @@ async def _resolve_judges( if isinstance(provider_value, Mapping) else None ) + # Judges cannot use tools. + judge_config = { + name: value for name, value in config.items() if name != "tools" + } resolved_judge = ResolvedJudge( key=judge.key, - config=dict(config), + config=judge_config, variation_key=str(meta.get("variationKey") or ""), version=int(meta["version"]) if isinstance(meta.get("version"), int) @@ -381,18 +383,28 @@ async def _resolve_judges( meta.get("mode") if isinstance(meta.get("mode"), str) else None ), ) - execution = _select_judge_handler( - resolved_judge, handler, available_judge_handlers + handler = select_eval_handler( + handlers, resolved_judge.provider, resolved_judge.mode ) - if execution is None: - raise EvaluationsError( - f"No handler can run LaunchDarkly judge {judge.key!r}: its " - f"config is served by provider {resolved_judge.provider!r} in " - f"{resolved_judge.mode!r} mode, which neither the generation " - "handler nor any judge_handlers entry provides for. Pass a " - "handler for that provider to run(judge_handlers=[...])." + if handler is None: + problems.append( + f"judge {judge.key!r} needs a handler for provider " + f"{resolved_judge.provider!r} in {resolved_judge.mode!r} mode" ) - resolved[judge.key] = execution + continue + prompt_error = prompt_mode_error(judge_config, resolved_judge.mode) + if prompt_error is not None: + problems.append(f"judge {judge.key!r}: {prompt_error}") + continue + resolved[judge.key] = JudgeExecution( + resolved=resolved_judge, handler=handler + ) + if problems: + raise EvaluationsError( + "Cannot run the judges of this evaluation: " + + "; ".join(problems) + + f". The handlers provide for: {describe_handlers(handlers)}." + ) return resolved def _fetch_dataset(self, project_key: str, dataset_key: str) -> DatasetRef: @@ -912,7 +924,6 @@ async def _run_scorer_for_result( async def _run_ld_judge_for_result( self, row: Mapping[str, Any], - tool_handlers: dict[str, ToolImplementation], judge: Judge, execution: JudgeExecution, ) -> dict[str, Any]: @@ -939,19 +950,11 @@ async def _run_ld_judge_for_result( # parse_template pass, so ``{{...}}`` sequences inside generated output # or dataset values are never re-expanded into the judge prompt. variables = self._judge_variables(row, judge) - # An agent-mode handler standing in for a messages-mode judge needs the - # messages folded into one instructions block, exactly as the online - # path does before handing a judge config to an agent handler. - judge_config = ( - collapse_messages_to_instructions(resolved.config) - if execution.collapse_messages - else resolved.config - ) try: result = await execution.handler( - dict(judge_config), + dict(resolved.config), row.get("output"), - tool_handlers, + {}, { **variables, "formatting_instructions": FORMATTING_INSTRUCTIONS, @@ -1007,7 +1010,6 @@ async def _run_ld_judge_for_result( async def _run_criteria_for_results( self, rows: list[dict[str, Any]], - tool_handlers: dict[str, ToolImplementation], criteria: list[Criterion], resolved_judges: Mapping[str, JudgeExecution], concurrency: int, @@ -1024,7 +1026,6 @@ async def run_one( return await self._run_scorer_for_result(row, criterion) return await self._run_ld_judge_for_result( row, - tool_handlers, criterion, resolved_judges[criterion.key], ) diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/types.py b/packages/client/src/launchdarkly_ai_server/evaluations/types.py index a359ba14..3422ff68 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/types.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/types.py @@ -2,7 +2,7 @@ from collections.abc import Mapping from dataclasses import dataclass, field -from typing import Any, Literal, TypedDict +from typing import Any, Literal, Required, TypedDict @dataclass @@ -26,8 +26,43 @@ def from_wire(cls, data: Mapping[str, Any]) -> Usage: ) +GenerationMode = Literal["completion", "agent"] +"""The mode of a generation config. It selects the handler that runs the config.""" + + class GenerationConfig(TypedDict, total=False): - """Generation settings stored on the evaluation and passed to its handler.""" + """Generation settings stored on the evaluation and passed to its handler. + + ``provider``, ``model`` and ``mode`` are required. + + ``mode`` must be the mode of a handler that you pass to ``run()``: + + - ``"completion"``: the prompt is a list of ``messages``, and the handler + sends one request to a chat or messages API. Use it with a messages + handler. + - ``"agent"``: the prompt is one ``instructions`` string, and an agent + framework runs the model, for example OpenAI Agents, Claude Agents, or + LangChain agents. Use it with an agent handler. + + If you are not sure, use the mode of the AI Config in LaunchDarkly. + """ + + provider: Required[str] + model: Required[str] + mode: Required[GenerationMode] + parameters: dict[str, Any] + instructions: str + messages: list[dict[str, Any]] + prompt_snippets: dict[str, str] + output_format: dict[str, Any] + + +class GenerationOverrides(TypedDict, total=False): + """Generation settings that replace the values of an AI Config variation. + + Pass these with ``ai_config``. Each field is optional. There is no + ``mode``, because the AI Config supplies it. + """ provider: str model: str @@ -107,7 +142,7 @@ class AIConfigVariation: the judges attached to the variation. """ - generation: GenerationConfig + generation: GenerationOverrides tool_versions: dict[str, int] = field(default_factory=dict) judge_keys: list[str] = field(default_factory=list) @@ -116,6 +151,7 @@ def from_api( cls, data: Mapping[str, Any], model_config: Mapping[str, Any] | None = None, + mode: Literal["agent", "messages"] = "messages", ) -> AIConfigVariation: """Build from one variation version and the model config it links. @@ -124,6 +160,11 @@ def from_api( a pure translation of API shapes. Provider and base parameters come from the model config; the variation's own parameters are layered over them, as the served flag payload layers them. + + ``mode`` is the AI Config's handler mode. It selects the prompt field: + ``instructions`` in agent mode and ``messages`` in messages mode. When + that field is empty, the other field is kept, so the prompt check can + reject it. """ model = data.get("model") model = model if isinstance(model, Mapping) else {} @@ -145,7 +186,7 @@ def from_api( if isinstance(config_model_id, str) and config_model_id: model_name = config_model_id - generation = GenerationConfig() + generation = GenerationOverrides() if isinstance(provider, str) and provider: generation["provider"] = provider if isinstance(model_name, str) and model_name: @@ -154,9 +195,12 @@ def from_api( generation["parameters"] = parameters instructions = data.get("instructions") messages = data.get("messages") - if isinstance(instructions, str) and instructions: + has_instructions = isinstance(instructions, str) and bool(instructions) + has_messages = isinstance(messages, list) and bool(messages) + use_instructions = has_instructions and (mode == "agent" or not has_messages) + if use_instructions and isinstance(instructions, str): generation["instructions"] = instructions - elif isinstance(messages, list) and messages: + elif has_messages and isinstance(messages, list): generation["messages"] = [ dict(message) for message in messages if isinstance(message, Mapping) ] diff --git a/packages/client/tests/test_evaluations_run.py b/packages/client/tests/test_evaluations_run.py index 796a95a1..408e9d57 100644 --- a/packages/client/tests/test_evaluations_run.py +++ b/packages/client/tests/test_evaluations_run.py @@ -96,6 +96,27 @@ def dataset_page( return {"items": items, "totalCount": total, "_links": links} +def tagged( + fn: Callable[..., Any], + provides_for: tuple[str, str] = ("OpenAI", "messages"), +) -> Callable[..., Any]: + """Return ``fn`` with ``provides_for``, keeping its four-argument call.""" + + async def call(*args: Any, **kwargs: Any) -> Any: + return await fn(*args, **kwargs) + + call.provides_for = provides_for # type: ignore[attr-defined] + return call + + +def prompt_text(config: dict[str, Any]) -> str: + """Return a config's instructions, or the content of its first message.""" + if config.get("instructions"): + return str(config["instructions"]) + messages = config.get("messages") or [] + return str(messages[0].get("content", "")) if messages else "" + + async def successful_handler( config: dict[str, Any], user_input: str | None, @@ -242,11 +263,12 @@ async def test_complete_run_with_zero_failed_and_error_rows_passes( result = await evals.run( key="support-qa-unique", dataset="golden", - handler=successful_handler, + handler=tagged(successful_handler, ("OpenAI", "agent")), tools=[await evals.tools.get("lookup_order", implementation=lookup_order)], generation={ "provider": "OpenAI", "model": "gpt-4o", + "mode": "agent", "parameters": {"temperature": 0.2}, "instructions": "Help the user.", }, @@ -419,8 +441,8 @@ async def handler(*args: object) -> dict[str, Any]: result = await evals.run( key="eval-key", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) assert result.passed is False @@ -481,8 +503,8 @@ async def handler(*args: object) -> dict[str, Any]: result = await evals.run( key="eval-key", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) summary_requests = [ @@ -542,8 +564,8 @@ async def handler(*args: object) -> dict[str, Any]: result = await evals.run( key="eval-key", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) summary_requests = [ @@ -600,8 +622,8 @@ async def handler(*args: object) -> dict[str, Any]: result = await evals.run( key="eval-key", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) summary_requests = [ @@ -653,8 +675,8 @@ async def handler(*args: object) -> dict[str, Any]: await evals.run( key="eval-key", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) summary_requests = [ @@ -697,8 +719,8 @@ async def handler(*args: object) -> dict[str, Any]: result = await evals.run( key="eval-key", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, poll_interval_seconds=0, poll_timeout_seconds=600, ) @@ -713,8 +735,8 @@ async def handler(*args: object) -> dict[str, Any]: await evals.run( key="eval-key", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, poll_timeout_seconds=-1, ) @@ -739,8 +761,8 @@ async def handler(*args: object) -> dict[str, Any]: await evals.run( key="eval-key", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, poll_interval_seconds=poll_interval_seconds, poll_timeout_seconds=poll_timeout_seconds, ) @@ -790,8 +812,8 @@ async def handler(*args: object) -> dict[str, Any]: result = await evals.run( key="eval-key", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) assert result.passed is True @@ -823,8 +845,8 @@ async def handler(*args: object) -> dict[str, Any]: await evals.run( key="eval-key", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) @@ -875,8 +897,8 @@ async def handler(*args: object) -> dict[str, Any]: result = await evals.run( key="eval-key", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) assert result.summary.failed_rows == 1 @@ -892,10 +914,11 @@ async def test_run_rejects_instructions_and_messages_before_network_io() -> None await evals.run( key="eval-key", dataset="golden", - handler=successful_handler, + handler=tagged(successful_handler), generation={ "provider": "OpenAI", "model": "gpt-4o", + "mode": "completion", "instructions": "System prompt", "messages": [{"role": "user", "content": "{{input}}"}], }, @@ -986,7 +1009,7 @@ async def handler( result = await evals.run( key="eval-key", dataset="golden", - handler=handler, + handler=tagged(handler), tools=[ EvalTool( key="lookup_order", @@ -995,7 +1018,7 @@ async def handler( description="Look up an order", ) ], - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) assert result.passed is True @@ -1048,13 +1071,13 @@ async def handler(*args: object) -> dict[str, Any]: await evals.run( key="eval-key", dataset="golden", - handler=handler, + handler=tagged(handler), tools=[ EvalTool( key="lookup_order", implementation=lookup_order, schema=ORDER_SCHEMA ) ], - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) assert transport.requests[2]["body"]["tools"] == [ @@ -1101,7 +1124,7 @@ async def handler( await evals.run( key="eval-key", dataset="golden", - handler=handler, + handler=tagged(handler), tools=[ await evals.tools.get("lookup_order", implementation=lookup_order), EvalTool( @@ -1111,7 +1134,7 @@ async def handler( description="Refund an order", ), ], - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) # Exactly one tool GET, for the library key only. @@ -1251,9 +1274,9 @@ async def test_bad_tool_entry_is_rejected_with_zero_requests( await evals.run( key="eval-key", dataset="golden", - handler=successful_handler, + handler=tagged(successful_handler), tools=tools, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) assert transport.requests == [] @@ -1270,7 +1293,7 @@ async def test_native_tool_paired_with_an_inline_definition_is_rejected() -> Non await evals.run( key="eval-key", dataset="golden", - handler=successful_handler, + handler=tagged(successful_handler), tools=[ EvalTool( key="lookup_order", @@ -1278,7 +1301,7 @@ async def test_native_tool_paired_with_an_inline_definition_is_rejected() -> Non schema=ORDER_SCHEMA, ) ], - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) assert transport.requests == [] @@ -1302,11 +1325,11 @@ async def handler(*args: object) -> dict[str, Any]: await evals.run( key="eval-key", dataset="golden", - handler=handler, + handler=tagged(handler), tools=[ await evals.tools.get("web_search", implementation=NativeTool("WebSearch")) ], - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) assert recorded_paths(transport)[0] == ("GET", "projects/proj/ai-tools/web_search") @@ -1329,7 +1352,7 @@ async def test_a_repeated_tool_key_is_rejected_with_zero_requests() -> None: await evals.run( key="eval-key", dataset="golden", - handler=successful_handler, + handler=tagged(successful_handler), tools=[ EvalTool( key="lookup_order", @@ -1342,7 +1365,7 @@ async def test_a_repeated_tool_key_is_rejected_with_zero_requests() -> None: schema=ORDER_SCHEMA, ), ], - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) assert transport.requests == [] @@ -1382,8 +1405,8 @@ async def test_empty_dataset_fails_before_evaluation_or_run_creation() -> None: await evals.run( key="eval-key", dataset="golden", - handler=successful_handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(successful_handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) assert [request["method"] for request in transport.requests] == ["GET", "GET"] @@ -1480,8 +1503,8 @@ async def fake_init_client(options: dict[str, Any]) -> MagicMock: result = await evals.run( key="eval-key", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) assert set(calls) == {"bad", "good"} @@ -1548,7 +1571,12 @@ async def fake_extract_variation( "config": { "provider": {"name": "OpenAI"}, "model": {"name": "gpt-4o"}, - "instructions": "Judge {{response_to_evaluate}} against {{expected_output}}", + "messages": [ + { + "role": "system", + "content": "Judge {{response_to_evaluate}} against {{expected_output}}", + } + ], }, "meta": {"variationKey": "default", "version": 12}, } @@ -1567,15 +1595,18 @@ async def handler( tool_handlers: dict[str, Callable[..., Any]], variables: dict[str, Any], ) -> dict[str, Any]: - if "Judge" in config.get("instructions", ""): + if "Judge" in prompt_text(config): assert user_input == "generated" assert variables["response_to_evaluate"] == "generated" assert variables["expected_output"] == "Answer A" # The SDK hands the judge config over unrendered; the handler owns # the single template pass. - assert config["instructions"] == ( + assert prompt_text(config) == ( "Judge {{response_to_evaluate}} against {{expected_output}}" ) + # Judges cannot use tools. + assert tool_handlers == {} + assert "tools" not in config assert variables["formatting_instructions"].startswith( "Your response MUST be in valid JSON" ) @@ -1600,8 +1631,8 @@ async def handler( result = await evals.run( key="support-qa", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -1696,7 +1727,12 @@ async def fake_extract_variation( config: dict[str, Any] = { "provider": {"name": "OpenAI"}, "model": {"name": "gpt-4o"}, - "instructions": "Judge {{response_to_evaluate}} against {{expected_output}}", + "messages": [ + { + "role": "system", + "content": "Judge {{response_to_evaluate}} against {{expected_output}}", + } + ], } if is_inverted is not None: config["isInverted"] = is_inverted @@ -1719,7 +1755,7 @@ async def handler( tool_handlers: dict[str, Callable[..., Any]], variables: dict[str, Any], ) -> dict[str, Any]: - if "Judge" in config.get("instructions", ""): + if "Judge" in prompt_text(config): return { "output": '{"score": 0.86, "reasoning": "matches policy"}', "usage": {"input_tokens": 640, "output_tokens": 48}, @@ -1732,8 +1768,8 @@ async def handler( result = await evals.run( key="support-qa", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy", threshold=threshold)], ) @@ -1796,7 +1832,9 @@ async def fake_extract_variation( "config": { "provider": {"name": "OpenAI"}, "model": {"name": "gpt-4o"}, - "instructions": "Judge {{response_to_evaluate}}", + "messages": [ + {"role": "system", "content": "Judge {{response_to_evaluate}}"} + ], }, "meta": {"variationKey": "default", "version": 12}, } @@ -1815,15 +1853,15 @@ async def handler( tool_handlers: dict[str, Callable[..., Any]], variables: dict[str, Any], ) -> dict[str, Any]: - if "Judge" in config.get("instructions", ""): + if "Judge" in prompt_text(config): return {"output": '{"score": 0.9, "reasoning": "fine"}'} return {"output": "generated"} await evals.run( key="support-qa", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -1894,8 +1932,8 @@ async def handler(*args: object) -> dict[str, Any]: await evals.run( key="eval-key", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="security-judge")], ) @@ -1953,8 +1991,8 @@ def check_refund(row: DatasetRow, output: Any) -> bool: result = await evals.run( key="support-qa", dataset="support-golden-v3", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Scorer(name="refund-exists", fn=check_refund)], ) @@ -2037,7 +2075,12 @@ async def fake_extract_variation( "config": { "provider": {"name": "OpenAI"}, "model": {"name": "gpt-4o"}, - "instructions": "Judge {{response_to_evaluate}} against {{expected_output}}", + "messages": [ + { + "role": "system", + "content": "Judge {{response_to_evaluate}} against {{expected_output}}", + } + ], }, "meta": {"variationKey": "default", "version": 12}, } @@ -2076,15 +2119,15 @@ async def handler( tool_handlers: dict[str, Callable[..., Any]], variables: dict[str, Any], ) -> dict[str, Any]: - if "Judge" in config.get("instructions", ""): + if "Judge" in prompt_text(config): return {"output": judge_output} return {"output": "generated"} result = await evals.run( key="support-qa", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -2117,8 +2160,8 @@ async def handler( tool_handlers: dict[str, Callable[..., Any]], variables: dict[str, Any], ) -> dict[str, Any]: - if "Judge" in config.get("instructions", ""): - rendered = parse_template(config["instructions"], variables) + if "Judge" in prompt_text(config): + rendered = parse_template(prompt_text(config), variables) # The placeholder smuggled in via the generated output must stay # literal text after the handler's single render pass. assert rendered == "Judge {{expected_output}} leaked? against Answer A" @@ -2128,8 +2171,8 @@ async def handler( result = await evals.run( key="support-qa", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -2174,7 +2217,7 @@ async def handler( tool_handlers: dict[str, Callable[..., Any]], variables: dict[str, Any], ) -> dict[str, Any]: - if "Judge" in config.get("instructions", ""): + if "Judge" in prompt_text(config): assert variables["expected_output"] == "" assert variables["ground_truth_context"] == "" return {"output": '{"score": 1, "reasoning": "ok"}'} @@ -2183,8 +2226,8 @@ async def handler( result = await evals.run( key="support-qa", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) assert result.passed is True @@ -2204,8 +2247,8 @@ async def handler(*args: object) -> dict[str, Any]: await evals.run( key="support-qa", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[ Judge(key="accuracy"), Scorer(name="accuracy", fn=lambda row, output: True), @@ -2233,8 +2276,8 @@ async def handler(*args: object) -> dict[str, Any]: await evals.run( key="support-qa", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[ Judge(key="Accuracy"), Scorer(name="accuracy", fn=lambda row, output: True), @@ -2263,15 +2306,15 @@ async def handler( tool_handlers: dict[str, Callable[..., Any]], variables: dict[str, Any], ) -> dict[str, Any]: - if "Judge" in config.get("instructions", ""): + if "Judge" in prompt_text(config): raise AssertionError("judges must not run for errored generations") raise RuntimeError("provider unavailable") result = await evals.run( key="support-qa", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -2315,7 +2358,7 @@ async def handler( tool_handlers: dict[str, Callable[..., Any]], variables: dict[str, Any], ) -> dict[str, Any]: - if "Judge" in config.get("instructions", ""): + if "Judge" in prompt_text(config): return {"output": '{"score": 1, "reasoning": "ok"}'} return {"output": "generated"} @@ -2323,8 +2366,8 @@ async def handler( await evals.run( key="support-qa", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[ Judge(key="$ld:ai:judge:accuracy"), Scorer(name="nonempty", fn=lambda row, output: bool(output)), @@ -2374,7 +2417,17 @@ async def fake_extract_variation( "config": { "provider": {"name": provider}, "model": {"name": "judge-model"}, - **(config or {"instructions": "Judge {{response_to_evaluate}}"}), + **( + config + or { + "messages": [ + { + "role": "system", + "content": "Judge {{response_to_evaluate}}", + } + ] + } + ), }, "meta": meta, } @@ -2416,17 +2469,18 @@ async def test_judge_on_another_provider_fails_before_any_records_are_created( key="support-qa", dataset="golden", handler=create_handler(("OpenAI", "messages"), _generation_only), - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) - assert "No handler can run LaunchDarkly judge" in str(error.value) + assert "Cannot run the judges" in str(error.value) assert "'Anthropic'" in str(error.value) + assert "('OpenAI', 'messages')" in str(error.value) assert transport.requests == [] @pytest.mark.asyncio -async def test_judge_handlers_route_a_judge_to_its_own_provider( +async def test_handlers_route_a_judge_to_its_own_provider( monkeypatch: pytest.MonkeyPatch, stub_sdk_client: MagicMock, ) -> None: @@ -2450,10 +2504,12 @@ async def anthropic_judge( result = await evals.run( key="support-qa", dataset="golden", - handler=create_handler(("OpenAI", "messages"), _generation_only), - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handlers=[ + create_handler(("OpenAI", "messages"), _generation_only), + create_handler(("Anthropic", "messages"), anthropic_judge), + ], + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], - judge_handlers=[create_handler(("Anthropic", "messages"), anthropic_judge)], ) assert result.passed is True @@ -2502,10 +2558,12 @@ async def run( result = await evals.run( key="support-qa", dataset="golden", - handler=create_handler(("OpenAI", "messages"), _generation_only), - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handlers=[ + create_handler(("OpenAI", "messages"), _generation_only), + *([wildcard, exact] if wildcard_first else [exact, wildcard]), + ], + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], - judge_handlers=[wildcard, exact] if wildcard_first else [exact, wildcard], ) assert result.passed is True @@ -2537,10 +2595,12 @@ async def wildcard_judge( result = await evals.run( key="support-qa", dataset="golden", - handler=create_handler(("OpenAI", "messages"), _generation_only), - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handlers=[ + create_handler(("OpenAI", "messages"), _generation_only), + create_handler(("*", "messages"), wildcard_judge), + ], + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], - judge_handlers=[create_handler(("*", "messages"), wildcard_judge)], ) assert result.passed is True @@ -2548,52 +2608,107 @@ async def wildcard_judge( @pytest.mark.asyncio -async def test_agent_handler_runs_a_messages_mode_judge_with_collapsed_messages( +async def test_a_messages_mode_judge_never_runs_on_an_agent_handler( monkeypatch: pytest.MonkeyPatch, - stub_sdk_client: MagicMock, ) -> None: - """Mirrors the online path's agent-mode fallback for a messages-mode judge.""" + """There is no mode fallback: the run fails before any record is created.""" transport = judge_run_transport() - judge_variation( - monkeypatch, - provider="Anthropic", - mode="messages", - config={ - "messages": [ - {"role": "system", "content": "Grade strictly."}, - {"role": "user", "content": "Judge {{response_to_evaluate}}"}, - ] - }, - ) + judge_variation(monkeypatch, provider="OpenAI", mode="messages") evals = init_evaluations( project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport ) - judged: list[dict[str, Any]] = [] + agent = AsyncMock(return_value={"output": "generated"}) - async def anthropic_agent_judge( - config: dict[str, Any], - user_input: str | None = None, - tool_handlers: dict[str, Callable[..., Any]] | None = None, - variables: dict[str, Any] | None = None, - history: list[dict[str, Any]] | None = None, + with pytest.raises(EvaluationsError) as error: + await evals.run( + key="support-qa", + dataset="golden", + handler=tagged(agent, ("OpenAI", "agent")), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "agent"}, + criteria=[Judge(key="$ld:ai:judge:accuracy")], + ) + + assert "'$ld:ai:judge:accuracy'" in str(error.value) + assert "'messages' mode" in str(error.value) + assert transport.requests == [] + agent.assert_not_called() + + +@pytest.mark.asyncio +async def test_every_judge_problem_is_reported_in_one_error( + monkeypatch: pytest.MonkeyPatch, +) -> None: + async def fake_extract_variation( + key: str, context: dict[str, Any] ) -> dict[str, Any]: - judged.append(config) - return {"output": '{"score": 1, "reasoning": "ok"}'} + config: dict[str, Any] = { + "provider": {"name": "Anthropic" if key == "uncovered" else "OpenAI"}, + "model": {"name": "judge-model"}, + } + if key == "wrong-prompt": + config["instructions"] = "Judge {{response_to_evaluate}}" + else: + config["messages"] = [{"role": "system", "content": "Judge"}] + return {"config": config, "meta": {"variationKey": "default", "version": 1}} + + monkeypatch.setattr( + "launchdarkly_ai_server.evaluations.runner.extract_variation", + fake_extract_variation, + ) + transport = judge_run_transport() + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + + with pytest.raises(EvaluationsError) as error: + await evals.run( + key="support-qa", + dataset="golden", + handler=tagged(_generation_only), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, + criteria=[Judge(key="uncovered"), Judge(key="wrong-prompt")], + ) + + message = str(error.value) + assert "judge 'uncovered' needs a handler for provider 'Anthropic'" in message + assert "judge 'wrong-prompt': prompt needs 'messages'" in message + assert transport.requests == [] + + +@pytest.mark.asyncio +async def test_mixed_modes_route_generation_and_judge_to_their_own_handlers( + monkeypatch: pytest.MonkeyPatch, + stub_sdk_client: MagicMock, +) -> None: + transport = judge_run_transport() + judge_variation(monkeypatch, provider="OpenAI", mode="completion") + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + agent = AsyncMock(return_value={"output": "generated"}) + messages = AsyncMock(return_value={"output": '{"score": 1, "reasoning": "ok"}'}) result = await evals.run( key="support-qa", dataset="golden", - handler=create_handler(("OpenAI", "messages"), _generation_only), - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handlers=[ + tagged(agent, ("OpenAI", "agent")), + tagged(messages, ("OpenAI", "messages")), + ], + generation={ + "provider": "OpenAI", + "model": "gpt-4o", + "mode": "agent", + "instructions": "Help the user.", + }, criteria=[Judge(key="$ld:ai:judge:accuracy")], - judge_handlers=[create_handler(("Anthropic", "agent"), anthropic_agent_judge)], ) assert result.passed is True - assert judged[0]["instructions"] == ( - "Grade strictly.\n\nJudge {{response_to_evaluate}}" - ) - assert judged[0]["messages"] == [] + assert agent.await_count == 1 + assert agent.await_args.args[0]["instructions"] == "Help the user." + assert messages.await_count == 1 + assert prompt_text(messages.await_args.args[0]) == "Judge {{response_to_evaluate}}" @pytest.mark.asyncio @@ -2615,8 +2730,8 @@ async def openai_handler( variables: dict[str, Any] | None = None, history: list[dict[str, Any]] | None = None, ) -> dict[str, Any]: - calls.append(config.get("instructions")) - if "Judge" in (config.get("instructions") or ""): + calls.append(prompt_text(config)) + if "Judge" in prompt_text(config): return {"output": '{"score": 0.9, "reasoning": "ok"}'} return {"output": "generated"} @@ -2624,7 +2739,7 @@ async def openai_handler( key="support-qa", dataset="golden", handler=create_handler(("OpenAI", "messages"), openai_handler), - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -2633,7 +2748,7 @@ async def openai_handler( @pytest.mark.asyncio -async def test_judge_handlers_must_declare_the_provider_they_serve( +async def test_every_handler_must_declare_the_provider_it_serves( monkeypatch: pytest.MonkeyPatch, ) -> None: """An unrouted judge handler would silently never be selected.""" @@ -2647,10 +2762,9 @@ async def test_judge_handlers_must_declare_the_provider_they_serve( await evals.run( key="support-qa", dataset="golden", - handler=_generation_only, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handlers=[tagged(_generation_only), _generation_only], + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], - judge_handlers=[_generation_only], ) assert transport.requests == [] @@ -2702,7 +2816,7 @@ async def handler( variables: dict[str, Any], ) -> dict[str, Any]: nonlocal in_flight, max_in_flight - if "Judge" in config.get("instructions", ""): + if "Judge" in prompt_text(config): in_flight += 1 max_in_flight = max(max_in_flight, in_flight) await asyncio.sleep(0.01) @@ -2713,8 +2827,8 @@ async def handler( result = await evals.run( key="support-qa", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], concurrency=2, ) @@ -2790,7 +2904,7 @@ async def handler( tool_handlers: dict[str, Callable[..., Any]], variables: dict[str, Any], ) -> dict[str, Any]: - if "Judge" in config.get("instructions", ""): + if "Judge" in prompt_text(config): seen["message_history"] = variables["message_history"] # The trajectory lives in message_history and nowhere else: this is # already the transcript variable every judge reads, so a second @@ -2805,9 +2919,9 @@ async def handler( result = await evals.run( key="support-qa", dataset="golden", - handler=handler, + handler=tagged(handler), tools=[await evals.tools.get("lookup_order", implementation=lookup_order)], - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -2855,7 +2969,7 @@ async def handler( tool_handlers: dict[str, Callable[..., Any]], variables: dict[str, Any], ) -> dict[str, Any]: - if "Judge" in config.get("instructions", ""): + if "Judge" in prompt_text(config): histories[str(user_input)] = variables["message_history"] return {"output": '{"score": 1, "reasoning": "ok"}'} row = str(user_input).split()[-1] @@ -2868,9 +2982,9 @@ async def handler( result = await evals.run( key="support-qa", dataset="golden", - handler=handler, + handler=tagged(handler), tools=[await evals.tools.get("lookup_order", implementation=lookup_order)], - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], concurrency=2, ) @@ -2901,7 +3015,7 @@ async def handler( tool_handlers: dict[str, Callable[..., Any]], variables: dict[str, Any], ) -> dict[str, Any]: - if "Judge" in config.get("instructions", ""): + if "Judge" in prompt_text(config): seen["message_history"] = variables["message_history"] return {"output": '{"score": 0, "reasoning": "should have looked it up"}'} return {"output": "I do not know."} @@ -2909,11 +3023,11 @@ async def handler( await evals.run( key="support-qa", dataset="golden", - handler=handler, + handler=tagged(handler), tools=[ await evals.tools.get("lookup_order", implementation=lambda args: "unused") ], - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -2946,7 +3060,7 @@ async def handler( tool_handlers: dict[str, Callable[..., Any]], variables: dict[str, Any], ) -> dict[str, Any]: - if "Judge" in config.get("instructions", ""): + if "Judge" in prompt_text(config): seen["message_history"] = variables["message_history"] return {"output": '{"score": 1, "reasoning": "ok"}'} return {"output": "generated"} @@ -2954,8 +3068,8 @@ async def handler( await evals.run( key="support-qa", dataset="golden", - handler=handler, - generation={"provider": "OpenAI", "model": "gpt-4o"}, + handler=tagged(handler), + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -2985,7 +3099,12 @@ async def fake_extract_variation( "config": { "provider": {"name": "OpenAI"}, "model": {"name": "gpt-4o"}, - "instructions": "Judge this history: {{message_history}}", + "messages": [ + { + "role": "system", + "content": "Judge this history: {{message_history}}", + } + ], }, "meta": {"variationKey": "default", "version": 12}, } @@ -3004,8 +3123,8 @@ async def handler( tool_handlers: dict[str, Callable[..., Any]], variables: dict[str, Any], ) -> dict[str, Any]: - if "Judge this history" in config.get("instructions", ""): - rendered = parse_template(config["instructions"], variables) + if "Judge this history" in prompt_text(config): + rendered = parse_template(prompt_text(config), variables) assert "result: {{expected_output}} leaked?" in rendered assert "Answer leaked?" not in rendered return {"output": '{"score": 1, "reasoning": "ok"}'} @@ -3015,14 +3134,14 @@ async def handler( result = await evals.run( key="support-qa", dataset="golden", - handler=handler, + handler=tagged(handler), tools=[ await evals.tools.get( "lookup_order", implementation=lambda args: "{{expected_output}} leaked?", ) ], - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -3063,6 +3182,9 @@ async def handler( assert results[0]["tool_calls"][0].result == "shipped" +AGENT_AI_CONFIG = {"key": "support-agent", "mode": "agent"} + + def config_variation_page(**overrides: Any) -> dict[str, Any]: """A getAIConfigVariation response holding two versions of one variation.""" latest: dict[str, Any] = { @@ -3100,6 +3222,7 @@ def config_variation_page(**overrides: Any) -> dict[str, Any]: def fetched_run_responses(variation_page: dict[str, Any]) -> list[HttpResponse]: """Every response a run seeded from an AI Config variation needs, in order.""" return [ + response(200, AGENT_AI_CONFIG), response(200, variation_page), response(200, MODEL_CONFIG), response(200, {"id": "dataset-id", "name": "golden"}), @@ -3151,17 +3274,20 @@ async def handler(config: dict[str, Any], *args: object) -> dict[str, Any]: result = await evals.run( key="eval-key", dataset="golden", - handler=handler, + handler=tagged(handler, ("OpenAI", "agent")), ai_config=AIConfig(key="support-agent", variation="control"), ) assert result.passed is True - assert transport.requests[0]["url"] == ( + assert transport.requests[0]["url"].endswith( + "/projects/proj/ai-configs/support-agent" + ) + assert transport.requests[1]["url"] == ( "https://app.launchdarkly.com/api/v2/projects/proj/ai-configs/" "support-agent/variations/control" ) # The pinned model-config version is the one read. - assert transport.requests[1]["url"].endswith( + assert transport.requests[2]["url"].endswith( "/projects/proj/ai-configs/model-configs/OpenAI.gpt-4o?version=3" ) body = evaluation_post(transport) @@ -3187,12 +3313,12 @@ async def handler(*args: object) -> dict[str, Any]: await evals.run( key="eval-key", dataset="golden", - handler=handler, + handler=tagged(handler, ("OpenAI", "agent")), ai_config=AIConfig(key="support-agent", variation="control"), generation={ "model": "gpt-4o-mini", "parameters": {"temperature": 0.1}, - "messages": [{"role": "system", "content": "Candidate prompt"}], + "instructions": "Candidate prompt", }, ) @@ -3201,7 +3327,7 @@ async def handler(*args: object) -> dict[str, Any]: assert body["generationModel"] == "gpt-4o-mini" # parameters merge key by key rather than replacing the fetched set. assert body["parameters"] == {"max_tokens": 100, "temperature": 0.1} - # messages replace the fetched instructions instead of clashing with them. + # The override replaces the fetched instructions. assert body["messages"] == [{"role": "system", "content": "Candidate prompt"}] @@ -3211,6 +3337,7 @@ async def test_variation_tools_without_implementations_fail_before_mutating_requ ): transport = SequencedTransport( [ + response(200, AGENT_AI_CONFIG), response( 200, config_variation_page(tools=[{"key": "lookup_order", "version": 4}]), @@ -3224,11 +3351,15 @@ async def test_variation_tools_without_implementations_fail_before_mutating_requ await evals.run( key="eval-key", dataset="golden", - handler=successful_handler, + handler=tagged(successful_handler, ("OpenAI", "agent")), ai_config=AIConfig(key="support-agent", variation="control"), ) - assert [request["method"] for request in transport.requests] == ["GET", "GET"] + assert [request["method"] for request in transport.requests] == [ + "GET", + "GET", + "GET", + ] @pytest.mark.asyncio @@ -3237,6 +3368,7 @@ async def test_variation_judges_become_the_default_criteria( ) -> None: transport = SequencedTransport( [ + response(200, AGENT_AI_CONFIG), response( 200, config_variation_page( @@ -3270,17 +3402,24 @@ async def handler(*args: object) -> dict[str, Any]: await evals.run( key="eval-key", dataset="golden", - handler=handler, + handler=tagged(handler, ("OpenAI", "agent")), ai_config=AIConfig(key="support-agent", variation="control"), ) - assert [request["method"] for request in transport.requests] == ["GET", "GET"] + assert [request["method"] for request in transport.requests] == [ + "GET", + "GET", + "GET", + ] @pytest.mark.asyncio async def test_unknown_variation_fails_before_any_records_are_created() -> None: transport = SequencedTransport( - [response(404, {"code": "not_found", "message": "not found"})] + [ + response(200, AGENT_AI_CONFIG), + response(404, {"code": "not_found", "message": "not found"}), + ] ) evals = init_evaluations(project_key="proj", api_key="token", transport=transport) @@ -3288,11 +3427,11 @@ async def test_unknown_variation_fails_before_any_records_are_created() -> None: await evals.run( key="eval-key", dataset="golden", - handler=successful_handler, + handler=tagged(successful_handler, ("OpenAI", "agent")), ai_config=AIConfig(key="support-agent", variation="missing"), ) - assert [request["method"] for request in transport.requests] == ["GET"] + assert [request["method"] for request in transport.requests] == ["GET", "GET"] @pytest.mark.asyncio @@ -3320,7 +3459,7 @@ async def test_config_source_is_validated_before_network_io( await evals.run( key="eval-key", dataset="golden", - handler=successful_handler, + handler=tagged(successful_handler, ("OpenAI", "agent")), **source, # type: ignore[arg-type] ) @@ -3351,7 +3490,7 @@ async def handler(*args: object) -> dict[str, Any]: await evals.run( key="eval-key", dataset="golden", - handler=handler, + handler=tagged(handler, ("OpenAI", "agent")), ai_config=AIConfig(key="support-agent", variation="control"), tools=[await evals.tools.get("lookup_order", implementation=lookup_order)], ) @@ -3369,7 +3508,10 @@ async def test_non_string_model_config_key_fails_loudly( model_config_key: object, ) -> None: transport = SequencedTransport( - [response(200, config_variation_page(modelConfigKey=model_config_key))] + [ + response(200, AGENT_AI_CONFIG), + response(200, config_variation_page(modelConfigKey=model_config_key)), + ] ) evals = init_evaluations(project_key="proj", api_key="token", transport=transport) @@ -3377,11 +3519,11 @@ async def test_non_string_model_config_key_fails_loudly( await evals.run( key="eval-key", dataset="golden", - handler=successful_handler, + handler=tagged(successful_handler, ("OpenAI", "agent")), ai_config=AIConfig(key="support-agent", variation="control"), ) - assert [request["method"] for request in transport.requests] == ["GET"] + assert [request["method"] for request in transport.requests] == ["GET", "GET"] @pytest.mark.asyncio @@ -3390,7 +3532,10 @@ async def test_variation_without_a_model_config_needs_an_explicit_provider( model_config_key: str | None, ) -> None: transport = SequencedTransport( - [response(200, config_variation_page(modelConfigKey=model_config_key))] + [ + response(200, AGENT_AI_CONFIG), + response(200, config_variation_page(modelConfigKey=model_config_key)), + ] ) evals = init_evaluations(project_key="proj", api_key="token", transport=transport) @@ -3399,11 +3544,11 @@ async def test_variation_without_a_model_config_needs_an_explicit_provider( await evals.run( key="eval-key", dataset="golden", - handler=successful_handler, + handler=tagged(successful_handler, ("OpenAI", "agent")), ai_config=AIConfig(key="support-agent", variation="control"), ) - assert [request["method"] for request in transport.requests] == ["GET"] + assert [request["method"] for request in transport.requests] == ["GET", "GET"] def test_ai_config_variation_from_api_layers_the_model_config() -> None: @@ -3522,9 +3667,9 @@ async def test_a_library_tool_without_a_project_is_rejected() -> None: await evals.run( key="eval-key", dataset="golden", - handler=successful_handler, + handler=tagged(successful_handler, ("OpenAI", "agent")), tools=[forged], - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) assert transport.requests == [] @@ -3571,14 +3716,14 @@ async def test_run_reads_no_tool_from_the_api() -> None: await evals.run( key="eval-key", dataset="golden", - handler=successful_handler, + handler=tagged(successful_handler), tools=[ library_tool, EvalTool( key="refund_order", implementation=refund_order, schema=ORDER_SCHEMA ), ], - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) run_paths = [path for _, path in recorded_paths(transport)[requests_before_run:]] @@ -3603,7 +3748,7 @@ async def handler(*args: object) -> dict[str, Any]: await evals.run( key="eval-key", dataset="golden", - handler=handler, + handler=tagged(handler, ("OpenAI", "agent")), ai_config=AIConfig(key="support-agent", variation="control"), tools=[], ) @@ -3634,9 +3779,9 @@ async def test_a_tool_from_another_project_is_rejected() -> None: await evals.run( key="eval-key", dataset="golden", - handler=successful_handler, + handler=tagged(successful_handler), tools=[foreign_tool], - generation={"provider": "OpenAI", "model": "gpt-4o"}, + generation={"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"}, ) @@ -3680,7 +3825,7 @@ async def echo_handler( return {"output": f"generated: {user_input}"} -INLINE_GENERATION: Any = {"provider": "OpenAI", "model": "gpt-4o"} +INLINE_GENERATION: Any = {"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"} def is_upload(request: dict[str, Any]) -> bool: @@ -3735,7 +3880,7 @@ async def handler( "metadata": {"suite": "orders"}, }, ], - handler=handler, + handler=tagged(handler), generation=INLINE_GENERATION, ) @@ -3799,7 +3944,7 @@ async def test_inline_dataset_uploads_in_batches_of_500() -> None: ).run( key="inline-eval", dataset=[{"rowIdx": index, "input": f"row {index}"} for index in range(1001)], - handler=echo_handler, + handler=tagged(echo_handler), generation=INLINE_GENERATION, ) @@ -3823,7 +3968,7 @@ async def test_inline_dataset_criterion_events_omit_dataset_id( ).run( key="inline-eval", dataset=[{"input": "hello"}], - handler=echo_handler, + handler=tagged(echo_handler), generation=INLINE_GENERATION, criteria=[Scorer(name="non-empty", fn=lambda row, output: bool(output))], ) @@ -3914,7 +4059,7 @@ async def test_malformed_inline_rows_fail_before_any_request( await evals.run( key="inline-eval", dataset=dataset, - handler=echo_handler, + handler=tagged(echo_handler), generation=INLINE_GENERATION, ) @@ -3936,7 +4081,7 @@ async def test_inline_upload_is_retried_after_a_server_error() -> None: result = await evals.run( key="inline-eval", dataset=[{"input": "hello"}], - handler=echo_handler, + handler=tagged(echo_handler), generation=INLINE_GENERATION, ) @@ -3966,7 +4111,7 @@ async def test_failed_inline_upload_stops_the_run_before_generation( ).run( key="inline-eval", dataset=[{"input": "hello"}], - handler=handler, + handler=tagged(handler), generation=INLINE_GENERATION, ) @@ -4001,7 +4146,7 @@ async def test_run_is_cancelled_when_a_later_upload_batch_fails() -> None: ).run( key="inline-eval", dataset=[{"input": f"row {index}"} for index in range(501)], - handler=handler, + handler=tagged(handler), generation=INLINE_GENERATION, ) @@ -4029,7 +4174,7 @@ async def test_failed_cancel_does_not_mask_the_upload_error( ).run( key="inline-eval", dataset=[{"input": "hello"}], - handler=echo_handler, + handler=tagged(echo_handler), generation=INLINE_GENERATION, ) @@ -4056,7 +4201,7 @@ async def test_hosted_dataset_event_identity_is_unchanged( ).run( key="inline-eval", dataset="golden", - handler=echo_handler, + handler=tagged(echo_handler), generation=INLINE_GENERATION, ) @@ -4107,7 +4252,7 @@ async def test_invalid_dataset_source_fails_before_any_request( await evals.run( key="inline-eval", dataset=dataset, - handler=echo_handler, + handler=tagged(echo_handler), generation=INLINE_GENERATION, ) @@ -4121,6 +4266,369 @@ async def test_dataset_is_required() -> None: with pytest.raises(TypeError, match="dataset"): await evals.run( # type: ignore[call-arg] key="inline-eval", - handler=echo_handler, + handler=tagged(echo_handler), generation=INLINE_GENERATION, ) + + +# Handler routing by provider and mode. + +COMPLETION: Any = {"provider": "OpenAI", "model": "gpt-4o", "mode": "completion"} + + +async def _echo(*args: Any, **kwargs: Any) -> dict[str, Any]: + return {"output": "generated"} + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("arguments", "message"), + [ + ({}, "Pass handler"), + ({"handler": tagged(_echo), "handlers": [tagged(_echo)]}, "not both"), + ({"handlers": []}, "handlers must not be empty"), + ({"handlers": tagged(_echo)}, "handlers must be a list"), + ({"handlers": [tagged(_echo), "nope"]}, r"handlers\[1\] must be callable"), + ({"handler": _echo}, "handler does not declare provides_for"), + ( + {"handlers": [tagged(_echo), tagged(_echo)]}, + r"handlers\[0\] and handlers\[1\] both provide for", + ), + ( + { + "handlers": [ + tagged(_echo, ("OpenAI", "completion")), + tagged(_echo, ("OpenAI", "messages")), + ] + }, + "both provide for", + ), + ], +) +async def test_handler_arguments_are_validated_before_network_io( + arguments: dict[str, Any], message: str +) -> None: + transport = SequencedTransport([]) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) + + with pytest.raises(EvaluationsError, match=message): + await evals.run( + key="eval-key", dataset="golden", generation=COMPLETION, **arguments + ) + + assert transport.requests == [] + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("generation", "message"), + [ + ( + {"provider": "OpenAI", "model": "gpt-4o"}, + r"generation\.mode is required\. Set it to 'completion' or 'agent'\.", + ), + ({**COMPLETION, "mode": "messages"}, "got 'messages'"), + ({**COMPLETION, "mode": "judge"}, "got 'judge'"), + ({**COMPLETION, "mode": "Completion"}, "got 'Completion'"), + ({**COMPLETION, "mode": "agnet"}, "got 'agnet'"), + ( + {**COMPLETION, "instructions": "Help."}, + "generation prompt needs 'messages' because the handler is a messages", + ), + ( + { + **COMPLETION, + "mode": "agent", + "messages": [{"role": "user", "content": "hi"}], + }, + "No handler can run generation", + ), + ( + {**COMPLETION, "provider": "Anthropic"}, + r"needs provider 'Anthropic' in 'completion' mode", + ), + ], +) +async def test_generation_mode_and_prompt_are_checked_before_network_io( + generation: dict[str, Any], message: str +) -> None: + transport = SequencedTransport([]) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) + + with pytest.raises(EvaluationsError, match=message): + await evals.run( + key="eval-key", + dataset="golden", + handler=tagged(_echo), + generation=generation, + ) + + assert transport.requests == [] + + +@pytest.mark.asyncio +async def test_an_agent_handler_rejects_a_messages_prompt() -> None: + transport = SequencedTransport([]) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) + + with pytest.raises( + EvaluationsError, + match="generation prompt needs 'instructions' because the handler is an agent", + ): + await evals.run( + key="eval-key", + dataset="golden", + handler=tagged(_echo, ("OpenAI", "agent")), + generation={ + **COMPLETION, + "mode": "agent", + "messages": [{"role": "user", "content": "hi"}], + }, + ) + + assert transport.requests == [] + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("pool", "generation", "expected"), + [ + ( + [("OpenAI", "messages"), ("Anthropic", "messages")], + {**COMPLETION, "provider": "Anthropic"}, + 1, + ), + ( + [("OpenAI", "messages"), ("OpenAI", "agent")], + {**COMPLETION, "mode": "agent"}, + 1, + ), + ([("*", "messages"), ("OpenAI", "messages")], COMPLETION, 1), + ([("*", "messages")], COMPLETION, 0), + ], +) +async def test_generation_selects_its_handler_by_provider_and_mode( + stub_sdk_client: MagicMock, + pool: list[tuple[str, str]], + generation: dict[str, Any], + expected: int, +) -> None: + transport = judge_run_transport() + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + stubs = [AsyncMock(return_value={"output": "generated"}) for _ in pool] + + await evals.run( + key="support-qa", + dataset="golden", + handlers=[tagged(stub, tag) for stub, tag in zip(stubs, pool, strict=True)], + generation=generation, + ) + + assert [stub.await_count for stub in stubs] == [ + 1 if index == expected else 0 for index in range(len(stubs)) + ] + body = evaluation_post(transport) + assert "mode" not in body + assert "generationMode" not in body + + +@pytest.mark.asyncio +async def test_judges_get_no_tools( + monkeypatch: pytest.MonkeyPatch, + stub_sdk_client: MagicMock, +) -> None: + transport = judge_run_transport() + judge_variation( + monkeypatch, + provider="OpenAI", + config={ + "messages": [ + {"role": "system", "content": "Judge {{response_to_evaluate}}"} + ], + "tools": [{"name": "lookup_order"}], + }, + ) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + calls: list[tuple[dict[str, Any], Any]] = [] + + async def handler( + config: dict[str, Any], + user_input: str | None, + tool_handlers: dict[str, Any], + variables: dict[str, Any], + ) -> dict[str, Any]: + calls.append((config, tool_handlers)) + if "Judge" in prompt_text(config): + return {"output": '{"score": 1, "reasoning": "ok"}'} + return {"output": "generated"} + + await evals.run( + key="support-qa", + dataset="golden", + handler=tagged(handler), + generation=COMPLETION, + tools=[ + EvalTool( + key="lookup_order", + implementation=lookup_order, + schema={"type": "object"}, + ) + ], + criteria=[Judge(key="$ld:ai:judge:accuracy")], + ) + + (generation_config, generation_tools), (judge_config, judge_tools) = calls + assert "lookup_order" in generation_tools + assert "lookup_order" in generation_config["tools"] + assert judge_tools == {} + assert "tools" not in judge_config + + +# AI Config generation source. + + +def ai_config_responses(mode: Any = "agent", **variation: Any) -> list[HttpResponse]: + config: dict[str, Any] = {"key": "support-agent"} + if mode is not None: + config["mode"] = mode + return [ + response(200, config), + *fetched_run_responses(config_variation_page(**variation))[1:], + ] + + +async def _run_ai_config( + transport: SequencedTransport, + handlers: list[Any], + generation: dict[str, Any] | None = None, +) -> None: + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + await evals.run( + key="eval-key", + dataset="golden", + handlers=handlers, + ai_config=AIConfig(key="support-agent", variation="control"), + generation=generation, # type: ignore[arg-type] + ) + + +@pytest.mark.asyncio +async def test_ai_config_mode_no_handler_covers_fails_after_one_get() -> None: + transport = SequencedTransport(ai_config_responses("agent")) + + with pytest.raises( + EvaluationsError, match="'agent' mode, which needs a handler in 'agent' mode" + ): + await _run_ai_config(transport, [tagged(_echo)]) + + assert len(transport.requests) == 1 + + +@pytest.mark.asyncio +@pytest.mark.parametrize("mode", ["chat", 3]) +async def test_unknown_ai_config_mode_fails_after_one_get(mode: Any) -> None: + transport = SequencedTransport(ai_config_responses(mode)) + + with pytest.raises(EvaluationsError, match="unknown mode"): + await _run_ai_config(transport, [tagged(_echo)]) + + assert len(transport.requests) == 1 + + +@pytest.mark.asyncio +@pytest.mark.parametrize("mode", [None, "completion", "judge"]) +async def test_completion_and_judge_ai_configs_use_messages_handlers( + stub_sdk_client: MagicMock, mode: Any +) -> None: + messages = [{"role": "system", "content": "Classify this."}] + transport = SequencedTransport(ai_config_responses(mode, messages=messages)) + stub = AsyncMock(return_value={"output": "generated"}) + + await _run_ai_config(transport, [tagged(stub)]) + + # The variation has both fields; messages mode reads only messages. + config = stub.await_args.args[0] + assert config["messages"] == messages + assert "instructions" not in config + + +@pytest.mark.asyncio +async def test_agent_ai_config_with_both_fields_uses_instructions( + stub_sdk_client: MagicMock, +) -> None: + transport = SequencedTransport( + ai_config_responses("agent", messages=[{"role": "user", "content": "x"}]) + ) + stub = AsyncMock(return_value={"output": "generated"}) + + await _run_ai_config(transport, [tagged(stub, ("OpenAI", "agent"))]) + + config = stub.await_args.args[0] + assert config["instructions"] == "You are a support agent." + assert "messages" not in config + + +@pytest.mark.asyncio +async def test_completion_ai_config_with_only_instructions_fails_after_two_gets() -> ( + None +): + transport = SequencedTransport(ai_config_responses("completion")) + + with pytest.raises( + EvaluationsError, + match=( + "AI Config 'support-agent' variation 'control' is in 'completion' " + "mode: the prompt needs 'messages'" + ), + ): + await _run_ai_config(transport, [tagged(_echo)]) + + assert len(transport.requests) == 2 + + +@pytest.mark.asyncio +async def test_a_messages_override_makes_a_completion_variation_valid( + stub_sdk_client: MagicMock, +) -> None: + transport = SequencedTransport(ai_config_responses("completion")) + stub = AsyncMock(return_value={"output": "generated"}) + override = [{"role": "system", "content": "Override"}] + + await _run_ai_config(transport, [tagged(stub)], {"messages": override}) + + config = stub.await_args.args[0] + assert config["messages"] == override + assert "instructions" not in config + + +@pytest.mark.asyncio +async def test_ai_config_provider_no_handler_covers_fails_after_the_model_config() -> ( + None +): + transport = SequencedTransport(ai_config_responses("agent")) + + with pytest.raises(EvaluationsError, match="needs provider 'OpenAI' in 'agent'"): + await _run_ai_config(transport, [tagged(_echo, ("Anthropic", "agent"))]) + + assert [request["method"] for request in transport.requests] == [ + "GET", + "GET", + "GET", + ] + assert "/model-configs/" in transport.requests[2]["url"] + + +@pytest.mark.asyncio +async def test_overrides_cannot_set_the_mode() -> None: + transport = SequencedTransport([]) + + with pytest.raises(EvaluationsError, match="the AI Config supplies the mode"): + await _run_ai_config(transport, [tagged(_echo)], {"mode": "agent"}) + + assert transport.requests == []