From df02eec64420c65768f5775f04fdde32505d694e Mon Sep 17 00:00:00 2001 From: Dylan O'Neill Date: Mon, 28 Sep 2026 15:39:26 -0700 Subject: [PATCH 1/3] feat(evaluations)!: take tools as a list of Tool Add Tool. Construct one to define a tool in code. A constructed Tool is always inline, because source and version are not constructor arguments. Add evals.tools.get(key, implementation=...). It reads the library tool now, pins the version now, and raises now when the tool is absent. It is the only way to make a library tool. run() reads no tool from the API. run(tools=...) now takes a list of Tool. Each Tool carries its own key. Validate the list before any network I/O. Reject a non-Tool entry, a blank or uppercase key, a non-object or non-serializable schema, a non-callable implementation, a repeated key, and a NativeTool on an inline tool. Compare keys without case. Move project_key to init_evaluations(). run() no longer takes it. Read LD_PROJECT_KEY when the argument is absent. Rename the api_token argument to api_key, in init_evaluations() and in LDApiClient. The LD_API_TOKEN variable name does not change. Give tools their own module. evaluations/tools.py owns Tool, the validation, the projections to a handler config and to a create body, and ToolsClient. ToolsClient reads the library itself, so the runner does not. Move segment(), require_mapping(), and require_string() to api.py. The create body and the event payloads do not change. Spec: launchdarkly/ai-sdks-monorepo#24 Co-Authored-By: Claude Opus 5 --- packages/ai/README.md | 7 +- packages/client/README.md | 57 +- packages/client/agents.md | 2 +- .../src/launchdarkly_ai_server/__init__.py | 2 + .../evaluations/__init__.py | 3 + .../launchdarkly_ai_server/evaluations/api.py | 32 +- .../evaluations/module.py | 142 ++-- .../evaluations/runner.py | 145 +--- .../evaluations/tools.py | 216 +++++ .../evaluations/types.py | 9 - packages/client/tests/test_evaluations.py | 65 +- packages/client/tests/test_evaluations_run.py | 754 +++++++++++++++--- 12 files changed, 1107 insertions(+), 327 deletions(-) create mode 100644 packages/client/src/launchdarkly_ai_server/evaluations/tools.py diff --git a/packages/ai/README.md b/packages/ai/README.md index 829bb567..f99a65b9 100644 --- a/packages/ai/README.md +++ b/packages/ai/README.md @@ -54,14 +54,13 @@ Never raises. Returns `{"enabled": bool, "config": dict | None, "meta": dict | N ## Evaluations from code -`init_evaluations`, the criterion types, and the evaluations result types are all re-exported: +`init_evaluations`, the criterion types, `Tool`, and the evaluations result types are all re-exported: ```python from launchdarkly_ai_python import Judge, Scorer, init_evaluations -evals = init_evaluations() +evals = init_evaluations(project_key="my-project") result = await evals.run( - project_key="my-project", key="unique-evaluation-key", dataset="golden-dataset", handler=my_handler, @@ -73,7 +72,7 @@ result = await evals.run( ) ``` -`LD_API_TOKEN` is required. Configure `LD_SDK_KEY` — or initialize your own client with `init_client(client=...)` — to emit one `$ld:ai:offline-evals:generation` event per generated row, plus one `$ld:ai:offline-evals:criterion` event per `(row, criterion)` when `criteria` are supplied, through the standard SDK event transport. The SDK reports scores; LaunchDarkly rules on them at ingest. A judge served by a different provider than `generation` needs a handler for it in `judge_handlers`. Each row's tool calls are recorded during generation and rendered into the judge's `{{message_history}}`, between the row input and the generated output, so a rubric can grade the tool trajectory as well as the final answer. Use `LD_API_BASE_URI` for staging or local management API traffic; it is separate from the SDK delivery setting `LD_BASE_URI`. Evaluation-run links use the explicit `ui_base_uri` option or `LD_UI_BASE_URI`, defaulting to `https://app.launchdarkly.com`; set it when the project is not in production, or a run created elsewhere still links to the production app. See the [core evaluations guide](../client/README.md#run-an-evaluation-from-code). +`LD_API_TOKEN` is required. `LD_SDK_KEY` is also required, because the harness needs an initialized client to send your results to LaunchDarkly. Supply your own client with `init_client(client=...)` if you prefer. The SDK reports a score for each row and criterion, and LaunchDarkly decides whether each one passes. A judge served by a different provider than `generation` needs a handler for it in `judge_handlers`. Each row's tool calls go into the judge's `{{message_history}}`, between the row input and the generated output, so a rubric can grade the tool calls and the final answer. `tools` is a list of `Tool`. Construct one to define a tool in code, or call `evals.tools.get()` for a tool that already exists in LaunchDarkly. See the [core evaluations guide](../client/README.md#run-an-evaluation-from-code). --- diff --git a/packages/client/README.md b/packages/client/README.md index b2607d4e..de974ce3 100644 --- a/packages/client/README.md +++ b/packages/client/README.md @@ -165,40 +165,47 @@ A tool result is now judge-prompt input. It stays literal for the same reason th Judges are resolved through flag delivery, and handlers are matched to them, **before** any evaluation records are created — a missing judge or one no handler covers fails the run up front rather than after the generation spend. After that point a criterion failure never aborts the run: an unparseable judge response, an out-of-range score, a raising handler or scorer, and a row whose generation errored each become a per-criterion `ERROR` event with a cause code (`invalid_judge_output`, `invalid_score`, `handler_raised`, `scorer_raised`, `generation_incomplete`) and a top-level `errorMessage`. Event *delivery* is different: the backend needs one result per `(row, criterion)` to finish row accounting, so if tracking a criterion event fails, every remaining result is still attempted and flushed and then `run()` raises — rather than polling to its timeout with the cause hidden. -The client uses **lazy initialization**: importing the package does not connect to LaunchDarkly. The singleton is created automatically on the first API call that needs it (`config().invoke()`, `graph().invoke()`, `resolve_graph()`, etc.), as long as `LD_SDK_KEY` is set in the environment. +### Give the evaluation tools -Call `init_client()` explicitly when you want to: -- Pass SDK or telemetry options programmatically (overriding env vars) -- Initialize at startup before the first AI call (e.g. to avoid latency on the first request) -- Fail fast at boot if `LD_SDK_KEY` is missing +Pass `tools` to `run()` as a list of `Tool`. Construct one to define a tool in code. Call `evals.tools.get()` to use a tool that already exists in your project's AI library, which reads the tool and pins its version at that point. One list can hold both kinds. ```python -import asyncio -from launchdarkly_ai_server import init_client, shutdown +from launchdarkly_ai_server import Tool, init_evaluations -async def main(): - # Standard path — auto-discovers launchdarkly-server-sdk. - client = await init_client({ - "sdkKey": "sdk-...", - "serviceName": "my-service", - "environment": "production", - }) - # Or skip init_client() and let the first model/graph call initialize lazily. +def lookup_order(order_id: str) -> str: + return f"order {order_id} shipped" - # Flush telemetry, flush LD events, and close the client. - await shutdown() -asyncio.run(main()) +evals = init_evaluations(project_key="my-project") + +# Read from the AI library now. Pins the version. Raises now if the tool is absent. +search_docs_tool = evals.tools.get("search_docs", implementation=search_docs) + +# Defined here. Needs no tool in LaunchDarkly. +lookup_order_tool = Tool( + key="lookup_order", + implementation=lookup_order, + schema={ + "type": "object", + "properties": {"order_id": {"type": "string"}}, + "required": ["order_id"], + }, + description="Look up an order by id", +) + +result = await evals.run( + key="support-qa-2026-08-20", + dataset="support-golden", + handler=create_openai_messages_handler(), + generation={"provider": "OpenAI", "model": "gpt-4o"}, + tools=[search_docs_tool, lookup_order_tool], +) ``` -| Export | Description | -|---|---| -| `init_client(options?)` | Auto-discover and initialize `launchdarkly-server-sdk`. Optional — the first AI API call triggers lazy init when `LD_SDK_KEY` is set. Returns `Awaitable[LDClientInterface]`. | -| `init_client(client=...)` | **BYOC overload** — accept a pre-initialized `LDClientInterface`. Skips SDK auto-discovery. | -| `get_client()` | Return the initialized `LDClientInterface`. Raises if `init_client` has not completed. | -| `shutdown()` | Flush all events and telemetry, then close the client. Await before process exit. | -| `inspect_config(key, context)` | Read an AI Config variation without invoking the model. Never raises. Returns `{"enabled", "config", "meta"}`. | +`run()` reads no tool from the API. A constructed `Tool` is always inline, because `source` and `version` are not constructor arguments. Only `tools.get()` produces a library tool. Handlers receive the same `{key: callable}` map whichever kind a tool is, so handler code needs no change. + +**The list is checked before any network I/O.** A blank key, an uppercase key, a `schema` that is not a JSON object, a schema that is not JSON-serializable (including a `NaN` or `Infinity` value), and a non-callable implementation each fail with zero requests issued. A repeated key fails too, and keys are compared without case. A `NativeTool` is valid only for a library tool, because the provider supplies its schema. ### `config(**args)` diff --git a/packages/client/agents.md b/packages/client/agents.md index cb117045..e374c60d 100644 --- a/packages/client/agents.md +++ b/packages/client/agents.md @@ -131,7 +131,7 @@ Handlers may return any of these — the client normalizes them before emitting `init_evaluations()` creates an evaluations harness using `LD_API_TOKEN` and the management API host `LD_API_BASE_URI`. Do not reuse `LD_BASE_URI`: that variable configures SDK delivery and may point at a relay proxy. Evaluation-run links use the separate `ui_base_uri` option, then `LD_UI_BASE_URI`, then `https://app.launchdarkly.com`; do not derive their host from `LD_API_BASE_URI`. An event transport is resolved in `init_evaluations()`, which raises before any network I/O when it finds neither an SDK key (`sdk_key` or `LD_SDK_KEY`) nor an already-initialized event-capable client: generation events are the only ingest path for row results, so a run without a transport could never complete. The lifecycle module's bring-your-own-client path (`init_client(client=...)`) therefore satisfies the check on its own, and `run()` reuses that singleton through `_resolve_client`; `run()` raises if the client disappears before it emits. Both polling arguments reject NaN, which would otherwise never compare past a deadline and hang the run. The harness always queues one `$ld:ai:offline-evals:generation` custom event per row through the standard SDK event transport and flushes before returning. No feature flag gates event emission. The harness polls the run summary endpoint until a nonzero `total_rows` has `pending_rows == 0` and `passed + failed + error` rows accounting for the total, polling every `poll_interval_seconds` (default 2s) until `poll_timeout_seconds` (default 180s); both are `run()` arguments so large datasets can widen them. The summary endpoint does not return run state, so `RunSummary` exposes row counts only. -`await EvaluationsModule.run(...)` takes `project_key` per call. Dataset lookup/row pagination, evaluation creation, and run creation are private helpers; only `run()` is public. Each call creates a new evaluation with `POST` and a run with `source="api"`, so its key must be unique. The harness directly invokes the supplied handler once per row and never retries it — event delivery is never a reason to rerun a handler because that would repeat tool side effects; retries apply only to management API requests. A 429 is replayed for any method, but 5xx responses and transport failures are replayed only for `GET`/`HEAD`, so an evaluation or run `POST` that the server may already have applied is never duplicated. Management API calls run in a worker thread (`asyncio.to_thread`) because the client is synchronous; the caller's event loop stays free. Generation events go through the already-initialized SDK client when the application has one — `init_client` is idempotent, so an existing singleton wins and the evaluations SDK key is ignored with a warning. Dataset-owned `input`, `expected_output`, `metadata`, and `variables` are deliberately excluded from the event payload. The harness flushes events, polls the run summary endpoint until row accounting is complete (`total_rows > 0`, `pending_rows == 0`, and `passed + failed + error == total_rows`), and raises a timeout once `poll_timeout_seconds` elapses if the backend never reaches one. `RunSummary` includes row counts only, and `EvalRunResult.passed` is true only when error and pending row counts are both zero. +`init_evaluations()` takes `project_key`; `run()` does not. Dataset lookup/row pagination, evaluation creation, and run creation are private helpers; `run()` and `tools.get()` are the public surface. `tools` is a `list[Tool]`. `evals.tools.get(key, implementation=...)` issues `GET projects/

/ai-tools/`, pins the returned version, and builds a library tool through `Tool._library`; a constructed `Tool` is always inline because `source` and `version` are `init=False`. `run()` issues no tool request at all: a library tool was already read by `tools.get()`. Library tools are sent as `{key, version, source: "library"}` and inline tools as `{key, schema, description, source: "inline"}`. `_validate_tools` runs in `run()` before any I/O, so a bad list — a non-`Tool` entry, a blank or uppercase key, a non-object or non-serializable `schema` (`allow_nan=False`), a non-callable implementation, a repeated key (compared without case), or a `NativeTool` on an inline tool — fails with zero requests recorded. `_validate_tool_key` is shared by `tools.get()` and `_validate_tools`, so the key rules are identical on both paths. Handler config synthesis is unforked: `config["tools"][tool.key] = {"description", "parameters"}` is fed from whichever kind the tool is, so handlers cannot tell them apart, and `_tool_handlers` maps each tool to its executable. Each call creates a new evaluation with `POST` and a run with `source="api"`, so its key must be unique. The harness directly invokes the supplied handler once per row and never retries it — event delivery is never a reason to rerun a handler because that would repeat tool side effects; retries apply only to management API requests. A 429 is replayed for any method, but 5xx responses and transport failures are replayed only for `GET`/`HEAD`, so an evaluation or run `POST` that the server may already have applied is never duplicated. Management API calls run in a worker thread (`asyncio.to_thread`) because the client is synchronous; the caller's event loop stays free. Generation events go through the already-initialized SDK client when the application has one — `init_client` is idempotent, so an existing singleton wins and the evaluations SDK key is ignored with a warning. Dataset-owned `input`, `expected_output`, `metadata`, and `variables` are deliberately excluded from the event payload. The harness flushes events, polls the run summary endpoint until row accounting is complete (`total_rows > 0`, `pending_rows == 0`, and `passed + failed + error == total_rows`), and raises a timeout once `poll_timeout_seconds` elapses if the backend never reaches one. `RunSummary` includes row counts only, and `EvalRunResult.passed` is true only when error and pending row counts are both zero. --- diff --git a/packages/client/src/launchdarkly_ai_server/__init__.py b/packages/client/src/launchdarkly_ai_server/__init__.py index a5e74f1d..26343578 100644 --- a/packages/client/src/launchdarkly_ai_server/__init__.py +++ b/packages/client/src/launchdarkly_ai_server/__init__.py @@ -36,6 +36,7 @@ Judge, RunSummary, Scorer, + Tool, init_evaluations, ) from .graph import GraphInstance, graph, resolve_graph @@ -194,6 +195,7 @@ "EvaluationsError", "EvaluationsModule", "GenerationConfig", + "Tool", "Judge", "RunSummary", "Scorer", diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/__init__.py b/packages/client/src/launchdarkly_ai_server/evaluations/__init__.py index 9cd942d9..8641b849 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/__init__.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/__init__.py @@ -11,6 +11,7 @@ ) from .criteria import Criterion, Judge, Scorer, SuccessDirection from .module import EvaluationsModule, init_evaluations +from .tools import Tool, ToolsClient from .types import ( AIConfig, DatasetRow, @@ -36,6 +37,8 @@ "RunSummary", "Scorer", "SuccessDirection", + "Tool", + "ToolsClient", "Transport", "Usage", "init_evaluations", diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/api.py b/packages/client/src/launchdarkly_ai_server/evaluations/api.py index 3a46d23d..96328b65 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/api.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/api.py @@ -6,7 +6,7 @@ import urllib.error import urllib.parse import urllib.request -from collections.abc import Callable +from collections.abc import Callable, Mapping from dataclasses import dataclass, field from datetime import UTC, datetime from email.utils import parsedate_to_datetime @@ -84,7 +84,7 @@ class LDApiClient: def __init__( self, - api_token: str, + api_key: str, base_uri: str = DEFAULT_BASE_URI, transport: Transport = urllib_transport, timeout: float = 30.0, @@ -92,7 +92,7 @@ def __init__( sleep: Callable[[float], None] = time.sleep, random_value: Callable[[], float] = random.random, ) -> None: - self.api_token = api_token + self.api_key = api_key self.base_uri = base_uri.rstrip("/") self._transport = transport self._timeout = timeout @@ -135,7 +135,7 @@ def request( params: dict[str, Any] | None = None, ) -> Any: headers = { - "Authorization": self.api_token, + "Authorization": self.api_key, "Accept": "application/json", "LD-API-Version": "20240415", "User-Agent": "launchdarkly-ai-evaluations-python", @@ -192,3 +192,27 @@ def get(self, path: str, params: dict[str, Any] | None = None) -> Any: def post(self, path: str, body: Any = None) -> Any: return self.request("POST", path, body=body) + + +def segment(value: str) -> str: + """Percent-encode one path segment.""" + return urllib.parse.quote(value, safe="") + + +def require_mapping(value: Any, *, description: str) -> Mapping[str, Any]: + """Return ``value`` as a mapping. Raises ``EvaluationsError``.""" + if not isinstance(value, Mapping): + raise EvaluationsError( + f"LaunchDarkly returned an invalid {description} response" + ) + return value + + +def require_string(data: Mapping[str, Any], key: str, description: str) -> str: + """Return the non-empty string at ``key``. Raises ``EvaluationsError``.""" + value = data.get(key) + if not isinstance(value, str) or not value: + raise EvaluationsError( + f"LaunchDarkly {description} response is missing string field {key!r}" + ) + return value diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/module.py b/packages/client/src/launchdarkly_ai_server/evaluations/module.py index 4cc684aa..92cfa3a0 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/module.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/module.py @@ -6,7 +6,7 @@ import math import os import time -from collections.abc import Mapping +from collections.abc import Mapping, Sequence from typing import Any, cast from ..lifecycle import get_client, init_client @@ -15,16 +15,12 @@ EvaluationsError, LDApiClient, Transport, + segment, urllib_transport, ) from .criteria import Criterion, Judge -from .runner import ( - EvalHandler, - EvaluationsRunner, - ToolImplementation, - _provides_for, - _segment, -) +from .runner import EvalHandler, EvaluationsRunner, _provides_for +from .tools import Tool, ToolsClient, tool_handlers, validate_tools from .types import AIConfig, EvalRunResult, GenerationConfig, RunSummary logger = logging.getLogger(__name__) @@ -98,18 +94,31 @@ class EvaluationsModule: def __init__( self, api_client: LDApiClient, + project_key: str, sdk_key: str | None, ui_base_uri: str = DEFAULT_UI_BASE_URI, ) -> None: self._api = api_client + self._project_key = project_key self._sdk_key = sdk_key self._ui_base_uri = ui_base_uri.rstrip("/") self._runner = EvaluationsRunner(api_client) + self._tools = ToolsClient(api_client, project_key) @property def api(self) -> LDApiClient: return self._api + @property + def project_key(self) -> str: + """Project that holds this module's evaluations, tools, and datasets.""" + return self._project_key + + @property + def tools(self) -> ToolsClient: + """Reader for tools in the LaunchDarkly tool library.""" + return self._tools + @property def sdk_key(self) -> str | None: """SDK key whose event transport carries generation results to LaunchDarkly.""" @@ -123,13 +132,12 @@ def ui_base_uri(self) -> str: async def run( self, *, - project_key: str, key: str, dataset: str, handler: EvalHandler, generation: GenerationConfig | None = None, ai_config: AIConfig | None = None, - tools: Mapping[str, ToolImplementation] | None = None, + tools: Sequence[Tool] | None = None, criteria: list[Criterion] | None = None, judge_handlers: list[EvalHandler] | None = None, concurrency: int = 10, @@ -145,6 +153,13 @@ async def run( generated row, and one evaluation event is emitted per ``(row, criterion)`` result. + ``tools`` is a list of :class:`Tool`. Construct one to define a tool + in code. Call ``evals.tools.get(key, implementation=...)`` to use a + tool from the LaunchDarkly tool library, which reads the tool and pins + its version at that point. One list may hold both kinds. ``run`` reads + no tool from the API, and it checks the list before any network I/O. + Handlers receive a ``{key: executable}`` map either way. + A :class:`Judge` is an independent AI Config and may be served by a different provider or mode than ``generation``. ``handler`` runs a judge only when it provides for that judge's provider; pass handlers for any @@ -171,7 +186,6 @@ async def run( if poll_timeout_seconds is None: poll_timeout_seconds = SUMMARY_POLL_TIMEOUT_SECONDS self._validate_run_args( - project_key=project_key, key=key, dataset=dataset, handler=handler, @@ -180,25 +194,32 @@ async def run( poll_timeout_seconds=poll_timeout_seconds, ) self._validate_config_source(generation=generation, ai_config=ai_config) + run_tools = list(tools or []) + validate_tools(run_tools) + run_tool_handlers = tool_handlers(run_tools) pinned_tool_versions: dict[str, int] = {} config_label = "" if ai_config is not None: config_label = f"{ai_config.key!r}/{ai_config.variation!r}" ai_config_variation = await asyncio.to_thread( self._runner._fetch_config_variation, - project_key, + self._project_key, ai_config.key, ai_config.variation, ) generation = _merge_generation(ai_config_variation.generation, generation) - if tools is None and ai_config_variation.tool_versions: + supplied = {tool.key for tool in run_tools} + missing = [ + name + for name in ai_config_variation.tool_versions + if name not in supplied + ] + if missing: raise EvaluationsError( f"AI Config variation {config_label} uses tools " "with no implementation: " - + ", ".join( - repr(name) for name in ai_config_variation.tool_versions - ) - + ". Pass tools= with an implementation for each." + + ", ".join(repr(name) for name in missing) + + ". Pass tools= with a Tool for each." ) pinned_tool_versions = ai_config_variation.tool_versions if criteria is None: @@ -206,7 +227,6 @@ async def run( Judge(key=judge_key) for judge_key in ai_config_variation.judge_keys ] generation = self._validate_generation(generation) - run_tools = dict(tools or {}) run_criteria = list(criteria or []) run_judge_handlers = list(judge_handlers or []) self._validate_criteria(run_criteria) @@ -218,58 +238,57 @@ async def run( # The management API client is synchronous; running it in a worker thread # keeps the caller's event loop free. - # Tool/judge verification is deliberately first: a typo must not create records. - resolved_tools = await asyncio.to_thread( - self._runner._resolve_tools, project_key, run_tools - ) - # The tool API serves only the latest version, so a variation pinned to - # an older one is evaluated against the current schema. + # Judge verification is first: a typo must not create records. + # A variation pins a tool version. Compare it with the version the run + # uses, which tools.get() already read. + tools_by_key = {tool.key: tool for tool in run_tools} for tool_key, pinned_version in pinned_tool_versions.items(): - resolved_tool = resolved_tools.get(tool_key) - if resolved_tool is not None and resolved_tool.version != pinned_version: - logger.warning( - "AI Config variation %s pins tool %r at version %d; " - "evaluating against the latest version %d.", - config_label, - tool_key, - pinned_version, - resolved_tool.version, - ) + tool = tools_by_key.get(tool_key) + if tool is None or tool.version is None or tool.version == pinned_version: + continue + logger.warning( + "AI Config variation %s pins tool %r at version %d; " + "the run uses version %d.", + config_label, + tool_key, + pinned_version, + tool.version, + ) resolved_judges = await self._runner._resolve_judges( - project_key, ld_judges, handler, run_judge_handlers + self._project_key, ld_judges, handler, run_judge_handlers ) dataset_ref = await asyncio.to_thread( - self._runner._fetch_dataset, project_key, dataset + self._runner._fetch_dataset, self._project_key, dataset ) rows = await asyncio.to_thread( - self._runner._get_dataset_rows, project_key, dataset + self._runner._get_dataset_rows, self._project_key, dataset ) evaluation = await asyncio.to_thread( self._runner._create_evaluation, - project_key, + self._project_key, key, generation, - resolved_tools, + run_tools, run_criteria, ) evaluation_run = await asyncio.to_thread( self._runner._create_evaluation_run, - project_key, + self._project_key, evaluation.id, dataset_ref.id, ) - config = self._runner._build_handler_config(generation, resolved_tools) + config = self._runner._build_handler_config(generation, run_tools) results = await self._runner._run_rows( rows, handler, config, - run_tools, + run_tool_handlers, concurrency, ) try: self._runner._emit_generation_events( client, - project_key=project_key, + project_key=self._project_key, evaluation=evaluation, evaluation_run=evaluation_run, dataset=dataset_ref, @@ -278,14 +297,14 @@ async def run( if run_criteria: criterion_results = await self._runner._run_criteria_for_results( results, - run_tools, + run_tool_handlers, run_criteria, resolved_judges, concurrency, ) self._runner._emit_evaluation_events( client, - project_key=project_key, + project_key=self._project_key, evaluation=evaluation, evaluation_run=evaluation_run, dataset=dataset_ref, @@ -298,15 +317,14 @@ async def run( if inspect.isawaitable(flush_result): await flush_result summary = await self._poll_summary_until_terminal( - project_key, evaluation.id, evaluation_run.id, poll_interval_seconds, poll_timeout_seconds, ) url = ( - f"{self._ui_base_uri}/projects/{_segment(project_key)}/ai/evaluations/" - f"{_segment(evaluation.id)}/runs/{_segment(evaluation_run.id)}" + f"{self._ui_base_uri}/projects/{segment(self._project_key)}/ai/evaluations/" + f"{segment(evaluation.id)}/runs/{segment(evaluation_run.id)}" ) return EvalRunResult( # failed_rows counts rows whose criteria were scored and did not @@ -326,7 +344,6 @@ async def run( async def _poll_summary_until_terminal( self, - project_key: str, evaluation_id: str, run_id: str, poll_interval_seconds: float, @@ -336,7 +353,7 @@ async def _poll_summary_until_terminal( last_summary = None while True: last_summary = await asyncio.to_thread( - self._runner._get_summary, project_key, evaluation_id, run_id + self._runner._get_summary, self._project_key, evaluation_id, run_id ) if _is_terminal_summary(last_summary): return last_summary @@ -429,7 +446,6 @@ def _validate_judge_handlers(judge_handlers: list[EvalHandler]) -> None: @staticmethod def _validate_run_args( *, - project_key: str, key: str, dataset: str, handler: EvalHandler, @@ -438,7 +454,6 @@ def _validate_run_args( poll_timeout_seconds: float, ) -> None: for name, value in ( - ("project_key", project_key), ("key", key), ("dataset", dataset), ): @@ -501,18 +516,30 @@ def _validate_generation(generation: GenerationConfig | None) -> GenerationConfi def init_evaluations( - api_token: str | None = None, + project_key: str | None = None, + api_key: str | None = None, sdk_key: str | None = None, base_uri: str | None = None, ui_base_uri: str | None = None, transport: Transport = urllib_transport, ) -> EvaluationsModule: - """Resolve credentials and construct the evaluations module.""" - token = api_token or _env("LD_API_TOKEN") + """Resolve credentials and construct the evaluations module. + + ``project_key`` names the project that holds the evaluations, the tools, + and the datasets this module uses. + """ + resolved_project_key = (project_key or "").strip() or _env("LD_PROJECT_KEY") + if not resolved_project_key: + raise EvaluationsError( + "No LaunchDarkly project key provided. Set the LD_PROJECT_KEY " + "environment variable or pass project_key to init_evaluations()." + ) + + token = api_key or _env("LD_API_TOKEN") if not token: raise EvaluationsError( "No LaunchDarkly API access token provided. Set the LD_API_TOKEN " - "environment variable or pass api_token to init_evaluations()." + "environment variable or pass api_key to init_evaluations()." ) resolved_sdk_key = sdk_key or _env("LD_SDK_KEY") @@ -529,12 +556,13 @@ def init_evaluations( ) api_client = LDApiClient( - api_token=token, + api_key=token, base_uri=base_uri or _env("LD_API_BASE_URI") or DEFAULT_BASE_URI, transport=transport, ) return EvaluationsModule( api_client=api_client, + project_key=resolved_project_key, sdk_key=resolved_sdk_key, ui_base_uri=ui_base_uri or _env("LD_UI_BASE_URI") or DEFAULT_UI_BASE_URI, ) diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/runner.py b/packages/client/src/launchdarkly_ai_server/evaluations/runner.py index 6240e980..9a4dd91d 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/runner.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/runner.py @@ -6,8 +6,7 @@ import json import logging import time -import urllib.parse -from collections.abc import Awaitable, Callable, Mapping +from collections.abc import Awaitable, Callable, Mapping, Sequence from dataclasses import dataclass from datetime import UTC, datetime from typing import Any, Literal @@ -24,7 +23,6 @@ render_row_trajectory, row_fields, ) -from ..types import NativeTool from ..utils import ( collapse_messages_to_instructions, normalize_mode, @@ -32,7 +30,14 @@ parse_usage, to_ld_context, ) -from .api import EvaluationsError, LDApiClient, LDApiError +from .api import ( + EvaluationsError, + LDApiClient, + LDApiError, + require_mapping, + require_string, + segment, +) from .criteria import Criterion, Judge, Scorer from .events import ( CriterionEventPayload, @@ -41,6 +46,12 @@ LDJudgeCriterionEventPayload, TokenUsage, ) +from .tools import ( + Tool, + ToolImplementation, + create_wire_tools, + handler_config_tools, +) from .types import ( AIConfigVariation, DatasetRef, @@ -49,7 +60,6 @@ EvaluationRunRef, GenerationConfig, ResolvedJudge, - ResolvedTool, RunSummary, ) @@ -60,7 +70,6 @@ CRITERION_EVENT_NAME = "$ld:ai:offline-evals:criterion" EvalHandler = Callable[..., Awaitable[dict[str, Any]]] -ToolImplementation = Callable[..., Any] | NativeTool @dataclass(frozen=True) @@ -164,27 +173,6 @@ def _select_judge_handler( return None -def _segment(value: str) -> str: - return urllib.parse.quote(value, safe="") - - -def _mapping(value: Any, *, description: str) -> Mapping[str, Any]: - if not isinstance(value, Mapping): - raise EvaluationsError( - f"LaunchDarkly returned an invalid {description} response" - ) - return value - - -def _required_string(data: Mapping[str, Any], key: str, description: str) -> str: - value = data.get(key) - if not isinstance(value, str) or not value: - raise EvaluationsError( - f"LaunchDarkly {description} response is missing string field {key!r}" - ) - return value - - class ConcurrencyController: """Owns row-worker permits.""" @@ -237,11 +225,11 @@ def _fetch_config_variation( """ description = f"AI Config variation {config_key!r}/{variation_key!r}" path = ( - f"projects/{_segment(project_key)}/ai-configs/{_segment(config_key)}" - f"/variations/{_segment(variation_key)}" + f"projects/{segment(project_key)}/ai-configs/{segment(config_key)}" + f"/variations/{segment(variation_key)}" ) try: - raw = _mapping(self._api.get(path), description=description) + raw = require_mapping(self._api.get(path), description=description) except LDApiError as error: if error.status == 404: raise EvaluationsError( @@ -282,13 +270,13 @@ def _fetch_model_config( version: Any, ) -> Mapping[str, Any]: path = ( - f"projects/{_segment(project_key)}/ai-configs/model-configs/" - f"{_segment(model_config_key)}" + f"projects/{segment(project_key)}/ai-configs/model-configs/" + f"{segment(model_config_key)}" ) # A pinned variation names the model-config version it was built against. params = {"version": version} if isinstance(version, int) else None try: - return _mapping( + return require_mapping( self._api.get(path, params=params), description=f"model config {model_config_key!r}", ) @@ -300,44 +288,6 @@ def _fetch_model_config( ) from error raise - def _resolve_tools( - self, - project_key: str, - tools: Mapping[str, ToolImplementation], - ) -> dict[str, ResolvedTool]: - resolved: dict[str, ResolvedTool] = {} - for key, implementation in tools.items(): - if not callable(implementation) and not isinstance( - implementation, NativeTool - ): - raise EvaluationsError( - f"Tool {key!r} must be callable or a NativeTool instance" - ) - path = f"projects/{_segment(project_key)}/ai-tools/{_segment(key)}" - try: - raw = _mapping(self._api.get(path), description=f"tool {key!r}") - except LDApiError as error: - if error.status == 404: - raise EvaluationsError( - f"LaunchDarkly AI tool {key!r} was not found in project {project_key!r}" - ) from error - raise - version = raw.get("version") - if not isinstance(version, int): - raise EvaluationsError( - f"LaunchDarkly AI tool {key!r} has no integer version" - ) - schema = raw.get("schema") - if not isinstance(schema, Mapping): - schema = {} - resolved[key] = ResolvedTool( - key=key, - version=version, - description=str(raw.get("description") or ""), - schema=dict(schema), - ) - return resolved - async def _resolve_judges( self, project_key: str, @@ -407,16 +357,18 @@ async def _resolve_judges( return resolved def _fetch_dataset(self, project_key: str, dataset_key: str) -> DatasetRef: - path = f"projects/{_segment(project_key)}/datasets/{_segment(dataset_key)}" + path = f"projects/{segment(project_key)}/datasets/{segment(dataset_key)}" try: - raw = _mapping(self._api.get(path), description=f"dataset {dataset_key!r}") + raw = require_mapping( + self._api.get(path), description=f"dataset {dataset_key!r}" + ) except LDApiError as error: if error.status == 404: raise EvaluationsError( f"LaunchDarkly dataset {dataset_key!r} was not found in project {project_key!r}" ) from error raise - dataset_id = _required_string(raw, "id", "dataset") + dataset_id = require_string(raw, "id", "dataset") response_key = raw.get("key", raw.get("name", dataset_key)) return DatasetRef(id=dataset_id, key=str(response_key)) @@ -427,8 +379,8 @@ def _fetch_dataset_rows_page( *, offset: int, ) -> Mapping[str, Any]: - path = f"projects/{_segment(project_key)}/datasets/{_segment(dataset_key)}/rows" - return _mapping( + path = f"projects/{segment(project_key)}/datasets/{segment(dataset_key)}/rows" + return require_mapping( self._api.get( path, params={ @@ -458,7 +410,7 @@ def _get_dataset_rows(self, project_key: str, dataset_key: str) -> list[DatasetR if not items: break for item_value in items: - item = _mapping(item_value, description="dataset row") + item = require_mapping(item_value, description="dataset row") row_index = item.get("rowIndex") if not isinstance(row_index, int): raise EvaluationsError( @@ -512,7 +464,7 @@ def _create_evaluation( project_key: str, key: str, generation: GenerationConfig, - tools: Mapping[str, ResolvedTool], + tools: Sequence[Tool], criteria: list[Criterion] | None = None, ) -> EvaluationRef: body: dict[str, Any] = { @@ -533,15 +485,13 @@ def _create_evaluation( if "prompt_snippets" in generation: body["promptSnippets"] = generation["prompt_snippets"] if tools: - body["tools"] = [ - {"key": tool.key, "version": tool.version} for tool in tools.values() - ] + body["tools"] = create_wire_tools(tools) if criteria: body["criteria"] = [criterion.to_criteria_wire() for criterion in criteria] - path = f"projects/{_segment(project_key)}/evaluations" - raw = _mapping(self._api.post(path, body=body), description="evaluation") - evaluation_id = _required_string(raw, "id", "evaluation") + path = f"projects/{segment(project_key)}/evaluations" + raw = require_mapping(self._api.post(path, body=body), description="evaluation") + evaluation_id = require_string(raw, "id", "evaluation") response_key = raw.get("name", raw.get("label", key)) version = raw.get("version") return EvaluationRef( @@ -557,14 +507,13 @@ def _create_evaluation_run( dataset_id: str, ) -> EvaluationRunRef: path = ( - f"projects/{_segment(project_key)}/evaluations/" - f"{_segment(evaluation_id)}/runs" + f"projects/{segment(project_key)}/evaluations/{segment(evaluation_id)}/runs" ) body: dict[str, Any] = { "source": "api", "datasetId": dataset_id, } - raw = _mapping( + raw = require_mapping( self._api.post(path, body=body), description="evaluation run", ) @@ -572,9 +521,9 @@ def _create_evaluation_run( def _run_ref(self, raw: Mapping[str, Any]) -> EvaluationRunRef: return EvaluationRunRef( - id=_required_string(raw, "id", "evaluation run"), - evaluation_id=_required_string(raw, "evaluationId", "evaluation run"), - state=_required_string(raw, "state", "evaluation run"), + id=require_string(raw, "id", "evaluation run"), + evaluation_id=require_string(raw, "evaluationId", "evaluation run"), + state=require_string(raw, "state", "evaluation run"), status_reason=( str(raw["statusReason"]) if raw.get("statusReason") is not None @@ -585,19 +534,13 @@ def _run_ref(self, raw: Mapping[str, Any]) -> EvaluationRunRef: def _build_handler_config( self, generation: GenerationConfig, - tools: Mapping[str, ResolvedTool], + tools: Sequence[Tool], ) -> dict[str, Any]: parameters = generation.get("parameters") config: dict[str, Any] = { "provider": {"name": generation["provider"]}, "model": {"name": generation["model"], "parameters": parameters}, - "tools": { - key: { - "description": tool.description, - "parameters": tool.schema, - } - for key, tool in tools.items() - }, + "tools": handler_config_tools(tools), } snippet_variables = {"snippet": generation.get("prompt_snippets", {})} if "instructions" in generation: @@ -1150,9 +1093,9 @@ def _get_summary( self, project_key: str, evaluation_id: str, run_id: str ) -> RunSummary: path = ( - f"projects/{_segment(project_key)}/evaluations/{_segment(evaluation_id)}" - f"/runs/{_segment(run_id)}/summary" + f"projects/{segment(project_key)}/evaluations/{segment(evaluation_id)}" + f"/runs/{segment(run_id)}/summary" ) return RunSummary.from_wire( - _mapping(self._api.get(path), description="evaluation run summary") + require_mapping(self._api.get(path), description="evaluation run summary") ) diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/tools.py b/packages/client/src/launchdarkly_ai_server/evaluations/tools.py new file mode 100644 index 00000000..3d3012f4 --- /dev/null +++ b/packages/client/src/launchdarkly_ai_server/evaluations/tools.py @@ -0,0 +1,216 @@ +"""Tools an evaluation run gives to its handler.""" + +from __future__ import annotations + +import json +from collections.abc import Callable, Mapping, Sequence +from dataclasses import dataclass, field +from typing import Any, Literal + +from ..types import NativeTool +from .api import ( + EvaluationsError, + LDApiClient, + LDApiError, + require_mapping, + segment, +) + +ToolImplementation = Callable[..., Any] | NativeTool + + +@dataclass +class Tool: + """A tool a run gives to its handler. Pass a list of these to ``run``. + + Construct one to define a tool in code. ``source`` is then ``"inline"`` and + ``version`` is ``None``. Neither is a constructor argument, so a + constructed tool is always inline. + + Call ``evals.tools.get(key, implementation=...)`` instead to use a tool + from the LaunchDarkly tool library. That returns a tool with ``source`` + ``"library"`` and the version it pinned. + + ``implementation`` is the function the handler calls. A key must be + lowercase. A :class:`~launchdarkly_ai_server.NativeTool` is valid only for + a library tool, because the provider supplies its schema. + """ + + key: str + implementation: ToolImplementation + schema: dict[str, Any] = field(default_factory=dict) + description: str = "" + source: Literal["library", "inline"] = field(default="inline", init=False) + version: int | None = field(default=None, init=False) + + @classmethod + def _library( + cls, + key: str, + implementation: ToolImplementation, + *, + version: int, + schema: dict[str, Any], + description: str, + ) -> Tool: + """Build a library tool. Used by ``ToolsClient.get``.""" + tool = cls( + key=key, + implementation=implementation, + schema=schema, + description=description, + ) + tool.source = "library" + tool.version = version + return tool + + def to_create_wire(self) -> dict[str, Any]: + """The entry this tool contributes to the evaluation-create body.""" + if self.source == "inline": + return { + "key": self.key, + "schema": self.schema, + "description": self.description, + "source": "inline", + } + return {"key": self.key, "version": self.version, "source": "library"} + + +def validate_tool_key(key: str) -> None: + """Validate a tool key. Raises ``EvaluationsError``.""" + if not isinstance(key, str) or not key.strip(): + raise EvaluationsError("tool keys must not be blank") + # Keys are lowercase. + if key != key.lower(): + raise EvaluationsError( + f"Tool key {key!r} must not use uppercase letters. Use " + f"{key.lower()!r} instead." + ) + + +def validate_tool_implementation(key: str, implementation: Any) -> None: + """Validate a tool implementation. Raises ``EvaluationsError``.""" + if not callable(implementation) and not isinstance(implementation, NativeTool): + raise EvaluationsError( + f"Tool {key!r} implementation must be callable or a NativeTool, got " + f"{type(implementation).__name__}" + ) + + +def _validate_inline_tool(tool: Tool) -> None: + """Validate one inline tool. Raises ``EvaluationsError``.""" + key = tool.key + if isinstance(tool.implementation, NativeTool): + raise EvaluationsError( + f"Inline tool {key!r} must not use a NativeTool. The provider " + "supplies the schema of a native tool. Call evals.tools.get() for " + "a library tool, or give this tool a function." + ) + if not isinstance(tool.schema, Mapping): + raise EvaluationsError( + f"Inline tool {key!r} schema must be a JSON object, got " + f"{type(tool.schema).__name__}" + ) + try: + # allow_nan=False rejects NaN and Infinity, which are not valid JSON. + json.dumps(dict(tool.schema), allow_nan=False) + except (TypeError, ValueError) as error: + raise EvaluationsError( + f"Inline tool {key!r} schema must be JSON-serializable: {error}" + ) from error + + +def validate_tools(tools: Sequence[Tool]) -> None: + """Validate the tools list. Raises ``EvaluationsError``. + + Issues no requests. + """ + keys_by_identity: dict[str, str] = {} + for tool in tools: + if not isinstance(tool, Tool): + raise EvaluationsError( + "each entry in tools must be a Tool. Construct one for an " + "inline tool, or call evals.tools.get() for a library tool, " + f"got {type(tool).__name__}" + ) + key = tool.key + validate_tool_key(key) + validate_tool_implementation(key, tool.implementation) + if not isinstance(tool.description, str): + raise EvaluationsError( + f"Tool {key!r} description must be a string, got " + f"{type(tool.description).__name__}" + ) + if tool.source == "inline": + _validate_inline_tool(tool) + # One key names one tool. Keys are compared case-insensitively. + identity = key.strip().lower() + collision = keys_by_identity.get(identity) + if collision is not None: + if collision == key: + raise EvaluationsError(f"Tool {key!r} appears more than once in tools.") + raise EvaluationsError( + f"Tool {key!r} collides with {collision!r}. Two tools in one " + "run must not have keys that differ only by case." + ) + keys_by_identity[identity] = key + + +def tool_handlers(tools: Sequence[Tool]) -> dict[str, ToolImplementation]: + """Return the tools list as ``{key: executable}``.""" + return {tool.key: tool.implementation for tool in tools} + + +def handler_config_tools(tools: Sequence[Tool]) -> dict[str, dict[str, Any]]: + """Return the ``tools`` entry of a handler config.""" + return { + tool.key: {"description": tool.description, "parameters": tool.schema} + for tool in tools + } + + +def create_wire_tools(tools: Sequence[Tool]) -> list[dict[str, Any]]: + """Return the ``tools`` array of an evaluation-create body.""" + return [tool.to_create_wire() for tool in tools] + + +class ToolsClient: + """Reads tools from the LaunchDarkly tool library.""" + + def __init__(self, api_client: LDApiClient, project_key: str) -> None: + self._api = api_client + self._project_key = project_key + + def get(self, key: str, *, implementation: ToolImplementation) -> Tool: + """Return the library tool ``key``, paired with ``implementation``. + + Reads the tool now and pins the version it returns. Raises + ``EvaluationsError`` when the tool does not exist in the project. + """ + validate_tool_key(key) + validate_tool_implementation(key, implementation) + path = f"projects/{segment(self._project_key)}/ai-tools/{segment(key)}" + try: + raw = require_mapping(self._api.get(path), description=f"tool {key!r}") + except LDApiError as error: + if error.status == 404: + raise EvaluationsError( + f"LaunchDarkly AI tool {key!r} was not found in project " + f"{self._project_key!r}" + ) from error + raise + version = raw.get("version") + if not isinstance(version, int): + raise EvaluationsError( + f"LaunchDarkly AI tool {key!r} has no integer version" + ) + schema = raw.get("schema") + if not isinstance(schema, Mapping): + schema = {} + return Tool._library( + key, + implementation, + version=version, + schema=dict(schema), + description=str(raw.get("description") or ""), + ) diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/types.py b/packages/client/src/launchdarkly_ai_server/evaluations/types.py index 75af1aaa..699ca183 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/types.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/types.py @@ -58,15 +58,6 @@ class DatasetRow: @dataclass -class ResolvedTool: - """The schema and pinned version returned by the LaunchDarkly tool API.""" - - key: str - version: int - description: str = "" - schema: dict[str, Any] = field(default_factory=dict) - - @dataclass(frozen=True) class AIConfig: """Identifies an existing AI Config variation to seed an evaluation run. diff --git a/packages/client/tests/test_evaluations.py b/packages/client/tests/test_evaluations.py index e4c27226..282b07ab 100644 --- a/packages/client/tests/test_evaluations.py +++ b/packages/client/tests/test_evaluations.py @@ -67,34 +67,37 @@ def failing_transport( def test_init_resolves_credentials_from_env(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("LD_PROJECT_KEY", "proj") monkeypatch.setenv("LD_API_TOKEN", "api-token-from-env") monkeypatch.setenv("LD_SDK_KEY", "sdk-key-from-env") evals = init_evaluations(transport=RecordingTransport()) - assert evals.api.api_token == "api-token-from-env" + assert evals.api.api_key == "api-token-from-env" assert evals.sdk_key == "sdk-key-from-env" assert evals.api.base_uri == DEFAULT_BASE_URI assert evals.ui_base_uri == "https://app.launchdarkly.com" def test_init_prefers_explicit_credentials(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("LD_PROJECT_KEY", "proj") monkeypatch.setenv("LD_API_TOKEN", "api-token-from-env") monkeypatch.setenv("LD_SDK_KEY", "sdk-key-from-env") evals = init_evaluations( - api_token="explicit-token", + api_key="explicit-token", sdk_key="explicit-sdk-key", transport=RecordingTransport(), ) - assert evals.api.api_token == "explicit-token" + assert evals.api.api_key == "explicit-token" assert evals.sdk_key == "explicit-sdk-key" def test_missing_api_token_raises_before_network_io( monkeypatch: pytest.MonkeyPatch, ) -> None: + monkeypatch.setenv("LD_PROJECT_KEY", "proj") monkeypatch.delenv("LD_API_TOKEN", raising=False) monkeypatch.setenv("LD_SDK_KEY", "sdk-key") @@ -105,6 +108,7 @@ def test_missing_api_token_raises_before_network_io( def test_blank_api_token_env_is_treated_as_unset( monkeypatch: pytest.MonkeyPatch, ) -> None: + monkeypatch.setenv("LD_PROJECT_KEY", "proj") monkeypatch.setenv("LD_API_TOKEN", " ") with pytest.raises(EvaluationsError): @@ -114,6 +118,7 @@ def test_blank_api_token_env_is_treated_as_unset( def test_missing_sdk_key_raises_before_network_io( monkeypatch: pytest.MonkeyPatch, ) -> None: + monkeypatch.setenv("LD_PROJECT_KEY", "proj") monkeypatch.setenv("LD_API_TOKEN", "api-token") monkeypatch.delenv("LD_SDK_KEY", raising=False) @@ -124,6 +129,7 @@ def test_missing_sdk_key_raises_before_network_io( def test_blank_sdk_key_env_is_treated_as_unset( monkeypatch: pytest.MonkeyPatch, ) -> None: + monkeypatch.setenv("LD_PROJECT_KEY", "proj") monkeypatch.setenv("LD_API_TOKEN", "api-token") monkeypatch.setenv("LD_SDK_KEY", " ") @@ -134,6 +140,7 @@ def test_blank_sdk_key_env_is_treated_as_unset( def test_missing_sdk_key_is_allowed_with_a_byoc_client( monkeypatch: pytest.MonkeyPatch, ) -> None: + monkeypatch.setenv("LD_PROJECT_KEY", "proj") monkeypatch.setenv("LD_API_TOKEN", "api-token") monkeypatch.delenv("LD_SDK_KEY", raising=False) byoc_client = MagicMock() @@ -149,6 +156,7 @@ def test_missing_sdk_key_is_allowed_with_a_byoc_client( def test_missing_sdk_key_raises_when_the_byoc_client_cannot_emit_events( monkeypatch: pytest.MonkeyPatch, ) -> None: + monkeypatch.setenv("LD_PROJECT_KEY", "proj") monkeypatch.setenv("LD_API_TOKEN", "api-token") monkeypatch.delenv("LD_SDK_KEY", raising=False) lifecycle_module._set_client_for_testing(object()) @@ -160,6 +168,7 @@ def test_missing_sdk_key_raises_when_the_byoc_client_cannot_emit_events( def test_base_uri_override_isolated_from_sdk_delivery_uri( monkeypatch: pytest.MonkeyPatch, ) -> None: + monkeypatch.setenv("LD_PROJECT_KEY", "proj") monkeypatch.setenv("LD_API_TOKEN", "api-token") monkeypatch.setenv("LD_SDK_KEY", "sdk-key") monkeypatch.setenv("LD_API_BASE_URI", "https://api.staging.example.com/") @@ -177,6 +186,7 @@ def test_base_uri_override_isolated_from_sdk_delivery_uri( def test_ui_base_uri_precedence_and_api_base_isolation( monkeypatch: pytest.MonkeyPatch, ) -> None: + monkeypatch.setenv("LD_PROJECT_KEY", "proj") monkeypatch.setenv("LD_API_TOKEN", "api-token") monkeypatch.setenv("LD_SDK_KEY", "sdk-key") monkeypatch.setenv("LD_API_BASE_URI", "https://api.staging.example.com") @@ -194,7 +204,7 @@ def test_ui_base_uri_precedence_and_api_base_isolation( def test_requests_carry_token_auth_and_json_body() -> None: transport = RecordingTransport([HttpResponse(status=201, body='{"key": "run-1"}')]) - client = LDApiClient(api_token="api-token", transport=transport) + client = LDApiClient(api_key="api-token", transport=transport) result = client.post("projects/proj/evaluations", body={"key": "support-qa"}) @@ -211,7 +221,7 @@ def test_requests_carry_token_auth_and_json_body() -> None: def test_get_encodes_query_params_and_omits_none() -> None: transport = RecordingTransport([HttpResponse(status=200, body='{"items": []}')]) client = LDApiClient( - api_token="api-token", base_uri="https://ld.example.com", transport=transport + api_key="api-token", base_uri="https://ld.example.com", transport=transport ) client.get("projects/proj/datasets/golden", params={"limit": 50, "offset": None}) @@ -238,7 +248,7 @@ def test_rate_limit_retries_and_honors_retry_after() -> None: ) sleeps: list[float] = [] client = LDApiClient( - api_token="api-token", + api_key="api-token", transport=transport, max_retries=1, sleep=sleeps.append, @@ -256,7 +266,7 @@ def test_server_error_retries_get_but_not_post() -> None: [server_error, HttpResponse(200, '{"ok": true}')] ) client = LDApiClient( - api_token="api-token", + api_key="api-token", transport=get_transport, max_retries=2, sleep=lambda _: None, @@ -268,7 +278,7 @@ def test_server_error_retries_get_but_not_post() -> None: post_transport = RecordingTransport([server_error]) client = LDApiClient( - api_token="api-token", + api_key="api-token", transport=post_transport, max_retries=2, sleep=lambda _: None, @@ -296,7 +306,7 @@ def timing_out_transport( raise TimeoutError("timed out") client = LDApiClient( - api_token="api-token", + api_key="api-token", transport=timing_out_transport, max_retries=2, sleep=lambda _: None, @@ -317,7 +327,7 @@ def test_rate_limited_post_is_retried() -> None: ] ) client = LDApiClient( - api_token="api-token", + api_key="api-token", transport=transport, max_retries=1, sleep=lambda _: None, @@ -334,7 +344,7 @@ def test_forbidden_response_is_not_retried() -> None: transport = RecordingTransport( [HttpResponse(status=403, body='{"message": "forbidden"}')] ) - client = LDApiClient(api_token="api-token", transport=transport, max_retries=3) + client = LDApiClient(api_key="api-token", transport=transport, max_retries=3) with pytest.raises(LDApiError) as excinfo: client.get("projects/proj/evaluations") @@ -347,7 +357,7 @@ def test_error_response_raises_ld_api_error() -> None: transport = RecordingTransport( [HttpResponse(status=404, body='{"message": "nope"}')] ) - client = LDApiClient(api_token="api-token", transport=transport) + client = LDApiClient(api_key="api-token", transport=transport) with pytest.raises(LDApiError) as excinfo: client.get("projects/proj/ai-tools/missing") @@ -358,7 +368,7 @@ def test_error_response_raises_ld_api_error() -> None: def test_empty_response_body_is_none() -> None: transport = RecordingTransport([HttpResponse(status=204, body="")]) - client = LDApiClient(api_token="api-token", transport=transport) + client = LDApiClient(api_key="api-token", transport=transport) assert client.post("projects/proj/evaluations/support-qa/runs") is None @@ -395,3 +405,32 @@ def test_run_summary_and_result() -> None: assert RunSummary.from_wire(None) == RunSummary() assert result.passed is False assert result.run_id == "run-1" + + +def test_project_key_resolves_from_the_environment(monkeypatch) -> None: + monkeypatch.setenv("LD_PROJECT_KEY", "proj-from-env") + monkeypatch.setenv("LD_API_TOKEN", "api-token") + monkeypatch.setenv("LD_SDK_KEY", "sdk-key") + + evals = init_evaluations(transport=RecordingTransport()) + + assert evals.project_key == "proj-from-env" + + +def test_explicit_project_key_wins_over_the_environment(monkeypatch) -> None: + monkeypatch.setenv("LD_PROJECT_KEY", "proj-from-env") + monkeypatch.setenv("LD_API_TOKEN", "api-token") + monkeypatch.setenv("LD_SDK_KEY", "sdk-key") + + evals = init_evaluations(project_key="explicit", transport=RecordingTransport()) + + assert evals.project_key == "explicit" + + +def test_missing_project_key_raises_before_network_io(monkeypatch) -> None: + monkeypatch.delenv("LD_PROJECT_KEY", raising=False) + monkeypatch.setenv("LD_API_TOKEN", "api-token") + monkeypatch.setenv("LD_SDK_KEY", "sdk-key") + + with pytest.raises(EvaluationsError, match="LD_PROJECT_KEY"): + init_evaluations(transport=failing_transport) diff --git a/packages/client/tests/test_evaluations_run.py b/packages/client/tests/test_evaluations_run.py index 4a4b8571..c96c2b12 100644 --- a/packages/client/tests/test_evaluations_run.py +++ b/packages/client/tests/test_evaluations_run.py @@ -8,7 +8,7 @@ import pytest -from launchdarkly_ai_server import create_handler +from launchdarkly_ai_server import NativeTool, create_handler from launchdarkly_ai_server.evaluations import ( AIConfig, DatasetRow, @@ -16,6 +16,7 @@ HttpResponse, Judge, Scorer, + Tool, init_evaluations, ) @@ -119,6 +120,10 @@ def lookup_order(order_id: str) -> str: return order_id +def refund_order(order_id: str) -> str: + return f"refunded {order_id}" + + @pytest.mark.asyncio async def test_complete_run_with_zero_failed_and_error_rows_passes( monkeypatch: pytest.MonkeyPatch, @@ -222,7 +227,8 @@ async def test_complete_run_with_zero_failed_and_error_rows_passes( client.flush = AsyncMock() init_client.return_value = client evals = init_evaluations( - api_token="token", + project_key="proj", + api_key="token", sdk_key="sdk-key", base_uri="https://api.example.com", ui_base_uri="https://ui.example.com/", @@ -231,11 +237,10 @@ async def test_complete_run_with_zero_failed_and_error_rows_passes( assert evals.sdk_key == "sdk-key" result = await evals.run( - project_key="proj", key="support-qa-unique", dataset="golden", handler=successful_handler, - tools={"lookup_order": lookup_order}, + tools=[evals.tools.get("lookup_order", implementation=lookup_order)], generation={ "provider": "OpenAI", "model": "gpt-4o", @@ -277,7 +282,7 @@ async def test_complete_run_with_zero_failed_and_error_rows_passes( "generationModel": "gpt-4o", "parameters": {"temperature": 0.2}, "messages": [{"role": "system", "content": "Help the user."}], - "tools": [{"key": "lookup_order", "version": 7}], + "tools": [{"key": "lookup_order", "version": 7, "source": "library"}], } assert transport.requests[5]["url"].endswith( "/api/v2/projects/proj/evaluations/11111111-1111-1111-1111-111111111111/runs" @@ -401,13 +406,14 @@ async def fake_init_client(options: dict[str, Any]) -> MagicMock: monkeypatch.setattr( "launchdarkly_ai_server.evaluations.module.init_client", fake_init_client ) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) async def handler(*args: object) -> dict[str, Any]: return {"output": "generated"} result = await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -464,13 +470,12 @@ async def test_summary_is_polled_until_rows_are_accounted( ), ] ) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) async def handler(*args: object) -> dict[str, Any]: return {"output": "generated"} result = await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -526,13 +531,12 @@ async def test_summary_polling_completes_for_real_backend_summary_without_state( ), ] ) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) async def handler(*args: object) -> dict[str, Any]: return {"output": "generated"} result = await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -585,13 +589,12 @@ async def test_summary_polling_ignores_missing_state_even_when_pending_is_zero( ), ] ) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) async def handler(*args: object) -> dict[str, Any]: return {"output": "generated"} result = await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -635,7 +638,7 @@ async def test_summary_polling_times_out_waiting_for_rows_to_be_accounted( ), ] ) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) async def handler(*args: object) -> dict[str, Any]: return {"output": "generated"} @@ -645,7 +648,6 @@ async def handler(*args: object) -> dict[str, Any]: match=r"Timed out after 0 seconds.*rows to be fully accounted.*pending_rows=1", ): await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -684,13 +686,12 @@ async def test_poll_timeout_and_interval_are_configurable_per_run() -> None: ), ] ) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) async def handler(*args: object) -> dict[str, Any]: return {"output": "generated"} result = await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -707,7 +708,6 @@ async def handler(*args: object) -> dict[str, Any]: with pytest.raises(EvaluationsError, match="poll_timeout_seconds"): await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -725,14 +725,15 @@ async def handler(*args: object) -> dict[str, Any]: async def test_nan_poll_values_are_rejected( poll_interval_seconds: float, poll_timeout_seconds: float ) -> None: - evals = init_evaluations(api_token="token", transport=failing_transport) + evals = init_evaluations( + project_key="proj", api_key="token", transport=failing_transport + ) async def handler(*args: object) -> dict[str, Any]: return {"output": "generated"} with pytest.raises(EvaluationsError, match="must be a number"): await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -777,14 +778,13 @@ async def test_run_uses_a_byoc_client_when_no_sdk_key_is_configured( ), ] ) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) assert evals.sdk_key is None async def handler(*args: object) -> dict[str, Any]: return {"output": "generated"} result = await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -808,7 +808,9 @@ async def test_run_raises_when_no_sdk_key_and_no_initialized_client( monkeypatch.setattr( "launchdarkly_ai_server.evaluations.module.get_client", get_client ) - evals = init_evaluations(api_token="token", transport=failing_transport) + evals = init_evaluations( + project_key="proj", api_key="token", transport=failing_transport + ) get_client.side_effect = RuntimeError("client not initialized") async def handler(*args: object) -> dict[str, Any]: @@ -816,7 +818,6 @@ async def handler(*args: object) -> dict[str, Any]: with pytest.raises(EvaluationsError, match="no initialized LaunchDarkly client"): await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -863,13 +864,12 @@ async def test_failed_rows_fail_the_result() -> None: ), ] ) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) async def handler(*args: object) -> dict[str, Any]: return {"output": "generated"} result = await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -883,11 +883,10 @@ async def handler(*args: object) -> dict[str, Any]: @pytest.mark.asyncio async def test_run_rejects_instructions_and_messages_before_network_io() -> None: transport = SequencedTransport([]) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) with pytest.raises(EvaluationsError, match=r"instructions.*messages"): await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=successful_handler, @@ -907,19 +906,453 @@ async def test_missing_tool_aborts_before_any_mutating_request() -> None: transport = SequencedTransport( [response(404, {"code": "not_found", "message": "not found"})] ) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) with pytest.raises(EvaluationsError, match="missing_tool"): + evals.tools.get("missing_tool", implementation=lookup_order) + + assert [request["method"] for request in transport.requests] == ["GET"] + + +ORDER_SCHEMA: dict[str, Any] = { + "type": "object", + "properties": {"order_id": {"type": "string"}}, + "required": ["order_id"], +} + + +def recorded_paths(transport: SequencedTransport) -> list[tuple[str, str]]: + """The recorded requests as ``(method, path)`` pairs, query strings dropped.""" + return [ + ( + request["method"], + request["url"].split("/api/v2/", 1)[-1].split("?", 1)[0], + ) + for request in transport.requests + ] + + +def hosted_dataset_responses() -> list[HttpResponse]: + """Canned responses for a one-row hosted dataset run that passes.""" + return [ + response(200, {"id": "dataset-id", "name": "golden"}), + response( + 200, + dataset_page( + [{"rowIndex": 0, "input": "Order A19", "variables": {}}], total=1 + ), + ), + response(201, {"id": "evaluation-id", "name": "eval-key", "version": 3}), + response( + 201, + {"id": "run-id", "evaluationId": "evaluation-id", "state": "PENDING"}, + ), + response( + 200, + { + "statusCounts": { + "total": 1, + "passed": 1, + "failed": 0, + "error": 0, + "pending": 0, + } + }, + ), + ] + + +@pytest.mark.asyncio +async def test_inline_tool_runs_without_reading_the_tool_api() -> None: + transport = SequencedTransport(hosted_dataset_responses()) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + seen: dict[str, Any] = {} + + async def handler( + config: dict[str, Any], + user_input: str | None, + tool_handlers: dict[str, Callable[..., Any]], + variables: dict[str, Any], + ) -> dict[str, Any]: + seen["config_tools"] = config["tools"] + seen["tool_handlers"] = tool_handlers + return {"output": "ok"} + + result = await evals.run( + key="eval-key", + dataset="golden", + handler=handler, + tools=[ + Tool( + key="lookup_order", + implementation=lookup_order, + schema=ORDER_SCHEMA, + description="Look up an order", + ) + ], + generation={"provider": "OpenAI", "model": "gpt-4o"}, + ) + + assert result.passed is True + # The full sequence, so a stray request anywhere in the run is visible. + assert recorded_paths(transport) == [ + ("GET", "projects/proj/datasets/golden"), + ("GET", "projects/proj/datasets/golden/rows"), + ("POST", "projects/proj/evaluations"), + ("POST", "projects/proj/evaluations/evaluation-id/runs"), + ("GET", "projects/proj/evaluations/evaluation-id/runs/run-id/summary"), + ] + # Asserted explicitly rather than left to the sequence above: a transport + # that tolerated a surplus request would not fail on a stray tool GET, and + # skipping that request is the whole point of an inline definition. + assert not any("/ai-tools" in request["url"] for request in transport.requests), ( + "an inline tool must not be looked up in the AI library" + ) + + assert transport.requests[2]["body"]["tools"] == [ + { + "key": "lookup_order", + "schema": ORDER_SCHEMA, + "description": "Look up an order", + "source": "inline", + } + ] + # Same config shape a library tool produces, fed from the inline body, so a + # handler cannot tell the two sources apart. + assert seen["config_tools"] == { + "lookup_order": { + "description": "Look up an order", + "parameters": ORDER_SCHEMA, + } + } + # The handler is passed the executable, not the definition wrapping it. + assert list(seen["tool_handlers"]) == ["lookup_order"] + assert seen["tool_handlers"]["lookup_order"]("A1") == "A1" + + +@pytest.mark.asyncio +async def test_inline_tool_description_defaults_to_empty_string() -> None: + transport = SequencedTransport(hosted_dataset_responses()) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + + async def handler(*args: object) -> dict[str, Any]: + return {"output": "ok"} + + await evals.run( + key="eval-key", + dataset="golden", + handler=handler, + tools=[ + Tool(key="lookup_order", implementation=lookup_order, schema=ORDER_SCHEMA) + ], + generation={"provider": "OpenAI", "model": "gpt-4o"}, + ) + + assert transport.requests[2]["body"]["tools"] == [ + { + "key": "lookup_order", + "schema": ORDER_SCHEMA, + "description": "", + "source": "inline", + } + ] + + +@pytest.mark.asyncio +async def test_mixed_library_and_inline_tools_each_keep_their_own_source() -> None: + transport = SequencedTransport( + [ + response( + 200, + { + "key": "lookup_order", + "version": 7, + "description": "Look up an order", + "schema": {"type": "object"}, + }, + ), + *hosted_dataset_responses(), + ] + ) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + seen: dict[str, Any] = {} + + async def handler( + config: dict[str, Any], + user_input: str | None, + tool_handlers: dict[str, Callable[..., Any]], + variables: dict[str, Any], + ) -> dict[str, Any]: + seen["config_tools"] = config["tools"] + seen["tool_handlers"] = tool_handlers + return {"output": "ok"} + + await evals.run( + key="eval-key", + dataset="golden", + handler=handler, + tools=[ + evals.tools.get("lookup_order", implementation=lookup_order), + Tool( + key="refund_order", + implementation=refund_order, + schema=ORDER_SCHEMA, + description="Refund an order", + ), + ], + generation={"provider": "OpenAI", "model": "gpt-4o"}, + ) + + # Exactly one tool GET, for the library key only. + assert [ + path for method, path in recorded_paths(transport) if "ai-tools" in path + ] == ["projects/proj/ai-tools/lookup_order"] + assert transport.requests[3]["body"]["tools"] == [ + {"key": "lookup_order", "version": 7, "source": "library"}, + { + "key": "refund_order", + "schema": ORDER_SCHEMA, + "description": "Refund an order", + "source": "inline", + }, + ] + assert seen["config_tools"] == { + "lookup_order": { + "description": "Look up an order", + "parameters": {"type": "object"}, + }, + "refund_order": { + "description": "Refund an order", + "parameters": ORDER_SCHEMA, + }, + } + assert sorted(seen["tool_handlers"]) == ["lookup_order", "refund_order"] + assert seen["tool_handlers"]["lookup_order"]("A1") == "A1" + assert seen["tool_handlers"]["refund_order"]("A1") == "refunded A1" + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("tools", "match"), + [ + pytest.param( + [Tool(key=" ", implementation=lookup_order, schema=ORDER_SCHEMA)], + "must not be blank", + id="blank_key", + ), + pytest.param( + [ + Tool( + key="Lookup_Order", implementation=lookup_order, schema=ORDER_SCHEMA + ) + ], + "must not use uppercase letters", + id="uppercase_key", + ), + pytest.param( + [Tool(key="lookup_order", implementation=lookup_order, schema=None)], # type: ignore[arg-type] + "schema must be a JSON object", + id="schema_is_none", + ), + pytest.param( + [ + Tool( + key="lookup_order", + implementation=lookup_order, + schema=[{"type": "object"}], + ) + ], # type: ignore[arg-type] + "schema must be a JSON object", + id="schema_is_a_list", + ), + pytest.param( + [ + Tool( + key="lookup_order", + implementation=lookup_order, + schema={"default": object()}, + ) + ], + "schema must be JSON-serializable", + id="schema_is_not_serializable", + ), + pytest.param( + [ + Tool( + key="lookup_order", + implementation=lookup_order, + schema={"default": float("nan")}, + ) + ], + "schema must be JSON-serializable", + id="schema_holds_nan", + ), + pytest.param( + [ + Tool( + key="lookup_order", + implementation=lookup_order, + schema={"default": float("inf")}, + ) + ], + "schema must be JSON-serializable", + id="schema_holds_infinity", + ), + pytest.param( + [ + Tool( + key="lookup_order", + implementation="not a function", + schema=ORDER_SCHEMA, + ) + ], # type: ignore[arg-type] + "implementation must be callable", + id="implementation_is_not_callable", + ), + pytest.param( + [ + Tool( + key="lookup_order", + implementation=lookup_order, + schema=ORDER_SCHEMA, + description=None, + ) + ], # type: ignore[arg-type] + "description must be a string", + id="description_is_not_a_string", + ), + pytest.param( + ["not a tool"], # type: ignore[dict-item] + "each entry in tools must be a Tool", + id="entry_is_not_a_tool", + ), + ], +) +async def test_bad_tool_entry_is_rejected_with_zero_requests( + tools: list[Any], match: str +) -> None: + transport = SequencedTransport([]) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + + with pytest.raises(EvaluationsError, match=match): await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=successful_handler, - tools={"missing_tool": lookup_order}, + tools=tools, generation={"provider": "OpenAI", "model": "gpt-4o"}, ) - assert [request["method"] for request in transport.requests] == ["GET"] + assert transport.requests == [] + + +@pytest.mark.asyncio +async def test_native_tool_paired_with_an_inline_definition_is_rejected() -> None: + transport = SequencedTransport([]) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + + with pytest.raises(EvaluationsError, match=r"lookup_order.*NativeTool"): + await evals.run( + key="eval-key", + dataset="golden", + handler=successful_handler, + tools=[ + Tool( + key="lookup_order", + implementation=NativeTool("WebSearch"), + schema=ORDER_SCHEMA, + ) + ], + generation={"provider": "OpenAI", "model": "gpt-4o"}, + ) + + assert transport.requests == [] + + +@pytest.mark.asyncio +async def test_native_tool_on_its_own_still_resolves_from_the_library() -> None: + transport = SequencedTransport( + [ + response(200, {"key": "web_search", "version": 2}), + *hosted_dataset_responses(), + ] + ) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + + async def handler(*args: object) -> dict[str, Any]: + return {"output": "ok"} + + await evals.run( + key="eval-key", + dataset="golden", + handler=handler, + tools=[evals.tools.get("web_search", implementation=NativeTool("WebSearch"))], + generation={"provider": "OpenAI", "model": "gpt-4o"}, + ) + + assert recorded_paths(transport)[0] == ("GET", "projects/proj/ai-tools/web_search") + assert transport.requests[3]["body"]["tools"] == [ + {"key": "web_search", "version": 2, "source": "library"} + ] + + +@pytest.mark.asyncio +async def test_a_repeated_tool_key_is_rejected_with_zero_requests() -> None: + """One key names one tool, whichever source each entry came from.""" + transport = SequencedTransport([]) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + + with pytest.raises( + EvaluationsError, match=r"'lookup_order' appears more than once" + ): + await evals.run( + key="eval-key", + dataset="golden", + handler=successful_handler, + tools=[ + Tool( + key="lookup_order", + implementation=lookup_order, + schema=ORDER_SCHEMA, + ), + Tool( + key="lookup_order", + implementation=refund_order, + schema=ORDER_SCHEMA, + ), + ], + generation={"provider": "OpenAI", "model": "gpt-4o"}, + ) + + assert transport.requests == [] + + +@pytest.mark.asyncio +async def test_tools_get_refuses_an_uppercase_key_before_any_request() -> None: + """A tool key is lowercase, whether it is constructed or read from the library.""" + transport = SequencedTransport([]) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + + with pytest.raises(EvaluationsError, match="must not use uppercase letters"): + evals.tools.get("Lookup_Order", implementation=lookup_order) + + assert transport.requests == [] @pytest.mark.asyncio @@ -936,11 +1369,10 @@ async def test_empty_dataset_fails_before_evaluation_or_run_creation() -> None: response(200, dataset_page([], total=0)), ] ) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) with pytest.raises(EvaluationsError, match="empty"): await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=successful_handler, @@ -1034,10 +1466,11 @@ async def fake_init_client(options: dict[str, Any]) -> MagicMock: monkeypatch.setattr( "launchdarkly_ai_server.evaluations.module.init_client", fake_init_client ) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) result = await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -1117,7 +1550,9 @@ async def fake_extract_variation( "launchdarkly_ai_server.evaluations.runner.extract_variation", fake_extract_variation, ) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) async def handler( config: dict[str, Any], @@ -1156,7 +1591,6 @@ async def handler( } result = await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, @@ -1268,7 +1702,9 @@ async def fake_extract_variation( "launchdarkly_ai_server.evaluations.runner.extract_variation", fake_extract_variation, ) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) async def handler( config: dict[str, Any], @@ -1287,7 +1723,6 @@ async def handler( } result = await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, @@ -1363,7 +1798,9 @@ async def fake_extract_variation( "launchdarkly_ai_server.evaluations.runner.extract_variation", fake_extract_variation, ) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) async def handler( config: dict[str, Any], @@ -1376,7 +1813,6 @@ async def handler( return {"output": "generated"} await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, @@ -1437,7 +1873,9 @@ async def fake_extract_variation( "launchdarkly_ai_server.evaluations.runner.extract_variation", fake_extract_variation, ) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) async def handler(*args: object) -> dict[str, Any]: return {"output": "generated"} @@ -1447,7 +1885,6 @@ async def handler(*args: object) -> dict[str, Any]: match=r"Failed to resolve LaunchDarkly judge 'security-judge': not found", ): await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -1490,7 +1927,9 @@ async def test_run_with_deterministic_scorer_emits_scorer_evaluation_event( ), ] ) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) async def handler(*args: object) -> dict[str, Any]: return { @@ -1505,7 +1944,6 @@ def check_refund(row: DatasetRow, output: Any) -> bool: return "refund" in str(output) result = await evals.run( - project_key="proj", key="support-qa", dataset="support-golden-v3", handler=handler, @@ -1621,7 +2059,9 @@ async def test_bad_judge_output_emits_error_event_instead_of_crashing( ) -> None: transport = judge_run_transport() accuracy_judge_variation(monkeypatch) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) async def handler( config: dict[str, Any], @@ -1634,7 +2074,6 @@ async def handler( return {"output": "generated"} result = await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, @@ -1661,7 +2100,9 @@ async def test_generated_placeholders_are_not_expanded_into_judge_prompt( transport = judge_run_transport() accuracy_judge_variation(monkeypatch) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) async def handler( config: dict[str, Any], @@ -1678,7 +2119,6 @@ async def handler( return {"output": "{{expected_output}} leaked?"} result = await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, @@ -1717,7 +2157,9 @@ async def test_missing_expected_output_renders_empty_judge_variables( ] ) accuracy_judge_variation(monkeypatch) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) async def handler( config: dict[str, Any], @@ -1732,7 +2174,6 @@ async def handler( return {"output": "generated"} result = await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, @@ -1745,14 +2186,15 @@ async def handler( @pytest.mark.asyncio async def test_duplicate_criteria_rejected_before_any_request() -> None: transport = SequencedTransport([]) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) async def handler(*args: object) -> dict[str, Any]: return {"output": "generated"} with pytest.raises(EvaluationsError, match="Duplicate evaluation criteria"): await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, @@ -1773,14 +2215,15 @@ async def test_duplicate_criteria_rejected_case_insensitively() -> None: only by case would still collide there even though they'd look distinct to a case-sensitive check.""" transport = SequencedTransport([]) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) async def handler(*args: object) -> dict[str, Any]: return {"output": "generated"} with pytest.raises(EvaluationsError, match="Duplicate evaluation criteria"): await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, @@ -1803,7 +2246,9 @@ async def test_errored_generation_row_emits_generation_incomplete_criterion_even summary={"statusCounts": {"total": 1, "passed": 0, "error": 1, "pending": 0}} ) accuracy_judge_variation(monkeypatch) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) async def handler( config: dict[str, Any], @@ -1816,7 +2261,6 @@ async def handler( raise RuntimeError("provider unavailable") result = await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, @@ -1854,7 +2298,9 @@ def track(event_name: str, *args: Any) -> None: raise RuntimeError("event pipeline unavailable") stub_sdk_client.track = MagicMock(side_effect=track) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) async def handler( config: dict[str, Any], @@ -1868,7 +2314,6 @@ async def handler( with pytest.raises(EvaluationsError) as error: await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, @@ -1955,11 +2400,12 @@ async def test_judge_on_another_provider_fails_before_any_records_are_created( """ transport = judge_run_transport() judge_variation(monkeypatch, provider="Anthropic") - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) with pytest.raises(EvaluationsError) as error: await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=create_handler(("OpenAI", "messages"), _generation_only), @@ -1979,7 +2425,9 @@ async def test_judge_handlers_route_a_judge_to_its_own_provider( ) -> None: transport = judge_run_transport() judge_variation(monkeypatch, provider="Anthropic") - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) judged: list[dict[str, Any]] = [] async def anthropic_judge( @@ -1993,7 +2441,6 @@ async def anthropic_judge( return {"output": '{"score": 0.75, "reasoning": "ok"}'} result = await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=create_handler(("OpenAI", "messages"), _generation_only), @@ -2024,7 +2471,9 @@ async def test_exact_provider_judge_handler_beats_a_wildcard_adapter( """ transport = judge_run_transport() judge_variation(monkeypatch, provider="Anthropic") - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) chosen: list[str] = [] def judge_handler(name: str) -> Any: @@ -2044,7 +2493,6 @@ async def run( exact = create_handler(("Anthropic", "messages"), judge_handler("exact")) result = await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=create_handler(("OpenAI", "messages"), _generation_only), @@ -2064,7 +2512,9 @@ async def test_wildcard_judge_handler_runs_a_judge_no_handler_names( ) -> None: transport = judge_run_transport() judge_variation(monkeypatch, provider="Anthropic") - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) judged: list[dict[str, Any]] = [] async def wildcard_judge( @@ -2078,7 +2528,6 @@ async def wildcard_judge( return {"output": '{"score": 1, "reasoning": "ok"}'} result = await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=create_handler(("OpenAI", "messages"), _generation_only), @@ -2109,7 +2558,9 @@ async def test_agent_handler_runs_a_messages_mode_judge_with_collapsed_messages( ] }, ) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) judged: list[dict[str, Any]] = [] async def anthropic_agent_judge( @@ -2123,7 +2574,6 @@ async def anthropic_agent_judge( return {"output": '{"score": 1, "reasoning": "ok"}'} result = await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=create_handler(("OpenAI", "messages"), _generation_only), @@ -2146,7 +2596,9 @@ async def test_generation_handler_runs_a_judge_on_the_same_provider( ) -> None: transport = judge_run_transport() judge_variation(monkeypatch, provider="OpenAI") - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) calls: list[str | None] = [] async def openai_handler( @@ -2162,7 +2614,6 @@ async def openai_handler( return {"output": "generated"} result = await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=create_handler(("OpenAI", "messages"), openai_handler), @@ -2181,11 +2632,12 @@ async def test_judge_handlers_must_declare_the_provider_they_serve( """An unrouted judge handler would silently never be selected.""" transport = judge_run_transport() accuracy_judge_variation(monkeypatch) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) with pytest.raises(EvaluationsError, match="does not declare provides_for"): await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=_generation_only, @@ -2229,7 +2681,9 @@ async def test_criteria_run_concurrently_within_the_concurrency_bound( ] ) accuracy_judge_variation(monkeypatch) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) in_flight = 0 max_in_flight = 0 @@ -2250,7 +2704,6 @@ async def handler( return {"output": "generated"} result = await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, @@ -2316,7 +2769,9 @@ async def test_tool_trajectory_reaches_the_judge_via_message_history( """ transport = tool_run_transport() accuracy_judge_variation(monkeypatch) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) seen: dict[str, str] = {} def lookup_order(args: dict[str, Any]) -> str: @@ -2341,11 +2796,10 @@ async def handler( return {"output": "Your order shipped."} result = await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, - tools={"lookup_order": lookup_order}, + tools=[evals.tools.get("lookup_order", implementation=lookup_order)], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -2379,7 +2833,9 @@ async def test_each_row_gets_only_its_own_tool_trajectory( transport = tool_run_transport(rows=2) accuracy_judge_variation(monkeypatch) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) histories: dict[str, str] = {} both_started = asyncio.Barrier(2) @@ -2403,11 +2859,10 @@ async def handler( return {"output": f"answered {row}"} result = await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, - tools={"lookup_order": lookup_order}, + tools=[evals.tools.get("lookup_order", implementation=lookup_order)], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], concurrency=2, @@ -2428,7 +2883,9 @@ async def test_a_row_that_called_no_tools_says_so_to_the_judge( """A judge grading tool selection needs to see the tool that went unused.""" transport = tool_run_transport() accuracy_judge_variation(monkeypatch) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) seen: dict[str, str] = {} async def handler( @@ -2443,11 +2900,10 @@ async def handler( return {"output": "I do not know."} await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, - tools={"lookup_order": lambda args: "unused"}, + tools=[evals.tools.get("lookup_order", implementation=lambda args: "unused")], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -2470,7 +2926,9 @@ async def test_a_run_without_tools_leaves_message_history_unchanged( """ transport = judge_run_transport() accuracy_judge_variation(monkeypatch) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) seen: dict[str, str] = {} async def handler( @@ -2485,7 +2943,6 @@ async def handler( return {"output": "generated"} await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, @@ -2528,7 +2985,9 @@ async def fake_extract_variation( "launchdarkly_ai_server.evaluations.runner.extract_variation", fake_extract_variation, ) - evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) async def handler( config: dict[str, Any], @@ -2545,11 +3004,15 @@ async def handler( return {"output": "done"} result = await evals.run( - project_key="proj", key="support-qa", dataset="golden", handler=handler, - tools={"lookup_order": lambda args: "{{expected_output}} leaked?"}, + tools=[ + evals.tools.get( + "lookup_order", + implementation=lambda args: "{{expected_output}} leaked?", + ) + ], generation={"provider": "OpenAI", "model": "gpt-4o"}, criteria=[Judge(key="$ld:ai:judge:accuracy")], ) @@ -2564,7 +3027,7 @@ async def test_a_failed_row_keeps_the_calls_made_before_the_handler_raised() -> from launchdarkly_ai_server.evaluations.runner import EvaluationsRunner runner = EvaluationsRunner( - LDApiClient(api_token="token", transport=failing_transport) + LDApiClient(api_key="token", transport=failing_transport) ) async def handler( @@ -2668,7 +3131,7 @@ def evaluation_post(transport: SequencedTransport) -> dict[str, Any]: @pytest.mark.asyncio async def test_run_seeds_generation_from_the_latest_ai_config_variation() -> None: transport = SequencedTransport(fetched_run_responses(config_variation_page())) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) seen_configs: list[dict[str, Any]] = [] async def handler(config: dict[str, Any], *args: object) -> dict[str, Any]: @@ -2676,7 +3139,6 @@ async def handler(config: dict[str, Any], *args: object) -> dict[str, Any]: return {"output": "generated"} result = await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -2707,13 +3169,12 @@ async def handler(config: dict[str, Any], *args: object) -> dict[str, Any]: @pytest.mark.asyncio async def test_explicit_generation_overrides_the_fetched_variation() -> None: transport = SequencedTransport(fetched_run_responses(config_variation_page())) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) async def handler(*args: object) -> dict[str, Any]: return {"output": "generated"} await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -2747,11 +3208,10 @@ async def test_variation_tools_without_implementations_fail_before_mutating_requ response(200, MODEL_CONFIG), ] ) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) with pytest.raises(EvaluationsError, match="'lookup_order'"): await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=successful_handler, @@ -2790,7 +3250,7 @@ async def fake_extract_variation( "launchdarkly_ai_server.evaluations.runner.extract_variation", fake_extract_variation, ) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) async def handler(*args: object) -> dict[str, Any]: return {"output": "generated"} @@ -2798,7 +3258,6 @@ async def handler(*args: object) -> dict[str, Any]: # Resolving the attached judge is what fails, so it was picked up as a criterion. with pytest.raises(EvaluationsError, match="'security-judge'"): await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, @@ -2813,11 +3272,10 @@ async def test_unknown_variation_fails_before_any_records_are_created() -> None: transport = SequencedTransport( [response(404, {"code": "not_found", "message": "not found"})] ) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) with pytest.raises(EvaluationsError, match="was not found"): await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=successful_handler, @@ -2846,11 +3304,10 @@ async def test_config_source_is_validated_before_network_io( source: dict[str, AIConfig], message: str ) -> None: transport = SequencedTransport([]) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) with pytest.raises(EvaluationsError, match=message): await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=successful_handler, @@ -2869,31 +3326,30 @@ async def test_tool_version_drift_from_the_variation_is_logged( config_variation_page(tools=[{"key": "lookup_order", "version": 4}]) ) responses.insert( - 2, + 0, response( 200, {"key": "lookup_order", "version": 7, "schema": {"type": "object"}}, ), ) transport = SequencedTransport(responses) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) async def handler(*args: object) -> dict[str, Any]: return {"output": "generated"} await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=handler, ai_config=AIConfig(key="support-agent", variation="control"), - tools={"lookup_order": lookup_order}, + tools=[evals.tools.get("lookup_order", implementation=lookup_order)], ) assert "pins tool 'lookup_order' at version 4" in caplog.text - assert "latest version 7" in caplog.text + assert "the run uses version 7" in caplog.text assert evaluation_post(transport)["tools"] == [ - {"key": "lookup_order", "version": 7} + {"key": "lookup_order", "version": 7, "source": "library"} ] @@ -2905,11 +3361,10 @@ async def test_non_string_model_config_key_fails_loudly( transport = SequencedTransport( [response(200, config_variation_page(modelConfigKey=model_config_key))] ) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) with pytest.raises(EvaluationsError, match="non-string modelConfigKey"): await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=successful_handler, @@ -2927,12 +3382,11 @@ async def test_variation_without_a_model_config_needs_an_explicit_provider( transport = SequencedTransport( [response(200, config_variation_page(modelConfigKey=model_config_key))] ) - evals = init_evaluations(api_token="token", transport=transport) + evals = init_evaluations(project_key="proj", api_key="token", transport=transport) # No model config is linked, so none is fetched and no provider is known. with pytest.raises(EvaluationsError, match=r"generation\.provider is required"): await evals.run( - project_key="proj", key="eval-key", dataset="golden", handler=successful_handler, @@ -2965,3 +3419,77 @@ def test_ai_config_variation_from_api_layers_the_model_config() -> None: unlinked = AIConfigVariation.from_api(latest) assert "provider" not in unlinked.generation assert unlinked.generation["parameters"] == {"temperature": 0.7} + + +@pytest.mark.asyncio +async def test_tools_get_pins_the_version_it_reads() -> None: + transport = SequencedTransport( + [ + response( + 200, + { + "key": "lookup_order", + "version": 7, + "description": "Look up an order", + "schema": ORDER_SCHEMA, + }, + ) + ] + ) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + + tool = evals.tools.get("lookup_order", implementation=lookup_order) + + assert tool.source == "library" + assert tool.version == 7 + assert tool.description == "Look up an order" + assert tool.schema == ORDER_SCHEMA + assert tool.implementation is lookup_order + assert recorded_paths(transport) == [("GET", "projects/proj/ai-tools/lookup_order")] + + +def test_a_constructed_tool_is_always_inline() -> None: + """``source`` and ``version`` are not constructor arguments.""" + tool = Tool(key="lookup_order", implementation=lookup_order, schema=ORDER_SCHEMA) + + assert tool.source == "inline" + assert tool.version is None + with pytest.raises(TypeError): + Tool( # type: ignore[call-arg] + key="lookup_order", + implementation=lookup_order, + schema=ORDER_SCHEMA, + source="library", + ) + + +@pytest.mark.asyncio +async def test_run_reads_no_tool_from_the_api() -> None: + """``run`` resolves nothing: a library tool was already read by ``tools.get``.""" + transport = SequencedTransport( + [ + response(200, {"key": "lookup_order", "version": 7}), + *hosted_dataset_responses(), + ] + ) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + library_tool = evals.tools.get("lookup_order", implementation=lookup_order) + requests_before_run = len(transport.requests) + + await evals.run( + key="eval-key", + dataset="golden", + handler=successful_handler, + tools=[ + library_tool, + Tool(key="refund_order", implementation=refund_order, schema=ORDER_SCHEMA), + ], + generation={"provider": "OpenAI", "model": "gpt-4o"}, + ) + + run_paths = [path for _, path in recorded_paths(transport)[requests_before_run:]] + assert not any("ai-tools" in path for path in run_paths), run_paths From 47e8c3e1634fba66ee8da4ad52faaf49e55124d5 Mon Sep 17 00:00:00 2001 From: Dylan O'Neill Date: Mon, 28 Sep 2026 16:02:57 -0700 Subject: [PATCH 2/3] docs(evaluations): call the credential an API key, not a token The option is api_key. The error message and the README still called the credential an API access token. The LD_API_TOKEN variable name does not change. Co-Authored-By: Claude Opus 5 --- packages/client/README.md | 2 +- .../client/src/launchdarkly_ai_server/evaluations/module.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/packages/client/README.md b/packages/client/README.md index de974ce3..075085d5 100644 --- a/packages/client/README.md +++ b/packages/client/README.md @@ -39,7 +39,7 @@ No code changes are required — `init_client()` detects the packages at runtime | `LD_SERVICE_NAME` | No | OTel `service.name` resource attribute (default: `python-sdk`) | | `LD_ENVIRONMENT` | No | `deployment.environment` resource attribute attached to telemetry | | `OTEL_EXPORTER_OTLP_ENDPOINT` | No | OTLP endpoint override (default: LaunchDarkly Observability backend) | -| `LD_API_TOKEN` | For evaluations | API access token used by the evaluations management API | +| `LD_API_TOKEN` | For evaluations | API key used by the evaluations management API | | `LD_SDK_KEY` | For evaluations | SDK key whose event transport carries generation results to LaunchDarkly | | `LD_API_BASE_URI` | No | Evaluations management API host override; intentionally separate from `LD_BASE_URI` | | `LD_UI_BASE_URI` | No | LaunchDarkly application host for evaluation-run links (default: `https://app.launchdarkly.com`). Set it for a non-production project, or its runs still link to the production app | diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/module.py b/packages/client/src/launchdarkly_ai_server/evaluations/module.py index 92cfa3a0..937347c5 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/module.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/module.py @@ -538,7 +538,7 @@ def init_evaluations( token = api_key or _env("LD_API_TOKEN") if not token: raise EvaluationsError( - "No LaunchDarkly API access token provided. Set the LD_API_TOKEN " + "No LaunchDarkly API key provided. Set the LD_API_TOKEN " "environment variable or pass api_key to init_evaluations()." ) From a9c40bc50490dc0f001b45c5ad108af6a1a85c27 Mon Sep 17 00:00:00 2001 From: Dylan O'Neill Date: Mon, 28 Sep 2026 16:27:02 -0700 Subject: [PATCH 3/3] fix(evaluations): let tools= override a variation's tools A caller who passes tools= replaces the variation's list, so an empty list runs the variation with no tools. The check rejected every variation tool that the list omitted, which made that impossible. It now raises only when the caller passes no tools at all, and logs a warning for each variation tool the run does not use. Record the project a library tool was read from, and reject a tool that came from a different project than the run. The handler would otherwise use one project's schema while the record named another project's tool. Say in the docstring and the README that tools.get() blocks, so a caller does not run it inside an event loop. Co-Authored-By: Claude Opus 5 --- packages/client/README.md | 2 +- .../evaluations/module.py | 25 ++++++--- .../evaluations/tools.py | 22 +++++++- packages/client/tests/test_evaluations_run.py | 55 +++++++++++++++++++ 4 files changed, 92 insertions(+), 12 deletions(-) diff --git a/packages/client/README.md b/packages/client/README.md index 075085d5..71079ae3 100644 --- a/packages/client/README.md +++ b/packages/client/README.md @@ -203,7 +203,7 @@ result = await evals.run( ) ``` -`run()` reads no tool from the API. A constructed `Tool` is always inline, because `source` and `version` are not constructor arguments. Only `tools.get()` produces a library tool. Handlers receive the same `{key: callable}` map whichever kind a tool is, so handler code needs no change. +`run()` reads no tool from the API. `tools.get()` blocks until its read completes, so call it while you set a run up, not inside a running event loop. A constructed `Tool` is always inline, because `source` and `version` are not constructor arguments. Only `tools.get()` produces a library tool. Handlers receive the same `{key: callable}` map whichever kind a tool is, so handler code needs no change. **The list is checked before any network I/O.** A blank key, an uppercase key, a `schema` that is not a JSON object, a schema that is not JSON-serializable (including a `NaN` or `Infinity` value), and a non-callable implementation each fail with zero requests issued. A repeated key fails too, and keys are compared without case. A `NativeTool` is valid only for a library tool, because the provider supplies its schema. diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/module.py b/packages/client/src/launchdarkly_ai_server/evaluations/module.py index 937347c5..99b630b2 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/module.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/module.py @@ -195,7 +195,7 @@ async def run( ) self._validate_config_source(generation=generation, ai_config=ai_config) run_tools = list(tools or []) - validate_tools(run_tools) + validate_tools(run_tools, self._project_key) run_tool_handlers = tool_handlers(run_tools) pinned_tool_versions: dict[str, int] = {} config_label = "" @@ -208,19 +208,26 @@ async def run( ai_config.variation, ) generation = _merge_generation(ai_config_variation.generation, generation) - supplied = {tool.key for tool in run_tools} - missing = [ - name - for name in ai_config_variation.tool_versions - if name not in supplied - ] - if missing: + if tools is None and ai_config_variation.tool_versions: raise EvaluationsError( f"AI Config variation {config_label} uses tools " "with no implementation: " - + ", ".join(repr(name) for name in missing) + + ", ".join( + repr(name) for name in ai_config_variation.tool_versions + ) + ". Pass tools= with a Tool for each." ) + # A caller who passes tools= replaces the variation's list, so an + # empty list runs the variation with no tools. + supplied = {tool.key for tool in run_tools} + for name in ai_config_variation.tool_versions: + if name not in supplied: + logger.warning( + "AI Config variation %s attaches tool %r, which this run " + "does not use.", + config_label, + name, + ) pinned_tool_versions = ai_config_variation.tool_versions if criteria is None: criteria = [ diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/tools.py b/packages/client/src/launchdarkly_ai_server/evaluations/tools.py index 3d3012f4..5b370393 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/tools.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/tools.py @@ -42,6 +42,7 @@ class Tool: description: str = "" source: Literal["library", "inline"] = field(default="inline", init=False) version: int | None = field(default=None, init=False) + project_key: str | None = field(default=None, init=False) @classmethod def _library( @@ -52,6 +53,7 @@ def _library( version: int, schema: dict[str, Any], description: str, + project_key: str, ) -> Tool: """Build a library tool. Used by ``ToolsClient.get``.""" tool = cls( @@ -62,6 +64,7 @@ def _library( ) tool.source = "library" tool.version = version + tool.project_key = project_key return tool def to_create_wire(self) -> dict[str, Any]: @@ -120,10 +123,11 @@ def _validate_inline_tool(tool: Tool) -> None: ) from error -def validate_tools(tools: Sequence[Tool]) -> None: +def validate_tools(tools: Sequence[Tool], project_key: str | None = None) -> None: """Validate the tools list. Raises ``EvaluationsError``. - Issues no requests. + Checks each library tool against ``project_key`` when one is given. Issues + no requests. """ keys_by_identity: dict[str, str] = {} for tool in tools: @@ -143,6 +147,16 @@ def validate_tools(tools: Sequence[Tool]) -> None: ) if tool.source == "inline": _validate_inline_tool(tool) + elif ( + project_key is not None + and tool.project_key is not None + and tool.project_key != project_key + ): + raise EvaluationsError( + f"Tool {key!r} was read from project {tool.project_key!r} and " + f"cannot run in project {project_key!r}. Read it from " + f"{project_key!r} instead." + ) # One key names one tool. Keys are compared case-insensitively. identity = key.strip().lower() collision = keys_by_identity.get(identity) @@ -186,6 +200,9 @@ def get(self, key: str, *, implementation: ToolImplementation) -> Tool: Reads the tool now and pins the version it returns. Raises ``EvaluationsError`` when the tool does not exist in the project. + + This call blocks until the read completes. Call it while you set a run + up, not inside a running event loop. """ validate_tool_key(key) validate_tool_implementation(key, implementation) @@ -213,4 +230,5 @@ def get(self, key: str, *, implementation: ToolImplementation) -> Tool: version=version, schema=dict(schema), description=str(raw.get("description") or ""), + project_key=self._project_key, ) diff --git a/packages/client/tests/test_evaluations_run.py b/packages/client/tests/test_evaluations_run.py index c96c2b12..0383e822 100644 --- a/packages/client/tests/test_evaluations_run.py +++ b/packages/client/tests/test_evaluations_run.py @@ -3493,3 +3493,58 @@ async def test_run_reads_no_tool_from_the_api() -> None: run_paths = [path for _, path in recorded_paths(transport)[requests_before_run:]] assert not any("ai-tools" in path for path in run_paths), run_paths + + +@pytest.mark.asyncio +async def test_an_empty_tools_list_runs_a_variation_with_no_tools() -> None: + """A caller who passes tools= replaces the variation's list.""" + transport = SequencedTransport( + fetched_run_responses( + config_variation_page(tools=[{"key": "lookup_order", "version": 4}]) + ) + ) + evals = init_evaluations( + project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport + ) + + async def handler(*args: object) -> dict[str, Any]: + return {"output": "generated"} + + await evals.run( + key="eval-key", + dataset="golden", + handler=handler, + ai_config=AIConfig(key="support-agent", variation="control"), + tools=[], + ) + + assert "tools" not in evaluation_post(transport) + + +@pytest.mark.asyncio +async def test_a_tool_from_another_project_is_rejected() -> None: + transport = SequencedTransport( + [response(200, {"key": "lookup_order", "version": 4, "schema": {}})] + ) + other = init_evaluations( + project_key="other-proj", + api_key="token", + sdk_key="sdk-key", + transport=transport, + ) + foreign_tool = other.tools.get("lookup_order", implementation=lookup_order) + evals = init_evaluations( + project_key="proj", + api_key="token", + sdk_key="sdk-key", + transport=SequencedTransport([]), + ) + + with pytest.raises(EvaluationsError, match="cannot run in project 'proj'"): + await evals.run( + key="eval-key", + dataset="golden", + handler=successful_handler, + tools=[foreign_tool], + generation={"provider": "OpenAI", "model": "gpt-4o"}, + )