From 4be7cb6bafa12c969cfa3f9536b2ac43576edd93 Mon Sep 17 00:00:00 2001 From: Tien Le Date: Fri, 18 Sep 2026 14:09:46 +0000 Subject: [PATCH 1/6] Add installed Codex tracing journey --- tests/integration/test_ug_codex_tracing.py | 68 ++++++++++++++++++++++ tests/integration/utils/sql.py | 47 +++++++++++++++ 2 files changed, 115 insertions(+) create mode 100644 tests/integration/test_ug_codex_tracing.py create mode 100644 tests/integration/utils/sql.py diff --git a/tests/integration/test_ug_codex_tracing.py b/tests/integration/test_ug_codex_tracing.py new file mode 100644 index 000000000..9bcdaa835 --- /dev/null +++ b/tests/integration/test_ug_codex_tracing.py @@ -0,0 +1,68 @@ +"""Installed-product customer journey for Codex OpenTelemetry export.""" + +import os +import time +import uuid + +import pytest +from utils.constants import CODEX_TEST_MODEL +from utils.evidence import FileTask +from utils.sql import query_count + +pytestmark = [pytest.mark.tracing, pytest.mark.codex] + + +def test_ug_codex_exports_trace_to_configured_table(live_session, workspace): + """Scenario: configure tracing and run Codex with a unique prompt marker. + + Expected: after the 30-second ingestion window, the configured tracing table + contains a Codex span carrying that marker, and the real agent task completed. + """ + session = live_session + marker = f"ug-codex-trace-{uuid.uuid4().hex}" + task = FileTask(session) + session.run( + "configure", + "--agents", + "codex", + "--workspace", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + assert session.workspace_state().get("codex_otel_tracing") is True + + result = session.run( + "codex", + "--", + "--config", + f'otel.span_attributes.ug_integration_marker="{marker}"', + "exec", + "--skip-git-repo-check", + "--json", + "--model", + CODEX_TEST_MODEL, + f"{task.prompt} Trace correlation marker: {marker}", + timeout=180, + ) + task.assert_headless_answer("codex", result) + + time.sleep(30) + table = os.environ.get("UG_INTEGRATION_TRACE_TABLE", "").strip() + warehouse_id = os.environ.get("UG_INTEGRATION_WAREHOUSE_ID", "").strip() + assert table, "Pass --trace-table for the staging tracing table" + assert warehouse_id, "Pass --warehouse-id for the staging SQL warehouse" + count = query_count( + workspace, + session.env["DATABRICKS_BEARER"], + warehouse_id, + ( + f"SELECT COUNT(*) FROM {table} " + "WHERE time > current_timestamp() - INTERVAL 10 MINUTES " + "AND variant_get(attributes, '$[\"ug_integration_marker\"]', 'STRING') = :marker" + ), + [{"name": "marker", "value": marker, "type": "STRING"}], + ) + session.record("trace-query.json", {"marker": marker, "table": table, "count": count}) + assert count > 0 diff --git a/tests/integration/utils/sql.py b/tests/integration/utils/sql.py new file mode 100644 index 000000000..33866f632 --- /dev/null +++ b/tests/integration/utils/sql.py @@ -0,0 +1,47 @@ +"""Query a Databricks SQL warehouse through the public Statement Execution API.""" + +from __future__ import annotations + +import json +import time +import urllib.request + + +def query_count( + workspace: str, + bearer: str, + warehouse_id: str, + statement: str, + parameters: list[dict[str, str]], +) -> int: + request = urllib.request.Request( + f"{workspace.rstrip('/')}/api/2.0/sql/statements", + data=json.dumps( + { + "warehouse_id": warehouse_id, + "statement": statement, + "parameters": parameters, + "wait_timeout": "50s", + "on_wait_timeout": "CONTINUE", + } + ).encode(), + headers={"Authorization": f"Bearer {bearer}", "Content-Type": "application/json"}, + ) + with urllib.request.urlopen(request, timeout=60) as response: # noqa: S310 + result = json.load(response) + + deadline = time.monotonic() + 180 + while result.get("status", {}).get("state") in {"PENDING", "RUNNING"}: + assert time.monotonic() < deadline, "SQL statement did not finish within 180 seconds" + time.sleep(2) + poll = urllib.request.Request( + f"{workspace.rstrip('/')}/api/2.0/sql/statements/{result['statement_id']}", + headers={"Authorization": f"Bearer {bearer}"}, + ) + with urllib.request.urlopen(poll, timeout=30) as response: # noqa: S310 + result = json.load(response) + + assert result.get("status", {}).get("state") == "SUCCEEDED", result.get("status") + rows = result.get("result", {}).get("data_array", []) + assert rows and rows[0], "SQL count query returned no row" + return int(rows[0][0]) From 6d8662557d9e3c98ffa9e9331262042ba474de27 Mon Sep 17 00:00:00 2001 From: Tien Le Date: Fri, 18 Sep 2026 19:16:04 +0000 Subject: [PATCH 2/6] Run Codex tracing in full integration suite --- .github/workflows/integration.yml | 3 ++- tests/integration/test_ug_codex_tracing.py | 28 ++++++++++++++-------- tests/integration/utils/managed.py | 3 +++ tests/integration/utils/sql.py | 21 ++++++++++++++++ 4 files changed, 44 insertions(+), 11 deletions(-) diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index 73692905b..85a44b9d9 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -241,7 +241,8 @@ jobs: uv run --no-project --python 3.12 python scripts/run_integration.py \ --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ --default-index "$PACKAGE_INDEX" --output "$RUNNER_TEMP/ug-integration" \ - "${args[@]}" -- -m "(managed or managed_fixture or workspace_switch) and $AGENT" + "${args[@]}" -- -m \ + "(managed or (managed_fixture and not live) or workspace_switch) and $AGENT" - name: Upload managed test evidence if: ${{ !cancelled() }} uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 diff --git a/tests/integration/test_ug_codex_tracing.py b/tests/integration/test_ug_codex_tracing.py index 9bcdaa835..5e183767a 100644 --- a/tests/integration/test_ug_codex_tracing.py +++ b/tests/integration/test_ug_codex_tracing.py @@ -7,24 +7,32 @@ import pytest from utils.constants import CODEX_TEST_MODEL from utils.evidence import FileTask -from utils.sql import query_count +from utils.managed import ( + build_codex_agent_config, + build_coding_agent_config, + set_managed_config_stub, +) +from utils.sql import query_count, resolve_warehouse_id -pytestmark = [pytest.mark.tracing, pytest.mark.codex] +pytestmark = [pytest.mark.live, pytest.mark.managed_fixture, pytest.mark.codex] -def test_ug_codex_exports_trace_to_configured_table(live_session, workspace): - """Scenario: configure tracing and run Codex with a unique prompt marker. +def test_ug_codex_exports_trace_to_configured_table(live_session, workspace, tmp_path): + """Scenario: configure Codex with tracing enabled and run a task with a unique marker. - Expected: after the 30-second ingestion window, the configured tracing table - contains a Codex span carrying that marker, and the real agent task completed. + Expected: the real agent task completes and, after the ingestion window, the + configured trace table contains a Codex span carrying the same marker. """ session = live_session marker = f"ug-codex-trace-{uuid.uuid4().hex}" task = FileTask(session) + config = build_coding_agent_config( + "CODING_AGENT_CODEX", + build_codex_agent_config(models=[CODEX_TEST_MODEL], otel_tracing_enabled=True), + ) + set_managed_config_stub(session, tmp_path, config) session.run( "configure", - "--agents", - "codex", "--workspace", workspace, "--skip-validate", @@ -50,9 +58,9 @@ def test_ug_codex_exports_trace_to_configured_table(live_session, workspace): time.sleep(30) table = os.environ.get("UG_INTEGRATION_TRACE_TABLE", "").strip() + assert table, "Pass --trace-table for the configured tracing table" warehouse_id = os.environ.get("UG_INTEGRATION_WAREHOUSE_ID", "").strip() - assert table, "Pass --trace-table for the staging tracing table" - assert warehouse_id, "Pass --warehouse-id for the staging SQL warehouse" + warehouse_id = warehouse_id or resolve_warehouse_id(workspace, session.env["DATABRICKS_BEARER"]) count = query_count( workspace, session.env["DATABRICKS_BEARER"], diff --git a/tests/integration/utils/managed.py b/tests/integration/utils/managed.py index 41c86817e..370e62557 100644 --- a/tests/integration/utils/managed.py +++ b/tests/integration/utils/managed.py @@ -134,6 +134,7 @@ def build_codex_agent_config( models: list[str], smart_routing: bool = False, http_headers: dict[str, str] | None = None, + otel_tracing_enabled: bool | None = None, ) -> dict: config = { "models": {"model_services": models}, @@ -143,6 +144,8 @@ def build_codex_agent_config( config["smart_routing"] = {"enabled": True} if http_headers is not None: config["http_headers"] = http_headers + if otel_tracing_enabled is not None: + config["tracing"] = {"enabled": otel_tracing_enabled} return {"agent": "CODING_AGENT_CODEX", "config": config} diff --git a/tests/integration/utils/sql.py b/tests/integration/utils/sql.py index 33866f632..f37bce14e 100644 --- a/tests/integration/utils/sql.py +++ b/tests/integration/utils/sql.py @@ -7,6 +7,27 @@ import urllib.request +def resolve_warehouse_id(workspace: str, bearer: str) -> str: + """Choose an existing warehouse, preferring one that is already running.""" + request = urllib.request.Request( + f"{workspace.rstrip('/')}/api/2.0/sql/warehouses", + headers={"Authorization": f"Bearer {bearer}"}, + ) + with urllib.request.urlopen(request, timeout=30) as response: # noqa: S310 + result = json.load(response) + + warehouses = [ + warehouse + for warehouse in result.get("warehouses", []) + if isinstance(warehouse, dict) and warehouse.get("id") + ] + assert warehouses, "The integration workspace has no SQL warehouse" + running = next( + (warehouse for warehouse in warehouses if warehouse.get("state") == "RUNNING"), None + ) + return str((running or warehouses[0])["id"]) + + def query_count( workspace: str, bearer: str, From 2eaf9c2545f9cfbd8451394b4c16bddf836dcb23 Mon Sep 17 00:00:00 2001 From: Tien Le Date: Fri, 18 Sep 2026 23:05:49 +0000 Subject: [PATCH 3/6] Verify model attribute in Codex tracing journey Point the tracing journey at the CI trace table main.aigw_tracing.unity_gateway_otel_spans and assert the marker span also carries Codex's `model` attribute, so the trace is attributable to the model that actually ran. Co-authored-by: Isaac --- tests/integration/test_ug_codex_tracing.py | 32 +++++++++++++++++----- 1 file changed, 25 insertions(+), 7 deletions(-) diff --git a/tests/integration/test_ug_codex_tracing.py b/tests/integration/test_ug_codex_tracing.py index 5e183767a..34ef63eb3 100644 --- a/tests/integration/test_ug_codex_tracing.py +++ b/tests/integration/test_ug_codex_tracing.py @@ -61,16 +61,34 @@ def test_ug_codex_exports_trace_to_configured_table(live_session, workspace, tmp assert table, "Pass --trace-table for the configured tracing table" warehouse_id = os.environ.get("UG_INTEGRATION_WAREHOUSE_ID", "").strip() warehouse_id = warehouse_id or resolve_warehouse_id(workspace, session.env["DATABRICKS_BEARER"]) + bearer = session.env["DATABRICKS_BEARER"] + marker_query = ( + f"SELECT COUNT(*) FROM {table} " + "WHERE time > current_timestamp() - INTERVAL 10 MINUTES " + "AND variant_get(attributes, '$[\"ug_integration_marker\"]', 'STRING') = :marker" + ) count = query_count( workspace, - session.env["DATABRICKS_BEARER"], + bearer, warehouse_id, - ( - f"SELECT COUNT(*) FROM {table} " - "WHERE time > current_timestamp() - INTERVAL 10 MINUTES " - "AND variant_get(attributes, '$[\"ug_integration_marker\"]', 'STRING') = :marker" - ), + marker_query, [{"name": "marker", "value": marker, "type": "STRING"}], ) - session.record("trace-query.json", {"marker": marker, "table": table, "count": count}) + # The span must also carry Codex's `model` attribute matching the model that ran, + # so the trace is attributable to a specific model and not just to this test run. + model_count = query_count( + workspace, + bearer, + warehouse_id, + marker_query + " AND variant_get(attributes, '$[\"model\"]', 'STRING') = :model", + [ + {"name": "marker", "value": marker, "type": "STRING"}, + {"name": "model", "value": CODEX_TEST_MODEL, "type": "STRING"}, + ], + ) + session.record( + "trace-query.json", + {"marker": marker, "table": table, "count": count, "model_count": model_count}, + ) assert count > 0 + assert model_count > 0, f"Codex span for {marker} lacked model={CODEX_TEST_MODEL}" From 8d92f7caf7215ad6b28243ebd9e1a053b6161190 Mon Sep 17 00:00:00 2001 From: Tien Le Date: Mon, 21 Sep 2026 14:07:23 +0000 Subject: [PATCH 4/6] Fix Codex tracing journey state assertion --- tests/integration/test_ug_codex_tracing.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/tests/integration/test_ug_codex_tracing.py b/tests/integration/test_ug_codex_tracing.py index 34ef63eb3..177f1c117 100644 --- a/tests/integration/test_ug_codex_tracing.py +++ b/tests/integration/test_ug_codex_tracing.py @@ -39,8 +39,6 @@ def test_ug_codex_exports_trace_to_configured_table(live_session, workspace, tmp "--skip-upgrade", "--disable-databricks-ai-tools", ) - assert session.workspace_state().get("codex_otel_tracing") is True - result = session.run( "codex", "--", From dd65d6df8103f3bf55d851c457242bae93eafb88 Mon Sep 17 00:00:00 2001 From: Tien Le Date: Mon, 21 Sep 2026 17:08:13 +0000 Subject: [PATCH 5/6] Resolve tracing table during integration setup --- scripts/run_integration.py | 3 ++ tests/integration/test_ug_codex_tracing.py | 13 ++++---- tests/integration/utils/sql.py | 39 +++++++++++++++++++++- 3 files changed, 47 insertions(+), 8 deletions(-) diff --git a/scripts/run_integration.py b/scripts/run_integration.py index fb04b5046..0c6183fb1 100644 --- a/scripts/run_integration.py +++ b/scripts/run_integration.py @@ -152,6 +152,7 @@ def arguments(): default=os.environ.get("UCODE_TEST_SECOND_WORKSPACE"), help="Second real workspace for workspace_switch CUJs; requires DATABRICKS_SECOND_BEARER.", ) + parser.add_argument("--warehouse-id", default=os.environ.get("UG_INTEGRATION_WAREHOUSE_ID")) parser.add_argument("--output", type=Path, help="New results directory; never reused.") parser.add_argument("--installation-only", action="store_true", help="No workspace calls.") parser.add_argument( @@ -323,6 +324,7 @@ def run(command, *, cwd=output, env=base_env, timeout=600) -> str: "dependencies": args.dependency, "workspace": args.workspace, "second_workspace": args.second_workspace, + "warehouse_id": args.warehouse_id, }, "platform": platform.platform(), "installation_only": args.installation_only, @@ -569,6 +571,7 @@ def run(command, *, cwd=output, env=base_env, timeout=600) -> str: "DATABRICKS_BEARER": bearer, "UCODE_TEST_SECOND_WORKSPACE": args.second_workspace or "", "DATABRICKS_SECOND_BEARER": second_bearer, + "UG_INTEGRATION_WAREHOUSE_ID": args.warehouse_id or "", } ) for agent in agents: diff --git a/tests/integration/test_ug_codex_tracing.py b/tests/integration/test_ug_codex_tracing.py index 177f1c117..3388a53a8 100644 --- a/tests/integration/test_ug_codex_tracing.py +++ b/tests/integration/test_ug_codex_tracing.py @@ -12,18 +12,22 @@ build_coding_agent_config, set_managed_config_stub, ) -from utils.sql import query_count, resolve_warehouse_id +from utils.sql import query_count, resolve_trace_table, resolve_warehouse_id pytestmark = [pytest.mark.live, pytest.mark.managed_fixture, pytest.mark.codex] def test_ug_codex_exports_trace_to_configured_table(live_session, workspace, tmp_path): - """Scenario: configure Codex with tracing enabled and run a task with a unique marker. + """Scenario: resolve the trace table, configure Codex, and run a uniquely marked task. Expected: the real agent task completes and, after the ingestion window, the configured trace table contains a Codex span carrying the same marker. """ session = live_session + bearer = session.env["DATABRICKS_BEARER"] + table = resolve_trace_table(workspace, bearer) + warehouse_id = os.environ.get("UG_INTEGRATION_WAREHOUSE_ID", "").strip() + warehouse_id = warehouse_id or resolve_warehouse_id(workspace, bearer) marker = f"ug-codex-trace-{uuid.uuid4().hex}" task = FileTask(session) config = build_coding_agent_config( @@ -55,11 +59,6 @@ def test_ug_codex_exports_trace_to_configured_table(live_session, workspace, tmp task.assert_headless_answer("codex", result) time.sleep(30) - table = os.environ.get("UG_INTEGRATION_TRACE_TABLE", "").strip() - assert table, "Pass --trace-table for the configured tracing table" - warehouse_id = os.environ.get("UG_INTEGRATION_WAREHOUSE_ID", "").strip() - warehouse_id = warehouse_id or resolve_warehouse_id(workspace, session.env["DATABRICKS_BEARER"]) - bearer = session.env["DATABRICKS_BEARER"] marker_query = ( f"SELECT COUNT(*) FROM {table} " "WHERE time > current_timestamp() - INTERVAL 10 MINUTES " diff --git a/tests/integration/utils/sql.py b/tests/integration/utils/sql.py index f37bce14e..03feb222e 100644 --- a/tests/integration/utils/sql.py +++ b/tests/integration/utils/sql.py @@ -1,4 +1,4 @@ -"""Query a Databricks SQL warehouse through the public Statement Execution API.""" +"""Resolve and query Databricks SQL resources through public APIs.""" from __future__ import annotations @@ -6,6 +6,43 @@ import time import urllib.request +DEFAULT_TRACE_TABLE_PREFIX = "unity_gateway" +TRACE_TABLE_SUFFIX = "_otel_spans" + + +def resolve_trace_table(workspace: str, bearer: str) -> str: + """Resolve the OTel span table from the workspace's tracing configuration.""" + workspace_request = urllib.request.Request( + f"{workspace.rstrip('/')}/api/2.0/preview/scim/v2/Me", + headers={"Authorization": f"Bearer {bearer}"}, + ) + with urllib.request.urlopen(workspace_request, timeout=30) as response: # noqa: S310 + workspace_id = response.headers.get("x-databricks-org-id") + + assert workspace_id, "Workspace response did not include x-databricks-org-id" + config_request = urllib.request.Request( + f"{workspace.rstrip('/')}/api/ai-gateway/v2/tracing-config/workspace/{workspace_id}", + headers={"Authorization": f"Bearer {bearer}"}, + ) + with urllib.request.urlopen(config_request, timeout=30) as response: # noqa: S310 + config = json.load(response) + + assert config.get("enabled") is True, "Workspace tracing is not enabled" + catalog = config.get("catalog_name") + schema = config.get("schema_name") + prefix = config.get("table_name_prefix") or DEFAULT_TRACE_TABLE_PREFIX + assert isinstance(catalog, str) and catalog, "Tracing config has no catalog_name" + assert isinstance(schema, str) and schema, "Tracing config has no schema_name" + assert isinstance(prefix, str), "Tracing config has an invalid table_name_prefix" + return ".".join( + _quote_identifier(identifier) + for identifier in (catalog, schema, prefix + TRACE_TABLE_SUFFIX) + ) + + +def _quote_identifier(identifier: str) -> str: + return f"`{identifier.replace('`', '``')}`" + def resolve_warehouse_id(workspace: str, bearer: str) -> str: """Choose an existing warehouse, preferring one that is already running.""" From 881da930afb42740ffcc1682c8f36090e353688d Mon Sep 17 00:00:00 2001 From: Tien Le Date: Tue, 22 Sep 2026 21:39:03 +0000 Subject: [PATCH 6/6] Add Claude tracing integration journey --- .github/workflows/integration.yml | 3 +- tests/README.md | 7 +- tests/integration/README.md | 29 +++++-- tests/integration/test_ug_claude_tracing.py | 91 +++++++++++++++++++++ tests/integration/test_ug_codex_tracing.py | 2 +- tests/integration/utils/constants.py | 1 + tests/integration/utils/managed.py | 3 + 7 files changed, 122 insertions(+), 14 deletions(-) create mode 100644 tests/integration/test_ug_claude_tracing.py diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index 85a44b9d9..73692905b 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -241,8 +241,7 @@ jobs: uv run --no-project --python 3.12 python scripts/run_integration.py \ --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ --default-index "$PACKAGE_INDEX" --output "$RUNNER_TEMP/ug-integration" \ - "${args[@]}" -- -m \ - "(managed or (managed_fixture and not live) or workspace_switch) and $AGENT" + "${args[@]}" -- -m "(managed or managed_fixture or workspace_switch) and $AGENT" - name: Upload managed test evidence if: ${{ !cancelled() }} uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 diff --git a/tests/README.md b/tests/README.md index 73dcba4b1..f672f431c 100644 --- a/tests/README.md +++ b/tests/README.md @@ -46,6 +46,7 @@ All tests live directly in `integration/`; shared mechanics live in `utils/`. | `test_case_14_*` | Launch configured and fresh Codex with a model location | The app-server list exactly matches the independent API-compatible parent catalog, includes the dedicated Codex service, and contains no out-of-schema models | | `test_ug_claude_headless_prompt_argument`, `test_ug_claude_headless_prompt_stdin`, `test_ug_claude_headless_prompt_after_separator` | Run Claude from a script using each prompt form | Structured final answer contains the file value; exit zero; no routing | | `test_ug_codex_headless_prompt_argument`, `test_ug_codex_headless_prompt_stdin`, `test_ug_codex_headless_prompt_after_separator` | Run Codex from a script using each prompt form | Completed turn and final answer contain the file value; exit zero; no routing | +| `test_ug_claude_exports_trace_to_configured_table`, `test_ug_codex_exports_trace_to_configured_table` | Configure tracing, complete a headless task carrying a unique trace marker, then wait for ingestion | The configured trace table contains an agent span with the same trace-safe marker and requested model | | `test_ug_claude_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` with routing enabled | Real file task completes; no routing wrapper | | `test_ug_codex_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` / `-m VALUE` with routing enabled | Real file task completes; no routing wrapper | | `test_ug_claude_preserves_caller_settings_and_hook` | Pass a settings path containing spaces | Real SessionStart hook executes; caller file unchanged; file task completes | @@ -74,7 +75,7 @@ All tests live directly in `integration/`; shared mechanics live in `utils/`. | `test_ug_and_ucode_auth_helpers_emit_only_the_supplied_bearer` | Run both auth helper commands with the public bearer override, with and without forced refresh | Exact token-only stdout, no warnings or ANSI escapes; no workspace authentication or saved state | | `test_ug_and_ucode_web_search_helpers_preserve_mcp_stdio` | Initialize and list tools through both web-search helper commands | Exactly the MCP JSON-RPC responses; no text/ANSI contamination; existing server/tool identities preserved; no model request | -With both agents selected there are **58 live cases** (12 interactive TUI cases), +With both agents selected there are **60 live cases** (12 interactive TUI cases), **4 managed-workspace cases** (marker `managed`, run against a separate workspace that publishes a CodingAgentConfig), **1 two-workspace case** (marker `workspace_switch`), **27 managed-fixture cases** (marker `managed_fixture`, with only @@ -121,7 +122,7 @@ dependency graph to reproduce a user's combination. Every relevant same-reposito PR and push to `main` runs both smoke and the full CUJ suite. Smoke covers the Databricks Hosted configure/TUI, custom OAuth CLI TUI, and headless argument journeys for both agents, in two parallel jobs. After smoke finishes, the full -suite runs all 58 live cases across two parallel agent jobs: one Claude VM and one +suite runs all 60 live cases across two parallel agent jobs: one Claude VM and one Codex VM, each running its configure, headless, and commands/lifecycle cases serially. Each agent is installed once for the full suite, and no two full jobs for the same agent overlap within a run. @@ -152,7 +153,7 @@ pending. The descriptive jobs provide the actual coverage and diagnostics. | Scenario | Status / requirement | | --- | --- | | Live MCP and skills functionality | Deferred; installation tests cover the local web-search MCP handshake and tool listing, not upstream proxying or a real search request | -| Broad configure flags, tracing, multiple workspaces, and PAT flows | Deferred while focusing on basic CUJs | +| Broad configure flags, multiple workspaces, and PAT flows | Deferred while focusing on basic CUJs | | Workspace-switch MCP cleanup | The `workspace_switch` CUJ covers real registration, cleanup, repeat configure, and a completed Claude task. Unit/component tests cover duplicate attempts and injected removal failures; the CUJ does not force an agent timeout. It runs in the existing non-blocking managed CI lane. | | Relayed/subscription MPS discovery | Not covered by the scoped discovery journeys | | Fresh provider/parent validation and mixed Bedrock filtering | Not covered after removing the duplicate model-discovery suites | diff --git a/tests/integration/README.md b/tests/integration/README.md index 32acf61ab..dc5d84fb0 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -100,7 +100,9 @@ test_ug_claude_custom_oauth.py # CLI custom-OAuth launch, profile, and test_ug_codex_custom_oauth.py # CLI custom-OAuth launch and profile test_ug_claude_headless.py # script prompts, models, caller settings test_ug_claude_relayed.py # relayed session: subscription + Databricks-hosted models +test_ug_claude_tracing.py # Claude OTLP export reaches the configured trace table test_ug_codex_headless.py # script prompts and model arguments +test_ug_codex_tracing.py # Codex OTLP export reaches the configured trace table test_ug_claude_commands.py # command help forwarding test_ug_codex_commands.py # command help and parser error forwarding test_ug_codex_app_server.py # actual client/server initialize exchange @@ -175,6 +177,15 @@ reproduce another existing service. Use `--claude-provider-model` / No service is created or modified. A missing service, permission, or OAuth token fails the selected CUJ, rather than skipping it. +The tracing journeys are part of their respective Full agent lanes and use the existing +e2e workspace and bearer. Because that workspace deliberately has no published managed +configuration, each journey injects only a tracing-enabled CodingAgentConfig input through +the suite's managed-config stub seam; the agents, inference, OTLP export, and table +verification remain real. Each adds the prompt's UUID as a trace-safe +`ug_integration_marker` attribute, resolves the destination table from the workspace tracing +configuration, waits 30 seconds, and queries that table through an existing SQL warehouse. +The tests assert that a span with the marker arrived and identifies the requested model. + Scoped discovery additionally requires Model Services `main.ucode.ci_e2e_claude` and `main.ucode.ci_e2e_codex`. Override them with `--parent-schema`, `--claude-parent-model`, or `--codex-parent-model`. The tests @@ -216,7 +227,7 @@ startup banners and footer text cannot satisfy discovery assertions. Cases 7–1 they only configure, list models, and open/close the picker. Other live CUJs perform real model tasks. -There are **58 live cases** (including 12 TUI journeys) and **7 installation +There are **60 live cases** (including 12 TUI journeys) and **7 installation checks** with both agents. A separate **4 managed-workspace cases** (one per agent, an idempotent re-configure, and a cache-TTL journey; marker `managed`) run against a workspace that publishes a CodingAgentConfig; see "Managed-workspace journeys" below. One **`workspace_switch` case** @@ -229,7 +240,7 @@ cover focused model, MCP, skills, and lifecycle shapes, including per-agent mode and managed skill cleanup. The two Claude default-model cases launch with injected MPS and Unity Catalog sources and verify both generated settings files retain all admin-authored family defaults. The 14 retained numbered scenarios comprise 24 explicit journeys: 12 managed and 12 unmanaged -executions; the complete integration suite collects 97 executions. See the named coverage and gaps matrix in +executions; the complete integration suite collects 99 executions. See the named coverage and gaps matrix in [../README.md](../README.md). ```bash @@ -237,6 +248,8 @@ executions; the complete integration suite collects 97 executions. See the named -- -m live # default: all live user journeys -- -m smoke # six Hosted, custom OAuth CLI TUI, and headless journeys -- -m 'live and tui' # twelve interactive live configuration/model-discovery journeys +-- -m 'live and claude' -k trace # installed Claude -> gateway -> configured trace table +-- -m 'live and codex' -k trace # installed Codex -> gateway -> configured trace table -- -k test_ug_codex_app_server_client_initializes # one named journey and its variants # Use --installation-only before -- for package checks without credentials. ``` @@ -245,7 +258,7 @@ The old focused checks are now descriptive CUJs with setup and outcomes visible in each test. Duplicate boot-only checks are incorporated into the Databricks configuration TUI journeys. Real failures, including generated config left after revert and banners on app-server stdout, remain assertions. -Live MCP/skills functionality, tracing, the broad configure-option matrix, and other +Live MCP/skills functionality, the broad configure-option matrix, and other agents are outside this focused revision. The workspace-switch CUJ is an exception to that deferred multi-workspace scope: @@ -338,13 +351,13 @@ each test; only explicit-model scenarios choose and record a discovered Every same-repository PR and push to `main` runs **Smoke journeys**, followed by **Full journeys** even if smoke fails. Smoke runs the Hosted configure/TUI, headless argument, and custom OAuth CLI TUI journeys for each agent (six cases, -two agent jobs). Full runs all 58 live cases, including those smoke cases, in two +two agent jobs). Full runs all 60 live cases, including those smoke cases, in two disjoint agent lanes: | Agent lane | Marker | Cases | | --- | --- | --- | -| Claude | `live and claude` | 25 | -| Codex | `live and codex` | 33 | +| Claude | `live and claude` | 26 | +| Codex | `live and codex` | 34 | Each lane installs only its agent CLI, once, and runs all its configure, headless, commands, lifecycle, and applicable app-server journeys. Cases remain serial @@ -358,7 +371,7 @@ No test retries or assertion changes compensate for capacity failures. Both matrices use `fail-fast: false` and upload uniquely named evidence even when the other agent fails. The **All integration tests** check requires installation, workspace validation, smoke, and -both full lanes to pass. The **Managed config** lanes run for signal but are temporarily +both full lanes to pass; each tracing journey is included in its agent's Full lane. The **Managed config** lanes run for signal but are temporarily non-blocking (`continue-on-error`): the managed workspace is now runner-reachable, but the lanes stay non-blocking until the managed-config apply path is proven stable. They neither fail the workflow nor gate merges until then. The @@ -623,7 +636,7 @@ uv run --no-project --python 3.12 python scripts/run_integration.py \ unset DATABRICKS_BEARER ``` -This runs all 58 live cases. For the seven installation checks, run the same +This runs all 60 live cases. For the seven installation checks, run the same runner/version/index arguments with `--installation-only` and omit `-- -m live`; no bearer or workspace is needed. Results remain under `.integration-runs/`. Each invocation needs a new output directory; an existing one is rejected. diff --git a/tests/integration/test_ug_claude_tracing.py b/tests/integration/test_ug_claude_tracing.py new file mode 100644 index 000000000..ecc8b8f9a --- /dev/null +++ b/tests/integration/test_ug_claude_tracing.py @@ -0,0 +1,91 @@ +"""Installed-product customer journey for Claude Code OpenTelemetry export.""" + +import os +import time +import uuid + +import pytest +from utils.constants import CLAUDE_TEST_MODEL +from utils.evidence import FileTask +from utils.managed import ( + build_claude_agent_config, + build_coding_agent_config, + set_managed_config_stub, +) +from utils.sql import query_count, resolve_trace_table, resolve_warehouse_id + +pytestmark = [pytest.mark.live, pytest.mark.claude] + + +def test_ug_claude_exports_trace_to_configured_table(live_session, workspace, tmp_path): + """Scenario: resolve the trace table, configure Claude, and run a uniquely marked task. + + Expected: the real agent task completes and, after the ingestion window, the + configured trace table contains a Claude span carrying the same marker. + """ + session = live_session + bearer = session.env["DATABRICKS_BEARER"] + table = resolve_trace_table(workspace, bearer) + warehouse_id = os.environ.get("UG_INTEGRATION_WAREHOUSE_ID", "").strip() + warehouse_id = warehouse_id or resolve_warehouse_id(workspace, bearer) + marker = f"ug-claude-trace-{uuid.uuid4().hex}" + task = FileTask(session) + config = build_coding_agent_config( + "CODING_AGENT_CLAUDE_CODE", + build_claude_agent_config([CLAUDE_TEST_MODEL], otel_tracing_enabled=True), + ) + set_managed_config_stub(session, tmp_path, config) + session.run( + "configure", + "--workspace", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["OTEL_RESOURCE_ATTRIBUTES"] = f"ug_integration_marker={marker}" + result = session.run( + "claude", + "--", + "-p", + f"{task.prompt} Trace correlation marker: {marker}", + "--output-format", + "json", + "--allowedTools", + "Read", + "--model", + CLAUDE_TEST_MODEL, + timeout=180, + ) + task.assert_headless_answer("claude", result) + + time.sleep(30) + marker_query = ( + f"SELECT COUNT(*) FROM {table} " + "WHERE time > current_timestamp() - INTERVAL 10 MINUTES " + "AND variant_get(resource.attributes, '$[\"ug_integration_marker\"]', 'STRING') = :marker" + ) + count = query_count( + workspace, + bearer, + warehouse_id, + marker_query, + [{"name": "marker", "value": marker, "type": "STRING"}], + ) + model_count = query_count( + workspace, + bearer, + warehouse_id, + marker_query + + " AND variant_get(attributes, '$[\"gen_ai.request.model\"]', 'STRING') = :model", + [ + {"name": "marker", "value": marker, "type": "STRING"}, + {"name": "model", "value": CLAUDE_TEST_MODEL, "type": "STRING"}, + ], + ) + session.record( + "trace-query.json", + {"marker": marker, "table": table, "count": count, "model_count": model_count}, + ) + assert count > 0 + assert model_count > 0, f"Claude span for {marker} lacked model={CLAUDE_TEST_MODEL}" diff --git a/tests/integration/test_ug_codex_tracing.py b/tests/integration/test_ug_codex_tracing.py index 3388a53a8..be935837e 100644 --- a/tests/integration/test_ug_codex_tracing.py +++ b/tests/integration/test_ug_codex_tracing.py @@ -14,7 +14,7 @@ ) from utils.sql import query_count, resolve_trace_table, resolve_warehouse_id -pytestmark = [pytest.mark.live, pytest.mark.managed_fixture, pytest.mark.codex] +pytestmark = [pytest.mark.live, pytest.mark.codex] def test_ug_codex_exports_trace_to_configured_table(live_session, workspace, tmp_path): diff --git a/tests/integration/utils/constants.py b/tests/integration/utils/constants.py index f5c6374c4..57cfab96c 100644 --- a/tests/integration/utils/constants.py +++ b/tests/integration/utils/constants.py @@ -1,5 +1,6 @@ """Shared constants for the integration CUJs.""" +CLAUDE_TEST_MODEL = "system.ai.claude-haiku-4-5" CODEX_TEST_MODEL = "system.ai.gpt-5-4-nano" MANAGED_CLAUDE_PROVIDER_SERVICE = "main.default.ci_e2e_anthropic_mps" diff --git a/tests/integration/utils/managed.py b/tests/integration/utils/managed.py index 370e62557..3599d38f9 100644 --- a/tests/integration/utils/managed.py +++ b/tests/integration/utils/managed.py @@ -114,6 +114,7 @@ def build_claude_agent_config( *, family_defaults: dict[str, str] | None = None, smart_routing: bool = False, + otel_tracing_enabled: bool | None = None, ) -> dict: default_models = {"default_model": models[0]} if family_defaults: @@ -126,6 +127,8 @@ def build_claude_agent_config( } if smart_routing: config["smart_routing"] = {"enabled": True} + if otel_tracing_enabled is not None: + config["tracing"] = {"enabled": otel_tracing_enabled} return {"agent": "CODING_AGENT_CLAUDE_CODE", "config": config}