diff --git a/.agents/logs/tone_model_eval/2026-09-10-report.json b/.agents/logs/tone_model_eval/2026-09-10-report.json new file mode 100644 index 00000000..6f0896f7 --- /dev/null +++ b/.agents/logs/tone_model_eval/2026-09-10-report.json @@ -0,0 +1,345 @@ +{ + "scope_boundary": "## What this eval can and cannot claim\nThis eval supports a **relative** claim: which candidate model most closely follows this repo's AGENTS.md tone/concision guidance when every model edits the identical \"before\" text for a fixture. The per-model, per-fixture and aggregate scores below are comparable to each other.\n\nThis eval does **not** support an **absolute** \"how much better than the original\" claim. Each fixture's \"before\" text has an unknown or uncontrolled authorship model, so no candidate's edit can be scored as an absolute improvement over it \u2014 only relative to the other candidates scored on the same fixture.", + "judge_model_id": "gemini-3.1-pro", + "fable_model_id": "claude-5-1-fable-high", + "default_model_id": "claude-4-5-sonnet", + "rows": [ + { + "fixture_id": "cli-agent-conversations-resume-menu-label", + "model_id": "claude-5-1-fable-high", + "mechanical": { + "tone_buzzword": 0, + "tone_meta_opener": 0, + "combined": 0 + }, + "word_count": { + "before_words": 1395, + "after_words": 1324, + "delta": -71, + "delta_pct": -0.05089605734767025 + }, + "judge": { + "concision": 5, + "avoids_over_explaining": 5, + "technical_fidelity": 4 + }, + "judge_composite": 4.666666666666667, + "judge_model_id": "gemini-3.1-pro" + }, + { + "fixture_id": "byollm-gemini-enterprise-google-cloud-setup", + "model_id": "claude-5-1-fable-high", + "mechanical": { + "tone_buzzword": 0, + "tone_meta_opener": 0, + "combined": 0 + }, + "word_count": { + "before_words": 2630, + "after_words": 2555, + "delta": -75, + "delta_pct": -0.028517110266159697 + }, + "judge": { + "concision": 4, + "avoids_over_explaining": 4, + "technical_fidelity": 2 + }, + "judge_composite": 3.3333333333333335, + "judge_model_id": "gemini-3.1-pro" + }, + { + "fixture_id": "quickstart-synthetic-verbose-seed", + "model_id": "claude-5-1-fable-high", + "mechanical": { + "tone_buzzword": 0, + "tone_meta_opener": 0, + "combined": 0 + }, + "word_count": { + "before_words": 235, + "after_words": 107, + "delta": -128, + "delta_pct": -0.5446808510638298 + }, + "judge": { + "concision": 5, + "avoids_over_explaining": 5, + "technical_fidelity": 5 + }, + "judge_composite": 5.0, + "judge_model_id": "gemini-3.1-pro" + }, + { + "fixture_id": "cli-agent-conversations-resume-menu-label", + "model_id": "claude-4-5-sonnet", + "mechanical": { + "tone_buzzword": 0, + "tone_meta_opener": 0, + "combined": 0 + }, + "word_count": { + "before_words": 1395, + "after_words": 1264, + "delta": -131, + "delta_pct": -0.0939068100358423 + }, + "judge": { + "concision": 4, + "avoids_over_explaining": 4, + "technical_fidelity": 2 + }, + "judge_composite": 3.3333333333333335, + "judge_model_id": "gemini-3.1-pro" + }, + { + "fixture_id": "byollm-gemini-enterprise-google-cloud-setup", + "model_id": "claude-4-5-sonnet", + "mechanical": { + "tone_buzzword": 0, + "tone_meta_opener": 0, + "combined": 0 + }, + "word_count": { + "before_words": 2630, + "after_words": 2595, + "delta": -35, + "delta_pct": -0.013307984790874524 + }, + "judge": { + "concision": 4, + "avoids_over_explaining": 4, + "technical_fidelity": 4 + }, + "judge_composite": 4.0, + "judge_model_id": "gemini-3.1-pro" + }, + { + "fixture_id": "quickstart-synthetic-verbose-seed", + "model_id": "claude-4-5-sonnet", + "mechanical": { + "tone_buzzword": 0, + "tone_meta_opener": 0, + "combined": 0 + }, + "word_count": { + "before_words": 235, + "after_words": 98, + "delta": -137, + "delta_pct": -0.5829787234042553 + }, + "judge": { + "concision": 5, + "avoids_over_explaining": 5, + "technical_fidelity": 5 + }, + "judge_composite": 5.0, + "judge_model_id": "gemini-3.1-pro" + }, + { + "fixture_id": "cli-agent-conversations-resume-menu-label", + "model_id": "claude-4-5-haiku", + "mechanical": { + "tone_buzzword": 0, + "tone_meta_opener": 0, + "combined": 0 + }, + "word_count": { + "before_words": 1395, + "after_words": 1251, + "delta": -144, + "delta_pct": -0.1032258064516129 + }, + "judge": { + "concision": 4, + "avoids_over_explaining": 4, + "technical_fidelity": 3 + }, + "judge_composite": 3.6666666666666665, + "judge_model_id": "gemini-3.1-pro" + }, + { + "fixture_id": "byollm-gemini-enterprise-google-cloud-setup", + "model_id": "claude-4-5-haiku", + "mechanical": { + "tone_buzzword": 0, + "tone_meta_opener": 0, + "combined": 0 + }, + "word_count": { + "before_words": 2630, + "after_words": 2579, + "delta": -51, + "delta_pct": -0.019391634980988594 + }, + "judge": { + "concision": 4, + "avoids_over_explaining": 4, + "technical_fidelity": 1 + }, + "judge_composite": 3.0, + "judge_model_id": "gemini-3.1-pro" + }, + { + "fixture_id": "quickstart-synthetic-verbose-seed", + "model_id": "claude-4-5-haiku", + "mechanical": { + "tone_buzzword": 0, + "tone_meta_opener": 0, + "combined": 0 + }, + "word_count": { + "before_words": 235, + "after_words": 88, + "delta": -147, + "delta_pct": -0.625531914893617 + }, + "judge": { + "concision": 5, + "avoids_over_explaining": 5, + "technical_fidelity": 5 + }, + "judge_composite": 5.0, + "judge_model_id": "gemini-3.1-pro" + }, + { + "fixture_id": "cli-agent-conversations-resume-menu-label", + "model_id": "gpt-5-mini", + "mechanical": { + "tone_buzzword": 0, + "tone_meta_opener": 0, + "combined": 0 + }, + "word_count": { + "before_words": 1395, + "after_words": 1188, + "delta": -207, + "delta_pct": -0.14838709677419354 + }, + "judge": { + "concision": 4, + "avoids_over_explaining": 4, + "technical_fidelity": 1 + }, + "judge_composite": 3.0, + "judge_model_id": "gemini-3.1-pro" + }, + { + "fixture_id": "byollm-gemini-enterprise-google-cloud-setup", + "model_id": "gpt-5-mini", + "mechanical": { + "tone_buzzword": 0, + "tone_meta_opener": 0, + "combined": 0 + }, + "word_count": { + "before_words": 2630, + "after_words": 2048, + "delta": -582, + "delta_pct": -0.22129277566539923 + }, + "judge": { + "concision": 4, + "avoids_over_explaining": 4, + "technical_fidelity": 3 + }, + "judge_composite": 3.6666666666666665, + "judge_model_id": "gemini-3.1-pro" + }, + { + "fixture_id": "quickstart-synthetic-verbose-seed", + "model_id": "gpt-5-mini", + "mechanical": { + "tone_buzzword": 0, + "tone_meta_opener": 0, + "combined": 0 + }, + "word_count": { + "before_words": 235, + "after_words": 117, + "delta": -118, + "delta_pct": -0.502127659574468 + }, + "judge": { + "concision": 5, + "avoids_over_explaining": 5, + "technical_fidelity": 1 + }, + "judge_composite": 3.6666666666666665, + "judge_model_id": "gemini-3.1-pro" + } + ], + "aggregates": { + "claude-5-1-fable-high": { + "fixture_count": 3, + "dimension_scores": { + "concision": 4.666666666666667, + "avoids_over_explaining": 4.666666666666667, + "technical_fidelity": 3.6666666666666665 + }, + "composite_judge_score": 4.333333333333333, + "combined_mechanical_violations": 0 + }, + "claude-4-5-sonnet": { + "fixture_count": 3, + "dimension_scores": { + "concision": 4.333333333333333, + "avoids_over_explaining": 4.333333333333333, + "technical_fidelity": 3.6666666666666665 + }, + "composite_judge_score": 4.111111111111112, + "combined_mechanical_violations": 0 + }, + "claude-4-5-haiku": { + "fixture_count": 3, + "dimension_scores": { + "concision": 4.333333333333333, + "avoids_over_explaining": 4.333333333333333, + "technical_fidelity": 3.0 + }, + "composite_judge_score": 3.888888888888889, + "combined_mechanical_violations": 0 + }, + "gpt-5-mini": { + "fixture_count": 3, + "dimension_scores": { + "concision": 4.333333333333333, + "avoids_over_explaining": 4.333333333333333, + "technical_fidelity": 1.6666666666666667 + }, + "composite_judge_score": 3.444444444444444, + "combined_mechanical_violations": 0 + } + }, + "recommendation": { + "adopt_fable_guidance": { + "passed": false, + "concision_margin": 0.3333333333333339, + "mechanical_violation_reduction_pct": 0.0 + }, + "cheaper_model_candidates": { + "claude-4-5-sonnet": { + "passed": false, + "judge_gap": 0.22222222222222143, + "mechanical_violations": 0, + "fable_mechanical_violations": 0, + "technical_fidelity": 3.6666666666666665 + }, + "claude-4-5-haiku": { + "passed": false, + "judge_gap": 0.4444444444444442, + "mechanical_violations": 0, + "fable_mechanical_violations": 0, + "technical_fidelity": 3.0 + }, + "gpt-5-mini": { + "passed": false, + "judge_gap": 0.8888888888888888, + "mechanical_violations": 0, + "fable_mechanical_violations": 0, + "technical_fidelity": 1.6666666666666667 + } + } + }, + "calibration_warning": null +} \ No newline at end of file diff --git a/.agents/logs/tone_model_eval/2026-09-10-report.md b/.agents/logs/tone_model_eval/2026-09-10-report.md new file mode 100644 index 00000000..8fe716c6 --- /dev/null +++ b/.agents/logs/tone_model_eval/2026-09-10-report.md @@ -0,0 +1,51 @@ +# Tone/concision model eval report + +## What this eval can and cannot claim +This eval supports a **relative** claim: which candidate model most closely follows this repo's AGENTS.md tone/concision guidance when every model edits the identical "before" text for a fixture. The per-model, per-fixture and aggregate scores below are comparable to each other. + +This eval does **not** support an **absolute** "how much better than the original" claim. Each fixture's "before" text has an unknown or uncontrolled authorship model, so no candidate's edit can be scored as an absolute improvement over it — only relative to the other candidates scored on the same fixture. + +**Judge model:** gemini-3.1-pro (check for a same-family match against any candidate before trusting its score) + +## Per-model scores +### claude-5-1-fable-high +- Fixtures scored: 3 +- Composite judge score: 4.33/5 + - concision: 4.67/5 + - avoids_over_explaining: 4.67/5 + - technical_fidelity: 3.67/5 +- Combined mechanical violations (tone-buzzword + tone-meta-opener): 0 + +### claude-4-5-sonnet +- Fixtures scored: 3 +- Composite judge score: 4.11/5 + - concision: 4.33/5 + - avoids_over_explaining: 4.33/5 + - technical_fidelity: 3.67/5 +- Combined mechanical violations (tone-buzzword + tone-meta-opener): 0 + +### claude-4-5-haiku +- Fixtures scored: 3 +- Composite judge score: 3.89/5 + - concision: 4.33/5 + - avoids_over_explaining: 4.33/5 + - technical_fidelity: 3.00/5 +- Combined mechanical violations (tone-buzzword + tone-meta-opener): 0 + +### gpt-5-mini +- Fixtures scored: 3 +- Composite judge score: 3.44/5 + - concision: 4.33/5 + - avoids_over_explaining: 4.33/5 + - technical_fidelity: 1.67/5 +- Combined mechanical violations (tone-buzzword + tone-meta-opener): 0 + +## Recommendation +### Adopt Fable-5.1-derived guidance +**No meaningful difference found** — neither the ≥1.0-point concision-dimension margin (0.33 observed) nor the ≥30% mechanical-violation reduction (0% observed) was met. + +### Recommend a cheaper model for production copy passes +**No candidate met the threshold** — every candidate either fell outside the 0.5-point judge tolerance, exceeded Fable 5.1's mechanical violation count, or scored below 4/5 on technical fidelity. +- claude-4-5-sonnet: Fail — judge gap 0.22 (tolerance ≤0.5), mechanical violations 0 vs Fable 5.1's 0, technical fidelity 3.67/5 (min 4.0) +- claude-4-5-haiku: Fail — judge gap 0.44 (tolerance ≤0.5), mechanical violations 0 vs Fable 5.1's 0, technical fidelity 3.00/5 (min 4.0) +- gpt-5-mini: Fail — judge gap 0.89 (tolerance ≤0.5), mechanical violations 0 vs Fable 5.1's 0, technical fidelity 1.67/5 (min 4.0) diff --git a/.agents/logs/tone_model_eval/2026-09-10-rows.jsonl b/.agents/logs/tone_model_eval/2026-09-10-rows.jsonl new file mode 100644 index 00000000..35bceeb8 --- /dev/null +++ b/.agents/logs/tone_model_eval/2026-09-10-rows.jsonl @@ -0,0 +1,12 @@ +{"fixture_id": "cli-agent-conversations-resume-menu-label", "model_id": "claude-5-1-fable-high", "mechanical": {"tone_buzzword": 0, "tone_meta_opener": 0, "combined": 0}, "word_count": {"before_words": 1395, "after_words": 1324, "delta": -71, "delta_pct": -0.05089605734767025}, "judge": {"concision": 5, "avoids_over_explaining": 5, "technical_fidelity": 4}, "judge_composite": 4.666666666666667, "judge_model_id": "gemini-3.1-pro"} +{"fixture_id": "byollm-gemini-enterprise-google-cloud-setup", "model_id": "claude-5-1-fable-high", "mechanical": {"tone_buzzword": 0, "tone_meta_opener": 0, "combined": 0}, "word_count": {"before_words": 2630, "after_words": 2555, "delta": -75, "delta_pct": -0.028517110266159697}, "judge": {"concision": 4, "avoids_over_explaining": 4, "technical_fidelity": 2}, "judge_composite": 3.3333333333333335, "judge_model_id": "gemini-3.1-pro"} +{"fixture_id": "quickstart-synthetic-verbose-seed", "model_id": "claude-5-1-fable-high", "mechanical": {"tone_buzzword": 0, "tone_meta_opener": 0, "combined": 0}, "word_count": {"before_words": 235, "after_words": 107, "delta": -128, "delta_pct": -0.5446808510638298}, "judge": {"concision": 5, "avoids_over_explaining": 5, "technical_fidelity": 5}, "judge_composite": 5.0, "judge_model_id": "gemini-3.1-pro"} +{"fixture_id": "cli-agent-conversations-resume-menu-label", "model_id": "claude-4-5-sonnet", "mechanical": {"tone_buzzword": 0, "tone_meta_opener": 0, "combined": 0}, "word_count": {"before_words": 1395, "after_words": 1264, "delta": -131, "delta_pct": -0.0939068100358423}, "judge": {"concision": 4, "avoids_over_explaining": 4, "technical_fidelity": 2}, "judge_composite": 3.3333333333333335, "judge_model_id": "gemini-3.1-pro"} +{"fixture_id": "byollm-gemini-enterprise-google-cloud-setup", "model_id": "claude-4-5-sonnet", "mechanical": {"tone_buzzword": 0, "tone_meta_opener": 0, "combined": 0}, "word_count": {"before_words": 2630, "after_words": 2595, "delta": -35, "delta_pct": -0.013307984790874524}, "judge": {"concision": 4, "avoids_over_explaining": 4, "technical_fidelity": 4}, "judge_composite": 4.0, "judge_model_id": "gemini-3.1-pro"} +{"fixture_id": "quickstart-synthetic-verbose-seed", "model_id": "claude-4-5-sonnet", "mechanical": {"tone_buzzword": 0, "tone_meta_opener": 0, "combined": 0}, "word_count": {"before_words": 235, "after_words": 98, "delta": -137, "delta_pct": -0.5829787234042553}, "judge": {"concision": 5, "avoids_over_explaining": 5, "technical_fidelity": 5}, "judge_composite": 5.0, "judge_model_id": "gemini-3.1-pro"} +{"fixture_id": "cli-agent-conversations-resume-menu-label", "model_id": "claude-4-5-haiku", "mechanical": {"tone_buzzword": 0, "tone_meta_opener": 0, "combined": 0}, "word_count": {"before_words": 1395, "after_words": 1251, "delta": -144, "delta_pct": -0.1032258064516129}, "judge": {"concision": 4, "avoids_over_explaining": 4, "technical_fidelity": 3}, "judge_composite": 3.6666666666666665, "judge_model_id": "gemini-3.1-pro"} +{"fixture_id": "byollm-gemini-enterprise-google-cloud-setup", "model_id": "claude-4-5-haiku", "mechanical": {"tone_buzzword": 0, "tone_meta_opener": 0, "combined": 0}, "word_count": {"before_words": 2630, "after_words": 2579, "delta": -51, "delta_pct": -0.019391634980988594}, "judge": {"concision": 4, "avoids_over_explaining": 4, "technical_fidelity": 1}, "judge_composite": 3.0, "judge_model_id": "gemini-3.1-pro"} +{"fixture_id": "quickstart-synthetic-verbose-seed", "model_id": "claude-4-5-haiku", "mechanical": {"tone_buzzword": 0, "tone_meta_opener": 0, "combined": 0}, "word_count": {"before_words": 235, "after_words": 88, "delta": -147, "delta_pct": -0.625531914893617}, "judge": {"concision": 5, "avoids_over_explaining": 5, "technical_fidelity": 5}, "judge_composite": 5.0, "judge_model_id": "gemini-3.1-pro"} +{"fixture_id": "cli-agent-conversations-resume-menu-label", "model_id": "gpt-5-mini", "mechanical": {"tone_buzzword": 0, "tone_meta_opener": 0, "combined": 0}, "word_count": {"before_words": 1395, "after_words": 1188, "delta": -207, "delta_pct": -0.14838709677419354}, "judge": {"concision": 4, "avoids_over_explaining": 4, "technical_fidelity": 1}, "judge_composite": 3.0, "judge_model_id": "gemini-3.1-pro"} +{"fixture_id": "byollm-gemini-enterprise-google-cloud-setup", "model_id": "gpt-5-mini", "mechanical": {"tone_buzzword": 0, "tone_meta_opener": 0, "combined": 0}, "word_count": {"before_words": 2630, "after_words": 2048, "delta": -582, "delta_pct": -0.22129277566539923}, "judge": {"concision": 4, "avoids_over_explaining": 4, "technical_fidelity": 3}, "judge_composite": 3.6666666666666665, "judge_model_id": "gemini-3.1-pro"} +{"fixture_id": "quickstart-synthetic-verbose-seed", "model_id": "gpt-5-mini", "mechanical": {"tone_buzzword": 0, "tone_meta_opener": 0, "combined": 0}, "word_count": {"before_words": 235, "after_words": 117, "delta": -118, "delta_pct": -0.502127659574468}, "judge": {"concision": 5, "avoids_over_explaining": 5, "technical_fidelity": 1}, "judge_composite": 3.6666666666666665, "judge_model_id": "gemini-3.1-pro"} diff --git a/.agents/logs/tone_model_eval/outputs/byollm-gemini-enterprise-google-cloud-setup__claude-4-5-haiku.txt b/.agents/logs/tone_model_eval/outputs/byollm-gemini-enterprise-google-cloud-setup__claude-4-5-haiku.txt new file mode 100644 index 00000000..580216ab --- /dev/null +++ b/.agents/logs/tone_model_eval/outputs/byollm-gemini-enterprise-google-cloud-setup__claude-4-5-haiku.txt @@ -0,0 +1,270 @@ +--- +title: "BYOLLM: Gemini Enterprise (Vertex AI)" +sidebar: + label: "BYOLLM: Gemini Enterprise" +description: >- + Route Warp Agent inference through your own Google Cloud project with Gemini + Enterprise BYOLLM. Short-lived Workload Identity Federation credentials, + admin-controlled models, and inference billed to your GCP account. +--- + +Warp's **Gemini Enterprise** BYOLLM integration routes agent inference through your own Google Cloud project using **Vertex AI** (the Gemini Enterprise Agent Platform). Your team uses Warp's agents as usual, while eligible requests execute against models in your GCP project, billed to your Google Cloud account and governed by your IAM controls. + +Gemini Enterprise is one of the providers supported by [Bring Your Own LLM (BYOLLM)](/enterprise/enterprise-features/bring-your-own-llm/). For AWS-based routing, see the [AWS Bedrock BYOLLM setup](/enterprise/enterprise-features/byollm-aws-bedrock/). + +:::note +Gemini Enterprise BYOLLM is available on Warp's Enterprise plan only. [Contact sales](https://www.warp.dev/contact-sales) to learn more. +::: + +:::caution +Gemini Enterprise BYOLLM currently applies to **interactive agent requests** in the Warp app. [Cloud agent](/platform/) runs don't route through Gemini Enterprise yet; cloud agent support is available with [AWS Bedrock BYOLLM](/enterprise/enterprise-features/byollm-aws-bedrock/#enabling-byollm-for-cloud-agents). +::: + +## Key features + +* **Session-based federated authentication** - Warp uses the member's signed-in Warp session to issue a short-lived OIDC identity token, then exchanges it through Google Workload Identity Federation for temporary Google Cloud credentials. +* **Admin-controlled routing and models** - Admins configure the GCP project, Vertex location, and WIF provider once in the Admin Panel, then choose which models are enabled and which Vertex model references they resolve to. +* **Gemini and Claude models** - Route native Gemini models and Claude partner models available on Vertex AI through your project. +* **Consolidated billing and quota attribution** - Inference runs against your project's Vertex AI quota and is billed to your Google Cloud account. Requests carry your project for quota attribution (`X-Goog-User-Project`). +* **No long-lived credentials** - Warp never stores service account keys, refresh tokens, or credential files. The client keeps only a short-lived Google Cloud access token in memory and replaces it automatically as it approaches expiration. + +## How it works + +When Gemini Enterprise is enabled, Warp redirects eligible inference calls to Vertex AI in your Google Cloud project instead of using model providers' direct APIs. + +The high-level flow: + +1. **Admin configures routing** - Your team admin enables the Gemini Enterprise host on the **Models** page of the [Admin Panel](/enterprise/team-management/admin-panel/) and sets the GCP project, Vertex location, and WIF provider audience string. +2. **Members enable credentials** - Each signed-in member turns on **Use Gemini Enterprise credentials** in Warp's Settings (or the admin enforces it team-wide). +3. **Warp mints a short-lived token** - The Warp client exchanges a Warp-signed identity token for a short-lived Google Cloud access token via Google's Security Token Service (STS), optionally impersonating a service account you designate. +4. **Warp routes requests** - Eligible agent requests carry that access token, and Warp's backend calls Vertex AI in your project with the configured location and model reference. +5. **Inference executes in your cloud** - The model runs in your GCP project. Responses stream back to the Warp client. + +### Credential lifecycle + +Gemini Enterprise uses **federated, short-lived credentials** instead of API keys or local cloud CLI sessions: + +* **Rooted in the Warp session** - Each mint starts from the member's signed-in Warp session. Warp issues a signed OpenID Connect (OIDC) token, exchanges it at Google STS for a federated access token, and, if configured, impersonates your designated service account. +* **Automatic refresh** - Tokens are held in memory and refreshed automatically about five minutes before expiration. Members don't need to re-authenticate during normal use. +* **Strict binding** - A minted token is only attached to requests while the signed-in user and the admin's WIF configuration still match. Signing out, switching accounts, or configuration changes invalidate it. +* **No storage or logging** - The client never uploads refresh tokens, credential JSON, or service account keys, and Warp's servers never persist or log the access token. + +### Model availability + +Gemini Enterprise supports models available both in Warp and through Vertex AI in your project: + +* **Native Gemini models** - Current Gemini Flash, Flash Lite, and Pro families (for example, Gemini 3.1 Pro and Gemini 3.6 Flash). +* **Claude partner models on Vertex AI** - Current Claude Sonnet, Opus, Haiku, and Fable families offered as Vertex AI partner models. + +To see which models you can use, check [Model Choice](/agent-platform/inference/model-choice/) for Warp's supported models and the **Models** page of the Admin Panel for your team's Gemini Enterprise enablement. + +## Enabling Gemini Enterprise + +Setup has two halves: a one-time Google Cloud configuration (WIF trust and IAM), then routing configuration in Warp's Admin Panel. + +### Prerequisites + +* **Warp Enterprise plan with admin access** - You configure routing on the **Models** page of the [Admin Panel](/enterprise/team-management/admin-panel/). +* **A Google Cloud project with Vertex AI enabled** - Enable the Vertex AI API (`aiplatform.googleapis.com`) in the project that should own inference, quota, and billing. +* **Model access in Vertex AI** - Gemini models are available by default; Claude partner models must be enabled for your project in the Vertex AI Model Garden. +* **GCP IAM admin access** - You need permission to create a Workload Identity Pool, provider, and IAM bindings (and optionally a service account). + +### 1. Create a Workload Identity Federation pool and provider (cloud admin) + +Before Google Cloud can trust tokens issued by Warp, register Warp as an OIDC identity provider in a Workload Identity Pool. This is a one-time setup per GCP project. + +Warp's OIDC tokens use the issuer `https://app.warp.dev` and carry these claims: + +* `sub` - Shaped as `scoped_principal:/:`, where `` is `user` for interactive requests. +* `teams` - The list of Warp team UIDs the user belongs to. Mapping this claim to `google.groups` lets you grant access to your whole Warp team with one IAM binding. + +The `` is the Warp team UID for your team. You can find it in your team's [Admin Panel](/enterprise/team-management/admin-panel/) URL as the path segment after `/admin/`. For example, in `https://app.warp.dev/admin/HzjUdNkg8Uiq8gp6FMgfxe/models`, the team UID is `HzjUdNkg8Uiq8gp6FMgfxe`. + +**Example: create the pool and provider with the gcloud CLI** +```bashgcloud iam workload-identity-pools create warp-byollm \ + --project=PROJECT_ID \ + --location=global \ + --display-name="Warp BYOLLM" + +gcloud iam workload-identity-pools providers create-oidc warp \ + --project=PROJECT_ID \ + --location=global \ + --workload-identity-pool=warp-byollm \ + --issuer-uri="https://app.warp.dev" \ + --attribute-mapping="google.subject=assertion.sub,google.groups=assertion.teams" \ + --attribute-condition="'TEAM_UID' in assertion.teams" +``` +Replace `PROJECT_ID` with your GCP project ID and `TEAM_UID` with your Warp team UID. The attribute condition restricts the provider to tokens from your Warp team. + +After creating the provider, note its full resource name; you'll paste it into Warp as the **WIF audience** in Step 3: +```text//iam.googleapis.com/projects/PROJECT_NUMBER/locations/global/workloadIdentityPools/warp-byollm/providers/warp +``` +Replace `PROJECT_NUMBER` with your GCP project number. For more detail, see Google's [Workload Identity Federation documentation](https://cloud.google.com/iam/docs/workload-identity-federation). + +### 2. Grant Vertex AI access (cloud admin) + +Grant the federated identities permission to run Vertex AI inference in your project. Use least-privilege IAM bindings scoped to your Warp team's group principal. + +The two required roles are: + +* `roles/aiplatform.user` - Allows Vertex AI inference calls. +* `roles/serviceusage.serviceUsageConsumer` - Allows quota attribution to your project (Warp sends your project as the quota project on each request). + +**Option A: Direct federated access** + +Grant the roles directly to your Warp team's principal set. Leave the service account field empty in Warp. +```bashgcloud projects add-iam-policy-binding PROJECT_ID \ + --member="principalSet://iam.googleapis.com/projects/PROJECT_NUMBER/locations/global/workloadIdentityPools/warp-byollm/group/TEAM_UID" \ + --role="roles/aiplatform.user" + +gcloud projects add-iam-policy-binding PROJECT_ID \ + --member="principalSet://iam.googleapis.com/projects/PROJECT_NUMBER/locations/global/workloadIdentityPools/warp-byollm/group/TEAM_UID" \ + --role="roles/serviceusage.serviceUsageConsumer" +``` +**Option B: Service account impersonation** + +If your organization prefers auditing through a dedicated service account, create one, grant it the two roles above, and allow the federated principal set to impersonate it: +```bashgcloud iam service-accounts add-iam-policy-binding \ + warp-byollm@PROJECT_ID.iam.gserviceaccount.com \ + --member="principalSet://iam.googleapis.com/projects/PROJECT_NUMBER/locations/global/workloadIdentityPools/warp-byollm/group/TEAM_UID" \ + --role="roles/iam.workloadIdentityUser" +``` +With this option, also enable the IAM Service Account Credentials API (`iamcredentials.googleapis.com`) and enter the service account's email in Warp in Step 3. Warp never receives the service account's keys; the client requests short-lived impersonated tokens at mint time. + +### 3. Configure routing in the Admin Panel (Warp admin) + +Connect Warp to your GCP configuration so eligible requests route to your project: + +1. In the [Admin Panel](https://app.warp.dev/admin/), go to the **Models** page and find the **Gemini Enterprise** host configuration. +2. Enter the **GCP project ID**, choose a **location**, and paste the **WIF audience** (the full provider resource name from Step 1). Optionally enter the **service account email** from Step 2, Option B. +3. Toggle **Gemini Enterprise** on, then enable the models that should route through your project. You can override the Vertex model reference per model; clearing an override restores Warp's default. +4. Optionally, disable **Direct API** access for those models to enforce provider-only routing. + +The default location is `global`. Multi-region values (`global`, `us`, `eu`) and specific regions (for example, `us-central1`) are supported; the location applies host-wide to all Gemini Enterprise models. + +You can also choose how member credentials behave: + +* **Enforce** - Every signed-in member's client uses Gemini Enterprise credentials automatically, and the member toggle is managed by the organization. +* **Respect user setting** - Each member opts in with the **Use Gemini Enterprise credentials** toggle in their Settings. + +### 4. Validate + +Run a test prompt in Warp using a model enabled for Gemini Enterprise. Verify: + +* The model shows the Gemini Enterprise badge in the model picker. +* The request completes successfully. +* The request appears in your project's Vertex AI monitoring or Cloud Logging. + +## Using Gemini Enterprise as a team member + +Members need to be signed in to Warp; the credential flow is rooted in the Warp session, so Gemini Enterprise isn't available to logged-out users. + +1. In the Warp app, go to **Settings** > **Agents** > **Warp Agent** and scroll to the **Gemini Enterprise** section. The section appears once your admin has enabled the host. +2. Toggle **Use Gemini Enterprise credentials** on. If your admin enforces credentials team-wide, the toggle is already on and managed by your organization. +3. Check the credential status card. It shows the current state (for example, loaded with the next scheduled refresh, refreshing, setup incomplete, or a failure) and a **Refresh** button to force a new credential mint. +4. Pick an eligible model in the model picker. Models routed through your project show a Gemini Enterprise badge. + +If the status card reports setup is incomplete, your workspace's Gemini Enterprise host is enabled but not fully configured; contact your team admin. + +## Routing and fallback behavior + +### Host priority + +For each request, Warp tries enabled hosts in a fixed order: + +1. AWS Bedrock +2. Gemini Enterprise +3. Direct API + +A Gemini Enterprise route is only used when the request carries valid Gemini Enterprise credentials; otherwise Warp falls back to the next enabled host. + +### Failover behavior + +If a Gemini Enterprise request fails (for example, due to IAM misconfiguration or Vertex AI quota limits), Warp attempts to fall back to the next available host your admin has enabled. If a fallback uses a Direct API model, that request consumes Warp credits. If no fallback is available, Warp displays an error message. + +### Auto model selection + +Auto model selection is disabled if an admin disables **any** Direct API model, regardless of Gemini Enterprise configuration. When Direct API models remain enabled, Auto picks the best model for the task; if the selected model is enabled for Gemini Enterprise and your credentials are active, the request routes through your project. + +## Billing behavior + +When a request routes through Gemini Enterprise: + +* **Warp doesn't consume AI credits** for that request. Inference is billed by Google Cloud to your project. +* **Platform credits still apply** - On Business and Enterprise plans, local agent runs that use customer-supplied inference consume [platform credits](/support-and-community/plans-and-billing/platform-credits/) for Warp's platform infrastructure. +* **Fallbacks are billed normally** - A request that falls back to a Direct API model consumes Warp credits at the standard rate. + +See [The three credit buckets](/support-and-community/plans-and-billing/platform-credits/#the-three-credit-buckets) for more on credit types. + +## Security and data handling + +### Credential security + +* **No long-lived credentials** - The only long-lived credential involved is the member's Warp session. Access tokens are short-lived, held in memory, and never persisted. +* **Nothing sensitive leaves your boundary** - Warp stores only non-secret routing configuration (project ID, location, WIF audience, and optional service account email). Service account keys, refresh tokens, and credential files are never uploaded to Warp. +* **Per-user identity** - Every token is minted for the individual signed-in member, so access control and revocation stay in your identity stack: remove a member from your Warp team (or restrict the WIF provider's attribute condition) and their tokens stop minting. + +### Zero Data Retention (ZDR) + +Warp maintains **SOC 2 compliance** and has **Zero Data Retention (ZDR)** agreements with its contracted LLM providers. + +When using Gemini Enterprise: + +* **Your** Google Cloud project settings determine data retention policies. +* Warp cannot enforce ZDR for requests routed through your infrastructure. +* Review Vertex AI's data governance settings for your project to control retention. + +### Auditability + +* Warp keeps all conversations fully steerable and logged within Warp. +* Your GCP project retains provider-side logs (usage, latency, errors) in Vertex AI monitoring and Cloud Logging, attributed to your project. + +## Troubleshooting + +### Common errors + +* **Credentials expired or invalid** - The request reached Vertex AI but was rejected as unauthenticated. Click **Refresh credentials** in the inline error, or use the **Refresh** button in **Settings** > **Agents** > **Warp Agent**, then retry. +* **Setup incomplete** - The host is enabled but the WIF audience is missing or blank. A team admin needs to complete the **Models** page configuration. +* **Token exchange or impersonation failed** - Verify the WIF provider's issuer (`https://app.warp.dev`), attribute mapping, and attribute condition, and (for Option B) that the principal set holds `roles/iam.workloadIdentityUser` on the service account. +* **Permission denied from Vertex AI** - Confirm the federated principal (or service account) holds `roles/aiplatform.user` and `roles/serviceusage.serviceUsageConsumer` on the project. +* **Model not found** - Confirm the model is available in your configured location and, for Claude partner models, enabled in the Vertex AI Model Garden. Check any per-model reference overrides on the **Models** page. +* **Provider quota limits** - Check your project's Vertex AI quotas and request increases if needed. + +### Debugging steps + +1. Confirm the WIF audience in the Admin Panel exactly matches the provider's full resource name. +2. Check the credential status card in **Settings** > **Agents** > **Warp Agent** for the failing state and recovery action. +3. Verify IAM bindings for the principal set (or service account) in your GCP project. +4. Confirm the model reference and location match what's available in your project. +5. Inspect Cloud Logging in your project for request details and errors. + +## FAQ + +### How is this different from BYOK with a Google API key? + +**BYOK** routes requests to the Gemini Developer API using a personal API key stored on each member's device. **Gemini Enterprise BYOLLM** routes requests to Vertex AI in your organization's GCP project using short-lived federated credentials, configured centrally by an admin, with your project's IAM, quota, and billing. See [Bring Your Own API Key](/agent-platform/inference/bring-your-own-api-key/) for the self-serve option. + +### Do members need the gcloud CLI installed? + +No. Unlike [AWS Bedrock BYOLLM](/enterprise/enterprise-features/bring-your-own-llm/), which uses each member's local AWS CLI session, Gemini Enterprise mints credentials from the member's signed-in Warp session. No local Google tooling or configuration is required. + +### Does Gemini Enterprise work with cloud agents? + +Not yet. Gemini Enterprise currently routes interactive agent requests in the Warp app. Cloud agent BYOLLM is available through [AWS Bedrock](/enterprise/enterprise-features/byollm-aws-bedrock/#enabling-byollm-for-cloud-agents), and Gemini Enterprise support for cloud agents is planned. + +### Which Claude models route through my project? + +Claude models offered as Vertex AI partner models (current Sonnet, Opus, Haiku, and Fable families) can route through Gemini Enterprise when your admin enables them. Eligible models show the Gemini Enterprise badge in the model picker. + +### Can admins enforce provider-only routing? + +Yes. Admins can disable Direct API access for models on the **Models** page so eligible requests only route through your project. Note that disabling any Direct API model also disables Auto model selection. + +## Related resources + +* [Bring Your Own LLM](/enterprise/enterprise-features/bring-your-own-llm/) - BYOLLM overview and AWS Bedrock setup +* [Team-managed API keys and endpoints](/enterprise/enterprise-features/team-managed-keys-and-endpoints/) - Admin-configured shared provider keys and custom endpoints +* [Bring Your Own API Key](/agent-platform/inference/bring-your-own-api-key/) - Self-serve, user-level API keys +* [Model Choice](/agent-platform/inference/model-choice/) - Full list of supported models +* [Admin Panel](/enterprise/team-management/admin-panel/) - Configure team settings +* [Contact sales](https://www.warp.dev/contact-sales) - Get help with Enterprise setup diff --git a/.agents/logs/tone_model_eval/outputs/byollm-gemini-enterprise-google-cloud-setup__claude-4-5-sonnet.txt b/.agents/logs/tone_model_eval/outputs/byollm-gemini-enterprise-google-cloud-setup__claude-4-5-sonnet.txt new file mode 100644 index 00000000..bea891e4 --- /dev/null +++ b/.agents/logs/tone_model_eval/outputs/byollm-gemini-enterprise-google-cloud-setup__claude-4-5-sonnet.txt @@ -0,0 +1,270 @@ +--- +title: "BYOLLM: Gemini Enterprise (Vertex AI)" +sidebar: + label: "BYOLLM: Gemini Enterprise" +description: >- + Route Warp Agent inference through your own Google Cloud project with Gemini + Enterprise BYOLLM. Short-lived Workload Identity Federation credentials, + admin-controlled models, and inference billed to your GCP account. +--- + +Warp's **Gemini Enterprise** BYOLLM integration routes agent inference through your own Google Cloud project using **Vertex AI** (the Gemini Enterprise Agent Platform). Your team keeps using Warp's agents, while eligible requests execute against models hosted in your GCP project, billed to your Google Cloud account and governed by your IAM controls. + +Gemini Enterprise is one of the providers supported by [Bring Your Own LLM (BYOLLM)](/enterprise/enterprise-features/bring-your-own-llm/). For AWS-based routing, see the [AWS Bedrock BYOLLM setup](/enterprise/enterprise-features/byollm-aws-bedrock/). + +:::note +Gemini Enterprise BYOLLM is only available on Warp's Enterprise plan. [Contact sales](https://www.warp.dev/contact-sales) to learn more. +::: + +:::caution +Gemini Enterprise BYOLLM currently applies to **interactive agent requests** in the Warp app. [Cloud agent](/platform/) runs don't route through Gemini Enterprise yet; cloud agent support is available today with [AWS Bedrock BYOLLM](/enterprise/enterprise-features/byollm-aws-bedrock/#enabling-byollm-for-cloud-agents). +::: + +## Key features + +* **Session-based federated authentication** - Warp uses the member's signed-in Warp session to issue a short-lived OIDC identity token, then exchanges it through Google Workload Identity Federation for temporary Google Cloud credentials. +* **Admin-controlled routing and models** - Admins configure the GCP project, Vertex location, and WIF provider once in the Admin Panel, then choose which models are enabled and which Vertex model references they resolve to. +* **Gemini and Claude models** - Route native Gemini models and Claude partner models available on Vertex AI through your project. +* **Consolidated billing and quota attribution** - Inference runs against your project's Vertex AI quota and is billed to your Google Cloud account. Requests carry your project for quota attribution (`X-Goog-User-Project`). +* **No long-lived credentials** - Warp never stores service account keys, refresh tokens, or credential files. The client keeps only a short-lived Google Cloud access token in memory and automatically replaces it as it approaches expiration. + +## How it works + +When Gemini Enterprise is enabled, Warp redirects eligible inference calls to Vertex AI in your Google Cloud project instead of using model providers' direct APIs. + +Here's the flow: + +1. **Admin configures routing** - Your team admin enables the Gemini Enterprise host on the **Models** page of the [Admin Panel](/enterprise/team-management/admin-panel/) and sets the GCP project, Vertex location, and WIF provider audience string. +2. **Members enable credentials** - Each signed-in member turns on **Use Gemini Enterprise credentials** in Warp's Settings (or the admin enforces it team-wide). +3. **Warp mints a short-lived token** - The Warp client exchanges a Warp-signed identity token for a short-lived Google Cloud access token via Google's Security Token Service (STS), optionally impersonating a service account you designate. +4. **Warp routes requests** - Eligible agent requests carry that access token, and Warp's backend uses it to call Vertex AI in your project with the configured location and model reference. +5. **Inference executes in your cloud** - The model runs in your GCP project. Responses stream back to the Warp client. + +### Credential lifecycle + +Gemini Enterprise uses **federated, short-lived credentials** instead of API keys or local cloud CLI sessions: + +* **Rooted in the Warp session** - Each mint starts from the member's signed-in Warp session. Warp issues a signed OpenID Connect (OIDC) token, exchanges it at Google STS for a federated access token, and, if configured, impersonates your designated service account. +* **Automatic refresh** - Tokens are held in memory and refreshed automatically about five minutes before they expire. Members don't need to re-authenticate during normal use. +* **Strict binding** - A minted token is only attached to requests while the signed-in user and the admin's WIF configuration still match. Signing out, switching accounts, or admin configuration changes invalidate it. +* **No storage or logging** - The client never uploads refresh tokens, credential JSON, or service account keys, and Warp's servers never persist or log the access token. + +### Model availability + +Gemini Enterprise supports the intersection of models that Warp supports and models available through Vertex AI in your project: + +* **Native Gemini models** - Current Gemini Flash, Flash Lite, and Pro families (e.g., Gemini 3.1 Pro and Gemini 3.6 Flash). +* **Claude partner models on Vertex AI** - Current Claude Sonnet, Opus, Haiku, and Fable families offered as Vertex AI partner models. + +To determine which models you can use, see [Model Choice](/agent-platform/inference/model-choice/) for Warp's supported models and the **Models** page of the Admin Panel for your team's Gemini Enterprise enablement. + +## Enabling Gemini Enterprise + +Setup has two halves: a one-time Google Cloud configuration (WIF trust and IAM), then routing configuration in Warp's Admin Panel. + +### Prerequisites + +* **Warp Enterprise plan with admin access** - You configure routing on the **Models** page of the [Admin Panel](/enterprise/team-management/admin-panel/). +* **A Google Cloud project with Vertex AI enabled** - Enable the Vertex AI API (`aiplatform.googleapis.com`) in the project that should own inference, quota, and billing. +* **Model access in Vertex AI** - Gemini models are available by default; Claude partner models must be enabled for your project in the Vertex AI Model Garden. +* **GCP IAM admin access** - You need permission to create a Workload Identity Pool, provider, and IAM bindings (and optionally a service account). + +### 1. Create a Workload Identity Federation pool and provider (cloud admin) + +Before Google Cloud can trust tokens issued by Warp, register Warp as an OIDC identity provider in a Workload Identity Pool. + +Warp's OIDC tokens use the issuer `https://app.warp.dev` and carry these claims: + +* `sub` - Shaped as `scoped_principal:/:`, where `` is `user` for interactive requests. +* `teams` - The list of Warp team UIDs the user belongs to. Mapping this claim to `google.groups` lets you grant access to your whole Warp team with one IAM binding. + +The `` is the Warp team UID for your team. You can find it in your team's [Admin Panel](/enterprise/team-management/admin-panel/) URL as the path segment after `/admin/`. For example, in `https://app.warp.dev/admin/HzjUdNkg8Uiq8gp6FMgfxe/models`, the team UID is `HzjUdNkg8Uiq8gp6FMgfxe`. + +**Example: create the pool and provider with the gcloud CLI** +```bashgcloud iam workload-identity-pools create warp-byollm \ + --project=PROJECT_ID \ + --location=global \ + --display-name="Warp BYOLLM" + +gcloud iam workload-identity-pools providers create-oidc warp \ + --project=PROJECT_ID \ + --location=global \ + --workload-identity-pool=warp-byollm \ + --issuer-uri="https://app.warp.dev" \ + --attribute-mapping="google.subject=assertion.sub,google.groups=assertion.teams" \ + --attribute-condition="'TEAM_UID' in assertion.teams" +``` +Replace `PROJECT_ID` with your GCP project ID and `TEAM_UID` with your Warp team UID. The attribute condition restricts the provider to tokens from your Warp team. + +After creating the provider, note its full resource name; you'll paste it into Warp as the **WIF audience** in Step 3: +```text//iam.googleapis.com/projects/PROJECT_NUMBER/locations/global/workloadIdentityPools/warp-byollm/providers/warp +``` +Replace `PROJECT_NUMBER` with your GCP project number. For more detail, see Google's [Workload Identity Federation documentation](https://cloud.google.com/iam/docs/workload-identity-federation). + +### 2. Grant Vertex AI access (cloud admin) + +Grant the federated identities permission to run Vertex AI inference in your project. Use least-privilege IAM bindings scoped to your Warp team's group principal. + +The two required roles are: + +* `roles/aiplatform.user` - Run Vertex AI inference calls. +* `roles/serviceusage.serviceUsageConsumer` - Quota attribution to your project (Warp sends your project as the quota project on each request). + +**Option A: Direct federated access** + +Grant the roles directly to your Warp team's principal set. Leave the service account field empty in Warp. +```bashgcloud projects add-iam-policy-binding PROJECT_ID \ + --member="principalSet://iam.googleapis.com/projects/PROJECT_NUMBER/locations/global/workloadIdentityPools/warp-byollm/group/TEAM_UID" \ + --role="roles/aiplatform.user" + +gcloud projects add-iam-policy-binding PROJECT_ID \ + --member="principalSet://iam.googleapis.com/projects/PROJECT_NUMBER/locations/global/workloadIdentityPools/warp-byollm/group/TEAM_UID" \ + --role="roles/serviceusage.serviceUsageConsumer" +``` +**Option B: Service account impersonation** + +If your organization prefers auditing through a dedicated service account, create one, grant it the two roles above, and allow the federated principal set to impersonate it: +```bashgcloud iam service-accounts add-iam-policy-binding \ + warp-byollm@PROJECT_ID.iam.gserviceaccount.com \ + --member="principalSet://iam.googleapis.com/projects/PROJECT_NUMBER/locations/global/workloadIdentityPools/warp-byollm/group/TEAM_UID" \ + --role="roles/iam.workloadIdentityUser" +``` +With this option, also enable the IAM Service Account Credentials API (`iamcredentials.googleapis.com`) and enter the service account's email in Warp in Step 3. Warp never receives the service account's keys; the client requests short-lived impersonated tokens at mint time. + +### 3. Configure routing in the Admin Panel (Warp admin) + +Connect Warp to your GCP configuration so eligible requests route to your project: + +1. In the [Admin Panel](https://app.warp.dev/admin/), go to the **Models** page and find the **Gemini Enterprise** host configuration. +2. Enter the **GCP project ID**, choose a **location**, and paste the **WIF audience** (the full provider resource name from Step 1). Optionally enter the **service account email** from Step 2, Option B. +3. Toggle **Gemini Enterprise** on, then enable the models that should route through your project. You can override the Vertex model reference per model; clearing an override restores Warp's default. +4. Optionally, disable **Direct API** access for those models to enforce provider-only routing. + +The default location is `global`. Multi-region values (`global`, `us`, `eu`) and specific regions (e.g., `us-central1`) are supported; the location applies host-wide to all Gemini Enterprise models. + +You can also choose how member credentials behave: + +* **Enforce** - Every signed-in member's client uses Gemini Enterprise credentials automatically, and the member toggle is managed by the organization. +* **Respect user setting** - Each member opts in with the **Use Gemini Enterprise credentials** toggle in their Settings. + +### 4. Validate + +Run a test prompt in Warp using a model enabled for Gemini Enterprise. Verify: + +* The model shows the Gemini Enterprise badge in the model picker. +* The request completes successfully. +* The request appears in your project's Vertex AI monitoring or Cloud Logging. + +## Using Gemini Enterprise as a team member + +Members need to be signed in to Warp; the credential flow is rooted in the Warp session, so Gemini Enterprise isn't available to logged-out users. + +1. In the Warp app, go to **Settings** > **Agents** > **Warp Agent** and scroll to the **Gemini Enterprise** section. The section appears once your admin has enabled the host. +2. Toggle **Use Gemini Enterprise credentials** on. If your admin enforces credentials team-wide, the toggle is already on and managed by your organization. +3. Check the credential status card. It shows the current state (e.g., loaded with the next scheduled refresh, refreshing, setup incomplete, or a failure) and a **Refresh** button to force a new credential mint. +4. Pick an eligible model in the model picker. Models routed through your project show a Gemini Enterprise badge. + +If the status card reports that setup is incomplete, your workspace's Gemini Enterprise host is enabled but not fully configured; contact your team admin. + +## Routing and fallback behavior + +### Host priority + +For each request, Warp expands the selected model into the hosts your admin has enabled and tries them in this order: + +1. AWS Bedrock +2. Gemini Enterprise +3. Direct API + +The priority is fixed and not admin-configurable. A Gemini Enterprise route is only used when the request carries valid Gemini Enterprise credentials; otherwise Warp falls back to the next enabled host. + +### Failover behavior + +If a Gemini Enterprise request fails (e.g., due to IAM misconfiguration or Vertex AI quota limits), Warp attempts to fall back to the next available host your admin has enabled. If a fallback uses a Direct API model, that request consumes Warp credits. If no fallback is available, Warp displays an error message. + +### Auto model selection + +Auto model selection is disabled if an admin disables **any** Direct API model, regardless of Gemini Enterprise configuration. When Direct API models remain enabled, Auto picks the best model for the task; if the selected model is enabled for Gemini Enterprise and your credentials are active, the request routes through your project. + +## Billing behavior + +When a request routes through Gemini Enterprise: + +* **Warp doesn't consume AI credits** for that request. Inference is billed by Google Cloud to your project. +* **Platform credits still apply** - On Business and Enterprise plans, local agent runs that use customer-supplied inference consume [platform credits](/support-and-community/plans-and-billing/platform-credits/) for Warp's platform infrastructure. +* **Fallbacks are billed normally** - A request that falls back to a Direct API model consumes Warp credits at the standard rate. + +See [The three credit buckets](/support-and-community/plans-and-billing/platform-credits/#the-three-credit-buckets) for more on credit types. + +## Security and data handling + +### Credential security + +* **No long-lived credentials** - The only long-lived credential involved is the member's Warp session. Access tokens are short-lived, held in memory, and never persisted. +* **Nothing sensitive leaves your boundary** - Warp stores only non-secret routing configuration (project ID, location, WIF audience, and optional service account email). Service account keys, refresh tokens, and credential files are never uploaded to Warp. +* **Per-user identity** - Every token is minted for the individual signed-in member, so access control and revocation stay in your identity stack: remove a member from your Warp team (or restrict the WIF provider's attribute condition) and their tokens stop minting. + +### Zero Data Retention (ZDR) + +Warp maintains **SOC 2 compliance** and has **Zero Data Retention (ZDR)** agreements with its contracted LLM providers. + +When using Gemini Enterprise: + +* **Your** Google Cloud project settings determine data retention policies. +* Warp cannot enforce ZDR for requests routed through your infrastructure. +* Review Vertex AI's data governance settings for your project to control retention. + +### Auditability + +* Warp keeps all conversations fully steerable and logged within Warp. +* Your GCP project retains provider-side logs (usage, latency, errors) in Vertex AI monitoring and Cloud Logging, attributed to your project. + +## Troubleshooting + +### Common errors + +* **Credentials expired or invalid** - The request reached Vertex AI but was rejected as unauthenticated. Click **Refresh credentials** in the inline error, or use the **Refresh** button in **Settings** > **Agents** > **Warp Agent**, then retry. +* **Setup incomplete** - The host is enabled but the WIF audience is missing or blank. A team admin needs to complete the **Models** page configuration. +* **Token exchange or impersonation failed** - Verify the WIF provider's issuer (`https://app.warp.dev`), attribute mapping, and attribute condition, and (for Option B) that the principal set holds `roles/iam.workloadIdentityUser` on the service account. +* **Permission denied from Vertex AI** - Confirm the federated principal (or service account) holds `roles/aiplatform.user` and `roles/serviceusage.serviceUsageConsumer` on the project. +* **Model not found** - Confirm the model is available in your configured location and, for Claude partner models, enabled in the Vertex AI Model Garden. Check any per-model reference overrides on the **Models** page. +* **Provider quota limits** - Check your project's Vertex AI quotas and request increases if needed. + +### Debugging steps + +1. Confirm the WIF audience in the Admin Panel exactly matches the provider's full resource name. +2. Check the credential status card in **Settings** > **Agents** > **Warp Agent** for the failing state and recovery action. +3. Verify IAM bindings for the principal set (or service account) in your GCP project. +4. Confirm the model reference and location match what's available in your project. +5. Inspect Cloud Logging in your project for request details and errors. + +## FAQ + +### How is this different from BYOK with a Google API key? + +**BYOK** routes requests to the Gemini Developer API using a personal API key stored on each member's device. **Gemini Enterprise BYOLLM** routes requests to Vertex AI in your organization's GCP project using short-lived federated credentials, configured centrally by an admin, with your project's IAM, quota, and billing. See [Bring Your Own API Key](/agent-platform/inference/bring-your-own-api-key/) for the self-serve option. + +### Do members need the gcloud CLI installed? + +No. Unlike [AWS Bedrock BYOLLM](/enterprise/enterprise-features/bring-your-own-llm/), which uses each member's local AWS CLI session, Gemini Enterprise mints credentials from the member's signed-in Warp session. No local Google tooling or configuration is required. + +### Does Gemini Enterprise work with cloud agents? + +Not yet. Gemini Enterprise currently routes interactive agent requests in the Warp app. Cloud agent BYOLLM is available today through [AWS Bedrock](/enterprise/enterprise-features/byollm-aws-bedrock/#enabling-byollm-for-cloud-agents), and Gemini Enterprise support for cloud agents is planned. + +### Which Claude models route through my project? + +Claude models offered as Vertex AI partner models (current Sonnet, Opus, Haiku, and Fable families) can route through Gemini Enterprise when your admin enables them. Eligible models show the Gemini Enterprise badge in the model picker. + +### Can admins enforce provider-only routing? + +Yes. Admins can disable Direct API access for models on the **Models** page so eligible requests only route through your project. Note that disabling any Direct API model also disables Auto model selection. + +## Related resources + +* [Bring Your Own LLM](/enterprise/enterprise-features/bring-your-own-llm/) - BYOLLM overview and AWS Bedrock setup +* [Team-managed API keys and endpoints](/enterprise/enterprise-features/team-managed-keys-and-endpoints/) - Admin-configured shared provider keys and custom endpoints +* [Bring Your Own API Key](/agent-platform/inference/bring-your-own-api-key/) - Self-serve, user-level API keys +* [Model Choice](/agent-platform/inference/model-choice/) - Full list of supported models +* [Admin Panel](/enterprise/team-management/admin-panel/) - Configure team settings +* [Contact sales](https://www.warp.dev/contact-sales) - Get help with Enterprise setup diff --git a/.agents/logs/tone_model_eval/outputs/byollm-gemini-enterprise-google-cloud-setup__claude-5-1-fable-high.txt b/.agents/logs/tone_model_eval/outputs/byollm-gemini-enterprise-google-cloud-setup__claude-5-1-fable-high.txt new file mode 100644 index 00000000..97475daa --- /dev/null +++ b/.agents/logs/tone_model_eval/outputs/byollm-gemini-enterprise-google-cloud-setup__claude-5-1-fable-high.txt @@ -0,0 +1,264 @@ +--- +title: "BYOLLM: Gemini Enterprise (Vertex AI)" +sidebar: + label: "BYOLLM: Gemini Enterprise" +description: >- + Route Warp Agent inference through your own Google Cloud project with Gemini + Enterprise BYOLLM. Short-lived Workload Identity Federation credentials, + admin-controlled models, and inference billed to your GCP account. +--- + +The **Gemini Enterprise** BYOLLM integration routes agent inference through your own Google Cloud project using **Vertex AI** (the Gemini Enterprise Agent Platform). Your team keeps using agents in Warp as usual. Eligible requests run against models hosted in your GCP project, billed to your Google Cloud account and governed by your IAM controls. + +Gemini Enterprise is one of the providers supported by [Bring Your Own LLM (BYOLLM)](/enterprise/enterprise-features/bring-your-own-llm/). For AWS-based routing, see the [AWS Bedrock BYOLLM setup](/enterprise/enterprise-features/byollm-aws-bedrock/). + +:::note +Gemini Enterprise BYOLLM is only available on the Enterprise plan. [Contact sales](https://www.warp.dev/contact-sales) to learn more. +::: + +:::caution +Gemini Enterprise BYOLLM applies to **interactive agent requests** in the Warp app. [Cloud agent](/platform/) runs don't route through Gemini Enterprise yet. Cloud agent support is available today with [AWS Bedrock BYOLLM](/enterprise/enterprise-features/byollm-aws-bedrock/#enabling-byollm-for-cloud-agents). +::: + +## Key features + +* **Session-based federated authentication** - Warp uses the member's signed-in Warp session to issue a short-lived OIDC identity token, then exchanges it through Google Workload Identity Federation for temporary Google Cloud credentials. +* **Admin-controlled routing and models** - Admins configure the GCP project, Vertex location, and WIF provider once in the Admin Panel, then choose which models are enabled and which Vertex model references they resolve to. +* **Gemini and Claude models** - Route native Gemini models and Claude partner models available on Vertex AI through your project. +* **Consolidated billing and quota attribution** - Inference runs against your project's Vertex AI quota and is billed to your Google Cloud account. Requests carry your project for quota attribution (`X-Goog-User-Project`). +* **No long-lived credentials** - Warp never stores service account keys, refresh tokens, or credential files. The client keeps only a short-lived Google Cloud access token in memory and replaces it as it approaches expiration. + +## How it works + +When Gemini Enterprise is enabled, Warp sends eligible inference calls to Vertex AI in your Google Cloud project instead of the model providers' direct APIs. + +1. **Admin configures routing** - Your team admin enables the Gemini Enterprise host on the **Models** page of the [Admin Panel](/enterprise/team-management/admin-panel/) and sets the GCP project, Vertex location, and WIF provider audience string. +2. **Members enable credentials** - Each signed-in member turns on **Use Gemini Enterprise credentials** in Settings, or the admin enforces it team-wide. +3. **Warp mints a short-lived token** - The Warp client exchanges a Warp-signed identity token for a short-lived Google Cloud access token through Google's Security Token Service (STS), optionally impersonating a service account you designate. +4. **Warp routes requests** - Eligible agent requests carry that access token, and Warp's backend uses it to call Vertex AI in your project with the configured location and model reference. +5. **Inference runs in your cloud** - The model runs in your GCP project. Responses stream back to the Warp client. + +### Credential lifecycle + +Gemini Enterprise uses federated, short-lived credentials instead of API keys or local cloud CLI sessions: + +* **Rooted in the Warp session** - Each mint starts from the member's signed-in Warp session. Warp issues a signed OpenID Connect (OIDC) token, exchanges it at Google STS for a federated access token, and, if configured, impersonates your designated service account. +* **Automatic refresh** - Tokens are held in memory and refreshed about five minutes before they expire. Members don't re-authenticate during normal use. +* **Strict binding** - A minted token is attached to requests only while the signed-in user and the admin's WIF configuration still match. Signing out, switching accounts, or admin configuration changes invalidate it. +* **No storage or logging** - The client never uploads refresh tokens, credential JSON, or service account keys, and Warp's servers never persist or log the access token. + +### Model availability + +Gemini Enterprise supports the intersection of models Warp supports and models available through Vertex AI in your project: + +* **Native Gemini models** - Current Gemini Flash, Flash Lite, and Pro families (for example, Gemini 3.1 Pro and Gemini 3.6 Flash). +* **Claude partner models on Vertex AI** - Current Claude Sonnet, Opus, Haiku, and Fable families offered as Vertex AI partner models. + +See [Model Choice](/agent-platform/inference/model-choice/) for Warp's supported models and the **Models** page of the Admin Panel for your team's Gemini Enterprise enablement. + +## Enabling Gemini Enterprise + +Setup has two halves: a one-time Google Cloud configuration (WIF trust and IAM), then routing configuration in the Admin Panel. + +### Prerequisites + +* **Warp Enterprise plan with admin access** - You configure routing on the **Models** page of the [Admin Panel](/enterprise/team-management/admin-panel/). +* **A Google Cloud project with Vertex AI enabled** - Enable the Vertex AI API (`aiplatform.googleapis.com`) in the project that should own inference, quota, and billing. +* **Model access in Vertex AI** - Gemini models are available by default. Claude partner models must be enabled for your project in the Vertex AI Model Garden. +* **GCP IAM admin access** - You need permission to create a Workload Identity Pool, provider, and IAM bindings (and optionally a service account). + +### 1. Create a Workload Identity Federation pool and provider (cloud admin) + +Register Warp as an OIDC identity provider in a Workload Identity Pool so Google Cloud trusts tokens issued by Warp. This is a one-time setup per GCP project. + +Warp's OIDC tokens use the issuer `https://app.warp.dev` and carry these claims: + +* `sub` - Shaped as `scoped_principal:/:`, where `` is `user` for interactive requests. +* `teams` - The list of Warp team UIDs the user belongs to. Mapping this claim to `google.groups` lets you grant access to your whole Warp team with one IAM binding. + +The `` is your Warp team UID. Find it in your team's [Admin Panel](/enterprise/team-management/admin-panel/) URL as the path segment after `/admin/`. For example, in `https://app.warp.dev/admin/HzjUdNkg8Uiq8gp6FMgfxe/models`, the team UID is `HzjUdNkg8Uiq8gp6FMgfxe`. + +**Example: create the pool and provider with the gcloud CLI** +```bashgcloud iam workload-identity-pools create warp-byollm \ + --project=PROJECT_ID \ + --location=global \ + --display-name="Warp BYOLLM" + +gcloud iam workload-identity-pools providers create-oidc warp \ + --project=PROJECT_ID \ + --location=global \ + --workload-identity-pool=warp-byollm \ + --issuer-uri="https://app.warp.dev" \ + --attribute-mapping="google.subject=assertion.sub,google.groups=assertion.teams" \ + --attribute-condition="'TEAM_UID' in assertion.teams" +``` +Replace `PROJECT_ID` with your GCP project ID and `TEAM_UID` with your Warp team UID. The attribute condition restricts the provider to tokens from your Warp team. + +Note the provider's full resource name. You'll paste it into Warp as the **WIF audience** in Step 3: +```text//iam.googleapis.com/projects/PROJECT_NUMBER/locations/global/workloadIdentityPools/warp-byollm/providers/warp +``` +Replace `PROJECT_NUMBER` with your GCP project number. For more detail, see Google's [Workload Identity Federation documentation](https://cloud.google.com/iam/docs/workload-identity-federation). + +### 2. Grant Vertex AI access (cloud admin) + +Grant the federated identities permission to run Vertex AI inference in your project, using least-privilege IAM bindings scoped to your Warp team's group principal. + +Two roles are required: + +* `roles/aiplatform.user` - Allows Vertex AI inference calls. +* `roles/serviceusage.serviceUsageConsumer` - Allows quota attribution to your project (Warp sends your project as the quota project on each request). + +**Option A: Direct federated access** + +Grant the roles directly to your Warp team's principal set. Leave the service account field empty in Warp. +```bashgcloud projects add-iam-policy-binding PROJECT_ID \ + --member="principalSet://iam.googleapis.com/projects/PROJECT_NUMBER/locations/global/workloadIdentityPools/warp-byollm/group/TEAM_UID" \ + --role="roles/aiplatform.user" + +gcloud projects add-iam-policy-binding PROJECT_ID \ + --member="principalSet://iam.googleapis.com/projects/PROJECT_NUMBER/locations/global/workloadIdentityPools/warp-byollm/group/TEAM_UID" \ + --role="roles/serviceusage.serviceUsageConsumer" +``` +**Option B: Service account impersonation** + +If your organization audits through a dedicated service account, create one, grant it the two roles above, and allow the federated principal set to impersonate it: +```bashgcloud iam service-accounts add-iam-policy-binding \ + warp-byollm@PROJECT_ID.iam.gserviceaccount.com \ + --member="principalSet://iam.googleapis.com/projects/PROJECT_NUMBER/locations/global/workloadIdentityPools/warp-byollm/group/TEAM_UID" \ + --role="roles/iam.workloadIdentityUser" +``` +With this option, also enable the IAM Service Account Credentials API (`iamcredentials.googleapis.com`) and enter the service account's email in Warp in Step 3. Warp never receives the service account's keys; the client requests short-lived impersonated tokens at mint time. + +### 3. Configure routing in the Admin Panel (Warp admin) + +1. In the [Admin Panel](https://app.warp.dev/admin/), go to the **Models** page and find the **Gemini Enterprise** host configuration. +2. Enter the **GCP project ID**, choose a **location**, and paste the **WIF audience** (the full provider resource name from Step 1). Optionally enter the **service account email** from Step 2, Option B. +3. Toggle **Gemini Enterprise** on, then enable the models that should route through your project. You can override the Vertex model reference per model; clearing an override restores Warp's default. +4. Optionally, disable **Direct API** access for those models to enforce provider-only routing. + +The default location is `global`. Multi-region values (`global`, `us`, `eu`) and specific regions (for example, `us-central1`) are supported. The location applies host-wide to all Gemini Enterprise models. + +You can also choose how member credentials behave: + +* **Enforce** - Every signed-in member's client uses Gemini Enterprise credentials, and the member toggle is managed by the organization. +* **Respect user setting** - Each member opts in with the **Use Gemini Enterprise credentials** toggle in their Settings. + +### 4. Validate + +Run a test prompt in Warp using a model enabled for Gemini Enterprise. Verify: + +* The model shows the Gemini Enterprise badge in the model picker. +* The request completes. +* The request appears in your project's Vertex AI monitoring or Cloud Logging. + +## Using Gemini Enterprise as a team member + +Members must be signed in to Warp. The credential flow is rooted in the Warp session, so Gemini Enterprise isn't available to logged-out users. + +1. In the Warp app, go to **Settings** > **Agents** > **Warp Agent** and scroll to the **Gemini Enterprise** section. The section appears once your admin has enabled the host. +2. Toggle **Use Gemini Enterprise credentials** on. If your admin enforces credentials team-wide, the toggle is already on and managed by your organization. +3. Check the credential status card. It shows the current state (for example, loaded with the next scheduled refresh, refreshing, setup incomplete, or a failure) and a **Refresh** button to force a new credential mint. +4. Pick an eligible model in the model picker. Models routed through your project show a Gemini Enterprise badge. + +If the status card reports that setup is incomplete, your workspace's Gemini Enterprise host is enabled but not fully configured. Contact your team admin. + +## Routing and fallback behavior + +### Host priority + +For each request, Warp expands the selected model into the hosts your admin has enabled and tries them in a fixed order: + +1. AWS Bedrock +2. Gemini Enterprise +3. Direct API + +The priority isn't admin-configurable. A Gemini Enterprise route is used only when the request carries valid Gemini Enterprise credentials; otherwise Warp falls back to the next enabled host. + +### Failover behavior + +If a Gemini Enterprise request fails (for example, from IAM misconfiguration or Vertex AI quota limits), Warp falls back to the next available host your admin has enabled. If the fallback uses a Direct API model, that request consumes Warp credits. If no fallback is available, Warp shows an error. + +### Auto model selection + +Auto model selection is disabled if an admin disables **any** Direct API model, regardless of Gemini Enterprise configuration. When Direct API models remain enabled, Auto picks the best model for the task. If the selected model is enabled for Gemini Enterprise and your credentials are active, the request routes through your project. + +## Billing behavior + +When a request routes through Gemini Enterprise: + +* **Warp doesn't consume AI credits** for that request. Google Cloud bills inference to your project. +* **Platform credits still apply** - On Business and Enterprise plans, local agent runs that use customer-supplied inference consume [platform credits](/support-and-community/plans-and-billing/platform-credits/) for Warp's platform infrastructure. +* **Fallbacks are billed normally** - A request that falls back to a Direct API model consumes Warp credits at the standard rate. + +See [The three credit buckets](/support-and-community/plans-and-billing/platform-credits/#the-three-credit-buckets) for more on credit types. + +## Security and data handling + +### Credential security + +* **No long-lived credentials** - The only long-lived credential involved is the member's Warp session. Access tokens are short-lived, held in memory, and never persisted. +* **Nothing sensitive leaves your boundary** - Warp stores only non-secret routing configuration (project ID, location, WIF audience, and optional service account email). Service account keys, refresh tokens, and credential files are never uploaded to Warp. +* **Per-user identity** - Every token is minted for the individual signed-in member, so access control and revocation stay in your identity stack. Remove a member from your Warp team (or restrict the WIF provider's attribute condition) and their tokens stop minting. + +### Zero Data Retention (ZDR) + +Warp maintains SOC 2 compliance and has Zero Data Retention (ZDR) agreements with its contracted LLM providers. When you use Gemini Enterprise: + +* Your Google Cloud project settings determine data retention policies. +* Warp can't enforce ZDR for requests routed through your infrastructure. +* Review Vertex AI's data governance settings for your project to control retention. + +### Auditability + +* Conversations stay steerable and logged within Warp. +* Your GCP project retains provider-side logs (usage, latency, errors) in Vertex AI monitoring and Cloud Logging, attributed to your project. + +## Troubleshooting + +### Common errors + +* **Credentials expired or invalid** - The request reached Vertex AI but was rejected as unauthenticated. Click **Refresh credentials** in the inline error, or use the **Refresh** button in **Settings** > **Agents** > **Warp Agent**, then retry. +* **Setup incomplete** - The host is enabled but the WIF audience is missing or blank. A team admin needs to complete the **Models** page configuration. +* **Token exchange or impersonation failed** - Verify the WIF provider's issuer (`https://app.warp.dev`), attribute mapping, and attribute condition, and (for Option B) that the principal set holds `roles/iam.workloadIdentityUser` on the service account. +* **Permission denied from Vertex AI** - Confirm the federated principal (or service account) holds `roles/aiplatform.user` and `roles/serviceusage.serviceUsageConsumer` on the project. +* **Model not found** - Confirm the model is available in your configured location and, for Claude partner models, enabled in the Vertex AI Model Garden. Check any per-model reference overrides on the **Models** page. +* **Provider quota limits** - Check your project's Vertex AI quotas and request increases if needed. + +### Debugging steps + +1. Confirm the WIF audience in the Admin Panel exactly matches the provider's full resource name. +2. Check the credential status card in **Settings** > **Agents** > **Warp Agent** for the failing state and recovery action. +3. Verify IAM bindings for the principal set (or service account) in your GCP project. +4. Confirm the model reference and location match what's available in your project. +5. Inspect Cloud Logging in your project for request details and errors. + +## FAQ + +### How is this different from BYOK with a Google API key? + +BYOK routes requests to the Gemini Developer API using a personal API key stored on each member's device. Gemini Enterprise BYOLLM routes requests to Vertex AI in your organization's GCP project using short-lived federated credentials, configured centrally by an admin, with your project's IAM, quota, and billing. See [Bring Your Own API Key](/agent-platform/inference/bring-your-own-api-key/) for the self-serve option. + +### Do members need the gcloud CLI installed? + +No. [AWS Bedrock BYOLLM](/enterprise/enterprise-features/bring-your-own-llm/) uses each member's local AWS CLI session; Gemini Enterprise mints credentials from the member's signed-in Warp session. No local Google tooling or configuration is required. + +### Does Gemini Enterprise work with cloud agents? + +Not yet. Cloud agent BYOLLM is available today through [AWS Bedrock](/enterprise/enterprise-features/byollm-aws-bedrock/#enabling-byollm-for-cloud-agents), and Gemini Enterprise support for cloud agents is planned. + +### Which Claude models route through my project? + +Claude models offered as Vertex AI partner models (current Sonnet, Opus, Haiku, and Fable families) route through Gemini Enterprise when your admin enables them. Eligible models show the Gemini Enterprise badge in the model picker. This is unrelated to any routing Warp does through its own infrastructure for Warp-billed requests. + +### Can admins enforce provider-only routing? + +Yes. Admins can disable Direct API access for models on the **Models** page so eligible requests route only through your project. Disabling any Direct API model also disables Auto model selection. + +## Related resources + +* [Bring Your Own LLM](/enterprise/enterprise-features/bring-your-own-llm/) - BYOLLM overview and AWS Bedrock setup +* [Team-managed API keys and endpoints](/enterprise/enterprise-features/team-managed-keys-and-endpoints/) - Admin-configured shared provider keys and custom endpoints +* [Bring Your Own API Key](/agent-platform/inference/bring-your-own-api-key/) - Self-serve, user-level API keys +* [Model Choice](/agent-platform/inference/model-choice/) - Full list of supported models +* [Admin Panel](/enterprise/team-management/admin-panel/) - Configure team settings +* [Contact sales](https://www.warp.dev/contact-sales) - Get help with Enterprise setup diff --git a/.agents/logs/tone_model_eval/outputs/byollm-gemini-enterprise-google-cloud-setup__gpt-5-mini.txt b/.agents/logs/tone_model_eval/outputs/byollm-gemini-enterprise-google-cloud-setup__gpt-5-mini.txt new file mode 100644 index 00000000..69bcc1e1 --- /dev/null +++ b/.agents/logs/tone_model_eval/outputs/byollm-gemini-enterprise-google-cloud-setup__gpt-5-mini.txt @@ -0,0 +1,244 @@ +title: "BYOLLM: Gemini Enterprise (Vertex AI)" +sidebar: + label: "BYOLLM: Gemini Enterprise" +description: >- + Route Warp Agent inference through your own Google Cloud project with Gemini + Enterprise BYOLLM. Short-lived Workload Identity Federation credentials, + admin-controlled models, and inference billed to your GCP account. + +Warp's Gemini Enterprise BYOLLM integration routes agent inference through your Google Cloud project using Vertex AI. Eligible requests execute against models hosted in your GCP project, billed to your Google Cloud account and governed by your IAM controls. Admins configure routing and model mappings in the Admin Panel; members mint short-lived federated credentials from their Warp session. + +Gemini Enterprise is one provider option for Bring Your Own LLM (BYOLLM). For AWS routing, see AWS Bedrock BYOLLM. + +:::note +Gemini Enterprise BYOLLM is available on Warp's Enterprise plan. Contact sales for access. +::: + +:::caution +Gemini Enterprise applies to interactive agent requests in the Warp app. Cloud agent runs do not route through Gemini Enterprise yet; use AWS Bedrock BYOLLM for cloud agent BYOLLM today. +::: + +## Key features + +* Session-based federated authentication — Warp issues a Warp-signed OIDC token and exchanges it via Google Workload Identity Federation for short-lived Google Cloud credentials. +* Admin-controlled routing and models — Admins set the GCP project, Vertex location, WIF provider, and which models map to Vertex references. +* Gemini and Claude models — Route native Gemini models and Claude partner models offered on Vertex AI. +* Consolidated billing and quota attribution — Inference runs against your project's Vertex AI quota and is billed to your Google Cloud account. Requests carry your project for quota attribution (X-Goog-User-Project). +* No long-lived credentials — Warp does not store service account keys, refresh tokens, or credential files. The client holds short-lived access tokens in memory and refreshes them automatically. + +## How it works + +When enabled, Warp redirects eligible inference calls to Vertex AI in your project instead of using direct provider APIs. High-level flow: + +1. Admin configures routing in the Admin Panel (Models page) with project, location, and WIF provider. +2. Members enable "Use Gemini Enterprise credentials" in Warp Settings or the admin enforces it. +3. The Warp client exchanges a Warp-signed identity token at Google STS for a short-lived access token, optionally impersonating a designated service account. +4. Warp routes eligible agent requests with that access token to Vertex AI in your project. +5. Inference executes in your cloud and responses stream back to the Warp client. + +### Credential lifecycle + +Gemini Enterprise uses federated, short-lived credentials: + +* Rooted in the Warp session — Warp issues an OIDC token, exchanges it at Google STS for a federated access token, and may impersonate a service account. +* Automatic refresh — Tokens are kept in memory and refreshed about five minutes before expiry. +* Strict binding — Tokens stop minting if the signed-in user changes or the admin's WIF configuration changes. +* No storage or logging — Warp does not persist or log the access token. + +### Model availability + +Gemini Enterprise supports models that Warp supports and that are available through Vertex AI in your project: + +* Native Gemini families (for example, Gemini 3.1 Pro and Gemini 3.6 Flash). +* Claude partner families offered as Vertex AI partner models (Sonnet, Opus, Haiku, Fable). + +Check Warp's Model Choice for Warp-supported models and the Admin Panel's Models page for your team's enabled Vertex models. + +## Enabling Gemini Enterprise + +Setup has two parts: a Google Cloud configuration (WIF trust and IAM) and routing configuration in Warp's Admin Panel. + +### Prerequisites + +* Warp Enterprise plan with admin access. +* A Google Cloud project with the Vertex AI API (aiplatform.googleapis.com) enabled. +* Model access in Vertex AI (Gemini available by default; Claude partner models require enablement). +* GCP IAM admin access to create a Workload Identity Pool, provider, and IAM bindings (or to set up a service account). + +### 1. Create a Workload Identity Federation pool and provider (cloud admin) + +Register Warp as an OIDC provider in a Workload Identity Pool so Google Cloud can trust Warp's tokens. + +Warp's OIDC tokens use issuer https://app.warp.dev and include these claims: + +* sub — scoped_principal:/: (actor-type is user for interactive requests). +* teams — list of Warp team UIDs the user belongs to. Map this claim to google.groups to grant access to your whole Warp team with one IAM binding. + +Find your team UID in the Admin Panel URL path after /admin/. + +Example gcloud commands: +```bashgcloud iam workload-identity-pools create warp-byollm \ + --project=PROJECT_ID \ + --location=global \ + --display-name="Warp BYOLLM" + +gcloud iam workload-identity-pools providers create-oidc warp \ + --project=PROJECT_ID \ + --location=global \ + --workload-identity-pool=warp-byollm \ + --issuer-uri="https://app.warp.dev" \ + --attribute-mapping="google.subject=assertion.sub,google.groups=assertion.teams" \ + --attribute-condition="'TEAM_UID' in assertion.teams" +``` +After creating the provider, note its full resource name; paste it as the WIF audience in the Admin Panel. + +### 2. Grant Vertex AI access (cloud admin) + +Grant federated identities permission to run Vertex AI inference. Use least-privilege IAM bindings. + +Required roles: + +* roles/aiplatform.user — Allows Vertex AI inference calls. +* roles/serviceusage.serviceUsageConsumer — Allows quota attribution to your project. + +Option A: Direct federated access — bind the roles directly to your Warp team's principal set. Leave the service account field empty in Warp. + +Option B: Service account impersonation — create a service account, grant it the roles, and allow the federated principal set to impersonate it by granting roles/iam.workloadIdentityUser on the service account. Enable the IAM Service Account Credentials API and enter the service account email in Warp. Warp never receives service account keys. + +### 3. Configure routing in the Admin Panel (Warp admin) + +Connect Warp to your GCP configuration: + +1. In the Admin Panel Models page, find Gemini Enterprise host configuration. +2. Enter the GCP project ID, choose a location, and paste the WIF audience (provider resource name). Optionally enter the service account email if using impersonation. +3. Toggle Gemini Enterprise on and enable the models that should route through your project. You can override the Vertex model reference per model; clearing an override restores Warp's default. +4. Optionally disable Direct API access for those models to enforce provider-only routing. + +Default location is global. Multi-region values (global, us, eu) and specific regions (for example, us-central1) are supported. + +Choose credential behavior: + +* Enforce — Admin forces all members' clients to use Gemini Enterprise credentials. +* Respect user setting — Members opt in with Use Gemini Enterprise credentials in Settings. + +### 4. Validate + +Run a test prompt in Warp with an enabled model and verify: + +* The model shows the Gemini Enterprise badge in the picker. +* The request completes successfully. +* The request appears in your project's Vertex AI monitoring or Cloud Logging. + +## Using Gemini Enterprise as a team member + +Members must be signed in; the flow is rooted in the Warp session. + +1. In the Warp app, go to Settings > Agents > Warp Agent and open the Gemini Enterprise section. +2. Toggle Use Gemini Enterprise credentials on, unless the admin enforces it. +3. Check the credential status card for the current state and use Refresh to force a new mint. +4. Pick an eligible model; routed models show a Gemini Enterprise badge. + +If setup is incomplete, contact your team admin. + +## Routing and fallback behavior + +### Host priority + +Warp expands the selected model into enabled hosts and tries them in this fixed order: + +1. AWS Bedrock +2. Gemini Enterprise +3. Direct API + +The priority is fixed and not admin-configurable. A Gemini Enterprise route is used only when the request carries valid Gemini Enterprise credentials; otherwise Warp falls back to the next enabled host. + +### Failover behavior + +If a Gemini Enterprise request fails (for example, due to IAM misconfiguration or Vertex quota limits), Warp attempts the next enabled host. A fallback to a Direct API model consumes Warp credits. If no fallback is available, Warp displays a clear error. + +### Auto model selection + +Auto model selection is disabled if an admin disables any Direct API model. When Direct API models remain enabled, Auto selects the best model; if that model is enabled for Gemini Enterprise and credentials are active, the request routes through your project. + +## Billing behavior + +When a request routes through Gemini Enterprise: + +* Warp does not consume AI credits for that request. Inference is billed to your Google Cloud project. +* Platform credits still apply — local agent runs that use customer-supplied inference consume platform credits for Warp's infrastructure on Business and Enterprise plans. +* Fallbacks are billed normally — a request that falls back to a Direct API model consumes Warp credits at the standard rate. + +See The three credit buckets for details. + +## Security and data handling + +### Credential security + +* No long-lived credentials — The only long-lived credential is the member's Warp session. Access tokens are short-lived, held in memory, and not persisted. +* Nothing sensitive leaves your boundary — Warp stores only non-secret routing configuration (project ID, location, WIF audience, and optional service account email). Service account keys, refresh tokens, and credential files are never uploaded. +* Per-user identity — Each token is minted for the signed-in member, so access control and revocation stay in your identity stack. + +### Zero Data Retention (ZDR) + +Warp maintains SOC 2 compliance and has Zero Data Retention agreements with contracted LLM providers. + +However, when routing through Gemini Enterprise: + +* Your Google Cloud project settings determine data retention. +* Warp cannot enforce ZDR for requests routed through your infrastructure. +* Review Vertex AI data governance settings to control retention. + +### Auditability + +* Warp keeps conversations logged in Warp. +* Your GCP project retains provider-side logs in Vertex AI monitoring and Cloud Logging, attributed to your project. + +## Troubleshooting + +### Common errors + +* Credentials expired or invalid — The request reached Vertex AI but was rejected as unauthenticated. Click Refresh credentials or use Settings > Agents > Warp Agent > Refresh, then retry. +* Setup incomplete — The host is enabled but the WIF audience is missing. A team admin must complete the Models page configuration. +* Token exchange or impersonation failed — Verify the WIF provider's issuer (https://app.warp.dev), attribute mapping, attribute condition, and (for impersonation) that the principal set holds roles/iam.workloadIdentityUser on the service account. +* Permission denied from Vertex AI — Confirm the federated principal (or service account) holds roles/aiplatform.user and roles/serviceusage.serviceUsageConsumer on the project. +* Model not found — Confirm the model is available in your configured location and, for Claude partner models, enabled in the Vertex AI Model Garden. Check per-model reference overrides on the Models page. +* Provider quota limits — Check Vertex AI quotas and request increases if needed. + +### Debugging steps + +1. Confirm the WIF audience in the Admin Panel exactly matches the provider's resource name. +2. Check the credential status card in Settings > Agents > Warp Agent for the failing state and recovery action. +3. Verify IAM bindings for the principal set or service account. +4. Confirm the model reference and location match what's available in your project. +5. Inspect Cloud Logging for request details and errors. + +## FAQ + +### How is this different from BYOK with a Google API key? + +BYOK uses a personal API key stored on each member's device to call the Gemini Developer API. Gemini Enterprise BYOLLM routes requests to Vertex AI in your organization's GCP project using short-lived federated credentials, configured centrally by an admin, and billed and governed by your project's IAM and quota. See Bring Your Own API Key for the self-serve option. + +### Do members need the gcloud CLI installed? + +No. Gemini Enterprise mints credentials from the signed-in Warp session. No local Google tooling or configuration is required. + +### Does Gemini Enterprise work with cloud agents? + +Not yet. Gemini Enterprise currently routes interactive agent requests in the Warp app. Use AWS Bedrock BYOLLM for cloud agent BYOLLM today. + +### Which Claude models route through my project? + +Claude partner models offered on Vertex AI (Sonnet, Opus, Haiku, Fable families) can route through Gemini Enterprise when your admin enables them. Eligible models show the Gemini Enterprise badge in the picker. + +### Can admins enforce provider-only routing? + +Yes. Admins can disable Direct API access for models on the Models page so eligible requests only route through your project. Note that disabling any Direct API model also disables Auto model selection. + +## Related resources + +* Bring Your Own LLM — BYOLLM overview and AWS Bedrock setup +* Team-managed API keys and endpoints — Admin-configured shared keys and custom endpoints +* Bring Your Own API Key — Self-serve, user-level API keys +* Model Choice — Full list of supported models +* Admin Panel — Configure team settings +* Contact sales — Get help with Enterprise setup diff --git a/.agents/logs/tone_model_eval/outputs/cli-agent-conversations-resume-menu-label__claude-4-5-haiku.txt b/.agents/logs/tone_model_eval/outputs/cli-agent-conversations-resume-menu-label__claude-4-5-haiku.txt new file mode 100644 index 00000000..e2c62278 --- /dev/null +++ b/.agents/logs/tone_model_eval/outputs/cli-agent-conversations-resume-menu-label__claude-4-5-haiku.txt @@ -0,0 +1,140 @@ +--- +title: "Agent conversations in the Warp Agent CLI" +description: >- + How agent conversations work in the Warp Agent CLI: streaming responses, tool + calls, diffs, task lists, plans, plus managing and resuming conversations. +--- +import { VARS } from '@data/vars'; + +When you send the agent a prompt in the {VARS.WARP_CLI}, the conversation appears in a scrollable transcript directly in your terminal. Responses stream in as they're generated, and everything the agent does renders inline: tool calls, file diffs, questions, task lists, and plans. Conversations persist as you work: you can start new ones, browse history, compact context, and [resume after exiting](#resuming-conversations). + +## The conversation transcript + +The agent's response streams into the transcript below your prompt as it's generated. Press `Ctrl+C` once to stop a response in progress. + +Responses render as formatted Markdown, including syntax-highlighted code blocks and tables. Mermaid diagrams appear as source code in a block, images show their alt text, and long code blocks are truncated for responsiveness. + +## Tool calls + +Every tool call the agent makes appears inline in the transcript in order. Most render as a one-line status row with a state glyph and label, such as reading a file or searching your codebase. + +Some tool calls render richer, interactive content: + +* **[Shell commands](/cli/shell-commands/)** - Run in your session and stream output into the transcript. +* **[File edits](#code-diffs)** - Expandable diffs with per-file headers. +* **[Questions](#agent-questions)** - Interactive option prompts. +* **[Plans](#planning)** - Inline plan documents. + +When a tool call needs your approval, an approval card appears. See [permissions in the CLI](/cli/permissions-and-profiles/) for how approvals work. + +## Code diffs + +When the agent edits files, the edit renders as a diff in the transcript with per-file headers showing the action and change counts. Multi-file edits group under one summary header with each file's section nested beneath it. + +Diffs open fully expanded while awaiting your approval and collapse to their headers once applied. Press `e` while the approval card is active to expand or collapse all diffs at once. + +## Thinking blocks + +For models that expose reasoning, the agent's thinking streams into a collapsible `Thinking...` section and collapses to a single `Thought for` row once complete. + +## Agent questions + +When the agent needs a decision, it asks with an interactive option list that replaces the input. Press an option's number to choose it, or use the arrow keys. **Other…** accepts a free-form answer when listed options don't fit. + +Options labeled `(recommended)` are the agent's best fit. Multi-select questions mark each chosen option with a check. When the agent asks several questions at once, the card advances through them. + +## Task lists + +For multi-step work, the agent tracks progress with a task list under a `≡ Tasks` header. Status glyphs: + +* `◌` - Pending +* - In progress +* - Completed + +Canceled tasks appear struck through. As the agent finishes tasks, compact rows such as `✓ Completed (2/5)` mark progress. + +Task lists in the CLI reflect the same agent behavior as in the Warp app. Learn more about [how task lists work](/agent-platform/capabilities/task-lists/). + +## Planning + +Use the `/plan` slash command followed by a task description to have the agent research and produce a plan before making changes. You can also ask for a plan in natural language. + +The plan renders inline as a formatted document with its own header row, and an `Updated plan` entry appears when the agent revises it. Toggle the latest plan with `Ctrl+Shift+P`; while a plan is open, a hint below shows the exact shortcut. + +Planning in the CLI follows the same workflow as the Warp app. See [Planning](/agent-platform/capabilities/planning/) for how plans are created, reviewed, and executed. + +## Selecting and copying output + +Select text anywhere in the transcript by dragging with the mouse. Releasing the mouse button copies the selection automatically. + +:::note +In local sessions, the CLI writes directly to your system clipboard. Over SSH, it copies through your terminal using OSC 52 escape sequences (including from inside tmux), so the text lands on your local clipboard. Terminals that disable OSC 52 may ignore the copy. +::: + +To copy an entire conversation as Markdown, use the `/export-to-clipboard` slash command, or use `/export-to-file` to save it to a file. + +## Managing conversations + +The CLI saves every agent conversation as you work, so closing your terminal never loses progress. + +### Conversation persistence + +Conversations save to your Warp account rather than only your machine, so the same history is available in the Warp app and on other devices. Reopening a conversation restores the full transcript, including agent responses, tool calls, and file-edit diffs. You can continue prompting from where it left off. + +The CLI shows one conversation at a time: opening a past conversation replaces the current transcript, and the previous one remains available in history. You can't switch conversations while the current one is responding or a command is running; finish or stop it with `Ctrl+C` first. + +### Starting a new conversation + +Use any of these slash commands to clear the transcript and start a fresh conversation: + +* **`/new`** - Starts a new conversation. +* **`/agent`** - Same as `/new`. +* **`/clear`** - Same as `/new`. + +Each command accepts an optional prompt. For example, `/new write tests for the parser` starts a new conversation and immediately sends that prompt. To keep the history but reduce its size instead, use [`/compact`](#compacting-context). + +### Conversation history + +To browse and reopen past conversations: + +* **`/conversations`** - Run the slash command from the input. +* **`←`** - Press the left arrow key when the input is empty and the cursor is at the start. An empty input shows a `← for conversations` hint. + +The menu lists your Warp Agent conversations, including conversations started in the Warp app and completed cloud agent runs tied to your account. Start typing to filter by title. + +:::caution +If the CLI can't load conversation data from Warp's servers, the menu shows conversations from your local device only and displays a warning. Conversations from other devices reappear once the connection recovers. +::: + +To continue a cloud agent run from the CLI, or to hand the current conversation off to a cloud agent, see [cloud handoff and orchestration](/cli/cloud-and-orchestration/). + +### Resuming conversations + +There are two ways to pick a past conversation back up: + +* **The [conversation menu](#conversation-history)** - The quickest route. From a running session, press `←` or run `/conversations`, then filter to the conversation you want. +* **`warp --resume`** - Reopens a specific conversation from your shell as the CLI starts, without going through the menu. + +When you exit the CLI with a non-empty conversation, it prints the `--resume` command: +```bashTo continue this conversation, run: +warp --resume YOUR_CONVERSATION_TOKEN +``` +YOUR_CONVERSATION_TOKEN is a conversation identifier generated by Warp. For the complete list of command-line flags, see the [CLI reference](/cli/reference/). + +### Compacting context + +Long conversations eventually fill the model's context window, which can degrade response quality. The `/compact` command frees up context by asking the agent to summarize the conversation history and carry only the summary forward. + +* **`/compact`** - Summarizes the conversation history with default instructions. +* **`/compact `** - Adds custom summarization instructions. For example, `/compact keep the API design decisions` tells the agent what to preserve. + +After compaction, a collapsed **Conversation summary** block appears in the transcript, and the conversation works normally with the summary standing in for the compacted history. + +## Related pages + +* **[Permissions and profiles](/cli/permissions-and-profiles/)** - Approve, reject, or auto-approve the agent's tool calls. +* **[Running shell commands](/cli/shell-commands/)** - How commands the agent (or you) run appear in the transcript. +* **[Cloud handoff and orchestration](/cli/cloud-and-orchestration/)** - Hand off conversations to cloud agents and resume cloud runs. +* **[{VARS.WARP_CLI} reference](/cli/reference/)** - Command-line flags, slash commands, and keyboard shortcuts. +* **[Planning](/agent-platform/capabilities/planning/)** - The full planning workflow. +* **[Task lists](/agent-platform/capabilities/task-lists/)** - How agents create and update task lists. diff --git a/.agents/logs/tone_model_eval/outputs/cli-agent-conversations-resume-menu-label__claude-4-5-sonnet.txt b/.agents/logs/tone_model_eval/outputs/cli-agent-conversations-resume-menu-label__claude-4-5-sonnet.txt new file mode 100644 index 00000000..da909bd7 --- /dev/null +++ b/.agents/logs/tone_model_eval/outputs/cli-agent-conversations-resume-menu-label__claude-4-5-sonnet.txt @@ -0,0 +1,143 @@ +--- +title: "Agent conversations in the Warp Agent CLI" +description: >- + How agent conversations work in the Warp Agent CLI: streaming responses, tool + calls, diffs, task lists, plans, plus managing and resuming conversations. +--- +import { VARS } from '@data/vars'; + +When you send a prompt in the {VARS.WARP_CLI}, the conversation appears as a scrollable transcript in your terminal. Responses stream as they're generated. Tool calls, file diffs, questions, task lists, and plans render inline. Conversations persist: you can start new ones, browse history, compact context, and [resume after exiting](#resuming-conversations). + +## The conversation transcript + +The agent's response streams below your prompt. Press `Ctrl+C` once to stop a response in progress. + +Responses render as formatted Markdown, including syntax-highlighted code blocks and tables. Mermaid diagrams appear as source in a code block, images show alt text, and long code blocks truncate to keep the transcript responsive. + +## Tool calls + +Every tool call appears inline in the order it happens. Most render as a one-line status row with a state glyph and label describing the action. + +Some tool calls render richer content: + +* **[Shell commands](/cli/shell-commands/)** - Run in your session and stream output into the transcript. +* **[File edits](#code-diffs)** - Expandable diffs with per-file headers. +* **[Questions](#agent-questions)** - Interactive option prompts. +* **[Plans](#planning)** - Inline plan documents. + +When a tool call needs approval, an approval card appears in place of the input. See [permissions in the CLI](/cli/permissions-and-profiles/). + +## Code diffs + +When the agent edits files, the edit renders as a diff: + +* **Per-file sections** - Each edited file gets a header with the action and change counts. +* **Multi-file edits** - Group under one summary header (e.g., `Edited 3 files`) with each file's section nested beneath it. + +Diffs open fully expanded while the agent waits for approval, and collapse to their headers once applied. Press `e` while the approval card is active to expand or collapse all diffs at once. + +## Thinking blocks + +For models that expose reasoning, the agent's thinking streams into a collapsible section headed `Thinking...`, which collapses to a single `Thought for` row once it finishes. + +## Agent questions + +When the agent needs a decision mid-task, it asks a question with an interactive option list that replaces the input. Press an option's number to choose it, or select **Other…** for a free-form answer when listed options don't fit. + +Options the agent suggests as best fit are labeled `(recommended)`. Multi-select questions mark each chosen option with a check mark. When the agent asks several questions at once, the card advances through them. + +## Task lists + +For multi-step work, the agent tracks progress with a task list rendered in the transcript under a `≡ Tasks` header. Each task row starts with a status glyph: + +* `◌` - Pending +* - In progress +* - Completed + +Canceled tasks appear struck through. As the agent finishes tasks, compact confirmation rows such as `✓ Completed (2/5)` mark progress without repeating the whole list. + +Task lists in the CLI reflect the same agent behavior as in the Warp app. Learn more about [how task lists work](/agent-platform/capabilities/task-lists/). + +## Planning + +Use the `/plan` slash command, followed by a description of your task, to have the agent research first and produce a plan before making changes. You can also ask for a plan in natural language. + +The plan renders inline as a formatted document with its own header row showing the plan's status, and an `Updated plan` entry appears when the agent revises it. Toggle the latest plan with `Ctrl+Shift+P`; while a plan is open, a hint below it shows the exact shortcut. + +Planning in the CLI follows the same workflow as the Warp app. See [Planning](/agent-platform/capabilities/planning/). + +## Selecting and copying output + +Select text anywhere in the transcript by dragging with the mouse. Releasing the mouse button copies the selection automatically. + +:::note +In local sessions, the CLI writes directly to your system clipboard. Over SSH, it copies through your terminal using OSC 52 escape sequences (including from inside tmux). Terminals that disable OSC 52 may ignore the copy. +::: + +To copy an entire conversation as Markdown, use the `/export-to-clipboard` slash command, or use `/export-to-file` to save it to a file. + +## Managing conversations + +The CLI saves every agent conversation as you work, so closing your terminal never loses your progress. + +### Conversation persistence + +Conversations save to your Warp account rather than only to your machine, so the same history is available in the Warp app and on your other devices. Reopening a conversation restores the full transcript, including agent responses, tool calls, and file-edit diffs. You can continue from where it left off. + +The CLI shows one conversation at a time: opening a past conversation replaces the current transcript, and the previous one remains available in history. You can't switch conversations while the current conversation is responding or a command is running; finish or stop it with `Ctrl+C` first. + +### Starting a new conversation + +Use any of these slash commands to clear the transcript and start a fresh conversation: + +* **`/new`** - Starts a new conversation. +* **`/agent`** - Same as `/new`. +* **`/clear`** - Same as `/new`. + +Each command accepts an optional prompt. For example, `/new write tests for the parser` starts a new conversation and immediately sends that prompt. To keep the history but reduce its size, use [`/compact`](#compacting-context). + +### Conversation history + +To browse and reopen past conversations, open the conversation menu in either of two ways: + +* **`/conversations`** - Run the slash command from the input. +* **`←`** - Press the left arrow key when the input is empty and the cursor is at the start. The empty input shows a `← for conversations` hint as a reminder. + +The menu lists your Warp Agent conversations, including conversations started in the Warp app and completed cloud agent runs tied to your account. Start typing to filter the list by title. + +:::caution +If the CLI can't load conversation data from Warp's servers, the menu shows conversations from your local device only and displays a warning. Conversations from other devices reappear once the connection recovers. +::: + +To continue a cloud agent run from the CLI, or to hand the current conversation off to a cloud agent, see [cloud handoff and orchestration](/cli/cloud-and-orchestration/). + +### Resuming conversations + +There are two ways to pick a past conversation back up: + +* **The [conversation menu](#conversation-history)** - The quickest route. From a running session, press `←` or run `/conversations`, then filter to the conversation you want. +* **`warp --resume`** - Reopens a specific conversation from your shell as the CLI starts. + +When you exit the CLI with a non-empty conversation, it prints the `--resume` command for that conversation: +```bashTo continue this conversation, run: +warp --resume YOUR_CONVERSATION_TOKEN +``` +YOUR_CONVERSATION_TOKEN is a conversation identifier generated by Warp. For the complete list of command-line flags, see the [CLI reference](/cli/reference/). + +### Compacting context + +Long conversations eventually fill the model's context window, which can degrade response quality. The `/compact` command frees up context by asking the agent to summarize the conversation history and carry only the summary forward. + +* **`/compact`** - Summarizes the conversation history with default instructions. +* **`/compact `** - Adds custom summarization instructions. For example, `/compact keep the API design decisions` tells the agent what to preserve in the summary. + +After compaction, a collapsed **Conversation summary** block appears in the transcript, and the conversation continues with the summary standing in for the compacted history. + +## Related pages + +* **[Permissions and profiles](/cli/permissions-and-profiles/)** - Approve, reject, or auto-approve the agent's tool calls. +* **[Running shell commands](/cli/shell-commands/)** - How commands the agent (or you) run appear in the transcript. +* **[Cloud handoff and orchestration](/cli/cloud-and-orchestration/)** - Hand off conversations to cloud agents and resume cloud runs. +* **[{VARS.WARP_CLI} reference](/cli/reference/)** - Command-line flags, slash commands, and keyboard shortcuts. +* **[Planning](/agent-platform/capabilities/planning/)** - The full planning workflow. +* **[Task lists](/agent-platform/capabilities/task-lists/)** - How agents create and update task lists. diff --git a/.agents/logs/tone_model_eval/outputs/cli-agent-conversations-resume-menu-label__claude-5-1-fable-high.txt b/.agents/logs/tone_model_eval/outputs/cli-agent-conversations-resume-menu-label__claude-5-1-fable-high.txt new file mode 100644 index 00000000..cdca1aba --- /dev/null +++ b/.agents/logs/tone_model_eval/outputs/cli-agent-conversations-resume-menu-label__claude-5-1-fable-high.txt @@ -0,0 +1,143 @@ +--- +title: "Agent conversations in the Warp Agent CLI" +description: >- + How agent conversations work in the Warp Agent CLI: streaming responses, tool + calls, diffs, task lists, plans, plus managing and resuming conversations. +--- +import { VARS } from '@data/vars'; + +When you send the agent a prompt in the {VARS.WARP_CLI}, the conversation appears in a scrollable transcript in your terminal. Responses stream in as they're generated, and everything the agent does renders inline: tool calls, file diffs, questions, task lists, and plans. Conversations persist as you work. You can start new ones, browse history, compact context, and [resume after exiting](#resuming-conversations). + +## The conversation transcript + +The agent's response streams into the transcript below your prompt. Press `Ctrl+C` once to stop a response in progress. + +Responses render as formatted Markdown, including syntax-highlighted code blocks and tables. Mermaid diagrams appear as their source in a code block, images show their alt text instead of rendering inline, and very long code blocks are truncated to keep the transcript responsive. + +## Tool calls + +Every tool call appears inline in the transcript, in the order it happens. Most render as a one-line status row with a state glyph and a label describing the action, such as reading a file or searching your codebase. + +Some tool calls render richer, interactive content: + +* **[Shell commands](/cli/shell-commands/)** - Run in your session and stream their output into the transcript. +* **[File edits](#code-diffs)** - Expandable diffs with per-file headers. +* **[Questions](#agent-questions)** - Interactive option prompts. +* **[Plans](#planning)** - Inline plan documents. + +When a tool call needs your approval before it runs, an approval card appears in place of the input. See [permissions in the CLI](/cli/permissions-and-profiles/). + +## Code diffs + +When the agent edits files, the edit renders as a diff in the transcript: + +* **Per-file sections** - Each edited file gets a header with the action and change counts. +* **Multi-file edits** - Group under one summary header (for example, `Edited 3 files`) with each file's section nested beneath it. + +Diffs open fully expanded while the agent waits for your approval and collapse to their headers once the edits are applied. Press `e` while the approval card is active to expand or collapse all diffs at once. + +## Thinking blocks + +For models that expose their reasoning, the agent's thinking streams into a collapsible section headed `Thinking...`. It collapses to a single `Thought for` row once it finishes. + +## Agent questions + +When the agent needs a decision mid-task, it asks a question with an interactive option list that temporarily replaces the input. Use the arrow keys or press an option's number to choose it. **Other…** accepts a free-form answer when the listed options don't fit. + +Options the agent suggests are labeled `(recommended)`. Multi-select questions mark each chosen option with a check mark. When the agent asks several questions at once, the card advances through them. + +## Task lists + +For multi-step work, the agent tracks its progress with a task list in the transcript under a `≡ Tasks` header. Each task row starts with a status glyph: + +* `◌` - Pending +* - In progress +* - Completed + +Canceled tasks appear struck through. As the agent finishes tasks, compact confirmation rows such as `✓ Completed (2/5)` mark progress without repeating the whole list. + +Task lists in the CLI reflect the same agent behavior as in the Warp app. See [how task lists work](/agent-platform/capabilities/task-lists/). + +## Planning + +Use the `/plan` slash command, followed by a description of your task, to have the agent research first and produce a plan before making changes. You can also ask for a plan in natural language. + +The plan renders inline in the transcript as a formatted document with a header row showing its status. An `Updated plan` entry appears when the agent revises it. Toggle the latest plan with `Ctrl+Shift+P`; while a plan is open, a hint below it shows the shortcut. + +Planning in the CLI follows the same workflow as the Warp app. See [Planning](/agent-platform/capabilities/planning/) for how plans are created, reviewed, and executed. + +## Selecting and copying output + +Select text anywhere in the transcript by dragging with the mouse. Releasing the mouse button copies the selection. + +:::note +In local sessions, the CLI writes directly to your system clipboard. Over SSH, it copies through your terminal using OSC 52 escape sequences (including from inside tmux), so the text lands on your local clipboard. Terminals that disable OSC 52 may ignore the copy. +::: + +To copy an entire conversation as Markdown, use `/export-to-clipboard`. To save it to a file, use `/export-to-file`. + +## Managing conversations + +The CLI saves every agent conversation as you work, so closing your terminal doesn't lose your progress. + +### Conversation persistence + +Conversations save to your Warp account, not only to your machine, so the same history is available in the Warp app and on your other devices. Reopening a conversation restores the full transcript, including agent responses, tool calls, and file-edit diffs, and you can keep prompting from where it left off. + +The CLI shows one conversation at a time. Opening a past conversation replaces the current transcript, and the previous one remains in history. You can't switch conversations while the current conversation is responding or a command is running; finish or stop it with `Ctrl+C` first. + +### Starting a new conversation + +Use any of these slash commands to clear the transcript and start a fresh conversation: + +* **`/new`** - Starts a new conversation. +* **`/agent`** - Same as `/new`. +* **`/clear`** - Same as `/new`. + +Each command accepts an optional prompt. For example, `/new write tests for the parser` starts a new conversation and sends that prompt to the agent. To keep the history but reduce its size, use [`/compact`](#compacting-context). + +### Conversation history + +To browse and reopen past conversations, open the conversation menu in either of two ways: + +* **`/conversations`** - Run the slash command from the input. +* **`←`** - Press the left arrow key when the input is empty and the cursor is at the start. The empty input shows a `← for conversations` hint. + +The menu lists your Warp Agent conversations, including conversations started in the Warp app and completed cloud agent runs tied to your account. Start typing to filter the list by title. + +:::caution +If the CLI can't load conversation data from Warp's servers, the menu shows conversations from your local device only and displays a warning. Conversations from other devices reappear once the connection recovers. +::: + +To continue a cloud agent run from the CLI, or to hand the current conversation off to a cloud agent, see [cloud handoff and orchestration](/cli/cloud-and-orchestration/). + +### Resuming conversations + +There are two ways to pick a past conversation back up: + +* **The [conversation menu](#conversation-history)** - Use this most of the time. From a running session, press `←` or run `/conversations`, then filter to the conversation you want. +* **`warp --resume`** - Reopens a specific conversation from your shell as the CLI starts, without the menu. + +When you exit the CLI with a non-empty conversation, it prints the `--resume` command for that conversation: +```bashTo continue this conversation, run: +warp --resume YOUR_CONVERSATION_TOKEN +``` +YOUR_CONVERSATION_TOKEN is a conversation identifier generated by Warp. For the complete list of command-line flags, see the [CLI reference](/cli/reference/). + +### Compacting context + +Long conversations eventually fill the model's context window, which can degrade response quality. `/compact` frees context by asking the agent to summarize the conversation so far and carry only the summary forward. + +* **`/compact`** - Summarizes the conversation history with default instructions. +* **`/compact `** - Adds custom summarization instructions. For example, `/compact keep the API design decisions` tells the agent what to preserve. + +After compaction, a collapsed **Conversation summary** block appears in the transcript, and the conversation continues with the summary standing in for the compacted history. + +## Related pages + +* **[Permissions and profiles](/cli/permissions-and-profiles/)** - Approve, reject, or auto-approve the agent's tool calls. +* **[Running shell commands](/cli/shell-commands/)** - How commands the agent (or you) run appear in the transcript. +* **[Cloud handoff and orchestration](/cli/cloud-and-orchestration/)** - Hand off conversations to cloud agents and resume cloud runs. +* **[{VARS.WARP_CLI} reference](/cli/reference/)** - Command-line flags, slash commands, and keyboard shortcuts. +* **[Planning](/agent-platform/capabilities/planning/)** - The full planning workflow. +* **[Task lists](/agent-platform/capabilities/task-lists/)** - How agents create and update task lists. diff --git a/.agents/logs/tone_model_eval/outputs/cli-agent-conversations-resume-menu-label__gpt-5-mini.txt b/.agents/logs/tone_model_eval/outputs/cli-agent-conversations-resume-menu-label__gpt-5-mini.txt new file mode 100644 index 00000000..d8b827f2 --- /dev/null +++ b/.agents/logs/tone_model_eval/outputs/cli-agent-conversations-resume-menu-label__gpt-5-mini.txt @@ -0,0 +1,138 @@ +title: "Agent conversations in the Warp Agent CLI" +description: >- + How agent conversations work in the Warp Agent CLI: streaming responses, tool + calls, diffs, task lists, plans, plus managing and resuming conversations. +import { VARS } from '@data/vars'; + +When you send a prompt in the {VARS.WARP_CLI}, the conversation appears as a scrollable transcript in your terminal. Responses stream as they are generated and the agent's actions render inline: tool calls, file diffs, questions, task lists, and plans. Conversations persist so you can start new ones, browse history, compact context, and resume after exiting. + +## The conversation transcript + +Responses stream into the transcript below your prompt. Press Ctrl+C once to stop a response in progress. + +Responses render as Markdown with syntax-highlighted code blocks and tables. Caveats: Mermaid diagrams show as source in a code block, images display alt text, and very long code blocks may be truncated to keep the transcript responsive. + +## Tool calls + +Tool calls appear inline in transcript order. Most show a one-line status row with a state glyph and a short label (for example, reading a file or searching the codebase). + +Some tool calls render richer content: + +* Shell commands - run in your session and stream output into the transcript. +* File edits - expandable diffs with per-file headers. +* Questions - interactive option prompts. +* Plans - inline plan documents. + +When a tool call requires approval, an approval card replaces the input. See Permissions and Profiles for approval behavior. + +## Code diffs + +File edits render as diffs: + +* Per-file sections include a header with the action and change counts. +* Multi-file edits group under a single summary header (for example, "Edited 3 files") with each file nested beneath it. + +Diffs are fully expanded while waiting for your approval and collapse to headers once applied. Press e while the approval card is active to expand or collapse all diffs. + +## Thinking blocks + +For models that expose reasoning, the agent's thinking streams into a collapsible "Thinking..." section that collapses to a single "Thought for" row when finished. + +## Agent questions + +When the agent needs a decision mid-task, it asks a question with an interactive option list that temporarily replaces the input. Use the arrow keys or press an option's number to choose it. Choose "Other…" to enter a free-form answer. + +Suggested options are labeled "(recommended)". Multi-select questions show checks for chosen options. If the agent asks several questions, the card advances through them in order. + +## Task lists + +For multi-step work, the agent tracks progress under a ≡ Tasks header. Each task row starts with a status glyph: + +* ◌ — Pending +* ● — In progress +* ✓ — Completed + +Canceled tasks appear struck through. As tasks complete, compact confirmation rows like "✓ Completed (2/5)" mark progress without repeating the whole list. + +Task lists in the CLI match agent behavior in the Warp app. See Task Lists for details. + +## Planning + +Use the /plan slash command plus a description to have the agent research first and produce a plan before making changes. You can also request a plan in natural language. + +A plan renders inline with its own header and status. An "Updated plan" entry appears when the agent revises it. Toggle the latest plan with Ctrl+Shift+P; while a plan is open a hint shows the exact shortcut. + +Planning in the CLI follows the same workflow as the Warp app. See Planning for the full process. + +## Selecting and copying output + +Select text by dragging with the mouse; releasing the mouse copies the selection automatically. + +Note: In local sessions the CLI writes to your system clipboard. Over SSH it uses OSC 52 escape sequences (including inside tmux) so the text lands on your local clipboard. Terminals that disable OSC 52 may ignore the copy. + +To copy an entire conversation as Markdown use /export-to-clipboard, or use /export-to-file to save it to a file. + +## Managing conversations + +The CLI saves every conversation as you work, so closing your terminal does not lose progress. + +### Conversation persistence + +Conversations save to your Warp account and are available in the Warp app and on other devices. Reopening a conversation restores the full transcript, including responses, tool calls, and diffs. You can continue prompting from where it left off. + +The CLI shows one conversation at a time: opening a past conversation replaces the current transcript. You cannot switch conversations while the current conversation is responding or a command is running; finish or stop it with Ctrl+C first. + +### Starting a new conversation + +Use any of these slash commands to clear the transcript and start a fresh conversation: + +* /new — Starts a new conversation. +* /agent — Same as /new. +* /clear — Same as /new. + +Each command accepts an optional prompt. For example, /new write tests for the parser starts a new conversation and immediately sends that prompt. To keep history but reduce its size, use /compact. + +### Conversation history + +Open the conversation menu in two ways: + +* /conversations — Run the slash command from the input. +* ← — Press the left arrow key when the input is empty and the cursor is at the start. A "← for conversations" hint appears as a reminder. + +The menu lists your Warp Agent conversations, including those started in the Warp app and completed cloud agent runs tied to your account. Start typing to filter by title. + +If the CLI cannot load conversation data from Warp's servers, the menu shows local conversations only and displays a warning. Conversations from other devices reappear once the connection recovers. + +To continue a cloud agent run from the CLI, or to hand the current conversation off to a cloud agent, see Cloud Handoff and Orchestration. + +### Resuming conversations + +Pick up a past conversation two ways: + +* Conversation menu — press ← or run /conversations, then filter to the conversation you want. +* warp --resume — Reopen a specific conversation from your shell as the CLI starts. + +When you exit the CLI with a non-empty conversation, it prints the --resume command for that conversation: + +To continue this conversation, run: +warp --resume YOUR_CONVERSATION_TOKEN + +YOUR_CONVERSATION_TOKEN is a conversation identifier generated by Warp. See the CLI reference for command-line flags. + +### Compacting context + +Long conversations can fill the model's context window and degrade responses. Use /compact to free context by asking the agent to summarize history and carry only the summary forward. + +* /compact — Summarizes the conversation history with default instructions. +* /compact — Adds custom summarization instructions (for example, /compact keep the API design decisions). + +After compaction a collapsed Conversation summary block appears in the transcript and the conversation continues with the summary standing in for compacted history. + +## Related pages + +* Permissions and profiles — Approve, reject, or auto-approve the agent's tool calls. +* Running shell commands — How commands the agent (or you) run appear in the transcript. +* Cloud handoff and orchestration — Hand off conversations to cloud agents and resume cloud runs. +* {VARS.WARP_CLI} reference — Command-line flags, slash commands, and keyboard shortcuts. +* Planning — The full planning workflow. +* Task lists — How agents create and update task lists. diff --git a/.agents/logs/tone_model_eval/outputs/quickstart-synthetic-verbose-seed__claude-4-5-haiku.txt b/.agents/logs/tone_model_eval/outputs/quickstart-synthetic-verbose-seed__claude-4-5-haiku.txt new file mode 100644 index 00000000..914b0aa1 --- /dev/null +++ b/.agents/logs/tone_model_eval/outputs/quickstart-synthetic-verbose-seed__claude-4-5-haiku.txt @@ -0,0 +1,19 @@ +--- +description: Get Warp installed and running your first agent conversation in about five minutes. +--- + +## Installing Warp + +Download the installer for your platform and run it. Follow the on-screen prompts; the installation takes about two minutes. + +## Signing in + +After installation, sign in. This unlocks Warp's full feature set and syncs your settings across devices. + +## Running your first agent conversation + +Open a new tab, type your request, and press Enter. The agent starts working on the task immediately. + +## Next steps + +You've installed Warp and run your first agent conversation. See [Agent Mode](/agents/using-agent-mode/) for a deeper look at multi-turn agent conversations. diff --git a/.agents/logs/tone_model_eval/outputs/quickstart-synthetic-verbose-seed__claude-4-5-sonnet.txt b/.agents/logs/tone_model_eval/outputs/quickstart-synthetic-verbose-seed__claude-4-5-sonnet.txt new file mode 100644 index 00000000..13a3738e --- /dev/null +++ b/.agents/logs/tone_model_eval/outputs/quickstart-synthetic-verbose-seed__claude-4-5-sonnet.txt @@ -0,0 +1,21 @@ +--- +description: Get Warp installed and running your first agent conversation in about five minutes. +--- + +Install Warp and run your first agent conversation in about five minutes. + +## Installing Warp + +Download the official installer for your platform and run it. Follow the on-screen prompts. Installation takes a couple of minutes. + +## Signing in + +After installation completes, sign in. Signing in syncs your settings across devices and unlocks Warp's features. + +## Running your first agent conversation + +Open a new tab, type your request, and press Enter. The agent handles the task. + +## Next steps + +You've installed Warp and run your first agent conversation. See [Agent Mode](/agents/using-agent-mode/) for a deeper look at multi-turn agent conversations. diff --git a/.agents/logs/tone_model_eval/outputs/quickstart-synthetic-verbose-seed__claude-5-1-fable-high.txt b/.agents/logs/tone_model_eval/outputs/quickstart-synthetic-verbose-seed__claude-5-1-fable-high.txt new file mode 100644 index 00000000..155cfe74 --- /dev/null +++ b/.agents/logs/tone_model_eval/outputs/quickstart-synthetic-verbose-seed__claude-5-1-fable-high.txt @@ -0,0 +1,19 @@ +--- +description: Get Warp installed and running your first agent conversation in about five minutes. +--- +Install Warp, sign in, and run your first agent conversation in about five minutes. + +## Installing Warp + +Download the official installer for your platform, run it, and follow the on-screen prompts. Installation takes a couple of minutes. + +## Signing in + +After installation, sign in. Signing in gives you access to all of Warp's features and syncs your settings across devices. + +## Running your first agent conversation + +Open a new tab, type your request, and press `Enter`. The agent works on the task for you. + +## Next steps +You've installed Warp and run your first agent conversation. See [Agent Mode](/agents/using-agent-mode/) for a deeper look at multi-turn agent conversations. diff --git a/.agents/logs/tone_model_eval/outputs/quickstart-synthetic-verbose-seed__gpt-5-mini.txt b/.agents/logs/tone_model_eval/outputs/quickstart-synthetic-verbose-seed__gpt-5-mini.txt new file mode 100644 index 00000000..cc39f189 --- /dev/null +++ b/.agents/logs/tone_model_eval/outputs/quickstart-synthetic-verbose-seed__gpt-5-mini.txt @@ -0,0 +1,19 @@ +description: Get Warp installed and running your first agent conversation in about five minutes. + +This quickstart shows how to install Warp and run your first agent conversation in about five minutes. + +## Install Warp + +Download and run the official installer for your platform. Follow the installer prompts to complete setup. + +## Sign in + +After installation, sign in to enable settings sync and agent features. Signing in also lets your Warp session mint short-lived credentials for integrations. + +## Run your first agent conversation + +Open a new tab, type a request, and press Enter. The agent will respond and continue the conversation as you follow up. + +## Next steps + +To learn more about multi-turn agent conversations, see Agent Mode. diff --git a/.agents/logs/tone_model_eval_runs.md b/.agents/logs/tone_model_eval_runs.md new file mode 100644 index 00000000..16abdb32 --- /dev/null +++ b/.agents/logs/tone_model_eval_runs.md @@ -0,0 +1,43 @@ +# Tone/concision model eval run log + +Written whenever `.agents/skills/tone_model_eval/` runs a full comparison +(GROW-6139 and any later re-run). Records the actual per-model scores and the +recommendation verdict, plus a pointer to the raw rows/report for the +regression-comparison uses cited in `out_of_repo_handoff.md` step 6. + +Newest entries first. Prepend, do not append. + +Entry format: + +```markdown +## YYYY-MM-DD — [adopt | no-difference | cheaper-model-recommended] +- **Fixtures**: N (ids) +- **Candidates**: model-id (role), ... +- **Judge model**: model-id +- **Composite scores**: model-id N.NN/5, ... +- **Verdict**: one line per recommendation arm +- **Data**: paths to rows.jsonl / report.json / report.md / per-row candidate output files +- **Oz run**: [URL] +- **Notes**: anything unusual (assumptions, retries, judge disagreements) +``` + +--- + +## 2026-09-10 — no-difference + +- **Fixtures**: 3 (`cli-agent-conversations-resume-menu-label` — feature-doc, `byollm-gemini-enterprise-google-cloud-setup` — procedural, `quickstart-synthetic-verbose-seed` — synthetic quickstart) +- **Candidates**: `claude-5-1-fable-high` (Fable 5.1), `claude-4-5-sonnet` (current-default stand-in — **documented assumption**, see Notes), `claude-4-5-haiku` (cheaper), `gpt-5-mini` (cheaper) +- **Judge model**: `gemini-3.1-pro` (distinct family from every candidate; no same-family bias) +- **Composite scores**: `claude-5-1-fable-high` 4.33/5 (concision 4.67, avoids-over-explaining 4.67, technical-fidelity 3.67); `claude-4-5-sonnet` 4.11/5 (4.33 / 4.33 / 3.67); `claude-4-5-haiku` 3.89/5 (4.33 / 4.33 / 3.00); `gpt-5-mini` 3.44/5 (4.33 / 4.33 / 1.67). Combined mechanical violations (tone-buzzword + tone-meta-opener): 0 for every candidate. +- **Verdict**: + - Adopt Fable-5.1-derived guidance: **no meaningful difference found** — concision-dimension margin over the default was 0.33 (need ≥1.0), mechanical-violation reduction was 0% (need ≥30%). + - Recommend a cheaper model for production copy passes: **no candidate met the threshold** — `claude-4-5-sonnet` (judge gap 0.22, technical fidelity 3.67 < 4.0 min), `claude-4-5-haiku` (gap 0.44, fidelity 3.00 < 4.0), `gpt-5-mini` (gap 0.89, fidelity 1.67 < 4.0) all fail on technical fidelity. +- **Data**: `.agents/logs/tone_model_eval/2026-09-10-rows.jsonl`, `.agents/logs/tone_model_eval/2026-09-10-report.json`, `.agents/logs/tone_model_eval/2026-09-10-report.md`, and the 12 raw candidate rewrites (all 4 candidates × all 3 fixtures) under `.agents/logs/tone_model_eval/outputs/__.txt` +- **Oz run**: see GROW-6139 +- **Notes**: First run (GROW-6139). `claude-4-5-sonnet` stands in for "the current production default model for docs drafting skills" as a **documented assumption**: live `oz-dev schedule list`/`schedule get` discovery against every schedule referencing docs drafting/audit skills found no explicit `model_id` in any schedule config — model selection for ad hoc/event-triggered drafting runs lives in the Warp app's Agent Profile UI (per `out_of_repo_handoff.md`), which isn't inspectable from this environment. `auto` was considered and rejected as the stand-in because it's a router that can resolve to different underlying models across calls, which would break the fixed-model comparison this eval depends on. + + Every technical-fidelity score below 5 was manually spot-checked against the source text and reflects a real defect, not judge noise. Two examples, each verifiable against the committed output file: + - `gpt-5-mini` introduced an unsupported claim not present in the original quickstart draft ("Signing in also lets your Warp session mint short-lived credentials for integrations.") — see `.agents/logs/tone_model_eval/outputs/quickstart-synthetic-verbose-seed__gpt-5-mini.txt` line 11, scored in the `technical_fidelity: 1` row for `(quickstart-synthetic-verbose-seed, gpt-5-mini)` in `2026-09-10-rows.jsonl`. + - `claude-4-5-haiku` dropped the original's "The priority is fixed and not admin-configurable" qualifier on host-priority order — compare `.agents/logs/tone_model_eval/outputs/byollm-gemini-enterprise-google-cloud-setup__claude-4-5-haiku.txt` ("### Host priority" section, no configurability qualifier) against the fixture's before text at commit `ca39ad1a0db873221fabf4bfc1f0e2417724b83a`, scored in the `technical_fidelity: 1` row for `(byollm-gemini-enterprise-google-cloud-setup, claude-4-5-haiku)` in `2026-09-10-rows.jsonl`. + + Per this outcome, `out_of_repo_handoff.md`'s checklist is skipped — no model/schedule/Agent Profile change to make.