diff --git a/crates/buzz-agent/README.md b/crates/buzz-agent/README.md index 6032dd11392..b972ffea1a4 100644 --- a/crates/buzz-agent/README.md +++ b/crates/buzz-agent/README.md @@ -242,7 +242,7 @@ lifecycle hook — see [MCP_DRIVEN_HOOKS.md](../../docs/MCP_DRIVEN_HOOKS.md). | Block Gateway | `openai` | `POST {base}/chat/completions` | gpt-5, claude | | OpenRouter | `openrouter` | `POST {base}/chat/completions` | anything they route (extended-thinking replay, provider-agnostic tool calling) | | Databricks | `databricks` | `POST {host}/serving-endpoints/{model}/invocations` | goose-claude-4-6-sonnet | -| Databricks AI Gateway v2 | `databricks_v2` | `POST {host}/ai-gateway/{provider}/v1/...` | workspace endpoints and Unity Catalog model-service FQNs; UC GPT-5+ services use OpenAI Responses; other UC FQNs use MLflow Chat Completions | +| Databricks AI Gateway v2 | `databricks_v2` | `POST {host}/ai-gateway/{provider}/v1/...` | workspace endpoints and Unity Catalog model-service FQNs; UC Claude services use Anthropic Messages, UC GPT-5+ services use OpenAI Responses, and unknown UC FQNs use MLflow Chat Completions | The optional `DATABRICKS_MODEL_FILTER` applies only to model discovery. Each comma-separated entry is trimmed and matched against the complete raw ID with case-sensitive `*` (zero or more characters) and `?` (one Unicode character) semantics; patterns are OR-ed. Unset or blank preserves the full authenticated catalog. A nonblank value containing no usable patterns is rejected. This controls picker visibility only; Databricks and Unity Catalog permissions remain the authorization boundary. A filtered-empty result is authoritative and does not restore the built-in fallback models. diff --git a/crates/buzz-agent/src/config.rs b/crates/buzz-agent/src/config.rs index 5b2f1d659d2..f8139358365 100644 --- a/crates/buzz-agent/src/config.rs +++ b/crates/buzz-agent/src/config.rs @@ -302,6 +302,15 @@ pub fn anthropic_thinking_config( } ThinkingMode::None | ThinkingMode::OmitFields => { // Non-thinking model, or unknown/unverified Anthropic name: omit rather than guess. + // A persisted explicit setting can outlive a model switch, so make the + // discarded choice observable even though fresh clients advertise no explicit effort. + if effort != ThinkingEffort::None { + tracing::warn!( + model = effective_model, + requested = effort.openai_effort_str(), + "BUZZ_AGENT_THINKING_EFFORT is unsupported for this unverified Anthropic model; omitting thinking fields" + ); + } (None, None) } } @@ -2187,6 +2196,29 @@ mod tests { ); } + #[test] + fn normalize_effort_for_uncurated_claude_fqn_is_a_noop() { + let model = "catalog.schema.claude-sonnet-custom"; + assert!(crate::model_capabilities::resolve("databricks_v2", model) + .supported_efforts + .is_empty()); + for effort in [ + ThinkingEffort::None, + ThinkingEffort::Minimal, + ThinkingEffort::Low, + ThinkingEffort::Medium, + ThinkingEffort::High, + ThinkingEffort::XHigh, + ThinkingEffort::Max, + ] { + assert_eq!( + normalize_effort_for_databricks_v2(effort, model), + effort, + "Anthropic-routed FQNs must never enter OpenAI effort resolution" + ); + } + } + // ---- normalize_effort_for_anthropic_route ---- #[test] diff --git a/crates/buzz-agent/src/llm.rs b/crates/buzz-agent/src/llm.rs index 433ece7f23e..afe8b87ede2 100644 --- a/crates/buzz-agent/src/llm.rs +++ b/crates/buzz-agent/src/llm.rs @@ -2856,10 +2856,13 @@ mod tests { } #[tokio::test] - async fn databricks_v2_model_service_fqn_summary_uses_mlflow_chat() { + async fn databricks_v2_claude_model_service_fqn_summary_uses_anthropic_messages() { let model = "catalog.schema.claude-gpt-5"; - let (base_url, captured) = - spawn_sequence_stub(vec![StubHttpResponse::ok(chat_response("summary"))]).await; + let response = json!({ + "content": [{"type": "text", "text": "summary"}], + "stop_reason": "end_turn" + }); + let (base_url, captured) = spawn_sequence_stub(vec![StubHttpResponse::ok(response)]).await; let mut config = cfg(Provider::DatabricksV2); config.base_url = base_url; let llm = Llm::new(&config).unwrap(); @@ -2875,11 +2878,11 @@ mod tests { .iter() .find(|request| request.method == "POST") .expect("summary must issue one POST"); - assert_eq!(request.path, "/v1/ai-gateway/mlflow/v1/chat/completions"); + assert_eq!(request.path, "/v1/ai-gateway/anthropic/v1/messages"); let body = request.body.as_ref().expect("summary body"); assert_eq!(body["model"], model); assert!(body["messages"].is_array()); - assert_eq!(body["max_completion_tokens"], 128); + assert_eq!(body["max_tokens"], 128); } fn image_history() -> Vec { @@ -3295,10 +3298,19 @@ mod tests { fn databricks_v2_model_service_fqn_shape_is_strict_and_precedes_manifest() { use crate::model_capabilities::{resolve, DatabricksV2Route as Manifest}; - for model in [ - "catalog.schema.service", - "catalog.schema.claude-gpt-5", - "data_tools.goose.kimi-k3", + for (model, expected) in [ + ( + "catalog.schema.service", + DatabricksV2Route::MlflowChatCompletions, + ), + ( + "catalog.schema.claude-gpt-5", + DatabricksV2Route::AnthropicMessages, + ), + ( + "data_tools.goose.kimi-k3", + DatabricksV2Route::MlflowChatCompletions, + ), ] { assert!( crate::model_capabilities::is_databricks_model_service_fqn(model), @@ -3306,8 +3318,8 @@ mod tests { ); assert_eq!( databricks_v2_route(model), - DatabricksV2Route::MlflowChatCompletions, - "FQN route must precede manifest family inference: {model}" + expected, + "only a Claude service component may select Anthropic Messages: {model}" ); } diff --git a/crates/buzz-agent/src/llm_fqn_tests.rs b/crates/buzz-agent/src/llm_fqn_tests.rs index e122318b01f..256f7417447 100644 --- a/crates/buzz-agent/src/llm_fqn_tests.rs +++ b/crates/buzz-agent/src/llm_fqn_tests.rs @@ -127,11 +127,11 @@ async fn opus_uc_replays_ordered_signed_content_with_high_effort() { } #[tokio::test] -async fn neighboring_opus_uc_services_stay_on_mlflow_chat() { +async fn neighboring_opus_uc_services_use_anthropic_messages_without_effort() { for model in ["system.ai.claude-opus-5-5", "other.goose.goose-claude-opus-5-5", "data_workflow_tools.goose.goose-claude-opus-5-5-preview"] { let (base_url, captured) = spawn_sequence_stub(vec![StubHttpResponse::ok(json!({ - "choices":[{"finish_reason":"stop", "message":{"content":"ok"}}] + "stop_reason":"end_turn", "content":[{"type":"text", "text":"ok"}] }))]).await; let mut config = cfg(Provider::DatabricksV2); config.base_url = base_url; @@ -139,11 +139,12 @@ async fn neighboring_opus_uc_services_stay_on_mlflow_chat() { Llm::new(&config).unwrap().complete(&config, "system", &[HistoryItem::User("hi".into())], &[], model).await.unwrap(); let requests = captured.lock().await; let post = requests.iter().find(|r| r.method == "POST").unwrap(); - assert_eq!(post.path, "/v1/ai-gateway/mlflow/v1/chat/completions"); + assert_eq!(post.path, "/v1/ai-gateway/anthropic/v1/messages"); let body = post.body.as_ref().unwrap(); assert_eq!(body["model"], model); - assert_eq!(body["reasoning_effort"], "high"); + assert!(body.get("reasoning_effort").is_none()); assert!(body.get("thinking").is_none()); + assert!(body.get("output_config").is_none()); } } @@ -163,3 +164,53 @@ fn native_thinking_tail_is_not_cache_stamped_or_leaked_to_chat() { assert!(!responses_body(&config, "system", &history, &[], "other", None).to_string().contains("anthropic_content")); } } + +#[tokio::test] +async fn uncurated_claude_fqn_completion_and_summary_use_anthropic_messages_without_effort() { + let response = json!({ + "content": [{"type": "text", "text": "ok"}], + "stop_reason": "end_turn" + }); + let (base_url, captured) = spawn_sequence_stub(vec![ + StubHttpResponse::ok(response.clone()), + StubHttpResponse::ok(response), + ]) + .await; + let mut config = cfg(Provider::DatabricksV2); + config.base_url = base_url; + config.thinking_effort = Some(ThinkingEffort::High); + let model = "catalog.schema.claude-sonnet-custom"; + let llm = Llm::new(&config).unwrap(); + + assert_eq!( + llm.complete( + &config, + "system", + &[HistoryItem::User("hello".into())], + &[], + model, + ) + .await + .unwrap() + .text, + "ok" + ); + assert_eq!( + llm.summarize(&config, "system", "conversation", 128, model) + .await + .unwrap(), + "ok" + ); + + let posts = captured.lock().await; + assert_eq!(posts.len(), 2); + for request in &*posts { + assert_eq!(request.path, "/v1/ai-gateway/anthropic/v1/messages"); + let body = request.body.as_ref().unwrap(); + assert_eq!(body["model"], model); + assert!(body.get("reasoning_effort").is_none()); + assert!(body.get("thinking").is_none()); + assert!(body.get("output_config").is_none()); + } + assert_eq!(posts[1].body.as_ref().unwrap()["max_tokens"], 128); +} diff --git a/crates/buzz-agent/src/model_capabilities.rs b/crates/buzz-agent/src/model_capabilities.rs index df5ff5cfa9f..207544a3380 100644 --- a/crates/buzz-agent/src/model_capabilities.rs +++ b/crates/buzz-agent/src/model_capabilities.rs @@ -293,8 +293,10 @@ fn prefix_matches(token: &str, s: &str) -> bool { /// Databricks Unity Catalog model-service names are catalog data, not model /// family hints. Both capability interpreters use this shape check before /// family matching so services cannot inherit endpoint capabilities accidentally. -/// Verified exact records may supply capabilities; other GPT-5+ services have -/// a route-only Responses exception. +/// Verified exact records may supply capabilities. Among uncurated services, +/// Claude routes through Anthropic Messages with no advertised effort choices, +/// GPT-5+ routes through OpenAI Responses with neutral fallback effort, and all +/// others retain the provider fallback route. pub(crate) fn is_databricks_model_service_fqn(model: &str) -> bool { let mut components = model.split('.'); let (Some(catalog), Some(schema), Some(service)) = @@ -309,6 +311,16 @@ pub(crate) fn is_databricks_model_service_fqn(model: &str) -> bool { }) && components.next().is_none() } +/// Route uncurated Claude UC services through Anthropic Messages without +/// borrowing effort capabilities from the service name. +fn fqn_requires_anthropic_messages(model: &str) -> bool { + let Some(service) = model.rsplit('.').next() else { + return false; + }; + let lower = service.to_ascii_lowercase(); + strip_catalog_prefix(&lower, &manifest().family_tokens).starts_with("claude-") +} + /// Route GPT-5+ UC services to Responses without borrowing endpoint effort facts. /// Match the first family token in the service only, preserving the existing /// boundary semantics (e.g. `claude-gpt-5` is not a GPT service). @@ -335,9 +347,10 @@ pub fn resolve(provider: &str, raw_model_id: &str) -> CapabilityResult { let canon = canonical_provider(provider); let blank = raw_model_id.trim().is_empty(); - // Uncurated FQNs keep neutral effort capabilities. Exact records are verified - // service contracts, not family-name inference. The GPT-5+ route-only fallback - // inspects just the service component, never catalog/schema names. + // Exact records are verified service contracts, not family-name inference. + // Among uncurated FQNs, Claude exposes no effort choices; other services + // retain neutral fallback effort. Route fallbacks inspect only the service + // component, never catalog/schema names. let model_service_fqn = canon == "databricks_v2" && is_databricks_model_service_fqn(raw_model_id); @@ -410,16 +423,33 @@ pub fn resolve(provider: &str, raw_model_id: &str) -> CapabilityResult { } else { &pair.concrete_unknown }; + let fqn_anthropic_messages = model_service_fqn && fqn_requires_anthropic_messages(raw_model_id); CapabilityResult { + // Route inference does not prove thinking support. Expose no choices for + // an unverified Claude FQN; verified exact records returned above. thinking_mode: state.thinking_mode, - supported_efforts: &state.supported_efforts, - default_effort: state.default_effort, - databricks_v2_wire_route: if model_service_fqn && fqn_requires_responses(raw_model_id) { + supported_efforts: if fqn_anthropic_messages { + &[] + } else { + &state.supported_efforts + }, + default_effort: if fqn_anthropic_messages { + None + } else { + state.default_effort + }, + databricks_v2_wire_route: if fqn_anthropic_messages { + DatabricksV2Route::AnthropicMessages + } else if model_service_fqn && fqn_requires_responses(raw_model_id) { DatabricksV2Route::OpenaiResponses } else { state.databricks_v2_wire_route }, - normalization_policy: state.normalization_policy, + normalization_policy: if fqn_anthropic_messages { + NormalizationPolicy::None + } else { + state.normalization_policy + }, registry_label: None, } } @@ -755,6 +785,11 @@ mod tests { Q::Vector { id: "dbv2-opus-uc-service", provider: "databricks_v2", raw_model_id: "data_workflow_tools.goose.goose-claude-opus-5-5-preview", note: None }, Q::Vector { id: "dbv2-opus-uc-system", provider: "databricks_v2", raw_model_id: "system.ai.claude-opus-5-5", note: None }, Q::Vector { id: "dbv2-opus-uc-provider", provider: "openai", raw_model_id: "data_workflow_tools.goose.goose-claude-opus-5-5", note: None }, + Q::Section { group: "Uncurated Databricks Claude FQN routing", note: Some("Only the service component selects Anthropic Messages; unverified Claude FQNs expose no effort choices.") }, + Q::Vector { id: "dbv2-fqn-claude-anthropic", provider: "databricks_v2", raw_model_id: "catalog.schema.claude-sonnet-custom", note: None }, + Q::Vector { id: "dbv2-fqn-claude-case", provider: "databricks_v2", raw_model_id: "catalog.schema.GOOSE-CLAUDE-SONNET-CUSTOM", note: None }, + Q::Vector { id: "dbv2-fqn-claude-catalog-inert", provider: "databricks_v2", raw_model_id: "claude-catalog.schema.service", note: None }, + Q::Vector { id: "dbv2-fqn-claude-schema-inert", provider: "databricks_v2", raw_model_id: "catalog.claude-schema.service", note: None }, Q::Section { group: "Databricks FQN GPT-5+ Responses routing", note: Some("Only the service component selects Responses; effort capabilities remain neutral.") }, Q::Vector { id: "dbv2-fqn-responses-0", provider: "databricks_v2", raw_model_id: "catalog.schema.goose-gpt-6-astra", note: None }, Q::Vector { id: "dbv2-fqn-responses-1", provider: "databricks_v2", raw_model_id: "catalog.schema.goose-gpt-5", note: None }, @@ -763,7 +798,7 @@ mod tests { Q::Vector { id: "dbv2-fqn-responses-4", provider: "databricks_v2", raw_model_id: "catalog.schema.GOOSE-GPT-6-ASTRA", note: None }, Q::Vector { id: "dbv2-fqn-responses-5", provider: "databricks_v2", raw_model_id: "gpt-6.schema.other", note: None }, Q::Vector { id: "dbv2-fqn-responses-6", provider: "databricks_v2", raw_model_id: "catalog.gpt-5.other", note: None }, - Q::Vector { id: "dbv2-fqn-responses-7", provider: "databricks_v2", raw_model_id: "catalog.schema.claude-gpt-6", note: None }, + Q::Vector { id: "dbv2-fqn-claude-precedes-gpt", provider: "databricks_v2", raw_model_id: "catalog.schema.claude-gpt-6", note: None }, Q::Vector { id: "dbv2-fqn-responses-8", provider: "databricks_v2", raw_model_id: "catalog.schema.my-gpt-6-astra", note: None }, Q::Vector { id: "dbv2-fqn-responses-9", provider: "databricks_v2", raw_model_id: "catalog.schema.mygpt-6-astra", note: None }, Q::Vector { id: "dbv2-fqn-responses-10", provider: "databricks_v2", raw_model_id: "catalog.schema.gpt-4", note: None }, @@ -894,7 +929,7 @@ mod tests { } #[test] - fn corpus_has_exactly_164_executable_vectors() { + fn corpus_has_exactly_168_executable_vectors() { // Locks the vector count so a silent INPUTS edit can't quietly drop // coverage; must equal the gate in the TS harness // (modelCapabilitiesCorpus.test.mjs). @@ -903,7 +938,7 @@ mod tests { .filter(|q| matches!(q, Q::Vector { .. })) .count(); assert_eq!( - vectors, 164, + vectors, 168, "corpus executable-vector count changed; update this gate deliberately" ); } @@ -974,7 +1009,9 @@ mod tests { #[test] fn test_every_resolve_yields_a_complete_result() { - // Complete-result invariant: supported_efforts is never empty on any path. + // Complete-result invariant: ordinary resolution paths advertise at + // least one effort; uncurated Claude FQNs intentionally opt out and are + // covered by dedicated fallback tests. let inputs = [ ("anthropic", "claude-opus-4-7"), ("anthropic", ""), diff --git a/crates/buzz-agent/tests/databricks_oauth.rs b/crates/buzz-agent/tests/databricks_oauth.rs index ac2b9578626..c979e0680bf 100644 --- a/crates/buzz-agent/tests/databricks_oauth.rs +++ b/crates/buzz-agent/tests/databricks_oauth.rs @@ -774,25 +774,21 @@ async fn databricks_v2_other_models_route_through_ai_gateway_mlflow_chat() { } #[tokio::test] -async fn databricks_v2_model_service_fqn_uses_mlflow_chat_and_preserves_full_id() { +async fn databricks_v2_claude_model_service_fqn_uses_messages_and_preserves_full_id() { let canned = vec![json!({ - "id": "x", - "object": "chat.completion", - "choices": [{ - "index": 0, - "message": { "role": "assistant", "content": "ok" }, - "finish_reason": "stop" - }] + "stop_reason": "end_turn", + "content": [{ "type": "text", "text": "ok" }] })]; // Family-looking text in a Unity Catalog namespace is data, not route - // authority. The full raw FQN must reach the MLflow model field. + // authority. Only the service component selects Anthropic Messages, and + // the full raw FQN must reach the model field. let model = "catalog.schema.claude-gpt-5"; let req = run_captured_prompt("databricks_v2", model, canned).await; assert_eq!( req.path.as_str(), - "/ai-gateway/mlflow/v1/chat/completions", - "Unity Catalog model-service FQNs must always use MLflow Chat" + "/ai-gateway/anthropic/v1/messages", + "a Claude model-service FQN must use Anthropic Messages" ); assert_eq!(req.body["model"], model); assert!( @@ -800,7 +796,7 @@ async fn databricks_v2_model_service_fqn_uses_mlflow_chat_and_preserves_full_id( .get("messages") .and_then(|value| value.as_array()) .is_some(), - "model-service FQN requests must use the Chat Completions envelope" + "Claude model-service FQN requests must use the Messages envelope" ); } diff --git a/crates/buzz-agent/tests/fqn_capabilities.rs b/crates/buzz-agent/tests/fqn_capabilities.rs index 1c513dc0118..c532d22f2fc 100644 --- a/crates/buzz-agent/tests/fqn_capabilities.rs +++ b/crates/buzz-agent/tests/fqn_capabilities.rs @@ -1,4 +1,5 @@ -//! UC route selection must change only the route, never effort or model identity. +//! UC fallback routing preserves model identity; GPT routes retain neutral effort, +//! while uncurated Claude routes intentionally advertise no effort controls. use buzz_agent::model_capabilities::{resolve, DatabricksV2Route}; #[test] @@ -30,11 +31,10 @@ fn gpt_fqn_route_preserves_neutral_capabilities() { } #[test] -fn unrelated_fqns_do_not_select_responses() { +fn unrelated_fqns_do_not_select_family_routes() { for model in [ "gpt-6.schema.other", "catalog.gpt-5.other", - "catalog.schema.claude-gpt-6", "catalog.schema.mygpt-6-astra", "catalog.schema.gpt-4", "catalog.schema.gpt-4o", diff --git a/desktop/src/features/agents/AGENTS.md b/desktop/src/features/agents/AGENTS.md index 1081bce81b4..9d3da397307 100644 --- a/desktop/src/features/agents/AGENTS.md +++ b/desktop/src/features/agents/AGENTS.md @@ -306,7 +306,7 @@ with a TypeScript lookup table or an id comparison in a component. refresh only local persona/team/managed-agent caches; they must never invalidate the remote relay directory. -17. **Databricks model discovery has one shared catalog authority.** Desktop and ACP call the shared `buzz-agent` discovery library; Desktop passes the effective merged `DATABRICKS_MODEL_FILTER` explicitly, and the library applies it to raw workspace endpoint IDs and Unity Catalog model-service FQNs after the additive union. A successful filtered-empty catalog is authoritative: it stays empty, disables switching, and never falls through to configured or known-model fallback. Verified provider-qualified exact records in `scripts/model-capabilities.json` take precedence for UC FQNs: `data_workflow_tools.goose.goose-claude-opus-5-5` uses Anthropic Messages, adaptive thinking, and `output_config.effort` (default medium). This does not confer capabilities on other namespaces or similarly named services. Uncurated UC FQNs retain neutral effort capabilities. A boundary-matched GPT-5-or-newer family in their service-name component selects OpenAI Responses so tools can coexist with reasoning; other uncurated FQNs use MLflow Chat Completions. Catalog/schema components never infer routing. Keep exact-record precedence and fallback rules identical in the Rust and TypeScript capability interpreters and shared corpus. Effort inheritance, persistence, and clearing semantics are unchanged. Global Defaults preserves the discovered model ID as the selected value while its closed trigger renders the provider-scoped display label; do not force the raw persisted ID over that label. +17. **Databricks model discovery has one shared catalog authority.** Desktop and ACP call the shared `buzz-agent` discovery library; Desktop passes the effective merged `DATABRICKS_MODEL_FILTER` explicitly, and the library applies it to raw workspace endpoint IDs and Unity Catalog model-service FQNs after the additive union. A successful filtered-empty catalog is authoritative: it stays empty, disables switching, and never falls through to configured or known-model fallback. Verified provider-qualified exact records in `scripts/model-capabilities.json` take precedence for UC FQNs: `data_workflow_tools.goose.goose-claude-opus-5-5` uses Anthropic Messages, adaptive thinking, and `output_config.effort` (default medium). This does not confer capabilities on other namespaces or similarly named services. Uncurated UC FQNs do not inherit family effort capabilities: Claude service components select Anthropic Messages and expose no effort choices, while GPT-5-or-newer service components select OpenAI Responses with neutral fallback effort capabilities; other uncurated FQNs use MLflow Chat Completions. Catalog/schema components never infer routing. Keep exact-record precedence and fallback rules identical in the Rust and TypeScript capability interpreters and shared corpus. Effort inheritance, persistence, and clearing semantics are unchanged. Global Defaults preserves the discovered model ID as the selected value while its closed trigger renders the provider-scoped display label; do not force the raw persisted ID over that label. ## Channel-only runtime controls diff --git a/desktop/src/features/agents/ui/buzzAgentConfig.test.mjs b/desktop/src/features/agents/ui/buzzAgentConfig.test.mjs index f0dd162f5be..b746e15dc1f 100644 --- a/desktop/src/features/agents/ui/buzzAgentConfig.test.mjs +++ b/desktop/src/features/agents/ui/buzzAgentConfig.test.mjs @@ -428,6 +428,15 @@ test("databricks_v2 with databricks-prefixed claude model strips prefix and rout assert.equal(defaultValue, "high"); }); +test("databricks_v2 Claude FQN exposes no effort controls", () => { + const { validValues, defaultValue } = getProviderEffortConfig( + "databricks_v2", + "catalog.schema.claude-sonnet-custom", + ); + assert.deepEqual([...validValues], []); + assert.equal(defaultValue, null); +}); + test("databricks_v2 with gpt-5.4 routes to openai gpt-5.5/5.4 table", () => { const { validValues, defaultValue } = getProviderEffortConfig( "databricks_v2", diff --git a/desktop/src/features/agents/ui/effortAutoClear.test.mjs b/desktop/src/features/agents/ui/effortAutoClear.test.mjs index f8a787a1027..35dd6eebb8c 100644 --- a/desktop/src/features/agents/ui/effortAutoClear.test.mjs +++ b/desktop/src/features/agents/ui/effortAutoClear.test.mjs @@ -114,6 +114,8 @@ dom.window.__TAURI_INTERNALS__ = globalThis.__TAURI_INTERNALS__; let act, render, screen, cleanup; let AgentConfigFields; +let EffortSelectField; +let getProviderEffortConfig; let fromRawAcpRuntimeCatalogEntry; let createElement, useState, useCallback; let setGlobalAgentConfig; @@ -121,6 +123,8 @@ let setGlobalAgentConfig; before(async () => { ({ act, render, screen, cleanup } = await import("@testing-library/react")); ({ AgentConfigFields } = await import("./AgentConfigFields.tsx")); + ({ EffortSelectField } = await import("./buzzAgentModelTuningFields.tsx")); + ({ getProviderEffortConfig } = await import("./buzzAgentConfig.ts")); ({ fromRawAcpRuntimeCatalogEntry } = await import( "../../../shared/api/tauri.ts" )); @@ -214,6 +218,33 @@ function SettingsParent({ runtime, initialConfig, saveRef }) { // ── Tests ───────────────────────────────────────────────────────────────────── +test("Claude FQN picker presents provider default with no selectable effort", async () => { + const { validValues, defaultValue } = getProviderEffortConfig( + "databricks_v2", + "catalog.schema.claude-sonnet-custom", + ); + render( + createElement(EffortSelectField, { + currentEffort: "", + effortDefault: defaultValue, + effortValid: validValues, + htmlFor: "claude-fqn-effort", + label: "Effort", + onChange: () => {}, + showUnavailableOptions: false, + testId: "claude-fqn-effort", + }), + ); + await act(async () => {}); + + const select = screen.getByTestId("claude-fqn-effort"); + assert.deepEqual( + [...select.options].map((option) => option.textContent), + ["Inherit (default)"], + ); + assert.doesNotMatch(select.textContent ?? "", /none/i); +}); + test("AgentConfigFields (useCustomSelect): Goose off renders 'Off' in custom trigger (mount)", async () => { // Regression: without the isHarnessNativeEffort branch, effortValidForRenderer // uses buzz-agent vocab (no "off"); the custom trigger shows the inherit diff --git a/desktop/src/features/agents/ui/modelCapabilities.ts b/desktop/src/features/agents/ui/modelCapabilities.ts index 045cc9de132..42cd450aa72 100644 --- a/desktop/src/features/agents/ui/modelCapabilities.ts +++ b/desktop/src/features/agents/ui/modelCapabilities.ts @@ -302,7 +302,17 @@ function isDatabricksModelServiceFqn(model: string): boolean { ); } -// Mirror fqn_requires_responses: routing is the only inferred FQN capability. +// Mirror fqn_requires_anthropic_messages: only the service component may +// select Anthropic Messages, and doing so does not infer effort support. +function fqnRequiresAnthropicMessages(model: string): boolean { + const service = model.split(".").at(-1) ?? ""; + const stripped = stripCatalogPrefix( + service.toLowerCase(), + MANIFEST.family_tokens, + ); + return stripped.startsWith("claude-"); +} + function fqnRequiresResponses(model: string): boolean { const service = model.split(".").at(-1) ?? ""; const stripped = stripCatalogPrefix( @@ -328,8 +338,9 @@ export function resolveModelCapabilities( ): CapabilityResult { const canon = canonicalizeProvider(provider); const blank = rawModelId.trim().length === 0; - // Uncurated FQNs keep neutral effort capabilities; verified exact records - // take precedence. Catalog/schema names never infer a protocol. + // Exact records are verified service contracts and take precedence. Among + // uncurated FQNs, Claude exposes no effort choices; other services retain + // neutral fallback effort. Routing inspects only the service component. const modelServiceFqn = canon === "databricks_v2" && isDatabricksModelServiceFqn(rawModelId); @@ -383,10 +394,27 @@ export function resolveModelCapabilities( // 3. Provider fallback (blank vs. concrete-unknown); never carries a label. const pair = fallbackPair(canon); const state = blank ? pair.blank : pair.concrete_unknown; - const route = - modelServiceFqn && fqnRequiresResponses(rawModelId) + const fqnAnthropicMessages = + modelServiceFqn && fqnRequiresAnthropicMessages(rawModelId); + const route = fqnAnthropicMessages + ? "anthropic-messages" + : modelServiceFqn && fqnRequiresResponses(rawModelId) ? "openai-responses" : state.databricks_v2_wire_route; + if (fqnAnthropicMessages) { + // Route inference does not prove thinking support. Verified exact records + // returned above; uncurated Claude services advertise no effort controls. + return toResult( + { + ...state, + supported_efforts: [], + default_effort: null, + normalization_policy: "none", + }, + route, + null, + ); + } return toResult(state, route, null); } diff --git a/desktop/src/features/agents/ui/modelCapabilitiesCorpus.test.mjs b/desktop/src/features/agents/ui/modelCapabilitiesCorpus.test.mjs index a27e0d675c5..c869d8bcbf9 100644 --- a/desktop/src/features/agents/ui/modelCapabilitiesCorpus.test.mjs +++ b/desktop/src/features/agents/ui/modelCapabilitiesCorpus.test.mjs @@ -25,10 +25,10 @@ const corpus = JSON.parse(readFileSync(fileURLToPath(corpusUrl), "utf8")); // (`_group`) are skipped. Mirrors the Rust corpus filter. const executable = corpus.filter((entry) => entry.expect != null); -test("corpus has exactly 164 executable vectors", () => { +test("corpus has exactly 168 executable vectors", () => { // Locks the vector count so a silent corpus edit can't quietly drop coverage; // must equal the gate in the Rust suite (model_capabilities.rs). - assert.equal(executable.length, 164); + assert.equal(executable.length, 168); }); test("registry label aliases refuse an unprefixed query", () => { diff --git a/scripts/normative-corpus.json b/scripts/normative-corpus.json index fe0b99d8468..33afdf984a3 100644 --- a/scripts/normative-corpus.json +++ b/scripts/normative-corpus.json @@ -2394,17 +2394,10 @@ "raw_model_id": "other.goose.goose-claude-opus-5-5", "expect": { "thinking_mode": "none", - "supported_efforts": [ - "none", - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "default_effort": "medium", - "databricks_v2_wire_route": "mlflow-chat", - "normalization_policy": "openai-clamp-max-to-xhigh", + "supported_efforts": [], + "default_effort": null, + "databricks_v2_wire_route": "anthropic-messages", + "normalization_policy": "none", "registry_label": null } }, @@ -2412,6 +2405,32 @@ "id": "dbv2-opus-uc-service", "provider": "databricks_v2", "raw_model_id": "data_workflow_tools.goose.goose-claude-opus-5-5-preview", + "expect": { + "thinking_mode": "none", + "supported_efforts": [], + "default_effort": null, + "databricks_v2_wire_route": "anthropic-messages", + "normalization_policy": "none", + "registry_label": null + } + }, + { + "id": "dbv2-opus-uc-system", + "provider": "databricks_v2", + "raw_model_id": "system.ai.claude-opus-5-5", + "expect": { + "thinking_mode": "none", + "supported_efforts": [], + "default_effort": null, + "databricks_v2_wire_route": "anthropic-messages", + "normalization_policy": "none", + "registry_label": null + } + }, + { + "id": "dbv2-opus-uc-provider", + "provider": "openai", + "raw_model_id": "data_workflow_tools.goose.goose-claude-opus-5-5", "expect": { "thinking_mode": "none", "supported_efforts": [ @@ -2423,15 +2442,45 @@ "xhigh" ], "default_effort": "medium", - "databricks_v2_wire_route": "mlflow-chat", + "databricks_v2_wire_route": "not-applicable", "normalization_policy": "openai-clamp-max-to-xhigh", "registry_label": null } }, { - "id": "dbv2-opus-uc-system", + "_group": "Uncurated Databricks Claude FQN routing", + "_note": "Only the service component selects Anthropic Messages; unverified Claude FQNs expose no effort choices." + }, + { + "id": "dbv2-fqn-claude-anthropic", "provider": "databricks_v2", - "raw_model_id": "system.ai.claude-opus-5-5", + "raw_model_id": "catalog.schema.claude-sonnet-custom", + "expect": { + "thinking_mode": "none", + "supported_efforts": [], + "default_effort": null, + "databricks_v2_wire_route": "anthropic-messages", + "normalization_policy": "none", + "registry_label": null + } + }, + { + "id": "dbv2-fqn-claude-case", + "provider": "databricks_v2", + "raw_model_id": "catalog.schema.GOOSE-CLAUDE-SONNET-CUSTOM", + "expect": { + "thinking_mode": "none", + "supported_efforts": [], + "default_effort": null, + "databricks_v2_wire_route": "anthropic-messages", + "normalization_policy": "none", + "registry_label": null + } + }, + { + "id": "dbv2-fqn-claude-catalog-inert", + "provider": "databricks_v2", + "raw_model_id": "claude-catalog.schema.service", "expect": { "thinking_mode": "none", "supported_efforts": [ @@ -2449,9 +2498,9 @@ } }, { - "id": "dbv2-opus-uc-provider", - "provider": "openai", - "raw_model_id": "data_workflow_tools.goose.goose-claude-opus-5-5", + "id": "dbv2-fqn-claude-schema-inert", + "provider": "databricks_v2", + "raw_model_id": "catalog.claude-schema.service", "expect": { "thinking_mode": "none", "supported_efforts": [ @@ -2463,7 +2512,7 @@ "xhigh" ], "default_effort": "medium", - "databricks_v2_wire_route": "not-applicable", + "databricks_v2_wire_route": "mlflow-chat", "normalization_policy": "openai-clamp-max-to-xhigh", "registry_label": null } @@ -2613,22 +2662,15 @@ } }, { - "id": "dbv2-fqn-responses-7", + "id": "dbv2-fqn-claude-precedes-gpt", "provider": "databricks_v2", "raw_model_id": "catalog.schema.claude-gpt-6", "expect": { "thinking_mode": "none", - "supported_efforts": [ - "none", - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "default_effort": "medium", - "databricks_v2_wire_route": "mlflow-chat", - "normalization_policy": "openai-clamp-max-to-xhigh", + "supported_efforts": [], + "default_effort": null, + "databricks_v2_wire_route": "anthropic-messages", + "normalization_policy": "none", "registry_label": null } },