Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion crates/buzz-agent/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -242,7 +242,7 @@ lifecycle hook — see [MCP_DRIVEN_HOOKS.md](../../docs/MCP_DRIVEN_HOOKS.md).
| Block Gateway | `openai` | `POST {base}/chat/completions` | gpt-5, claude |
| OpenRouter | `openrouter` | `POST {base}/chat/completions` | anything they route (extended-thinking replay, provider-agnostic tool calling) |
| Databricks | `databricks` | `POST {host}/serving-endpoints/{model}/invocations` | goose-claude-4-6-sonnet |
| Databricks AI Gateway v2 | `databricks_v2` | `POST {host}/ai-gateway/{provider}/v1/...` | workspace endpoints and Unity Catalog model-service FQNs; UC GPT-5+ services use OpenAI Responses; other UC FQNs use MLflow Chat Completions |
| Databricks AI Gateway v2 | `databricks_v2` | `POST {host}/ai-gateway/{provider}/v1/...` | workspace endpoints and Unity Catalog model-service FQNs; UC Claude services use Anthropic Messages, UC GPT-5+ services use OpenAI Responses, and unknown UC FQNs use MLflow Chat Completions |

The optional `DATABRICKS_MODEL_FILTER` applies only to model discovery. Each comma-separated entry is trimmed and matched against the complete raw ID with case-sensitive `*` (zero or more characters) and `?` (one Unicode character) semantics; patterns are OR-ed. Unset or blank preserves the full authenticated catalog. A nonblank value containing no usable patterns is rejected. This controls picker visibility only; Databricks and Unity Catalog permissions remain the authorization boundary. A filtered-empty result is authoritative and does not restore the built-in fallback models.

Expand Down
32 changes: 32 additions & 0 deletions crates/buzz-agent/src/config.rs
Original file line number Diff line number Diff line change
Expand Up @@ -302,6 +302,15 @@ pub fn anthropic_thinking_config(
}
ThinkingMode::None | ThinkingMode::OmitFields => {
// Non-thinking model, or unknown/unverified Anthropic name: omit rather than guess.
// A persisted explicit setting can outlive a model switch, so make the
// discarded choice observable even though fresh clients advertise no explicit effort.
if effort != ThinkingEffort::None {
tracing::warn!(
model = effective_model,
requested = effort.openai_effort_str(),
"BUZZ_AGENT_THINKING_EFFORT is unsupported for this unverified Anthropic model; omitting thinking fields"
);
}
(None, None)
}
}
Expand Down Expand Up @@ -2187,6 +2196,29 @@ mod tests {
);
}

#[test]
fn normalize_effort_for_uncurated_claude_fqn_is_a_noop() {
let model = "catalog.schema.claude-sonnet-custom";
assert!(crate::model_capabilities::resolve("databricks_v2", model)
.supported_efforts
.is_empty());
for effort in [
ThinkingEffort::None,
ThinkingEffort::Minimal,
ThinkingEffort::Low,
ThinkingEffort::Medium,
ThinkingEffort::High,
ThinkingEffort::XHigh,
ThinkingEffort::Max,
] {
assert_eq!(
normalize_effort_for_databricks_v2(effort, model),
effort,
"Anthropic-routed FQNs must never enter OpenAI effort resolution"
);
}
}

// ---- normalize_effort_for_anthropic_route ----

#[test]
Expand Down
34 changes: 23 additions & 11 deletions crates/buzz-agent/src/llm.rs
Original file line number Diff line number Diff line change
Expand Up @@ -2856,10 +2856,13 @@ mod tests {
}

#[tokio::test]
async fn databricks_v2_model_service_fqn_summary_uses_mlflow_chat() {
async fn databricks_v2_claude_model_service_fqn_summary_uses_anthropic_messages() {
let model = "catalog.schema.claude-gpt-5";
let (base_url, captured) =
spawn_sequence_stub(vec![StubHttpResponse::ok(chat_response("summary"))]).await;
let response = json!({
"content": [{"type": "text", "text": "summary"}],
"stop_reason": "end_turn"
});
let (base_url, captured) = spawn_sequence_stub(vec![StubHttpResponse::ok(response)]).await;
let mut config = cfg(Provider::DatabricksV2);
config.base_url = base_url;
let llm = Llm::new(&config).unwrap();
Expand All @@ -2875,11 +2878,11 @@ mod tests {
.iter()
.find(|request| request.method == "POST")
.expect("summary must issue one POST");
assert_eq!(request.path, "/v1/ai-gateway/mlflow/v1/chat/completions");
assert_eq!(request.path, "/v1/ai-gateway/anthropic/v1/messages");
let body = request.body.as_ref().expect("summary body");
assert_eq!(body["model"], model);
assert!(body["messages"].is_array());
assert_eq!(body["max_completion_tokens"], 128);
assert_eq!(body["max_tokens"], 128);
}

fn image_history() -> Vec<HistoryItem> {
Expand Down Expand Up @@ -3295,19 +3298,28 @@ mod tests {
fn databricks_v2_model_service_fqn_shape_is_strict_and_precedes_manifest() {
use crate::model_capabilities::{resolve, DatabricksV2Route as Manifest};

for model in [
"catalog.schema.service",
"catalog.schema.claude-gpt-5",
"data_tools.goose.kimi-k3",
for (model, expected) in [
(
"catalog.schema.service",
DatabricksV2Route::MlflowChatCompletions,
),
(
"catalog.schema.claude-gpt-5",
DatabricksV2Route::AnthropicMessages,
),
(
"data_tools.goose.kimi-k3",
DatabricksV2Route::MlflowChatCompletions,
),
] {
assert!(
crate::model_capabilities::is_databricks_model_service_fqn(model),
"expected FQN shape: {model}"
);
assert_eq!(
databricks_v2_route(model),
DatabricksV2Route::MlflowChatCompletions,
"FQN route must precede manifest family inference: {model}"
expected,
"only a Claude service component may select Anthropic Messages: {model}"
);
}

Expand Down
59 changes: 55 additions & 4 deletions crates/buzz-agent/src/llm_fqn_tests.rs
Original file line number Diff line number Diff line change
Expand Up @@ -127,23 +127,24 @@ async fn opus_uc_replays_ordered_signed_content_with_high_effort() {
}

#[tokio::test]
async fn neighboring_opus_uc_services_stay_on_mlflow_chat() {
async fn neighboring_opus_uc_services_use_anthropic_messages_without_effort() {
for model in ["system.ai.claude-opus-5-5", "other.goose.goose-claude-opus-5-5",
"data_workflow_tools.goose.goose-claude-opus-5-5-preview"] {
let (base_url, captured) = spawn_sequence_stub(vec![StubHttpResponse::ok(json!({
"choices":[{"finish_reason":"stop", "message":{"content":"ok"}}]
"stop_reason":"end_turn", "content":[{"type":"text", "text":"ok"}]
}))]).await;
let mut config = cfg(Provider::DatabricksV2);
config.base_url = base_url;
config.thinking_effort = Some(ThinkingEffort::High);
Llm::new(&config).unwrap().complete(&config, "system", &[HistoryItem::User("hi".into())], &[], model).await.unwrap();
let requests = captured.lock().await;
let post = requests.iter().find(|r| r.method == "POST").unwrap();
assert_eq!(post.path, "/v1/ai-gateway/mlflow/v1/chat/completions");
assert_eq!(post.path, "/v1/ai-gateway/anthropic/v1/messages");
let body = post.body.as_ref().unwrap();
assert_eq!(body["model"], model);
assert_eq!(body["reasoning_effort"], "high");
assert!(body.get("reasoning_effort").is_none());
assert!(body.get("thinking").is_none());
assert!(body.get("output_config").is_none());
}
}

Expand All @@ -163,3 +164,53 @@ fn native_thinking_tail_is_not_cache_stamped_or_leaked_to_chat() {
assert!(!responses_body(&config, "system", &history, &[], "other", None).to_string().contains("anthropic_content"));
}
}

#[tokio::test]
async fn uncurated_claude_fqn_completion_and_summary_use_anthropic_messages_without_effort() {
let response = json!({
"content": [{"type": "text", "text": "ok"}],
"stop_reason": "end_turn"
});
let (base_url, captured) = spawn_sequence_stub(vec![
StubHttpResponse::ok(response.clone()),
StubHttpResponse::ok(response),
])
.await;
let mut config = cfg(Provider::DatabricksV2);
config.base_url = base_url;
config.thinking_effort = Some(ThinkingEffort::High);
let model = "catalog.schema.claude-sonnet-custom";
let llm = Llm::new(&config).unwrap();

assert_eq!(
llm.complete(
&config,
"system",
&[HistoryItem::User("hello".into())],
&[],
model,
)
.await
.unwrap()
.text,
"ok"
);
assert_eq!(
llm.summarize(&config, "system", "conversation", 128, model)
.await
.unwrap(),
"ok"
);

let posts = captured.lock().await;
assert_eq!(posts.len(), 2);
for request in &*posts {
assert_eq!(request.path, "/v1/ai-gateway/anthropic/v1/messages");
let body = request.body.as_ref().unwrap();
assert_eq!(body["model"], model);
assert!(body.get("reasoning_effort").is_none());
assert!(body.get("thinking").is_none());
assert!(body.get("output_config").is_none());
}
assert_eq!(posts[1].body.as_ref().unwrap()["max_tokens"], 128);
}
63 changes: 50 additions & 13 deletions crates/buzz-agent/src/model_capabilities.rs
Original file line number Diff line number Diff line change
Expand Up @@ -293,8 +293,10 @@ fn prefix_matches(token: &str, s: &str) -> bool {
/// Databricks Unity Catalog model-service names are catalog data, not model
/// family hints. Both capability interpreters use this shape check before
/// family matching so services cannot inherit endpoint capabilities accidentally.
/// Verified exact records may supply capabilities; other GPT-5+ services have
/// a route-only Responses exception.
/// Verified exact records may supply capabilities. Among uncurated services,
/// Claude routes through Anthropic Messages with no advertised effort choices,
/// GPT-5+ routes through OpenAI Responses with neutral fallback effort, and all
/// others retain the provider fallback route.
pub(crate) fn is_databricks_model_service_fqn(model: &str) -> bool {
let mut components = model.split('.');
let (Some(catalog), Some(schema), Some(service)) =
Expand All @@ -309,6 +311,16 @@ pub(crate) fn is_databricks_model_service_fqn(model: &str) -> bool {
}) && components.next().is_none()
}

/// Route uncurated Claude UC services through Anthropic Messages without
/// borrowing effort capabilities from the service name.
fn fqn_requires_anthropic_messages(model: &str) -> bool {
let Some(service) = model.rsplit('.').next() else {
return false;
};
let lower = service.to_ascii_lowercase();
strip_catalog_prefix(&lower, &manifest().family_tokens).starts_with("claude-")
}

/// Route GPT-5+ UC services to Responses without borrowing endpoint effort facts.
/// Match the first family token in the service only, preserving the existing
/// boundary semantics (e.g. `claude-gpt-5` is not a GPT service).
Expand All @@ -335,9 +347,10 @@ pub fn resolve(provider: &str, raw_model_id: &str) -> CapabilityResult {
let canon = canonical_provider(provider);
let blank = raw_model_id.trim().is_empty();

// Uncurated FQNs keep neutral effort capabilities. Exact records are verified
// service contracts, not family-name inference. The GPT-5+ route-only fallback
// inspects just the service component, never catalog/schema names.
// Exact records are verified service contracts, not family-name inference.
// Among uncurated FQNs, Claude exposes no effort choices; other services
// retain neutral fallback effort. Route fallbacks inspect only the service
// component, never catalog/schema names.
let model_service_fqn =
canon == "databricks_v2" && is_databricks_model_service_fqn(raw_model_id);

Expand Down Expand Up @@ -410,16 +423,33 @@ pub fn resolve(provider: &str, raw_model_id: &str) -> CapabilityResult {
} else {
&pair.concrete_unknown
};
let fqn_anthropic_messages = model_service_fqn && fqn_requires_anthropic_messages(raw_model_id);
CapabilityResult {
// Route inference does not prove thinking support. Expose no choices for
// an unverified Claude FQN; verified exact records returned above.
thinking_mode: state.thinking_mode,
supported_efforts: &state.supported_efforts,
default_effort: state.default_effort,
databricks_v2_wire_route: if model_service_fqn && fqn_requires_responses(raw_model_id) {
supported_efforts: if fqn_anthropic_messages {
Comment thread
kalvinnchau marked this conversation as resolved.
&[]
} else {
&state.supported_efforts
},
default_effort: if fqn_anthropic_messages {
None
} else {
state.default_effort
},
databricks_v2_wire_route: if fqn_anthropic_messages {
Comment thread
kalvinnchau marked this conversation as resolved.
DatabricksV2Route::AnthropicMessages
} else if model_service_fqn && fqn_requires_responses(raw_model_id) {
DatabricksV2Route::OpenaiResponses
} else {
state.databricks_v2_wire_route
},
normalization_policy: state.normalization_policy,
normalization_policy: if fqn_anthropic_messages {
NormalizationPolicy::None
} else {
state.normalization_policy
},
registry_label: None,
}
}
Expand Down Expand Up @@ -755,6 +785,11 @@ mod tests {
Q::Vector { id: "dbv2-opus-uc-service", provider: "databricks_v2", raw_model_id: "data_workflow_tools.goose.goose-claude-opus-5-5-preview", note: None },
Q::Vector { id: "dbv2-opus-uc-system", provider: "databricks_v2", raw_model_id: "system.ai.claude-opus-5-5", note: None },
Q::Vector { id: "dbv2-opus-uc-provider", provider: "openai", raw_model_id: "data_workflow_tools.goose.goose-claude-opus-5-5", note: None },
Q::Section { group: "Uncurated Databricks Claude FQN routing", note: Some("Only the service component selects Anthropic Messages; unverified Claude FQNs expose no effort choices.") },
Q::Vector { id: "dbv2-fqn-claude-anthropic", provider: "databricks_v2", raw_model_id: "catalog.schema.claude-sonnet-custom", note: None },
Q::Vector { id: "dbv2-fqn-claude-case", provider: "databricks_v2", raw_model_id: "catalog.schema.GOOSE-CLAUDE-SONNET-CUSTOM", note: None },
Q::Vector { id: "dbv2-fqn-claude-catalog-inert", provider: "databricks_v2", raw_model_id: "claude-catalog.schema.service", note: None },
Q::Vector { id: "dbv2-fqn-claude-schema-inert", provider: "databricks_v2", raw_model_id: "catalog.claude-schema.service", note: None },
Q::Section { group: "Databricks FQN GPT-5+ Responses routing", note: Some("Only the service component selects Responses; effort capabilities remain neutral.") },
Q::Vector { id: "dbv2-fqn-responses-0", provider: "databricks_v2", raw_model_id: "catalog.schema.goose-gpt-6-astra", note: None },
Q::Vector { id: "dbv2-fqn-responses-1", provider: "databricks_v2", raw_model_id: "catalog.schema.goose-gpt-5", note: None },
Expand All @@ -763,7 +798,7 @@ mod tests {
Q::Vector { id: "dbv2-fqn-responses-4", provider: "databricks_v2", raw_model_id: "catalog.schema.GOOSE-GPT-6-ASTRA", note: None },
Q::Vector { id: "dbv2-fqn-responses-5", provider: "databricks_v2", raw_model_id: "gpt-6.schema.other", note: None },
Q::Vector { id: "dbv2-fqn-responses-6", provider: "databricks_v2", raw_model_id: "catalog.gpt-5.other", note: None },
Q::Vector { id: "dbv2-fqn-responses-7", provider: "databricks_v2", raw_model_id: "catalog.schema.claude-gpt-6", note: None },
Q::Vector { id: "dbv2-fqn-claude-precedes-gpt", provider: "databricks_v2", raw_model_id: "catalog.schema.claude-gpt-6", note: None },
Q::Vector { id: "dbv2-fqn-responses-8", provider: "databricks_v2", raw_model_id: "catalog.schema.my-gpt-6-astra", note: None },
Q::Vector { id: "dbv2-fqn-responses-9", provider: "databricks_v2", raw_model_id: "catalog.schema.mygpt-6-astra", note: None },
Q::Vector { id: "dbv2-fqn-responses-10", provider: "databricks_v2", raw_model_id: "catalog.schema.gpt-4", note: None },
Expand Down Expand Up @@ -894,7 +929,7 @@ mod tests {
}

#[test]
fn corpus_has_exactly_164_executable_vectors() {
fn corpus_has_exactly_168_executable_vectors() {
// Locks the vector count so a silent INPUTS edit can't quietly drop
// coverage; must equal the gate in the TS harness
// (modelCapabilitiesCorpus.test.mjs).
Expand All @@ -903,7 +938,7 @@ mod tests {
.filter(|q| matches!(q, Q::Vector { .. }))
.count();
assert_eq!(
vectors, 164,
vectors, 168,
"corpus executable-vector count changed; update this gate deliberately"
);
}
Expand Down Expand Up @@ -974,7 +1009,9 @@ mod tests {

#[test]
fn test_every_resolve_yields_a_complete_result() {
// Complete-result invariant: supported_efforts is never empty on any path.
// Complete-result invariant: ordinary resolution paths advertise at
// least one effort; uncurated Claude FQNs intentionally opt out and are
// covered by dedicated fallback tests.
let inputs = [
("anthropic", "claude-opus-4-7"),
("anthropic", ""),
Expand Down
20 changes: 8 additions & 12 deletions crates/buzz-agent/tests/databricks_oauth.rs
Original file line number Diff line number Diff line change
Expand Up @@ -774,33 +774,29 @@ async fn databricks_v2_other_models_route_through_ai_gateway_mlflow_chat() {
}

#[tokio::test]
async fn databricks_v2_model_service_fqn_uses_mlflow_chat_and_preserves_full_id() {
async fn databricks_v2_claude_model_service_fqn_uses_messages_and_preserves_full_id() {
let canned = vec![json!({
"id": "x",
"object": "chat.completion",
"choices": [{
"index": 0,
"message": { "role": "assistant", "content": "ok" },
"finish_reason": "stop"
}]
"stop_reason": "end_turn",
"content": [{ "type": "text", "text": "ok" }]
})];
// Family-looking text in a Unity Catalog namespace is data, not route
// authority. The full raw FQN must reach the MLflow model field.
// authority. Only the service component selects Anthropic Messages, and
// the full raw FQN must reach the model field.
let model = "catalog.schema.claude-gpt-5";
let req = run_captured_prompt("databricks_v2", model, canned).await;

assert_eq!(
req.path.as_str(),
"/ai-gateway/mlflow/v1/chat/completions",
"Unity Catalog model-service FQNs must always use MLflow Chat"
"/ai-gateway/anthropic/v1/messages",
"a Claude model-service FQN must use Anthropic Messages"
);
assert_eq!(req.body["model"], model);
assert!(
req.body
.get("messages")
.and_then(|value| value.as_array())
.is_some(),
"model-service FQN requests must use the Chat Completions envelope"
"Claude model-service FQN requests must use the Messages envelope"
);
}

Expand Down
6 changes: 3 additions & 3 deletions crates/buzz-agent/tests/fqn_capabilities.rs
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
//! UC route selection must change only the route, never effort or model identity.
//! UC fallback routing preserves model identity; GPT routes retain neutral effort,
//! while uncurated Claude routes intentionally advertise no effort controls.
use buzz_agent::model_capabilities::{resolve, DatabricksV2Route};

#[test]
Expand Down Expand Up @@ -30,11 +31,10 @@ fn gpt_fqn_route_preserves_neutral_capabilities() {
}

#[test]
fn unrelated_fqns_do_not_select_responses() {
fn unrelated_fqns_do_not_select_family_routes() {
for model in [
"gpt-6.schema.other",
"catalog.gpt-5.other",
"catalog.schema.claude-gpt-6",
"catalog.schema.mygpt-6-astra",
"catalog.schema.gpt-4",
"catalog.schema.gpt-4o",
Expand Down
Loading
Loading