Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 10 additions & 2 deletions crates/buzz-agent/src/model_capabilities.rs
Original file line number Diff line number Diff line change
Expand Up @@ -622,6 +622,8 @@ mod tests {
Q::Vector { id: "dbv2-claude-opus-4-7-probe", provider: "databricks_v2", raw_model_id: "claude-opus-4-7", note: None },
Q::Vector { id: "dbv2-databricks-prefix-probe", provider: "databricks_v2", raw_model_id: "databricks-claude-opus-4-7", note: Some("Probes stripping of the databricks- catalog prefix.") },
Q::Vector { id: "dbv2-goose-claude-prefix-probe", provider: "databricks_v2", raw_model_id: "goose-claude-fable-5", note: Some("Probes stripping of the goose- catalog prefix.") },
Q::Vector { id: "dbv2-goose-claude-4-6-sonnet-alias-probe", provider: "databricks_v2", raw_model_id: "goose-claude-4-6-sonnet", note: Some("Probes the discovered Goose Sonnet 4.6 endpoint spelling and label.") },
Q::Vector { id: "dbv2-goose-claude-4-7-opus-alias-probe", provider: "databricks_v2", raw_model_id: "goose-claude-4-7-opus", note: Some("Probes the discovered Goose Opus 4.7 endpoint spelling and label.") },
Q::Vector { id: "dbv2-team-prefix-probe", provider: "databricks_v2", raw_model_id: "team-x-claude-opus-4-7", note: Some("Probes stripping of a team-x- catalog prefix.") },
Q::Vector { id: "dbv2-consolidated-llama-substring-probe", provider: "databricks_v2", raw_model_id: "consolidated-llama", note: Some("Probes a name where a code word ('sol') appears only as a substring, not a boundary-aligned segment.") },
Q::Vector { id: "dbv2-terraform-coder-substring-probe", provider: "databricks_v2", raw_model_id: "terraform-coder", note: Some("Probes a name where a code word ('terra') is only a segment prefix, not a full segment.") },
Expand All @@ -638,6 +640,7 @@ mod tests {
Q::Vector { id: "dbv2-goose-claude-opus-5-alias-probe", provider: "databricks_v2", raw_model_id: "goose-claude-opus-5", note: Some("Probes a prefixed alias of the Databricks Opus 5 endpoint.") },
Q::Vector { id: "dbv2-claude-sonnet-5-exact-record-probe", provider: "databricks_v2", raw_model_id: "databricks-claude-sonnet-5", note: Some("Probes the canonical Databricks Sonnet 5 endpoint record.") },
Q::Vector { id: "dbv2-goose-claude-sonnet-5-alias-probe", provider: "databricks_v2", raw_model_id: "goose-claude-sonnet-5", note: Some("Probes a prefixed alias of the Databricks Sonnet 5 endpoint.") },
Q::Vector { id: "dbv2-kimi-2-7-exact-record-probe", provider: "databricks_v2", raw_model_id: "databricks-kimi-2-7", note: Some("Probes the canonical Databricks Kimi 2.7 endpoint record.") },
Q::Vector { id: "dbv2-kimi-k3-exact-record-probe", provider: "databricks_v2", raw_model_id: "databricks-kimi-k3", note: Some("Probes the canonical Databricks Kimi K3 endpoint record.") },
Q::Vector { id: "dbv2-goose-kimi-k3-alias-probe", provider: "databricks_v2", raw_model_id: "goose-kimi-k3", note: Some("Probes a prefixed alias of the Databricks Kimi K3 endpoint.") },
Q::Vector { id: "resolver-prefixed-alias-probe", provider: "databricks_v2", raw_model_id: "team-x-databricks-gpt-5-4-mini", note: Some("Probes a prefixed alias of an exact-record id (raw exact key differs).") },
Expand Down Expand Up @@ -728,6 +731,7 @@ mod tests {
Q::Vector { id: "dbv2-gemini-3-pro-image-exact-record-probe", provider: "databricks_v2", raw_model_id: "databricks-gemini-3-pro-image", note: Some("Probes the Gemini 3 Pro Image endpoint record and label.") },
Q::Vector { id: "dbv2-deepseek-v4-flash-exact-record-probe", provider: "databricks_v2", raw_model_id: "databricks-deepseek-v4-flash-0731", note: Some("Probes the DeepSeek V4 Flash endpoint record and label.") },
Q::Vector { id: "dbv2-deepseek-v4-pro-exact-record-probe", provider: "databricks_v2", raw_model_id: "databricks-deepseek-v4-pro-0813", note: Some("Probes the DeepSeek V4 Pro endpoint record and label.") },
Q::Vector { id: "dbv2-glm-5-3-exact-record-probe", provider: "databricks_v2", raw_model_id: "databricks-glm-5-3", note: Some("Probes the GLM-5.3 endpoint record and label.") },
Q::Vector { id: "dbv2-glm-5-3-flash-exact-record-probe", provider: "databricks_v2", raw_model_id: "databricks-glm-5-3-flash", note: Some("Probes the GLM-5.3 Flash endpoint record and label.") },
Q::Vector { id: "dbv2-grok-4-6-exact-record-probe", provider: "databricks_v2", raw_model_id: "databricks-grok-4-6", note: Some("Probes the Grok 4.6 endpoint record and label.") },
Q::Vector { id: "dbv2-llama-4-maverick-exact-record-probe", provider: "databricks_v2", raw_model_id: "databricks-llama-4-maverick", note: Some("Probes the Llama 4 Maverick endpoint record and label.") },
Expand Down Expand Up @@ -839,7 +843,7 @@ mod tests {
}

#[test]
fn corpus_has_exactly_135_executable_vectors() {
fn corpus_has_exactly_139_executable_vectors() {
// Locks the vector count so a silent INPUTS edit can't quietly drop
// coverage; must equal the gate in the TS harness
// (modelCapabilitiesCorpus.test.mjs).
Expand All @@ -848,7 +852,7 @@ mod tests {
.filter(|q| matches!(q, Q::Vector { .. }))
.count();
assert_eq!(
vectors, 135,
vectors, 139,
"corpus executable-vector count changed; update this gate deliberately"
);
}
Expand Down Expand Up @@ -1018,9 +1022,12 @@ mod tests {
Some("Claude Fable 5")
);
for (alias, label) in [
("goose-claude-4-6-sonnet", "Claude Sonnet 4.6"),
("goose-claude-4-7-opus", "Claude Opus 4.7"),
("goose-claude-opus-4-8", "Claude Opus 4.8"),
("goose-claude-opus-5", "Claude Opus 5"),
("goose-claude-sonnet-5", "Claude Sonnet 5"),
("goose-kimi-2-7", "Kimi 2.7"),
("goose-kimi-k3", "Kimi K3"),
] {
assert_eq!(
Expand Down Expand Up @@ -1053,6 +1060,7 @@ mod tests {
"data_workflow_tools.goose.goose-deepseek-v4-flash-0731",
"DeepSeek V4 Flash",
),
("data_workflow_tools.goose.goose-glm-5-3", "GLM-5.3"),
(
"data_workflow_tools.goose.goose-glm-5-3-flash",
"GLM-5.3 Flash",
Expand Down
2 changes: 1 addition & 1 deletion desktop/src/features/agents/AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -281,7 +281,7 @@ with a TypeScript lookup table or an id comparison in a component.
refresh only local persona/team/managed-agent caches; they must never
invalidate the remote relay directory.

17. **Databricks model discovery has one shared catalog authority.** Desktop and ACP call the shared `buzz-agent` discovery library; Desktop passes the effective merged `DATABRICKS_MODEL_FILTER` explicitly, and the library applies it to raw workspace endpoint IDs and Unity Catalog model-service FQNs after the additive union. A successful filtered-empty catalog is authoritative: it stays empty, disables switching, and never falls through to configured or known-model fallback. UC FQNs are catalog data and always use the MLflow Chat Completions route, regardless of family-looking text in their components.
17. **Databricks model discovery has one shared catalog authority.** Desktop and ACP call the shared `buzz-agent` discovery library; Desktop passes the effective merged `DATABRICKS_MODEL_FILTER` explicitly, and the library applies it to raw workspace endpoint IDs and Unity Catalog model-service FQNs after the additive union. A successful filtered-empty catalog is authoritative: it stays empty, disables switching, and never falls through to configured or known-model fallback. UC FQNs are catalog data and always use the MLflow Chat Completions route, regardless of family-looking text in their components. Global Defaults preserves the discovered model ID as the selected value while its closed trigger renders the provider-scoped display label; do not force the raw persisted ID over that label.

## The tests that enforce this

Expand Down
1 change: 0 additions & 1 deletion desktop/src/features/agents/ui/AgentConfigFields.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -818,7 +818,6 @@ export function AgentConfigFields({
fallbackModel === null &&
!dependentFieldsDisabled
}
keepSelectedModelValueLabel
model={dependentFieldsDisabled ? "" : (config.model ?? "")}
modelDiscoveryLoading={
dependentFieldsDisabled ? false : modelDiscoveryLoading
Expand Down
10 changes: 0 additions & 10 deletions desktop/src/features/agents/ui/agentConfigControls.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -339,7 +339,6 @@ export function AgentModelField({
allowDefaultModel = true,
defaultModelLabel,
disableSelectDuringDiscovery = true,
keepSelectedModelValueLabel = false,
id = "agent-model",
isCustomModelEditing,
isRequired,
Expand Down Expand Up @@ -371,8 +370,6 @@ export function AgentModelField({
defaultModelLabel?: string;
/** Disable the trigger while live model discovery refreshes the option list. */
disableSelectDuringDiscovery?: boolean;
/** Keep the closed trigger from swapping to discovered display labels. */
keepSelectedModelValueLabel?: boolean;
/** DOM id for the model select. Defaults to `"agent-model"`. Override in
* contexts where multiple instances coexist on the same page (e.g. the
* global-config settings card) to avoid duplicate DOM ids. */
Expand Down Expand Up @@ -513,12 +510,6 @@ export function AgentModelField({
// yields an empty list and discovery has finished, add a disabled sentinel
// row so the user sees "No models found" instead of a bare white bar.
appendNoModelsSentinel(modelOptions, modelDiscoveryLoading);
const stableSelectedModelLabel =
keepSelectedModelValueLabel &&
modelSelectValue === trimmedModel &&
trimmedModel.length > 0
? trimmedModel
: undefined;
// While discovery is in flight with nothing selected, the closed field
// reads "Loading models…" instead of a select-prompt β€” the field isn't
// waiting on the user, it's waiting on the harness.
Expand Down Expand Up @@ -547,7 +538,6 @@ export function AgentModelField({
placeholder={restingPlaceholder}
placeholderClassName={placeholderClassName}
searchable
selectedLabel={stableSelectedModelLabel}
testId={testId ?? id}
value={modelSelectValue}
/>
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -25,10 +25,10 @@ const corpus = JSON.parse(readFileSync(fileURLToPath(corpusUrl), "utf8"));
// (`_group`) are skipped. Mirrors the Rust corpus filter.
const executable = corpus.filter((entry) => entry.expect != null);

test("corpus has exactly 135 executable vectors", () => {
test("corpus has exactly 139 executable vectors", () => {
// Locks the vector count so a silent corpus edit can't quietly drop coverage;
// must equal the gate in the Rust suite (model_capabilities.rs).
assert.equal(executable.length, 135);
assert.equal(executable.length, 139);
});

test("registry label aliases refuse an unprefixed query", () => {
Expand All @@ -50,6 +50,9 @@ test("UC model-family FQNs and goose- aliases humanize onto their base records",
// must resolve onto the same base databricks_v2 records via the new family
// tokens. Mirrors the Rust `test_databricks_registry_label_lookup` coverage.
const cases = [
["goose-claude-4-6-sonnet", "Claude Sonnet 4.6"],
["goose-claude-4-7-opus", "Claude Opus 4.7"],
["goose-kimi-2-7", "Kimi 2.7"],
["system.ai.gemini-3-5-flash", "Gemini 3.5 Flash"],
["system.ai.gemini-3-pro-image", "Gemini 3 Pro Image"],
["system.ai.deepseek-v4-pro-0813", "DeepSeek V4 Pro"],
Expand All @@ -65,6 +68,7 @@ test("UC model-family FQNs and goose- aliases humanize onto their base records",
"data_workflow_tools.goose.goose-deepseek-v4-flash-0731",
"DeepSeek V4 Flash",
],
["data_workflow_tools.goose.goose-glm-5-3", "GLM-5.3"],
["data_workflow_tools.goose.goose-glm-5-3-flash", "GLM-5.3 Flash"],
["data_workflow_tools.goose.goose-grok-4-6", "Grok 4.6"],
];
Expand Down
2 changes: 1 addition & 1 deletion desktop/tests/e2e/agents.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -746,7 +746,7 @@ test("agent defaults stays in the header without an actions menu", async ({
defaultsDialog.getByTestId("global-agent-model"),
).toHaveAttribute("data-value", "gpt-5.5[high]");
await expect(defaultsDialog.getByTestId("global-agent-model")).toContainText(
"gpt-5.5[high]",
"GPT-5.5 (high)",
);
await page.keyboard.press("Escape");
await expect(defaultsDialog).toHaveCount(0);
Expand Down
43 changes: 43 additions & 0 deletions desktop/tests/e2e/global-agent-config-screenshots.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -248,6 +248,49 @@ test.describe("global agent config screenshots", () => {
expect(saved).toMatchObject({ preferred_runtime: "codex" });
});

test("defaults render Databricks model labels without changing persisted ids", async ({
page,
}) => {
const modelId = "data_workflow_tools.goose.goose-glm-5-3";
await installMockBridge(page, {
globalAgentConfig: {
preferred_runtime: "goose",
provider: "databricks_v2",
model: modelId,
env_vars: {},
},
discoverAgentModels: {
models: [{ id: modelId, name: modelId }],
supportsSwitching: true,
selectedModel: modelId,
},
runtimeFileConfigs: {
goose: {
provider: "databricks_v2",
model: modelId,
satisfiedEnvKeys: ["DATABRICKS_HOST"],
},
},
});

await openAiDefaultsSettings(page);

const model = page.getByTestId("global-agent-model");
await expect(model).toHaveText("GLM-5.3");

const persisted = await page.evaluate(async () =>
(
window as typeof window & {
__BUZZ_E2E_INVOKE_MOCK_COMMAND__?: (
command: string,
payload: unknown,
) => Promise<unknown>;
}
).__BUZZ_E2E_INVOKE_MOCK_COMMAND__?.("get_global_agent_config", null),
);
expect(persisted).toMatchObject({ model: modelId });
});

test("defaults honor credentials set in the harness config file", async ({
page,
}) => {
Expand Down
73 changes: 73 additions & 0 deletions scripts/model-capabilities.json
Original file line number Diff line number Diff line change
Expand Up @@ -601,6 +601,24 @@
"_reconciliation_note": "models.dev advertises reasoning_options=[{\"type\":\"budget_tokens\",\"min\":1024}]. This is a different capability axis (extended thinking token budget), not an effort-level selector. No effort divergence to reconcile β€” efforts for this model come from the anthropic family rule (anthropic-adaptive-xhigh-opus-4-7).",
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-07-31, SHA-256 d5a4974cd69f19b0f67713acaa6bb3b16e920defdc07ecbdf6b0a936181bb0e0): providers.databricks.models[\"databricks-claude-opus-4-7\"].reasoning_options=[{\"type\":\"budget_tokens\",\"min\":1024}]"
},
{
"provider": "databricks_v2",
"raw_model_id": "goose-claude-4-7-opus",
"registry_label": "Claude Opus 4.7",
"thinking_mode": "adaptive",
"supported_efforts": [
"low",
"medium",
"high",
"xhigh",
"max"
],
"default_effort": "high",
"databricks_v2_wire_route": "anthropic-messages",
"normalization_policy": "none",
"_provenance": "all axes copied from equivalent registry endpoint databricks-claude-opus-4-7 for the discovered Goose endpoint id",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gpt-5-6-luna",
Expand Down Expand Up @@ -761,6 +779,23 @@
"_provenance": "all axes materialized from family:anthropic-adaptive-no-xhigh-sonnet-4-6",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "goose-claude-4-6-sonnet",
"registry_label": "Claude Sonnet 4.6",
"thinking_mode": "adaptive",
"supported_efforts": [
"low",
"medium",
"high",
"max"
],
"default_effort": "high",
"databricks_v2_wire_route": "anthropic-messages",
"normalization_policy": "none",
"_provenance": "all axes copied from equivalent registry endpoint databricks-claude-sonnet-4-6 for the discovered Goose endpoint id",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-gemini-2-5-flash",
Expand Down Expand Up @@ -1031,6 +1066,25 @@
"_reconciliation_note": "The exact databricks endpoint is absent from the models.dev databricks catalog, but the first-party deepseek entry advertises a reasoning toggle plus effort [high, max]. Per the Kimi-K3 precedent, adopt the model-level effort list despite endpoint absence. The toggle is not representable on the MLflow Chat request schema (reasoning_effort only), so thinking_mode stays none and default_effort is null β€” mirroring the kimi-k3 record.",
"_reconciliation_doc": "https://models.dev/api.json (pinned SHA-256 9266003029c4ea265e923637826211f597db08e8f0f40123aad8547411029238): providers.deepseek.models[\"deepseek-v4-pro\"].reasoning_options=[{\"type\":\"toggle\"},{\"type\":\"effort\",\"values\":[\"high\",\"max\"]}]"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-glm-5-3",
"registry_label": "GLM-5.3",
"thinking_mode": "none",
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"default_effort": "medium",
"databricks_v2_wire_route": "mlflow-chat",
"normalization_policy": "openai-clamp-max-to-xhigh",
"_provenance": "all axes materialized from provider fallback (databricks_v2/concrete_unknown)",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-glm-5-3-flash",
Expand Down Expand Up @@ -1385,6 +1439,25 @@
"_reconciliation_note": "models.dev advertises toggleable reasoning with [low, high, max], but no default. The Databricks endpoint is absent from its provider catalog, so retain the established MLflow Chat route, adopt the upstream capability set, and leave the effort unset rather than invent a Databricks default. The MLflow Chat request schema cannot express a reasoning toggle (only reasoning_effort is representable), so thinking_mode maps to none rather than the upstream toggle, matching the kimi-k2-7-code precedent.",
"_reconciliation_doc": "https://models.dev/api.json (retrieved 2026-08-20, SHA-256 7ccb5635f682e4248ad8d39f515fbe3f2bb10e67bbc7fa1b94e66dddb00779e5): providers.moonshotai.models[\"kimi-k3\"].reasoning_options=[{\"type\":\"toggle\"},{\"type\":\"effort\",\"values\":[\"low\",\"high\",\"max\"]}]; Databricks endpoint absent from providers.databricks.models"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-kimi-2-7",
"registry_label": "Kimi 2.7",
"thinking_mode": "none",
"supported_efforts": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"default_effort": "medium",
"databricks_v2_wire_route": "mlflow-chat",
"normalization_policy": "openai-clamp-max-to-xhigh",
"_provenance": "all axes materialized from provider fallback (databricks_v2/concrete_unknown)",
"_source": "registry_labels"
},
{
"provider": "databricks_v2",
"raw_model_id": "databricks-kimi-k2-7-code",
Expand Down
Loading
Loading