From 18e5b908ad90dda19628104222437b63327c8877 Mon Sep 17 00:00:00 2001 From: James Blair Date: Sat, 26 Sep 2026 12:11:22 -0600 Subject: [PATCH 1/4] Normalize array delta content in OpenAI-compatible streams Databricks streams reasoning models (e.g. GPT OSS) with delta.content as an array of reasoning/text parts instead of a string. The AI SDK rejects the whole chunk, and Databricks sends tool call arguments in the same chunk as reasoning, so tool calls completed with {} input. Collapse array content to its text parts (or null) in the shared OpenAI-compatible SSE transform, dropping reasoning parts. --- memory-bank/aiConfig.md | 9 +- .../openai-compat-reasoning-content.test.ts | 110 ++++++++++++++++++ .../src/model-clients/openai-compat-fetch.ts | 47 +++++++- 3 files changed, 160 insertions(+), 6 deletions(-) create mode 100644 packages/ai-provider-bridge/src/model-clients/__tests__/openai-compat-reasoning-content.test.ts diff --git a/memory-bank/aiConfig.md b/memory-bank/aiConfig.md index a391c69..6919f58 100644 --- a/memory-bank/aiConfig.md +++ b/memory-bank/aiConfig.md @@ -638,9 +638,12 @@ models it does not proxy natively. Endpoints advertising the routes to `{host}/ai-gateway/mlflow/v1` in the SDK's Responses mode. The chat surface loses on all three counts that matter: it rejects `store`, rejects `max_completion_tokens`, and streams reasoning as a `delta.content` block array -that the OpenAI chunk schema cannot represent (so reasoning is discarded), -while Responses accepts the same thinking controls and carries reasoning as -first-class items. +that the OpenAI chunk schema cannot represent, while Responses accepts the same +thinking controls and carries reasoning as first-class items. On the chat +surface the OpenAI-compatible fetch (transform 7) collapses that array to its +text parts: the AI SDK otherwise rejects the whole chunk, and Databricks sends +tool call arguments in the same chunk as reasoning (GPT OSS on classic serving +completed every tool call with `{}`). Reasoning itself is still discarded. The stamp is **gateway-only**: classic serving has no unified Responses route (`/serving-endpoints/responses` is native passthrough and refuses diff --git a/packages/ai-provider-bridge/src/model-clients/__tests__/openai-compat-reasoning-content.test.ts b/packages/ai-provider-bridge/src/model-clients/__tests__/openai-compat-reasoning-content.test.ts new file mode 100644 index 0000000..1b90435 --- /dev/null +++ b/packages/ai-provider-bridge/src/model-clients/__tests__/openai-compat-reasoning-content.test.ts @@ -0,0 +1,110 @@ +/*--------------------------------------------------------------------------------------------- + * Copyright (C) 2026 Posit Software, PBC. All rights reserved. + *--------------------------------------------------------------------------------------------*/ + +import { createOpenAI } from "@ai-sdk/openai"; +import { jsonSchema, streamText, tool } from "ai"; +import { describe, expect, it } from "vitest"; + +import { createOpenAICompatibleFetchMiddleware } from "../openai-compat-fetch"; + +/** + * Databricks streams reasoning models (e.g. `databricks-gpt-oss-120b`) with + * `delta.content` as an array of `{type: "reasoning", summary: [...]}` blocks + * instead of a string. The AI SDK rejects such a chunk whole — and Databricks + * sends a tool call's arguments in the same chunk as its reasoning, so the tool + * call completed with `{}` input. The tool-call and reasoning chunks below are + * captured from the wire (ids shortened); the `{type: "text"}` part is modeled + * on Databricks' non-streaming content shape. + */ +function chunk(delta: Record, finishReason: string | null = null): string { + return JSON.stringify({ + id: "chatcmpl_1", + object: "chat.completion.chunk", + created: 1790444153, + model: "gpt-oss-120b-080525", + choices: [{ index: 0, delta, finish_reason: finishReason, logprobs: null }], + }); +} + +function reasoning(text: string) { + return [{ type: "reasoning", summary: [{ type: "summary_text", text }] }]; +} + +const toolCallStream = [ + chunk({ + role: "assistant", + content: "", + tool_calls: [ + { + id: "call_1", + index: 0, + type: "function", + function: { name: "get_weather", arguments: "" }, + }, + ], + }), + chunk({ + content: reasoning('We need to call the get_weather function with city "Paris".'), + tool_calls: [{ index: 0, function: { arguments: '{\n "city": "Paris"\n}' } }], + }), + chunk({ content: "", tool_calls: [{ index: 0, function: { arguments: "" } }] }, "tool_calls"), +]; + +const textStream = [ + chunk({ role: "assistant", content: "" }), + chunk({ content: reasoning("We need") }), + chunk({ content: reasoning(" to say hello.") }), + chunk({ content: [{ type: "text", text: "Hello" }] }), + chunk({ content: "!" }), + chunk({ content: "" }, "stop"), +]; + +function run(sseChunks: string[]) { + const wire = async () => + new Response(sseChunks.map((c) => `data: ${c}\n\n`).join("") + "data: [DONE]\n\n", { + headers: { "content-type": "text/event-stream" }, + }); + const fetch = createOpenAICompatibleFetchMiddleware("Databricks", "key")(wire); + const provider = createOpenAI({ apiKey: "key", baseURL: "https://example.invalid", fetch }); + return streamText({ + model: provider.chat("databricks-gpt-oss-120b"), + prompt: "What's the weather in Paris?", + tools: { + get_weather: tool({ + inputSchema: jsonSchema<{ city: string }>({ + type: "object", + properties: { city: { type: "string" } }, + required: ["city"], + }), + }), + }, + }); +} + +async function parts(sseChunks: string[]) { + const result = run(sseChunks); + const collected = []; + for await (const part of result.fullStream) { + collected.push(part); + } + return collected; +} + +describe("createOpenAICompatibleFetch — array delta.content", () => { + it("keeps tool arguments that share a chunk with reasoning content", async () => { + const collected = await parts(toolCallStream); + + expect(collected.filter((p) => p.type === "error")).toEqual([]); + const toolCalls = collected.flatMap((p) => (p.type === "tool-call" ? [p.input] : [])); + expect(toolCalls).toEqual([{ city: "Paris" }]); + }); + + it("keeps text parts and drops reasoning parts without stream errors", async () => { + const collected = await parts(textStream); + + expect(collected.filter((p) => p.type === "error")).toEqual([]); + const text = collected.flatMap((p) => (p.type === "text-delta" ? [p.text] : [])).join(""); + expect(text).toBe("Hello!"); + }); +}); diff --git a/packages/ai-provider-bridge/src/model-clients/openai-compat-fetch.ts b/packages/ai-provider-bridge/src/model-clients/openai-compat-fetch.ts index 2bfa992..3981475 100644 --- a/packages/ai-provider-bridge/src/model-clients/openai-compat-fetch.ts +++ b/packages/ai-provider-bridge/src/model-clients/openai-compat-fetch.ts @@ -49,9 +49,18 @@ * 6. Empty tool `type` `""` → `"function"` in tool call chunks * Spec requires `type` to be `"function"`. Some providers send `""`. * + * 7. Array `content` → string (or `null`) in delta chunks + * Spec requires `content` to be a string or null. Databricks streams + * reasoning models (e.g. GPT OSS) with an array of content parts — + * `{type: "reasoning", summary: [...]}` and `{type: "text", text}`. The AI + * SDK rejects the whole chunk, and Databricks sends tool call arguments in + * the same chunk as reasoning, so the tool call completes with `{}`. Text + * parts are kept; reasoning parts are dropped (the chat completions path + * has no reasoning channel to carry them). + * * ## Auth * - * 7. Strip `Authorization` header when `apiKey === ""` + * 8. Strip `Authorization` header when `apiKey === ""` * For unauthenticated endpoints (e.g., local servers with no auth). * Only matches empty string — `undefined` means the caller manages * auth separately (e.g. Foundry injects its own token). @@ -88,13 +97,23 @@ interface MalformedToolCall { function: MalformedToolCallFunction; } +/** + * A content part in an array-valued delta `content`. Databricks sends + * `{type: "reasoning", summary: [...]}` and `{type: "text", text}` parts. + */ +interface MalformedContentPart { + type?: string; + text?: unknown; +} + /** * A delta where `role` may be empty string instead of `"assistant"`, - * and `tool_calls` may contain malformed entries. + * `content` may be an array of parts instead of a string, and `tool_calls` + * may contain malformed entries. */ interface MalformedDelta { role?: "assistant" | ""; // may be "" instead of "assistant" - content?: string | null; + content?: string | null | MalformedContentPart[]; // may be an array of parts tool_calls?: MalformedToolCall[]; } @@ -375,6 +394,16 @@ function fixMalformedChunk(chunk: MalformedChatCompletionChunk, noArgTools: stri delta.role = "assistant"; } + // Transform 7: Array content → string (or null). + // Spec: content is a string or null. + // Broken: Databricks streams reasoning models with an array of + // `{type: "reasoning"}` / `{type: "text"}` parts. + // Impact: AI SDK Zod validation rejects the chunk, dropping any tool + // call arguments it carries. + if (Array.isArray(delta.content)) { + delta.content = flattenContentParts(delta.content); + } + if (!Array.isArray(delta.tool_calls)) continue; for (const tc of delta.tool_calls) { @@ -403,3 +432,15 @@ function fixMalformedChunk(chunk: MalformedChatCompletionChunk, noArgTools: stri } } } + +/** + * Collapse array-valued delta `content` to the string the spec requires: + * the concatenated text of its `text` parts, or `null` when it has none. + * Reasoning parts are dropped. + */ +function flattenContentParts(parts: MalformedContentPart[]): string | null { + const text = parts + .flatMap((part) => (part.type === "text" && typeof part.text === "string" ? [part.text] : [])) + .join(""); + return text === "" ? null : text; +} From 63a60d4ef39ca8ad830f59066438df0e37890b4f Mon Sep 17 00:00:00 2001 From: James Blair Date: Sat, 26 Sep 2026 12:41:53 -0600 Subject: [PATCH 2/4] Keep Databricks GPT OSS on gateway chat completions The Unity AI Gateway's unified Responses API streams GPT OSS tool calls with empty arguments in every event, response.completed included, so every tool call reached the client as {} (or not at all). The same request non-streamed, or streamed over gateway chat completions, carries the arguments; Qwen and Llama stream them fine on Responses. Withhold the mlflow-responses stamp from gpt-oss identities so they fall back to openai-chat, which they advertise. --- memory-bank/aiConfig.md | 9 +++++++ .../__tests__/databricks-helpers.test.ts | 5 ++++ .../model-capabilities/databricks-helpers.ts | 25 ++++++++++++++++++- 3 files changed, 38 insertions(+), 1 deletion(-) diff --git a/memory-bank/aiConfig.md b/memory-bank/aiConfig.md index 6919f58..2bb9ce3 100644 --- a/memory-bank/aiConfig.md +++ b/memory-bank/aiConfig.md @@ -645,6 +645,15 @@ text parts: the AI SDK otherwise rejects the whole chunk, and Databricks sends tool call arguments in the same chunk as reasoning (GPT OSS on classic serving completed every tool call with `{}`). Reasoning itself is still discarded. +**GPT OSS is the exception: it stays on chat completions on the gateway.** Its +streamed `mlflow/v1/responses` output carries `function_call` items with +`arguments: ""` in every event (including `response.completed`) and no +argument deltas, so every tool call reaches the client as `{}`; the same +request non-streamed, or streamed over gateway chat completions, carries the +arguments, and other hosted families (Qwen, Llama) stream them fine. The +classifier therefore withholds the `mlflow-responses` stamp from `gpt-oss*` +identities, and they fall back to `openai-chat` (which they advertise). + The stamp is **gateway-only**: classic serving has no unified Responses route (`/serving-endpoints/responses` is native passthrough and refuses non-passthrough models), so serving keeps `openai-chat`. The advertised diff --git a/packages/ai-config/src/model-capabilities/__tests__/databricks-helpers.test.ts b/packages/ai-config/src/model-capabilities/__tests__/databricks-helpers.test.ts index e7188fe..d1e0962 100644 --- a/packages/ai-config/src/model-capabilities/__tests__/databricks-helpers.test.ts +++ b/packages/ai-config/src/model-capabilities/__tests__/databricks-helpers.test.ts @@ -379,6 +379,11 @@ describe("inferDatabricksModelProfile — unified MLflow Responses route", () => ]); expect(responseProtocol("gateway", [claude])).toBe("anthropic-messages"); }); + + it("keeps GPT OSS on chat, where its streamed tool arguments survive", () => { + const gptOss = foundationEntity("system.ai.gpt-oss-120b", [GATEWAY_CHAT, GATEWAY_RESPONSES]); + expect(responseProtocol("gateway", [gptOss])).toBe("openai-chat"); + }); }); describe("inferDatabricksModelProfile — gateway chat availability", () => { diff --git a/packages/ai-config/src/model-capabilities/databricks-helpers.ts b/packages/ai-config/src/model-capabilities/databricks-helpers.ts index 3627235..842176d 100644 --- a/packages/ai-config/src/model-capabilities/databricks-helpers.ts +++ b/packages/ai-config/src/model-capabilities/databricks-helpers.ts @@ -221,6 +221,19 @@ function isResponsesCompatibleOpenAIId(modelId: string): boolean { return /^gpt-5/.test(bare) || /^gpt-4o/.test(bare); } +/** + * Whether the gateway's unified Responses API streams this model's tool calls + * intact. GPT OSS does not: its streamed `function_call` items carry + * `arguments: ""` in every event, `response.completed` included, with no + * `function_call_arguments.delta` events — so every tool call arrives with `{}` + * input. The same request non-streamed, or over the gateway's chat completions + * route, carries the arguments, and other hosted families (Qwen, Llama) stream + * them fine on Responses. + */ +function streamsUnifiedResponsesToolArguments(modelId: string): boolean { + return !/^gpt-oss/.test(modelId.toLowerCase()); +} + // --------------------------------------------------------------------------- // Per-entity resolution // --------------------------------------------------------------------------- @@ -240,6 +253,8 @@ interface EntityResolution { readonly vendor: string | undefined; readonly apiTypes: readonly string[]; readonly gatewayV2Supported: boolean; + /** Whether the unified MLflow Responses route may serve this entity. */ + readonly unifiedResponsesEligible: boolean; } /** OpenAI table capabilities with the shared-context input reservation applied. */ @@ -353,6 +368,7 @@ function resolveEntity( vendor: recognizedVendor(normalizedIdentity), apiTypes: foundationModel?.api_types ?? [], gatewayV2Supported: foundationModel?.ai_gateway_v2_supported === true, + unifiedResponsesEligible: streamsUnifiedResponsesToolArguments(normalizedIdentity), } as const; if (!nativeAllowed) { @@ -586,7 +602,14 @@ export function inferDatabricksModelProfile( // Gateway-only: classic serving has no unified Responses route // (`/serving-endpoints/responses` is native passthrough and refuses // non-passthrough models), so serving keeps chat completions. - if (input.surface === "gateway" && everyGatewayEntityAdvertises(GATEWAY_RESPONSES_API_TYPE)) { + // + // Advertising the route is not enough for every family: GPT OSS streams + // empty tool arguments on it (see `streamsUnifiedResponsesToolArguments`). + if ( + input.surface === "gateway" && + everyGatewayEntityAdvertises(GATEWAY_RESPONSES_API_TYPE) && + resolutions.every((entity) => entity.unifiedResponsesEligible) + ) { return { excluded: false, protocol: "mlflow-responses", From 2afd83a1ad3fdcda78d601bae7e4f792d36c85c2 Mon Sep 17 00:00:00 2001 From: James Blair Date: Tue, 29 Sep 2026 10:44:22 -0600 Subject: [PATCH 3/4] Tighten GPT OSS docs and text-part handling Keep output_text parts as well as text parts when collapsing array content, and test parts whose text is not a string. Note that gateway chat completions can also drop GPT OSS output on large requests, so it is the better route, not a clean one. Add Databricks to the fetch wrapper's provider list. --- memory-bank/aiConfig.md | 20 ++++++++----- .../__tests__/databricks-helpers.test.ts | 2 +- .../model-capabilities/databricks-helpers.ts | 8 ++++-- .../openai-compat-reasoning-content.test.ts | 13 +++++++++ .../src/model-clients/openai-compat-fetch.ts | 28 ++++++++++++------- 5 files changed, 50 insertions(+), 21 deletions(-) diff --git a/memory-bank/aiConfig.md b/memory-bank/aiConfig.md index 2bb9ce3..8cb3e54 100644 --- a/memory-bank/aiConfig.md +++ b/memory-bank/aiConfig.md @@ -641,18 +641,24 @@ surface loses on all three counts that matter: it rejects `store`, rejects that the OpenAI chunk schema cannot represent, while Responses accepts the same thinking controls and carries reasoning as first-class items. On the chat surface the OpenAI-compatible fetch (transform 7) collapses that array to its -text parts: the AI SDK otherwise rejects the whole chunk, and Databricks sends -tool call arguments in the same chunk as reasoning (GPT OSS on classic serving -completed every tool call with `{}`). Reasoning itself is still discarded. +text parts: the AI SDK otherwise rejects the whole chunk, and Databricks can +send tool call arguments in the same chunk as reasoning (GPT OSS on classic +serving completed every tool call with `{}`). With that transform in place the +block array no longer breaks tool calls, so the remaining reason to prefer +Responses here is reasoning fidelity: chat still discards the reasoning. **GPT OSS is the exception: it stays on chat completions on the gateway.** Its streamed `mlflow/v1/responses` output carries `function_call` items with `arguments: ""` in every event (including `response.completed`) and no argument deltas, so every tool call reaches the client as `{}`; the same -request non-streamed, or streamed over gateway chat completions, carries the -arguments, and other hosted families (Qwen, Llama) stream them fine. The -classifier therefore withholds the `mlflow-responses` stamp from `gpt-oss*` -identities, and they fall back to `openai-chat` (which they advertise). +request non-streamed carries the arguments, and other hosted families (Qwen, +Llama) stream them fine. The classifier therefore withholds the +`mlflow-responses` stamp from `gpt-oss*` identities, and they fall back to +`openai-chat` (which they advertise). Chat completions is the better of two +imperfect routes rather than a clean one: it streams the arguments, but it has +also been observed to drop GPT OSS output entirely on large, tool-heavy +requests (observed 2026-09, on both surfaces; re-test before relaxing the +rule). The stamp is **gateway-only**: classic serving has no unified Responses route (`/serving-endpoints/responses` is native passthrough and refuses diff --git a/packages/ai-config/src/model-capabilities/__tests__/databricks-helpers.test.ts b/packages/ai-config/src/model-capabilities/__tests__/databricks-helpers.test.ts index d1e0962..819a4f1 100644 --- a/packages/ai-config/src/model-capabilities/__tests__/databricks-helpers.test.ts +++ b/packages/ai-config/src/model-capabilities/__tests__/databricks-helpers.test.ts @@ -380,7 +380,7 @@ describe("inferDatabricksModelProfile — unified MLflow Responses route", () => expect(responseProtocol("gateway", [claude])).toBe("anthropic-messages"); }); - it("keeps GPT OSS on chat, where its streamed tool arguments survive", () => { + it("keeps GPT OSS off unified Responses (empty streamed tool arguments)", () => { const gptOss = foundationEntity("system.ai.gpt-oss-120b", [GATEWAY_CHAT, GATEWAY_RESPONSES]); expect(responseProtocol("gateway", [gptOss])).toBe("openai-chat"); }); diff --git a/packages/ai-config/src/model-capabilities/databricks-helpers.ts b/packages/ai-config/src/model-capabilities/databricks-helpers.ts index 842176d..3df9e31 100644 --- a/packages/ai-config/src/model-capabilities/databricks-helpers.ts +++ b/packages/ai-config/src/model-capabilities/databricks-helpers.ts @@ -226,9 +226,11 @@ function isResponsesCompatibleOpenAIId(modelId: string): boolean { * intact. GPT OSS does not: its streamed `function_call` items carry * `arguments: ""` in every event, `response.completed` included, with no * `function_call_arguments.delta` events — so every tool call arrives with `{}` - * input. The same request non-streamed, or over the gateway's chat completions - * route, carries the arguments, and other hosted families (Qwen, Llama) stream - * them fine on Responses. + * input. The same request non-streamed carries the arguments, and other hosted + * families (Qwen, Llama) stream them fine on Responses. Chat completions is the + * better of two imperfect routes for GPT OSS: it streams the arguments, though + * it has also been observed to drop output entirely on large, tool-heavy + * requests (observed 2026-09; re-test before relaxing this rule). */ function streamsUnifiedResponsesToolArguments(modelId: string): boolean { return !/^gpt-oss/.test(modelId.toLowerCase()); diff --git a/packages/ai-provider-bridge/src/model-clients/__tests__/openai-compat-reasoning-content.test.ts b/packages/ai-provider-bridge/src/model-clients/__tests__/openai-compat-reasoning-content.test.ts index 1b90435..b2fa723 100644 --- a/packages/ai-provider-bridge/src/model-clients/__tests__/openai-compat-reasoning-content.test.ts +++ b/packages/ai-provider-bridge/src/model-clients/__tests__/openai-compat-reasoning-content.test.ts @@ -107,4 +107,17 @@ describe("createOpenAICompatibleFetch — array delta.content", () => { const text = collected.flatMap((p) => (p.type === "text-delta" ? [p.text] : [])).join(""); expect(text).toBe("Hello!"); }); + + it("keeps output_text parts and ignores parts without string text", async () => { + const collected = await parts([ + chunk({ role: "assistant", content: "" }), + chunk({ content: [{ type: "output_text", text: "Hi" }] }), + chunk({ content: [{ type: "text", text: 42 }] }), + chunk({ content: "" }, "stop"), + ]); + + expect(collected.filter((p) => p.type === "error")).toEqual([]); + const text = collected.flatMap((p) => (p.type === "text-delta" ? [p.text] : [])).join(""); + expect(text).toBe("Hi"); + }); }); diff --git a/packages/ai-provider-bridge/src/model-clients/openai-compat-fetch.ts b/packages/ai-provider-bridge/src/model-clients/openai-compat-fetch.ts index 3981475..10edb98 100644 --- a/packages/ai-provider-bridge/src/model-clients/openai-compat-fetch.ts +++ b/packages/ai-provider-bridge/src/model-clients/openai-compat-fetch.ts @@ -5,8 +5,8 @@ /** * Shared OpenAI-compatible fetch wrapper. * - * Many OpenAI-compatible providers (Snowflake Cortex, MS Foundry, generic - * endpoints) return responses that deviate from the OpenAI Chat Completions + * Many OpenAI-compatible providers (Snowflake Cortex, MS Foundry, Databricks, + * generic endpoints) return responses that deviate from the OpenAI Chat Completions * spec in small but breaking ways. The AI SDK's Zod schema validation * rejects these malformed chunks, crashing the stream. * @@ -53,10 +53,11 @@ * Spec requires `content` to be a string or null. Databricks streams * reasoning models (e.g. GPT OSS) with an array of content parts — * `{type: "reasoning", summary: [...]}` and `{type: "text", text}`. The AI - * SDK rejects the whole chunk, and Databricks sends tool call arguments in - * the same chunk as reasoning, so the tool call completes with `{}`. Text - * parts are kept; reasoning parts are dropped (the chat completions path - * has no reasoning channel to carry them). + * SDK rejects the whole chunk, and Databricks can send tool call arguments + * in the same chunk as reasoning, so the tool call completes with `{}`. + * Text parts (`text` or `output_text`) are kept; every other part type, + * reasoning included, is dropped (the chat completions path has no + * reasoning channel to carry them). * * ## Auth * @@ -399,7 +400,7 @@ function fixMalformedChunk(chunk: MalformedChatCompletionChunk, noArgTools: stri // Broken: Databricks streams reasoning models with an array of // `{type: "reasoning"}` / `{type: "text"}` parts. // Impact: AI SDK Zod validation rejects the chunk, dropping any tool - // call arguments it carries. + // call arguments it can carry alongside the reasoning. if (Array.isArray(delta.content)) { delta.content = flattenContentParts(delta.content); } @@ -433,14 +434,21 @@ function fixMalformedChunk(chunk: MalformedChatCompletionChunk, noArgTools: stri } } +/** Content part types whose `text` is visible output rather than reasoning. */ +const TEXT_PART_TYPES: ReadonlySet = new Set(["text", "output_text"]); + /** * Collapse array-valued delta `content` to the string the spec requires: - * the concatenated text of its `text` parts, or `null` when it has none. - * Reasoning parts are dropped. + * the concatenated text of its text parts, or `null` when it has none. + * Every other part type (reasoning included) is dropped deliberately. */ function flattenContentParts(parts: MalformedContentPart[]): string | null { const text = parts - .flatMap((part) => (part.type === "text" && typeof part.text === "string" ? [part.text] : [])) + .flatMap((part) => + part.type !== undefined && TEXT_PART_TYPES.has(part.type) && typeof part.text === "string" + ? [part.text] + : [], + ) .join(""); return text === "" ? null : text; } From c4a378a8d4f8733a18e880e33ef7839031faa30a Mon Sep 17 00:00:00 2001 From: Winston Chang Date: Tue, 29 Sep 2026 13:50:46 -0500 Subject: [PATCH 4/4] Cover Databricks GPT OSS routing and malformed content --- memory-bank/aiConfig.md | 24 +++++++------ .../model-capabilities/databricks-helpers.ts | 11 +++--- .../openai-compat-reasoning-content.test.ts | 13 +++++++ .../src/model-clients/openai-compat-fetch.ts | 36 ++++++++----------- .../__tests__/databricks-provider.test.ts | 34 +++++++++++++++++- 5 files changed, 80 insertions(+), 38 deletions(-) diff --git a/memory-bank/aiConfig.md b/memory-bank/aiConfig.md index 8cb3e54..855b674 100644 --- a/memory-bank/aiConfig.md +++ b/memory-bank/aiConfig.md @@ -581,14 +581,15 @@ Classification rules, in order: not read — splits change independently of the configuration observed here); the native route requires every entity to resolve to the same protocol (mixed/empty/ambiguous prefers unified MLflow Responses when every entity - advertises it, then falls back to `openai-chat` when chat is unanimously - advertised); capabilities aggregate conservatively across + advertises it and is eligible, then falls back to `openai-chat` when chat is + unanimously advertised); capabilities aggregate conservatively across same-protocol entities (minimum numeric limits, intersection of media-type/effort-level sets, boolean AND, `vendor`/`family` only when unanimous); and the whole computation is entity-order invariant. - **Fallback stamps are always explicit**, never `undefined`: a gateway endpoint - gets `mlflow-responses` when unanimously advertised, otherwise `openai-chat` - wherever chat exists. `undefined` stays reserved for providers that made no + gets `mlflow-responses` when every entity advertises it and can stream tool + arguments on that route (unlike GPT OSS), otherwise `openai-chat` wherever + chat exists. `undefined` stays reserved for providers that made no routing decision at all. **Two documented limitations, not fixed here:** @@ -633,13 +634,14 @@ table's levels. completions.** The gateway exposes `/ai-gateway/mlflow/v1/responses` alongside `/ai-gateway/mlflow/v1/chat/completions`. It is a different API from the `openai/v1/responses` **native passthrough**, which Databricks refuses for -models it does not proxy natively. Endpoints advertising the -`mlflow/v1/responses` api_type are therefore stamped `mlflow-responses`, which -routes to `{host}/ai-gateway/mlflow/v1` in the SDK's Responses mode. The chat -surface loses on all three counts that matter: it rejects `store`, rejects -`max_completion_tokens`, and streams reasoning as a `delta.content` block array -that the OpenAI chunk schema cannot represent, while Responses accepts the same -thinking controls and carries reasoning as first-class items. On the chat +models it does not proxy natively. When every served entity advertises +`mlflow/v1/responses` and can stream tool arguments on it, the endpoint is +stamped `mlflow-responses`. This routes to `{host}/ai-gateway/mlflow/v1` in the +SDK's Responses mode. The chat surface loses on all three counts that matter: +it rejects `store` and `max_completion_tokens`, and streams reasoning as a +`delta.content` block array that the OpenAI chunk schema cannot represent, +while Responses accepts those thinking controls and carries reasoning as +first-class items. On the chat surface the OpenAI-compatible fetch (transform 7) collapses that array to its text parts: the AI SDK otherwise rejects the whole chunk, and Databricks can send tool call arguments in the same chunk as reasoning (GPT OSS on classic diff --git a/packages/ai-config/src/model-capabilities/databricks-helpers.ts b/packages/ai-config/src/model-capabilities/databricks-helpers.ts index 3df9e31..37df691 100644 --- a/packages/ai-config/src/model-capabilities/databricks-helpers.ts +++ b/packages/ai-config/src/model-capabilities/databricks-helpers.ts @@ -19,9 +19,10 @@ * The rules are deliberately a **positive identification**: a native protocol is * stamped only when the endpoint's structure, the model's identity, and (on the * gateway surface) the advertised `api_types` all agree. A non-native gateway - * endpoint uses unified MLflow Responses when every entity advertises it, then - * falls back to an explicit `openai-chat` stamp when every entity advertises - * chat. `undefined` is never returned as a protocol. + * endpoint uses unified MLflow Responses when every entity advertises it and + * streams tool arguments on it (GPT OSS does not), then falls back to an + * explicit `openai-chat` stamp when every entity advertises chat. `undefined` + * is never returned as a protocol. * * Two outcomes exist because unavailability is real: on the gateway surface an * endpoint whose entities cannot all serve any supported route must not be @@ -592,8 +593,8 @@ export function inferDatabricksModelProfile( // --- Unified MLflow Responses, preferred over chat completions --- // The gateway exposes a unified Responses API alongside chat completions. - // For everything that did not qualify for a native vendor route it is the - // better target: chat completions rejects `store`, rejects + // For eligible models that did not qualify for a native vendor route it is + // the better target: chat completions rejects `store`, rejects // `max_completion_tokens`, and streams reasoning as a block array that the // OpenAI chunk schema cannot represent (so reasoning is dropped), while // Responses carries reasoning as first-class items and accepts the same diff --git a/packages/ai-provider-bridge/src/model-clients/__tests__/openai-compat-reasoning-content.test.ts b/packages/ai-provider-bridge/src/model-clients/__tests__/openai-compat-reasoning-content.test.ts index b2fa723..a29c18a 100644 --- a/packages/ai-provider-bridge/src/model-clients/__tests__/openai-compat-reasoning-content.test.ts +++ b/packages/ai-provider-bridge/src/model-clients/__tests__/openai-compat-reasoning-content.test.ts @@ -100,6 +100,19 @@ describe("createOpenAICompatibleFetch — array delta.content", () => { expect(toolCalls).toEqual([{ city: "Paris" }]); }); + it("keeps co-chunked tool arguments when content contains a non-object part", async () => { + const malformed = [...toolCallStream]; + malformed[1] = chunk({ + content: [null, ...reasoning("Calling get_weather")], + tool_calls: [{ index: 0, function: { arguments: '{"city":"Paris"}' } }], + }); + const collected = await parts(malformed); + + expect(collected.filter((p) => p.type === "error")).toEqual([]); + const toolCalls = collected.flatMap((p) => (p.type === "tool-call" ? [p.input] : [])); + expect(toolCalls).toEqual([{ city: "Paris" }]); + }); + it("keeps text parts and drops reasoning parts without stream errors", async () => { const collected = await parts(textStream); diff --git a/packages/ai-provider-bridge/src/model-clients/openai-compat-fetch.ts b/packages/ai-provider-bridge/src/model-clients/openai-compat-fetch.ts index 10edb98..eb049e5 100644 --- a/packages/ai-provider-bridge/src/model-clients/openai-compat-fetch.ts +++ b/packages/ai-provider-bridge/src/model-clients/openai-compat-fetch.ts @@ -98,15 +98,6 @@ interface MalformedToolCall { function: MalformedToolCallFunction; } -/** - * A content part in an array-valued delta `content`. Databricks sends - * `{type: "reasoning", summary: [...]}` and `{type: "text", text}` parts. - */ -interface MalformedContentPart { - type?: string; - text?: unknown; -} - /** * A delta where `role` may be empty string instead of `"assistant"`, * `content` may be an array of parts instead of a string, and `tool_calls` @@ -114,7 +105,7 @@ interface MalformedContentPart { */ interface MalformedDelta { role?: "assistant" | ""; // may be "" instead of "assistant" - content?: string | null | MalformedContentPart[]; // may be an array of parts + content?: string | null | unknown[]; // may be an array of untrusted content parts tool_calls?: MalformedToolCall[]; } @@ -434,21 +425,24 @@ function fixMalformedChunk(chunk: MalformedChatCompletionChunk, noArgTools: stri } } -/** Content part types whose `text` is visible output rather than reasoning. */ -const TEXT_PART_TYPES: ReadonlySet = new Set(["text", "output_text"]); - /** * Collapse array-valued delta `content` to the string the spec requires: * the concatenated text of its text parts, or `null` when it has none. * Every other part type (reasoning included) is dropped deliberately. */ -function flattenContentParts(parts: MalformedContentPart[]): string | null { - const text = parts - .flatMap((part) => - part.type !== undefined && TEXT_PART_TYPES.has(part.type) && typeof part.text === "string" - ? [part.text] - : [], - ) - .join(""); +function flattenContentParts(parts: unknown[]): string | null { + let text = ""; + for (const part of parts) { + if ( + typeof part === "object" && + part !== null && + "type" in part && + (part.type === "text" || part.type === "output_text") && + "text" in part && + typeof part.text === "string" + ) { + text += part.text; + } + } return text === "" ? null : text; } diff --git a/packages/ai-provider-bridge/src/providers/__tests__/databricks-provider.test.ts b/packages/ai-provider-bridge/src/providers/__tests__/databricks-provider.test.ts index d27e7a6..42ee7ec 100644 --- a/packages/ai-provider-bridge/src/providers/__tests__/databricks-provider.test.ts +++ b/packages/ai-provider-bridge/src/providers/__tests__/databricks-provider.test.ts @@ -181,6 +181,21 @@ const FOUNDATION_MODELS_FIXTURE = { ], }, }, + // GPT OSS advertises both routes, but its Responses stream drops tool arguments. + { + name: "databricks-gpt-oss-120b", + config: { + served_entities: [ + { + foundation_model: { + name: "databricks-gpt-oss-120b", + api_types: ["mlflow/v1/chat/completions", "mlflow/v1/responses"], + ai_gateway_v2_supported: true, + }, + }, + ], + }, + }, // Embeddings model — excluded (never advertises the chat api_type) { name: "databricks-gte-large-en", @@ -361,6 +376,7 @@ describe("registerDatabricksProvider model fetcher", () => { expect(models.map((m) => [m.id, m.protocol])).toEqual([ ["databricks-claude-opus-4-8", "anthropic-messages"], ["databricks-llama-4-maverick", "openai-chat"], + ["databricks-gpt-oss-120b", "openai-chat"], ]); expect(models[0]).toMatchObject({ name: "Claude Opus 4.8", @@ -373,6 +389,22 @@ describe("registerDatabricksProvider model fetcher", () => { expect(workspace.urlsCalled()).toEqual([PROBE_URL, FOUNDATION_LIST_URL]); }); + it("routes discovered gateway GPT OSS over chat even when Responses is advertised", async () => { + const workspace = stubWorkspaceFetch({ probeStatus: 200 }); + const models = await registry.getModelsForProvider("databricks", CREDENTIALS); + const gptOss = models.find((model) => model.id === "databricks-gpt-oss-120b"); + if (!gptOss) throw new Error("GPT OSS endpoint was not discovered"); + + expect(gptOss.protocol).toBe("openai-chat"); + await runChat(registry.getClientForProvider("databricks", CREDENTIALS), { + model: gptOss.id, + protocol: gptOss.protocol, + }); + expect(workspace.chatRequests.map((request) => request.url)).toEqual([ + `${HOST}/ai-gateway/mlflow/v1/chat/completions`, + ]); + }); + it("probes once per registration and shares the pin with concurrent callers", async () => { const workspace = stubWorkspaceFetch({ probeStatus: 200 }); @@ -674,7 +706,7 @@ describe("registerDatabricksProvider route seam", () => { it("routes gateway unified MLflow Responses under /ai-gateway/mlflow/v1", async () => { expect( await routedUrl(200, { - model: "databricks-gpt-oss-120b", + model: "databricks-qwen3-235b-a22b", protocol: "mlflow-responses", }), ).toBe(`${HOST}/ai-gateway/mlflow/v1/responses`);