From 411a8cebc7ecad220a14e66e5ecf4d834334648a Mon Sep 17 00:00:00 2001 From: James Magahern Date: Fri, 2 Oct 2026 13:56:44 -0700 Subject: [PATCH] Expose chat-latest and default Quick Question to it --- docs/api/rest.md | 4 +- docs/api/streaming-chat.md | 1 + .../Sources/Sybil/SybilSettingsStore.swift | 4 +- .../Sybil/Sources/Sybil/SybilViewModel.swift | 2 +- .../Sybil/Tests/SybilTests/SybilTests.swift | 44 +++++++++++++++++- ios/README.md | 3 ++ server/src/llm/provider-adapters.ts | 2 +- server/tests/provider-adapters.test.ts | 46 ++++++++++++++++++- web/src/App.tsx | 10 ++-- web/src/lib/quick-question.ts | 8 ++++ web/tests/quick-question.test.mjs | 23 +++++++++- 11 files changed, 132 insertions(+), 15 deletions(-) diff --git a/docs/api/rest.md b/docs/api/rest.md index 9d922cf..58d1955 100644 --- a/docs/api/rest.md +++ b/docs/api/rest.md @@ -31,7 +31,7 @@ Chat upload limits: ```json { "providers": { - "openai": { "models": ["gpt-4.1-mini"], "loadedAt": "2026-02-14T00:00:00.000Z", "error": null }, + "openai": { "models": ["chat-latest", "gpt-4.1-mini"], "loadedAt": "2026-02-14T00:00:00.000Z", "error": null }, "anthropic": { "models": ["claude-3-5-sonnet-latest"], "loadedAt": null, "error": null }, "xai": { "models": ["grok-3-mini"], "loadedAt": null, "error": null }, "gemini": { "models": ["gemini-3.5-flash"], "loadedAt": null, "error": null }, @@ -39,7 +39,7 @@ Chat upload limits: } } ``` -- OpenAI model lists are filtered to models that are expected to work with the backend's Responses API implementation. +- OpenAI model lists are filtered to models that are expected to work with the backend's Responses API implementation, including the exact `chat-latest` alias when returned by OpenAI. Audio, image-generation, embedding, moderation, and other specialized model ids remain excluded. See [OpenAI's Chat Latest documentation](https://developers.openai.com/api/docs/models/chat-latest) for supported capabilities. - Gemini model lists are loaded from Google's native Models API and filtered to Gemini `generateContent` model ids. - `hermes-agent` is included only when `HERMES_AGENT_API_KEY` is configured. Set it to Hermes `API_SERVER_KEY`, or any non-empty value if that local server does not require auth. `HERMES_AGENT_API_BASE_URL` defaults to `http://127.0.0.1:8642/v1`; set `HERMES_AGENT_MODEL` only when you need an additional fallback/override model id. - The backend loads provider model lists at startup and refreshes them about once every 24 hours. If a later provider refresh fails, the response keeps the last loaded model list for that provider and sets `error` to the latest failure message. diff --git a/docs/api/streaming-chat.md b/docs/api/streaming-chat.md index ac99b63..e520cde 100644 --- a/docs/api/streaming-chat.md +++ b/docs/api/streaming-chat.md @@ -109,6 +109,7 @@ Request body: ``` Behavior notes: +- `provider` and `model` are required. Web and native clients default Quick Question to `provider: "openai"` and `model: "chat-latest"` when no Quick Question selection has been saved. Saved Quick Question choices take precedence over this default and are separate from regular chat preferences. - `question` is required, trimmed by the server, and must not be empty. - The server prepends a Quick Question system prompt that asks for a succinct, direct, self-contained answer without follow-up questions. Clients do not send or maintain this prompt. - Quick Questions are always non-persistent. The endpoint does not create a chat or store messages, tool-call logs, assistant output, or `LlmCall` metadata. diff --git a/ios/Packages/Sybil/Sources/Sybil/SybilSettingsStore.swift b/ios/Packages/Sybil/Sources/Sybil/SybilSettingsStore.swift index 55e5145..e4e6d8a 100644 --- a/ios/Packages/Sybil/Sources/Sybil/SybilSettingsStore.swift +++ b/ios/Packages/Sybil/Sources/Sybil/SybilSettingsStore.swift @@ -52,9 +52,9 @@ final class SybilSettingsStore { self.preferredModelByProvider = preferredModels self.quickQuestionPreferredProvider = - defaults.string(forKey: Keys.quickQuestionPreferredProvider).flatMap(Provider.init(rawValue:)) ?? provider + defaults.string(forKey: Keys.quickQuestionPreferredProvider).flatMap(Provider.init(rawValue:)) ?? .openai self.quickQuestionPreferredModelByProvider = [ - .openai: defaults.string(forKey: Keys.quickQuestionPreferredOpenAIModel) ?? preferredModels[.openai] ?? "gpt-4.1-mini", + .openai: defaults.string(forKey: Keys.quickQuestionPreferredOpenAIModel) ?? "chat-latest", .anthropic: defaults.string(forKey: Keys.quickQuestionPreferredAnthropicModel) ?? preferredModels[.anthropic] ?? "claude-3-5-sonnet-latest", .xai: defaults.string(forKey: Keys.quickQuestionPreferredXAIModel) ?? preferredModels[.xai] ?? "grok-3-mini", .gemini: defaults.string(forKey: Keys.quickQuestionPreferredGeminiModel) ?? preferredModels[.gemini] ?? "gemini-3.5-flash", diff --git a/ios/Packages/Sybil/Sources/Sybil/SybilViewModel.swift b/ios/Packages/Sybil/Sources/Sybil/SybilViewModel.swift index 947693f..c94a4dc 100644 --- a/ios/Packages/Sybil/Sources/Sybil/SybilViewModel.swift +++ b/ios/Packages/Sybil/Sources/Sybil/SybilViewModel.swift @@ -222,7 +222,7 @@ final class SybilViewModel { private let clientFactory: (APIConfiguration) -> any SybilAPIClienting private let fallbackModels: [Provider: [String]] = [ - .openai: ["gpt-4.1-mini"], + .openai: ["gpt-4.1-mini", "chat-latest"], .anthropic: ["claude-3-5-sonnet-latest"], .xai: ["grok-3-mini"], .gemini: ["gemini-3.5-flash", "gemini-flash-latest"], diff --git a/ios/Packages/Sybil/Tests/SybilTests/SybilTests.swift b/ios/Packages/Sybil/Tests/SybilTests/SybilTests.swift index abbab3a..2898ae1 100644 --- a/ios/Packages/Sybil/Tests/SybilTests/SybilTests.swift +++ b/ios/Packages/Sybil/Tests/SybilTests/SybilTests.swift @@ -52,6 +52,7 @@ private actor MockSybilClient: SybilAPIClienting { private let updateChatStarResponses: [String: ChatSummary] private let updateSearchStarResponses: [String: SearchSummary] private let activeRunsResponse: ActiveRunsResponse + private let modelCatalogResponse: ModelCatalogResponse private var snapshot = MockClientCallSnapshot() private var lastCreateChatCall: ChatCreateCallSnapshot? @@ -84,7 +85,8 @@ private actor MockSybilClient: SybilAPIClienting { updateChatStarResponses: [String: ChatSummary] = [:], updateSearchStarResponses: [String: SearchSummary] = [:], activeRunsResponse: ActiveRunsResponse = ActiveRunsResponse(), - workspaceItemsResponse: [WorkspaceItem]? = nil + workspaceItemsResponse: [WorkspaceItem]? = nil, + modelCatalogResponse: ModelCatalogResponse = ModelCatalogResponse(providers: [:]) ) { self.chatsResponse = chatsResponse self.searchesResponse = searchesResponse @@ -98,6 +100,7 @@ private actor MockSybilClient: SybilAPIClienting { self.updateChatStarResponses = updateChatStarResponses self.updateSearchStarResponses = updateSearchStarResponses self.activeRunsResponse = activeRunsResponse + self.modelCatalogResponse = modelCatalogResponse } private static func makeWorkspaceItems(chats: [ChatSummary], searches: [SearchSummary]) -> [WorkspaceItem] { @@ -306,7 +309,7 @@ private actor MockSybilClient: SybilAPIClienting { } func listModels() async throws -> ModelCatalogResponse { - ModelCatalogResponse(providers: [:]) + modelCatalogResponse } func getActiveRuns() async throws -> ActiveRunsResponse { @@ -1257,6 +1260,7 @@ private func makeCompletionToolCall( #expect(snapshot.runQuickQuestionStream == 1) #expect(snapshot.runCompletionStream == 0) #expect(body?.provider == .openai) + #expect(body?.model == "chat-latest") #expect(body?.question == "How do I reset my password?") #expect(viewModel.quickQuestionAnswerText == "Reset it from Settings.") #expect(!viewModel.isQuickQuestionSending) @@ -1370,6 +1374,42 @@ private func makeCompletionToolCall( #expect(viewModel.quickQuestionPrompt.isEmpty) } +@MainActor +@Test func quickQuestionDefaultsToChatLatestIndependentlyOfRegularChatPreferences() async { + let defaults = UserDefaults(suiteName: #function)! + defaults.removePersistentDomain(forName: #function) + defer { defaults.removePersistentDomain(forName: #function) } + defaults.set("anthropic", forKey: "sybil.ios.preferredProvider") + defaults.set("gpt-4o", forKey: "sybil.ios.preferredOpenAIModel") + let settings = SybilSettingsStore(defaults: defaults) + let client = MockSybilClient(modelCatalogResponse: ModelCatalogResponse(providers: [ + .openai: ProviderModelInfo(models: ["gpt-4.1-mini", "chat-latest"], loadedAt: nil, error: nil), + .anthropic: ProviderModelInfo(models: ["claude-3-5-sonnet-latest"], loadedAt: nil, error: nil) + ])) + let viewModel = SybilViewModel(settings: settings) { _ in client } + + #expect(viewModel.provider == .anthropic) + #expect(settings.preferredModelByProvider[.openai] == "gpt-4o") + #expect(viewModel.quickQuestionProvider == .openai) + #expect(viewModel.quickQuestionModel == "chat-latest") + #expect(viewModel.quickQuestionProviderModelOptions.contains("chat-latest")) + + viewModel.setQuickQuestionProvider(.anthropic) + viewModel.setQuickQuestionProvider(.openai) + #expect(viewModel.quickQuestionModel == "chat-latest") + + await viewModel.bootstrap() + #expect(viewModel.quickQuestionModel == "chat-latest") + #expect(viewModel.quickQuestionProviderModelOptions == ["gpt-4.1-mini", "chat-latest"]) + + viewModel.setQuickQuestionProvider(.anthropic) + viewModel.setQuickQuestionProvider(.openai) + #expect(viewModel.quickQuestionModel == "chat-latest") + #expect(SybilSettingsStore(defaults: defaults).quickQuestionPreferredModelByProvider[.openai] == "chat-latest") + #expect(settings.preferredProvider == .anthropic) + #expect(settings.preferredModelByProvider[.openai] == "gpt-4o") +} + @MainActor @Test func quickQuestionProviderAndModelSelectionPersistSeparately() async throws { let defaults = UserDefaults(suiteName: #function)! diff --git a/ios/README.md b/ios/README.md index ad5001e..9b79dbc 100644 --- a/ios/README.md +++ b/ios/README.md @@ -16,6 +16,9 @@ second **Option+Space**, the close button, or clicking outside dismisses it. Hiding the panel preserves the draft and answer and lets an in-flight answer finish. **Open in Chat** saves the question and answer in the regular workspace. +Quick Question defaults to OpenAI `chat-latest` on iPhone, iPad, and Mac. Its saved +provider/model selection takes precedence and is separate from regular chat preferences. + The regular Mac chat/search composer uses the same Return/Shift+Return rules. Holding Return does not repeatedly submit. On iOS, Return submits from Quick Question, while the normal chat/search composer continues to use Return for a diff --git a/server/src/llm/provider-adapters.ts b/server/src/llm/provider-adapters.ts index d2d8057..aab82d9 100644 --- a/server/src/llm/provider-adapters.ts +++ b/server/src/llm/provider-adapters.ts @@ -95,7 +95,7 @@ function isLikelyResponsesApiModel(model: string) { if (id.includes("audio") || id.includes("realtime") || id.includes("transcribe") || id.includes("tts")) return false; if (id.includes("image") || id.includes("dall-e") || id.includes("sora")) return false; if (id.includes("search") || id.includes("computer-use")) return false; - return /^(gpt-|o\d|chatgpt-)/.test(id); + return id === "chat-latest" || /^(gpt-|o\d|chatgpt-)/.test(id); } function isLikelyGeminiChatModel(model: string) { diff --git a/server/tests/provider-adapters.test.ts b/server/tests/provider-adapters.test.ts index e48a519..94accfd 100644 --- a/server/tests/provider-adapters.test.ts +++ b/server/tests/provider-adapters.test.ts @@ -1,6 +1,50 @@ import assert from "node:assert/strict"; import test from "node:test"; -import { describeProviderChatBackend } from "../src/llm/provider-adapters.js"; +import { env } from "../src/env.js"; +import { describeProviderChatBackend, fetchProviderCatalogModels } from "../src/llm/provider-adapters.js"; + +for (const { name, listedModels, expectedModels } of [ + { + name: "OpenAI catalog includes chat-latest while excluding non-chat models", + listedModels: [ + "gpt-4.1-mini", + "chat-latest", + "o3", + "chatgpt-4o-latest", + "chat-latest", + "text-embedding-3-small", + "omni-moderation-latest", + "gpt-4o-audio-preview", + "gpt-realtime", + "gpt-4o-transcribe", + "gpt-4o-mini-tts", + "gpt-image-1", + "chatgpt-image-latest", + "gpt-4o-search-preview", + "computer-use-preview", + "sora-2", + ], + expectedModels: ["chat-latest", "chatgpt-4o-latest", "gpt-4.1-mini", "o3"], + }, + { + name: "OpenAI catalog does not invent models absent from the provider response", + listedModels: ["gpt-4.1-mini"], + expectedModels: ["gpt-4.1-mini"], + }, +]) { + test(name, async (t) => { + const originalKey = env.OPENAI_API_KEY; + env.OPENAI_API_KEY = "test-openai-key"; + t.after(() => { env.OPENAI_API_KEY = originalKey; }); + t.mock.method(globalThis, "fetch", async (input: RequestInfo | URL) => { + const url = input instanceof Request ? input.url : String(input); + assert.equal(url, "https://api.openai.com/v1/models"); + return Response.json({ data: listedModels.map((id) => ({ id })) }); + }); + + assert.deepEqual(await fetchProviderCatalogModels("openai"), expectedModels); + }); +} test("provider backend registry selects chat protocol and managed-tool mode", () => { assert.deepEqual(describeProviderChatBackend("openai", []), { diff --git a/web/src/App.tsx b/web/src/App.tsx index 5129d80..97a02f6 100644 --- a/web/src/App.tsx +++ b/web/src/App.tsx @@ -80,7 +80,7 @@ import { resolveSidebarSelectionAfterRefresh, type SidebarSelection, } from "@/lib/sidebar-selection"; -import { buildQuickQuestionRequest } from "@/lib/quick-question"; +import { buildQuickQuestionRequest, pickQuickQuestionModel, QUICK_QUESTION_DEFAULT_MODEL } from "@/lib/quick-question"; import { appendStreamingDelta, clearStreamingAssistantTrace, @@ -161,7 +161,7 @@ function buildWorkspaceUrl(selection: SidebarSelection | null) { } const PROVIDER_FALLBACK_MODELS: Record = { - openai: ["gpt-4.1-mini"], + openai: ["gpt-4.1-mini", QUICK_QUESTION_DEFAULT_MODEL], anthropic: ["claude-3-5-sonnet-latest"], xai: ["grok-3-mini"], gemini: ["gemini-3.5-flash", "gemini-flash-latest"], @@ -991,7 +991,7 @@ export default function App() { ); const [quickModel, setQuickModel] = useState(() => { const stored = loadStoredQuickQuestionModelSelection(); - return stored.modelPreferences[stored.provider] ?? PROVIDER_FALLBACK_MODELS[stored.provider][0]; + return pickQuickQuestionModel(stored.provider, PROVIDER_FALLBACK_MODELS[stored.provider], stored.modelPreferences[stored.provider]); }); const [isQuickQuestionOpen, setIsQuickQuestionOpen] = useState(false); const [quickPrompt, setQuickPrompt] = useState(""); @@ -1449,7 +1449,7 @@ export default function App() { useEffect(() => { if (quickModel.trim()) return; setQuickModel((current) => { - return current.trim() || pickProviderModel(quickProviderModelOptions, quickProviderModelPreferences[quickProvider]); + return current.trim() || pickQuickQuestionModel(quickProvider, quickProviderModelOptions, quickProviderModelPreferences[quickProvider]); }); }, [quickModel, quickProvider, quickProviderModelOptions, quickProviderModelPreferences]); @@ -4176,7 +4176,7 @@ export default function App() { const nextProvider = event.currentTarget.value as Provider; setQuickProvider(nextProvider); const options = getModelOptions(modelCatalog, nextProvider); - setQuickModel(pickProviderModel(options, quickProviderModelPreferences[nextProvider])); + setQuickModel(pickQuickQuestionModel(nextProvider, options, quickProviderModelPreferences[nextProvider])); }} disabled={isQuickQuestionSending || isConvertingQuickQuestion} aria-label="Quick question provider" diff --git a/web/src/lib/quick-question.ts b/web/src/lib/quick-question.ts index 40944ca..ebebb61 100644 --- a/web/src/lib/quick-question.ts +++ b/web/src/lib/quick-question.ts @@ -1,5 +1,13 @@ import type { Provider } from "./api"; +export const QUICK_QUESTION_DEFAULT_MODEL = "chat-latest"; + +export function pickQuickQuestionModel(provider: Provider, options: string[], preferred: string | null = null) { + if (preferred?.trim()) return preferred.trim(); + if (provider === "openai") return QUICK_QUESTION_DEFAULT_MODEL; + return options[0] ?? ""; +} + export function buildQuickQuestionRequest({ provider, model, diff --git a/web/tests/quick-question.test.mjs b/web/tests/quick-question.test.mjs index 4b1beab..3a6c032 100644 --- a/web/tests/quick-question.test.mjs +++ b/web/tests/quick-question.test.mjs @@ -1,6 +1,27 @@ import assert from "node:assert/strict"; import test from "node:test"; -import { buildQuickQuestionRequest } from "../src/lib/quick-question.ts"; +import { buildQuickQuestionRequest, pickQuickQuestionModel } from "../src/lib/quick-question.ts"; + +test("quick questions default to chat-latest before and after the model catalog loads", () => { + for (const options of [[], ["gpt-4.1-mini"], ["gpt-4.1-mini", "chat-latest"]]) { + const model = pickQuickQuestionModel("openai", options); + assert.equal(model, "chat-latest"); + assert.deepEqual(buildQuickQuestionRequest({ provider: "openai", model, content: "What is UTC?" }), { + provider: "openai", + model: "chat-latest", + question: "What is UTC?", + }); + } +}); + +test("quick questions preserve saved models and use provider-specific choices when switching providers", () => { + assert.equal(pickQuickQuestionModel("openai", ["chat-latest"], " gpt-4.1-mini "), "gpt-4.1-mini"); + assert.equal(pickQuickQuestionModel("openai", ["gpt-4.1-mini"], " "), "chat-latest"); + assert.equal(pickQuickQuestionModel("anthropic", ["claude-3-5-sonnet-latest"]), "claude-3-5-sonnet-latest"); + assert.equal(pickQuickQuestionModel("anthropic", ["claude-3-5-sonnet-latest"], "claude-3-haiku"), "claude-3-haiku"); + assert.equal(pickQuickQuestionModel("anthropic", []), ""); + assert.equal(pickQuickQuestionModel("openai", ["gpt-4.1-mini", "chat-latest"]), "chat-latest"); +}); test("quick question requests use the dedicated server endpoint shape", () => { const request = buildQuickQuestionRequest({