Compare commits

...
Author SHA1 Message Date
buzzert 048456b8a5 Preserve draft views during collection refreshes 2026-07-25 17:14:39 -07:00
buzzert 211fad972a avoid mobile focus on conversation selection 2026-07-23 20:29:53 -07:00
buzzert a69c641481 add mobile workspace gestures 2026-07-23 18:05:57 -07:00
buzzert 9229896ad7 harden PWA resume and title fallback 2026-07-23 17:29:02 -07:00
buzzert c5217b2710 make PWA chat resume idempotent 2026-07-23 17:12:11 -07:00
buzzert abc1124d27 fix for PWA background/foreground 2026-07-23 16:32:20 -07:00
buzzert 4721717022 server: respect Brave search rate limits 2026-07-19 21:56:35 -07:00
buzzert 87b7d9502f server: add Brave search provider 2026-07-19 21:35:47 -07:00
buzzert 1952f4f358 web: open links in chat in new tab 2026-07-19 16:08:38 -07:00
buzzert 622659f6ca ios: preserve chat scroll position on resume
TestFlight / testflight (push) Successful in 1m46s
2026-07-12 13:57:45 -07:00
buzzert 93ca8a76c3 adds gemini support
TestFlight / testflight (push) Successful in 1m53s
2026-07-11 14:16:21 -07:00
buzzertandClaude Fable 5 69f50064a3 ci: restore Ruby PATH export for fastlane step
TestFlight / testflight (push) Successful in 1m58s
act_runner does not carry setup-ruby's PATH changes into later steps,
so without this the step runs the toolcache default Ruby (3.3.11) and
bundler cannot find the gems installed for 3.1.7.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-11 13:04:47 -07:00
35 changed files with 2119 additions and 397 deletions
+6 -1
View File
@@ -36,4 +36,9 @@ jobs:
SYBIL_BUILD_NUMBER: ${{ github.run_number }} SYBIL_BUILD_NUMBER: ${{ github.run_number }}
FASTLANE_SKIP_UPDATE_CHECK: "1" FASTLANE_SKIP_UPDATE_CHECK: "1"
FASTLANE_XCODEBUILD_SETTINGS_TIMEOUT: "120" FASTLANE_XCODEBUILD_SETTINGS_TIMEOUT: "120"
run: bundle exec fastlane ios beta # act_runner does not propagate setup-ruby's PATH changes into later
# steps, so put the selected Ruby back on PATH or `bundle` resolves to
# the toolcache default and misses the gems installed above.
run: |
export PATH="/Users/runner/hostedtoolcache/Ruby/3.1.7/arm64/bin:${PATH}"
bundle exec fastlane ios beta
+26
View File
@@ -12,17 +12,43 @@ server {
location /api/ { location /api/ {
proxy_pass http://server:8787/; proxy_pass http://server:8787/;
proxy_http_version 1.1; proxy_http_version 1.1;
proxy_buffering off;
proxy_cache off;
proxy_read_timeout 3600s;
proxy_send_timeout 3600s;
proxy_set_header Host $host; proxy_set_header Host $host;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme; proxy_set_header X-Forwarded-Proto $scheme;
proxy_set_header Connection "";
}
location = /sw.js {
add_header Cache-Control "no-store, no-cache, must-revalidate" always;
expires -1;
try_files $uri =404;
} }
location = /manifest.webmanifest { location = /manifest.webmanifest {
default_type application/manifest+json; default_type application/manifest+json;
add_header Cache-Control "no-store, no-cache, must-revalidate" always;
expires -1;
try_files $uri =404;
}
location = /index.html {
add_header Cache-Control "no-store, no-cache, must-revalidate" always;
expires -1;
try_files $uri =404;
}
location /assets/ {
add_header Cache-Control "public, max-age=31536000, immutable" always;
try_files $uri =404; try_files $uri =404;
} }
location / { location / {
add_header Cache-Control "no-store, no-cache, must-revalidate" always;
expires -1;
try_files $uri $uri/ /index.html; try_files $uri $uri/ /index.html;
} }
} }
+2
View File
@@ -12,10 +12,12 @@ services:
OPENAI_API_KEY: ${OPENAI_API_KEY:-} OPENAI_API_KEY: ${OPENAI_API_KEY:-}
ANTHROPIC_API_KEY: ${ANTHROPIC_API_KEY:-} ANTHROPIC_API_KEY: ${ANTHROPIC_API_KEY:-}
XAI_API_KEY: ${XAI_API_KEY:-} XAI_API_KEY: ${XAI_API_KEY:-}
GEMINI_API_KEY: ${GEMINI_API_KEY:-}
HERMES_AGENT_API_BASE_URL: ${HERMES_AGENT_API_BASE_URL:-http://127.0.0.1:8642/v1} HERMES_AGENT_API_BASE_URL: ${HERMES_AGENT_API_BASE_URL:-http://127.0.0.1:8642/v1}
HERMES_AGENT_API_KEY: ${HERMES_AGENT_API_KEY:-} HERMES_AGENT_API_KEY: ${HERMES_AGENT_API_KEY:-}
HERMES_AGENT_MODEL: ${HERMES_AGENT_MODEL:-} HERMES_AGENT_MODEL: ${HERMES_AGENT_MODEL:-}
EXA_API_KEY: ${EXA_API_KEY:-} EXA_API_KEY: ${EXA_API_KEY:-}
BRAVE_SEARCH_API_KEY: ${BRAVE_SEARCH_API_KEY:-}
CHAT_WEB_SEARCH_ENGINE: ${CHAT_WEB_SEARCH_ENGINE:-exa} CHAT_WEB_SEARCH_ENGINE: ${CHAT_WEB_SEARCH_ENGINE:-exa}
SEARXNG_BASE_URL: ${SEARXNG_BASE_URL:-} SEARXNG_BASE_URL: ${SEARXNG_BASE_URL:-}
CHAT_MAX_TOOL_ROUNDS: ${CHAT_MAX_TOOL_ROUNDS:-100} CHAT_MAX_TOOL_ROUNDS: ${CHAT_MAX_TOOL_ROUNDS:-100}
+25 -9
View File
@@ -34,11 +34,13 @@ Chat upload limits:
"openai": { "models": ["gpt-4.1-mini"], "loadedAt": "2026-02-14T00:00:00.000Z", "error": null }, "openai": { "models": ["gpt-4.1-mini"], "loadedAt": "2026-02-14T00:00:00.000Z", "error": null },
"anthropic": { "models": ["claude-3-5-sonnet-latest"], "loadedAt": null, "error": null }, "anthropic": { "models": ["claude-3-5-sonnet-latest"], "loadedAt": null, "error": null },
"xai": { "models": ["grok-3-mini"], "loadedAt": null, "error": null }, "xai": { "models": ["grok-3-mini"], "loadedAt": null, "error": null },
"gemini": { "models": ["gemini-3.5-flash"], "loadedAt": null, "error": null },
"hermes-agent": { "models": ["hermes-agent"], "loadedAt": null, "error": null } "hermes-agent": { "models": ["hermes-agent"], "loadedAt": null, "error": null }
} }
} }
``` ```
- OpenAI model lists are filtered to models that are expected to work with the backend's Responses API implementation. - OpenAI model lists are filtered to models that are expected to work with the backend's Responses API implementation.
- Gemini model lists are loaded from Google's native Models API and filtered to Gemini `generateContent` model ids.
- `hermes-agent` is included only when `HERMES_AGENT_API_KEY` is configured. Set it to Hermes `API_SERVER_KEY`, or any non-empty value if that local server does not require auth. `HERMES_AGENT_API_BASE_URL` defaults to `http://127.0.0.1:8642/v1`; set `HERMES_AGENT_MODEL` only when you need an additional fallback/override model id. - `hermes-agent` is included only when `HERMES_AGENT_API_KEY` is configured. Set it to Hermes `API_SERVER_KEY`, or any non-empty value if that local server does not require auth. `HERMES_AGENT_API_BASE_URL` defaults to `http://127.0.0.1:8642/v1`; set `HERMES_AGENT_MODEL` only when you need an additional fallback/override model id.
- The backend loads provider model lists at startup and refreshes them about once every 24 hours. If a later provider refresh fails, the response keeps the last loaded model list for that provider and sets `error` to the latest failure message. - The backend loads provider model lists at startup and refreshes them about once every 24 hours. If a later provider refresh fails, the response keeps the last loaded model list for that provider and sets `error` to the latest failure message.
@@ -56,7 +58,7 @@ Chat upload limits:
``` ```
Behavior notes: Behavior notes:
- Lists Sybil-managed chat tools that can be enabled for `openai`, `anthropic`, and `xai` chat completions. - Lists Sybil-managed chat tools that can be enabled for `openai`, `anthropic`, `xai`, and `gemini` chat completions.
- Optional tools such as `codex_exec` and `shell_exec` appear only when enabled by server environment configuration. - Optional tools such as `codex_exec` and `shell_exec` appear only when enabled by server environment configuration.
## Active Runs ## Active Runs
@@ -128,7 +130,7 @@ Behavior notes:
```json ```json
{ {
"title": "optional title", "title": "optional title",
"provider": "optional openai|anthropic|xai|hermes-agent", "provider": "optional openai|anthropic|xai|gemini|hermes-agent",
"model": "optional model id", "model": "optional model id",
"additionalSystemPrompt": "optional stored system prompt", "additionalSystemPrompt": "optional stored system prompt",
"enabledTools": ["web_search", "fetch_url"], "enabledTools": ["web_search", "fetch_url"],
@@ -184,6 +186,7 @@ Behavior notes:
- If the chat already has a non-empty title, server returns the existing chat unchanged. - If the chat already has a non-empty title, server returns the existing chat unchanged.
- If a title is set while suggestion generation is in flight, server returns the current chat instead of overwriting that title. - If a title is set while suggestion generation is in flight, server returns the current chat instead of overwriting that title.
- When no title exists at write time, server uses OpenAI `gpt-4.1-mini` to generate a one-line title (up to ~4 words), updates the chat title, and returns the updated chat. - When no title exists at write time, server uses OpenAI `gpt-4.1-mini` to generate a one-line title (up to ~4 words), updates the chat title, and returns the updated chat.
- If the title provider is unavailable or rejects the request, server still persists a deterministic title derived from the first line of `content` instead of leaving the chat untitled.
### `DELETE /v1/chats/:chatId` ### `DELETE /v1/chats/:chatId`
- Response: `{ "deleted": true }` - Response: `{ "deleted": true }`
@@ -234,7 +237,7 @@ Notes:
```json ```json
{ {
"chatId": "optional-chat-id", "chatId": "optional-chat-id",
"provider": "openai|anthropic|xai|hermes-agent", "provider": "openai|anthropic|xai|gemini|hermes-agent",
"model": "string", "model": "string",
"messages": [ "messages": [
{ {
@@ -294,13 +297,16 @@ Behavior notes:
- For `openai`, backend calls OpenAI's Responses API and enables internal tool use with an internal system instruction. - For `openai`, backend calls OpenAI's Responses API and enables internal tool use with an internal system instruction.
- For `anthropic`, backend calls Anthropic's Messages API and enables internal tool use with Anthropic `tool_use`/`tool_result` content blocks. - For `anthropic`, backend calls Anthropic's Messages API and enables internal tool use with Anthropic `tool_use`/`tool_result` content blocks.
- For `xai`, backend calls xAI's OpenAI-compatible Chat Completions API and enables internal tool use with the same internal system instruction. - For `xai`, backend calls xAI's OpenAI-compatible Chat Completions API and enables internal tool use with the same internal system instruction.
- For `gemini`, backend calls Google's native Gemini `generateContent` API and enables internal tool use with Gemini function calling.
- For `hermes-agent`, backend calls the configured Hermes Agent OpenAI-compatible Chat Completions API without adding Sybil-managed tool definitions; Hermes Agent handles its own tools server-side. - For `hermes-agent`, backend calls the configured Hermes Agent OpenAI-compatible Chat Completions API without adding Sybil-managed tool definitions; Hermes Agent handles its own tools server-side.
- For `openai`, image attachments are sent as Responses `input_image` items and text attachments are sent as `input_text` items. - For `openai`, image attachments are sent as Responses `input_image` items and text attachments are sent as `input_text` items.
- For `gemini`, image attachments are sent as native Gemini `inlineData` parts and text attachments are sent as text parts.
- For `xai` and `hermes-agent`, image attachments are sent as Chat Completions content parts alongside text. - For `xai` and `hermes-agent`, image attachments are sent as Chat Completions content parts alongside text.
- For `openai`, Responses calls that can enter the server-managed tool loop use `store: true` so reasoning and function-call items can be passed between tool rounds. - For `openai`, Responses calls that can enter the server-managed tool loop use `store: true` so reasoning and function-call items can be passed between tool rounds.
- For `anthropic`, image attachments are sent as Messages API `image` blocks using base64 source data; text attachments are added as `text` blocks. - For `anthropic`, image attachments are sent as Messages API `image` blocks using base64 source data; text attachments are added as `text` blocks.
- Available Sybil-managed tool calls for `openai`, `anthropic`, and `xai`: `web_search` and `fetch_url`. When `CHAT_CODEX_TOOL_ENABLED=true`, `codex_exec` is also available. When `CHAT_SHELL_TOOL_ENABLED=true`, `shell_exec` is also available. - Available Sybil-managed tool calls for `openai`, `anthropic`, `xai`, and `gemini`: `web_search` and `fetch_url`. When `CHAT_CODEX_TOOL_ENABLED=true`, `codex_exec` is also available. When `CHAT_SHELL_TOOL_ENABLED=true`, `shell_exec` is also available.
- `web_search` returns ranked results with per-result summaries/snippets. Its backend engine is selected by `CHAT_WEB_SEARCH_ENGINE` (`exa` default, or `searxng` with `SEARXNG_BASE_URL` set). SearXNG mode requires the instance to allow `format=json`. - `web_search` returns ranked results with per-result summaries/snippets. Its backend engine is selected by `CHAT_WEB_SEARCH_ENGINE`: `exa` (default), `brave` (requires `BRAVE_SEARCH_API_KEY`), or `searxng` (requires `SEARXNG_BASE_URL`; the instance must allow `format=json`).
- Brave searches are queued and evenly paced according to the shortest window in Brave's `X-RateLimit-Policy` response header. The backend also honors `X-RateLimit-Remaining`/`X-RateLimit-Reset` and retries `429` responses up to three times with reset-aware exponential backoff; quota resets beyond the bounded retry window fail immediately.
- `fetch_url` fetches a URL with browser-like navigation headers and returns plaintext page content (HTML converted to text server-side). - `fetch_url` fetches a URL with browser-like navigation headers and returns plaintext page content (HTML converted to text server-side).
- `codex_exec` delegates coding, shell, repository inspection, and other complex software tasks to a persistent remote Codex CLI workspace over SSH. The server runs `codex exec --dangerously-bypass-approvals-and-sandbox --skip-git-repo-check <non-interactive wrapped prompt>` on the configured devbox inside `CHAT_CODEX_REMOTE_WORKDIR`, with SSH stdin closed. - `codex_exec` delegates coding, shell, repository inspection, and other complex software tasks to a persistent remote Codex CLI workspace over SSH. The server runs `codex exec --dangerously-bypass-approvals-and-sandbox --skip-git-repo-check <non-interactive wrapped prompt>` on the configured devbox inside `CHAT_CODEX_REMOTE_WORKDIR`, with SSH stdin closed.
- `shell_exec` runs arbitrary non-interactive shell commands on the same configured devbox, starting in `CHAT_CODEX_REMOTE_WORKDIR`. It uses `bash -lc` when bash exists, otherwise `sh -lc`, closes SSH stdin, and does not run inside the Sybil server container. - `shell_exec` runs arbitrary non-interactive shell commands on the same configured devbox, starting in `CHAT_CODEX_REMOTE_WORKDIR`. It uses `bash -lc` when bash exists, otherwise `sh -lc`, closes SSH stdin, and does not run inside the Sybil server container.
@@ -318,6 +324,16 @@ Behavior notes:
- `CHAT_SHELL_EXEC_TIMEOUT_MS=120000` (optional) - `CHAT_SHELL_EXEC_TIMEOUT_MS=120000` (optional)
- When a tool call is executed, backend stores a chat `Message` with `role: "tool"` and tool metadata (`metadata.kind = "tool_call"`). Streaming requests emit an initiated SSE `tool_call` event before execution, then persist each completed or failed tool call as its terminal SSE `tool_call` event is emitted, then store the assistant output when the completion finishes. - When a tool call is executed, backend stores a chat `Message` with `role: "tool"` and tool metadata (`metadata.kind = "tool_call"`). Streaming requests emit an initiated SSE `tool_call` event before execution, then persist each completed or failed tool call as its terminal SSE `tool_call` event is emitted, then store the assistant output when the completion finishes.
## Streaming Chat
### `POST /v1/chat-completions/stream`
- The request accepts the chat-completion fields above plus optional `persist` and `clientRequestId` fields.
- `clientRequestId` is only valid for a persisted request with a `chatId`, may be up to 128 characters, and should be a stable unique value generated once per user submission.
- Retrying with the same `chatId` and `clientRequestId` replays the matching active or completed stream rather than starting a duplicate provider call.
- The server persists the ID in `metadata.clientRequestId` on the submitted user message and completed assistant message.
- The complete request, SSE event, persistence, retry, and attach contracts are defined in `docs/api/streaming-chat.md`.
## Searches ## Searches
### `GET /v1/searches` ### `GET /v1/searches`
@@ -417,9 +433,9 @@ Behavior notes:
"updatedAt": "...", "updatedAt": "...",
"starred": false, "starred": false,
"starredAt": null, "starredAt": null,
"initiatedProvider": "openai|anthropic|xai|hermes-agent|null", "initiatedProvider": "openai|anthropic|xai|gemini|hermes-agent|null",
"initiatedModel": "string|null", "initiatedModel": "string|null",
"lastUsedProvider": "openai|anthropic|xai|hermes-agent|null", "lastUsedProvider": "openai|anthropic|xai|gemini|hermes-agent|null",
"lastUsedModel": "string|null", "lastUsedModel": "string|null",
"additionalSystemPrompt": null, "additionalSystemPrompt": null,
"enabledTools": ["web_search", "fetch_url"] "enabledTools": ["web_search", "fetch_url"]
@@ -469,9 +485,9 @@ Behavior notes:
"updatedAt": "...", "updatedAt": "...",
"starred": false, "starred": false,
"starredAt": null, "starredAt": null,
"initiatedProvider": "openai|anthropic|xai|hermes-agent|null", "initiatedProvider": "openai|anthropic|xai|gemini|hermes-agent|null",
"initiatedModel": "string|null", "initiatedModel": "string|null",
"lastUsedProvider": "openai|anthropic|xai|hermes-agent|null", "lastUsedProvider": "openai|anthropic|xai|gemini|hermes-agent|null",
"lastUsedModel": "string|null", "lastUsedModel": "string|null",
"additionalSystemPrompt": null, "additionalSystemPrompt": null,
"enabledTools": ["web_search", "fetch_url"], "enabledTools": ["web_search", "fetch_url"],
+11 -4
View File
@@ -21,7 +21,8 @@ Authentication:
{ {
"chatId": "optional-chat-id", "chatId": "optional-chat-id",
"persist": true, "persist": true,
"provider": "openai|anthropic|xai|hermes-agent", "clientRequestId": "optional-client-generated-id",
"provider": "openai|anthropic|xai|gemini|hermes-agent",
"model": "string", "model": "string",
"messages": [ "messages": [
{ {
@@ -61,6 +62,9 @@ Notes:
- If `persist` is `true` and `chatId` is omitted, backend creates a new chat. - If `persist` is `true` and `chatId` is omitted, backend creates a new chat.
- If `chatId` is provided, backend validates it exists. - If `chatId` is provided, backend validates it exists.
- If `persist` is `false`, `chatId` must be omitted. Backend does not create a chat and does not persist input messages, tool-call messages, assistant output, or `LlmCall` metadata. - If `persist` is `false`, `chatId` must be omitted. Backend does not create a chat and does not persist input messages, tool-call messages, assistant output, or `LlmCall` metadata.
- `clientRequestId` is optional and is only valid for a persisted stream with a `chatId`. Clients should generate one stable, unique value per user submission and reuse it when retrying a disconnected request.
- A retry with the same `chatId` and `clientRequestId` attaches to and replays the matching active stream. If that submission already completed, the endpoint replays `meta` and `done` without invoking the provider again. This makes retrying the initial streaming `POST` idempotent.
- `clientRequestId` values may be up to 128 characters. The server stores the value in `metadata.clientRequestId` on the submitted user message and completed assistant message.
- For persisted streams, backend stores only new non-assistant input history rows to avoid duplicates. - For persisted streams, backend stores only new non-assistant input history rows to avoid duplicates.
- `additionalSystemPrompt`, when present directly or loaded from stored chat settings, is prepended to the provider request as a `system` message and is not inserted into the persisted chat transcript by this endpoint. - `additionalSystemPrompt`, when present directly or loaded from stored chat settings, is prepended to the provider request as a `system` message and is not inserted into the persisted chat transcript by this endpoint.
- `enabledTools` limits Sybil-managed tools for this request. When omitted for a saved chat, the stored chat setting is used; otherwise all available tools are enabled by default. An empty array disables Sybil-managed tools. - `enabledTools` limits Sybil-managed tools for this request. When omitted for a saved chat, the stored chat setting is used; otherwise all available tools are enabled by default. An empty array disables Sybil-managed tools.
@@ -70,7 +74,7 @@ Notes:
Persisted chat streams with a `chatId` are backend-owned active runs: Persisted chat streams with a `chatId` are backend-owned active runs:
- Once started, the backend keeps the stream running even if the HTTP client disconnects or refreshes. - Once started, the backend keeps the stream running even if the HTTP client disconnects or refreshes.
- While running, `GET /v1/active-runs` includes the `chatId`. - While running, `GET /v1/active-runs` includes the `chatId`.
- Starting a second persisted stream for the same active `chatId` returns `409`. - Starting a second persisted stream for the same active `chatId` returns `409`, unless its `clientRequestId` matches the active submission, in which case the existing stream is replayed.
- Clients can reattach with `POST /v1/chats/:chatId/stream/attach`. - Clients can reattach with `POST /v1/chats/:chatId/stream/attach`.
## Attach Endpoint ## Attach Endpoint
@@ -174,18 +178,21 @@ Terminal tool-call event:
- `openai`: backend uses OpenAI's Responses API and may execute internal function tool calls (`web_search`, `fetch_url`, optional `codex_exec`, and optional `shell_exec`) before producing final text. - `openai`: backend uses OpenAI's Responses API and may execute internal function tool calls (`web_search`, `fetch_url`, optional `codex_exec`, and optional `shell_exec`) before producing final text.
- `anthropic`: backend uses Anthropic's Messages API and may execute the same internal tools with `tool_use`/`tool_result` content blocks before producing final text. - `anthropic`: backend uses Anthropic's Messages API and may execute the same internal tools with `tool_use`/`tool_result` content blocks before producing final text.
- `xai`: backend uses xAI's OpenAI-compatible Chat Completions API and may execute the same internal tool calls before producing final text. - `xai`: backend uses xAI's OpenAI-compatible Chat Completions API and may execute the same internal tool calls before producing final text.
- `gemini`: backend uses Google's native Gemini `streamGenerateContent` API and may execute the same internal tool calls before producing final text.
- `fetch_url` sends browser-like navigation headers for outbound URL requests to reduce false 403s from sites that reject generic server clients. - `fetch_url` sends browser-like navigation headers for outbound URL requests to reduce false 403s from sites that reject generic server clients.
- `hermes-agent`: backend uses the configured Hermes Agent OpenAI-compatible Chat Completions API. Sybil does not add its own tool definitions for this provider; Hermes Agent handles its own tools server-side. Custom Hermes stream events are normalized away unless they produce text deltas in this SSE contract. - `hermes-agent`: backend uses the configured Hermes Agent OpenAI-compatible Chat Completions API. Sybil does not add its own tool definitions for this provider; Hermes Agent handles its own tools server-side. Custom Hermes stream events are normalized away unless they produce text deltas in this SSE contract.
- `openai`: image attachments are sent as Responses `input_image` items; text attachments are sent as `input_text` items. - `openai`: image attachments are sent as Responses `input_image` items; text attachments are sent as `input_text` items.
- `gemini`: image attachments are sent as native Gemini `inlineData` parts; text attachments are inlined as text parts.
- `xai` and `hermes-agent`: image attachments are sent as Chat Completions content parts; text attachments are inlined as text parts. - `xai` and `hermes-agent`: image attachments are sent as Chat Completions content parts; text attachments are inlined as text parts.
- `openai`: Responses calls that can enter the server-managed tool loop use `store: true` so reasoning and function-call items can be passed between tool rounds. - `openai`: Responses calls that can enter the server-managed tool loop use `store: true` so reasoning and function-call items can be passed between tool rounds.
- `anthropic`: streamed via event stream; emits `delta` from `content_block_delta` with `text_delta`, and emits normalized `tool_call` SSE events when Anthropic `tool_use` blocks are executed. Image attachments are sent as base64 `image` blocks and text attachments are appended as `text` blocks. - `anthropic`: streamed via event stream; emits `delta` from `content_block_delta` with `text_delta`, and emits normalized `tool_call` SSE events when Anthropic `tool_use` blocks are executed. Image attachments are sent as base64 `image` blocks and text attachments are appended as `text` blocks.
- `web_search` uses `CHAT_WEB_SEARCH_ENGINE` (`exa` default, or `searxng` with `SEARXNG_BASE_URL` set). SearXNG mode requires the instance to allow `format=json`. This only affects chat-mode tool calls, not search-mode endpoints. - `web_search` uses `CHAT_WEB_SEARCH_ENGINE`: `exa` (default), `brave` (requires `BRAVE_SEARCH_API_KEY`), or `searxng` (requires `SEARXNG_BASE_URL`; the instance must allow `format=json`). This only affects chat-mode tool calls, not search-mode endpoints.
- Brave searches are queued and evenly paced according to the shortest window in Brave's `X-RateLimit-Policy` response header. The backend also honors `X-RateLimit-Remaining`/`X-RateLimit-Reset` and retries `429` responses up to three times with reset-aware exponential backoff; quota resets beyond the bounded retry window fail immediately.
- `codex_exec` is available only when `CHAT_CODEX_TOOL_ENABLED=true`. It SSHes to `CHAT_CODEX_REMOTE_HOST`, creates/uses `CHAT_CODEX_REMOTE_WORKDIR`, and runs `codex exec --dangerously-bypass-approvals-and-sandbox --skip-git-repo-check <non-interactive wrapped prompt>` there with SSH stdin closed. Prefer `CHAT_CODEX_SSH_KEY_PATH` with a read-only mounted private key; `CHAT_CODEX_SSH_PRIVATE_KEY_B64` is also supported. - `codex_exec` is available only when `CHAT_CODEX_TOOL_ENABLED=true`. It SSHes to `CHAT_CODEX_REMOTE_HOST`, creates/uses `CHAT_CODEX_REMOTE_WORKDIR`, and runs `codex exec --dangerously-bypass-approvals-and-sandbox --skip-git-repo-check <non-interactive wrapped prompt>` there with SSH stdin closed. Prefer `CHAT_CODEX_SSH_KEY_PATH` with a read-only mounted private key; `CHAT_CODEX_SSH_PRIVATE_KEY_B64` is also supported.
- `shell_exec` is available only when `CHAT_SHELL_TOOL_ENABLED=true`. It uses the same devbox SSH configuration, starts in `CHAT_CODEX_REMOTE_WORKDIR`, and runs non-interactive shell commands there with SSH stdin closed, not inside the Sybil server container. - `shell_exec` is available only when `CHAT_SHELL_TOOL_ENABLED=true`. It uses the same devbox SSH configuration, starts in `CHAT_CODEX_REMOTE_WORKDIR`, and runs non-interactive shell commands there with SSH stdin closed, not inside the Sybil server container.
- `CHAT_MAX_TOOL_ROUNDS` controls how many model/tool result cycles may occur before the backend returns a tool-call limit message; default is 100. - `CHAT_MAX_TOOL_ROUNDS` controls how many model/tool result cycles may occur before the backend returns a tool-call limit message; default is 100.
Tool-enabled streaming notes (`openai`/`anthropic`/`xai`): Tool-enabled streaming notes (`openai`/`anthropic`/`xai`/`gemini`):
- Stream still emits standard `meta`, `delta`, `done|error` events. - Stream still emits standard `meta`, `delta`, `done|error` events.
- Stream may emit `tool_call` events while tool calls are executed. - Stream may emit `tool_call` events while tool calls are executed.
- `delta` events carry assistant text and are emitted incrementally for normal text rounds. The backend may buffer model-native text briefly while determining whether a provider round contains tool calls. - `delta` events carry assistant text and are emitted incrementally for normal text rounds. The backend may buffer model-native text briefly while determining whether a provider round contains tool calls.
+1
View File
@@ -57,4 +57,5 @@ Instructions for work under `/Users/buzzert/src/sybil-2/ios`.
- OpenAI: `gpt-4.1-mini` - OpenAI: `gpt-4.1-mini`
- Anthropic: `claude-3-5-sonnet-latest` - Anthropic: `claude-3-5-sonnet-latest`
- xAI: `grok-3-mini` - xAI: `grok-3-mini`
- Gemini: `gemini-3.5-flash`
- Hermes Agent: `hermes-agent` - Hermes Agent: `hermes-agent`
@@ -67,7 +67,6 @@ struct SybilChatTranscriptView: View {
.scrollDismissesKeyboard(.interactively) .scrollDismissesKeyboard(.interactively)
.onAppear { .onAppear {
syncKnownToolCallMessageIDs() syncKnownToolCallMessageIDs()
scrollToBottom(with: proxy, animated: false)
} }
.onChange(of: toolCallMessageIDSignature) { _, _ in .onChange(of: toolCallMessageIDSignature) { _, _ in
syncKnownToolCallMessageIDs() syncKnownToolCallMessageIDs()
@@ -4,6 +4,7 @@ public enum Provider: String, Codable, CaseIterable, Hashable, Sendable {
case openai case openai
case anthropic case anthropic
case xai case xai
case gemini
case hermesAgent = "hermes-agent" case hermesAgent = "hermes-agent"
public var displayName: String { public var displayName: String {
@@ -11,6 +12,7 @@ public enum Provider: String, Codable, CaseIterable, Hashable, Sendable {
case .openai: return "OpenAI" case .openai: return "OpenAI"
case .anthropic: return "Anthropic" case .anthropic: return "Anthropic"
case .xai: return "xAI" case .xai: return "xAI"
case .gemini: return "Gemini"
case .hermesAgent: return "Hermes Agent" case .hermesAgent: return "Hermes Agent"
} }
} }
@@ -11,11 +11,13 @@ final class SybilSettingsStore {
static let preferredOpenAIModel = "sybil.ios.preferredOpenAIModel" static let preferredOpenAIModel = "sybil.ios.preferredOpenAIModel"
static let preferredAnthropicModel = "sybil.ios.preferredAnthropicModel" static let preferredAnthropicModel = "sybil.ios.preferredAnthropicModel"
static let preferredXAIModel = "sybil.ios.preferredXAIModel" static let preferredXAIModel = "sybil.ios.preferredXAIModel"
static let preferredGeminiModel = "sybil.ios.preferredGeminiModel"
static let preferredHermesAgentModel = "sybil.ios.preferredHermesAgentModel" static let preferredHermesAgentModel = "sybil.ios.preferredHermesAgentModel"
static let quickQuestionPreferredProvider = "sybil.ios.quickQuestionPreferredProvider" static let quickQuestionPreferredProvider = "sybil.ios.quickQuestionPreferredProvider"
static let quickQuestionPreferredOpenAIModel = "sybil.ios.quickQuestionPreferredOpenAIModel" static let quickQuestionPreferredOpenAIModel = "sybil.ios.quickQuestionPreferredOpenAIModel"
static let quickQuestionPreferredAnthropicModel = "sybil.ios.quickQuestionPreferredAnthropicModel" static let quickQuestionPreferredAnthropicModel = "sybil.ios.quickQuestionPreferredAnthropicModel"
static let quickQuestionPreferredXAIModel = "sybil.ios.quickQuestionPreferredXAIModel" static let quickQuestionPreferredXAIModel = "sybil.ios.quickQuestionPreferredXAIModel"
static let quickQuestionPreferredGeminiModel = "sybil.ios.quickQuestionPreferredGeminiModel"
static let quickQuestionPreferredHermesAgentModel = "sybil.ios.quickQuestionPreferredHermesAgentModel" static let quickQuestionPreferredHermesAgentModel = "sybil.ios.quickQuestionPreferredHermesAgentModel"
} }
@@ -44,6 +46,7 @@ final class SybilSettingsStore {
.openai: defaults.string(forKey: Keys.preferredOpenAIModel) ?? "gpt-4.1-mini", .openai: defaults.string(forKey: Keys.preferredOpenAIModel) ?? "gpt-4.1-mini",
.anthropic: defaults.string(forKey: Keys.preferredAnthropicModel) ?? "claude-3-5-sonnet-latest", .anthropic: defaults.string(forKey: Keys.preferredAnthropicModel) ?? "claude-3-5-sonnet-latest",
.xai: defaults.string(forKey: Keys.preferredXAIModel) ?? "grok-3-mini", .xai: defaults.string(forKey: Keys.preferredXAIModel) ?? "grok-3-mini",
.gemini: defaults.string(forKey: Keys.preferredGeminiModel) ?? "gemini-3.5-flash",
.hermesAgent: defaults.string(forKey: Keys.preferredHermesAgentModel) ?? "hermes-agent" .hermesAgent: defaults.string(forKey: Keys.preferredHermesAgentModel) ?? "hermes-agent"
] ]
self.preferredModelByProvider = preferredModels self.preferredModelByProvider = preferredModels
@@ -54,6 +57,7 @@ final class SybilSettingsStore {
.openai: defaults.string(forKey: Keys.quickQuestionPreferredOpenAIModel) ?? preferredModels[.openai] ?? "gpt-4.1-mini", .openai: defaults.string(forKey: Keys.quickQuestionPreferredOpenAIModel) ?? preferredModels[.openai] ?? "gpt-4.1-mini",
.anthropic: defaults.string(forKey: Keys.quickQuestionPreferredAnthropicModel) ?? preferredModels[.anthropic] ?? "claude-3-5-sonnet-latest", .anthropic: defaults.string(forKey: Keys.quickQuestionPreferredAnthropicModel) ?? preferredModels[.anthropic] ?? "claude-3-5-sonnet-latest",
.xai: defaults.string(forKey: Keys.quickQuestionPreferredXAIModel) ?? preferredModels[.xai] ?? "grok-3-mini", .xai: defaults.string(forKey: Keys.quickQuestionPreferredXAIModel) ?? preferredModels[.xai] ?? "grok-3-mini",
.gemini: defaults.string(forKey: Keys.quickQuestionPreferredGeminiModel) ?? preferredModels[.gemini] ?? "gemini-3.5-flash",
.hermesAgent: defaults.string(forKey: Keys.quickQuestionPreferredHermesAgentModel) ?? preferredModels[.hermesAgent] ?? "hermes-agent" .hermesAgent: defaults.string(forKey: Keys.quickQuestionPreferredHermesAgentModel) ?? preferredModels[.hermesAgent] ?? "hermes-agent"
] ]
} }
@@ -72,12 +76,14 @@ final class SybilSettingsStore {
defaults.set(preferredModelByProvider[.openai], forKey: Keys.preferredOpenAIModel) defaults.set(preferredModelByProvider[.openai], forKey: Keys.preferredOpenAIModel)
defaults.set(preferredModelByProvider[.anthropic], forKey: Keys.preferredAnthropicModel) defaults.set(preferredModelByProvider[.anthropic], forKey: Keys.preferredAnthropicModel)
defaults.set(preferredModelByProvider[.xai], forKey: Keys.preferredXAIModel) defaults.set(preferredModelByProvider[.xai], forKey: Keys.preferredXAIModel)
defaults.set(preferredModelByProvider[.gemini], forKey: Keys.preferredGeminiModel)
defaults.set(preferredModelByProvider[.hermesAgent], forKey: Keys.preferredHermesAgentModel) defaults.set(preferredModelByProvider[.hermesAgent], forKey: Keys.preferredHermesAgentModel)
defaults.set(quickQuestionPreferredProvider.rawValue, forKey: Keys.quickQuestionPreferredProvider) defaults.set(quickQuestionPreferredProvider.rawValue, forKey: Keys.quickQuestionPreferredProvider)
defaults.set(quickQuestionPreferredModelByProvider[.openai], forKey: Keys.quickQuestionPreferredOpenAIModel) defaults.set(quickQuestionPreferredModelByProvider[.openai], forKey: Keys.quickQuestionPreferredOpenAIModel)
defaults.set(quickQuestionPreferredModelByProvider[.anthropic], forKey: Keys.quickQuestionPreferredAnthropicModel) defaults.set(quickQuestionPreferredModelByProvider[.anthropic], forKey: Keys.quickQuestionPreferredAnthropicModel)
defaults.set(quickQuestionPreferredModelByProvider[.xai], forKey: Keys.quickQuestionPreferredXAIModel) defaults.set(quickQuestionPreferredModelByProvider[.xai], forKey: Keys.quickQuestionPreferredXAIModel)
defaults.set(quickQuestionPreferredModelByProvider[.gemini], forKey: Keys.quickQuestionPreferredGeminiModel)
defaults.set(quickQuestionPreferredModelByProvider[.hermesAgent], forKey: Keys.quickQuestionPreferredHermesAgentModel) defaults.set(quickQuestionPreferredModelByProvider[.hermesAgent], forKey: Keys.quickQuestionPreferredHermesAgentModel)
} }
@@ -160,6 +160,7 @@ final class SybilViewModel {
.openai: ["gpt-4.1-mini"], .openai: ["gpt-4.1-mini"],
.anthropic: ["claude-3-5-sonnet-latest"], .anthropic: ["claude-3-5-sonnet-latest"],
.xai: ["grok-3-mini"], .xai: ["grok-3-mini"],
.gemini: ["gemini-3.5-flash", "gemini-flash-latest"],
.hermesAgent: ["hermes-agent"] .hermesAgent: ["hermes-agent"]
] ]
@@ -1751,13 +1752,16 @@ final class SybilViewModel {
switch target { switch target {
case let .chat(chatID): case let .chat(chatID):
SybilLog.debug(SybilLog.app, "Refreshing chat \(chatID)") SybilLog.debug(SybilLog.app, "Refreshing chat \(chatID)")
let isSelectingDifferentChat = selectedChat?.id != chatID
let chat = try await client.getChat(chatID: chatID) let chat = try await client.getChat(chatID: chatID)
guard selectedItem == target, draftKind == nil else { guard selectedItem == target, draftKind == nil else {
return return
} }
selectedChat = chat selectedChat = chat
selectedSearch = nil selectedSearch = nil
if isSelectingDifferentChat {
requestChatBottomPin() requestChatBottomPin()
}
if let provider = chat.lastUsedProvider, if let provider = chat.lastUsedProvider,
let model = chat.lastUsedModel, let model = chat.lastUsedModel,
@@ -544,12 +544,14 @@ private func makeToolCallMessage(id: String, date: Date, summary: String = "Ran
@MainActor @MainActor
@Test func foregroundChatRefreshReloadsSelectedTranscript() async throws { @Test func foregroundChatRefreshReloadsSelectedTranscript() async throws {
let date = Date(timeIntervalSince1970: 1_700_000_100) let date = Date(timeIntervalSince1970: 1_700_000_100)
let staleDetail = makeChatDetail(id: "chat-2", date: date, body: "stale transcript")
let detail = makeChatDetail(id: "chat-2", date: date, body: "refreshed transcript") let detail = makeChatDetail(id: "chat-2", date: date, body: "refreshed transcript")
let client = MockSybilClient(chatDetails: ["chat-2": detail]) let client = MockSybilClient(chatDetails: ["chat-2": detail])
let viewModel = SybilViewModel(settings: testSettings(named: #function)) { _ in client } let viewModel = SybilViewModel(settings: testSettings(named: #function)) { _ in client }
viewModel.isAuthenticated = true viewModel.isAuthenticated = true
viewModel.isCheckingSession = false viewModel.isCheckingSession = false
viewModel.selectedItem = .chat("chat-2") viewModel.selectedItem = .chat("chat-2")
viewModel.selectedChat = staleDetail
await viewModel.refreshVisibleContent(refreshCollections: false, refreshSelection: true) await viewModel.refreshVisibleContent(refreshCollections: false, refreshSelection: true)
@@ -559,7 +561,7 @@ private func makeToolCallMessage(id: String, date: Date, summary: String = "Ran
#expect(snapshot.listSearches == 0) #expect(snapshot.listSearches == 0)
#expect(snapshot.getChat == 1) #expect(snapshot.getChat == 1)
#expect(viewModel.selectedChat?.messages.first?.content == "refreshed transcript") #expect(viewModel.selectedChat?.messages.first?.content == "refreshed transcript")
#expect(viewModel.chatBottomPinRequestID == 1) #expect(viewModel.chatBottomPinRequestID == 0)
} }
@MainActor @MainActor
@@ -675,6 +677,7 @@ private func makeToolCallMessage(id: String, date: Date, summary: String = "Ran
#expect(viewModel.displayedMessages.first?.content == "fresh transcript") #expect(viewModel.displayedMessages.first?.content == "fresh transcript")
#expect(!viewModel.isLoadingSelection) #expect(!viewModel.isLoadingSelection)
#expect(viewModel.chatBottomPinRequestID == 1)
} }
@MainActor @MainActor
+6 -4
View File
@@ -1,7 +1,7 @@
# Sybil Server # Sybil Server
Backend API for: Backend API for:
- LLM multiplexer (OpenAI Responses / Anthropic / xAI Chat Completions-compatible Grok / Hermes Agent) - LLM multiplexer (OpenAI Responses / Anthropic / xAI Chat Completions-compatible Grok / Gemini / Hermes Agent)
- Personal chat database (chats/messages + LLM call log) - Personal chat database (chats/messages + LLM call log)
## Stack ## Stack
@@ -43,14 +43,16 @@ If `ADMIN_TOKEN` is not set, the server runs in open mode (dev).
- `OPENAI_API_KEY` - `OPENAI_API_KEY`
- `ANTHROPIC_API_KEY` - `ANTHROPIC_API_KEY`
- `XAI_API_KEY` - `XAI_API_KEY`
- `GEMINI_API_KEY`
- `HERMES_AGENT_API_BASE_URL` (`http://127.0.0.1:8642/v1` by default; include the `/v1` suffix) - `HERMES_AGENT_API_BASE_URL` (`http://127.0.0.1:8642/v1` by default; include the `/v1` suffix)
- `HERMES_AGENT_API_KEY` (enables the Hermes Agent provider; set to Hermes `API_SERVER_KEY`, or any non-empty value if that local server does not require auth) - `HERMES_AGENT_API_KEY` (enables the Hermes Agent provider; set to Hermes `API_SERVER_KEY`, or any non-empty value if that local server does not require auth)
- `HERMES_AGENT_MODEL` (optional fallback/override model id; defaults client-side to `hermes-agent`) - `HERMES_AGENT_MODEL` (optional fallback/override model id; defaults client-side to `hermes-agent`)
- `EXA_API_KEY` - `EXA_API_KEY`
- `CHAT_WEB_SEARCH_ENGINE` (`exa` by default, or `searxng` for chat tool calls only) - `BRAVE_SEARCH_API_KEY` (required when `CHAT_WEB_SEARCH_ENGINE=brave`)
- `CHAT_WEB_SEARCH_ENGINE` (`exa` by default; `brave` and `searxng` are also supported for chat tool calls only)
- `SEARXNG_BASE_URL` (required when `CHAT_WEB_SEARCH_ENGINE=searxng`; instance must allow `format=json`) - `SEARXNG_BASE_URL` (required when `CHAT_WEB_SEARCH_ENGINE=searxng`; instance must allow `format=json`)
- `CHAT_MAX_TOOL_ROUNDS` (`100` by default; maximum model/tool result cycles per chat completion) - `CHAT_MAX_TOOL_ROUNDS` (`100` by default; maximum model/tool result cycles per chat completion)
- `CHAT_CODEX_TOOL_ENABLED` (`false` by default; enables the `codex_exec` chat tool for OpenAI/xAI) - `CHAT_CODEX_TOOL_ENABLED` (`false` by default; enables the `codex_exec` chat tool for managed-tool providers)
- `CHAT_CODEX_REMOTE_HOST` (required when Codex tool is enabled; SSH host/IP or `user@host`) - `CHAT_CODEX_REMOTE_HOST` (required when Codex tool is enabled; SSH host/IP or `user@host`)
- `CHAT_CODEX_REMOTE_USER` (optional SSH user when host does not include one) - `CHAT_CODEX_REMOTE_USER` (optional SSH user when host does not include one)
- `CHAT_CODEX_REMOTE_PORT` (`22` by default) - `CHAT_CODEX_REMOTE_PORT` (`22` by default)
@@ -58,7 +60,7 @@ If `ADMIN_TOKEN` is not set, the server runs in open mode (dev).
- `CHAT_CODEX_SSH_KEY_PATH` (recommended: path to a read-only mounted private key) - `CHAT_CODEX_SSH_KEY_PATH` (recommended: path to a read-only mounted private key)
- `CHAT_CODEX_SSH_PRIVATE_KEY_B64` (optional fallback private key delivery) - `CHAT_CODEX_SSH_PRIVATE_KEY_B64` (optional fallback private key delivery)
- `CHAT_CODEX_EXEC_TIMEOUT_MS` (`600000` by default) - `CHAT_CODEX_EXEC_TIMEOUT_MS` (`600000` by default)
- `CHAT_SHELL_TOOL_ENABLED` (`false` by default; enables the `shell_exec` chat tool for OpenAI/xAI on the same devbox) - `CHAT_SHELL_TOOL_ENABLED` (`false` by default; enables the `shell_exec` chat tool for managed-tool providers on the same devbox)
- `CHAT_SHELL_EXEC_TIMEOUT_MS` (`120000` by default) - `CHAT_SHELL_EXEC_TIMEOUT_MS` (`120000` by default)
## API ## API
+1
View File
@@ -13,6 +13,7 @@ enum Provider {
openai openai
anthropic anthropic
xai xai
gemini
hermes_agent @map("hermes-agent") hermes_agent @map("hermes-agent")
} }
+11 -1
View File
@@ -24,7 +24,7 @@ const ChatWebSearchEngineSchema = z.preprocess(
const trimmed = value.trim(); const trimmed = value.trim();
return trimmed ? trimmed.toLowerCase() : undefined; return trimmed ? trimmed.toLowerCase() : undefined;
}, },
z.enum(["exa", "searxng"]).default("exa") z.enum(["exa", "searxng", "brave"]).default("exa")
); );
const BooleanFlagSchema = z.preprocess((value) => { const BooleanFlagSchema = z.preprocess((value) => {
@@ -66,10 +66,12 @@ const EnvSchema = z.object({
OPENAI_API_KEY: z.string().optional(), OPENAI_API_KEY: z.string().optional(),
ANTHROPIC_API_KEY: z.string().optional(), ANTHROPIC_API_KEY: z.string().optional(),
XAI_API_KEY: z.string().optional(), XAI_API_KEY: z.string().optional(),
GEMINI_API_KEY: z.string().optional(),
HERMES_AGENT_API_BASE_URL: HermesAgentApiBaseUrlSchema, HERMES_AGENT_API_BASE_URL: HermesAgentApiBaseUrlSchema,
HERMES_AGENT_API_KEY: OptionalTrimmedStringSchema, HERMES_AGENT_API_KEY: OptionalTrimmedStringSchema,
HERMES_AGENT_MODEL: OptionalTrimmedStringSchema, HERMES_AGENT_MODEL: OptionalTrimmedStringSchema,
EXA_API_KEY: z.string().optional(), EXA_API_KEY: z.string().optional(),
BRAVE_SEARCH_API_KEY: OptionalTrimmedStringSchema,
// Chat-mode web_search tool configuration. Search mode remains Exa-only for now. // Chat-mode web_search tool configuration. Search mode remains Exa-only for now.
CHAT_WEB_SEARCH_ENGINE: ChatWebSearchEngineSchema, CHAT_WEB_SEARCH_ENGINE: ChatWebSearchEngineSchema,
@@ -99,6 +101,14 @@ const EnvSchema = z.object({
}); });
} }
if (value.CHAT_WEB_SEARCH_ENGINE === "brave" && !value.BRAVE_SEARCH_API_KEY) {
ctx.addIssue({
code: "custom",
path: ["BRAVE_SEARCH_API_KEY"],
message: "BRAVE_SEARCH_API_KEY is required when CHAT_WEB_SEARCH_ENGINE=brave",
});
}
if ((value.CHAT_CODEX_TOOL_ENABLED || value.CHAT_SHELL_TOOL_ENABLED) && !value.CHAT_CODEX_REMOTE_HOST) { if ((value.CHAT_CODEX_TOOL_ENABLED || value.CHAT_SHELL_TOOL_ENABLED) && !value.CHAT_CODEX_REMOTE_HOST) {
ctx.addIssue({ ctx.addIssue({
code: "custom", code: "custom",
+23
View File
@@ -7,6 +7,7 @@ import { convert as htmlToText } from "html-to-text";
import { z } from "zod"; import { z } from "zod";
import { buildBrowserLikeNavigationHeaders } from "../browser-fetch-headers.js"; import { buildBrowserLikeNavigationHeaders } from "../browser-fetch-headers.js";
import { env } from "../env.js"; import { env } from "../env.js";
import { searchBrave } from "../search/brave.js";
import { exaClient } from "../search/exa.js"; import { exaClient } from "../search/exa.js";
import { searchSearxng } from "../search/searxng.js"; import { searchSearxng } from "../search/searxng.js";
import type { ChatMessage } from "./types.js"; import type { ChatMessage } from "./types.js";
@@ -507,11 +508,33 @@ async function runSearxngWebSearchTool(args: WebSearchArgs): Promise<ToolRunOutc
}; };
} }
async function runBraveWebSearchTool(args: WebSearchArgs): Promise<ToolRunOutcome> {
const response = await searchBrave(args.query, {
numResults: args.numResults ?? DEFAULT_WEB_RESULTS,
includeDomains: args.includeDomains,
excludeDomains: args.excludeDomains,
});
return {
ok: true,
searchEngine: "brave",
query: args.query,
requestId: response.requestId,
results: response.results.map((result, index) => ({
rank: index + 1,
...result,
})),
};
}
async function runWebSearchTool(input: unknown): Promise<ToolRunOutcome> { async function runWebSearchTool(input: unknown): Promise<ToolRunOutcome> {
const args = WebSearchArgsSchema.parse(input); const args = WebSearchArgsSchema.parse(input);
if (env.CHAT_WEB_SEARCH_ENGINE === "searxng") { if (env.CHAT_WEB_SEARCH_ENGINE === "searxng") {
return runSearxngWebSearchTool(args); return runSearxngWebSearchTool(args);
} }
if (env.CHAT_WEB_SEARCH_ENGINE === "brave") {
return runBraveWebSearchTool(args);
}
return runExaWebSearchTool(args); return runExaWebSearchTool(args);
} }
+501
View File
@@ -0,0 +1,501 @@
import {
buildChatToolSystemPrompt,
executeToolCallAndBuildEvent,
getEnabledChatTools,
getUnstreamedText,
looksLikeDanglingToolIntent,
MAX_DANGLING_TOOL_INTENT_RETRIES,
MAX_TOOL_ROUNDS,
prepareToolCallExecution,
type NormalizedToolCall,
type ToolAwareCompletionParams,
type ToolAwareCompletionResult,
type ToolAwareStreamingEvent,
type ToolAwareUsage,
type ToolExecutionEvent,
} from "../chat-tools.js";
import {
buildImageSummaryText,
buildTextAttachmentPrompt,
buildTopLevelSystemPrompt,
getImageAttachments,
getTextAttachments,
parseImageDataUrl,
} from "../message-content.js";
import type { ChatMessage } from "../types.js";
type GeminiClient = {
apiKey: string;
baseURL: string;
};
const INTERNAL_CORRECTION =
"Internal correction: the previous assistant message claimed it would run a tool, but no tool call was made. If the task needs an available tool, call it now. Otherwise provide the final answer directly without saying you will run a tool.";
function normalizeModelResourceName(model: string) {
const trimmed = model.trim().replace(/^\/+/, "");
return trimmed.startsWith("models/") || trimmed.startsWith("tunedModels/") ? trimmed : `models/${trimmed}`;
}
function geminiUrl(client: GeminiClient, model: string, method: "generateContent" | "streamGenerateContent", extraParams: Record<string, string> = {}) {
const url = new URL(`${client.baseURL.replace(/\/+$/, "")}/${normalizeModelResourceName(model)}:${method}`);
url.searchParams.set("key", client.apiKey);
for (const [key, value] of Object.entries(extraParams)) {
url.searchParams.set(key, value);
}
return url;
}
function generationConfig(params: Pick<ToolAwareCompletionParams, "temperature" | "maxTokens">) {
const config: Record<string, unknown> = {};
if (params.temperature !== undefined) config.temperature = params.temperature;
if (params.maxTokens !== undefined) config.maxOutputTokens = params.maxTokens;
return Object.keys(config).length ? config : undefined;
}
function toGeminiJsonSchema(schema: unknown): Record<string, unknown> | undefined {
if (!schema || typeof schema !== "object" || Array.isArray(schema)) return undefined;
const input = schema as Record<string, unknown>;
const output: Record<string, unknown> = {};
if (typeof input.type === "string") output.type = input.type;
if (typeof input.description === "string") output.description = input.description;
if (typeof input.format === "string") output.format = input.format;
if (typeof input.nullable === "boolean") output.nullable = input.nullable;
if (Array.isArray(input.enum)) output.enum = input.enum.filter((value) => typeof value === "string");
if (Array.isArray(input.required)) output.required = input.required.filter((value) => typeof value === "string");
const items = toGeminiJsonSchema(input.items);
if (items) output.items = items;
if (input.properties && typeof input.properties === "object" && !Array.isArray(input.properties)) {
const properties: Record<string, unknown> = {};
for (const [key, value] of Object.entries(input.properties)) {
const propertySchema = toGeminiJsonSchema(value);
if (propertySchema) properties[key] = propertySchema;
}
if (Object.keys(properties).length) output.properties = properties;
}
return Object.keys(output).length ? output : undefined;
}
function toGeminiTools(tools: any[]) {
const functionDeclarations = tools
.map((tool) => {
if (tool?.type !== "function") return null;
const declaration: Record<string, unknown> = {
name: tool.function.name,
description: tool.function.description,
};
const parameters = toGeminiJsonSchema(tool.function.parameters);
if (parameters) declaration.parameters = parameters;
return declaration;
})
.filter(Boolean);
return functionDeclarations.length ? [{ functionDeclarations }] : undefined;
}
function toContentParts(message: ChatMessage) {
const imageAttachments = getImageAttachments(message);
const textAttachments = getTextAttachments(message);
const parts: Array<Record<string, unknown>> = [];
for (const attachment of imageAttachments) {
const source = parseImageDataUrl(attachment);
parts.push({
inlineData: {
mimeType: source.mediaType,
data: source.data,
},
});
}
const imageSummary = buildImageSummaryText(imageAttachments);
if (imageSummary) {
parts.push({ text: imageSummary });
}
for (const attachment of textAttachments) {
parts.push({ text: buildTextAttachmentPrompt(attachment) });
}
if (message.content.trim()) {
parts.push({ text: message.content });
}
return parts.length ? parts : [{ text: "" }];
}
function buildConversationContent(message: ChatMessage) {
if (message.role === "system") {
throw new Error("System messages must be handled separately for Gemini.");
}
if (message.role === "tool") {
const name = message.name?.trim() || "tool";
return {
role: "user",
parts: [{ text: `Tool output (${name}):\n${message.content}` }],
};
}
return {
role: message.role === "assistant" ? "model" : "user",
parts: toContentParts(message),
};
}
function buildBaseContents(messages: ChatMessage[]) {
return messages.filter((message) => message.role !== "system").map((message) => buildConversationContent(message));
}
function buildSystemInstruction(params: ToolAwareCompletionParams, toolSystemPrompt?: string) {
const text = buildTopLevelSystemPrompt(params.messages, params.userLocation, toolSystemPrompt);
return text ? { parts: [{ text }] } : undefined;
}
function mergeUsage(acc: Required<ToolAwareUsage>, usage: any) {
const normalized = normalizeUsage(usage);
if (!normalized) return false;
acc.inputTokens += normalized.inputTokens;
acc.outputTokens += normalized.outputTokens;
acc.totalTokens += normalized.totalTokens;
return true;
}
function normalizeUsage(usage: any) {
if (!usage) return null;
const inputTokens = usage.promptTokenCount ?? 0;
const outputTokens = usage.candidatesTokenCount ?? 0;
const totalTokens = usage.totalTokenCount ?? inputTokens + outputTokens;
return { inputTokens, outputTokens, totalTokens };
}
function getCandidate(response: any) {
return Array.isArray(response?.candidates) ? response.candidates[0] : null;
}
function getParts(response: any) {
const parts = getCandidate(response)?.content?.parts;
return Array.isArray(parts) ? parts : [];
}
function extractText(response: any) {
return getParts(response)
.map((part: any) => (typeof part?.text === "string" ? part.text : ""))
.join("");
}
function stringifyToolArgs(args: unknown) {
try {
return JSON.stringify(args ?? {});
} catch {
return "{}";
}
}
function normalizeToolCallsFromParts(parts: any[], round: number): NormalizedToolCall[] {
return parts
.filter((part) => part?.functionCall)
.map((part, index) => ({
id: part.functionCall.id ?? `tool_call_${round}_${index}`,
name: part.functionCall.name ?? "unknown_tool",
arguments: stringifyToolArgs(part.functionCall.args),
}));
}
function buildFunctionResponsePart(call: NormalizedToolCall, toolResult: unknown) {
return {
functionResponse: {
id: call.id,
name: call.name,
response: toolResult,
},
};
}
function appendCorrection(conversation: any[], text: string) {
conversation.push({ role: "model", parts: [{ text }] });
conversation.push({ role: "user", parts: [{ text: INTERNAL_CORRECTION }] });
}
async function parseGeminiResponse(response: Response) {
const bodyText = await response.text();
let body: any = null;
try {
body = bodyText ? JSON.parse(bodyText) : null;
} catch {
body = { raw: bodyText };
}
if (!response.ok) {
throw new Error(body?.error?.message ?? `Gemini API request failed with status ${response.status}.`);
}
return body;
}
async function generateContent(params: ToolAwareCompletionParams, body: Record<string, unknown>) {
const response = await fetch(geminiUrl(params.client, params.model, "generateContent"), {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify(body),
});
return parseGeminiResponse(response);
}
function getFailureMessage(response: any, text: string, toolCallCount: number) {
const promptBlockReason = response?.promptFeedback?.blockReason;
if (promptBlockReason) return `Gemini prompt blocked: ${promptBlockReason}.`;
const candidate = getCandidate(response);
const finishReason = candidate?.finishReason;
if (!finishReason || finishReason === "STOP" || finishReason === "MAX_TOKENS") return null;
if (text || toolCallCount > 0) return null;
return candidate?.finishMessage ?? `Gemini response stopped: ${finishReason}.`;
}
function buildRequest(params: ToolAwareCompletionParams, conversation: any[], enabledTools: any[] = []) {
const tools = toGeminiTools(enabledTools);
return {
contents: conversation,
systemInstruction: buildSystemInstruction(params, enabledTools.length ? buildChatToolSystemPrompt(params) : undefined),
generationConfig: generationConfig(params),
tools,
toolConfig: tools ? { functionCallingConfig: { mode: "AUTO" } } : undefined,
};
}
export async function completeWithGeminiApi(params: ToolAwareCompletionParams): Promise<ToolAwareCompletionResult> {
const enabledTools = getEnabledChatTools(params);
const conversation = buildBaseContents(params.messages);
const rawResponses: unknown[] = [];
const toolEvents: ToolExecutionEvent[] = [];
const usageAcc: Required<ToolAwareUsage> = { inputTokens: 0, outputTokens: 0, totalTokens: 0 };
let sawUsage = false;
let totalToolCalls = 0;
let danglingToolIntentRetries = 0;
for (let round = 0; round < MAX_TOOL_ROUNDS; round += 1) {
const response = await generateContent(params, buildRequest(params, conversation, enabledTools));
rawResponses.push(response);
sawUsage = mergeUsage(usageAcc, response?.usageMetadata) || sawUsage;
const parts = getParts(response);
const text = extractText(response);
const normalizedToolCalls = normalizeToolCallsFromParts(parts, round);
const failureMessage = getFailureMessage(response, text, normalizedToolCalls.length);
if (failureMessage) throw new Error(failureMessage);
if (!normalizedToolCalls.length) {
if (danglingToolIntentRetries < MAX_DANGLING_TOOL_INTENT_RETRIES && looksLikeDanglingToolIntent(text)) {
danglingToolIntentRetries += 1;
appendCorrection(conversation, text);
continue;
}
return {
text,
usage: sawUsage ? usageAcc : undefined,
raw: { responses: rawResponses, toolCallsUsed: totalToolCalls, api: "gemini.generateContent" },
toolEvents,
};
}
totalToolCalls += normalizedToolCalls.length;
conversation.push({ role: "model", parts });
const toolResultParts: any[] = [];
for (const call of normalizedToolCalls) {
const { execution } = prepareToolCallExecution(call);
const { event, toolResult } = await executeToolCallAndBuildEvent(call, execution, params);
toolEvents.push(event);
toolResultParts.push(buildFunctionResponsePart(call, toolResult));
}
conversation.push({ role: "user", parts: toolResultParts });
}
return {
text: "I reached the tool-call limit while gathering information. Please narrow the request and try again.",
usage: sawUsage ? usageAcc : undefined,
raw: { responses: rawResponses, toolCallsUsed: totalToolCalls, toolCallLimitReached: true, api: "gemini.generateContent" },
toolEvents,
};
}
function findSseBoundary(buffer: string) {
const crlf = buffer.indexOf("\r\n\r\n");
const lf = buffer.indexOf("\n\n");
if (crlf === -1) return lf === -1 ? null : { index: lf, length: 2 };
if (lf === -1) return { index: crlf, length: 4 };
return crlf < lf ? { index: crlf, length: 4 } : { index: lf, length: 2 };
}
function parseSseEvent(rawEvent: string) {
const data = rawEvent
.split(/\r?\n/)
.filter((line) => line.startsWith("data:"))
.map((line) => line.slice("data:".length).trimStart())
.join("\n")
.trim();
if (!data || data === "[DONE]") return null;
return JSON.parse(data);
}
async function* streamGeminiResponses(params: ToolAwareCompletionParams, body: Record<string, unknown>) {
const response = await fetch(geminiUrl(params.client, params.model, "streamGenerateContent", { alt: "sse" }), {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify(body),
});
if (!response.ok) {
await parseGeminiResponse(response);
return;
}
if (!response.body) {
throw new Error("Gemini stream response did not include a body.");
}
const reader = response.body.getReader();
const decoder = new TextDecoder();
let buffer = "";
while (true) {
const { value, done } = await reader.read();
if (done) break;
buffer += decoder.decode(value, { stream: true });
let boundary = findSseBoundary(buffer);
while (boundary) {
const rawEvent = buffer.slice(0, boundary.index);
buffer = buffer.slice(boundary.index + boundary.length);
const event = parseSseEvent(rawEvent);
if (event) yield event;
boundary = findSseBoundary(buffer);
}
}
buffer += decoder.decode();
const tail = buffer.trim();
if (tail) {
const event = parseSseEvent(tail);
if (event) yield event;
}
}
export async function* streamWithGeminiApi(params: ToolAwareCompletionParams): AsyncGenerator<ToolAwareStreamingEvent> {
const enabledTools = getEnabledChatTools(params);
const conversation = buildBaseContents(params.messages);
const rawResponses: unknown[] = [];
const toolEvents: ToolExecutionEvent[] = [];
const usageAcc: Required<ToolAwareUsage> = { inputTokens: 0, outputTokens: 0, totalTokens: 0 };
let sawUsage = false;
let totalToolCalls = 0;
let danglingToolIntentRetries = 0;
if (!enabledTools.length) {
let text = "";
let latestUsage: any = null;
for await (const response of streamGeminiResponses(params, buildRequest(params, conversation))) {
rawResponses.push(response);
if (response?.usageMetadata) latestUsage = response.usageMetadata;
const failureMessage = getFailureMessage(response, extractText(response), 0);
if (failureMessage) throw new Error(failureMessage);
const delta = extractText(response);
if (delta) {
text += delta;
yield { type: "delta", text: delta };
}
}
sawUsage = mergeUsage(usageAcc, latestUsage) || sawUsage;
yield {
type: "done",
result: {
text,
usage: sawUsage ? usageAcc : undefined,
raw: { streamed: true, responses: rawResponses, toolCallsUsed: 0, api: "gemini.streamGenerateContent" },
toolEvents: [],
},
};
return;
}
for (let round = 0; round < MAX_TOOL_ROUNDS; round += 1) {
const roundParts: any[] = [];
let roundText = "";
let latestRoundResponse: any = null;
let latestRoundUsage: any = null;
for await (const response of streamGeminiResponses(params, buildRequest(params, conversation, enabledTools))) {
rawResponses.push(response);
latestRoundResponse = response;
if (response?.usageMetadata) latestRoundUsage = response.usageMetadata;
roundParts.push(...getParts(response));
roundText += extractText(response);
}
sawUsage = mergeUsage(usageAcc, latestRoundUsage) || sawUsage;
const normalizedToolCalls = normalizeToolCallsFromParts(roundParts, round);
const failureMessage = getFailureMessage(latestRoundResponse ?? { candidates: [{ content: { parts: roundParts } }] }, roundText, normalizedToolCalls.length);
if (failureMessage) throw new Error(failureMessage);
if (!normalizedToolCalls.length) {
if (danglingToolIntentRetries < MAX_DANGLING_TOOL_INTENT_RETRIES && looksLikeDanglingToolIntent(roundText)) {
danglingToolIntentRetries += 1;
appendCorrection(conversation, roundText);
continue;
}
const unstreamedText = getUnstreamedText(roundText, "");
if (unstreamedText) {
yield { type: "delta", text: unstreamedText };
}
yield {
type: "done",
result: {
text: roundText,
usage: sawUsage ? usageAcc : undefined,
raw: { streamed: true, responses: rawResponses, toolCallsUsed: totalToolCalls, api: "gemini.streamGenerateContent" },
toolEvents,
},
};
return;
}
totalToolCalls += normalizedToolCalls.length;
conversation.push({ role: "model", parts: roundParts });
const toolResultParts: any[] = [];
for (const call of normalizedToolCalls) {
const { event: initiatedEvent, execution } = prepareToolCallExecution(call);
yield { type: "tool_call", event: initiatedEvent };
const { event, toolResult } = await executeToolCallAndBuildEvent(call, execution, params);
toolEvents.push(event);
yield { type: "tool_call", event };
toolResultParts.push(buildFunctionResponsePart(call, toolResult));
}
conversation.push({ role: "user", parts: toolResultParts });
}
yield {
type: "done",
result: {
text: "I reached the tool-call limit while gathering information. Please narrow the request and try again.",
usage: sawUsage ? usageAcc : undefined,
raw: {
streamed: true,
responses: rawResponses,
toolCallsUsed: totalToolCalls,
toolCallLimitReached: true,
api: "gemini.streamGenerateContent",
},
toolEvents,
},
};
}
+70 -3
View File
@@ -5,10 +5,11 @@ import {
type ToolAwareStreamingEvent, type ToolAwareStreamingEvent,
} from "./chat-tools.js"; } from "./chat-tools.js";
import { completeWithChatCompletionsApi, streamWithChatCompletionsApi } from "./protocols/chat-completions-api.js"; import { completeWithChatCompletionsApi, streamWithChatCompletionsApi } from "./protocols/chat-completions-api.js";
import { completeWithGeminiApi, streamWithGeminiApi } from "./protocols/gemini-api.js";
import { completeWithMessagesApi, streamWithMessagesApi } from "./protocols/messages-api.js"; import { completeWithMessagesApi, streamWithMessagesApi } from "./protocols/messages-api.js";
import { completeWithResponsesApi, streamWithResponsesApi } from "./protocols/responses-api.js"; import { completeWithResponsesApi, streamWithResponsesApi } from "./protocols/responses-api.js";
import { env } from "../env.js"; import { env } from "../env.js";
import { anthropicClient, hermesAgentClient, isHermesAgentConfigured, openaiClient, xaiClient } from "./providers.js"; import { anthropicClient, geminiClient, hermesAgentClient, isHermesAgentConfigured, openaiClient, xaiClient } from "./providers.js";
import type { ChatMessage, Provider } from "./types.js"; import type { ChatMessage, Provider } from "./types.js";
type ProviderAdapterParams = { type ProviderAdapterParams = {
@@ -27,7 +28,7 @@ export type ProviderChatAdapter = {
stream(params: ProviderAdapterParams): AsyncGenerator<ToolAwareStreamingEvent>; stream(params: ProviderAdapterParams): AsyncGenerator<ToolAwareStreamingEvent>;
}; };
type ChatProtocolId = "chat-completions" | "messages" | "responses"; type ChatProtocolId = "chat-completions" | "gemini" | "messages" | "responses";
type ChatProtocol = { type ChatProtocol = {
id: ChatProtocolId; id: ChatProtocolId;
@@ -39,6 +40,7 @@ type ModelCatalogSpec = {
enabled?: () => boolean; enabled?: () => boolean;
fetchModels(client: any): Promise<string[]>; fetchModels(client: any): Promise<string[]>;
fallbackModels?: () => string[]; fallbackModels?: () => string[];
sortModels?: (models: string[]) => string[];
}; };
type ProviderBackendSpec = { type ProviderBackendSpec = {
@@ -61,6 +63,12 @@ const messagesProtocol: ChatProtocol = {
stream: streamWithMessagesApi, stream: streamWithMessagesApi,
}; };
const geminiProtocol: ChatProtocol = {
id: "gemini",
complete: completeWithGeminiApi,
stream: streamWithGeminiApi,
};
const responsesProtocol: ChatProtocol = { const responsesProtocol: ChatProtocol = {
id: "responses", id: "responses",
complete: completeWithResponsesApi, complete: completeWithResponsesApi,
@@ -77,6 +85,10 @@ function modelIdsFromListResponse(page: any) {
: []; : [];
} }
function stripModelResourcePrefix(model: string) {
return model.startsWith("models/") ? model.slice("models/".length) : model;
}
function isLikelyResponsesApiModel(model: string) { function isLikelyResponsesApiModel(model: string) {
const id = model.toLowerCase(); const id = model.toLowerCase();
if (id.includes("embedding") || id.includes("moderation")) return false; if (id.includes("embedding") || id.includes("moderation")) return false;
@@ -86,6 +98,37 @@ function isLikelyResponsesApiModel(model: string) {
return /^(gpt-|o\d|chatgpt-)/.test(id); return /^(gpt-|o\d|chatgpt-)/.test(id);
} }
function isLikelyGeminiChatModel(model: string) {
const id = model.toLowerCase();
if (!id.startsWith("gemini-")) return false;
if (id.includes("embedding") || id.includes("embed")) return false;
if (id.includes("image") || id.includes("imagen") || id.includes("veo")) return false;
if (id.includes("audio") || id.includes("tts") || id.includes("live")) return false;
if (id.includes("computer-use") || id.includes("robotics")) return false;
return true;
}
function preferGeminiModels(models: string[]) {
const preferred = [
"gemini-3.5-flash",
"gemini-flash-latest",
"gemini-3.1-flash-lite",
"gemini-3-flash-preview",
"gemini-pro-latest",
];
const modelSet = new Set(models);
return [...preferred.filter((model) => modelSet.delete(model)), ...[...modelSet].sort((a, b) => a.localeCompare(b))];
}
async function fetchJson(url: URL): Promise<any> {
const response = await fetch(url);
const body: any = await response.json().catch(() => null);
if (!response.ok) {
throw new Error(body?.error?.message ?? `Gemini model fetch failed with status ${response.status}.`);
}
return body;
}
function withClient(params: ProviderAdapterParams, client: any, enabledTools?: string[]): ToolAwareCompletionParams { function withClient(params: ProviderAdapterParams, client: any, enabledTools?: string[]): ToolAwareCompletionParams {
return { return {
client, client,
@@ -160,6 +203,29 @@ const backendSpecs: Record<Provider, ProviderBackendSpec> = {
}, },
}, },
}, },
gemini: {
createClient: geminiClient,
plainProtocol: geminiProtocol,
toolProtocol: geminiProtocol,
managedTools: true,
modelCatalog: {
async fetchModels(client) {
const url = new URL(`${client.baseURL.replace(/\/+$/, "")}/models`);
url.searchParams.set("key", client.apiKey);
url.searchParams.set("pageSize", "1000");
const page = await fetchJson(url);
return Array.isArray(page?.models)
? page.models
.filter((model: any) => Array.isArray(model?.supportedGenerationMethods) && model.supportedGenerationMethods.includes("generateContent"))
.map((model: any) => model?.name)
.filter((id: unknown): id is string => typeof id === "string")
.map(stripModelResourcePrefix)
.filter(isLikelyGeminiChatModel)
: [];
},
sortModels: preferGeminiModels,
},
},
"hermes-agent": { "hermes-agent": {
createClient: hermesAgentClient, createClient: hermesAgentClient,
plainProtocol: chatCompletionsProtocol, plainProtocol: chatCompletionsProtocol,
@@ -209,7 +275,8 @@ export function listModelCatalogProviders(): Provider[] {
export async function fetchProviderCatalogModels(provider: Provider) { export async function fetchProviderCatalogModels(provider: Provider) {
const spec = backendSpecs[provider].modelCatalog; const spec = backendSpecs[provider].modelCatalog;
if (!spec) return []; if (!spec) return [];
return uniqSorted(await spec.fetchModels(backendSpecs[provider].createClient())); const models = uniqSorted(await spec.fetchModels(backendSpecs[provider].createClient()));
return spec.sortModels ? spec.sortModels(models) : models;
} }
export function getProviderCatalogFallbackModels(provider: Provider) { export function getProviderCatalogFallbackModels(provider: Provider) {
+2
View File
@@ -6,6 +6,7 @@ const apiToPrismaProvider = {
openai: "openai", openai: "openai",
anthropic: "anthropic", anthropic: "anthropic",
xai: "xai", xai: "xai",
gemini: "gemini",
"hermes-agent": "hermes_agent", "hermes-agent": "hermes_agent",
} as const satisfies Record<Provider, PrismaProvider>; } as const satisfies Record<Provider, PrismaProvider>;
@@ -13,6 +14,7 @@ const prismaToApiProvider = {
openai: "openai", openai: "openai",
anthropic: "anthropic", anthropic: "anthropic",
xai: "xai", xai: "xai",
gemini: "gemini",
hermes_agent: "hermes-agent", hermes_agent: "hermes-agent",
"hermes-agent": "hermes-agent", "hermes-agent": "hermes-agent",
} as const satisfies Record<PrismaProvider | "hermes-agent", Provider>; } as const satisfies Record<PrismaProvider | "hermes-agent", Provider>;
+9 -1
View File
@@ -1,5 +1,5 @@
import OpenAI from "openai";
import Anthropic from "@anthropic-ai/sdk"; import Anthropic from "@anthropic-ai/sdk";
import OpenAI from "openai";
import { env } from "../env.js"; import { env } from "../env.js";
export function openaiClient() { export function openaiClient() {
@@ -13,6 +13,14 @@ export function xaiClient() {
return new OpenAI({ apiKey: env.XAI_API_KEY, baseURL: "https://api.x.ai/v1" }); return new OpenAI({ apiKey: env.XAI_API_KEY, baseURL: "https://api.x.ai/v1" });
} }
export function geminiClient() {
if (!env.GEMINI_API_KEY) throw new Error("GEMINI_API_KEY not set");
return {
apiKey: env.GEMINI_API_KEY,
baseURL: "https://generativelanguage.googleapis.com/v1beta",
};
}
export function isHermesAgentConfigured() { export function isHermesAgentConfigured() {
return Boolean(env.HERMES_AGENT_API_KEY); return Boolean(env.HERMES_AGENT_API_KEY);
} }
+6 -1
View File
@@ -119,7 +119,12 @@ export async function* runMultiplexStream(req: MultiplexRequest): AsyncGenerator
if (shouldPersist && chatId && call) { if (shouldPersist && chatId && call) {
await prisma.$transaction(async (tx) => { await prisma.$transaction(async (tx) => {
await tx.message.create({ await tx.message.create({
data: { chatId, role: "assistant" as any, content: text }, data: {
chatId,
role: "assistant" as any,
content: text,
metadata: req.clientRequestId ? ({ clientRequestId: req.clientRequestId } as any) : undefined,
},
}); });
await tx.llmCall.update({ await tx.llmCall.update({
where: { id: call.id }, where: { id: call.id },
+2 -1
View File
@@ -1,4 +1,4 @@
export const PROVIDERS = ["openai", "anthropic", "xai", "hermes-agent"] as const; export const PROVIDERS = ["openai", "anthropic", "xai", "gemini", "hermes-agent"] as const;
export type Provider = (typeof PROVIDERS)[number]; export type Provider = (typeof PROVIDERS)[number];
@@ -33,6 +33,7 @@ export type ChatMessage = {
export type MultiplexRequest = { export type MultiplexRequest = {
chatId?: string; chatId?: string;
persist?: boolean; persist?: boolean;
clientRequestId?: string;
provider: Provider; provider: Provider;
model: string; model: string;
messages: ChatMessage[]; messages: ChatMessage[];
+120 -13
View File
@@ -16,7 +16,7 @@ import { exaClient } from "./search/exa.js";
import { isFreshSearchCacheHit, normalizeSearchQuery } from "./search-cache.js"; import { isFreshSearchCacheHit, normalizeSearchQuery } from "./search-cache.js";
import type { ChatAttachment } from "./llm/types.js"; import type { ChatAttachment } from "./llm/types.js";
const ProviderSchema = z.enum(["openai", "anthropic", "xai", "hermes-agent"]); const ProviderSchema = z.enum(["openai", "anthropic", "xai", "gemini", "hermes-agent"]);
const MAX_ADDITIONAL_SYSTEM_PROMPT_CHARS = 12_000; const MAX_ADDITIONAL_SYSTEM_PROMPT_CHARS = 12_000;
const EnabledToolsSchema = z.array(z.string().trim().min(1).max(80)).max(20).transform((value) => normalizeEnabledChatTools(value)); const EnabledToolsSchema = z.array(z.string().trim().min(1).max(80)).max(20).transform((value) => normalizeEnabledChatTools(value));
@@ -88,7 +88,7 @@ function withRequestUserLocation<T extends { userLocation?: string }>(body: T, r
return body.userLocation ? body : { ...body, userLocation: inferRequestUserLocation(req) }; return body.userLocation ? body : { ...body, userLocation: inferRequestUserLocation(req) };
} }
async function storeNonAssistantMessages(chatId: string, messages: IncomingChatMessage[]) { async function storeNonAssistantMessages(chatId: string, messages: IncomingChatMessage[], clientRequestId?: string) {
const incoming = messages.filter((m) => m.role !== "assistant"); const incoming = messages.filter((m) => m.role !== "assistant");
if (!incoming.length) return; if (!incoming.length) return;
@@ -109,14 +109,21 @@ async function storeNonAssistantMessages(chatId: string, messages: IncomingChatM
const toInsert = sharedPrefix === existingNonAssistant.length ? incoming.slice(existingNonAssistant.length) : incoming; const toInsert = sharedPrefix === existingNonAssistant.length ? incoming.slice(existingNonAssistant.length) : incoming;
if (!toInsert.length) return; if (!toInsert.length) return;
const finalUserMessageIndex = toInsert.map((message) => message.role).lastIndexOf("user");
await prisma.message.createMany({ await prisma.message.createMany({
data: toInsert.map((m) => ({ data: toInsert.map((m, index) => {
const metadata = {
...(m.attachments?.length ? { attachments: m.attachments } : {}),
...(clientRequestId && index === finalUserMessageIndex ? { clientRequestId } : {}),
};
return {
chatId, chatId,
role: m.role as any, role: m.role as any,
content: m.content, content: m.content,
name: m.name, name: m.name,
metadata: m.attachments?.length ? ({ attachments: m.attachments } as any) : undefined, metadata: Object.keys(metadata).length ? (metadata as any) : undefined,
})), };
}),
}); });
} }
@@ -169,6 +176,7 @@ const CompletionStreamBody = z
.object({ .object({
chatId: z.string().optional(), chatId: z.string().optional(),
persist: z.boolean().optional(), persist: z.boolean().optional(),
clientRequestId: z.string().trim().min(1).max(128).optional(),
provider: ProviderSchema, provider: ProviderSchema,
model: z.string().min(1), model: z.string().min(1),
messages: z.array(CompletionMessageSchema), messages: z.array(CompletionMessageSchema),
@@ -186,6 +194,13 @@ const CompletionStreamBody = z
path: ["chatId"], path: ["chatId"],
}); });
} }
if (value.clientRequestId && (value.persist === false || !value.chatId)) {
ctx.addIssue({
code: z.ZodIssueCode.custom,
message: "clientRequestId requires a persisted stream with chatId",
path: ["clientRequestId"],
});
}
}); });
function mergeAttachmentsIntoMetadata(metadata: unknown, attachments?: ChatAttachment[]) { function mergeAttachmentsIntoMetadata(metadata: unknown, attachments?: ChatAttachment[]) {
@@ -399,6 +414,7 @@ function buildSseHeaders(originHeader: string | undefined) {
type SearchRunRequest = z.infer<typeof SearchRunBody>; type SearchRunRequest = z.infer<typeof SearchRunBody>;
const activeChatStreams = new Map<string, ActiveSseStream>(); const activeChatStreams = new Map<string, ActiveSseStream>();
const activeChatStreamRequestIds = new Map<string, string>();
const activeSearchStreams = new Map<string, ActiveSseStream>(); const activeSearchStreams = new Map<string, ActiveSseStream>();
const STARRED_PROJECT_ID = "starred"; const STARRED_PROJECT_ID = "starred";
@@ -554,6 +570,7 @@ function writeSseEvent(reply: FastifyReply, event: SseStreamEvent) {
} }
async function streamActiveRun(req: FastifyRequest, reply: FastifyReply, stream: ActiveSseStream) { async function streamActiveRun(req: FastifyRequest, reply: FastifyReply, stream: ActiveSseStream) {
if (reply.raw.destroyed || reply.raw.writableEnded) return reply;
reply.raw.writeHead(200, buildSseHeaders(typeof req.headers.origin === "string" ? req.headers.origin : undefined)); reply.raw.writeHead(200, buildSseHeaders(typeof req.headers.origin === "string" ? req.headers.origin : undefined));
reply.raw.flushHeaders?.(); reply.raw.flushHeaders?.();
@@ -588,10 +605,24 @@ function mapChatStreamEvent(ev: StreamEvent): SseStreamEvent {
return { event: ev.type, data: ev }; return { event: ev.type, data: ev };
} }
function startActiveChatStream(chatId: string, body: z.infer<typeof CompletionStreamBody>) { function registerActiveChatStream(chatId: string, clientRequestId?: string) {
const stream = new ActiveSseStream(); const stream = new ActiveSseStream();
activeChatStreams.set(chatId, stream); activeChatStreams.set(chatId, stream);
if (clientRequestId) {
activeChatStreamRequestIds.set(chatId, clientRequestId);
} else {
activeChatStreamRequestIds.delete(chatId);
}
return stream;
}
function clearActiveChatStream(chatId: string, stream: ActiveSseStream) {
if (activeChatStreams.get(chatId) !== stream) return;
activeChatStreams.delete(chatId);
activeChatStreamRequestIds.delete(chatId);
}
function executeActiveChatStream(chatId: string, body: z.infer<typeof CompletionStreamBody>, stream: ActiveSseStream) {
void (async () => { void (async () => {
let sawTerminalEvent = false; let sawTerminalEvent = false;
try { try {
@@ -611,13 +642,54 @@ function startActiveChatStream(chatId: string, body: z.infer<typeof CompletionSt
} catch (err) { } catch (err) {
stream.complete({ event: "error", data: { message: getErrorMessage(err) } }); stream.complete({ event: "error", data: { message: getErrorMessage(err) } });
} finally { } finally {
activeChatStreams.delete(chatId); clearActiveChatStream(chatId, stream);
} }
})(); })();
}
function startActiveChatStream(chatId: string, body: z.infer<typeof CompletionStreamBody>) {
const stream = registerActiveChatStream(chatId, body.clientRequestId);
executeActiveChatStream(chatId, body, stream);
return stream; return stream;
} }
function getMetadataClientRequestId(metadata: unknown) {
if (!metadata || typeof metadata !== "object" || Array.isArray(metadata)) return null;
const clientRequestId = (metadata as Record<string, unknown>).clientRequestId;
return typeof clientRequestId === "string" ? clientRequestId : null;
}
async function findCompletedChatSubmission(chatId: string, clientRequestId: string) {
const assistantMessages = await prisma.message.findMany({
where: { chatId, role: "assistant" as any },
orderBy: { createdAt: "desc" },
select: { content: true, metadata: true },
});
return assistantMessages.find((message) => getMetadataClientRequestId(message.metadata) === clientRequestId) ?? null;
}
function completeChatSubmissionStream(
stream: ActiveSseStream,
chatId: string,
body: z.infer<typeof CompletionStreamBody>,
assistantText: string
) {
stream.emit("meta", {
type: "meta",
chatId,
callId: null,
provider: body.provider,
model: body.model,
});
stream.complete({
event: "done",
data: {
type: "done",
text: assistantText,
},
});
}
async function executeSearchRunStream(searchId: string, body: SearchRunRequest, stream: ActiveSseStream) { async function executeSearchRunStream(searchId: string, body: SearchRunRequest, stream: ActiveSseStream) {
const startedAt = performance.now(); const startedAt = performance.now();
const query = body.query?.trim(); const query = body.query?.trim();
@@ -935,7 +1007,18 @@ export async function registerRoutes(app: FastifyInstance) {
if (existing.title?.trim()) return { chat: serializeChatLike(existing) }; if (existing.title?.trim()) return { chat: serializeChatLike(existing) };
const fallback = body.content.split(/\r?\n/)[0]?.trim().slice(0, 48) || "New chat"; const fallback = body.content.split(/\r?\n/)[0]?.trim().slice(0, 48) || "New chat";
const suggestedRaw = await generateChatTitle(body.content); let suggestedRaw = "";
try {
suggestedRaw = await generateChatTitle(body.content);
} catch (err) {
req.log.warn(
{
chatId: body.chatId,
err: getErrorMessage(err),
},
"chat title generation failed; using fallback"
);
}
const title = normalizeSuggestedTitle(suggestedRaw, fallback); const title = normalizeSuggestedTitle(suggestedRaw, fallback);
await prisma.chat.updateMany({ await prisma.chat.updateMany({
@@ -1353,15 +1436,39 @@ export async function registerRoutes(app: FastifyInstance) {
if (!exists) return app.httpErrors.notFound("chat not found"); if (!exists) return app.httpErrors.notFound("chat not found");
} }
// Store only new non-assistant messages to avoid duplicate history entries.
if (body.persist !== false && body.chatId) { if (body.persist !== false && body.chatId) {
await storeNonAssistantMessages(body.chatId, body.messages); const activeStream = activeChatStreams.get(body.chatId);
if (activeStream) {
if (body.clientRequestId && activeChatStreamRequestIds.get(body.chatId) === body.clientRequestId) {
return streamActiveRun(req, reply, activeStream);
} }
if (body.persist !== false && body.chatId) {
if (activeChatStreams.has(body.chatId)) {
return app.httpErrors.conflict("chat completion already running"); return app.httpErrors.conflict("chat completion already running");
} }
if (body.clientRequestId) {
const reservedStream = registerActiveChatStream(body.chatId, body.clientRequestId);
try {
const completedSubmission = await findCompletedChatSubmission(body.chatId, body.clientRequestId);
if (completedSubmission) {
completeChatSubmissionStream(reservedStream, body.chatId, body, completedSubmission.content);
clearActiveChatStream(body.chatId, reservedStream);
return streamActiveRun(req, reply, reservedStream);
}
// Store only new non-assistant messages to avoid duplicate history entries.
await storeNonAssistantMessages(body.chatId, body.messages, body.clientRequestId);
const configuredBody = await applyStoredChatSettings(body);
executeActiveChatStream(body.chatId, configuredBody, reservedStream);
return streamActiveRun(req, reply, reservedStream);
} catch (err) {
reservedStream.complete({ event: "error", data: { message: getErrorMessage(err) } });
clearActiveChatStream(body.chatId, reservedStream);
throw err;
}
}
// Legacy requests without an idempotency key retain the original behavior.
await storeNonAssistantMessages(body.chatId, body.messages);
const stream = startActiveChatStream(body.chatId, await applyStoredChatSettings(body)); const stream = startActiveChatStream(body.chatId, await applyStoredChatSettings(body));
return streamActiveRun(req, reply, stream); return streamActiveRun(req, reply, stream);
} }
+305
View File
@@ -0,0 +1,305 @@
import { buildBrowserLikeRequestHeaders } from "../browser-fetch-headers.js";
import { env } from "../env.js";
const BRAVE_WEB_SEARCH_URL = "https://api.search.brave.com/res/v1/web/search";
const BRAVE_SEARCH_TIMEOUT_MS = 12_000;
const DEFAULT_BRAVE_REQUEST_INTERVAL_MS = 1_000;
const RATE_LIMIT_INTERVAL_SAFETY_RATIO = 0.05;
const MIN_RATE_LIMIT_INTERVAL_SAFETY_MS = 2;
const RATE_LIMIT_RESET_SAFETY_MS = 50;
const MAX_RATE_LIMIT_RETRIES = 3;
const MAX_RATE_LIMIT_RETRY_DELAY_MS = 8_000;
type RateLimitPolicy = {
limit: number;
windowSeconds: number;
};
let requestIntervalMs = addIntervalSafety(DEFAULT_BRAVE_REQUEST_INTERVAL_MS);
let lastRequestAtMs = 0;
let nextRequestAtMs = 0;
let quotaUnavailableUntilMs = 0;
let requestQueue = Promise.resolve();
export type BraveSearchOptions = {
numResults: number;
includeDomains?: string[];
excludeDomains?: string[];
};
export type BraveSearchResult = {
title: string | null;
url: string | null;
publishedDate: string | null;
author: string | null;
summary: string | null;
text: string | null;
highlights: string[];
};
export type BraveSearchResponse = {
query: string;
requestId: string | null;
results: BraveSearchResult[];
};
function clipText(input: string, maxCharacters: number) {
return input.length <= maxCharacters ? input : `${input.slice(0, maxCharacters)}...`;
}
function compactWhitespace(input: string) {
return input.replace(/\r/g, "").replace(/[ \t]+\n/g, "\n").replace(/\n{3,}/g, "\n\n").replace(/\s+/g, " ").trim();
}
function requireBraveSearchApiKey() {
if (!env.BRAVE_SEARCH_API_KEY) {
throw new Error("BRAVE_SEARCH_API_KEY not set");
}
return env.BRAVE_SEARCH_API_KEY;
}
function sleep(milliseconds: number) {
return new Promise<void>((resolve) => setTimeout(resolve, milliseconds));
}
function addIntervalSafety(intervalMs: number) {
return intervalMs + Math.max(MIN_RATE_LIMIT_INTERVAL_SAFETY_MS, Math.ceil(intervalMs * RATE_LIMIT_INTERVAL_SAFETY_RATIO));
}
function parseCommaSeparatedNumbers(value: string | null) {
if (!value) return [];
return value.split(",").map((part) => Number(part.trim())).map((number) => (Number.isFinite(number) ? number : null));
}
function parseRateLimitPolicy(value: string | null): RateLimitPolicy[] {
if (!value) return [];
return value.split(",").flatMap((part) => {
const match = part.trim().match(/^(\d+)\s*;\s*w=(\d+)$/i);
if (!match) return [];
const limit = Number(match[1]);
const windowSeconds = Number(match[2]);
return limit > 0 && windowSeconds > 0 ? [{ limit, windowSeconds }] : [];
});
}
function getBurstPolicyIndex(policies: RateLimitPolicy[]) {
if (!policies.length) return null;
let burstIndex = 0;
for (let index = 1; index < policies.length; index += 1) {
if (policies[index]!.windowSeconds < policies[burstIndex]!.windowSeconds) burstIndex = index;
}
return burstIndex;
}
function updateRateLimitState(headers: Headers) {
const policies = parseRateLimitPolicy(headers.get("x-ratelimit-policy"));
const burstIndex = getBurstPolicyIndex(policies);
if (burstIndex === null) return;
const burstPolicy = policies[burstIndex]!;
const learnedIntervalMs = addIntervalSafety(Math.ceil((burstPolicy.windowSeconds * 1_000) / burstPolicy.limit));
if (learnedIntervalMs < requestIntervalMs && lastRequestAtMs > 0) {
nextRequestAtMs = Math.min(nextRequestAtMs, lastRequestAtMs + learnedIntervalMs);
}
requestIntervalMs = learnedIntervalMs;
const remaining = parseCommaSeparatedNumbers(headers.get("x-ratelimit-remaining"));
const resetSeconds = parseCommaSeparatedNumbers(headers.get("x-ratelimit-reset"));
for (let index = 0; index < policies.length; index += 1) {
if ((remaining[index] ?? null) === null || remaining[index]! >= 1 || (resetSeconds[index] ?? 0) <= 0) continue;
const unavailableUntilMs = Date.now() + resetSeconds[index]! * 1_000 + RATE_LIMIT_RESET_SAFETY_MS;
if (index === burstIndex) {
nextRequestAtMs = Math.max(nextRequestAtMs, unavailableUntilMs);
} else {
quotaUnavailableUntilMs = Math.max(quotaUnavailableUntilMs, unavailableUntilMs);
}
}
}
function assertLongTermQuotaAvailable() {
if (quotaUnavailableUntilMs <= Date.now()) {
quotaUnavailableUntilMs = 0;
return;
}
const resetSeconds = Math.ceil((quotaUnavailableUntilMs - Date.now()) / 1_000);
throw new Error(`Brave Search API long-term quota is exhausted; reset is expected in ${resetSeconds} seconds.`);
}
async function waitForRateLimitSlot() {
const reservation = requestQueue.then(async () => {
while (true) {
assertLongTermQuotaAvailable();
const waitMs = nextRequestAtMs - Date.now();
if (waitMs <= 0) break;
await sleep(waitMs);
}
lastRequestAtMs = Date.now();
nextRequestAtMs = lastRequestAtMs + requestIntervalMs;
});
requestQueue = reservation.catch(() => undefined);
await reservation;
}
function get429RetryDelayMs(headers: Headers, retryNumber: number) {
const remaining = parseCommaSeparatedNumbers(headers.get("x-ratelimit-remaining"));
const resetSeconds = parseCommaSeparatedNumbers(headers.get("x-ratelimit-reset"));
const exhaustedResetSeconds = resetSeconds.filter((reset, index): reset is number => reset !== null && (remaining[index] ?? 0) < 1);
const headerDelayMs = exhaustedResetSeconds.length ? Math.max(...exhaustedResetSeconds) * 1_000 : 0;
const exponentialDelayMs = 2 ** retryNumber * 1_000;
const delayMs = Math.max(headerDelayMs + RATE_LIMIT_RESET_SAFETY_MS, exponentialDelayMs);
return delayMs <= MAX_RATE_LIMIT_RETRY_DELAY_MS ? delayMs : null;
}
async function fetchBrave(url: URL) {
const apiKey = requireBraveSearchApiKey();
for (let attempt = 0; attempt <= MAX_RATE_LIMIT_RETRIES; attempt += 1) {
await waitForRateLimitSlot();
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(), BRAVE_SEARCH_TIMEOUT_MS);
let response: Response;
try {
response = await fetch(url, {
signal: controller.signal,
headers: {
...buildBrowserLikeRequestHeaders("application/json"),
"X-Subscription-Token": apiKey,
},
});
} finally {
clearTimeout(timeout);
}
updateRateLimitState(response.headers);
if (response.status !== 429 || attempt === MAX_RATE_LIMIT_RETRIES) return response;
const retryDelayMs = get429RetryDelayMs(response.headers, attempt);
await response.arrayBuffer();
if (retryDelayMs === null) {
throw new Error("Brave Search API rate limit quota is exhausted beyond the retry window.");
}
await sleep(retryDelayMs);
}
throw new Error("Brave Search API request failed after rate-limit retries.");
}
function normalizeDomain(input: string) {
const trimmed = input.trim().toLowerCase();
if (!trimmed) return null;
try {
const parsed = new URL(trimmed.includes("://") ? trimmed : `https://${trimmed}`);
return parsed.hostname.replace(/^www\./, "");
} catch {
return trimmed.split(/[/?#]/, 1)[0]?.replace(/^www\./, "") || null;
}
}
function normalizeDomains(input: string[] | undefined) {
return Array.from(new Set((input ?? []).map(normalizeDomain).filter((domain): domain is string => Boolean(domain))));
}
function hostnameMatchesDomain(urlRaw: string | null, domain: string) {
if (!urlRaw) return false;
try {
const hostname = new URL(urlRaw).hostname.toLowerCase().replace(/^www\./, "");
return hostname === domain || hostname.endsWith(`.${domain}`);
} catch {
return false;
}
}
function filterResultsByDomains(results: BraveSearchResult[], options: BraveSearchOptions) {
const includeDomains = normalizeDomains(options.includeDomains);
const excludeDomains = normalizeDomains(options.excludeDomains);
return results.filter((result) => {
if (includeDomains.length && !includeDomains.some((domain) => hostnameMatchesDomain(result.url, domain))) return false;
if (excludeDomains.some((domain) => hostnameMatchesDomain(result.url, domain))) return false;
return true;
});
}
function buildBraveQuery(query: string, options: BraveSearchOptions) {
const includeDomains = normalizeDomains(options.includeDomains);
const excludeDomains = normalizeDomains(options.excludeDomains);
const includeClause =
includeDomains.length === 0
? ""
: includeDomains.length === 1
? `site:${includeDomains[0]}`
: `(${includeDomains.map((domain) => `site:${domain}`).join(" OR ")})`;
const excludeClause = excludeDomains.map((domain) => `-site:${domain}`).join(" ");
return [query, includeClause, excludeClause].filter(Boolean).join(" ");
}
function buildSearchUrl(query: string, options: BraveSearchOptions) {
const url = new URL(BRAVE_WEB_SEARCH_URL);
url.searchParams.set("q", buildBraveQuery(query, options));
url.searchParams.set("count", String(options.numResults));
url.searchParams.set("safesearch", "moderate");
url.searchParams.set("result_filter", "web");
url.searchParams.set("text_decorations", "false");
url.searchParams.set("extra_snippets", "true");
return url;
}
function stringOrNull(value: unknown) {
if (typeof value !== "string") return null;
const normalized = compactWhitespace(value);
return normalized || null;
}
function stringArray(value: unknown) {
if (!Array.isArray(value)) return [];
return value.filter((item): item is string => typeof item === "string").map(compactWhitespace).filter(Boolean);
}
function mapWebResult(result: any): BraveSearchResult {
const description = stringOrNull(result?.description);
const extraSnippets = stringArray(result?.extra_snippets);
const snippets = [description, ...extraSnippets].filter((snippet): snippet is string => Boolean(snippet));
const combinedText = snippets.join("\n\n");
return {
title: stringOrNull(result?.title),
url: stringOrNull(result?.url),
publishedDate: stringOrNull(result?.page_age),
author: stringOrNull(result?.profile?.name) ?? stringOrNull(result?.article?.author),
summary: description ? clipText(description, 1_400) : null,
text: combinedText ? clipText(combinedText, 700) : null,
highlights: snippets.slice(0, 3).map((snippet) => clipText(snippet, 280)),
};
}
export async function searchBrave(query: string, options: BraveSearchOptions): Promise<BraveSearchResponse> {
const url = buildSearchUrl(query, options);
const response = await fetchBrave(url);
if (!response.ok) {
await response.arrayBuffer();
throw new Error(`Brave Search API request failed with status ${response.status}.`);
}
const contentType = response.headers.get("content-type")?.toLowerCase() ?? "";
if (!contentType.includes("application/json")) {
await response.arrayBuffer();
throw new Error(`Brave Search API returned ${contentType || "unknown content type"}.`);
}
const data: any = await response.json();
const results = Array.isArray(data?.web?.results) ? data.web.results.map(mapWebResult) : [];
return {
query,
requestId: response.headers.get("x-request-id"),
results: filterResultsByDomains(results, options).slice(0, options.numResults),
};
}
export function resetBraveRateLimitStateForTests() {
requestIntervalMs = addIntervalSafety(DEFAULT_BRAVE_REQUEST_INTERVAL_MS);
lastRequestAtMs = 0;
nextRequestAtMs = 0;
quotaUnavailableUntilMs = 0;
requestQueue = Promise.resolve();
}
+284
View File
@@ -0,0 +1,284 @@
import assert from "node:assert/strict";
import test from "node:test";
import { env } from "../src/env.js";
import { resetBraveRateLimitStateForTests, searchBrave } from "../src/search/brave.js";
test("searchBrave authenticates, builds filters, and normalizes web results", async () => {
const originalFetch = globalThis.fetch;
const originalApiKey = env.BRAVE_SEARCH_API_KEY;
const fetchCalls: Array<{ input: RequestInfo | URL; init?: RequestInit }> = [];
resetBraveRateLimitStateForTests();
env.BRAVE_SEARCH_API_KEY = "test-brave-key";
globalThis.fetch = (async (input: RequestInfo | URL, init?: RequestInit) => {
fetchCalls.push({ input, init });
return new Response(
JSON.stringify({
web: {
results: [
{
title: " Brave result ",
url: "https://docs.example.com/article",
description: "Main\n snippet",
extra_snippets: ["Extra snippet one", "Extra snippet two"],
page_age: "2026-07-18T12:00:00Z",
profile: { name: "Example Docs" },
},
{
title: "Excluded result",
url: "https://blocked.example.com/article",
description: "Should be filtered",
},
],
},
}),
{
status: 200,
headers: {
"content-type": "application/json; charset=utf-8",
"x-request-id": "brave-request-1",
},
}
);
}) as typeof fetch;
try {
const response = await searchBrave("latest docs", {
numResults: 5,
includeDomains: ["https://example.com/path"],
excludeDomains: ["blocked.example.com"],
});
assert.equal(fetchCalls.length, 1);
const requestUrl = new URL(String(fetchCalls[0]?.input));
assert.equal(requestUrl.origin + requestUrl.pathname, "https://api.search.brave.com/res/v1/web/search");
assert.equal(requestUrl.searchParams.get("q"), "latest docs site:example.com -site:blocked.example.com");
assert.equal(requestUrl.searchParams.get("count"), "5");
assert.equal(requestUrl.searchParams.get("safesearch"), "moderate");
assert.equal(requestUrl.searchParams.get("result_filter"), "web");
assert.equal(requestUrl.searchParams.get("text_decorations"), "false");
assert.equal(requestUrl.searchParams.get("extra_snippets"), "true");
assert.equal((fetchCalls[0]?.init?.headers as Record<string, string>)["X-Subscription-Token"], "test-brave-key");
assert.deepEqual(response, {
query: "latest docs",
requestId: "brave-request-1",
results: [
{
title: "Brave result",
url: "https://docs.example.com/article",
publishedDate: "2026-07-18T12:00:00Z",
author: "Example Docs",
summary: "Main snippet",
text: "Main snippet\n\nExtra snippet one\n\nExtra snippet two",
highlights: ["Main snippet", "Extra snippet one", "Extra snippet two"],
},
],
});
} finally {
globalThis.fetch = originalFetch;
env.BRAVE_SEARCH_API_KEY = originalApiKey;
}
});
test("searchBrave rejects requests without an API key", async () => {
const originalApiKey = env.BRAVE_SEARCH_API_KEY;
resetBraveRateLimitStateForTests();
env.BRAVE_SEARCH_API_KEY = undefined;
try {
await assert.rejects(() => searchBrave("test", { numResults: 1 }), /BRAVE_SEARCH_API_KEY not set/);
} finally {
env.BRAVE_SEARCH_API_KEY = originalApiKey;
}
});
test("searchBrave reports non-JSON responses", async () => {
const originalFetch = globalThis.fetch;
const originalApiKey = env.BRAVE_SEARCH_API_KEY;
resetBraveRateLimitStateForTests();
env.BRAVE_SEARCH_API_KEY = "test-brave-key";
globalThis.fetch = (async () =>
new Response("upstream error", {
status: 200,
headers: { "content-type": "text/plain" },
})) as typeof fetch;
try {
await assert.rejects(
() => searchBrave("test", { numResults: 1 }),
/Brave Search API returned text\/plain/
);
} finally {
globalThis.fetch = originalFetch;
env.BRAVE_SEARCH_API_KEY = originalApiKey;
}
});
test("searchBrave evenly paces concurrent bursts using Brave's shortest policy window", async () => {
const originalFetch = globalThis.fetch;
const originalApiKey = env.BRAVE_SEARCH_API_KEY;
const requestStartedAt: number[] = [];
resetBraveRateLimitStateForTests();
env.BRAVE_SEARCH_API_KEY = "test-brave-key";
globalThis.fetch = (async () => {
requestStartedAt.push(Date.now());
return new Response(JSON.stringify({ web: { results: [] } }), {
status: 200,
headers: {
"content-type": "application/json",
"x-ratelimit-policy": "1;w=1, 2000;w=2678400",
"x-ratelimit-remaining": "1, 1999",
"x-ratelimit-reset": "1, 2678400",
},
});
}) as typeof fetch;
try {
await Promise.all([
searchBrave("burst one", { numResults: 1 }),
searchBrave("burst two", { numResults: 1 }),
searchBrave("burst three", { numResults: 1 }),
]);
assert.equal(requestStartedAt.length, 3);
assert.ok(requestStartedAt[1]! - requestStartedAt[0]! >= 1_000);
assert.ok(requestStartedAt[2]! - requestStartedAt[1]! >= 1_000);
} finally {
globalThis.fetch = originalFetch;
env.BRAVE_SEARCH_API_KEY = originalApiKey;
resetBraveRateLimitStateForTests();
}
});
test("searchBrave adapts its pacing to a 50 request-per-second Search plan", async () => {
const originalFetch = globalThis.fetch;
const originalApiKey = env.BRAVE_SEARCH_API_KEY;
const requestStartedAt: number[] = [];
resetBraveRateLimitStateForTests();
env.BRAVE_SEARCH_API_KEY = "test-brave-key";
globalThis.fetch = (async () => {
requestStartedAt.push(Date.now());
return new Response(JSON.stringify({ web: { results: [] } }), {
status: 200,
headers: {
"content-type": "application/json",
"x-ratelimit-policy": "50;w=1, 0;w=2678400",
"x-ratelimit-remaining": "49, 0",
"x-ratelimit-reset": "1, 2678400",
},
});
}) as typeof fetch;
try {
await searchBrave("learn upgraded policy", { numResults: 1 });
await Promise.all(Array.from({ length: 8 }, (_, index) => searchBrave(`fast burst ${index}`, { numResults: 1 })));
assert.equal(requestStartedAt.length, 9);
const burstStartedAt = requestStartedAt.slice(1);
for (let index = 1; index < burstStartedAt.length; index += 1) {
assert.ok(burstStartedAt[index]! - burstStartedAt[index - 1]! >= 18);
}
assert.ok(burstStartedAt.at(-1)! - burstStartedAt[0]! < 500);
} finally {
globalThis.fetch = originalFetch;
env.BRAVE_SEARCH_API_KEY = originalApiKey;
resetBraveRateLimitStateForTests();
}
});
test("searchBrave retries 429 responses after the burst window resets", async () => {
const originalFetch = globalThis.fetch;
const originalApiKey = env.BRAVE_SEARCH_API_KEY;
let fetchCount = 0;
resetBraveRateLimitStateForTests();
env.BRAVE_SEARCH_API_KEY = "test-brave-key";
globalThis.fetch = (async () => {
fetchCount += 1;
const rateLimitHeaders = {
"content-type": "application/json",
"x-ratelimit-policy": "1;w=1, 2000;w=2678400",
"x-ratelimit-remaining": fetchCount === 1 ? "0, 1999" : "1, 1998",
"x-ratelimit-reset": "1, 2678400",
};
if (fetchCount === 1) {
return new Response(JSON.stringify({ error: { detail: "Rate limit exceeded" } }), {
status: 429,
headers: rateLimitHeaders,
});
}
return new Response(JSON.stringify({ web: { results: [] } }), { status: 200, headers: rateLimitHeaders });
}) as typeof fetch;
try {
const startedAt = Date.now();
await searchBrave("retry burst", { numResults: 1 });
assert.equal(fetchCount, 2);
assert.ok(Date.now() - startedAt >= 1_000);
} finally {
globalThis.fetch = originalFetch;
env.BRAVE_SEARCH_API_KEY = originalApiKey;
resetBraveRateLimitStateForTests();
}
});
test("searchBrave does not wait for exhausted long-term quotas", async () => {
const originalFetch = globalThis.fetch;
const originalApiKey = env.BRAVE_SEARCH_API_KEY;
resetBraveRateLimitStateForTests();
env.BRAVE_SEARCH_API_KEY = "test-brave-key";
globalThis.fetch = (async () =>
new Response(JSON.stringify({ error: { detail: "Quota exceeded" } }), {
status: 429,
headers: {
"content-type": "application/json",
"x-ratelimit-policy": "1;w=1, 2000;w=2678400",
"x-ratelimit-remaining": "0, 0",
"x-ratelimit-reset": "1, 100000",
},
})) as typeof fetch;
try {
const startedAt = Date.now();
await assert.rejects(
() => searchBrave("quota exhausted", { numResults: 1 }),
/rate limit quota is exhausted beyond the retry window/
);
assert.ok(Date.now() - startedAt < 1_000);
} finally {
globalThis.fetch = originalFetch;
env.BRAVE_SEARCH_API_KEY = originalApiKey;
resetBraveRateLimitStateForTests();
}
});
test("searchBrave blocks locally after a successful request exhausts the long-term quota", async () => {
const originalFetch = globalThis.fetch;
const originalApiKey = env.BRAVE_SEARCH_API_KEY;
let fetchCount = 0;
resetBraveRateLimitStateForTests();
env.BRAVE_SEARCH_API_KEY = "test-brave-key";
globalThis.fetch = (async () => {
fetchCount += 1;
return new Response(JSON.stringify({ web: { results: [] } }), {
status: 200,
headers: {
"content-type": "application/json",
"x-ratelimit-policy": "1;w=1, 2000;w=2678400",
"x-ratelimit-remaining": "0, 0",
"x-ratelimit-reset": "1, 100000",
},
});
}) as typeof fetch;
try {
await searchBrave("last allowed query", { numResults: 1 });
await assert.rejects(
() => searchBrave("over quota query", { numResults: 1 }),
/long-term quota is exhausted/
);
assert.equal(fetchCount, 1);
} finally {
globalThis.fetch = originalFetch;
env.BRAVE_SEARCH_API_KEY = originalApiKey;
resetBraveRateLimitStateForTests();
}
});
+6
View File
@@ -27,6 +27,12 @@ test("provider backend registry selects chat protocol and managed-tool mode", ()
managedTools: true, managedTools: true,
enabledTools: ["web_search"], enabledTools: ["web_search"],
}); });
assert.deepEqual(describeProviderChatBackend("gemini", ["web_search"]), {
provider: "gemini",
protocol: "gemini",
managedTools: true,
enabledTools: ["web_search"],
});
assert.deepEqual(describeProviderChatBackend("hermes-agent", ["web_search"]), { assert.deepEqual(describeProviderChatBackend("hermes-agent", ["web_search"]), {
provider: "hermes-agent", provider: "hermes-agent",
protocol: "chat-completions", protocol: "chat-completions",
+4 -2
View File
@@ -5,8 +5,10 @@ import { fromPrismaProvider, serializeProviderFields, toPrismaProvider } from ".
test("Hermes Agent provider id maps between API and Prisma enum forms", () => { test("Hermes Agent provider id maps between API and Prisma enum forms", () => {
assert.equal(toPrismaProvider("hermes-agent"), "hermes_agent"); assert.equal(toPrismaProvider("hermes-agent"), "hermes_agent");
assert.equal(fromPrismaProvider("hermes_agent"), "hermes-agent"); assert.equal(fromPrismaProvider("hermes_agent"), "hermes-agent");
assert.deepEqual(serializeProviderFields({ initiatedProvider: "hermes_agent", lastUsedProvider: "xai" }), { assert.equal(toPrismaProvider("gemini"), "gemini");
assert.equal(fromPrismaProvider("gemini"), "gemini");
assert.deepEqual(serializeProviderFields({ initiatedProvider: "hermes_agent", lastUsedProvider: "gemini" }), {
initiatedProvider: "hermes-agent", initiatedProvider: "hermes-agent",
lastUsedProvider: "xai", lastUsedProvider: "gemini",
}); });
}); });
+1 -1
View File
@@ -1,6 +1,6 @@
import type { Provider } from "./types.js"; import type { Provider } from "./types.js";
const PROVIDERS: Provider[] = ["openai", "anthropic", "xai", "hermes-agent"]; const PROVIDERS: Provider[] = ["openai", "anthropic", "xai", "gemini", "hermes-agent"];
function normalizeBaseUrl(value: string) { function normalizeBaseUrl(value: string) {
const trimmed = value.trim(); const trimmed = value.trim();
+5 -1
View File
@@ -42,12 +42,13 @@ type ToolLogMetadata = {
resultPreview?: string | null; resultPreview?: string | null;
}; };
const BASE_PROVIDERS: Provider[] = ["openai", "anthropic", "xai"]; const BASE_PROVIDERS: Provider[] = ["openai", "anthropic", "xai", "gemini"];
const PROVIDERS: Provider[] = [...BASE_PROVIDERS, "hermes-agent"]; const PROVIDERS: Provider[] = [...BASE_PROVIDERS, "hermes-agent"];
const PROVIDER_FALLBACK_MODELS: Record<Provider, string[]> = { const PROVIDER_FALLBACK_MODELS: Record<Provider, string[]> = {
openai: ["gpt-4.1-mini"], openai: ["gpt-4.1-mini"],
anthropic: ["claude-3-5-sonnet-latest"], anthropic: ["claude-3-5-sonnet-latest"],
xai: ["grok-3-mini"], xai: ["grok-3-mini"],
gemini: ["gemini-3.5-flash", "gemini-flash-latest"],
"hermes-agent": ["hermes-agent"], "hermes-agent": ["hermes-agent"],
}; };
@@ -55,6 +56,7 @@ const EMPTY_MODEL_CATALOG: ModelCatalogResponse["providers"] = {
openai: { models: [], loadedAt: null, error: null }, openai: { models: [], loadedAt: null, error: null },
anthropic: { models: [], loadedAt: null, error: null }, anthropic: { models: [], loadedAt: null, error: null },
xai: { models: [], loadedAt: null, error: null }, xai: { models: [], loadedAt: null, error: null },
gemini: { models: [], loadedAt: null, error: null },
}; };
function escapeTags(value: string) { function escapeTags(value: string) {
@@ -79,6 +81,7 @@ function getProviderLabel(provider: Provider | null | undefined) {
if (provider === "openai") return "OpenAI"; if (provider === "openai") return "OpenAI";
if (provider === "anthropic") return "Anthropic"; if (provider === "anthropic") return "Anthropic";
if (provider === "xai") return "xAI"; if (provider === "xai") return "xAI";
if (provider === "gemini") return "Gemini";
if (provider === "hermes-agent") return "Hermes Agent"; if (provider === "hermes-agent") return "Hermes Agent";
return ""; return "";
} }
@@ -266,6 +269,7 @@ async function main() {
openai: null, openai: null,
anthropic: null, anthropic: null,
xai: null, xai: null,
gemini: null,
"hermes-agent": null, "hermes-agent": null,
}; };
let model: string = config.defaultModel ?? pickProviderModel(getModelOptions(modelCatalog, provider), null); let model: string = config.defaultModel ?? pickProviderModel(getModelOptions(modelCatalog, provider), null);
+1 -1
View File
@@ -1,4 +1,4 @@
export type Provider = "openai" | "anthropic" | "xai" | "hermes-agent"; export type Provider = "openai" | "anthropic" | "xai" | "gemini" | "hermes-agent";
export type ProviderModelInfo = { export type ProviderModelInfo = {
models: string[]; models: string[];
+16 -2
View File
@@ -3,10 +3,24 @@ self.addEventListener("install", () => {
}); });
self.addEventListener("activate", (event) => { self.addEventListener("activate", (event) => {
event.waitUntil(self.clients.claim()); event.waitUntil(
(async () => {
await self.clients.claim();
const windows = await self.clients.matchAll({ type: "window", includeUncontrolled: true });
await Promise.all(
windows.map(async (client) => {
try {
await client.navigate(client.url);
} catch {
// The client may have closed while the new worker was activating.
}
})
);
})()
);
}); });
self.addEventListener("fetch", (event) => { self.addEventListener("fetch", (event) => {
if (event.request.mode !== "navigate") return; if (event.request.mode !== "navigate") return;
event.respondWith(fetch(event.request)); event.respondWith(fetch(new Request(event.request, { cache: "no-store" })));
}); });
+464 -36
View File
@@ -1,4 +1,5 @@
import { useEffect, useMemo, useRef, useState } from "preact/hooks"; import { useEffect, useMemo, useRef, useState } from "preact/hooks";
import type { TargetedTouchEvent } from "preact";
import { import {
Check, Check,
ChevronDown, ChevronDown,
@@ -92,6 +93,11 @@ type ActiveRunsState = {
chats: Record<string, true>; chats: Record<string, true>;
searches: Record<string, true>; searches: Record<string, true>;
}; };
type RefreshCollectionsOptions = {
preferredSelection?: SidebarSelection;
reportTransientError?: boolean;
selectFallback?: boolean;
};
function readSidebarSelectionFromUrl(): SidebarSelection | null { function readSidebarSelectionFromUrl(): SidebarSelection | null {
if (typeof window === "undefined") return null; if (typeof window === "undefined") return null;
@@ -123,6 +129,7 @@ const PROVIDER_FALLBACK_MODELS: Record<Provider, string[]> = {
openai: ["gpt-4.1-mini"], openai: ["gpt-4.1-mini"],
anthropic: ["claude-3-5-sonnet-latest"], anthropic: ["claude-3-5-sonnet-latest"],
xai: ["grok-3-mini"], xai: ["grok-3-mini"],
gemini: ["gemini-3.5-flash", "gemini-flash-latest"],
"hermes-agent": ["hermes-agent"], "hermes-agent": ["hermes-agent"],
}; };
@@ -130,13 +137,30 @@ const EMPTY_MODEL_CATALOG: ModelCatalogResponse["providers"] = {
openai: { models: [], loadedAt: null, error: null }, openai: { models: [], loadedAt: null, error: null },
anthropic: { models: [], loadedAt: null, error: null }, anthropic: { models: [], loadedAt: null, error: null },
xai: { models: [], loadedAt: null, error: null }, xai: { models: [], loadedAt: null, error: null },
gemini: { models: [], loadedAt: null, error: null },
}; };
const BASE_PROVIDERS: Provider[] = ["openai", "anthropic", "xai"]; const BASE_PROVIDERS: Provider[] = ["openai", "anthropic", "xai", "gemini"];
const ALL_PROVIDERS: Provider[] = [...BASE_PROVIDERS, "hermes-agent"]; const ALL_PROVIDERS: Provider[] = [...BASE_PROVIDERS, "hermes-agent"];
const MODEL_PREFERENCES_STORAGE_KEY = "sybil:modelPreferencesByProvider"; const MODEL_PREFERENCES_STORAGE_KEY = "sybil:modelPreferencesByProvider";
const QUICK_QUESTION_MODEL_SELECTION_STORAGE_KEY = "sybil:quickQuestionModelSelection"; const QUICK_QUESTION_MODEL_SELECTION_STORAGE_KEY = "sybil:quickQuestionModelSelection";
const STREAM_RESUME_RETRY_DELAYS_MS = [0, 250, 750, 1500];
const MOBILE_LAYOUT_MEDIA_QUERY = "(max-width: 767px)";
const MOBILE_SWIPE_ACTIVATION_DISTANCE = 18;
const MOBILE_SWIPE_DIRECTION_DOMINANCE = 1.22;
const MOBILE_SWIPE_VELOCITY_PROJECTION_SECONDS = 0.18;
const MOBILE_SWIPE_COMPLETION_VELOCITY = 620;
type MobileWorkspaceSwipe = {
touchIdentifier: number;
startX: number;
startY: number;
lastX: number;
lastTimestamp: number;
velocityX: number;
direction: -1 | 1 | null;
};
type ProviderModelPreferences = Record<Provider, string | null>; type ProviderModelPreferences = Record<Provider, string | null>;
@@ -149,6 +173,7 @@ const EMPTY_MODEL_PREFERENCES: ProviderModelPreferences = {
openai: null, openai: null,
anthropic: null, anthropic: null,
xai: null, xai: null,
gemini: null,
"hermes-agent": null, "hermes-agent": null,
}; };
const EMPTY_ACTIVE_RUNS: ActiveRunsState = { const EMPTY_ACTIVE_RUNS: ActiveRunsState = {
@@ -156,6 +181,95 @@ const EMPTY_ACTIVE_RUNS: ActiveRunsState = {
searches: {}, searches: {},
}; };
function isMobileLayout() {
return typeof window !== "undefined" && window.matchMedia(MOBILE_LAYOUT_MEDIA_QUERY).matches;
}
function getMobileSwipeLatchDistance() {
if (typeof window === "undefined") return 112;
return Math.min(Math.max(window.innerWidth * 0.28, 112), 152);
}
function findTouch(touches: TouchList, identifier: number) {
for (let index = 0; index < touches.length; index += 1) {
const touch = touches.item(index);
if (touch?.identifier === identifier) return touch;
}
return null;
}
function waitForAppForeground() {
if (typeof document === "undefined" || typeof window === "undefined") return Promise.resolve();
if (document.visibilityState !== "hidden") return Promise.resolve();
return new Promise<void>((resolve) => {
const finishIfReady = () => {
if (document.visibilityState === "hidden") return;
document.removeEventListener("visibilitychange", finishIfReady);
window.removeEventListener("pageshow", finishIfReady);
resolve();
};
document.addEventListener("visibilitychange", finishIfReady);
window.addEventListener("pageshow", finishIfReady);
});
}
function waitForRetry(delayMs: number) {
if (!delayMs) return Promise.resolve();
return new Promise<void>((resolve) => window.setTimeout(resolve, delayMs));
}
async function retryAfterAppResume<T>(operation: () => Promise<T>) {
await waitForAppForeground();
let lastError: unknown = new Error("Unable to reconnect");
for (const delayMs of STREAM_RESUME_RETRY_DELAYS_MS) {
await waitForRetry(delayMs);
try {
return await operation();
} catch (err) {
lastError = err;
const message = err instanceof Error ? err.message : String(err);
if (message.includes("bearer token")) throw err;
}
}
throw lastError;
}
function getActiveRunsAfterResume() {
return retryAfterAppResume(getActiveRuns);
}
function createClientRequestId() {
if (typeof crypto !== "undefined" && typeof crypto.randomUUID === "function") {
return crypto.randomUUID();
}
return `web-${Date.now()}-${Math.random().toString(36).slice(2)}`;
}
function isRecoverableStreamDisconnect(error: unknown) {
if (error instanceof TypeError) return true;
const message = (error instanceof Error ? error.message : String(error)).toLowerCase();
return (
message.includes("failed to fetch") ||
message.includes("load failed") ||
message.includes("network error") ||
message.includes("networkerror") ||
message.includes("no response stream") ||
message.includes("stream disconnected before completion")
);
}
function hasNewPersistedUserMessage(chat: ChatDetail, previousMessageIds: Set<string>, content: string, attachments: ChatAttachment[]) {
return chat.messages.some((message) => {
if (previousMessageIds.has(message.id) || message.role !== "user" || message.content !== content) return false;
const persistedAttachmentIds = new Set(getMessageAttachments(message.metadata).map((attachment) => attachment.id));
return attachments.length === persistedAttachmentIds.size && attachments.every((attachment) => persistedAttachmentIds.has(attachment.id));
});
}
const TRANSCRIPT_BOTTOM_GAP = 20; const TRANSCRIPT_BOTTOM_GAP = 20;
const REPLY_SCROLL_BUFFER_MIN = 288; const REPLY_SCROLL_BUFFER_MIN = 288;
const REPLY_SCROLL_BUFFER_MAX = 576; const REPLY_SCROLL_BUFFER_MAX = 576;
@@ -345,6 +459,7 @@ function loadStoredModelPreferences() {
openai: typeof parsed.openai === "string" && parsed.openai.trim() ? parsed.openai.trim() : null, openai: typeof parsed.openai === "string" && parsed.openai.trim() ? parsed.openai.trim() : null,
anthropic: typeof parsed.anthropic === "string" && parsed.anthropic.trim() ? parsed.anthropic.trim() : null, anthropic: typeof parsed.anthropic === "string" && parsed.anthropic.trim() ? parsed.anthropic.trim() : null,
xai: typeof parsed.xai === "string" && parsed.xai.trim() ? parsed.xai.trim() : null, xai: typeof parsed.xai === "string" && parsed.xai.trim() ? parsed.xai.trim() : null,
gemini: typeof parsed.gemini === "string" && parsed.gemini.trim() ? parsed.gemini.trim() : null,
"hermes-agent": "hermes-agent":
typeof parsed["hermes-agent"] === "string" && parsed["hermes-agent"].trim() ? parsed["hermes-agent"].trim() : null, typeof parsed["hermes-agent"] === "string" && parsed["hermes-agent"].trim() ? parsed["hermes-agent"].trim() : null,
}; };
@@ -354,7 +469,9 @@ function loadStoredModelPreferences() {
} }
function normalizeStoredProvider(value: unknown): Provider { function normalizeStoredProvider(value: unknown): Provider {
return value === "anthropic" || value === "xai" || value === "openai" || value === "hermes-agent" ? value : "openai"; return value === "anthropic" || value === "xai" || value === "gemini" || value === "openai" || value === "hermes-agent"
? value
: "openai";
} }
function normalizeStoredModelPreferences(value: unknown): ProviderModelPreferences { function normalizeStoredModelPreferences(value: unknown): ProviderModelPreferences {
@@ -364,6 +481,7 @@ function normalizeStoredModelPreferences(value: unknown): ProviderModelPreferenc
openai: typeof parsed.openai === "string" && parsed.openai.trim() ? parsed.openai.trim() : null, openai: typeof parsed.openai === "string" && parsed.openai.trim() ? parsed.openai.trim() : null,
anthropic: typeof parsed.anthropic === "string" && parsed.anthropic.trim() ? parsed.anthropic.trim() : null, anthropic: typeof parsed.anthropic === "string" && parsed.anthropic.trim() ? parsed.anthropic.trim() : null,
xai: typeof parsed.xai === "string" && parsed.xai.trim() ? parsed.xai.trim() : null, xai: typeof parsed.xai === "string" && parsed.xai.trim() ? parsed.xai.trim() : null,
gemini: typeof parsed.gemini === "string" && parsed.gemini.trim() ? parsed.gemini.trim() : null,
"hermes-agent": "hermes-agent":
typeof parsed["hermes-agent"] === "string" && parsed["hermes-agent"].trim() ? parsed["hermes-agent"].trim() : null, typeof parsed["hermes-agent"] === "string" && parsed["hermes-agent"].trim() ? parsed["hermes-agent"].trim() : null,
}; };
@@ -395,6 +513,7 @@ function getProviderLabel(provider: Provider | null | undefined) {
if (provider === "openai") return "OpenAI"; if (provider === "openai") return "OpenAI";
if (provider === "anthropic") return "Anthropic"; if (provider === "anthropic") return "Anthropic";
if (provider === "xai") return "xAI"; if (provider === "xai") return "xAI";
if (provider === "gemini") return "Gemini";
if (provider === "hermes-agent") return "Hermes Agent"; if (provider === "hermes-agent") return "Hermes Agent";
return ""; return "";
} }
@@ -862,11 +981,14 @@ export default function App() {
const selectedItemRef = useRef<SidebarSelection | null>(null); const selectedItemRef = useRef<SidebarSelection | null>(null);
const pendingTitleGenerationRef = useRef<Set<string>>(new Set()); const pendingTitleGenerationRef = useRef<Set<string>>(new Set());
const chatStreamAbortRefs = useRef<Map<string, AbortController>>(new Map()); const chatStreamAbortRefs = useRef<Map<string, AbortController>>(new Map());
const backgroundSuspendedChatStreamsRef = useRef<Set<string>>(new Set());
const searchRunAbortRefs = useRef<Map<string, AbortController>>(new Map()); const searchRunAbortRefs = useRef<Map<string, AbortController>>(new Map());
const quickQuestionAbortRef = useRef<AbortController | null>(null); const quickQuestionAbortRef = useRef<AbortController | null>(null);
const searchRunCountersRef = useRef<Map<string, number>>(new Map()); const searchRunCountersRef = useRef<Map<string, number>>(new Map());
const shouldAutoScrollRef = useRef(true); const shouldAutoScrollRef = useRef(true);
const wasSendingRef = useRef(false); const wasSendingRef = useRef(false);
const suppressNextMobileComposerFocusRef = useRef(false);
const mobileWorkspaceSwipeRef = useRef<MobileWorkspaceSwipe | null>(null);
const pendingReplyScrollRef = useRef(false); const pendingReplyScrollRef = useRef(false);
const transcriptTailSpacerHeightRef = useRef(TRANSCRIPT_BOTTOM_GAP); const transcriptTailSpacerHeightRef = useRef(TRANSCRIPT_BOTTOM_GAP);
const transcriptTailSpacerSettleFrameRef = useRef<number | null>(null); const transcriptTailSpacerSettleFrameRef = useRef<number | null>(null);
@@ -969,6 +1091,7 @@ export default function App() {
controller.abort(); controller.abort();
} }
chatStreamAbortRefs.current.clear(); chatStreamAbortRefs.current.clear();
backgroundSuspendedChatStreamsRef.current.clear();
for (const controller of searchRunAbortRefs.current.values()) { for (const controller of searchRunAbortRefs.current.values()) {
controller.abort(); controller.abort();
} }
@@ -1007,7 +1130,11 @@ export default function App() {
resetWorkspaceState(); resetWorkspaceState();
}; };
const refreshCollections = async (preferredSelection?: SidebarSelection) => { const refreshCollections = async ({
preferredSelection,
reportTransientError = true,
selectFallback = false,
}: RefreshCollectionsOptions = {}) => {
setIsLoadingCollections(true); setIsLoadingCollections(true);
try { try {
const nextWorkspaceItems = await listWorkspaceItems(); const nextWorkspaceItems = await listWorkspaceItems();
@@ -1028,6 +1155,9 @@ export default function App() {
if (hasItem(current)) { if (hasItem(current)) {
return current; return current;
} }
if (!selectFallback) {
return null;
}
const first = nextWorkspaceItems[0]; const first = nextWorkspaceItems[0];
return first ? { kind: first.type, id: first.id } : null; return first ? { kind: first.type, id: first.id } : null;
}); });
@@ -1035,7 +1165,7 @@ export default function App() {
const message = err instanceof Error ? err.message : String(err); const message = err instanceof Error ? err.message : String(err);
if (message.includes("bearer token")) { if (message.includes("bearer token")) {
handleAuthFailure(message); handleAuthFailure(message);
} else { } else if (reportTransientError || !isRecoverableStreamDisconnect(err)) {
setError(message); setError(message);
} }
} finally { } finally {
@@ -1094,7 +1224,7 @@ export default function App() {
const message = err instanceof Error ? err.message : String(err); const message = err instanceof Error ? err.message : String(err);
if (message.includes("bearer token")) { if (message.includes("bearer token")) {
handleAuthFailure(message); handleAuthFailure(message);
} else { } else if (!isRecoverableStreamDisconnect(err)) {
setError(message); setError(message);
} }
} finally { } finally {
@@ -1112,7 +1242,7 @@ export default function App() {
const message = err instanceof Error ? err.message : String(err); const message = err instanceof Error ? err.message : String(err);
if (message.includes("bearer token")) { if (message.includes("bearer token")) {
handleAuthFailure(message); handleAuthFailure(message);
} else { } else if (!isRecoverableStreamDisconnect(err)) {
setError(message); setError(message);
} }
} finally { } finally {
@@ -1124,7 +1254,12 @@ export default function App() {
if (!isAuthenticated) return; if (!isAuthenticated) return;
const preferredSelection = initialRouteSelectionRef.current; const preferredSelection = initialRouteSelectionRef.current;
initialRouteSelectionRef.current = null; initialRouteSelectionRef.current = null;
void Promise.all([refreshCollections(preferredSelection ?? undefined), refreshModels(), refreshChatTools(), refreshActiveRuns()]); void Promise.all([
refreshCollections({ preferredSelection: preferredSelection ?? undefined, selectFallback: true }),
refreshModels(),
refreshChatTools(),
refreshActiveRuns(),
]);
}, [isAuthenticated]); }, [isAuthenticated]);
useEffect(() => { useEffect(() => {
@@ -1135,6 +1270,44 @@ export default function App() {
return () => window.clearInterval(interval); return () => window.clearInterval(interval);
}, [isAuthenticated]); }, [isAuthenticated]);
useEffect(() => {
if (!isAuthenticated) return;
const suspendChatStreams = () => {
for (const [chatId, controller] of chatStreamAbortRefs.current) {
backgroundSuspendedChatStreamsRef.current.add(chatId);
controller.abort();
}
};
const handleVisibilityChange = () => {
if (document.visibilityState === "hidden") {
suspendChatStreams();
return;
}
setError((current) => (current && isRecoverableStreamDisconnect(new Error(current)) ? null : current));
void refreshActiveRuns();
const currentSelection = selectedItemRef.current;
void refreshCollections({ reportTransientError: false });
if (currentSelection?.kind === "chat") {
void refreshChat(currentSelection.id);
} else if (currentSelection?.kind === "search") {
void refreshSearch(currentSelection.id);
}
};
const handlePageShow = () => {
void refreshActiveRuns();
};
document.addEventListener("visibilitychange", handleVisibilityChange);
window.addEventListener("pagehide", suspendChatStreams);
window.addEventListener("pageshow", handlePageShow);
return () => {
document.removeEventListener("visibilitychange", handleVisibilityChange);
window.removeEventListener("pagehide", suspendChatStreams);
window.removeEventListener("pageshow", handlePageShow);
};
}, [isAuthenticated]);
useEffect(() => { useEffect(() => {
const onPopState = () => { const onPopState = () => {
setContextMenu(null); setContextMenu(null);
@@ -1229,10 +1402,18 @@ export default function App() {
const selectedKey = selectedItem ? `${selectedItem.kind}:${selectedItem.id}` : null; const selectedKey = selectedItem ? `${selectedItem.kind}:${selectedItem.id}` : null;
const transcriptViewKey = draftKind ? `draft:${draftKind}` : selectedKey ?? "empty"; const transcriptViewKey = draftKind ? `draft:${draftKind}` : selectedKey ?? "empty";
const selectedChatPendingState = selectedItem?.kind === "chat" ? pendingChatStates[selectedItem.id] ?? null : null; const selectedChatPendingState =
const selectedSearchRunState = selectedItem?.kind === "search" ? runningSearchStates[selectedItem.id] ?? null : null; draftKind === null && selectedItem?.kind === "chat" ? pendingChatStates[selectedItem.id] ?? null : null;
const selectedChatIsActive = selectedItem?.kind === "chat" && (!!selectedChatPendingState || !!activeRuns.chats[selectedItem.id]); const selectedSearchRunState =
const selectedSearchIsActive = selectedItem?.kind === "search" && (!!selectedSearchRunState || !!activeRuns.searches[selectedItem.id]); draftKind === null && selectedItem?.kind === "search" ? runningSearchStates[selectedItem.id] ?? null : null;
const selectedChatIsActive =
draftKind === null &&
selectedItem?.kind === "chat" &&
(!!selectedChatPendingState || !!activeRuns.chats[selectedItem.id]);
const selectedSearchIsActive =
draftKind === null &&
selectedItem?.kind === "search" &&
(!!selectedSearchRunState || !!activeRuns.searches[selectedItem.id]);
const isSearchMode = draftKind ? draftKind === "search" : selectedItem?.kind === "search"; const isSearchMode = draftKind ? draftKind === "search" : selectedItem?.kind === "search";
const isSearchRunning = !!selectedSearchIsActive; const isSearchRunning = !!selectedSearchIsActive;
const isSendingActiveChat = draftKind !== "search" && !!selectedChatIsActive; const isSendingActiveChat = draftKind !== "search" && !!selectedChatIsActive;
@@ -1287,6 +1468,7 @@ export default function App() {
wasSendingRef.current = isSendingActiveChat; wasSendingRef.current = isSendingActiveChat;
if (isSendingActiveChat) return; if (isSendingActiveChat) return;
if (wasSending) { if (wasSending) {
suppressNextMobileComposerFocusRef.current = isMobileLayout();
shouldAutoScrollRef.current = false; shouldAutoScrollRef.current = false;
return; return;
} }
@@ -1310,6 +1492,12 @@ export default function App() {
if (isActiveSelectionSending) return; if (isActiveSelectionSending) return;
const hasWorkspaceSelection = Boolean(selectedItem) || draftKind !== null; const hasWorkspaceSelection = Boolean(selectedItem) || draftKind !== null;
if (!hasWorkspaceSelection) return; if (!hasWorkspaceSelection) return;
const mobileLayout = isMobileLayout();
if (mobileLayout && (Boolean(selectedItem) || suppressNextMobileComposerFocusRef.current)) {
suppressNextMobileComposerFocusRef.current = false;
return;
}
suppressNextMobileComposerFocusRef.current = false;
focusComposer(); focusComposer();
}, [draftKind, isActiveSelectionSending, selectedKey]); }, [draftKind, isActiveSelectionSending, selectedKey]);
@@ -1332,7 +1520,10 @@ export default function App() {
}; };
}, []); }, []);
const messages = selectedChat?.messages ?? []; const messages =
draftKind === null && selectedItem?.kind === "chat" && selectedChat?.id === selectedItem.id
? selectedChat.messages
: [];
useEffect(() => { useEffect(() => {
if (isSearchMode && pendingAttachments.length) { if (isSearchMode && pendingAttachments.length) {
@@ -1439,6 +1630,7 @@ export default function App() {
setError(null); setError(null);
setContextMenu(null); setContextMenu(null);
setDraftKind("chat"); setDraftKind("chat");
selectedItemRef.current = null;
setSelectedItem(null); setSelectedItem(null);
setSelectedChat(null); setSelectedChat(null);
setSelectedSearch(null); setSelectedSearch(null);
@@ -1451,6 +1643,120 @@ export default function App() {
setIsMobileSidebarOpen(false); setIsMobileSidebarOpen(false);
}; };
const resetMobileWorkspaceSwipe = () => {
mobileWorkspaceSwipeRef.current = null;
};
const handleMobileWorkspaceTouchStart = (event: TargetedTouchEvent<HTMLElement>) => {
if (
event.touches.length !== 1 ||
!isMobileLayout() ||
isMobileSidebarOpen ||
isQuickQuestionOpen ||
isChatSettingsOpen ||
renameChatDialog !== null
) {
return;
}
const target = event.target;
if (
target instanceof Element &&
target.closest("input, textarea, select, button, a, [role='dialog'], [data-mobile-swipe-ignore]")
) {
return;
}
const touch = event.touches.item(0);
if (!touch) return;
mobileWorkspaceSwipeRef.current = {
touchIdentifier: touch.identifier,
startX: touch.clientX,
startY: touch.clientY,
lastX: touch.clientX,
lastTimestamp: event.timeStamp,
velocityX: 0,
direction: null,
};
};
const handleMobileWorkspaceTouchMove = (event: TargetedTouchEvent<HTMLElement>) => {
const swipe = mobileWorkspaceSwipeRef.current;
if (!swipe) return;
const touch = findTouch(event.touches, swipe.touchIdentifier);
if (!touch) return;
const deltaX = touch.clientX - swipe.startX;
const deltaY = touch.clientY - swipe.startY;
const horizontalTravel = Math.abs(deltaX);
const verticalTravel = Math.abs(deltaY);
if (swipe.direction === null) {
if (
verticalTravel >= MOBILE_SWIPE_ACTIVATION_DISTANCE &&
verticalTravel > horizontalTravel * MOBILE_SWIPE_DIRECTION_DOMINANCE
) {
resetMobileWorkspaceSwipe();
return;
}
if (
horizontalTravel < MOBILE_SWIPE_ACTIVATION_DISTANCE ||
horizontalTravel < verticalTravel * MOBILE_SWIPE_DIRECTION_DOMINANCE
) {
return;
}
swipe.direction = deltaX > 0 ? 1 : -1;
}
const elapsedMs = event.timeStamp - swipe.lastTimestamp;
if (elapsedMs > 0 && touch.clientX !== swipe.lastX) {
swipe.velocityX = ((touch.clientX - swipe.lastX) / elapsedMs) * 1000;
}
swipe.lastX = touch.clientX;
swipe.lastTimestamp = event.timeStamp;
if (event.cancelable) event.preventDefault();
};
const handleMobileWorkspaceTouchEnd = (event: TargetedTouchEvent<HTMLElement>) => {
const swipe = mobileWorkspaceSwipeRef.current;
if (!swipe) return;
const touch = findTouch(event.changedTouches, swipe.touchIdentifier);
if (!touch) return;
const direction = swipe.direction;
const deltaX = touch.clientX - swipe.startX;
const elapsedMs = event.timeStamp - swipe.lastTimestamp;
if (elapsedMs > 0 && touch.clientX !== swipe.lastX) {
swipe.velocityX = ((touch.clientX - swipe.lastX) / elapsedMs) * 1000;
}
resetMobileWorkspaceSwipe();
if (direction === null) return;
const directionalDistance = deltaX * direction;
const directionalVelocity = swipe.velocityX * direction;
const projectedDistance =
directionalDistance + directionalVelocity * MOBILE_SWIPE_VELOCITY_PROJECTION_SECONDS;
const shouldComplete =
directionalVelocity <= -MOBILE_SWIPE_COMPLETION_VELOCITY
? false
: directionalVelocity >= MOBILE_SWIPE_COMPLETION_VELOCITY ||
directionalDistance >= getMobileSwipeLatchDistance() ||
projectedDistance >= getMobileSwipeLatchDistance();
if (!shouldComplete) return;
if (direction > 0) {
setIsMobileSidebarOpen(true);
} else {
handleCreateChat();
}
};
const handleMobileWorkspaceTouchCancel = (event: TargetedTouchEvent<HTMLElement>) => {
const swipe = mobileWorkspaceSwipeRef.current;
if (!swipe || !findTouch(event.changedTouches, swipe.touchIdentifier)) return;
resetMobileWorkspaceSwipe();
};
const handleOpenQuickQuestion = () => { const handleOpenQuickQuestion = () => {
setQuickQuestionError(null); setQuickQuestionError(null);
setIsQuickQuestionOpen(true); setIsQuickQuestionOpen(true);
@@ -1461,6 +1767,7 @@ export default function App() {
setError(null); setError(null);
setContextMenu(null); setContextMenu(null);
setDraftKind("search"); setDraftKind("search");
selectedItemRef.current = null;
setSelectedItem(null); setSelectedItem(null);
setSelectedChat(null); setSelectedChat(null);
setSelectedSearch(null); setSelectedSearch(null);
@@ -1788,7 +2095,7 @@ export default function App() {
} else { } else {
await deleteSearch(target.id); await deleteSearch(target.id);
} }
await refreshCollections(); await refreshCollections({ selectFallback: true });
} catch (err) { } catch (err) {
const message = err instanceof Error ? err.message : String(err); const message = err instanceof Error ? err.message : String(err);
if (message.includes("bearer token")) { if (message.includes("bearer token")) {
@@ -2043,6 +2350,7 @@ export default function App() {
if (!baseChat || baseChat.id !== chatId) { if (!baseChat || baseChat.id !== chatId) {
baseChat = await getChat(chatId); baseChat = await getChat(chatId);
} }
const previousMessageIds = new Set(baseChat.messages.map((message) => message.id));
setPendingChatStates((current) => ({ setPendingChatStates((current) => ({
...current, ...current,
@@ -2095,12 +2403,20 @@ export default function App() {
}); });
} }
const target: SidebarSelection = { kind: "chat", id: chatId };
const clientRequestId = createClientRequestId();
while (true) {
let streamErrorMessage: string | null = null; let streamErrorMessage: string | null = null;
let replayedAssistantText = "";
const abortController = new AbortController();
chatStreamAbortRefs.current.set(chatId, abortController);
try { try {
await runCompletionStream( await runCompletionStream(
{ {
chatId, chatId,
clientRequestId,
provider, provider,
model: selectedModel, model: selectedModel,
messages: requestMessages, messages: requestMessages,
@@ -2123,6 +2439,7 @@ export default function App() {
}, },
onDelta: (payload) => { onDelta: (payload) => {
if (!payload.text) return; if (!payload.text) return;
replayedAssistantText += payload.text;
setPendingChatStates((current) => { setPendingChatStates((current) => {
const pendingState = current[chatId]; const pendingState = current[chatId];
if (!pendingState) return current; if (!pendingState) return current;
@@ -2131,7 +2448,7 @@ export default function App() {
const isTarget = index === all.length - 1 && message.id.startsWith("temp-assistant-"); const isTarget = index === all.length - 1 && message.id.startsWith("temp-assistant-");
if (!isTarget) return message; if (!isTarget) return message;
updated = true; updated = true;
return { ...message, content: message.content + payload.text }; return { ...message, content: replayedAssistantText };
}); });
return updated ? { ...current, [chatId]: { messages: nextMessages } } : current; return updated ? { ...current, [chatId]: { messages: nextMessages } } : current;
}); });
@@ -2153,30 +2470,90 @@ export default function App() {
onError: (payload) => { onError: (payload) => {
streamErrorMessage = payload.message; streamErrorMessage = payload.message;
}, },
} },
{ signal: abortController.signal }
); );
if (streamErrorMessage) { if (streamErrorMessage) {
throw new Error(streamErrorMessage); throw new Error(streamErrorMessage);
} }
await refreshCollections(); backgroundSuspendedChatStreamsRef.current.delete(chatId);
const currentSelection = selectedItemRef.current; const persistedChat = await retryAfterAppResume(() => getChat(chatId));
if (currentSelection?.kind === "chat" && currentSelection.id === chatId) { await refreshCollections({ preferredSelection: target, reportTransientError: false });
await refreshChat(chatId); if (isCurrentSelection(target)) {
setSelectedChat(persistedChat);
setSelectedSearch(null);
} }
removePendingChatState(chatId); removePendingChatState(chatId);
removeActiveRun("chat", chatId); removeActiveRun("chat", chatId);
if (currentSelection?.kind === "chat" && currentSelection.id === chatId) { if (isCurrentSelection(target)) {
requestSettleTranscriptTailSpacer(); requestSettleTranscriptTailSpacer();
} }
return { kind: "chat", id: chatId }; return target;
} catch (err) { } catch (caughtError) {
let err: unknown = caughtError;
const wasBackgroundSuspended =
abortController.signal.aborted && backgroundSuspendedChatStreamsRef.current.delete(chatId);
const shouldResumeStream =
wasBackgroundSuspended ||
(!abortController.signal.aborted && streamErrorMessage === null && isRecoverableStreamDisconnect(err));
if (chatStreamAbortRefs.current.get(chatId) === abortController) {
chatStreamAbortRefs.current.delete(chatId);
}
if (shouldResumeStream) {
try {
const persistedChat = await retryAfterAppResume(() => getChat(chatId));
setError(null);
if (isCurrentSelection(target)) {
setSelectedChat(persistedChat);
setSelectedSearch(null);
}
// Retrying the same idempotent request either starts an unaccepted
// submission or replays the existing backend-owned stream. Keep
// the current optimistic transcript visible until replay begins.
continue;
} catch (resumeError) {
if (isRecoverableStreamDisconnect(resumeError)) {
setError(null);
await waitForAppForeground();
await waitForRetry(1000);
continue;
}
err = resumeError;
}
} else if (streamErrorMessage) {
try {
const persistedChat = await retryAfterAppResume(() => getChat(chatId));
if (hasNewPersistedUserMessage(persistedChat, previousMessageIds, content, attachments)) {
if (isCurrentSelection(target)) {
setSelectedChat(persistedChat);
setSelectedSearch(null);
setError(streamErrorMessage);
}
removePendingChatState(chatId);
removeActiveRun("chat", chatId);
requestSettleTranscriptTailSpacer();
return target;
}
} catch (reconcileError) {
err = reconcileError;
}
}
removePendingChatState(chatId); removePendingChatState(chatId);
removeActiveRun("chat", chatId); removeActiveRun("chat", chatId);
pendingReplyScrollRef.current = false; pendingReplyScrollRef.current = false;
setTranscriptTailSpacer(TRANSCRIPT_BOTTOM_GAP); setTranscriptTailSpacer(TRANSCRIPT_BOTTOM_GAP);
throw err; throw err;
} finally {
if (chatStreamAbortRefs.current.get(chatId) === abortController) {
chatStreamAbortRefs.current.delete(chatId);
}
}
} }
}; };
@@ -2317,23 +2694,28 @@ export default function App() {
} }
} }
await refreshCollections(target); await refreshCollections({ preferredSelection: target });
return target; return target;
}; };
const attachToActiveChatStream = async (chatId: string) => { async function attachToActiveChatStream(chatId: string, resetFromServer = false) {
if (chatStreamAbortRefs.current.has(chatId)) return; if (chatStreamAbortRefs.current.has(chatId)) return;
const target: SidebarSelection = { kind: "chat", id: chatId }; const target: SidebarSelection = { kind: "chat", id: chatId };
addActiveRun("chat", chatId);
let shouldResetFromServer = resetFromServer;
try {
while (true) {
const abortController = new AbortController(); const abortController = new AbortController();
chatStreamAbortRefs.current.set(chatId, abortController); chatStreamAbortRefs.current.set(chatId, abortController);
addActiveRun("chat", chatId);
let streamErrorMessage: string | null = null; let streamErrorMessage: string | null = null;
try { try {
const baseChat = selectedChat?.id === chatId ? selectedChat : await getChat(chatId); const baseChat = await getChat(chatId);
const resetPendingState = shouldResetFromServer;
shouldResetFromServer = false;
setPendingChatStates((current) => { setPendingChatStates((current) => {
if (current[chatId]) return current; if (!resetPendingState && current[chatId]) return current;
return { return {
...current, ...current,
[chatId]: { [chatId]: {
@@ -2404,12 +2786,41 @@ export default function App() {
throw new Error(streamErrorMessage); throw new Error(streamErrorMessage);
} }
backgroundSuspendedChatStreamsRef.current.delete(chatId);
await refreshCollections(); await refreshCollections();
if (isCurrentSelection(target)) { if (isCurrentSelection(target)) {
await refreshChat(chatId); await refreshChat(chatId);
} }
return;
} catch (err) { } catch (err) {
if (abortController.signal.aborted) return; const wasBackgroundSuspended =
abortController.signal.aborted && backgroundSuspendedChatStreamsRef.current.delete(chatId);
const shouldResumeStream =
wasBackgroundSuspended ||
(!abortController.signal.aborted && streamErrorMessage === null && isRecoverableStreamDisconnect(err));
if (shouldResumeStream) {
shouldResetFromServer = true;
try {
const resumedRuns = await getActiveRunsAfterResume();
setActiveRuns(buildActiveRunsState(resumedRuns));
if (resumedRuns.chats.includes(chatId)) {
continue;
}
const persistedChat = await retryAfterAppResume(() => getChat(chatId));
if (isCurrentSelection(target)) {
setSelectedChat(persistedChat);
setSelectedSearch(null);
}
void refreshCollections({ preferredSelection: target });
return;
} catch (resumeError) {
err = resumeError;
}
}
if (abortController.signal.aborted && !shouldResumeStream) return;
const message = err instanceof Error ? err.message : String(err); const message = err instanceof Error ? err.message : String(err);
if (message.includes("active chat stream not found")) { if (message.includes("active chat stream not found")) {
await refreshActiveRuns(); await refreshActiveRuns();
@@ -2419,15 +2830,22 @@ export default function App() {
} else if (isCurrentSelection(target)) { } else if (isCurrentSelection(target)) {
setError(message); setError(message);
} }
return;
} finally { } finally {
if (chatStreamAbortRefs.current.get(chatId) === abortController) {
chatStreamAbortRefs.current.delete(chatId); chatStreamAbortRefs.current.delete(chatId);
}
}
}
} finally {
backgroundSuspendedChatStreamsRef.current.delete(chatId);
removePendingChatState(chatId); removePendingChatState(chatId);
removeActiveRun("chat", chatId); removeActiveRun("chat", chatId);
if (isCurrentSelection(target)) { if (isCurrentSelection(target)) {
requestSettleTranscriptTailSpacer(); requestSettleTranscriptTailSpacer();
} }
} }
}; }
const attachToActiveSearchStream = async (searchId: string) => { const attachToActiveSearchStream = async (searchId: string) => {
if (searchRunAbortRefs.current.has(searchId)) return; if (searchRunAbortRefs.current.has(searchId)) return;
@@ -2501,7 +2919,7 @@ export default function App() {
{ signal: abortController.signal } { signal: abortController.signal }
); );
await refreshCollections(target); await refreshCollections({ preferredSelection: target });
if (isCurrentSelection(target)) { if (isCurrentSelection(target)) {
await refreshSearch(searchId); await refreshSearch(searchId);
} }
@@ -2569,7 +2987,7 @@ export default function App() {
messages: [], messages: [],
}); });
setSelectedSearch(null); setSelectedSearch(null);
await refreshCollections({ kind: "chat", id: chat.id }); await refreshCollections({ preferredSelection: { kind: "chat", id: chat.id } });
await refreshChat(chat.id); await refreshChat(chat.id);
} catch (err) { } catch (err) {
const message = err instanceof Error ? err.message : String(err); const message = err instanceof Error ? err.message : String(err);
@@ -2731,7 +3149,7 @@ export default function App() {
messages: [], messages: [],
}); });
setSelectedSearch(null); setSelectedSearch(null);
await refreshCollections({ kind: "chat", id: chat.id }); await refreshCollections({ preferredSelection: { kind: "chat", id: chat.id } });
await refreshChat(chat.id); await refreshChat(chat.id);
} catch (err) { } catch (err) {
const message = err instanceof Error ? err.message : String(err); const message = err instanceof Error ? err.message : String(err);
@@ -2790,7 +3208,8 @@ export default function App() {
await refreshSearch(sentTarget.id); await refreshSearch(sentTarget.id);
} }
} finally { } finally {
if (!sentTarget || isCurrentSelection(sentTarget)) { const shouldSuppressMobileChatFocus = sentTarget?.kind === "chat" && isMobileLayout();
if ((!sentTarget || isCurrentSelection(sentTarget)) && !shouldSuppressMobileChatFocus) {
focusComposer(); focusComposer();
} }
} }
@@ -2977,7 +3396,13 @@ export default function App() {
</div> </div>
</aside> </aside>
<main className="glass-panel relative flex min-w-0 flex-1 flex-col overflow-hidden border-violet-300/18 md:rounded-2xl md:border"> <main
className="glass-panel relative flex min-w-0 flex-1 touch-pan-y flex-col overflow-hidden border-violet-300/18 md:touch-auto md:rounded-2xl md:border"
onTouchStart={handleMobileWorkspaceTouchStart}
onTouchMove={handleMobileWorkspaceTouchMove}
onTouchEnd={handleMobileWorkspaceTouchEnd}
onTouchCancel={handleMobileWorkspaceTouchCancel}
>
<header className="flex items-center justify-between gap-2 border-b border-violet-300/12 bg-[linear-gradient(180deg,hsl(243_48%_10%_/_0.86),hsl(236_48%_6%_/_0.66))] px-4 py-3 md:gap-3 md:px-7"> <header className="flex items-center justify-between gap-2 border-b border-violet-300/12 bg-[linear-gradient(180deg,hsl(243_48%_10%_/_0.86),hsl(236_48%_6%_/_0.66))] px-4 py-3 md:gap-3 md:px-7">
<div className="flex min-w-0 items-center gap-2"> <div className="flex min-w-0 items-center gap-2">
<Button <Button
@@ -3052,7 +3477,10 @@ export default function App() {
<div ref={transcriptEndRef} /> <div ref={transcriptEndRef} />
</div> </div>
<footer className="pointer-events-none absolute inset-x-0 bottom-0 z-10 bg-[linear-gradient(to_top,hsl(235_50%_4%)_0%,hsl(235_50%_4%_/_0.92)_58%,transparent)] p-3 pt-14 md:p-6 md:pt-20"> <footer
className="pointer-events-none absolute inset-x-0 bottom-0 z-10 bg-[linear-gradient(to_top,hsl(235_50%_4%)_0%,hsl(235_50%_4%_/_0.92)_58%,transparent)] p-3 pt-14 md:p-6 md:pt-20"
data-mobile-swipe-ignore
>
<div <div
className={cn( className={cn(
"pointer-events-auto mx-auto max-w-4xl rounded-2xl border bg-[linear-gradient(135deg,hsl(235_48%_7%_/_0.96),hsl(258_48%_11%_/_0.94))] p-2 shadow-lg shadow-black/20 transition", "pointer-events-auto mx-auto max-w-4xl rounded-2xl border bg-[linear-gradient(135deg,hsl(235_48%_7%_/_0.96),hsl(258_48%_11%_/_0.94))] p-2 shadow-lg shadow-black/20 transition",
@@ -478,6 +478,7 @@ export function ChatMessagesPanel({ messages, isLoading, isSending }: Props) {
) : message.content.trim() ? ( ) : message.content.trim() ? (
<MarkdownContent <MarkdownContent
markdown={message.content} markdown={message.content}
openLinksInNewTab
className={cn("[&_a]:text-inherit [&_a]:underline", isUser ? "leading-[1.78] text-fuchsia-50" : "leading-[1.82] text-violet-50")} className={cn("[&_a]:text-inherit [&_a]:underline", isUser ? "leading-[1.78] text-fuchsia-50" : "leading-[1.82] text-violet-50")}
/> />
) : null} ) : null}
@@ -10,6 +10,7 @@ type Props = {
className?: string; className?: string;
mode?: MarkdownMode; mode?: MarkdownMode;
resolveCitationIndex?: (href: string) => number | undefined; resolveCitationIndex?: (href: string) => number | undefined;
openLinksInNewTab?: boolean;
}; };
function replaceMarkdownLinksWithCitationTokens(markdown: string, resolveCitationIndex?: (href: string) => number | undefined) { function replaceMarkdownLinksWithCitationTokens(markdown: string, resolveCitationIndex?: (href: string) => number | undefined) {
@@ -28,17 +29,30 @@ markdownRenderer.table = (token) => {
return `<div class="md-table-scroll">${renderTable(token)}</div>`; return `<div class="md-table-scroll">${renderTable(token)}</div>`;
}; };
function renderMarkdown(markdown: string) { function setNewTabLinkAttributes(currentNode: Element) {
const rawHtml = marked.parse(markdown, { gfm: true, breaks: true, renderer: markdownRenderer }) as string; if (currentNode.tagName !== "A") return;
return DOMPurify.sanitize(rawHtml, { ADD_ATTR: ["class", "target", "rel"] }); currentNode.setAttribute("target", "_blank");
currentNode.setAttribute("rel", "noopener noreferrer");
} }
export function MarkdownContent({ markdown, className, mode = "default", resolveCitationIndex }: Props) { function renderMarkdown(markdown: string, openLinksInNewTab: boolean) {
const rawHtml = marked.parse(markdown, { gfm: true, breaks: true, renderer: markdownRenderer }) as string;
if (!openLinksInNewTab) return DOMPurify.sanitize(rawHtml, { ADD_ATTR: ["class", "target", "rel"] });
DOMPurify.addHook("afterSanitizeAttributes", setNewTabLinkAttributes);
try {
return DOMPurify.sanitize(rawHtml, { ADD_ATTR: ["class", "target", "rel"] });
} finally {
DOMPurify.removeHook("afterSanitizeAttributes", setNewTabLinkAttributes);
}
}
export function MarkdownContent({ markdown, className, mode = "default", resolveCitationIndex, openLinksInNewTab = false }: Props) {
const html = useMemo(() => { const html = useMemo(() => {
const prepared = const prepared =
mode === "citationTokens" ? replaceMarkdownLinksWithCitationTokens(markdown, resolveCitationIndex) : markdown; mode === "citationTokens" ? replaceMarkdownLinksWithCitationTokens(markdown, resolveCitationIndex) : markdown;
return renderMarkdown(prepared); return renderMarkdown(prepared, openLinksInNewTab);
}, [markdown, mode, resolveCitationIndex]); }, [markdown, mode, openLinksInNewTab, resolveCitationIndex]);
return <div className={cn("md-content", className)} dangerouslySetInnerHTML={{ __html: html }} />; return <div className={cn("md-content", className)} dangerouslySetInnerHTML={{ __html: html }} />;
} }
+14 -151
View File
@@ -149,7 +149,7 @@ export type CompletionRequestMessage = {
attachments?: ChatAttachment[]; attachments?: ChatAttachment[];
}; };
export type Provider = "openai" | "anthropic" | "xai" | "hermes-agent"; export type Provider = "openai" | "anthropic" | "xai" | "gemini" | "hermes-agent";
export type ProviderModelInfo = { export type ProviderModelInfo = {
models: string[]; models: string[];
@@ -450,6 +450,7 @@ async function readSseStream(response: Response, dispatch: (eventName: string, p
let buffer = ""; let buffer = "";
let eventName = "message"; let eventName = "message";
let dataLines: string[] = []; let dataLines: string[] = [];
let sawTerminalEvent = false;
const flushEvent = () => { const flushEvent = () => {
if (!dataLines.length) { if (!dataLines.length) {
@@ -466,6 +467,9 @@ async function readSseStream(response: Response, dispatch: (eventName: string, p
} }
dispatch(eventName, payload); dispatch(eventName, payload);
if (eventName === "done" || eventName === "error") {
sawTerminalEvent = true;
}
dataLines = []; dataLines = [];
eventName = "message"; eventName = "message";
@@ -505,6 +509,10 @@ async function readSseStream(response: Response, dispatch: (eventName: string, p
} }
} }
flushEvent(); flushEvent();
if (!sawTerminalEvent) {
throw new Error("Stream disconnected before completion");
}
} }
export async function runSearchStream( export async function runSearchStream(
@@ -528,87 +536,14 @@ export async function runSearchStream(
signal: options?.signal, signal: options?.signal,
}); });
if (!response.ok) { await readSseStream(response, (eventName, payload) => {
const fallback = `${response.status} ${response.statusText}`;
let message = fallback;
try {
const body = (await response.json()) as { message?: string };
if (body.message) message = body.message;
} catch {
// keep fallback message
}
throw new Error(message);
}
if (!response.body) {
throw new Error("No response stream");
}
const reader = response.body.getReader();
const decoder = new TextDecoder();
let buffer = "";
let eventName = "message";
let dataLines: string[] = [];
const flushEvent = () => {
if (!dataLines.length) {
eventName = "message";
return;
}
const dataText = dataLines.join("\n");
let payload: any = null;
try {
payload = JSON.parse(dataText);
} catch {
payload = { message: dataText };
}
if (eventName === "search_results") handlers.onSearchResults?.(payload); if (eventName === "search_results") handlers.onSearchResults?.(payload);
else if (eventName === "search_error") handlers.onSearchError?.(payload); else if (eventName === "search_error") handlers.onSearchError?.(payload);
else if (eventName === "answer") handlers.onAnswer?.(payload); else if (eventName === "answer") handlers.onAnswer?.(payload);
else if (eventName === "answer_error") handlers.onAnswerError?.(payload); else if (eventName === "answer_error") handlers.onAnswerError?.(payload);
else if (eventName === "done") handlers.onDone?.(payload); else if (eventName === "done") handlers.onDone?.(payload);
else if (eventName === "error") handlers.onError?.(payload); else if (eventName === "error") handlers.onError?.(payload);
});
dataLines = [];
eventName = "message";
};
while (true) {
const { value, done } = await reader.read();
if (done) break;
buffer += decoder.decode(value, { stream: true });
let newlineIndex = buffer.indexOf("\n");
while (newlineIndex >= 0) {
const rawLine = buffer.slice(0, newlineIndex);
buffer = buffer.slice(newlineIndex + 1);
const line = rawLine.endsWith("\r") ? rawLine.slice(0, -1) : rawLine;
if (!line) {
flushEvent();
} else if (line.startsWith("event:")) {
eventName = line.slice("event:".length).trim();
} else if (line.startsWith("data:")) {
dataLines.push(line.slice("data:".length).trimStart());
}
newlineIndex = buffer.indexOf("\n");
}
}
buffer += decoder.decode();
if (buffer.length) {
const line = buffer.endsWith("\r") ? buffer.slice(0, -1) : buffer;
if (line.startsWith("event:")) {
eventName = line.slice("event:".length).trim();
} else if (line.startsWith("data:")) {
dataLines.push(line.slice("data:".length).trimStart());
}
}
flushEvent();
} }
export async function attachSearchStream(searchId: string, handlers: RunSearchStreamHandlers, options?: { signal?: AbortSignal }) { export async function attachSearchStream(searchId: string, handlers: RunSearchStreamHandlers, options?: { signal?: AbortSignal }) {
@@ -654,6 +589,7 @@ export async function runCompletionStream(
body: { body: {
chatId?: string | null; chatId?: string | null;
persist?: boolean; persist?: boolean;
clientRequestId?: string;
provider: Provider; provider: Provider;
model: string; model: string;
messages: CompletionRequestMessage[]; messages: CompletionRequestMessage[];
@@ -679,86 +615,13 @@ export async function runCompletionStream(
signal: options?.signal, signal: options?.signal,
}); });
if (!response.ok) { await readSseStream(response, (eventName, payload) => {
const fallback = `${response.status} ${response.statusText}`;
let message = fallback;
try {
const body = (await response.json()) as { message?: string };
if (body.message) message = body.message;
} catch {
// keep fallback message
}
throw new Error(message);
}
if (!response.body) {
throw new Error("No response stream");
}
const reader = response.body.getReader();
const decoder = new TextDecoder();
let buffer = "";
let eventName = "message";
let dataLines: string[] = [];
const flushEvent = () => {
if (!dataLines.length) {
eventName = "message";
return;
}
const dataText = dataLines.join("\n");
let payload: any = null;
try {
payload = JSON.parse(dataText);
} catch {
payload = { message: dataText };
}
if (eventName === "meta") handlers.onMeta?.(payload); if (eventName === "meta") handlers.onMeta?.(payload);
else if (eventName === "tool_call") handlers.onToolCall?.(payload); else if (eventName === "tool_call") handlers.onToolCall?.(payload);
else if (eventName === "delta") handlers.onDelta?.(payload); else if (eventName === "delta") handlers.onDelta?.(payload);
else if (eventName === "done") handlers.onDone?.(payload); else if (eventName === "done") handlers.onDone?.(payload);
else if (eventName === "error") handlers.onError?.(payload); else if (eventName === "error") handlers.onError?.(payload);
});
dataLines = [];
eventName = "message";
};
while (true) {
const { value, done } = await reader.read();
if (done) break;
buffer += decoder.decode(value, { stream: true });
let newlineIndex = buffer.indexOf("\n");
while (newlineIndex >= 0) {
const rawLine = buffer.slice(0, newlineIndex);
buffer = buffer.slice(newlineIndex + 1);
const line = rawLine.endsWith("\r") ? rawLine.slice(0, -1) : rawLine;
if (!line) {
flushEvent();
} else if (line.startsWith("event:")) {
eventName = line.slice("event:".length).trim();
} else if (line.startsWith("data:")) {
dataLines.push(line.slice("data:".length).trimStart());
}
newlineIndex = buffer.indexOf("\n");
}
}
buffer += decoder.decode();
if (buffer.length) {
const line = buffer.endsWith("\r") ? buffer.slice(0, -1) : buffer;
if (line.startsWith("event:")) {
eventName = line.slice("event:".length).trim();
} else if (line.startsWith("data:")) {
dataLines.push(line.slice("data:".length).trimStart());
}
}
flushEvent();
} }
export async function attachCompletionStream(chatId: string, handlers: CompletionStreamHandlers, options?: { signal?: AbortSignal }) { export async function attachCompletionStream(chatId: string, handlers: CompletionStreamHandlers, options?: { signal?: AbortSignal }) {
+4 -1
View File
@@ -2,7 +2,10 @@ export function registerServiceWorker() {
if (!import.meta.env.PROD || !("serviceWorker" in navigator)) return; if (!import.meta.env.PROD || !("serviceWorker" in navigator)) return;
window.addEventListener("load", () => { window.addEventListener("load", () => {
void navigator.serviceWorker.register("/sw.js").catch((error: unknown) => { void navigator.serviceWorker
.register("/sw.js", { updateViaCache: "none" })
.then((registration) => registration.update())
.catch((error: unknown) => {
console.warn("Sybil service worker registration failed", error); console.warn("Sybil service worker registration failed", error);
}); });
}); });