diff --git a/.env.example b/.env.example index 89fec9c..2d44424 100644 --- a/.env.example +++ b/.env.example @@ -32,7 +32,7 @@ OPENAI_COMPATIBLE_API_KEY= OPENAI_COMPATIBLE_BASE_URL= OPENAI_COMPATIBLE_MODEL= -# 默认模型 (可选,如 deepseek:deepseek-v4-flash) +# 默认模型 (可选,如 deepseek:deepseek-flash) X_CODE_MODEL= # Tavily 搜索 (可选,免费 1000 次/月,https://tavily.com) diff --git a/.github/workflows/pr-check.yml b/.github/workflows/pr-check.yml index 824d00c..48768c4 100644 --- a/.github/workflows/pr-check.yml +++ b/.github/workflows/pr-check.yml @@ -23,8 +23,6 @@ jobs: - uses: actions/checkout@v7 - uses: pnpm/action-setup@v4 - with: - version: 10 - uses: actions/setup-node@v7 with: @@ -43,8 +41,6 @@ jobs: - uses: actions/checkout@v7 - uses: pnpm/action-setup@v4 - with: - version: 10 - uses: actions/setup-node@v7 with: @@ -67,8 +63,6 @@ jobs: - uses: actions/checkout@v7 - uses: pnpm/action-setup@v4 - with: - version: 10 - uses: actions/setup-node@v7 with: @@ -85,8 +79,6 @@ jobs: - uses: actions/checkout@v7 - uses: pnpm/action-setup@v4 - with: - version: 10 - uses: actions/setup-node@v7 with: @@ -121,8 +113,6 @@ jobs: - uses: actions/checkout@v7 - uses: pnpm/action-setup@v4 - with: - version: 10 - uses: actions/setup-node@v7 with: @@ -141,8 +131,6 @@ jobs: fetch-depth: 0 - uses: pnpm/action-setup@v4 - with: - version: 10 - uses: actions/setup-node@v7 with: diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 0e3c3b2..58178c9 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -59,8 +59,6 @@ jobs: - uses: actions/checkout@v7 - uses: pnpm/action-setup@v4 - with: - version: 10 - uses: actions/setup-node@v7 with: @@ -101,8 +99,6 @@ jobs: fetch-depth: 0 - uses: pnpm/action-setup@v4 - with: - version: 10 - uses: actions/setup-node@v7 with: diff --git a/README.md b/README.md index c1f7162..5a520be 100644 --- a/README.md +++ b/README.md @@ -60,6 +60,8 @@ The current release stores ChatGPT tokens in a plaintext credential file under ` Alternatively, configure at least one provider API key: > **Recommended: [DeepSeek](https://platform.deepseek.com/)** — affordable and capable enough for everyday coding. Promotional credits and prices can change; check the official console for current terms. +> +> The default DeepSeek model is V4.1 Flash (`deepseek-flash`). It receives supported image attachments and tool screenshots natively; X-Code does not send them to a separate vision model. | Variable | Provider | Sign up | | ------------------------------ | ------------------ | --------------------------------------------------------------------------- | @@ -72,6 +74,21 @@ Alternatively, configure at least one provider API key: | `ZHIPU_API_KEY` | Zhipu (GLM) | [open.bigmodel.cn](https://open.bigmodel.cn/usercenter/apikeys) | | `MOONSHOT_API_KEY` | Moonshot (Kimi) | [Choose a service](#moonshot-kimi-endpoints) | +The curated `/model` catalog tracks the current API families below. Older supported entries remain selectable. + +| Provider | Current catalog highlights | Smart default | +| --------- | --------------------------------------------------------------- | ------------------- | +| Anthropic | Claude Fable 5.1, Opus 5, Sonnet 5, Haiku 4.5 | Claude Sonnet 5 | +| OpenAI | GPT-6 Astra, GPT-5.6 Sol / Terra / Luna | GPT-5.6 Sol | +| DeepSeek | DeepSeek V4.1 Flash (native vision), V4 Pro (text only) | DeepSeek V4.1 Flash | +| Alibaba | Qwen3.8 Max / Flash (native vision), Qwen3.7 and Qwen3 variants | Qwen3.8 Max | +| Google | Gemini 3.8 / 3.7 / 3.6 / 3.5 Flash, Gemini 2.5 | Gemini 3.8 Flash | +| xAI | Grok 4.6, Grok 4.20 reasoning / non-reasoning | Grok 4.6 | +| Zhipu | GLM-5.3, GLM-5.3 Flash (native vision), earlier GLM models | GLM-5.3 | +| Moonshot | Kimi K3, K2.7 Code / Highspeed, K2.6 | Kimi K3 | + +GPT-6 Astra is intentionally not the OpenAI smart default: GPT-5.6 Sol remains the safer default for ChatGPT subscription compatibility and lower accidental API cost. Models whose reasoning cannot be disabled automatically map `/thinking off` to their lowest supported effort. + **OpenAI-compatible escape hatch** (vLLM / OpenRouter / internal gateways): set both `OPENAI_COMPATIBLE_API_KEY` and `OPENAI_COMPATIBLE_BASE_URL`, then address models as `custom:`.
@@ -134,7 +151,7 @@ To enable the `webSearch` tool, configure any one of the following: > Tavily is recommended for first-time setup: simpler signup, LLM-optimized responses. When several keys are set, the first in the table order above wins. Set `X_CODE_WEB_SEARCH_PROVIDER` to `tavily`, `brave`, `exa`, `perplexity`, `firecrawl`, or `deepseek` to force a specific provider. > -> **DeepSeek users need no extra key**: when the active model is a DeepSeek model and `DEEPSEEK_API_KEY` is set, `webSearch` automatically uses DeepSeek's built-in server-side web search. Note that each search is billed as a model turn (default `deepseek-v4-flash`), not as a flat search request. +> **DeepSeek users need no extra key**: when the active model is a DeepSeek model and `DEEPSEEK_API_KEY` is set, `webSearch` automatically uses DeepSeek's built-in server-side web search. Note that each search is billed as a model turn (default `deepseek-flash`), not as a flat search request.
@@ -147,7 +164,7 @@ Moonshot/Kimi credentials come from three separate services. A key only works wi - China Open Platform: [platform.kimi.com](https://platform.kimi.com/console/api-keys) → `https://api.moonshot.cn/v1` - International Open Platform: [platform.kimi.ai](https://platform.kimi.ai/console/api-keys) → `https://api.moonshot.ai/v1` -After selecting a Kimi model via `/model`, an endpoint picker appears automatically. +After selecting a Kimi model via `/model`, an endpoint picker appears automatically. On Coding Plan, the stable `kimi-for-coding` wire id currently auto-routes to K2.8 Preview; the public Moonshot API model ids remain unchanged. @@ -312,6 +329,8 @@ Log path: `~/.x-code/logs/debug.log` (Windows: `%USERPROFILE%\.x-code\logs\debug Requires Node.js 22+ and pnpm 10.x. +The repository pins pnpm 10.28.2 through `packageManager`. If pnpm 11 is installed globally, run `corepack enable` and use the repository-local `pnpm` (or `corepack pnpm`); running pnpm 11 directly is intentionally rejected by `engines.pnpm`. + ```bash git clone https://github.com/woai3c/x-code-cli.git cd x-code-cli diff --git a/README.zh-CN.md b/README.zh-CN.md index 3915bc0..6bc2136 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -60,6 +60,8 @@ xc logout # 退出并删除本地 ChatGPT 凭据 也可以配置至少一个厂商的 API Key: > **推荐 [DeepSeek](https://platform.deepseek.com/)**:价格低、国内访问稳定,适合首次试用。赠送额度与价格可能变化,请以官方控制台为准。 +> +> DeepSeek 默认模型为 V4.1 Flash(`deepseek-flash`)。它会原生接收支持的图片附件和工具截图,X-Code 不会将它们发送给其他视觉模型。 | 环境变量 | 厂商 | 注册地址 | | ------------------------------ | ------------------- | --------------------------------------------------------------------------- | @@ -72,6 +74,21 @@ xc logout # 退出并删除本地 ChatGPT 凭据 | `ZHIPU_API_KEY` | 智谱(GLM) | [open.bigmodel.cn](https://open.bigmodel.cn/usercenter/apikeys) | | `MOONSHOT_API_KEY` | Moonshot(Kimi) | [按服务选择](#moonshot-kimi-endpoints) | +内置 `/model` 目录已跟进以下当前 API 型号;仍可用的旧型号继续保留。 + +| 厂商 | 当前目录重点型号 | 智能默认值 | +| --------- | ------------------------------------------------------ | ------------------- | +| Anthropic | Claude Fable 5.1、Opus 5、Sonnet 5、Haiku 4.5 | Claude Sonnet 5 | +| OpenAI | GPT-6 Astra、GPT-5.6 Sol / Terra / Luna | GPT-5.6 Sol | +| DeepSeek | DeepSeek V4.1 Flash(原生视觉)、V4 Pro(仅文本) | DeepSeek V4.1 Flash | +| 阿里通义 | Qwen3.8 Max / Flash(原生视觉)、Qwen3.7 与 Qwen3 系列 | Qwen3.8 Max | +| Google | Gemini 3.8 / 3.7 / 3.6 / 3.5 Flash、Gemini 2.5 | Gemini 3.8 Flash | +| xAI | Grok 4.6、Grok 4.20 推理版 / 非推理版 | Grok 4.6 | +| 智谱 | GLM-5.3、GLM-5.3 Flash(原生视觉)及较早 GLM 型号 | GLM-5.3 | +| Moonshot | Kimi K3、K2.7 Code / Highspeed、K2.6 | Kimi K3 | + +GPT-6 Astra 不会自动成为 OpenAI 智能默认值:保留 GPT-5.6 Sol 可以兼顾 ChatGPT 订阅兼容性,并降低误用高价 API 的风险。对于不能关闭推理的型号,`/thinking off` 会自动降到该型号支持的最低推理档位。 + **OpenAI 兼容接入**(vLLM / OpenRouter / 代理网关等):同时设置 `OPENAI_COMPATIBLE_API_KEY` 与 `OPENAI_COMPATIBLE_BASE_URL`,模型 ID 写成 `custom:`。
@@ -134,7 +151,7 @@ setx DEEPSEEK_API_KEY "sk-..." > 推荐首次配 Tavily:注册简便,返回格式针对 LLM 优化。配置多个 key 时按上表顺序取第一个;也可通过 `X_CODE_WEB_SEARCH_PROVIDER` 显式指定(可选值:`tavily`、`brave`、`exa`、`perplexity`、`firecrawl`、`deepseek`)。 > -> **DeepSeek 用户无需额外 key**:当前模型为 DeepSeek 且已配置 `DEEPSEEK_API_KEY` 时,`webSearch` 自动使用 DeepSeek 内置的服务端联网搜索。注意每次搜索按一次模型调用计费(默认 `deepseek-v4-flash`),而非按搜索次数计费。 +> **DeepSeek 用户无需额外 key**:当前模型为 DeepSeek 且已配置 `DEEPSEEK_API_KEY` 时,`webSearch` 自动使用 DeepSeek 内置的服务端联网搜索。注意每次搜索按一次模型调用计费(默认 `deepseek-flash`),而非按搜索次数计费。
@@ -147,7 +164,7 @@ Moonshot/Kimi 提供三套独立的凭证与端点,API Key 只能用于签发 - 国内开放平台:[platform.kimi.com](https://platform.kimi.com/console/api-keys) → `https://api.moonshot.cn/v1` - 国际开放平台:[platform.kimi.ai](https://platform.kimi.ai/console/api-keys) → `https://api.moonshot.ai/v1` -通过 `/model` 选择 Kimi 模型后,X-Code CLI 会自动显示端点选择器。 +通过 `/model` 选择 Kimi 模型后,X-Code CLI 会自动显示端点选择器。Coding Plan 的稳定 wire id `kimi-for-coding` 当前会自动路由到 K2.8 Preview;Moonshot 开放平台的模型 ID 不变。 @@ -312,6 +329,8 @@ set DEBUG_STDOUT=1 && xc 需要 Node.js 22+ 和 pnpm 10.x。 +仓库通过 `packageManager` 固定使用 pnpm 10.28.2。如果全局安装的是 pnpm 11,请先运行 `corepack enable`,再在仓库内使用 `pnpm`(或直接使用 `corepack pnpm`);`engines.pnpm` 会有意拒绝 pnpm 11 直接执行。 + ```bash git clone https://github.com/woai3c/x-code-cli.git cd x-code-cli diff --git a/package.json b/package.json index aa4fec6..34c6fdd 100644 --- a/package.json +++ b/package.json @@ -1,5 +1,6 @@ { "private": true, + "packageManager": "pnpm@10.28.2", "type": "module", "engines": { "node": ">=22.0.0", diff --git a/packages/cli/src/ui/app/App.tsx b/packages/cli/src/ui/app/App.tsx index 83b8f91..aa1ab50 100644 --- a/packages/cli/src/ui/app/App.tsx +++ b/packages/cli/src/ui/app/App.tsx @@ -2039,7 +2039,7 @@ export function App({ // automatically on /model switch since switchModel updates // state.modelId. When the model has a reasoning-effort tier configured // (via the /model tier picker), it appends next to the name — e.g. - // "deepseek-v4-flash · High". + // "deepseek-flash · High". modelLabel={ reasoningTierLabel ? `${renderModelLabel(state.modelId)} · ${reasoningTierLabel}` diff --git a/packages/cli/src/ui/utils.ts b/packages/cli/src/ui/utils.ts index 898a13f..a02282e 100644 --- a/packages/cli/src/ui/utils.ts +++ b/packages/cli/src/ui/utils.ts @@ -108,7 +108,8 @@ const SHELL_LABELS: Record = { export function getToolLabel(toolName: string): string { const n = normalizeToolName(toolName) if (n === 'shell' || n === 'bash') return SHELL_LABELS[getShellProvider().type] ?? 'Shell' - if (n === 'readfile' || n === 'read' || n === 'fileingest') return 'Read' + if (n === 'fileingest') return 'Attach' + if (n === 'readfile' || n === 'read') return 'Read' if (n === 'writefile' || n === 'write') return 'Write' if (n === 'edit' || n === 'update') return 'Update' if (n === 'glob') return 'Glob' @@ -192,6 +193,7 @@ export interface ReadGroupSummary { const TABLE_OUTPUT_MAX_LINES = 30 export function formatReadGroupSummary(tools: readonly DisplayToolCall[]): ReadGroupSummary { + let attachCount = 0 let readCount = 0 let grepCount = 0 let globCount = 0 @@ -200,7 +202,11 @@ export function formatReadGroupSummary(tools: readonly DisplayToolCall[]): ReadG for (const tc of tools) { const n = normalizeToolName(tc.toolName) - if (n === 'read' || n === 'readfile' || n === 'fileingest') { + if (n === 'fileingest') { + attachCount++ + const p = (tc.input.filePath as string) || (tc.input.file_path as string) || (tc.input.path as string) || '' + if (p) readPaths.push(basename(p)) + } else if (n === 'read' || n === 'readfile') { readCount++ const p = (tc.input.filePath as string) || (tc.input.file_path as string) || (tc.input.path as string) || '' if (p) readPaths.push(basename(p)) @@ -214,6 +220,7 @@ export function formatReadGroupSummary(tools: readonly DisplayToolCall[]): ReadG } const clauses: string[] = [] + if (attachCount > 0) clauses.push(`attached ${attachCount} file${attachCount === 1 ? '' : 's'}`) if (readCount > 0) clauses.push(`read ${readCount} file${readCount === 1 ? '' : 's'}`) if (grepCount > 0) clauses.push(`searched for ${grepCount} pattern${grepCount === 1 ? '' : 's'}`) if (globCount > 0) clauses.push(`globbed ${globCount} pattern${globCount === 1 ? '' : 's'}`) diff --git a/packages/cli/tests/e2e/README.md b/packages/cli/tests/e2e/README.md index aa46d6c..ce464cc 100644 --- a/packages/cli/tests/e2e/README.md +++ b/packages/cli/tests/e2e/README.md @@ -16,7 +16,7 @@ pnpm test:e2e ``` The runner detects which `*_API_KEY` you have set, lists the matching models, -and asks you to pick one. Default is `deepseek:deepseek-v4-flash` (cheap, fast). +and asks you to pick one. Default is `deepseek:deepseek-flash` (V4.1 Flash, cheap, fast, and vision-capable). ## Flags @@ -34,8 +34,8 @@ pnpm test:e2e --max-turns 8 # cap agent loop turns ## Cost -`deepseek-v4-flash` runs the whole suite (26 scenarios, ~50-100K tokens total) -for roughly **$0.10–0.18 per full run**. Each scenario takes 5–30 seconds. +`deepseek-flash` runs the whole suite (26 scenarios, ~50-100K tokens total). +Cost varies with input/output mix, cache hits, and DeepSeek peak/off-peak pricing. Each scenario takes 5–30 seconds. Full suite: 4–8 minutes typically. If you only want to verify changes near a specific area, use `--filter` and @@ -85,7 +85,7 @@ export default scenario | Method | What it does | | ----------------------------- | ---------------------------------------------------------------- | | `ctx.tmpDir` | Absolute path to this scenario's temp directory. CLI runs there. | -| `ctx.modelId` | Resolved model id (e.g. `deepseek:deepseek-v4-flash`). | +| `ctx.modelId` | Resolved model id (e.g. `deepseek:deepseek-flash`). | | `ctx.writeFile(rel, content)` | Write a file inside tmpDir. | | `ctx.readFile(rel)` | Read a file. | | `ctx.fileExists(rel)` | Returns `Promise`. | diff --git a/packages/cli/tests/e2e/framework/models.ts b/packages/cli/tests/e2e/framework/models.ts index 35910ad..3e0265d 100644 --- a/packages/cli/tests/e2e/framework/models.ts +++ b/packages/cli/tests/e2e/framework/models.ts @@ -7,39 +7,70 @@ import { fileURLToPath } from 'node:url' const __dirname = path.dirname(fileURLToPath(import.meta.url)) const REPO_ROOT = path.resolve(__dirname, '..', '..', '..', '..', '..') -export const DEFAULT_MODEL = 'deepseek:deepseek-v4-flash' +export const DEFAULT_MODEL = 'deepseek:deepseek-flash' /** Provider env-var → list of model ids that can be selected when that key is set. * Keep aligned with `packages/core/src/types/index.ts::PROVIDER_DETECTION_ORDER`. */ const PROVIDER_MODELS: Record = { - DEEPSEEK_API_KEY: ['deepseek:deepseek-v4-flash', 'deepseek:deepseek-v4-pro'], + DEEPSEEK_API_KEY: ['deepseek:deepseek-flash', 'deepseek:deepseek-v4-pro'], ANTHROPIC_API_KEY: [ + 'anthropic:claude-fable-5-1', + 'anthropic:claude-opus-5', 'anthropic:claude-fable-5', 'anthropic:claude-opus-4-8', 'anthropic:claude-sonnet-5', 'anthropic:claude-haiku-4-5', ], - OPENAI_API_KEY: ['openai:gpt-5.6-sol', 'openai:gpt-5.6-terra', 'openai:gpt-5.6-luna', 'openai:gpt-5.4-mini'], - GOOGLE_GENERATIVE_AI_API_KEY: ['google:gemini-3.5-flash', 'google:gemini-2.5-pro', 'google:gemini-2.5-flash'], - XAI_API_KEY: ['xai:grok-4.3', 'xai:grok-4.5'], - ALIBABA_API_KEY: ['alibaba:qwen3.7-max', 'alibaba:qwen3.7-plus', 'alibaba:qwen3-coder-plus', 'alibaba:qwq-plus'], - ZHIPU_API_KEY: ['zhipu:glm-5.2', 'zhipu:glm-5', 'zhipu:glm-4.7'], - MOONSHOT_API_KEY: ['moonshotai:kimi-k2.6', 'moonshotai:kimi-k3'], + OPENAI_API_KEY: [ + 'openai:gpt-6-astra', + 'openai:gpt-5.6-sol', + 'openai:gpt-5.6-terra', + 'openai:gpt-5.6-luna', + 'openai:gpt-5.4-mini', + ], + GOOGLE_GENERATIVE_AI_API_KEY: [ + 'google:gemini-3.8-flash', + 'google:gemini-3.7-flash', + 'google:gemini-3.6-flash', + 'google:gemini-3.5-flash', + 'google:gemini-2.5-pro', + 'google:gemini-2.5-flash', + ], + XAI_API_KEY: ['xai:grok-4.6', 'xai:grok-4.20', 'xai:grok-4.20-non-reasoning', 'xai:grok-4.5'], + ALIBABA_API_KEY: [ + 'alibaba:qwen3.8-max', + 'alibaba:qwen3.8-flash', + 'alibaba:qwen3.7-max', + 'alibaba:qwen3.7-plus', + 'alibaba:qwen3.7-flash', + 'alibaba:qwen3-coder-plus', + 'alibaba:qwq-plus', + ], + ZHIPU_API_KEY: ['zhipu:glm-5.3', 'zhipu:glm-5.3-flash', 'zhipu:glm-5.2', 'zhipu:glm-5', 'zhipu:glm-4.7'], + MOONSHOT_API_KEY: [ + 'moonshotai:kimi-k3', + 'moonshotai:kimi-k2.7-code', + 'moonshotai:kimi-k2.7-code-highspeed', + 'moonshotai:kimi-k2.6', + ], } /** Short aliases — accepted on CLI `--model` flag. Aligns with product's * `MODEL_ALIASES` table; keep them in sync. */ export const ALIASES: Record = { - fable: 'anthropic:claude-fable-5', + fable: 'anthropic:claude-fable-5-1', sonnet: 'anthropic:claude-sonnet-5', - opus: 'anthropic:claude-opus-4-8', + opus: 'anthropic:claude-opus-5', haiku: 'anthropic:claude-haiku-4-5', + gpt6: 'openai:gpt-6-astra', + astra: 'openai:gpt-6-astra', gpt5: 'openai:gpt-5.6-sol', - gemini: 'google:gemini-3.5-flash', - deepseek: 'deepseek:deepseek-v4-flash', + gemini: 'google:gemini-3.8-flash', + deepseek: 'deepseek:deepseek-flash', 'deepseek-pro': 'deepseek:deepseek-v4-pro', - qwen: 'alibaba:qwen3.7-max', - glm: 'zhipu:glm-5.2', + qwen: 'alibaba:qwen3.8-max', + grok: 'xai:grok-4.6', + glm: 'zhipu:glm-5.3', kimi: 'moonshotai:kimi-k3', } diff --git a/packages/cli/tests/e2e/scenarios/14-large-file.ts b/packages/cli/tests/e2e/scenarios/14-large-file.ts index 30d2f23..07e9744 100644 --- a/packages/cli/tests/e2e/scenarios/14-large-file.ts +++ b/packages/cli/tests/e2e/scenarios/14-large-file.ts @@ -13,7 +13,7 @@ const scenario: Scenario = { lines.push(`line ${i}: lorem ipsum dolor sit amet`) } // 关键标记放在尾部。前 4999 行模式完全一致 — 一个偷懒的模型只看 head - // 就会"按模式外推"答 `lorem ipsum dolor sit amet`(deepseek-v4-flash 实测会); + // 就会"按模式外推"答 `lorem ipsum dolor sit amet`(deepseek-flash 实测会); // 只有真的二次调 readFile(offset≈5000) 拿到尾部才能引用出这个 token。 lines[4999] = 'line 5000: FINAL_SENTINEL_TOKEN_XYZ' await ctx.writeFile('big.txt', lines.join('\n')) diff --git a/packages/cli/tests/openai-auth-transition-state.test.ts b/packages/cli/tests/openai-auth-transition-state.test.ts index 5388bf2..02ef5ce 100644 --- a/packages/cli/tests/openai-auth-transition-state.test.ts +++ b/packages/cli/tests/openai-auth-transition-state.test.ts @@ -69,7 +69,7 @@ describe('OpenAI auth transition state', () => { it('falls back or disables deterministically when the new account has no models', () => { expect(planOpenAIModelReconciliation('openai:old', [], ['deepseek'], 'blocked', (id) => id)).toMatchObject({ - modelId: 'deepseek:deepseek-v4-flash', + modelId: 'deepseek:deepseek-flash', }) expect(planOpenAIModelReconciliation('openai:old', [], [], 'blocked', (id) => id)).toEqual({ blockedMessage: 'blocked', diff --git a/packages/cli/tests/pty/tui-input.test.ts b/packages/cli/tests/pty/tui-input.test.ts index d5f3330..6c7111f 100644 --- a/packages/cli/tests/pty/tui-input.test.ts +++ b/packages/cli/tests/pty/tui-input.test.ts @@ -115,10 +115,10 @@ describe('TUI input and lifecycle', () => { const raw = harness.raw() const oldReplyTail = raw.lastIndexOf('initial-finished') const queuedUser = raw.lastIndexOf('@"queued attachment.txt" analyze this file') - const readSummary = raw.lastIndexOf('Read') + const attachSummary = raw.lastIndexOf('Attach') expect(oldReplyTail).toBeGreaterThanOrEqual(0) expect(queuedUser).toBeGreaterThan(oldReplyTail) - expect(readSummary).toBeGreaterThan(queuedUser) + expect(attachSummary).toBeGreaterThan(queuedUser) }, { beforeStart: async (workspace) => { diff --git a/packages/cli/tests/stdout-writer-spacing.test.ts b/packages/cli/tests/stdout-writer-spacing.test.ts index 94bbd60..d1aea91 100644 --- a/packages/cli/tests/stdout-writer-spacing.test.ts +++ b/packages/cli/tests/stdout-writer-spacing.test.ts @@ -152,7 +152,7 @@ describe('stdout writer spacing', () => { }) const plain = output.replace(/\x1b\[[0-9;]*m/g, '') - expect(plain).toContain('Read 2 files(invoice.pdf, analysis.docx)') + expect(plain).toContain('Attached 2 files(invoice.pdf, analysis.docx)') expect(plain.match(/Prepared for analysis\./g)).toBeNull() }) diff --git a/packages/cli/tests/utils.test.ts b/packages/cli/tests/utils.test.ts index 73ea176..2a787e6 100644 --- a/packages/cli/tests/utils.test.ts +++ b/packages/cli/tests/utils.test.ts @@ -50,8 +50,8 @@ describe('CLI tool result summaries', () => { }) describe('local file ingestion tool display', () => { - it('renders the built-in preflight as Read with portable file-name previews', () => { - expect(getToolLabel('fileIngest')).toBe('Read') + it('renders the built-in preflight as Attach with portable file-name previews', () => { + expect(getToolLabel('fileIngest')).toBe('Attach') expect(getToolInputPreview('fileIngest', { filePath: 'C:\\reports\\invoice.pdf' })).toBe('C:\\reports\\invoice.pdf') }) }) diff --git a/packages/core/package.json b/packages/core/package.json index 2fb2d6b..2489d5b 100644 --- a/packages/core/package.json +++ b/packages/core/package.json @@ -28,7 +28,7 @@ "dependencies": { "@ai-sdk/alibaba": "^2.0.0", "@ai-sdk/anthropic": "^4.0.0", - "@ai-sdk/deepseek": "^3.0.28", + "@ai-sdk/deepseek": "^3.0.45", "@ai-sdk/google": "^4.0.0", "@ai-sdk/moonshotai": "^3.0.0", "@ai-sdk/openai": "^4.0.0", diff --git a/packages/core/src/agent/compression.ts b/packages/core/src/agent/compression.ts index 796a970..3056f40 100644 --- a/packages/core/src/agent/compression.ts +++ b/packages/core/src/agent/compression.ts @@ -362,6 +362,9 @@ async function finishLightweightCompression( ): Promise { await markBoundaryAndReflush(state, undefined, candidate) setTrackedTranscript(state, candidate) + // A compacted transcript may no longer contain the file bodies that made + // these de-dup entries valid. Let readFile deliver them again on demand. + state.readFileCache.clear() state.lastInputTokens = 0 markExpectedCacheMiss(state, 'compaction') emitCompactionHook(hookCtx, { @@ -389,6 +392,7 @@ async function summarizeLoopState( ) await markBoundaryAndReflush(state, compressed.summary, compressed.trackedMessages) setTrackedTranscript(state, compressed.trackedMessages) + state.readFileCache.clear() await recordCompressionUsage(state, compressed, callbacks, hookCtx) state.lastInputTokens = 0 markExpectedCacheMiss(state, 'compaction') diff --git a/packages/core/src/agent/context-window.ts b/packages/core/src/agent/context-window.ts index 85753bd..7fc8c09 100644 --- a/packages/core/src/agent/context-window.ts +++ b/packages/core/src/agent/context-window.ts @@ -49,41 +49,57 @@ export function setContextWindowOverride(value: unknown): number | undefined { /** Context window sizes per model (tokens). */ const MODEL_CONTEXT_WINDOWS: ReadonlyMap = new Map([ // Anthropic + ['anthropic:claude-fable-5-1', 1000000], + ['anthropic:claude-opus-5', 1000000], ['anthropic:claude-fable-5', 1000000], ['anthropic:claude-opus-4-8', 1000000], ['anthropic:claude-sonnet-5', 1000000], ['anthropic:claude-haiku-4-5', 200000], // OpenAI + ['openai:gpt-6-astra', 1050000], ['openai:gpt-5.6-sol', 1047576], ['openai:gpt-5.6-terra', 1047576], ['openai:gpt-5.6-luna', 1047576], ['openai:gpt-5.4-mini', 1047576], ['openai:gpt-5.4-nano', 1047576], // Google + ['google:gemini-3.8-flash', 1048576], + ['google:gemini-3.7-flash', 1048576], + ['google:gemini-3.6-flash', 1048576], ['google:gemini-3.5-flash', 1000000], ['google:gemini-2.5-pro', 1000000], ['google:gemini-2.5-flash', 1000000], // DeepSeek + ['deepseek:deepseek-flash', 1000000], ['deepseek:deepseek-v4-flash', 1000000], ['deepseek:deepseek-v4-pro', 1000000], - // Alibaba — per DashScope docs: qwen3.7-max and qwen3-coder-plus extend to 1M; + // Alibaba — current Qwen3.8 models, qwen3.7-max/plus, and qwen3-coder-plus extend to 1M; // qwen-max still caps at 32k. Values verified against // https://help.aliyun.com/zh/model-studio/models. + ['alibaba:qwen3.8-max', 1000000], + ['alibaba:qwen3.8-flash', 1000000], ['alibaba:qwen3.7-max', 1000000], - ['alibaba:qwen3.7-plus', 131072], + ['alibaba:qwen3.7-plus', 1000000], + ['alibaba:qwen3.7-flash', 1000000], ['alibaba:qwen3-coder-plus', 1000000], ['alibaba:qwq-plus', 131072], ['alibaba:qwen-max', 32768], - // xAI — grok-4.5 has 500k window; grok-4.3 has 1M. - ['xai:grok-4.5', 512000], + // xAI — Grok 4.6/4.5 have 500k windows; Grok 4.20/4.3 have 1M. + ['xai:grok-4.6', 500000], + ['xai:grok-4.20', 1000000], + ['xai:grok-4.20-non-reasoning', 1000000], + ['xai:grok-4.5', 500000], ['xai:grok-4.3', 1000000], // Zhipu + ['zhipu:glm-5.3', 1000000], + ['zhipu:glm-5.3-flash', 1000000], ['zhipu:glm-5.2', 1000000], ['zhipu:glm-5', 200000], ['zhipu:glm-4.7', 128000], // Moonshot ['moonshotai:kimi-k3', 1000000], ['moonshotai:kimi-k2.7-code', 262144], + ['moonshotai:kimi-k2.7-code-highspeed', 262144], ['moonshotai:kimi-k2.6', 262144], ]) @@ -125,15 +141,31 @@ export function getCompressionThreshold(modelId: string): number { */ const DEFAULT_MAX_OUTPUT_TOKENS = 16384 const MODEL_MAX_OUTPUT_TOKENS: ReadonlyMap = new Map([ - // DeepSeek V4: both flash and pro advertise up to 384K output tokens. + // Current flagships with provider-documented extended output ceilings. + ['anthropic:claude-fable-5-1', 128000], + ['anthropic:claude-opus-5', 128000], + ['anthropic:claude-fable-5', 128000], + ['anthropic:claude-opus-4-8', 128000], + ['anthropic:claude-sonnet-5', 128000], + ['anthropic:claude-haiku-4-5', 64000], + ['openai:gpt-6-astra', 128000], + ['google:gemini-3.8-flash', 65536], + ['google:gemini-3.7-flash', 65536], + ['google:gemini-3.6-flash', 65536], + // DeepSeek V4/V4.1: flash and pro advertise up to 384K output tokens. // We cap at a generous but conservative 131072 to avoid edge-case 400s. + ['deepseek:deepseek-flash', 131072], ['deepseek:deepseek-v4-flash', 131072], ['deepseek:deepseek-v4-pro', 131072], - // Alibaba — Qwen3.7 models support 32768 (non-thinking) / 81920 (thinking). + // Alibaba — Qwen3.8 and multimodal Qwen3.7 Plus support 64k output. + ['alibaba:qwen3.8-max', 65536], + ['alibaba:qwen3.8-flash', 65536], + ['alibaba:qwen3.7-plus', 65536], + ['alibaba:qwen3.7-flash', 65536], + // Older Qwen3.7 models support 32768 (non-thinking) / 81920 (thinking). // We cap at the non-thinking ceiling so the request always succeeds. ['alibaba:qwen-max', 8192], ['alibaba:qwen3.7-max', 32000], - ['alibaba:qwen3.7-plus', 32000], ['alibaba:qwen3-coder-plus', 32000], ['alibaba:qwq-plus', 32000], ]) diff --git a/packages/core/src/agent/file-ingest.ts b/packages/core/src/agent/file-ingest.ts index 56a1ebd..d989fd0 100644 --- a/packages/core/src/agent/file-ingest.ts +++ b/packages/core/src/agent/file-ingest.ts @@ -659,6 +659,7 @@ export async function ingestFile( parts.push({ type: 'text', text: buildCompressionCaption(compressed) }) onNotice?.(`Normalized image: ${formatBytes(buffer.length)} → ${formatBytes(compressed.data.length)}`) } + parts.push({ type: 'text', text: BUILT_IN_MEDIA_ANALYSIS_NOTE }) return parts } @@ -732,7 +733,11 @@ function hasCompleteLocalFile(parts: IngestedPart[], filePath: string): boolean if (part.type !== 'text' || !part.text.startsWith('<>') if (openingEnd === -1 || !part.text.slice(0, openingEnd).includes(pathAttribute)) return false - return part.text.indexOf('<>', openingEnd + 2) !== -1 + if (part.text.indexOf('<>', openingEnd + 2) !== -1) return true + return ( + part.text.slice(0, openingEnd).includes('kind="image"') && + parts.some((candidate) => candidate.type === 'file' && candidate.mediaType.startsWith('image/')) + ) }) } diff --git a/packages/core/src/agent/local-media.ts b/packages/core/src/agent/local-media.ts index 3c44ceb..1929a2b 100644 --- a/packages/core/src/agent/local-media.ts +++ b/packages/core/src/agent/local-media.ts @@ -16,7 +16,7 @@ export type ProcessedLocalPart = export const BUILT_IN_MEDIA_ANALYSIS_NOTE = '[Built-in local media processing succeeded. Analyze the supplied content directly. ' + - 'Do not invoke shell, Node.js, Python, FFmpeg, or other external programs merely to re-read, parse, OCR, ' + + 'Do not invoke readFile, shell, Node.js, Python, FFmpeg, or other external programs merely to re-read, parse, OCR, ' + 'transcribe, or independently validate values from this attachment. Use external programs only if the built-in ' + 'pipeline reports a failure, or the user explicitly asks for conversion, codec diagnostics, or independent validation.]' diff --git a/packages/core/src/agent/loop.ts b/packages/core/src/agent/loop.ts index 8110b96..5af3e9b 100644 --- a/packages/core/src/agent/loop.ts +++ b/packages/core/src/agent/loop.ts @@ -21,8 +21,13 @@ import { listMcpResources, readMcpResource } from '../mcp/resources.js' import { bridgeMcpTool, toSystemPromptEntries } from '../mcp/tool-bridge.js' import { listAgentsTool, sendMessageTool } from '../peers/tools.js' import { applyCacheControl, openAICacheComparisonTtlMs } from '../providers/cache-control.js' -import { withZhipuReasoningHeader } from '../providers/registry.js' -import { getReasoningLevel, getThinkingProviderOptions, mergeThinkingOptions } from '../providers/thinking.js' +import { withXaiReasoningHeader, withZhipuReasoningHeader } from '../providers/registry.js' +import { + getReasoningEffort, + getReasoningLevel, + getThinkingProviderOptions, + mergeThinkingOptions, +} from '../providers/thinking.js' import { createActivateSkillTool } from '../tools/activate-skill.js' import { BROWSER_VISUAL_CHECK_TOOL_NAME, browserVisualCheck } from '../tools/browser-visual-check.js' import { createGetGoalTool } from '../tools/get-goal.js' @@ -516,11 +521,14 @@ async function runTurnAttempt( // /thinking toggle. If the user explicitly chose a reasoning effort level // for this model (stored in config.modelReasoningEffort), we use it. const effort = userConfig.modelReasoningEffort?.[options.modelId] + const effectiveEffort = getReasoningEffort(options.modelId, options.thinking ?? false, effort) const reasoningLevel = getReasoningLevel(options.modelId, options.thinking ?? false, effort) const thinkingOptions = getThinkingProviderOptions(options.modelId, options.thinking ?? false, effort) const mergedProviderOptions = mergeThinkingOptions(cached.providerOptions, thinkingOptions) - const requestHeaders = - options.modelId.split(':')[0] === 'zhipu' ? withZhipuReasoningHeader(cached.headers, effort) : cached.headers + const provider = options.modelId.split(':')[0] + let requestHeaders = cached.headers + if (provider === 'zhipu') requestHeaders = withZhipuReasoningHeader(requestHeaders, effectiveEffort) + if (provider === 'xai') requestHeaders = withXaiReasoningHeader(requestHeaders, effectiveEffort) const requestTimestamp = new Date().toISOString() let result: StreamResult @@ -600,7 +608,7 @@ async function runTurnAttempt( // notices and retry once. stripBinaryPartsFromMessages returns false // when nothing matched — then the bad part isn't in a shape we // recognize, so fall through and report instead of looping forever. - if (stripBinaryPartsFromMessages(state.messages)) { + if (stripBinaryPartsFromMessages(state.messages, state.readFileCache)) { recalculateContextSecurity(state) state.transcriptRequiresSnapshot = true await flushPendingMessages(state) diff --git a/packages/core/src/agent/provider-compat.ts b/packages/core/src/agent/provider-compat.ts index 7cd6f90..727c808 100644 --- a/packages/core/src/agent/provider-compat.ts +++ b/packages/core/src/agent/provider-compat.ts @@ -24,8 +24,8 @@ import type { ToolImage } from './messages.js' // ── Image/PDF downgrade for text-only providers ─────────────────────────── // -// If the active provider can't receive image/file parts (DeepSeek today, -// plus `custom` unless the user opts in), walk every message that would be +// If the active model can't receive image/file parts (for example DeepSeek V4 +// Pro, plus `custom` unless the user opts in), walk every message that would be // sent on the next turn and replace each binary part with something the // provider CAN accept. // @@ -320,7 +320,10 @@ function imagePartToBuffer(part: { image: unknown; mediaType?: string }): Buffer * retry the turn. Returns true only when something actually changed, so the * caller can tell "retry is worthwhile" from "the bad part isn't in a shape * we recognize — report the error instead of looping". */ -export function stripBinaryPartsFromMessages(messages: ModelMessage[]): boolean { +export function stripBinaryPartsFromMessages( + messages: ModelMessage[], + deliveredFileCache?: { clear(): void }, +): boolean { let changed = false const fileNotice = (mediaType?: string) => `[File attachment omitted (${mediaType ?? 'binary'}) — the provider rejected it; removed so the session can continue.]` @@ -358,6 +361,7 @@ export function stripBinaryPartsFromMessages(messages: ModelMessage[]): boolean } } } + if (changed) deliveredFileCache?.clear() return changed } diff --git a/packages/core/src/agent/vision-fallback.ts b/packages/core/src/agent/vision-fallback.ts index c0a885b..d1b9908 100644 --- a/packages/core/src/agent/vision-fallback.ts +++ b/packages/core/src/agent/vision-fallback.ts @@ -47,6 +47,7 @@ export interface VisionUsageEvent { const VISION_MODELS: Record = { google: { modelId: 'google:gemini-2.5-flash', label: 'Gemini 2.5 Flash' }, zhipu: { modelId: 'zhipu:glm-4.6v', label: 'GLM-4.6V' }, + deepseek: { modelId: 'deepseek:deepseek-flash', label: 'DeepSeek V4.1 Flash' }, alibaba: { modelId: 'alibaba:qwen3-vl-flash', label: 'Qwen3-VL Flash' }, openai: { modelId: 'openai:gpt-5.4-mini', label: 'GPT-5.4 Mini' }, anthropic: { modelId: 'anthropic:claude-haiku-4-5', label: 'Claude Haiku 4.5' }, @@ -59,8 +60,9 @@ const VISION_MODELS: Record = { * last. Gemini 2.5 Flash leads because its free tier is the most * generous (1500/day) and the model is also the strongest at the * free price point. GLM-4.6V is second because it's cheap/free - * and reachable from China without a proxy. */ -const VISION_PRIORITY = ['google', 'zhipu', 'alibaba', 'openai', 'anthropic', 'moonshotai', 'xai'] + * and reachable from China without a proxy. DeepSeek V4.1 Flash follows as + * the cheapest configured paid fallback with native vision. */ +const VISION_PRIORITY = ['google', 'zhipu', 'deepseek', 'alibaba', 'openai', 'anthropic', 'moonshotai', 'xai'] /** * Pick the best available vision sub-agent given the keys the user has diff --git a/packages/core/src/knowledge/memory/inference.ts b/packages/core/src/knowledge/memory/inference.ts index 06a4816..4163460 100644 --- a/packages/core/src/knowledge/memory/inference.ts +++ b/packages/core/src/knowledge/memory/inference.ts @@ -6,7 +6,7 @@ import { z } from 'zod' import type { MemoryReasoningMode } from '../../config/index.js' import { providerOf } from '../../providers/capabilities.js' import { getOpenAIChatGPTReasoningTiers } from '../../providers/openai-chatgpt-models.js' -import { getReasoningLevel, getThinkingProviderOptions } from '../../providers/thinking.js' +import { acceptsReasoningControl, getReasoningLevel, getThinkingProviderOptions } from '../../providers/thinking.js' import { debugLog } from '../../utils.js' export type MemoryReasoningControl = 'off' | 'low' | 'provider-default' @@ -174,6 +174,7 @@ function inferenceSettings( omitTemperature: boolean, ): MemoryInferenceSettings { if (!modelId || control === 'provider-default') return { maxOutputTokens } + if (!acceptsReasoningControl(modelId)) return { maxOutputTokens } if (control === 'off') { const reasoning = getReasoningLevel(modelId, false) const providerOptions = getThinkingProviderOptions(modelId, false) @@ -186,7 +187,7 @@ function inferenceSettings( } const provider = providerOf(modelId) if (provider === 'alibaba' || provider === 'zhipu') { - const providerOptions = getThinkingProviderOptions(modelId, true) + const providerOptions = getThinkingProviderOptions(modelId, true, 'low') return { maxOutputTokens, ...(Object.keys(providerOptions).length ? { providerOptions } : {}) } } if (provider === 'custom') return { maxOutputTokens } @@ -201,9 +202,11 @@ function requiresThinking(modelId: string): boolean { if (provider === 'openai') { const chatGPTTiers = getOpenAIChatGPTReasoningTiers(modelId) if (chatGPTTiers?.length && !chatGPTTiers.some((tier) => tier.value === 'none')) return true - return /(?:^|:)(?:gpt-5(?:$|[-.])|o[134](?:$|[-.]))/.test(normalized) + return /(?:^|:)(?:gpt-(?:5|6)(?:$|[-.])|o[134](?:$|[-.]))/.test(normalized) } - return provider === 'anthropic' && /claude-opus-4-5/.test(normalized) + if (provider === 'anthropic') return /claude-(?:fable-5|opus-4-5)/.test(normalized) + if (provider === 'zhipu') return /glm-5\.3(?:$|-)/.test(normalized) + return provider === 'xai' && /grok-(?:4\.6|4\.20)$/.test(normalized) } function rejectControl(modelId: string, control: MemoryReasoningControl): void { diff --git a/packages/core/src/providers/capabilities.ts b/packages/core/src/providers/capabilities.ts index b393909..c188544 100644 --- a/packages/core/src/providers/capabilities.ts +++ b/packages/core/src/providers/capabilities.ts @@ -37,7 +37,7 @@ const CAPS: Record = { moonshotai: { image: true, pdf: true, audio: false, filesApi: true, toolImageTransport: 'user-message' }, alibaba: { image: true, pdf: true, audio: false, filesApi: true, toolImageTransport: 'user-message' }, zhipu: { image: true, pdf: true, audio: false, filesApi: true, toolImageTransport: 'user-message' }, - deepseek: { image: false, pdf: false, audio: false, filesApi: false, toolImageTransport: 'unsupported' }, + deepseek: { image: true, pdf: false, audio: false, filesApi: true, toolImageTransport: 'user-message' }, custom: { image: false, pdf: false, audio: false, filesApi: false, toolImageTransport: 'unsupported' }, } @@ -141,8 +141,8 @@ export function capabilitiesOf(modelId: string): ProviderCapabilities { /** Can this specific MODEL natively see images? Unlike `capabilitiesOf` (which * is provider-level — "does the API accept image parts"), this is per-model, * because providers mix vision and text-only models under one id namespace - * (Qwen-VL vs Qwen-Max, GLM-4V vs GLM-5, kimi-k2.6 is multimodal but a plain - * DeepSeek is not). Used to gate the browser agent's `--caps vision` so a + * (DeepSeek Flash vs V4 Pro, Qwen-VL vs Qwen-Max, GLM-4V vs GLM-5). Used to + * gate the browser agent's `--caps vision` so a * text-only model never gets screenshots it can't read. * * Resolution: alias-expand, look the id up in the curated catalog and trust diff --git a/packages/core/src/providers/catalog.ts b/packages/core/src/providers/catalog.ts index 3d8d38f..5ee5c82 100644 --- a/packages/core/src/providers/catalog.ts +++ b/packages/core/src/providers/catalog.ts @@ -20,8 +20,8 @@ export interface ProviderModel { * just whether its provider's API accepts image parts. Drives the browser * agent's visual gating (modelSupportsVision) so a text-only model never * gets `--caps vision` / screenshots. Set per-model because providers mix - * vision and text-only models under one id namespace (e.g. Qwen-VL vs - * Qwen-Max, GLM-4V vs GLM-5). */ + * vision and text-only models under one id namespace (e.g. DeepSeek Flash + * vs V4 Pro, Qwen-VL vs Qwen-Max, GLM-4V vs GLM-5). */ vision: boolean } @@ -36,6 +36,14 @@ export interface ReasoningTierOption { description: string } +export interface ReasoningTierProfile { + /** Restrict this profile to matching model ids. Omit for a provider-wide profile. */ + modelPattern?: RegExp + options: readonly ReasoningTierOption[] + /** Lowest valid effort for models that reject disabling reasoning. */ + offValue?: string +} + export interface ProviderInfo { /** Provider key used in `:` ids and config maps. */ name: string @@ -48,9 +56,9 @@ export interface ProviderInfo { /** Hand-curated models shown in the interactive `/model` picker. Users can * still type any full id into `/model :` for variants * not listed here. Vision flags reflect model FAMILY: Claude / GPT / - * Gemini / Grok flagships and Kimi K2.x are multimodal; DeepSeek and the - * Qwen-Max / GLM text flagships are text-only; the dedicated *-VL / - * GLM-4V / *-vision-preview models see images. */ + * Gemini / Grok flagships, DeepSeek Flash, and Kimi K2.x are multimodal; + * Qwen-Max / GLM text flagships and DeepSeek V4 Pro are text-only; the + * dedicated *-VL / GLM-4V / *-vision-preview models see images. */ models: readonly ProviderModel[] /** Providers that serve multiple endpoints for the same API (regional * platforms, plan-specific gateways, etc.). When a user picks a model @@ -67,12 +75,12 @@ export interface ProviderInfo { * Providers with no entry here (alibaba) only support the binary * /thinking toggle — skip the tier picker. * - * `modelPattern` gates the tier to the models that actually honor it: + * `modelPattern` gates each profile to the models that actually honor it: * within a provider, only some model families expose the granular knob * (e.g. thinkingLevel is Gemini 3-only, Kimi's reasoningEffort is * K3-only). Models that don't match fall back to the binary /thinking * toggle. */ - reasoningTiers?: { modelPattern?: RegExp; options: readonly ReasoningTierOption[] } + reasoningTiers?: readonly ReasoningTierProfile[] } // ─── The table ─── @@ -88,6 +96,18 @@ export const PROVIDERS: readonly ProviderInfo[] = [ defaultModel: 'anthropic:claude-sonnet-5', keyUrl: 'https://console.anthropic.com/', models: [ + { + id: 'anthropic:claude-fable-5-1', + label: 'Fable 5.1', + description: 'Newest adaptive-reasoning flagship, 1M context', + vision: true, + }, + { + id: 'anthropic:claude-opus-5', + label: 'Opus 5', + description: 'Latest Opus for complex reasoning and agentic coding, 1M context', + vision: true, + }, { id: 'anthropic:claude-fable-5', label: 'Fable 5', @@ -113,13 +133,29 @@ export const PROVIDERS: readonly ProviderInfo[] = [ vision: true, }, ], - reasoningTiers: { - options: [ - { label: 'Low', value: 'low', description: 'Minimal reasoning, fastest' }, - { label: 'Medium', value: 'medium', description: 'Balanced reasoning' }, - { label: 'High', value: 'high', description: 'Thorough reasoning (default)' }, - ], - }, + reasoningTiers: [ + { + modelPattern: /claude-fable-5(?:-1)?$/, + offValue: 'low', + options: [ + { label: 'Low', value: 'low', description: 'Minimum adaptive reasoning' }, + { label: 'Medium', value: 'medium', description: 'Balanced reasoning' }, + { label: 'High', value: 'high', description: 'Thorough reasoning' }, + { label: 'XHigh', value: 'xhigh', description: 'Very deep reasoning' }, + { label: 'Max', value: 'max', description: 'Maximum reasoning depth' }, + ], + }, + { + modelPattern: /claude-(?:opus-(?:5|4-8)|sonnet-5)$/, + options: [ + { label: 'Low', value: 'low', description: 'Minimal reasoning, fastest' }, + { label: 'Medium', value: 'medium', description: 'Balanced reasoning' }, + { label: 'High', value: 'high', description: 'Thorough reasoning' }, + { label: 'XHigh', value: 'xhigh', description: 'Very deep reasoning' }, + { label: 'Max', value: 'max', description: 'Maximum reasoning depth' }, + ], + }, + ], }, { name: 'openai', @@ -127,22 +163,28 @@ export const PROVIDERS: readonly ProviderInfo[] = [ defaultModel: 'openai:gpt-5.6-sol', keyUrl: 'https://platform.openai.com/api-keys', models: [ + { + id: 'openai:gpt-6-astra', + label: 'GPT-6 Astra', + description: 'Newest flagship, 1.05M context and 128k output', + vision: true, + }, { id: 'openai:gpt-5.6-sol', label: 'GPT-5.6 Sol', - description: 'Flagship, top reasoning + coding, $5/$30, 1M context', + description: 'Flagship reasoning and coding tier, $4/$20, 1M context', vision: true, }, { id: 'openai:gpt-5.6-terra', label: 'GPT-5.6 Terra', - description: 'Balanced tier, $2.50/$15, 1M context', + description: 'Balanced tier, $2/$12, 1M context', vision: true, }, { id: 'openai:gpt-5.6-luna', label: 'GPT-5.6 Luna', - description: 'Budget tier, $1/$6, 1M context', + description: 'Budget tier, $0.20/$1.20, 1M context', vision: true, }, { @@ -158,50 +200,97 @@ export const PROVIDERS: readonly ProviderInfo[] = [ vision: true, }, ], - reasoningTiers: { - options: [ - { label: 'Minimal', value: 'minimal', description: 'Bare-minimum reasoning' }, - { label: 'Low', value: 'low', description: 'Fast, concise reasoning' }, - { label: 'Medium', value: 'medium', description: 'Balanced (default)' }, - { label: 'High', value: 'high', description: 'Thorough reasoning' }, - ], - }, + reasoningTiers: [ + { + modelPattern: /gpt-6-astra$/, + offValue: 'low', + options: [ + { label: 'Low', value: 'low', description: 'Minimum supported reasoning' }, + { label: 'Medium', value: 'medium', description: 'Balanced reasoning' }, + { label: 'High', value: 'high', description: 'Thorough reasoning' }, + { label: 'XHigh', value: 'xhigh', description: 'Very deep reasoning' }, + { label: 'Max', value: 'max', description: 'Maximum reasoning depth' }, + ], + }, + { + modelPattern: /gpt-5\.6(?:-(?:sol|terra|luna))?$/, + options: [ + { label: 'Low', value: 'low', description: 'Fast, concise reasoning' }, + { label: 'Medium', value: 'medium', description: 'Balanced reasoning' }, + { label: 'High', value: 'high', description: 'Thorough reasoning' }, + { label: 'XHigh', value: 'xhigh', description: 'Very deep reasoning' }, + { label: 'Max', value: 'max', description: 'Maximum reasoning depth' }, + ], + }, + { + modelPattern: /gpt-5\.5$/, + options: [ + { label: 'Low', value: 'low', description: 'Fast, concise reasoning' }, + { label: 'Medium', value: 'medium', description: 'Balanced reasoning' }, + { label: 'High', value: 'high', description: 'Thorough reasoning' }, + { label: 'XHigh', value: 'xhigh', description: 'Maximum reasoning depth' }, + ], + }, + { + modelPattern: /gpt-5\.4-(?:mini|nano)$/, + options: [ + { label: 'Low', value: 'low', description: 'Fast, concise reasoning' }, + { label: 'Medium', value: 'medium', description: 'Balanced (default)' }, + { label: 'High', value: 'high', description: 'Thorough reasoning' }, + { label: 'XHigh', value: 'xhigh', description: 'Maximum reasoning depth' }, + ], + }, + ], }, { name: 'deepseek', envKey: 'DEEPSEEK_API_KEY', - defaultModel: 'deepseek:deepseek-v4-flash', + defaultModel: 'deepseek:deepseek-flash', keyUrl: 'https://platform.deepseek.com/api_keys', models: [ { - id: 'deepseek:deepseek-v4-flash', - label: 'DeepSeek V4 Flash', - description: 'Fast, efficient general-purpose, $0.14/$0.28, 1M context (text-only)', - vision: false, + id: 'deepseek:deepseek-flash', + label: 'DeepSeek V4.1 Flash', + description: 'Latest fast, efficient model with native vision and 1M context', + vision: true, }, { id: 'deepseek:deepseek-v4-pro', label: 'DeepSeek V4 Pro', - description: 'Flagship, stronger reasoning, $0.44/$0.87, 1M context (text-only)', + description: 'Flagship V4 model with 1M context and text-only API surface', vision: false, }, ], - reasoningTiers: { - // V4 Flash and Pro both support low/high/max; medium/xhigh map to high server-side. - modelPattern: /deepseek-v4/, - options: [ - { label: 'Low', value: 'low', description: 'Faster, less reasoning' }, - { label: 'High', value: 'high', description: 'Standard reasoning (default)' }, - { label: 'Max', value: 'max', description: 'Maximum reasoning depth' }, - ], - }, + reasoningTiers: [ + { + // Flash and V4 Pro support low/high/max; medium/xhigh map to high server-side. + modelPattern: /deepseek-(?:flash|v4)/, + options: [ + { label: 'Low', value: 'low', description: 'Faster, less reasoning' }, + { label: 'High', value: 'high', description: 'Standard reasoning (default)' }, + { label: 'Max', value: 'max', description: 'Maximum reasoning depth' }, + ], + }, + ], }, { name: 'alibaba', envKey: 'ALIBABA_API_KEY', - defaultModel: 'alibaba:qwen3.7-max', + defaultModel: 'alibaba:qwen3.8-max', keyUrl: 'https://dashscope.console.aliyun.com/apiKey', models: [ + { + id: 'alibaba:qwen3.8-max', + label: 'Qwen3.8 Max', + description: 'Newest flagship with native vision, 1M context', + vision: true, + }, + { + id: 'alibaba:qwen3.8-flash', + label: 'Qwen3.8 Flash', + description: 'Fast multimodal model with 1M context', + vision: true, + }, { id: 'alibaba:qwen3.7-max', label: 'Qwen3.7 Max', @@ -211,8 +300,14 @@ export const PROVIDERS: readonly ProviderInfo[] = [ { id: 'alibaba:qwen3.7-plus', label: 'Qwen3.7 Plus', - description: 'Mid-tier, balanced cost/quality', - vision: false, + description: 'Multimodal mid-tier with 1M context', + vision: true, + }, + { + id: 'alibaba:qwen3.7-flash', + label: 'Qwen3.7 Flash', + description: 'Fast multimodal model with 1M context', + vision: true, }, { id: 'alibaba:qwen3-coder-plus', @@ -243,9 +338,27 @@ export const PROVIDERS: readonly ProviderInfo[] = [ { name: 'google', envKey: 'GOOGLE_GENERATIVE_AI_API_KEY', - defaultModel: 'google:gemini-3.5-flash', + defaultModel: 'google:gemini-3.8-flash', keyUrl: 'https://aistudio.google.com/apikey', models: [ + { + id: 'google:gemini-3.8-flash', + label: 'Gemini 3.8 Flash', + description: 'Newest multimodal flagship, 1M context and 64k output', + vision: true, + }, + { + id: 'google:gemini-3.7-flash', + label: 'Gemini 3.7 Flash', + description: 'Previous-generation multimodal agentic model, 1M context', + vision: true, + }, + { + id: 'google:gemini-3.6-flash', + label: 'Gemini 3.6 Flash', + description: 'Fast multimodal agentic model, 1M context', + vision: true, + }, { id: 'google:gemini-3.5-flash', label: 'Gemini 3.5 Flash', @@ -265,21 +378,52 @@ export const PROVIDERS: readonly ProviderInfo[] = [ vision: true, }, ], - reasoningTiers: { - // thinkingLevel is a Gemini 3 feature; Gemini 2.5 uses thinkingBudget. - modelPattern: /gemini-3/, - options: [ - { label: 'Low', value: 'low', description: 'Lower latency, lower cost' }, - { label: 'High', value: 'high', description: 'Deeper reasoning, higher quality' }, - ], - }, + reasoningTiers: [ + { + modelPattern: /gemini-3\.[78]-flash$/, + offValue: 'low', + options: [ + { label: 'Low', value: 'low', description: 'Minimum supported reasoning' }, + { label: 'Medium', value: 'medium', description: 'Balanced reasoning' }, + { label: 'High', value: 'high', description: 'Deeper reasoning, higher quality' }, + ], + }, + { + modelPattern: /gemini-3\.[56]-flash$/, + offValue: 'minimal', + options: [ + { label: 'Minimal', value: 'minimal', description: 'Minimum supported reasoning' }, + { label: 'Low', value: 'low', description: 'Lower latency, lower cost' }, + { label: 'Medium', value: 'medium', description: 'Balanced reasoning' }, + { label: 'High', value: 'high', description: 'Deeper reasoning, higher quality' }, + ], + }, + ], }, { name: 'xai', envKey: 'XAI_API_KEY', - defaultModel: 'xai:grok-4.5', + defaultModel: 'xai:grok-4.6', keyUrl: 'https://console.x.ai/', models: [ + { + id: 'xai:grok-4.6', + label: 'Grok 4.6', + description: 'Recommended flagship, native vision and 500k context', + vision: true, + }, + { + id: 'xai:grok-4.20', + label: 'Grok 4.20', + description: 'Fixed-reasoning model with native vision and 1M context', + vision: true, + }, + { + id: 'xai:grok-4.20-non-reasoning', + label: 'Grok 4.20 Non-Reasoning', + description: 'Low-latency non-reasoning variant with 1M context', + vision: true, + }, { id: 'xai:grok-4.5', label: 'Grok 4.5', @@ -293,19 +437,55 @@ export const PROVIDERS: readonly ProviderInfo[] = [ vision: true, }, ], - reasoningTiers: { - options: [ - { label: 'Low', value: 'low', description: 'Faster, cheaper responses' }, - { label: 'High', value: 'high', description: 'Deeper reasoning' }, - ], - }, + reasoningTiers: [ + { + modelPattern: /grok-4\.6$/, + offValue: 'low', + options: [ + { label: 'Low', value: 'low', description: 'Minimum supported reasoning' }, + { label: 'Medium', value: 'medium', description: 'Balanced reasoning' }, + { label: 'High', value: 'high', description: 'Deeper reasoning' }, + { label: 'XHigh', value: 'xhigh', description: 'Maximum supported reasoning' }, + ], + }, + { + modelPattern: /grok-4\.5$/, + offValue: 'low', + options: [ + { label: 'Low', value: 'low', description: 'Minimum supported reasoning' }, + { label: 'Medium', value: 'medium', description: 'Balanced reasoning' }, + { label: 'High', value: 'high', description: 'Maximum reasoning depth' }, + ], + }, + { + modelPattern: /grok-4\.3$/, + options: [ + { label: 'Low', value: 'low', description: 'Faster, cheaper responses' }, + { label: 'Medium', value: 'medium', description: 'Balanced reasoning' }, + { label: 'High', value: 'high', description: 'Deeper reasoning' }, + { label: 'XHigh', value: 'xhigh', description: 'Maximum supported reasoning' }, + ], + }, + ], }, { name: 'zhipu', envKey: 'ZHIPU_API_KEY', - defaultModel: 'zhipu:glm-5.2', + defaultModel: 'zhipu:glm-5.3', keyUrl: 'https://open.bigmodel.cn/usercenter/apikeys', models: [ + { + id: 'zhipu:glm-5.3', + label: 'GLM-5.3', + description: 'Newest reasoning flagship with 1M context', + vision: false, + }, + { + id: 'zhipu:glm-5.3-flash', + label: 'GLM-5.3 Flash', + description: 'Fast native-multimodal model with 1M context', + vision: true, + }, { id: 'zhipu:glm-5.2', label: 'GLM-5.2', @@ -337,14 +517,24 @@ export const PROVIDERS: readonly ProviderInfo[] = [ vision: true, }, ], - reasoningTiers: { - // reasoning_effort is GLM-5.2+; earlier models use the binary thinking switch. - modelPattern: /glm-5\.2/, - options: [ - { label: 'High', value: 'high', description: 'Enhanced reasoning' }, - { label: 'Max', value: 'max', description: 'Deep reasoning (default)' }, - ], - }, + reasoningTiers: [ + { + modelPattern: /glm-5\.3(?:-flash)?$/, + offValue: 'low', + options: [ + { label: 'Low', value: 'low', description: 'Minimum supported reasoning' }, + { label: 'High', value: 'high', description: 'Enhanced reasoning' }, + { label: 'Max', value: 'max', description: 'Deep reasoning (default)' }, + ], + }, + { + modelPattern: /glm-5\.2$/, + options: [ + { label: 'High', value: 'high', description: 'Enhanced reasoning' }, + { label: 'Max', value: 'max', description: 'Deep reasoning (default)' }, + ], + }, + ], }, { name: 'moonshotai', @@ -364,6 +554,12 @@ export const PROVIDERS: readonly ProviderInfo[] = [ description: 'Dedicated coding model, 256k context', vision: true, }, + { + id: 'moonshotai:kimi-k2.7-code-highspeed', + label: 'Kimi K2.7 Code Highspeed', + description: 'High-speed coding variant, 256k context', + vision: true, + }, { id: 'moonshotai:kimi-k2.6', label: 'Kimi K2.6', @@ -376,31 +572,36 @@ export const PROVIDERS: readonly ProviderInfo[] = [ { label: 'api.moonshot.cn (China)', url: 'https://api.moonshot.cn/v1' }, { label: 'api.moonshot.ai (International)', url: 'https://api.moonshot.ai/v1' }, ], - reasoningTiers: { - // reasoning_effort is K3-only; K2.x uses the binary thinking switch. - modelPattern: /kimi-k3/, - options: [ - { label: 'Low', value: 'low', description: 'Faster, concise reasoning' }, - { label: 'High', value: 'high', description: 'Deeper reasoning' }, - { label: 'Max', value: 'max', description: 'Maximum reasoning (default)' }, - ], - }, + reasoningTiers: [ + { + // reasoning_effort is K3-only; K2.x uses the binary thinking switch. + modelPattern: /kimi-k3/, + options: [ + { label: 'Low', value: 'low', description: 'Faster, concise reasoning' }, + { label: 'High', value: 'high', description: 'Deeper reasoning' }, + { label: 'Max', value: 'max', description: 'Maximum reasoning (default)' }, + ], + }, + ], }, ] // ─── Model aliases ─── export const MODEL_ALIASES: Record = { - fable: 'anthropic:claude-fable-5', + fable: 'anthropic:claude-fable-5-1', sonnet: 'anthropic:claude-sonnet-5', - opus: 'anthropic:claude-opus-4-8', + opus: 'anthropic:claude-opus-5', haiku: 'anthropic:claude-haiku-4-5', + gpt6: 'openai:gpt-6-astra', + astra: 'openai:gpt-6-astra', gpt5: 'openai:gpt-5.6-sol', - gemini: 'google:gemini-3.5-flash', - deepseek: 'deepseek:deepseek-v4-flash', + gemini: 'google:gemini-3.8-flash', + deepseek: 'deepseek:deepseek-flash', 'deepseek-pro': 'deepseek:deepseek-v4-pro', - qwen: 'alibaba:qwen3.7-max', - glm: 'zhipu:glm-5.2', + qwen: 'alibaba:qwen3.8-max', + grok: 'xai:grok-4.6', + glm: 'zhipu:glm-5.3', kimi: 'moonshotai:kimi-k3', } @@ -425,7 +626,24 @@ export const PROVIDER_BASE_URLS: Record p.baseUrlOptions).map((p) => [p.name, { options: p.baseUrlOptions! }]), ) +export const PROVIDER_REASONING_PROFILES: Record = Object.fromEntries( + PROVIDERS.filter((p) => p.reasoningTiers).map((p) => [p.name, p.reasoningTiers!]), +) + +/** Legacy provider-wide view retained for public API compatibility. Internal + * model selection uses PROVIDER_REASONING_PROFILES so each family gets only + * the effort values it actually supports. */ export const PROVIDER_REASONING_TIERS: Record< string, { modelPattern?: RegExp; options: readonly ReasoningTierOption[] } -> = Object.fromEntries(PROVIDERS.filter((p) => p.reasoningTiers).map((p) => [p.name, p.reasoningTiers!])) +> = Object.fromEntries( + Object.entries(PROVIDER_REASONING_PROFILES).map(([name, profiles]) => { + const options = Array.from( + new Map(profiles.flatMap((profile) => profile.options).map((option) => [option.value, option])).values(), + ) + const patterns = profiles.map((profile) => profile.modelPattern).filter((pattern) => pattern !== undefined) + const modelPattern = + patterns.length === profiles.length ? new RegExp(patterns.map((pattern) => pattern.source).join('|')) : undefined + return [name, { ...(modelPattern ? { modelPattern } : {}), options }] + }), +) diff --git a/packages/core/src/providers/registry.ts b/packages/core/src/providers/registry.ts index fb2bc0f..21a2210 100644 --- a/packages/core/src/providers/registry.ts +++ b/packages/core/src/providers/registry.ts @@ -18,6 +18,7 @@ import { createOpenAIChatGPTFetch } from './openai-chatgpt-fetch.js' const KIMI_CODING_MODEL_IDS = { 'kimi-k3': 'k3', + // The stable Coding Plan id currently auto-routes to K2.8 Preview. 'kimi-k2.7-code': 'kimi-for-coding', 'kimi-k2.7-code-highspeed': 'kimi-for-coding-highspeed', // Coding Plan exposes K2.6 by disabling thinking on kimi-for-coding rather @@ -194,6 +195,7 @@ const moonshotConvertUsage = (usage: any) => { */ let _zhipuReasoningEffort: string | undefined const ZHIPU_REASONING_HEADER = 'x-x-code-zhipu-reasoning-effort' +const XAI_REASONING_HEADER = 'x-x-code-xai-reasoning-effort' export function setZhipuReasoningEffort(effort: string | undefined): void { _zhipuReasoningEffort = effort @@ -202,8 +204,28 @@ export function setZhipuReasoningEffort(effort: string | undefined): void { export function withZhipuReasoningHeader( headers: Record | undefined, effort: string | undefined, -): Record { - return { ...headers, [ZHIPU_REASONING_HEADER]: effort ?? '' } +): Record { + return withInternalReasoningHeader(headers, ZHIPU_REASONING_HEADER, effort) +} + +export function withXaiReasoningHeader( + headers: Record | undefined, + effort: string | undefined, +): Record { + return withInternalReasoningHeader(headers, XAI_REASONING_HEADER, effort) +} + +function withInternalReasoningHeader( + headers: Record | undefined, + name: string, + effort: string | undefined, +): Record { + const result: Record = {} + for (const [key, value] of Object.entries(headers ?? {})) { + if (value !== undefined) result[key] = value + } + result[name] = effort ?? '' + return result } /** Move x-code's per-session affinity hint to the wire location required by @@ -212,16 +234,18 @@ export function withZhipuReasoningHeader( const xaiPromptCacheFetch: typeof fetch = async (input, init) => { const headers = new Headers(init?.headers ?? (input instanceof Request ? input.headers : undefined)) const promptCacheKey = headers.get(XAI_PROMPT_CACHE_KEY_HEADER) + const reasoningEffort = headers.get(XAI_REASONING_HEADER) || undefined headers.delete(XAI_PROMPT_CACHE_KEY_HEADER) + headers.delete(XAI_REASONING_HEADER) const sanitizedInit = { ...init, headers } - if (!promptCacheKey) return permanentErrorFetch(input, sanitizedInit) + if (!promptCacheKey && !reasoningEffort) return permanentErrorFetch(input, sanitizedInit) const url = input instanceof URL ? input : new URL(typeof input === 'string' ? input : input.url) if (url.pathname.endsWith('/chat/completions')) { - headers.set('x-grok-conv-id', promptCacheKey) + if (promptCacheKey) headers.set('x-grok-conv-id', promptCacheKey) + } else if (!url.pathname.endsWith('/responses')) { return permanentErrorFetch(input, sanitizedInit) } - if (!url.pathname.endsWith('/responses')) return permanentErrorFetch(input, sanitizedInit) const rawBody = init?.body ?? (input instanceof Request ? await input.clone().text() : undefined) if (typeof rawBody !== 'string') return permanentErrorFetch(input, sanitizedInit) @@ -231,7 +255,18 @@ const xaiPromptCacheFetch: typeof fetch = async (input, init) => { return permanentErrorFetch(input, sanitizedInit) } const body = parsed as Record - if (!body.prompt_cache_key) body.prompt_cache_key = promptCacheKey + if (url.pathname.endsWith('/chat/completions')) { + if (reasoningEffort) body.reasoning_effort = reasoningEffort + } else { + if (promptCacheKey && !body.prompt_cache_key) body.prompt_cache_key = promptCacheKey + if (reasoningEffort) { + const current = + body.reasoning && typeof body.reasoning === 'object' && !Array.isArray(body.reasoning) + ? (body.reasoning as Record) + : {} + body.reasoning = { ...current, effort: reasoningEffort } + } + } return permanentErrorFetch(input, { ...sanitizedInit, body: JSON.stringify(body) }) } catch { return permanentErrorFetch(input, sanitizedInit) diff --git a/packages/core/src/providers/thinking.ts b/packages/core/src/providers/thinking.ts index ce56a3d..35d7740 100644 --- a/packages/core/src/providers/thinking.ts +++ b/packages/core/src/providers/thinking.ts @@ -5,9 +5,13 @@ // We use it as the primary mechanism for reasoning control. // // Exceptions that still need providerOptions / fetch shim injection: +// - deepseek: explicit tiers use native reasoningEffort so `max` does not +// round-trip through portable `xhigh` and emit a compatibility warning. // - zhipu: goes through @ai-sdk/openai-compatible, SDK doesn't auto-translate // `reasoning` for it. We inject `reasoning_effort` via fetch shim. // - alibaba: uses `enableThinking` in providerOptions (no top-level support). +// - native `max` controls for newer Anthropic/OpenAI models also travel via +// providerOptions because the portable reasoning enum tops out at xhigh. // // The user-facing controls: // /thinking on|off — binary toggle (maps to 'high' / 'none') @@ -17,8 +21,8 @@ // over the `enabled` flag. The /thinking toggle is only used as a fallback // for models without an explicit tier. import { providerOf } from './capabilities.js' -import { PROVIDER_REASONING_TIERS } from './catalog.js' -import type { ReasoningTierOption } from './catalog.js' +import { PROVIDER_REASONING_PROFILES } from './catalog.js' +import type { ReasoningTierOption, ReasoningTierProfile } from './catalog.js' import { getOpenAIChatGPTReasoningTiers, getOpenAIChatGPTRuntimeModel } from './openai-chatgpt-models.js' /** Whether the model exposes a granular reasoning-effort tier (vs. the @@ -26,20 +30,30 @@ import { getOpenAIChatGPTReasoningTiers, getOpenAIChatGPTRuntimeModel } from './ * model families honor them — modelPattern in PROVIDER_REASONING_TIERS * gates that. Drives both the /model tier picker and the effort branch * in getReasoningLevel. */ +function getReasoningTierProfile(modelId: string): ReasoningTierProfile | undefined { + return PROVIDER_REASONING_PROFILES[providerOf(modelId)]?.find( + (profile) => !profile.modelPattern || profile.modelPattern.test(modelId), + ) +} + +/** Some fixed-reasoning xAI models reject the effort field entirely. The + * public grok-4.20 alias is not recognized by the SDK's equivalent guard. */ +export function acceptsReasoningControl(modelId: string): boolean { + if (providerOf(modelId) !== 'xai') return true + const providerModelId = modelId.slice(modelId.indexOf(':') + 1) + return !/^grok-4\.20(?!-multi-agent(?:$|-))/.test(providerModelId) +} + export function supportsReasoningTier(modelId: string): boolean { const chatGPTTiers = getOpenAIChatGPTReasoningTiers(modelId) if (chatGPTTiers !== undefined) return chatGPTTiers.length > 0 - const config = PROVIDER_REASONING_TIERS[providerOf(modelId)] - if (!config) return false - return !config.modelPattern || config.modelPattern.test(modelId) + return getReasoningTierProfile(modelId) !== undefined } export function getReasoningTierOptions(modelId: string): readonly ReasoningTierOption[] | undefined { const chatGPTTiers = getOpenAIChatGPTReasoningTiers(modelId) if (chatGPTTiers !== undefined) return chatGPTTiers - const config = PROVIDER_REASONING_TIERS[providerOf(modelId)] - if (!config || (config.modelPattern && !config.modelPattern.test(modelId))) return undefined - return config.options + return getReasoningTierProfile(modelId)?.options } export type ReasoningLevel = 'provider-default' | 'none' | 'minimal' | 'low' | 'medium' | 'high' | 'xhigh' @@ -55,6 +69,15 @@ const TIER_TO_REASONING: Record = { max: 'xhigh', } +/** Resolve the provider-native effort after applying model support and the + * minimum valid effort for models that cannot turn reasoning off. */ +export function getReasoningEffort(modelId: string, enabled: boolean, effort?: string): string | undefined { + const profile = getReasoningTierProfile(modelId) + if (!profile) return undefined + if (effort && profile.options.some((option) => option.value === effort)) return effort + return !enabled ? profile.offValue : undefined +} + /** * Compute the top-level `reasoning` value for streamText/generateText. * Returns undefined when the model doesn't support reasoning control @@ -69,7 +92,7 @@ export function getReasoningLevel(modelId: string, enabled: boolean, effort?: st const chatGPTTiers = getOpenAIChatGPTReasoningTiers(modelId) // These providers use separate mechanisms (providerOptions / fetch shim) - if (provider === 'alibaba' || provider === 'zhipu' || provider === 'custom') { + if (provider === 'alibaba' || provider === 'zhipu' || provider === 'custom' || !acceptsReasoningControl(modelId)) { return undefined } @@ -83,20 +106,22 @@ export function getReasoningLevel(modelId: string, enabled: boolean, effort?: st return chatGPTTiers[Math.floor((chatGPTTiers.length - 1) / 2)]?.value as ReasoningLevel } - // Tiered reasoning — user picked an explicit effort level AND the model - // honors it. - if (effort && supportsReasoningTier(modelId)) { - return TIER_TO_REASONING[effort] ?? (effort as ReasoningLevel) + const effectiveEffort = getReasoningEffort(modelId, enabled, effort) + // DeepSeek exposes provider-native low/high/max values. Sending the + // portable max equivalent (`xhigh`) makes the SDK map it back to `max` + // and emit a compatibility warning, so explicit tiers travel only through + // providerOptions.deepseek.reasoningEffort. + if (provider === 'deepseek' && effectiveEffort) return undefined + if (effectiveEffort) { + return TIER_TO_REASONING[effectiveEffort] ?? (effectiveEffort as ReasoningLevel) } return enabled ? 'high' : 'none' } /** - * Build providerOptions for providers that can't use the top-level - * `reasoning` parameter: Alibaba (enableThinking) and Zhipu (thinking toggle). - * - * Returns an empty object for providers that use top-level `reasoning`. + * Build provider-native reasoning options where the portable top-level + * `reasoning` value cannot express the exact provider control. */ export function getThinkingProviderOptions( modelId: string, @@ -106,17 +131,32 @@ export function getThinkingProviderOptions( const provider = providerOf(modelId) switch (provider) { + case 'deepseek': + const deepseekEffort = getReasoningEffort(modelId, enabled, effort) + return deepseekEffort ? { deepseek: { reasoningEffort: deepseekEffort } } : {} + case 'alibaba': return { alibaba: { enableThinking: enabled } } case 'zhipu': // Binary toggle via providerOptions for models that don't use tiers. // Tiered models get reasoning_effort injected by the fetch shim. - if (effort && supportsReasoningTier(modelId)) { - return { zhipu: { thinking: { type: 'enabled' } } } + const zhipuEffort = getReasoningEffort(modelId, enabled, effort) + if (zhipuEffort) { + return { zhipu: { thinking: { type: 'enabled' }, reasoningEffort: zhipuEffort } } } return enabled ? { zhipu: { thinking: { type: 'enabled' } } } : { zhipu: { thinking: { type: 'disabled' } } } + case 'anthropic': + return effort === 'max' && getReasoningTierOptions(modelId)?.some((option) => option.value === 'max') + ? { anthropic: { thinking: { type: 'adaptive' }, effort: 'max' } } + : {} + + case 'openai': + return effort === 'max' && getReasoningTierOptions(modelId)?.some((option) => option.value === 'max') + ? { openai: { reasoningEffort: 'max' } } + : {} + default: return {} } diff --git a/packages/core/src/tools/read-file.ts b/packages/core/src/tools/read-file.ts index aa80ea1..f7b5370 100644 --- a/packages/core/src/tools/read-file.ts +++ b/packages/core/src/tools/read-file.ts @@ -247,7 +247,8 @@ async function checkReadCache( stub: `[readFile: ${filePath} is unchanged since its full content was added to this conversation ` + `(same mtime and size); its full content is already in the conversation above. ` + - `Re-read with an explicit offset/limit to revisit a specific range, or use grep to search within it.]`, + `Analyze that content directly instead of reading it again. For text files, use an explicit offset/limit ` + + `only to revisit a specific range, or use grep to search within it.]`, } } return { hit: false, entry: { mtimeMs: stat.mtimeMs, size: stat.size } } @@ -337,6 +338,8 @@ Usage: } if (kind === 'image') { + const verdict = await checkReadCache(cache, filePath, false) + if (verdict && verdict.hit) return verdict.stub const stats = await fs.stat(filePath) if (stats.size > MAX_IMAGE_SOURCE_BYTES) { return `[Image ${filePath} is too large to process safely (${(stats.size / (1024 * 1024)).toFixed(1)} MB, cap ${MAX_IMAGE_SOURCE_BYTES / (1024 * 1024)} MB).]` @@ -380,6 +383,7 @@ Usage: const header = compressed.changed ? `Loaded image: ${filePath} (compressed from ${buffer.length} to ${compressed.data.length} bytes)` : `Loaded image: ${filePath}` + if (verdict && !verdict.hit) cache?.set(filePath, verdict.entry) return { type: 'content', value: [ @@ -390,6 +394,7 @@ Usage: mediaType: finalMime, filename: path.basename(filePath), }, + { type: 'text', text: BUILT_IN_MEDIA_ANALYSIS_NOTE }, ], } } diff --git a/packages/core/src/tools/web-search.ts b/packages/core/src/tools/web-search.ts index ecad6f7..6cd2251 100644 --- a/packages/core/src/tools/web-search.ts +++ b/packages/core/src/tools/web-search.ts @@ -236,7 +236,7 @@ async function searchWithDeepseek(query: string, maxResults: number, signal?: Ab body: JSON.stringify({ // A full (cheap) model turn powers each search; override if DeepSeek // retires the id or a cheaper search-capable model appears. - model: process.env.DEEPSEEK_SEARCH_MODEL || 'deepseek-v4-flash', + model: process.env.DEEPSEEK_SEARCH_MODEL || 'deepseek-flash', max_tokens: 4096, messages: [{ role: 'user', content: [{ type: 'text', text: `Perform a web search for the query: ${query}` }] }], // Each use is billed server-side, so don't grant more searches than the diff --git a/packages/core/tests/agent-loop.test.ts b/packages/core/tests/agent-loop.test.ts index 248cf3b..cd8d303 100644 --- a/packages/core/tests/agent-loop.test.ts +++ b/packages/core/tests/agent-loop.test.ts @@ -999,7 +999,7 @@ describe('agent loop', () => { await agentLoop( 'Try this task', {} as any, - { modelId: 'deepseek:deepseek-v4-flash', trustMode: false, maxTurns: 10, printMode: false }, + { modelId: 'deepseek:deepseek-flash', trustMode: false, maxTurns: 10, printMode: false }, mockCallbacks, ) @@ -1025,7 +1025,7 @@ describe('agent loop', () => { const { state } = await agentLoop( 'Give a complete answer', {} as any, - { modelId: 'deepseek:deepseek-v4-flash', trustMode: false, maxTurns: 10, printMode: false }, + { modelId: 'deepseek:deepseek-flash', trustMode: false, maxTurns: 10, printMode: false }, mockCallbacks, ) diff --git a/packages/core/tests/capabilities.test.ts b/packages/core/tests/capabilities.test.ts index ffe0e17..c109452 100644 --- a/packages/core/tests/capabilities.test.ts +++ b/packages/core/tests/capabilities.test.ts @@ -14,7 +14,18 @@ describe('modelSupportsVision', () => { expect(modelSupportsVision('anthropic:claude-haiku-4-5')).toBe(true) expect(modelSupportsVision('moonshotai:kimi-k3')).toBe(true) expect(modelSupportsVision('moonshotai:kimi-k2.6')).toBe(true) + expect(modelSupportsVision('deepseek:deepseek-flash')).toBe(true) + expect(modelSupportsVision('deepseek:deepseek-v4-flash')).toBe(true) + expect(modelSupportsVision('alibaba:qwen3.8-max')).toBe(true) + expect(modelSupportsVision('alibaba:qwen3.8-flash')).toBe(true) + expect(modelSupportsVision('alibaba:qwen3.7-plus')).toBe(true) + expect(modelSupportsVision('alibaba:qwen3.7-flash')).toBe(true) expect(modelSupportsVision('alibaba:qwen3-vl-flash')).toBe(true) + expect(modelSupportsVision('google:gemini-3.8-flash')).toBe(true) + expect(modelSupportsVision('google:gemini-3.7-flash')).toBe(true) + expect(modelSupportsVision('xai:grok-4.6')).toBe(true) + expect(modelSupportsVision('zhipu:glm-5.3-flash')).toBe(true) + expect(modelSupportsVision('moonshotai:kimi-k2.7-code-highspeed')).toBe(true) expect(modelSupportsVision('zhipu:glm-4.6v')).toBe(true) }) @@ -23,18 +34,18 @@ describe('modelSupportsVision', () => { // specific models are text-only — the per-model flag must win. expect(modelSupportsVision('alibaba:qwen3.7-max')).toBe(false) expect(modelSupportsVision('zhipu:glm-5.2')).toBe(false) - expect(modelSupportsVision('deepseek:deepseek-v4-flash')).toBe(false) + expect(modelSupportsVision('deepseek:deepseek-v4-pro')).toBe(false) }) it('expands aliases before lookup', () => { - expect(modelSupportsVision('opus')).toBe(true) // → anthropic:claude-opus-4-8 - expect(modelSupportsVision('deepseek')).toBe(false) // → deepseek:deepseek-v4-flash + expect(modelSupportsVision('opus')).toBe(true) // → anthropic:claude-opus-5 + expect(modelSupportsVision('deepseek')).toBe(true) // → deepseek:deepseek-flash }) it('falls back to provider-level capability for unlisted ids', () => { // Not in the catalog → defer to the provider's image capability. expect(modelSupportsVision('anthropic:claude-some-future-model')).toBe(true) - expect(modelSupportsVision('deepseek:some-future-model')).toBe(false) + expect(modelSupportsVision('deepseek:some-future-model')).toBe(true) expect(modelSupportsVision('unknownprovider:whatever')).toBe(false) }) @@ -71,13 +82,19 @@ describe('toolImageTransport capability', () => { }) it('reattaches media in a following user message for Chat Completions providers', () => { - for (const id of ['moonshotai:kimi-k2.6', 'alibaba:qwen3-vl-flash', 'zhipu:glm-4.6v', 'xai:grok-4.3']) { + for (const id of [ + 'deepseek:deepseek-flash', + 'moonshotai:kimi-k2.6', + 'alibaba:qwen3-vl-flash', + 'zhipu:glm-4.6v', + 'xai:grok-4.3', + ]) { expect(capabilitiesOf(id).toolImageTransport, id).toBe('user-message') } }) it('marks text-only and unknown providers unsupported', () => { - for (const id of ['deepseek:deepseek-v4', 'custom:whatever', 'unknownprovider:whatever']) { + for (const id of ['custom:whatever', 'unknownprovider:whatever']) { expect(capabilitiesOf(id).toolImageTransport, id).toBe('unsupported') } }) diff --git a/packages/core/tests/compression.test.ts b/packages/core/tests/compression.test.ts index 62b8420..d9f5403 100644 --- a/packages/core/tests/compression.test.ts +++ b/packages/core/tests/compression.test.ts @@ -269,11 +269,46 @@ describe('checkAndCompressContext', () => { const state = createLoopState() state.messages = padMessages(10) state.lastInputTokens = 999_999 + state.readFileCache.set('/tmp/previously-read.png', { mtimeMs: 1, size: 1 }) await checkAndCompressContext(state, fakeModel, 1, makeCallbacks()) expect(state.lastInputTokens).toBe(0) expect(state.expectCacheMiss).toBe(true) + expect(state.readFileCache.size).toBe(0) + }) + + it('clears delivered-file de-dup state after lightweight compression rewrites history', async () => { + const state = createLoopState() + state.messages = [ + { role: 'user', content: 'start' }, + { + role: 'assistant', + content: [{ type: 'tool-call', toolCallId: 'looped', toolName: 'readFile', input: { filePath: '/tmp/a' } }], + }, + { + role: 'tool', + content: [ + { + type: 'tool-result', + toolCallId: 'looped', + toolName: 'readFile', + output: { type: 'text', value: `[loop-guard] ${'x'.repeat(10_000)}` }, + }, + ], + }, + { role: 'user', content: 'one' }, + { role: 'assistant', content: 'two' }, + { role: 'user', content: 'three' }, + { role: 'assistant', content: 'four' }, + ] as ModelMessage[] + state.lastInputTokens = 999_999 + state.readFileCache.set('/tmp/previously-read.png', { mtimeMs: 1, size: 1 }) + + await checkAndCompressContext(state, fakeModel, 100, makeCallbacks()) + + expect(generateText).not.toHaveBeenCalled() + expect(state.readFileCache.size).toBe(0) }) it('does not return until the compact boundary and recall-window reset are durable', async () => { diff --git a/packages/core/tests/config.test.ts b/packages/core/tests/config.test.ts index 784e134..d4aa087 100644 --- a/packages/core/tests/config.test.ts +++ b/packages/core/tests/config.test.ts @@ -101,8 +101,15 @@ describe('resolveModelId', () => { it('resolves alias from CLI argument', () => { expect(resolveModelId('sonnet')).toBe('anthropic:claude-sonnet-5') - expect(resolveModelId('opus')).toBe('anthropic:claude-opus-4-8') - expect(resolveModelId('deepseek')).toBe('deepseek:deepseek-v4-flash') + expect(resolveModelId('fable')).toBe('anthropic:claude-fable-5-1') + expect(resolveModelId('opus')).toBe('anthropic:claude-opus-5') + expect(resolveModelId('astra')).toBe('openai:gpt-6-astra') + expect(resolveModelId('gpt6')).toBe('openai:gpt-6-astra') + expect(resolveModelId('gemini')).toBe('google:gemini-3.8-flash') + expect(resolveModelId('qwen')).toBe('alibaba:qwen3.8-max') + expect(resolveModelId('grok')).toBe('xai:grok-4.6') + expect(resolveModelId('glm')).toBe('zhipu:glm-5.3') + expect(resolveModelId('deepseek')).toBe('deepseek:deepseek-flash') }) it('falls back to env var X_CODE_MODEL', () => { @@ -130,12 +137,26 @@ describe('resolveModelId', () => { expect(resolveModelId()).toBe('openai:gpt-5.6-sol') }) + it('uses the updated provider smart defaults', () => { + const cases: Array<[string, string]> = [ + ['GOOGLE_GENERATIVE_AI_API_KEY', 'google:gemini-3.8-flash'], + ['ALIBABA_API_KEY', 'alibaba:qwen3.8-max'], + ['XAI_API_KEY', 'xai:grok-4.6'], + ['ZHIPU_API_KEY', 'zhipu:glm-5.3'], + ] + for (const [envKey, expected] of cases) { + process.env[envKey] = 'test-key' + expect(resolveModelId()).toBe(expected) + delete process.env[envKey] + } + }) + it('returns null when no providers configured', () => { expect(resolveModelId()).toBeNull() }) it('returns model even if provider key missing when explicitly requested', () => { - expect(resolveModelId('deepseek')).toBe('deepseek:deepseek-v4-flash') + expect(resolveModelId('deepseek')).toBe('deepseek:deepseek-flash') }) it('uses ChatGPT authentication as the OpenAI smart default without an API key', async () => { diff --git a/packages/core/tests/context-usage.test.ts b/packages/core/tests/context-usage.test.ts index 9e81a54..7e4b621 100644 --- a/packages/core/tests/context-usage.test.ts +++ b/packages/core/tests/context-usage.test.ts @@ -235,7 +235,7 @@ describe('calibrateContextBreakdown', () => { describe('buildContextBreakdownInput', () => { it('returns null before the system prompt has been built', () => { const state = createLoopState() - expect(buildContextBreakdownInput({ modelId: 'deepseek:deepseek-v4-flash' } as any, state)).toBeNull() + expect(buildContextBreakdownInput({ modelId: 'deepseek:deepseek-flash' } as any, state)).toBeNull() }) it('derives the deferred-tools block from the catalog', () => { @@ -252,7 +252,7 @@ describe('buildContextBreakdownInput', () => { def: makeTool('Search the web'), }, ] - const options = { modelId: 'deepseek:deepseek-v4-flash' } as any + const options = { modelId: 'deepseek:deepseek-flash' } as any const input = buildContextBreakdownInput(options, state)! expect(input.systemPrompt).toBe('prompt') expect(input.mcpDeferredBlock).toBe(formatDeferredCapabilities([{ name: 'webSearch', source: 'builtin' }])) @@ -264,7 +264,7 @@ describe('buildContextBreakdownInput', () => { const state = createLoopState() state.systemPromptCache = 'prompt with embedded blocks' state.systemPromptBlocks = { knowledge: 'knowledge', skill: 'skills block', mcpDeferred: 'mcp block' } - const options = { modelId: 'deepseek:deepseek-v4-flash' } as any + const options = { modelId: 'deepseek:deepseek-flash' } as any const input = buildContextBreakdownInput(options, state)! expect(input.knowledgeContext).toBe('knowledge') expect(input.skillBlock).toBe('skills block') @@ -286,7 +286,7 @@ describe('buildContextBreakdownInput', () => { state.deferredCatalog = catalog const executor = async () => 'ok' state.manualToolExecutors.set('existing', executor) - buildContextBreakdownInput({ modelId: 'deepseek:deepseek-v4-flash' } as any, state) + buildContextBreakdownInput({ modelId: 'deepseek:deepseek-flash' } as any, state) expect(state.deferredCatalog).toBe(catalog) expect([...state.manualToolExecutors]).toEqual([['existing', executor]]) }) diff --git a/packages/core/tests/context-window.test.ts b/packages/core/tests/context-window.test.ts index d59bc1a..00b1667 100644 --- a/packages/core/tests/context-window.test.ts +++ b/packages/core/tests/context-window.test.ts @@ -25,10 +25,18 @@ afterEach(() => { describe('getContextWindow', () => { it('returns exact value for known models', () => { + expect(getContextWindow('anthropic:claude-fable-5-1')).toBe(1000000) expect(getContextWindow('anthropic:claude-opus-4-8')).toBe(1000000) + expect(getContextWindow('openai:gpt-6-astra')).toBe(1050000) expect(getContextWindow('openai:gpt-5.6-sol')).toBe(1047576) + expect(getContextWindow('google:gemini-3.8-flash')).toBe(1048576) + expect(getContextWindow('google:gemini-3.7-flash')).toBe(1048576) expect(getContextWindow('google:gemini-2.5-flash')).toBe(1000000) - expect(getContextWindow('deepseek:deepseek-v4-flash')).toBe(1000000) + expect(getContextWindow('deepseek:deepseek-flash')).toBe(1000000) + expect(getContextWindow('alibaba:qwen3.8-max')).toBe(1000000) + expect(getContextWindow('alibaba:qwen3.7-plus')).toBe(1000000) + expect(getContextWindow('xai:grok-4.20')).toBe(1000000) + expect(getContextWindow('zhipu:glm-5.3-flash')).toBe(1000000) expect(getContextWindow('alibaba:qwen-max')).toBe(32768) }) @@ -72,8 +80,14 @@ describe('getCompressionThreshold', () => { describe('getMaxOutputTokens', () => { it('returns specific ceiling for known models', () => { - expect(getMaxOutputTokens('deepseek:deepseek-v4-flash')).toBe(131072) - expect(getMaxOutputTokens('alibaba:qwen3.7-plus')).toBe(32000) + expect(getMaxOutputTokens('deepseek:deepseek-flash')).toBe(131072) + expect(getMaxOutputTokens('anthropic:claude-sonnet-5')).toBe(128000) + expect(getMaxOutputTokens('anthropic:claude-haiku-4-5')).toBe(64000) + expect(getMaxOutputTokens('openai:gpt-6-astra')).toBe(128000) + expect(getMaxOutputTokens('google:gemini-3.8-flash')).toBe(65536) + expect(getMaxOutputTokens('google:gemini-3.7-flash')).toBe(65536) + expect(getMaxOutputTokens('alibaba:qwen3.8-flash')).toBe(65536) + expect(getMaxOutputTokens('alibaba:qwen3.7-plus')).toBe(65536) expect(getMaxOutputTokens('alibaba:qwen-max')).toBe(8192) }) diff --git a/packages/core/tests/file-ingest.test.ts b/packages/core/tests/file-ingest.test.ts index 89db111..f67c0ca 100644 --- a/packages/core/tests/file-ingest.test.ts +++ b/packages/core/tests/file-ingest.test.ts @@ -584,6 +584,24 @@ describe('ingestFile', () => { expect(imagePart.data).toEqual({ type: 'data', data: source.toString('base64') }) expect(JSON.parse(JSON.stringify(imagePart))).toEqual(imagePart) } + expect(JSON.stringify(parts)).toContain('Analyze the supplied content directly') + expect(JSON.stringify(parts)).toContain('Do not invoke readFile') + }) + + it('marks a natively attached image as already delivered for readFile de-duplication', async () => { + const cache = new Map() + + await buildUserContent( + `'${imageFile}' describe this image`, + multimodalCaps, + undefined, + undefined, + undefined, + 'deepseek:deepseek-flash', + cache, + ) + + expect(cache.has(imageFile)).toBe(true) }) it('rejects GIF before session insertion for an xAI model', async () => { diff --git a/packages/core/tests/memory-inference.test.ts b/packages/core/tests/memory-inference.test.ts index a4f62e2..14124b1 100644 --- a/packages/core/tests/memory-inference.test.ts +++ b/packages/core/tests/memory-inference.test.ts @@ -43,6 +43,52 @@ describe('memory inference policy', () => { expect(generate).toHaveBeenCalledWith({ maxOutputTokens: 1500, reasoning: 'low' }) }) + it('keeps current always-thinking families at their minimum effort', async () => { + const generate = vi.fn().mockResolvedValue({ output: {} }) + + await runMemoryInference({ + modelId: 'anthropic:claude-fable-5-1', + maxOutputTokens: 1500, + maxTotalOutputTokens: 8192, + generate, + }) + + expect(generate).toHaveBeenCalledWith({ maxOutputTokens: 1500, reasoning: 'low' }) + }) + + it('uses the minimum GLM-5.3 effort for memory inference', async () => { + for (const modelId of ['zhipu:glm-5.3', 'zhipu:glm-5.3-flash']) { + const generate = vi.fn().mockResolvedValue({ output: {} }) + + await runMemoryInference({ + modelId, + maxOutputTokens: 1500, + maxTotalOutputTokens: 8192, + generate, + }) + + expect(generate).toHaveBeenCalledWith({ + maxOutputTokens: 1500, + providerOptions: { + zhipu: { thinking: { type: 'enabled' }, reasoningEffort: 'low' }, + }, + }) + } + }) + + it('omits unsupported reasoning controls for fixed-reasoning Grok 4.20', async () => { + const generate = vi.fn().mockResolvedValue({ output: {} }) + + await runMemoryInference({ + modelId: 'xai:grok-4.20', + maxOutputTokens: 1500, + maxTotalOutputTokens: 8192, + generate, + }) + + expect(generate).toHaveBeenCalledWith({ maxOutputTokens: 1500 }) + }) + it('starts OpenAI models without an off tier at low effort without temperature', async () => { const generate = vi.fn().mockResolvedValue({ output: {} }) diff --git a/packages/core/tests/openai-chatgpt-models.test.ts b/packages/core/tests/openai-chatgpt-models.test.ts index b3dc4be..9b886ec 100644 --- a/packages/core/tests/openai-chatgpt-models.test.ts +++ b/packages/core/tests/openai-chatgpt-models.test.ts @@ -26,7 +26,7 @@ import { refreshOpenAIChatGPTModelsAfterNotFound, resetOpenAIChatGPTModelsForTesting, } from '../src/providers/openai-chatgpt-models.js' -import { getReasoningLevel } from '../src/providers/thinking.js' +import { getReasoningLevel, getThinkingProviderOptions } from '../src/providers/thinking.js' describe('OpenAI ChatGPT model catalog', () => { let testHome: string @@ -107,6 +107,28 @@ describe('OpenAI ChatGPT model catalog', () => { expect(String(fetchMock.mock.calls[0]?.[0])).not.toContain('client_version=test') }) + it('does not reintroduce a stale max effort after the server catalog removes it', async () => { + const fetchMock = vi.fn(async () => + Response.json({ + models: [ + { + slug: 'gpt-5.6-sol', + display_name: 'GPT-5.6 Sol', + input_modalities: ['text', 'image'], + default_reasoning_level: 'low', + supported_reasoning_levels: [{ effort: 'low' }], + visibility: 'list', + }, + ], + }), + ) + + await refreshOpenAIChatGPTModels('test', { fetch: fetchMock, force: true }) + + expect(getReasoningLevel('openai:gpt-5.6-sol', false, 'max')).toBe('low') + expect(getThinkingProviderOptions('openai:gpt-5.6-sol', false, 'max')).toEqual({}) + }) + it('uses the raw ChatGPT context window ahead of the static OpenAI model table', async () => { const fetchMock = vi.fn(async () => Response.json({ diff --git a/packages/core/tests/provider-compat.test.ts b/packages/core/tests/provider-compat.test.ts index 23630de..1bb25d0 100644 --- a/packages/core/tests/provider-compat.test.ts +++ b/packages/core/tests/provider-compat.test.ts @@ -1,5 +1,6 @@ import { describe, expect, it, vi } from 'vitest' +import { createDeepSeek } from '@ai-sdk/deepseek' import { createOpenAICompatible } from '@ai-sdk/openai-compatible' import { generateText } from 'ai' import type { ModelMessage } from 'ai' @@ -38,6 +39,7 @@ function imageToolResult(toolCallId: string, data: string, toolName = 'readFile' describe('stripBinaryPartsFromMessages', () => { it('replaces user image/file parts and tool-result media with text notices', () => { + const deliveredFileCache = new Map([['/tmp/image.png', { mtimeMs: 1, size: 1 }]]) const messages = [ { role: 'user', @@ -65,7 +67,8 @@ describe('stripBinaryPartsFromMessages', () => { }, ] as unknown as ModelMessage[] - expect(stripBinaryPartsFromMessages(messages)).toBe(true) + expect(stripBinaryPartsFromMessages(messages, deliveredFileCache)).toBe(true) + expect(deliveredFileCache.size).toBe(0) const userContent = messages[0]!.content as Array<{ type: string; text?: string }> expect(userContent.map((p) => p.type)).toEqual(['text', 'text']) @@ -87,7 +90,7 @@ describe('stripBinaryPartsFromMessages', () => { }) describe('reattachToolResultImagesForProvider', () => { - it('moves a contiguous Kimi tool-image group into one following user message', () => { + it('moves a contiguous DeepSeek tool-image group into one following user message', () => { const messages: ModelMessage[] = [ { role: 'assistant', content: [] }, imageToolResult('tc-1', 'AAAA1'), @@ -95,7 +98,7 @@ describe('reattachToolResultImagesForProvider', () => { { role: 'assistant', content: 'continued' }, ] - const requestMessages = reattachToolResultImagesForProvider(messages, 'moonshotai:kimi-k3') + const requestMessages = reattachToolResultImagesForProvider(messages, 'deepseek:deepseek-flash') expect(requestMessages.map((message) => message.role)).toEqual(['assistant', 'tool', 'tool', 'user', 'assistant']) for (const message of requestMessages.slice(1, 3)) { @@ -169,7 +172,7 @@ describe('reattachToolResultImagesForProvider', () => { }) it('leaves native and text-only transports unchanged', () => { - for (const modelId of ['openai:gpt-5.6-sol', 'anthropic:claude-sonnet-5', 'deepseek:deepseek-v4-flash']) { + for (const modelId of ['openai:gpt-5.6-sol', 'anthropic:claude-sonnet-5', 'deepseek:deepseek-v4-pro']) { const messages = [imageToolResult('tc-1', 'AAAA1')] const original = structuredClone(messages) const requestMessages = reattachToolResultImagesForProvider(messages, modelId) @@ -178,13 +181,13 @@ describe('reattachToolResultImagesForProvider', () => { } }) - it('serializes reattached base64 as image_url rather than tool text', async () => { + it('serializes DeepSeek Flash direct and tool images natively as image_url parts', async () => { let requestBody: { - messages?: Array<{ role?: string; content?: unknown }> + model?: string + messages?: Array<{ role?: string; content?: unknown; reasoning_content?: string }> } = {} - const provider = createOpenAICompatible({ - name: 'moonshotai', - baseURL: 'https://example.test/v1', + const provider = createDeepSeek({ + baseURL: 'https://example.test', apiKey: 'test-key', fetch: async (_input, init) => { requestBody = JSON.parse(String(init?.body)) @@ -193,7 +196,7 @@ describe('reattachToolResultImagesForProvider', () => { id: 'response-1', object: 'chat.completion', created: 0, - model: 'k3', + model: 'deepseek-flash', choices: [ { index: 0, @@ -222,10 +225,12 @@ describe('reattachToolResultImagesForProvider', () => { }, imageToolResult('tc-1', 'QUFBQQ=='), ] - const requestMessages = reattachToolResultImagesForProvider(messages, 'moonshotai:kimi-k3') + const requestMessages = reattachToolResultImagesForProvider(messages, 'deepseek:deepseek-flash') - await generateText({ model: provider('k3'), messages: requestMessages }) + await generateText({ model: provider('deepseek-flash'), messages: requestMessages }) + expect(requestBody.model).toBe('deepseek-flash') + expect(requestBody.messages?.find((message) => message.role === 'assistant')?.reasoning_content).toBe('') const toolMessage = requestBody.messages?.find((message) => message.role === 'tool') expect(toolMessage?.content).toBe('Loaded tc-1') const imageMessage = requestBody.messages?.find( @@ -243,6 +248,34 @@ describe('reattachToolResultImagesForProvider', () => { image_url: { url: 'data:image/png;base64,QUFBQQ==' }, }, ]) + + await generateText({ + model: provider('deepseek-flash'), + messages: [ + { + role: 'user', + content: [ + { type: 'text', text: 'Describe this image directly.' }, + { + type: 'file', + data: { type: 'data', data: 'QUFBQQ==' }, + mediaType: 'image/png', + filename: 'image.png', + }, + { type: 'text', text: 'Analyze the supplied content directly; do not re-read the local path.' }, + ], + }, + ], + }) + + expect(requestBody.messages?.[0]?.content).toEqual([ + { type: 'text', text: 'Describe this image directly.' }, + { + type: 'image_url', + image_url: { url: 'data:image/png;base64,QUFBQQ==' }, + }, + { type: 'text', text: 'Analyze the supplied content directly; do not re-read the local path.' }, + ]) }) }) @@ -280,9 +313,12 @@ describe('downgradeBinaryPartsForProvider', () => { const messages = [{ role: 'user', content: [{ type: 'text', text: 'look' }, imagePart] }] as ModelMessage[] const canonical = structuredClone(messages) - const vision = await downgradeBinaryPartsForProvider(messages, 'moonshotai:kimi-k3') - expect(vision[0]).toEqual(messages[0]) - const text = await downgradeBinaryPartsForProvider(messages, 'deepseek:deepseek-v4-flash') + for (const modelId of ['moonshotai:kimi-k3', 'deepseek:deepseek-flash']) { + const vision = await downgradeBinaryPartsForProvider(messages, modelId) + expect(vision[0], modelId).toEqual(messages[0]) + expect(JSON.stringify(vision), modelId).not.toContain('mock compatibility OCR') + } + const text = await downgradeBinaryPartsForProvider(messages, 'deepseek:deepseek-v4-pro') expect(JSON.stringify(text)).toContain('mock compatibility OCR') expect(JSON.stringify(text)).not.toContain('"type":"file"') expect(messages).toEqual(canonical) @@ -309,8 +345,8 @@ describe('downgradeBinaryPartsForProvider', () => { ] as ModelMessage[] const callsBefore = vi.mocked(ocrImage).mock.calls.length - await downgradeBinaryPartsForProvider(asMessages(first), 'deepseek:deepseek-v4-flash') - await downgradeBinaryPartsForProvider(asMessages(second), 'deepseek:deepseek-v4-flash') + await downgradeBinaryPartsForProvider(asMessages(first), 'deepseek:deepseek-v4-pro') + await downgradeBinaryPartsForProvider(asMessages(second), 'deepseek:deepseek-v4-pro') expect(vi.mocked(ocrImage).mock.calls.length - callsBefore).toBe(2) }) diff --git a/packages/core/tests/provider-registry.test.ts b/packages/core/tests/provider-registry.test.ts index bf40fc6..d3bc16b 100644 --- a/packages/core/tests/provider-registry.test.ts +++ b/packages/core/tests/provider-registry.test.ts @@ -22,7 +22,12 @@ import { XAI_PROMPT_CACHE_KEY_HEADER, applyCacheControl, } from '../src/providers/cache-control.js' -import { createModelRegistry, kimiCodingModelId } from '../src/providers/registry.js' +import { + createModelRegistry, + kimiCodingModelId, + withXaiReasoningHeader, + withZhipuReasoningHeader, +} from '../src/providers/registry.js' function sseResponse(events: unknown[]): Response { return new Response(`${events.map((event) => `data: ${JSON.stringify(event)}\n\n`).join('')}data: [DONE]\n\n`, { @@ -50,6 +55,7 @@ describe('Kimi endpoint model ids', () => { delete process.env.MOONSHOT_API_KEY delete process.env.OPENAI_API_KEY delete process.env.XAI_API_KEY + delete process.env.ZHIPU_API_KEY fs.rmSync(testHome, { recursive: true, force: true }) }) @@ -74,7 +80,7 @@ describe('Kimi endpoint model ids', () => { expect(registry.languageModel('moonshotai:kimi-k2.7-code').modelId).toBe('kimi-k2.7-code') }) - it('moves xAI session affinity into the Responses prompt_cache_key body field', async () => { + it('moves xAI session affinity and xhigh reasoning into the Responses body', async () => { const fetchMock = vi.fn( async () => new Response(JSON.stringify({ error: { message: 'test stop' } }), { @@ -87,14 +93,15 @@ describe('Kimi endpoint model ids', () => { const cache = applyCacheControl({ instructions: 'stable instructions', messages: [{ role: 'user', content: 'hello' }], - modelId: 'xai:grok-4.5', + modelId: 'xai:grok-4.6', sessionId: 'session-1', }) const result = streamText({ - model: createModelRegistry().languageModel('xai:grok-4.5'), + model: createModelRegistry().languageModel('xai:grok-4.6'), instructions: cache.instructions, messages: cache.messages, - headers: cache.headers, + headers: withXaiReasoningHeader(cache.headers, 'xhigh'), + reasoning: 'xhigh', abortSignal: controller.signal, onError: () => undefined, }) @@ -105,13 +112,84 @@ describe('Kimi endpoint model ids', () => { expect(fetchMock).toHaveBeenCalledOnce() const [url, init] = fetchMock.mock.calls[0]! expect(String(url)).toBe('https://api.x.ai/v1/responses') - expect(JSON.parse(String(init?.body))).toMatchObject({ prompt_cache_key: 'session-1' }) + expect(JSON.parse(String(init?.body))).toMatchObject({ + prompt_cache_key: 'session-1', + reasoning: { effort: 'xhigh' }, + }) const headers = new Headers(init?.headers) expect(headers.get(XAI_PROMPT_CACHE_KEY_HEADER)).toBeNull() expect(headers.get('x-grok-conv-id')).toBeNull() expect(init?.signal).toBe(controller.signal) }) + it('omits unsupported reasoning parameters for the Grok 4.20 alias', async () => { + const fetchMock = vi.fn(async () => + Response.json({ error: { message: 'test stop' } }, { status: 400 }), + ) + vi.stubGlobal('fetch', fetchMock) + const result = streamText({ + model: createModelRegistry().languageModel('xai:grok-4.20'), + messages: [{ role: 'user', content: 'hello' }], + headers: withXaiReasoningHeader(undefined, undefined), + onError: () => undefined, + }) + for await (const _chunk of result.textStream) { + // The mock returns a deliberate error after the outbound request is captured. + } + + const [, init] = fetchMock.mock.calls[0]! + const body = JSON.parse(String(init?.body)) + expect(body).not.toHaveProperty('reasoning') + expect(body).not.toHaveProperty('reasoning_effort') + expect(new Headers(init?.headers).get('x-x-code-xai-reasoning-effort')).toBeNull() + }) + + it('serializes max Anthropic effort with adaptive thinking enabled', async () => { + process.env.ANTHROPIC_API_KEY = 'test-key' + const fetchMock = vi.fn(async () => + Response.json({ error: { message: 'test stop' } }, { status: 400 }), + ) + vi.stubGlobal('fetch', fetchMock) + const result = streamText({ + model: createModelRegistry().languageModel('anthropic:claude-opus-5'), + messages: [{ role: 'user', content: 'hello' }], + reasoning: 'xhigh', + providerOptions: { + anthropic: { thinking: { type: 'adaptive' }, effort: 'max' }, + }, + onError: () => undefined, + }) + for await (const _chunk of result.textStream) { + // The mock returns a deliberate error after the outbound request is captured. + } + + const body = JSON.parse(String(fetchMock.mock.calls[0]?.[1]?.body)) + expect(body.thinking).toEqual({ type: 'adaptive' }) + expect(body.output_config).toMatchObject({ effort: 'max' }) + }) + + it('serializes the minimum Zhipu reasoning effort for always-thinking models', async () => { + process.env.ZHIPU_API_KEY = 'test-key' + const fetchMock = vi.fn(async () => + Response.json({ error: { message: 'test stop' } }, { status: 400 }), + ) + vi.stubGlobal('fetch', fetchMock) + const result = streamText({ + model: createModelRegistry().languageModel('zhipu:glm-5.3-flash'), + messages: [{ role: 'user', content: 'hello' }], + headers: withZhipuReasoningHeader(undefined, 'low'), + onError: () => undefined, + }) + for await (const _chunk of result.textStream) { + // The mock returns a deliberate error after the outbound request is captured. + } + + const [url, init] = fetchMock.mock.calls[0]! + expect(String(url)).toBe('https://open.bigmodel.cn/api/paas/v4/chat/completions') + expect(JSON.parse(String(init?.body))).toMatchObject({ model: 'glm-5.3-flash', reasoning_effort: 'low' }) + expect(new Headers(init?.headers).get('x-x-code-zhipu-reasoning-effort')).toBeNull() + }) + it('serializes Anthropic cache breakpoints without putting a system role in messages', async () => { process.env.ANTHROPIC_API_KEY = 'test-key' vi.stubEnv('ANTHROPIC_BASE_URL', 'https://api.anthropic.com') diff --git a/packages/core/tests/read-file.test.ts b/packages/core/tests/read-file.test.ts index 96b8f1a..0f116b2 100644 --- a/packages/core/tests/read-file.test.ts +++ b/packages/core/tests/read-file.test.ts @@ -228,7 +228,7 @@ describe('readFile tool', () => { const filePath = path.join(tmpDir, 'image.png') const { Jimp } = await import('jimp') await fs.writeFile(filePath, await new Jimp({ width: 3, height: 2, color: 0xffffffff }).getBuffer('image/png')) - const tool = createReadFileTool(undefined, { modelId: 'deepseek:deepseek-v4-flash' }) + const tool = createReadFileTool(undefined, { modelId: 'deepseek:deepseek-v4-pro' }) const result = await tool.execute!({ filePath }, { toolCallId: 'image-ocr-test', messages: [], @@ -239,6 +239,45 @@ describe('readFile tool', () => { expect(JSON.stringify(result)).not.toContain('image-data') await fs.rm(tmpDir, { recursive: true }) }) + + it('returns native image content to DeepSeek Flash without local OCR', async () => { + const tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'xc-rf-deepseek-image-')) + const filePath = path.join(tmpDir, 'image.png') + const { Jimp } = await import('jimp') + await fs.writeFile(filePath, await new Jimp({ width: 3, height: 2, color: 0xffffffff }).getBuffer('image/png')) + const tool = createReadFileTool(undefined, { modelId: 'deepseek:deepseek-flash' }) + const result = await tool.execute!({ filePath }, { + toolCallId: 'deepseek-image-test', + messages: [], + abortSignal: undefined, + } as never) + + expect(result).toMatchObject({ + type: 'content', + value: expect.arrayContaining([ + expect.objectContaining({ type: 'file', mediaType: 'image/png' }), + expect.objectContaining({ type: 'text', text: expect.stringContaining('Do not invoke readFile') }), + ]), + }) + expect(JSON.stringify(result)).not.toContain('mock readFile OCR') + await fs.rm(tmpDir, { recursive: true }) + }) + + it('does not deliver an unchanged image twice through readFile', async () => { + const tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'xc-rf-image-cache-')) + const filePath = path.join(tmpDir, 'image.png') + const { Jimp } = await import('jimp') + await fs.writeFile(filePath, await new Jimp({ width: 3, height: 2, color: 0xffffffff }).getBuffer('image/png')) + const tool = createReadFileTool(new Map(), { modelId: 'deepseek:deepseek-flash' }) + const options = { toolCallId: 'deepseek-image-cache-test', messages: [], abortSignal: undefined } as never + + const first = await tool.execute!({ filePath }, options) + const second = await tool.execute!({ filePath }, options) + + expect(first).toMatchObject({ type: 'content' }) + expect(second).toContain('is unchanged since its full content was added') + await fs.rm(tmpDir, { recursive: true }) + }) }) describe('parsePdfPageRange', () => { diff --git a/packages/core/tests/thinking.test.ts b/packages/core/tests/thinking.test.ts index 5e25a9f..728bdb5 100644 --- a/packages/core/tests/thinking.test.ts +++ b/packages/core/tests/thinking.test.ts @@ -1,4 +1,13 @@ -import { getReasoningLevel, getThinkingProviderOptions, supportsReasoningTier } from '../src/providers/thinking.js' +import { createDeepSeek } from '@ai-sdk/deepseek' +import { generateText } from 'ai' + +import { + getReasoningEffort, + getReasoningLevel, + getReasoningTierOptions, + getThinkingProviderOptions, + supportsReasoningTier, +} from '../src/providers/thinking.js' import { isolateOpenAIAuth } from './provider-env.js' let restoreOpenAIAuth: () => void @@ -12,10 +21,18 @@ afterEach(() => restoreOpenAIAuth()) describe('supportsReasoningTier', () => { it('is true for tier-capable models', () => { expect(supportsReasoningTier('moonshotai:kimi-k3')).toBe(true) + expect(supportsReasoningTier('google:gemini-3.8-flash')).toBe(true) + expect(supportsReasoningTier('google:gemini-3.7-flash')).toBe(true) + expect(supportsReasoningTier('google:gemini-3.6-flash')).toBe(true) expect(supportsReasoningTier('google:gemini-3.5-flash')).toBe(true) + expect(supportsReasoningTier('openai:gpt-5.6')).toBe(true) expect(supportsReasoningTier('openai:gpt-5.6-sol')).toBe(true) + expect(supportsReasoningTier('openai:gpt-5.5')).toBe(true) expect(supportsReasoningTier('anthropic:claude-sonnet-5')).toBe(true) expect(supportsReasoningTier('xai:grok-4.5')).toBe(true) + expect(supportsReasoningTier('xai:grok-4.6')).toBe(true) + expect(supportsReasoningTier('anthropic:claude-fable-5-1')).toBe(true) + expect(supportsReasoningTier('openai:gpt-6-astra')).toBe(true) }) it('is false for models whose provider has tiers but the model family does not', () => { @@ -23,15 +40,19 @@ describe('supportsReasoningTier', () => { expect(supportsReasoningTier('google:gemini-2.5-flash')).toBe(false) expect(supportsReasoningTier('moonshotai:kimi-k2.6')).toBe(false) expect(supportsReasoningTier('moonshotai:kimi-k2.7-code')).toBe(false) + expect(supportsReasoningTier('xai:grok-4.20')).toBe(false) + expect(supportsReasoningTier('xai:grok-4.20-non-reasoning')).toBe(false) }) - it('is true for DeepSeek V4 models', () => { - expect(supportsReasoningTier('deepseek:deepseek-v4-flash')).toBe(true) + it('is true for current DeepSeek Flash and V4 models', () => { + expect(supportsReasoningTier('deepseek:deepseek-flash')).toBe(true) expect(supportsReasoningTier('deepseek:deepseek-v4-pro')).toBe(true) }) it('is true for Zhipu GLM-5.2', () => { expect(supportsReasoningTier('zhipu:glm-5.2')).toBe(true) + expect(supportsReasoningTier('zhipu:glm-5.3')).toBe(true) + expect(supportsReasoningTier('zhipu:glm-5.3-flash')).toBe(true) }) it('is false for providers without any tier support', () => { @@ -46,25 +67,85 @@ describe('supportsReasoningTier', () => { }) }) +describe('model-specific reasoning profiles', () => { + it('returns only tiers supported by the selected model family', () => { + expect(getReasoningTierOptions('google:gemini-3.8-flash')?.map((tier) => tier.value)).toEqual([ + 'low', + 'medium', + 'high', + ]) + expect(getReasoningTierOptions('google:gemini-3.6-flash')?.map((tier) => tier.value)).toEqual([ + 'minimal', + 'low', + 'medium', + 'high', + ]) + expect(getReasoningTierOptions('google:gemini-3.5-flash')?.map((tier) => tier.value)).toEqual([ + 'minimal', + 'low', + 'medium', + 'high', + ]) + expect(getReasoningTierOptions('openai:gpt-5.4-mini')?.map((tier) => tier.value)).toEqual([ + 'low', + 'medium', + 'high', + 'xhigh', + ]) + expect(getReasoningTierOptions('openai:gpt-5.6')?.map((tier) => tier.value)).toEqual([ + 'low', + 'medium', + 'high', + 'xhigh', + 'max', + ]) + expect(getReasoningTierOptions('openai:gpt-5.5')?.map((tier) => tier.value)).toEqual([ + 'low', + 'medium', + 'high', + 'xhigh', + ]) + }) + + it('floors disabled reasoning for models that reject off', () => { + for (const id of [ + 'anthropic:claude-fable-5-1', + 'openai:gpt-6-astra', + 'google:gemini-3.8-flash', + 'google:gemini-3.7-flash', + 'xai:grok-4.6', + 'xai:grok-4.5', + 'zhipu:glm-5.3', + ]) { + expect(getReasoningEffort(id, false), id).toBe('low') + } + expect(getReasoningLevel('google:gemini-3.8-flash', false)).toBe('low') + expect(getReasoningEffort('google:gemini-3.6-flash', false)).toBe('minimal') + expect(getReasoningLevel('openai:gpt-6-astra', false)).toBe('low') + }) +}) + describe('getReasoningLevel', () => { it('returns the effort tier for tier-capable models', () => { - expect(getReasoningLevel('deepseek:deepseek-v4-flash', false, 'high')).toBe('high') + expect(getReasoningLevel('deepseek:deepseek-flash', false, 'high')).toBeUndefined() + expect(getReasoningLevel('openai:gpt-5.6', false, 'low')).toBe('low') expect(getReasoningLevel('openai:gpt-5.6-sol', false, 'medium')).toBe('medium') + expect(getReasoningLevel('openai:gpt-5.5', false, 'xhigh')).toBe('xhigh') expect(getReasoningLevel('anthropic:claude-sonnet-5', true, 'low')).toBe('low') }) - it('maps "max" tier to "xhigh" reasoning level', () => { - expect(getReasoningLevel('deepseek:deepseek-v4-flash', false, 'max')).toBe('xhigh') + it('maps portable "max" tiers to "xhigh" while leaving DeepSeek on its native option', () => { + expect(getReasoningLevel('deepseek:deepseek-flash', false, 'max')).toBeUndefined() expect(getReasoningLevel('moonshotai:kimi-k3', false, 'max')).toBe('xhigh') }) it('returns high when enabled and no effort specified', () => { - expect(getReasoningLevel('deepseek:deepseek-v4-flash', true)).toBe('high') + expect(getReasoningLevel('deepseek:deepseek-flash', true)).toBe('high') expect(getReasoningLevel('openai:gpt-5.6-sol', true)).toBe('high') }) it('returns none when disabled and no effort specified', () => { - expect(getReasoningLevel('deepseek:deepseek-v4-flash', false)).toBe('none') + expect(getReasoningLevel('deepseek:deepseek-flash', false)).toBe('none') expect(getReasoningLevel('openai:gpt-5.6-sol', false)).toBe('none') }) @@ -74,6 +155,12 @@ describe('getReasoningLevel', () => { expect(getReasoningLevel('custom:my-model', true)).toBeUndefined() }) + it('omits reasoning control for fixed-reasoning Grok 4.20 variants', () => { + expect(getReasoningLevel('xai:grok-4.20', false)).toBeUndefined() + expect(getReasoningLevel('xai:grok-4.20', true, 'low')).toBeUndefined() + expect(getReasoningLevel('xai:grok-4.20-non-reasoning', true, 'high')).toBeUndefined() + }) + it('ignores effort for models that do not support tiers', () => { expect(getReasoningLevel('deepseek:deepseek-chat', false, 'high')).toBe('none') expect(getReasoningLevel('deepseek:deepseek-chat', true, 'high')).toBe('high') @@ -101,14 +188,73 @@ describe('getThinkingProviderOptions', () => { it('returns zhipu thinking enabled when tier is set', () => { expect(getThinkingProviderOptions('zhipu:glm-5.2', false, 'high')).toEqual({ - zhipu: { thinking: { type: 'enabled' } }, + zhipu: { thinking: { type: 'enabled' }, reasoningEffort: 'high' }, + }) + }) + + it('keeps always-thinking Zhipu models enabled at low effort when switched off', () => { + expect(getThinkingProviderOptions('zhipu:glm-5.3-flash', false)).toEqual({ + zhipu: { thinking: { type: 'enabled' }, reasoningEffort: 'low' }, }) }) + it.each(['low', 'high', 'max'])('uses the provider-native DeepSeek %s tier', (effort) => { + expect(getThinkingProviderOptions('deepseek:deepseek-flash', true, effort)).toEqual({ + deepseek: { reasoningEffort: effort }, + }) + }) + + it('uses provider-native max where the portable reasoning enum stops at xhigh', () => { + expect(getThinkingProviderOptions('anthropic:claude-opus-5', true, 'max')).toEqual({ + anthropic: { thinking: { type: 'adaptive' }, effort: 'max' }, + }) + expect(getThinkingProviderOptions('openai:gpt-6-astra', true, 'max')).toEqual({ + openai: { reasoningEffort: 'max' }, + }) + expect(getThinkingProviderOptions('openai:gpt-5.4-mini', true, 'max')).toEqual({}) + }) + it('returns empty object for providers using top-level reasoning', () => { - expect(getThinkingProviderOptions('deepseek:deepseek-v4-flash', true)).toEqual({}) + expect(getThinkingProviderOptions('deepseek:deepseek-flash', true)).toEqual({}) expect(getThinkingProviderOptions('openai:gpt-5.6-sol', true, 'high')).toEqual({}) expect(getThinkingProviderOptions('anthropic:claude-sonnet-5', true)).toEqual({}) expect(getThinkingProviderOptions('google:gemini-3.5-flash', true, 'low')).toEqual({}) }) + + it('sends DeepSeek max natively without an xhigh compatibility warning', async () => { + let requestBody: { reasoning_effort?: string } = {} + const provider = createDeepSeek({ + baseURL: 'https://example.test', + apiKey: 'test-key', + fetch: async (_input, init) => { + requestBody = JSON.parse(String(init?.body)) + return new Response( + JSON.stringify({ + id: 'response-1', + object: 'chat.completion', + created: 0, + model: 'deepseek-flash', + choices: [{ index: 0, message: { role: 'assistant', content: 'done' }, finish_reason: 'stop' }], + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, + }), + { status: 200, headers: { 'content-type': 'application/json' } }, + ) + }, + }) + const modelId = 'deepseek:deepseek-flash' + const reasoning = getReasoningLevel(modelId, true, 'max') + const providerOptions = getThinkingProviderOptions(modelId, true, 'max') + + const result = await generateText({ + model: provider('deepseek-flash'), + messages: [{ role: 'user', content: 'hello' }], + ...(reasoning ? { reasoning } : {}), + providerOptions: providerOptions as Parameters[0]['providerOptions'], + }) + + expect(requestBody.reasoning_effort).toBe('max') + expect(result.warnings ?? []).not.toEqual( + expect.arrayContaining([expect.objectContaining({ feature: 'reasoning' })]), + ) + }) }) diff --git a/packages/core/tests/tool-result-images.test.ts b/packages/core/tests/tool-result-images.test.ts index 4b08aca..5def177 100644 --- a/packages/core/tests/tool-result-images.test.ts +++ b/packages/core/tests/tool-result-images.test.ts @@ -45,21 +45,25 @@ describe('deliverToolImages', () => { expect(captionImageBuffer).not.toHaveBeenCalled() }) - it('keeps Kimi images in canonical tool history for request-time reattachment', async () => { - const r = await deliverToolImages(ctx('moonshotai:kimi-k3'), 'shot taken', IMG) - expect(r.images).toEqual(IMG) - expect(r.text).toBe('shot taken') + it('keeps Chat Completions vision images in canonical tool history for request-time reattachment', async () => { + for (const modelId of ['moonshotai:kimi-k3', 'deepseek:deepseek-flash']) { + const r = await deliverToolImages(ctx(modelId), 'shot taken', IMG) + expect(r.images, modelId).toEqual(IMG) + expect(r.text, modelId).toBe('shot taken') + } expect(captionImageBuffer).not.toHaveBeenCalled() }) - it('does not borrow a separate vision provider when the active Kimi model can view the image', async () => { + it('does not borrow a separate vision provider when the active model can view the image', async () => { vi.mocked(pickVisionProvider).mockReturnValue({ provider: 'google', modelId: 'google:gemini-2.5-flash', label: 'Gemini 2.5 Flash', }) - const r = await deliverToolImages(ctx('moonshotai:kimi-k3'), 'shot taken', IMG) - expect(r.images).toEqual(IMG) + for (const modelId of ['moonshotai:kimi-k3', 'deepseek:deepseek-flash']) { + const r = await deliverToolImages(ctx(modelId), 'shot taken', IMG) + expect(r.images, modelId).toEqual(IMG) + } expect(captionImageBuffer).not.toHaveBeenCalled() }) @@ -69,7 +73,7 @@ describe('deliverToolImages', () => { modelId: 'google:gemini-2.5-flash', label: 'Gemini 2.5 Flash', }) - const r = await deliverToolImages(ctx('deepseek:deepseek-v4-flash'), 'shot taken', IMG) + const r = await deliverToolImages(ctx('deepseek:deepseek-v4-pro'), 'shot taken', IMG) expect(r.images).toBeUndefined() expect(r.text).toContain('A MAP OF BERLIN') expect(r.text).toContain('Privacy notice') @@ -83,7 +87,7 @@ describe('deliverToolImages', () => { modelId: 'google:gemini-2.5-flash', label: 'Gemini 2.5 Flash', }) - await deliverToolImages(ctx('deepseek:deepseek-v4-flash'), 'shot taken', IMG, { + await deliverToolImages(ctx('deepseek:deepseek-v4-pro'), 'shot taken', IMG, { captionPrompt: 'Report only visible UI defects.', maxOutputTokens: 400, }) @@ -115,7 +119,7 @@ describe('deliverToolImages', () => { releaseUsageWrite = resolve }), ) - const context = ctx('deepseek:deepseek-v4-flash') + const context = ctx('deepseek:deepseek-v4-pro') let completed = false const resultPromise = deliverToolImages(context, 'shot taken', IMG).then((result) => { completed = true @@ -134,14 +138,14 @@ describe('deliverToolImages', () => { }) it('drops the image with a clear note when no vision model is available', async () => { - const r = await deliverToolImages(ctx('deepseek:deepseek-v4-flash'), 'shot taken', IMG) + const r = await deliverToolImages(ctx('deepseek:deepseek-v4-pro'), 'shot taken', IMG) expect(r.images).toBeUndefined() expect(r.text).toContain('no vision model is available') expect(captionImageBuffer).not.toHaveBeenCalled() }) it('uses a caller-specific fallback when no accessibility snapshot exists', async () => { - const r = await deliverToolImages(ctx('deepseek:deepseek-v4-flash'), 'shot taken', IMG, { + const r = await deliverToolImages(ctx('deepseek:deepseek-v4-pro'), 'shot taken', IMG, { unavailableFallback: 'No accessibility snapshot was returned for this check.', }) @@ -157,7 +161,7 @@ describe('deliverToolImages', () => { label: 'Gemini 2.5 Flash', }) const two = [...IMG, { data: Buffer.from('second').toString('base64'), mediaType: 'image/png' }] - const r = await deliverToolImages(ctx('deepseek:deepseek-v4-flash'), 'shots', two) + const r = await deliverToolImages(ctx('deepseek:deepseek-v4-pro'), 'shots', two) expect(captionImageBuffer).toHaveBeenCalledTimes(2) expect(r.text).toContain('Screenshot 1') expect(r.text).toContain('Screenshot 2') diff --git a/packages/core/tests/vision-fallback.test.ts b/packages/core/tests/vision-fallback.test.ts index b274c1c..294cbeb 100644 --- a/packages/core/tests/vision-fallback.test.ts +++ b/packages/core/tests/vision-fallback.test.ts @@ -89,10 +89,19 @@ describe('pickVisionProvider', () => { expect(pickVisionProvider()?.provider).toBe('xai') }) - it('ignores DeepSeek key when picking — still selects vision provider if present', () => { + it('uses DeepSeek Flash when it is the only configured vision provider', () => { + process.env.DEEPSEEK_API_KEY = 'test' + expect(pickVisionProvider()).toEqual({ + provider: 'deepseek', + modelId: 'deepseek:deepseek-flash', + label: 'DeepSeek V4.1 Flash', + }) + }) + + it('prefers DeepSeek Flash over Anthropic for borrowed vision', () => { process.env.DEEPSEEK_API_KEY = 'test' process.env.ANTHROPIC_API_KEY = 'test' - expect(pickVisionProvider()?.provider).toBe('anthropic') + expect(pickVisionProvider()?.provider).toBe('deepseek') }) }) diff --git a/packages/core/tests/web-search.test.ts b/packages/core/tests/web-search.test.ts index ca05493..e014bf9 100644 --- a/packages/core/tests/web-search.test.ts +++ b/packages/core/tests/web-search.test.ts @@ -200,6 +200,7 @@ describe('webSearch', () => { }), }) const body = JSON.parse(init!.body as string) + expect(body.model).toBe('deepseek-flash') expect(body.tools).toEqual([{ type: 'web_search_20250305', name: 'web_search', max_uses: 5 }]) // Deduped by URL, snippet stitched from the text-block citation. expect(result).toContain('cited snippet') diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index b3600ab..13d1c33 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -138,8 +138,8 @@ importers: specifier: ^4.0.0 version: 4.0.25(zod@3.25.76) '@ai-sdk/deepseek': - specifier: ^3.0.28 - version: 3.0.28(zod@3.25.76) + specifier: ^3.0.45 + version: 3.0.45(zod@3.25.76) '@ai-sdk/google': specifier: ^4.0.0 version: 4.0.29(zod@3.25.76) @@ -261,8 +261,8 @@ packages: peerDependencies: zod: ^3.25.76 || ^4.1.8 - '@ai-sdk/deepseek@3.0.28': - resolution: {integrity: sha512-ySiafZL/1KQq8NF49eGWmVRYP4ezBjxAwAlMK+hhTV+9ee37LcCxlSIbURe4C4KFq8kYaTFKsTGI2poLFLOu0w==} + '@ai-sdk/deepseek@3.0.45': + resolution: {integrity: sha512-b2rEm2IdG1/JfLpry0rNi8A7dav9cvXZviFUo0g0XRRYAkNPkOgsO2XMEYfCQvKKymt2uWiqmDijxcZh17oYXg==} engines: {node: '>=22'} peerDependencies: zod: ^3.25.76 || ^4.1.8 @@ -303,18 +303,18 @@ packages: peerDependencies: zod: ^3.25.76 || ^4.1.8 - '@ai-sdk/provider-utils@5.0.27': - resolution: {integrity: sha512-EzAn4pdgG5g0xXtH6lE2zyNmfjDQIDjATkfqzuidEI35g++hh4+07vnjzkT/RmGmIClPZiRj/Q2GMPV2V7mkHw==} + '@ai-sdk/provider-utils@5.0.41': + resolution: {integrity: sha512-hSosMII5k33z0buaZ6dugH/g8yIJjPQBILuQvr4ThdiMJxrkiZxFWTgIkvKsT6Dk493JbsnGz5W/q88Y5bR7WA==} engines: {node: '>=22'} peerDependencies: zod: ^3.25.76 || ^4.1.8 - '@ai-sdk/provider@4.0.4': - resolution: {integrity: sha512-tbHKNLirllUNF3ZlkCsXnwab2ZV1Sl4b1H/Cp9ruCce15IBmskE8Gwkk0yo9xDWY+jho2of7lVXtwSsyrq7cwQ==} + '@ai-sdk/provider@4.0.15': + resolution: {integrity: sha512-LIDCrT8xgWVMiVUV13+v9FXPJ8B5uX/5ByEMXxp9Sspr2SyB1PwMSCPshZoI/t+4AgxyZCFEVutdaigCnqje3w==} engines: {node: '>=22'} - '@ai-sdk/provider@4.0.7': - resolution: {integrity: sha512-6or44XprPzKbr8zkmzosowSE0pxkvJcoojBL+mCZvPUt3kvXp3XSNqeVun9golb1acEfSo6yaEBRT18h2VU+1Q==} + '@ai-sdk/provider@4.0.4': + resolution: {integrity: sha512-tbHKNLirllUNF3ZlkCsXnwab2ZV1Sl4b1H/Cp9ruCce15IBmskE8Gwkk0yo9xDWY+jho2of7lVXtwSsyrq7cwQ==} engines: {node: '>=22'} '@ai-sdk/xai@4.0.23': @@ -3363,10 +3363,10 @@ snapshots: '@ai-sdk/provider-utils': 5.0.16(zod@3.25.76) zod: 3.25.76 - '@ai-sdk/deepseek@3.0.28(zod@3.25.76)': + '@ai-sdk/deepseek@3.0.45(zod@3.25.76)': dependencies: - '@ai-sdk/provider': 4.0.7 - '@ai-sdk/provider-utils': 5.0.27(zod@3.25.76) + '@ai-sdk/provider': 4.0.15 + '@ai-sdk/provider-utils': 5.0.41(zod@3.25.76) zod: 3.25.76 '@ai-sdk/gateway@4.0.33(zod@3.25.76)': @@ -3410,20 +3410,20 @@ snapshots: undici: 7.29.0 zod: 3.25.76 - '@ai-sdk/provider-utils@5.0.27(zod@3.25.76)': + '@ai-sdk/provider-utils@5.0.41(zod@3.25.76)': dependencies: - '@ai-sdk/provider': 4.0.7 + '@ai-sdk/provider': 4.0.15 '@standard-schema/spec': 1.1.0 '@workflow/serde': 4.1.0 eventsource-parser: 3.1.1 undici: 7.29.0 zod: 3.25.76 - '@ai-sdk/provider@4.0.4': + '@ai-sdk/provider@4.0.15': dependencies: json-schema: 0.4.0 - '@ai-sdk/provider@4.0.7': + '@ai-sdk/provider@4.0.4': dependencies: json-schema: 0.4.0