From f72373ceebb0b3370573d27baa42c1ef30f7ef6d Mon Sep 17 00:00:00 2001 From: fylorn <249551762+fylorn@users.noreply.github.com> Date: Tue, 6 Oct 2026 10:25:23 +0800 Subject: [PATCH] The model listing says whether a model reasons and takes images, and its output limit; both can be set by hand, v0.64.0 - /v1/models carries max_output_tokens, supports_reasoning and input_modalities (OpenAI/Anthropic shapes) and thinking (Gemini), each only when known - model_specs gain reasoning and image_input; the price table keeps 'not stated' apart from 'no' and reads supports_vision - ModelRow/ModelSpecSave gain the two fields; protocol 40 - the speed probe leaves room to reason by the resolved spec Co-Authored-By: Claude Opus 5.5 --- Cargo.lock | 38 +++--- Cargo.toml | 2 +- crates/tw-api/msg-codes.txt | 2 +- crates/tw-api/src/lib.rs | 26 +++- crates/tw-config/src/edit.rs | 3 +- crates/tw-config/src/lib.rs | 2 +- crates/tw-config/src/model_specs.rs | 157 +++++++++++++++-------- crates/tw-config/src/validate.rs | 13 +- crates/tw-config/tests/manual/schema.rs | 24 +++- crates/tw-control/src/aliases.rs | 2 +- crates/tw-control/src/lib.rs | 18 +-- crates/tw-control/src/resources.rs | 21 ++- crates/tw-control/tests/model_specs.rs | 30 ++++- crates/tw-gateway/src/l3.rs | 80 +++++++----- crates/tw-gateway/src/server/listing.rs | 73 +++++++++-- crates/tw-gateway/src/server/pipeline.rs | 2 +- crates/tw-gateway/src/translate.rs | 4 +- crates/tw-gateway/tests/chatgpt.rs | 6 +- crates/tw-pricing/src/book.rs | 1 + crates/tw-pricing/src/lib.rs | 12 +- crates/tw-pricing/src/sheet.rs | 5 +- crates/tw-pricing/src/table.rs | 17 ++- docs/config.md | 4 +- docs/config.zh-CN.md | 4 +- release-notes/0.64.0.md | 22 ++++ 25 files changed, 391 insertions(+), 177 deletions(-) create mode 100644 release-notes/0.64.0.md diff --git a/Cargo.lock b/Cargo.lock index d5a29f3e..1d096996 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3143,7 +3143,7 @@ dependencies = [ [[package]] name = "tw-api" -version = "0.63.0" +version = "0.64.0" dependencies = [ "serde", "serde_json", @@ -3154,7 +3154,7 @@ dependencies = [ [[package]] name = "tw-bedrock" -version = "0.63.0" +version = "0.64.0" dependencies = [ "aws-credential-types", "aws-sigv4", @@ -3175,7 +3175,7 @@ dependencies = [ [[package]] name = "tw-breaker" -version = "0.63.0" +version = "0.64.0" dependencies = [ "serde", "serde_json", @@ -3183,7 +3183,7 @@ dependencies = [ [[package]] name = "tw-config" -version = "0.63.0" +version = "0.64.0" dependencies = [ "blake3", "libc", @@ -3208,7 +3208,7 @@ dependencies = [ [[package]] name = "tw-control" -version = "0.63.0" +version = "0.64.0" dependencies = [ "axum", "base64", @@ -3250,7 +3250,7 @@ dependencies = [ [[package]] name = "tw-dialect" -version = "0.63.0" +version = "0.64.0" dependencies = [ "serde", "serde_json", @@ -3258,7 +3258,7 @@ dependencies = [ [[package]] name = "tw-engine" -version = "0.63.0" +version = "0.64.0" dependencies = [ "serde", "serde_json", @@ -3271,7 +3271,7 @@ dependencies = [ [[package]] name = "tw-gateway" -version = "0.63.0" +version = "0.64.0" dependencies = [ "arc-swap", "async-stream", @@ -3320,7 +3320,7 @@ dependencies = [ [[package]] name = "tw-guard" -version = "0.63.0" +version = "0.64.0" dependencies = [ "base64", "bytes", @@ -3335,7 +3335,7 @@ dependencies = [ [[package]] name = "tw-link" -version = "0.63.0" +version = "0.64.0" dependencies = [ "serde", "serde_json", @@ -3349,7 +3349,7 @@ dependencies = [ [[package]] name = "tw-observe" -version = "0.63.0" +version = "0.64.0" dependencies = [ "tokio", "tracing", @@ -3358,7 +3358,7 @@ dependencies = [ [[package]] name = "tw-plugin" -version = "0.63.0" +version = "0.64.0" dependencies = [ "libc", "rand 0.10.2", @@ -3373,7 +3373,7 @@ dependencies = [ [[package]] name = "tw-pricing" -version = "0.63.0" +version = "0.64.0" dependencies = [ "arc-swap", "flate2", @@ -3386,14 +3386,14 @@ dependencies = [ [[package]] name = "tw-secret" -version = "0.63.0" +version = "0.64.0" dependencies = [ "thiserror", ] [[package]] name = "tw-store" -version = "0.63.0" +version = "0.64.0" dependencies = [ "blake3", "bytes", @@ -3413,7 +3413,7 @@ dependencies = [ [[package]] name = "tw-types" -version = "0.63.0" +version = "0.64.0" dependencies = [ "serde", "serde_json", @@ -3422,7 +3422,7 @@ dependencies = [ [[package]] name = "tw-watch" -version = "0.63.0" +version = "0.64.0" dependencies = [ "notify", "tempfile", @@ -3432,7 +3432,7 @@ dependencies = [ [[package]] name = "tw-yaml" -version = "0.63.0" +version = "0.64.0" dependencies = [ "saphyr-parser", "serde_yaml_ng", @@ -3442,7 +3442,7 @@ dependencies = [ [[package]] name = "twcore" -version = "0.63.0" +version = "0.64.0" dependencies = [ "anyhow", "axum", diff --git a/Cargo.toml b/Cargo.toml index e8e39972..f91dda61 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -54,7 +54,7 @@ categories = ["network-programming", "web-programming::http-server"] # 二进制走 CalVer,crate 走 SemVer —— 两者是两回事,卖给不同的人。 # 这里是 crate 的版本。 -version = "0.63.0" +version = "0.64.0" [workspace.dependencies] # ── 内部 crate ─────────────────────────────────────────── diff --git a/crates/tw-api/msg-codes.txt b/crates/tw-api/msg-codes.txt index b9162218..2513b074 100644 --- a/crates/tw-api/msg-codes.txt +++ b/crates/tw-api/msg-codes.txt @@ -75,7 +75,7 @@ config.key_limit_not_positive config.key_limit_retention config.key_limit_two_measures config.model_spec_blank_model -config.model_spec_empty +config.model_spec_nothing_set config.model_spec_wildcard config.model_spec_zero config.name_collision diff --git a/crates/tw-api/src/lib.rs b/crates/tw-api/src/lib.rs index 3a662816..122ae583 100644 --- a/crates/tw-api/src/lib.rs +++ b/crates/tw-api/src/lib.rs @@ -794,7 +794,11 @@ pub const MSG_CODES: &str = include_str!("../msg-codes.txt"); /// Responses 的 WebSocket 连接上每个 `response.create` 是一行请求(带用量和费用,按轮算上限、 /// 并发、快慢和成败),连接本身不留行;Realtime 和别的路径照旧整条连接一行。照 38 写的界面 /// 保存组和密钥时会把权重、上限丢掉。 -pub const CONTROL_API_VERSION: u32 = 39; +/// +/// **40 起模型规格多了会不会推理、收不收图**:[`ModelRow`] 多了 `reasoning`、`image_input` +/// 和它们的来源,[`ModelSpecSave`] 多了 `reasoning`、`image_input`(四项都空才是删掉)。 +/// 照 39 写的界面保存规格时会把这两项丢掉。 +pub const CONTROL_API_VERSION: u32 = 40; #[derive(Debug, Clone, Serialize, Deserialize)] #[cfg_attr(feature = "ts", derive(ts_rs::TS))] @@ -3098,6 +3102,18 @@ pub struct ModelRow { /// `max_output_tokens` 从哪儿来。不知道输出上限时没有 #[serde(default, skip_serializing_if = "Option::is_none")] pub max_output_tokens_source: Option, + /// 会不会推理:这一家手写的,没写时来自价目表。不知道时没有 + #[serde(default, skip_serializing_if = "Option::is_none")] + pub reasoning: Option, + /// `reasoning` 从哪儿来。不知道时没有 + #[serde(default, skip_serializing_if = "Option::is_none")] + pub reasoning_source: Option, + /// 收不收图:这一家手写的,没写时来自价目表。不知道时没有 + #[serde(default, skip_serializing_if = "Option::is_none")] + pub image_input: Option, + /// `image_input` 从哪儿来。不知道时没有 + #[serde(default, skip_serializing_if = "Option::is_none")] + pub image_input_source: Option, /// 按这个上游选的价目表查到的价格。空 = 无法计价 #[serde(default, skip_serializing_if = "Option::is_none")] pub price: Option, @@ -3162,7 +3178,7 @@ pub struct ProviderSave { } /// 设一家上游的一个模型的规格(`PUT /provider-model-spec`):价目表不认识这个模型、 -/// 或者写错了时手写。**两项都空就是删掉这一项**,回到价目表。 +/// 或者写错了时手写。**四项都空就是删掉这一项**,回到价目表。 #[derive(Debug, Clone, Default, Serialize, Deserialize)] #[cfg_attr(feature = "ts", derive(ts_rs::TS))] pub struct ModelSpecSave { @@ -3175,6 +3191,12 @@ pub struct ModelSpecSave { /// 输出上限(token)。空 = 用价目表的 #[serde(default, skip_serializing_if = "Option::is_none")] pub max_output_tokens: Option, + /// 会不会推理。空 = 用价目表的 + #[serde(default, skip_serializing_if = "Option::is_none")] + pub reasoning: Option, + /// 收不收图。空 = 用价目表的 + #[serde(default, skip_serializing_if = "Option::is_none")] + pub image_input: Option, /// 你基于哪一版。**对不上就是 409** #[serde(default, skip_serializing_if = "Option::is_none")] pub base_version: Option, diff --git a/crates/tw-config/src/edit.rs b/crates/tw-config/src/edit.rs index 22bf0508..ee9f944a 100644 --- a/crates/tw-config/src/edit.rs +++ b/crates/tw-config/src/edit.rs @@ -631,7 +631,7 @@ pub fn remove_alias(text: &str, name: &str) -> Result { /// - 那一家写成了行内(`- { name: a, … }`):整项换成块式,位置不变(和 [`upsert`] 一样); /// `model_specs` 写成了行内:整张换掉 /// -/// 两项都空的 `spec` 交过来是调用方的错 —— 那是「删掉」,传 `None`。 +/// 全空的 `spec` 交过来是调用方的错 —— 那是「删掉」,传 `None`。 pub fn set_model_spec( text: &str, provider: &str, @@ -1314,6 +1314,7 @@ routes: [] crate::ModelSpec { context_window, max_output_tokens, + ..Default::default() } } diff --git a/crates/tw-config/src/lib.rs b/crates/tw-config/src/lib.rs index 2b898f9f..60fa46b5 100644 --- a/crates/tw-config/src/lib.rs +++ b/crates/tw-config/src/lib.rs @@ -37,7 +37,7 @@ pub use aliases::{Alias, Aliases}; pub use credential::{CredentialError, Header, Headers, Secret, SecretResolveError, auth_header}; pub use init::{generate_control_key, generate_initial, generate_key}; pub use limits::{KeyLimit, LimitMeasure, LimitPer}; -pub use model_specs::{ModelLimits, ModelSpec, Sourced, SpecSource}; +pub use model_specs::{ModelSpec, ResolvedSpec, Sourced, SpecSource}; pub use plugins::Plugin; pub use proxy::{DIRECT, OnProxyFail, Proxy, ProxyKind, SYSTEM}; pub use validate::ValidationError; diff --git a/crates/tw-config/src/model_specs.rs b/crates/tw-config/src/model_specs.rs index 6ec9335c..1a0a2e19 100644 --- a/crates/tw-config/src/model_specs.rs +++ b/crates/tw-config/src/model_specs.rs @@ -1,25 +1,25 @@ -//! 手写的模型规格:某家上游的某个模型的上下文窗口和输出上限。 +//! 手写的模型规格:某家上游的某个模型的上下文窗口、输出上限、会不会推理、收不收图。 //! //! ```yaml //! providers: //! - name: relay //! base_url: https://relay.example.com/v1 //! model_specs: -//! glm-5-air: { context_window: 128000, max_output_tokens: 16384 } +//! glm-5-air: { context_window: 128000, max_output_tokens: 16384, reasoning: true } //! ``` //! -//! 价目表里没有的模型(中转站自己的名字)说不出上下文窗口,价目表写错的也有。手写的 +//! 价目表里没有的模型(中转站自己的名字)说不出这几项,价目表写错的也有。手写的 //! **只管这一家的这一个模型**(名字要完全相等,没有通配),写了就优先于价目表。 //! -//! **哪个数优先只在 [`resolve`] 里定一次。**列模型(`/v1/models` 的各种格式)、上游页的 -//! 模型清单、别名、管线里「输入超出了上下文」、转换到 Anthropic 时补的输出上限,全都 -//! 经它取数 —— 哪一处另起炉灶,哪一处就会忘了手写的那个数。 +//! **哪个值优先只在 [`resolve`] 里定一次。**列模型(`/v1/models` 的各种格式)、上游页的 +//! 模型清单、别名、管线里「输入超出了上下文」、转换到 Anthropic 时补的输出上限、测速 +//! 给推理留的额度,全都经它取值 —— 哪一处另起炉灶,哪一处就会忘了手写的那个值。 use serde::{Deserialize, Serialize}; use crate::{Config, Provider}; -/// `providers[].model_specs` 的一项。两项都可选,但至少写一项(校验管)。 +/// `providers[].model_specs` 的一项。每项都可选,但至少写一项(校验管)。 #[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)] #[serde(deny_unknown_fields)] pub struct ModelSpec { @@ -29,13 +29,19 @@ pub struct ModelSpec { /// 一次最多输出多少 token #[serde(default, skip_serializing_if = "Option::is_none")] pub max_output_tokens: Option, + /// 会不会推理 + #[serde(default, skip_serializing_if = "Option::is_none")] + pub reasoning: Option, + /// 收不收图 + #[serde(default, skip_serializing_if = "Option::is_none")] + pub image_input: Option, } impl ModelSpec { - /// 两项都没写。写进配置里的不许这样([`crate::ValidationError::ModelSpecEmpty`]); - /// 界面交过来两项都空,是要删掉这一项 + /// 一项都没写。写进配置里的不许这样([`crate::ValidationError::ModelSpecEmpty`]); + /// 界面交过来全空,是要删掉这一项 pub fn is_empty(&self) -> bool { - self.context_window.is_none() && self.max_output_tokens.is_none() + *self == Self::default() } } @@ -48,27 +54,38 @@ pub enum SpecSource { Manual, } -/// 一个 token 数和它的来源。 +/// 一个值和它的来源。 #[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub struct Sourced { - pub tokens: u64, +pub struct Sourced { + pub value: T, pub source: SpecSource, } -/// 一个模型的上下文窗口和输出上限。**不知道就是 `None`**:编出来的数客户端会照着截断。 +/// 一个模型的规格,每一项连同来源。**不知道就是 `None`**:编出来的数客户端会照着截断, +/// 编出来的「不会推理」客户端会照着不让选推理档。 #[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] -pub struct ModelLimits { - pub context_window: Option, - pub max_output_tokens: Option, +pub struct ResolvedSpec { + pub context_window: Option>, + pub max_output_tokens: Option>, + pub reasoning: Option>, + pub image_input: Option>, } -impl ModelLimits { +impl ResolvedSpec { pub fn context_window(&self) -> Option { - self.context_window.map(|s| s.tokens) + self.context_window.map(|s| s.value) } pub fn max_output_tokens(&self) -> Option { - self.max_output_tokens.map(|s| s.tokens) + self.max_output_tokens.map(|s| s.value) + } + + pub fn reasoning(&self) -> Option { + self.reasoning.map(|s| s.value) + } + + pub fn image_input(&self) -> Option { + self.image_input.map(|s| s.value) } /// 不知道哪家上游服务它时:只看默认价目表。 @@ -78,21 +95,21 @@ impl ModelLimits { } impl Provider { - /// 这一家的这个模型的上下文窗口和输出上限:手写的优先,没写的取它选的价目表。 - pub fn model_limits(&self, book: &tw_pricing::PriceBook, model: &str) -> ModelLimits { + /// 这一家的这个模型的规格:手写的优先,没写的取它选的价目表。 + pub fn model_spec(&self, book: &tw_pricing::PriceBook, model: &str) -> ResolvedSpec { resolve(book, Some(&self.name), self.model_specs.get(model), model) } } impl Config { - /// 同 [`Provider::model_limits`],按上游的名字找。配置里没有这一家(刚删掉)时只看 + /// 同 [`Provider::model_spec`],按上游的名字找。配置里没有这一家(刚删掉)时只看 /// 价目表。 - pub fn model_limits( + pub fn model_spec( &self, book: &tw_pricing::PriceBook, provider: &str, model: &str, - ) -> ModelLimits { + ) -> ResolvedSpec { let spec = self .providers .iter() @@ -102,41 +119,51 @@ impl Config { } } -/// **唯一定先后的地方**:一项一项看,手写了就用手写的,没写的那一项取价目表。两项各管 +/// **唯一定先后的地方**:一项一项看,手写了就用手写的,没写的那一项取价目表。各项各管 /// 各的 —— 只写了上下文窗口,输出上限照样来自价目表。 /// -/// 价目表按这一家选的那张查(`provider`),和计价同一个查法;上下文窗口和输出上限在 -/// 每张价目表里都取自默认价目表,所以给不给上游通常是同一个数。 +/// 价目表按这一家选的那张查(`provider`),和计价同一个查法;这几项在每张价目表里都 +/// 取自默认价目表(自定义价目表只改单价),所以给不给上游通常是同一个值。 fn resolve( book: &tw_pricing::PriceBook, provider: Option<&str>, spec: Option<&ModelSpec>, model: &str, -) -> ModelLimits { +) -> ResolvedSpec { let priced = match provider { Some(p) => book.resolve_for(p, model), None => book.resolve(None, model), }; let price = priced.as_ref().map(|r| &r.price); - let pick = |manual: Option, table: Option| match manual { - Some(n) => Some(Sourced { - tokens: n.into(), - source: SpecSource::Manual, - }), - None => table.map(|n| Sourced { - tokens: n, - source: SpecSource::PriceTable, - }), - }; - ModelLimits { + fn pick(manual: Option, table: Option) -> Option> { + match manual { + Some(value) => Some(Sourced { + value, + source: SpecSource::Manual, + }), + None => table.map(|value| Sourced { + value, + source: SpecSource::PriceTable, + }), + } + } + ResolvedSpec { context_window: pick( - spec.and_then(|s| s.context_window), + spec.and_then(|s| s.context_window).map(u64::from), price.and_then(|p| p.max_input_tokens), ), max_output_tokens: pick( - spec.and_then(|s| s.max_output_tokens), + spec.and_then(|s| s.max_output_tokens).map(u64::from), price.and_then(|p| p.max_output_tokens), ), + reasoning: pick( + spec.and_then(|s| s.reasoning), + price.and_then(|p| p.reasoning), + ), + image_input: pick( + spec.and_then(|s| s.image_input), + price.and_then(|p| p.image_input), + ), } } @@ -164,14 +191,15 @@ mod tests { "claude-sonnet-4-5", ModelSpec { context_window: Some(1_000_000), - max_output_tokens: None, + reasoning: Some(false), + ..Default::default() }, )]); - let l = p.model_limits(&b, "claude-sonnet-4-5"); + let l = p.model_spec(&b, "claude-sonnet-4-5"); assert_eq!( l.context_window, Some(Sourced { - tokens: 1_000_000, + value: 1_000_000, source: SpecSource::Manual }) ); @@ -179,13 +207,28 @@ mod tests { assert_eq!( l.max_output_tokens, Some(Sourced { - tokens: 64_000, + value: 64_000, + source: SpecSource::PriceTable + }) + ); + // 推理也是一项一项来:手写的「不会」盖过价目表的「会」,收不收图照样取价目表 + assert_eq!( + l.reasoning, + Some(Sourced { + value: false, + source: SpecSource::Manual + }) + ); + assert_eq!( + l.image_input, + Some(Sourced { + value: true, source: SpecSource::PriceTable }) ); // 别的模型、别的上游不受影响 assert_eq!( - p.model_limits(&b, "claude-haiku-4-5") + p.model_spec(&b, "claude-haiku-4-5") .context_window .map(|s| s.source), Some(SpecSource::PriceTable) @@ -195,12 +238,12 @@ mod tests { ..Default::default() }; assert_eq!( - cfg.model_limits(&b, "relay", "claude-sonnet-4-5") + cfg.model_spec(&b, "relay", "claude-sonnet-4-5") .context_window(), Some(1_000_000) ); assert_eq!( - ModelLimits::priced(&b, "claude-sonnet-4-5").context_window(), + ResolvedSpec::priced(&b, "claude-sonnet-4-5").context_window(), Some(200_000) ); } @@ -211,23 +254,27 @@ mod tests { let p = relay(&[( "中转自有模型", ModelSpec { - context_window: None, max_output_tokens: Some(8_000), + image_input: Some(true), + ..Default::default() }, )]); - let l = p.model_limits(&b, "中转自有模型"); + let l = p.model_spec(&b, "中转自有模型"); assert_eq!(l.context_window, None); assert_eq!(l.max_output_tokens(), Some(8_000)); + assert_eq!(l.image_input(), Some(true)); + // 价目表不认识、也没手写的:不知道,不是「不会」 + assert_eq!(l.reasoning, None); // 名字要完全相等:带日期的另一个名字不算 assert_eq!( - p.model_limits(&b, "中转自有模型-2026"), - ModelLimits::default() + p.model_spec(&b, "中转自有模型-2026"), + ResolvedSpec::default() ); // 配置里没有这一家:只看价目表 let cfg = Config::default(); assert_eq!( - cfg.model_limits(&b, "relay", "中转自有模型"), - ModelLimits::default() + cfg.model_spec(&b, "relay", "中转自有模型"), + ResolvedSpec::default() ); } } diff --git a/crates/tw-config/src/validate.rs b/crates/tw-config/src/validate.rs index 4f877f25..3d74cf6d 100644 --- a/crates/tw-config/src/validate.rs +++ b/crates/tw-config/src/validate.rs @@ -358,9 +358,9 @@ impl ValidationError { is for one exact model id" ), ModelSpecEmpty { upstream, model } => msg!( - "config.model_spec_empty", upstream = upstream, model = model => - "the model spec `{model}` of upstream `{upstream}` sets neither context_window \ - nor max_output_tokens. Set at least one, or remove it" + "config.model_spec_nothing_set", upstream = upstream, model = model => + "the model spec `{model}` of upstream `{upstream}` sets none of context_window, \ + max_output_tokens, reasoning and image_input. Set at least one, or remove it" ), ModelSpecZero { upstream, @@ -933,10 +933,13 @@ mod tests { )) }; let ok = parse( - " glm-5-air: { context_window: 128000, max_output_tokens: 16384 }\n \"us.anthropic.claude-fable-5-v1:0\": { max_output_tokens: 32000 }\n", + " glm-5-air: { context_window: 128000, max_output_tokens: 16384 }\n \"us.anthropic.claude-fable-5-v1:0\": { max_output_tokens: 32000 }\n kimi-k3: { reasoning: true, image_input: false }\n", ) .unwrap(); let specs = &ok.providers[0].model_specs; + // 只写会不会推理、收不收图也算写了 + assert_eq!(specs["kimi-k3"].reasoning, Some(true)); + assert_eq!(specs["kimi-k3"].image_input, Some(false)); assert_eq!(specs["glm-5-air"].context_window, Some(128_000)); assert_eq!( specs["us.anthropic.claude-fable-5-v1:0"].max_output_tokens, @@ -951,7 +954,7 @@ mod tests { " glm-*: { context_window: 1000 }\n", "config.model_spec_wildcard", ), - (" glm-5-air: {}\n", "config.model_spec_empty"), + (" glm-5-air: {}\n", "config.model_spec_nothing_set"), ( " glm-5-air: { context_window: 0 }\n", "config.model_spec_zero", diff --git a/crates/tw-config/tests/manual/schema.rs b/crates/tw-config/tests/manual/schema.rs index c4a7010e..8d162cb6 100644 --- a/crates/tw-config/tests/manual/schema.rs +++ b/crates/tw-config/tests/manual/schema.rs @@ -629,8 +629,8 @@ pub fn sections() -> Vec
{ Kind::ObjMap(t("model id", "模型 ID"), "providers[].model_specs.*"), Def::Is("{}"), t( - "Context window and output limit of single models of this upstream, written by hand, by exact model id. They take precedence over the price table: for models it does not know, or gets wrong.", - "手写这家上游某些模型的上下文窗口和输出上限,按模型 ID 完全匹配。写了就优先于价目表,用于价目表里没有或写错的模型。", + "Context window, output limit, reasoning and image input of single models of this upstream, written by hand, by exact model id. They take precedence over the price table: for models it does not know, or gets wrong.", + "手写这家上游某些模型的上下文窗口、输出上限、会不会推理、收不收图,按模型 ID 完全匹配。写了就优先于价目表,用于价目表里没有或写错的模型。", ), ), row( @@ -655,7 +655,7 @@ pub fn sections() -> Vec
{ }, Section { path: "providers[].model_specs.*", - // 两项都可选,至少写一项由校验管 + // 每项都可选,至少写一项由校验管 ty: checked!(ModelSpec, "{}"), rows: vec![ row( @@ -676,6 +676,24 @@ pub fn sections() -> Vec
{ "一次回答最多输出多少 token。不写:取价目表的。", ), ), + row( + "reasoning", + Kind::Bool, + Def::Unset, + t( + "Whether the model reasons. The model list (`GET /v1/models`) carries it, so clients offer reasoning levels for it. Unset: the price table's; the list leaves it out when the price table does not say.", + "这个模型会不会推理。模型列表(`GET /v1/models`)带着它,客户端据此给出推理档位。不写:取价目表的;价目表也没写时,列表里不给这一项。", + ), + ), + row( + "image_input", + Kind::Bool, + Def::Unset, + t( + "Whether the model takes images as input. The model list carries it, as `input_modalities`. Unset: the price table's; the list leaves it out when the price table does not say.", + "这个模型收不收图。模型列表里以 `input_modalities` 给出。不写:取价目表的;价目表也没写时,列表里不给这一项。", + ), + ), ], }, Section { diff --git a/crates/tw-control/src/aliases.rs b/crates/tw-control/src/aliases.rs index b3a834ca..cd1823a2 100644 --- a/crates/tw-control/src/aliases.rs +++ b/crates/tw-control/src/aliases.rs @@ -70,7 +70,7 @@ async fn list(State(s): State) -> Json { shadows: shadows(&lists, a), // 和 `/v1/models` 列别名时同一个查法:那一家手写的优先 context_window: served_by.first().and_then(|first| { - cfg.model_limits(&book, &first.provider, &first.model) + cfg.model_spec(&book, &first.provider, &first.model) .context_window() }), served_by, diff --git a/crates/tw-control/src/lib.rs b/crates/tw-control/src/lib.rs index eb00f992..9ee2f0fe 100644 --- a/crates/tw-control/src/lib.rs +++ b/crates/tw-control/src/lib.rs @@ -884,15 +884,9 @@ async fn speed_quote( Ok(m) => (m, None), Err(skip) => (req.model.clone(), Some(skip)), }; - // 和记账同一个口径:按这家的计费方式和价目表、发给它的那个模型名。**协议 - // 也要给**:输出上限报多少由它决定(见 `tw_gateway::l3::max_output_tokens`) - let e = tw_gateway::l3::estimate( - &book, - &p.name, - &model, - p.billing, - p.effective_protocol(), - ); + // 和记账同一个口径:按这家的计费方式和价目表、发给它的那个模型名。输出上限 + // 报多少由这一家的协议和规格决定(见 `tw_gateway::l3::max_output_tokens`) + let e = tw_gateway::l3::estimate(&book, p, &model); (e, skip) }) .collect(); @@ -958,11 +952,7 @@ async fn speed_run( p, &headers, &model, - tw_gateway::l3::max_output_tokens( - &s.gateway.pricing.load(), - &model, - p.effective_protocol(), - ), + tw_gateway::l3::max_output_tokens(&s.gateway.pricing.load(), p, &model), ) .await; out.push(tw_api::SpeedResult { diff --git a/crates/tw-control/src/resources.rs b/crates/tw-control/src/resources.rs index 40432825..b3ed9477 100644 --- a/crates/tw-control/src/resources.rs +++ b/crates/tw-control/src/resources.rs @@ -143,7 +143,8 @@ fn chatgpt_login(cfg: &tw_config::Config, name: &str, token_endpoint: &str) -> O (ours && !shared).then(|| o.refresh.clone()) } -/// 手写一家上游的一个模型的上下文窗口、输出上限;两项都空就删掉那一项,回到价目表。 +/// 手写一家上游的一个模型的上下文窗口、输出上限、会不会推理、收不收图;四项都空就删掉 +/// 那一项,回到价目表。 /// /// **只动这一家 `model_specs` 里的这一项**([`edit::set_model_spec`]),不走编辑上游那条 /// 路:那条路按读进来的结构把整项写回去,用户写成和默认值一样的字段会被顺手删掉。 @@ -157,6 +158,8 @@ async fn set_model_spec( let spec = tw_config::ModelSpec { context_window: req.context_window, max_output_tokens: req.max_output_tokens, + reasoning: req.reasoning, + image_input: req.image_input, }; let spec = (!spec.is_empty()).then_some(spec); let version = s @@ -343,14 +346,18 @@ fn models_view( .into_iter() .map(|id| { let r = book.resolve_for(&p.name, &id); - // 上下文窗口、输出上限和 `/v1/models` 给客户端的是同一个查法 - let limits = p.model_limits(&book, &id); + // 规格和 `/v1/models` 给客户端的是同一个查法 + let spec = p.model_spec(&book, &id); tw_api::ModelRow { enabled: p.uses_model(&id), - context_window: limits.context_window(), - context_window_source: limits.context_window.map(|s| s.source.into()), - max_output_tokens: limits.max_output_tokens(), - max_output_tokens_source: limits.max_output_tokens.map(|s| s.source.into()), + context_window: spec.context_window(), + context_window_source: spec.context_window.map(|s| s.source.into()), + max_output_tokens: spec.max_output_tokens(), + max_output_tokens_source: spec.max_output_tokens.map(|s| s.source.into()), + reasoning: spec.reasoning(), + reasoning_source: spec.reasoning.map(|s| s.source.into()), + image_input: spec.image_input(), + image_input_source: spec.image_input.map(|s| s.source.into()), price: r.as_ref().map(|r| { crate::pricing::price_fields(&tw_pricing::PerMillion::of(&r.price)) }), diff --git a/crates/tw-control/tests/model_specs.rs b/crates/tw-control/tests/model_specs.rs index f55f9b6e..520f8905 100644 --- a/crates/tw-control/tests/model_specs.rs +++ b/crates/tw-control/tests/model_specs.rs @@ -121,8 +121,8 @@ async fn each_row_says_whether_its_numbers_were_written_by_hand() { let b = bed(&config( up, " model_specs: - claude-sonnet-4-5: { max_output_tokens: 8000 } - 中转自有模型: { context_window: 32000 } + claude-sonnet-4-5: { max_output_tokens: 8000, reasoning: false } + 中转自有模型: { context_window: 32000, image_input: true } ", "", )); @@ -133,6 +133,10 @@ async fn each_row_says_whether_its_numbers_were_written_by_hand() { assert_eq!(sonnet["context_window_source"], "price_table"); assert_eq!(sonnet["max_output_tokens"], 8_000); assert_eq!(sonnet["max_output_tokens_source"], "manual"); + assert_eq!(sonnet["reasoning"], false); + assert_eq!(sonnet["reasoning_source"], "manual"); + assert_eq!(sonnet["image_input"], true); + assert_eq!(sonnet["image_input_source"], "price_table"); let own = row(&rows, "中转自有模型"); assert_eq!(own["context_window"], 32_000, "{own}"); @@ -140,6 +144,10 @@ async fn each_row_says_whether_its_numbers_were_written_by_hand() { // 不知道就整个不出现,来源也没有 assert!(own.get("max_output_tokens").is_none(), "{own}"); assert!(own.get("max_output_tokens_source").is_none(), "{own}"); + assert!(own.get("reasoning").is_none(), "{own}"); + assert!(own.get("reasoning_source").is_none(), "{own}"); + assert_eq!(own["image_input"], true); + assert_eq!(own["image_input_source"], "manual"); let haiku = row(&rows, "claude-haiku-4-5"); assert_eq!(haiku["context_window_source"], "price_table", "{haiku}"); @@ -174,7 +182,7 @@ async fn a_spec_is_set_changed_and_cleared_through_the_endpoint() { cfg.providers[0].model_specs["中转自有模型"], tw_config::ModelSpec { context_window: Some(64_000), - max_output_tokens: None, + ..Default::default() }, "模型 ID 去掉首尾空白:{}", b.file() @@ -196,7 +204,21 @@ async fn a_spec_is_set_changed_and_cleared_through_the_endpoint() { assert_eq!(own["max_output_tokens"], 4_096); assert_eq!(own["max_output_tokens_source"], "manual"); - // 两项都空:删掉,回到价目表(它不认识这个模型),文件回到原样 + // 只写会不会推理也是一项规格 + let (st, v) = call( + &b.app, + "PUT", + "/provider-model-spec", + json!({ "provider": "relay", "model": "中转自有模型", "reasoning": true }), + ) + .await; + assert_eq!(st, StatusCode::OK, "{v}"); + let own = row(&rows(&b).await, "中转自有模型").clone(); + assert_eq!(own["reasoning"], true); + assert_eq!(own["reasoning_source"], "manual"); + assert!(own.get("context_window").is_none(), "整项换掉:{own}"); + + // 四项都空:删掉,回到价目表(它不认识这个模型),文件回到原样 let (st, v) = call( &b.app, "PUT", diff --git a/crates/tw-gateway/src/l3.rs b/crates/tw-gateway/src/l3.rs index 1e5f1e3a..c87458b4 100644 --- a/crates/tw-gateway/src/l3.rs +++ b/crates/tw-gateway/src/l3.rs @@ -67,16 +67,18 @@ pub fn probe_input_tokens() -> u64 { /// 三种情况: /// - ChatGPT 账号:Codex 后端不认 `max_output_tokens`,带上去是 400(见 /// [`crate::chatgpt`])。上限报不出来,费用也就报不出来 -/// - 会推理的模型:留出推理的量,见 [`REASONING_MAX_TOKENS`] -/// - 其余:一句话的长度就够 +/// - 会推理的模型:留出推理的量,见 [`REASONING_MAX_TOKENS`]。会不会推理按这一家的 +/// 规格算([`Provider::model_spec`]):手写的优先,再看价目表 +/// - 其余(包括不知道会不会推理的):一句话的长度就够 /// /// Anthropic 那边不看模型会不会推理:**要推理得在请求里写**,而探测请求不写, /// 所以它不会花额度在推理上。Bedrock 上的 Claude 也一样。 pub fn max_output_tokens( book: &tw_pricing::PriceBook, + provider: &Provider, model: &str, - protocol: Option, ) -> Option { + let protocol = provider.effective_protocol(); if protocol == Some(Protocol::Chatgpt) { return None; } @@ -85,7 +87,7 @@ pub fn max_output_tokens( Dialect::Bedrock => model.to_ascii_lowercase().contains("anthropic.claude"), _ => false, }; - let reasons = !asks_to_think && book.table().get(model).is_some_and(|p| p.reasoning); + let reasons = !asks_to_think && provider.model_spec(book, model).reasoning() == Some(true); Some(if reasons { REASONING_MAX_TOKENS } else { @@ -381,29 +383,24 @@ pub async fn run( } } -/// 算一次测速要花多少。 -pub fn estimate( - book: &tw_pricing::PriceBook, - provider: &str, - model: &str, - billing: Billing, - protocol: Option, -) -> Estimate { +/// 算一次测速要花多少:按这一家的计费方式和价目表、发给它的那个模型名,和记账同一个口径。 +pub fn estimate(book: &tw_pricing::PriceBook, provider: &Provider, model: &str) -> Estimate { + let billing = provider.billing; let input = probe_input_tokens(); - let cap = max_output_tokens(book, model, protocol); + let cap = max_output_tokens(book, provider, model); let usage = tw_pricing::Usage { input, output: cap.unwrap_or_default(), ..Default::default() }; - let mut quote = crate::quote::quote(book, provider, model, &usage, billing); + let mut quote = crate::quote::quote(book, &provider.name, model, &usage, billing); // 上限报不出来时,按量计费这次要花多少就是**不知道**,不是 0 —— 回答有多长 // 由模型决定。不计费的那档照样是 0 if cap.is_none() && billing == Billing::PerToken { quote.cost_micros = None; } Estimate { - provider: provider.to_string(), + provider: provider.name.clone(), model: model.to_string(), input_tokens: input, max_output_tokens: cap, @@ -523,7 +520,7 @@ mod tests { let r = probe_request( &p, "gpt-5.5", - max_output_tokens(&prices(), "gpt-5.5", p.effective_protocol()), + max_output_tokens(&prices(), &p, "gpt-5.5"), &[], ); assert_eq!(r.path, "/responses"); @@ -543,7 +540,8 @@ mod tests { fn a_model_that_reasons_gets_room_to_reason() { // 推理 token 也算输出:上限按一句话给,模型一个可见的 token 都不吐 let book = prices(); - let cap = |model, protocol| max_output_tokens(&book, model, protocol); + let cap = + |model, protocol| max_output_tokens(&book, &provider(protocol, "https://x"), model); assert_eq!( cap("gpt-5", Some(Protocol::OpenaiResponses)), Some(REASONING_MAX_TOKENS) @@ -559,6 +557,22 @@ mod tests { cap("中转站自己起的名字", Some(Protocol::OpenaiChat)), Some(MAX_TOKENS) ); + // 手写成会推理的,照样留出推理的量 + let relay = Provider { + model_specs: [( + "中转站自己起的名字".to_string(), + tw_config::ModelSpec { + reasoning: Some(true), + ..Default::default() + }, + )] + .into(), + ..provider(Some(Protocol::OpenaiChat), "https://x") + }; + assert_eq!( + max_output_tokens(&book, &relay, "中转站自己起的名字"), + Some(REASONING_MAX_TOKENS) + ); // Codex 后端不接受上限 assert_eq!(cap("gpt-5.5", Some(Protocol::Chatgpt)), None); } @@ -569,20 +583,24 @@ mod tests { // 对话框里是两种可信度。 let e = estimate( &prices(), - "官方", + &Provider { + name: "官方".into(), + billing: Billing::PerToken, + ..provider(Some(Protocol::Anthropic), "https://x") + }, "claude-sonnet-4-5", - Billing::PerToken, - Some(Protocol::Anthropic), ); assert_eq!(e.input_tokens, probe_input_tokens()); assert_eq!(e.max_output_tokens, Some(MAX_TOKENS)); assert!(e.quote.cost_micros.is_some()); let e = estimate( &prices(), - "本地", + &Provider { + name: "本地".into(), + billing: Billing::Free, + ..provider(Some(Protocol::Anthropic), "https://x") + }, "claude-sonnet-4-5", - Billing::Free, - Some(Protocol::Anthropic), ); assert_eq!(e.quote.cost_micros, Some(0)); assert_eq!(e.input_tokens, probe_input_tokens()); @@ -593,20 +611,24 @@ mod tests { // 回答有多长由模型决定:报一个看起来确定的数字,那是编的 let e = estimate( &prices(), - "chatgpt", + &Provider { + name: "chatgpt".into(), + billing: Billing::PerToken, + ..provider(Some(Protocol::Chatgpt), "https://x") + }, "gpt-5.5", - Billing::PerToken, - Some(Protocol::Chatgpt), ); assert_eq!(e.max_output_tokens, None); assert_eq!(e.quote.cost_micros, None); // 不计费的那档照样是 0 let free = estimate( &prices(), - "chatgpt", + &Provider { + name: "chatgpt".into(), + billing: Billing::Free, + ..provider(Some(Protocol::Chatgpt), "https://x") + }, "gpt-5.5", - Billing::Free, - Some(Protocol::Chatgpt), ); assert_eq!(free.quote.cost_micros, Some(0)); } diff --git a/crates/tw-gateway/src/server/listing.rs b/crates/tw-gateway/src/server/listing.rs index c2f146d9..a7114708 100644 --- a/crates/tw-gateway/src/server/listing.rs +++ b/crates/tw-gateway/src/server/listing.rs @@ -64,21 +64,25 @@ impl ListingShape { /// 列表里一个模型带给客户端的元数据。 /// /// 这一家手写的(`model_specs`)优先,没写的来自价目表(见 [`tw_config::model_specs`]), -/// 和上游页模型一格的「上下文」(`ModelRow.context_window`)是同一个数。 +/// 和上游页模型一格的「上下文」(`ModelRow.context_window`)是同一个值。 /// **查不到就是 `None`,对应的字段整个不出现** —— 客户端读不到会用自己的默认值, -/// 一个编出来的数它却会照着截断对话。 +/// 一个编出来的数它却会照着截断对话,编出来的「不会推理」它会照着不让选推理档。 #[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] pub(crate) struct ModelMeta { /// 一次最多输入多少 token,也就是上下文窗口 pub max_input_tokens: Option, - /// 一次最多输出多少 token。只有 Gemini 的模型对象有这一项 + /// 一次最多输出多少 token pub max_output_tokens: Option, + /// 会不会推理 + pub reasoning: Option, + /// 收不收图 + pub image_input: Option, } /// 查一个模型的元数据。 /// /// `provider`:知道这个名称发给哪家上游时给上,先看这一家手写的,再按它选的价目表查, -/// 和上游页那一格是同一个查法([`tw_config::Config::model_limits`]);不给就只查默认 +/// 和上游页那一格是同一个查法([`tw_config::Config::model_spec`]);不给就只查默认 /// 价目表。自定义价目表只改单价,上下文窗口照样取自默认价目表。 pub(crate) fn model_meta( cfg: &tw_config::Config, @@ -86,13 +90,15 @@ pub(crate) fn model_meta( provider: Option<&str>, model: &str, ) -> ModelMeta { - let limits = match provider { - Some(p) => cfg.model_limits(book, p, model), - None => tw_config::ModelLimits::priced(book, model), + let spec = match provider { + Some(p) => cfg.model_spec(book, p, model), + None => tw_config::ResolvedSpec::priced(book, model), }; ModelMeta { - max_input_tokens: limits.context_window(), - max_output_tokens: limits.max_output_tokens(), + max_input_tokens: spec.context_window(), + max_output_tokens: spec.max_output_tokens(), + reasoning: spec.reasoning(), + image_input: spec.image_input(), } } @@ -185,6 +191,10 @@ const RELEASED_AT: i64 = 0; /// /// 知道上下文窗口时,同一个数写成三个字段:各家客户端读的不是同一个名字 —— /// Grok Build 读 `context_window`,oh-my-pi 读 `context_length`,Hermes 三个依次试。 +/// +/// 另外三项各写一个字段,知道才写:输出上限 `max_output_tokens`、会不会推理 +/// `supports_reasoning`、收不收图 `input_modalities`(`["text", "image"]` 或 `["text"]`, +/// 和 OpenRouter 的写法一样)。桌面端接管客户端时照着它们写进客户端的配置。 fn openai_model(id: &str, meta: &ModelMeta) -> serde_json::Value { let mut m = serde_json::json!({ "id": id, "object": "model", "created": RELEASED_AT }); if let Some(n) = meta.max_input_tokens { @@ -192,6 +202,19 @@ fn openai_model(id: &str, meta: &ModelMeta) -> serde_json::Value { m["context_length"] = n.into(); m["max_input_tokens"] = n.into(); } + if let Some(n) = meta.max_output_tokens { + m["max_output_tokens"] = n.into(); + } + if let Some(r) = meta.reasoning { + m["supports_reasoning"] = r.into(); + } + if let Some(image) = meta.image_input { + m["input_modalities"] = if image { + serde_json::json!(["text", "image"]) + } else { + serde_json::json!(["text"]) + }; + } m } @@ -225,7 +248,8 @@ fn anthropic_model(id: &str, meta: &ModelMeta) -> serde_json::Value { m } -/// Gemini 格式的一个模型对象。上下文窗口和输出上限用 Gemini 自己的字段名。 +/// Gemini 格式的一个模型对象。上下文窗口、输出上限、会不会推理用 Gemini 自己的字段名 +/// (`thinking`);收不收图 Gemini 的模型对象里没有这一项。 fn gemini_model(id: &str, meta: &ModelMeta) -> serde_json::Value { let mut m = serde_json::json!({ "name": format!("models/{id}") }); if let Some(n) = meta.max_input_tokens { @@ -234,6 +258,9 @@ fn gemini_model(id: &str, meta: &ModelMeta) -> serde_json::Value { if let Some(n) = meta.max_output_tokens { m["outputTokenLimit"] = n.into(); } + if let Some(r) = meta.reasoning { + m["thinking"] = r.into(); + } m } @@ -400,6 +427,8 @@ mod tests { const KNOWN: ModelMeta = ModelMeta { max_input_tokens: Some(200_000), max_output_tokens: Some(64_000), + reasoning: Some(true), + image_input: Some(true), }; #[test] @@ -409,8 +438,18 @@ mod tests { assert_eq!(m["context_window"], 200_000); assert_eq!(m["context_length"], 200_000); assert_eq!(m["max_input_tokens"], 200_000); - // 输出上限 OpenAI 的模型对象里没有人读 - assert!(m.get("max_output_tokens").is_none(), "{m}"); + assert_eq!(m["max_output_tokens"], 64_000); + assert_eq!(m["supports_reasoning"], true); + assert_eq!(m["input_modalities"], serde_json::json!(["text", "image"])); + // 「不会」和「不收图」照样说出来:那和不知道是两回事 + let plain = ModelMeta { + reasoning: Some(false), + image_input: Some(false), + ..KNOWN + }; + let m = openai_model("deepseek-chat", &plain); + assert_eq!(m["supports_reasoning"], false); + assert_eq!(m["input_modalities"], serde_json::json!(["text"])); } #[test] @@ -443,6 +482,7 @@ mod tests { assert_eq!(m["name"], "models/gemini-2.5-pro"); assert_eq!(m["inputTokenLimit"], 200_000); assert_eq!(m["outputTokenLimit"], 64_000); + assert_eq!(m["thinking"], true); } #[test] @@ -453,8 +493,12 @@ mod tests { "context_length", "max_input_tokens", "supports_1m", + "max_output_tokens", + "supports_reasoning", + "input_modalities", "inputTokenLimit", "outputTokenLimit", + "thinking", ]; for m in [ openai_model("my-model", &unknown), @@ -502,7 +546,8 @@ mod tests { "claude-sonnet-4-5".to_string(), tw_config::ModelSpec { context_window: Some(1_000_000), - max_output_tokens: None, + reasoning: Some(false), + ..Default::default() }, )] .into(), @@ -513,6 +558,8 @@ mod tests { let m = model_meta(&cfg, &book, Some("up"), "claude-sonnet-4-5"); assert_eq!(m.max_input_tokens, Some(1_000_000)); assert_eq!(m.max_output_tokens, Some(64_000), "没写的照样取价目表"); + assert_eq!(m.reasoning, Some(false), "手写的「不会」盖过价目表"); + assert_eq!(m.image_input, Some(true)); // 不知道是哪一家时没有手写的可看 assert_eq!( model_meta(&cfg, &book, None, "claude-sonnet-4-5").max_input_tokens, diff --git a/crates/tw-gateway/src/server/pipeline.rs b/crates/tw-gateway/src/server/pipeline.rs index 87b5e8ec..243dccc4 100644 --- a/crates/tw-gateway/src/server/pipeline.rs +++ b/crates/tw-gateway/src/server/pipeline.rs @@ -401,7 +401,7 @@ fn outgrown( .iter() .filter_map(|s| { rt.config - .model_limits(&book, &s.provider, s.model.as_deref().ok()?) + .model_spec(&book, &s.provider, s.model.as_deref().ok()?) .context_window() }) .min() diff --git a/crates/tw-gateway/src/translate.rs b/crates/tw-gateway/src/translate.rs index f3c48a51..8a74ee0f 100644 --- a/crates/tw-gateway/src/translate.rs +++ b/crates/tw-gateway/src/translate.rs @@ -74,7 +74,7 @@ pub fn default_max_tokens( model: &str, ) -> u64 { provider - .model_limits(book, model) + .model_spec(book, model) .max_output_tokens() .unwrap_or_else(|| tw_dialect::official::fallback_max_output_tokens(model)) } @@ -147,8 +147,8 @@ mod tests { ( m.to_string(), tw_config::ModelSpec { - context_window: None, max_output_tokens: Some(n), + ..Default::default() }, ) }) diff --git a/crates/tw-gateway/tests/chatgpt.rs b/crates/tw-gateway/tests/chatgpt.rs index ed12ddb7..202ed41f 100644 --- a/crates/tw-gateway/tests/chatgpt.rs +++ b/crates/tw-gateway/tests/chatgpt.rs @@ -761,11 +761,7 @@ async fn an_inference_test_reaches_the_codex_backend_and_times_the_first_token() .unwrap(); // 上限:Codex 后端不接受,所以报价里是空的,发出去的请求里也不该有 - let cap = tw_gateway::l3::max_output_tokens( - &state.pricing.load(), - "gpt-5.5", - Some(Protocol::Chatgpt), - ); + let cap = tw_gateway::l3::max_output_tokens(&state.pricing.load(), &p, "gpt-5.5"); assert_eq!(cap, None); let http = state.client_for(&p.name); diff --git a/crates/tw-pricing/src/book.rs b/crates/tw-pricing/src/book.rs index 12bd7eb0..fda78907 100644 --- a/crates/tw-pricing/src/book.rs +++ b/crates/tw-pricing/src/book.rs @@ -241,6 +241,7 @@ impl ModelPrice { max_input_tokens: self.max_input_tokens, max_output_tokens: self.max_output_tokens, reasoning: self.reasoning, + image_input: self.image_input, } } diff --git a/crates/tw-pricing/src/lib.rs b/crates/tw-pricing/src/lib.rs index 7959f0d3..8bf4005a 100644 --- a/crates/tw-pricing/src/lib.rs +++ b/crates/tw-pricing/src/lib.rs @@ -55,9 +55,15 @@ pub struct ModelPrice { pub max_output_tokens: Option, /// 这个模型自己会推理。**推理 token 也算输出** —— 测速的探测请求要给它留出 /// 量(见 `tw_gateway::l3`),按一句话的长度设上限的话,它会把额度全用在 - /// 推理上,一个可见的 token 都不吐 - #[serde(default, skip_serializing_if = "std::ops::Not::not")] - pub reasoning: bool, + /// 推理上,一个可见的 token 都不吐。 + /// + /// **数据集没写就是 `None`**,不是「不会」:接管客户端时不知道的不写,客户端用它 + /// 自己的默认值;写成「不会」,它就不让用户选推理档了 + #[serde(default, skip_serializing_if = "Option::is_none")] + pub reasoning: Option, + /// 收不收图(数据集的 `supports_vision`)。没写就是 `None`,理由同上 + #[serde(default, skip_serializing_if = "Option::is_none")] + pub image_input: Option, } /// 长上下文那一档:一次请求的输入超过 `above` 个 token,**整个请求**改按这一档的 diff --git a/crates/tw-pricing/src/sheet.rs b/crates/tw-pricing/src/sheet.rs index 8e349728..412b3258 100644 --- a/crates/tw-pricing/src/sheet.rs +++ b/crates/tw-pricing/src/sheet.rs @@ -279,8 +279,9 @@ impl PerMillion { }), max_input_tokens: base.and_then(|b| b.max_input_tokens), max_output_tokens: base.and_then(|b| b.max_output_tokens), - // 覆盖价只改单价:模型会不会推理不是价格的一部分 - reasoning: base.is_some_and(|b| b.reasoning), + // 覆盖价只改单价:模型会不会推理、收不收图不是价格的一部分 + reasoning: base.and_then(|b| b.reasoning), + image_input: base.and_then(|b| b.image_input), } } diff --git a/crates/tw-pricing/src/table.rs b/crates/tw-pricing/src/table.rs index cb8f2e98..15be9eff 100644 --- a/crates/tw-pricing/src/table.rs +++ b/crates/tw-pricing/src/table.rs @@ -176,10 +176,8 @@ fn price_from(v: &serde_json::Value) -> Option { long: long_tier(v, output), max_input_tokens: v.get("max_input_tokens").and_then(|x| x.as_u64()), max_output_tokens: v.get("max_output_tokens").and_then(|x| x.as_u64()), - reasoning: v - .get("supports_reasoning") - .and_then(|x| x.as_bool()) - .unwrap_or(false), + reasoning: v.get("supports_reasoning").and_then(|x| x.as_bool()), + image_input: v.get("supports_vision").and_then(|x| x.as_bool()), }) } @@ -239,8 +237,15 @@ mod tests { .is_some_and(|n| n >= 32_000) ); // 会不会推理也要读进来:测速的探测请求按它决定留多少输出额度 - assert!(t.get("gpt-5").is_some_and(|p| p.reasoning)); - assert!(t.get("gpt-4o").is_some_and(|p| !p.reasoning)); + assert_eq!(t.get("gpt-5").and_then(|p| p.reasoning), Some(true)); + // 数据集没写的是「不知道」,不是「不会」 + assert_eq!(t.get("gpt-4o").and_then(|p| p.reasoning), None); + // 收不收图也是,写了「不收」的照样读进来 + assert_eq!(t.get("gpt-4o").and_then(|p| p.image_input), Some(true)); + assert_eq!( + t.get("deepseek-v4-flash").and_then(|p| p.image_input), + Some(false) + ); } #[test] diff --git a/docs/config.md b/docs/config.md index a84134c1..01c93867 100644 --- a/docs/config.md +++ b/docs/config.md @@ -402,7 +402,7 @@ Upstreams: the APIs requests are forwarded to. | `models_only` | list of strings | — | Use only these of the upstream's models, as ids or globs. Others are not listed and are not routed here. Unset: all of them. Empty is refused; use `disabled`. | | `billing` | `per-token` \| `free` | `per-token` | `per-token`: cost is usage times the price in the upstream's price sheet, subscription accounts included. `free`: cost is recorded as 0. | | `pricing` | string | — | Name of a price sheet under `pricing.sheets`. Unset: the default price table. | -| `model_specs` | map of model id → [`providers[].model_specs.*`](#cfg-providers-model_specs) | `{}` | Context window and output limit of single models of this upstream, written by hand, by exact model id. They take precedence over the price table: for models it does not know, or gets wrong. | +| `model_specs` | map of model id → [`providers[].model_specs.*`](#cfg-providers-model_specs) | `{}` | Context window, output limit, reasoning and image input of single models of this upstream, written by hand, by exact model id. They take precedence over the price table: for models it does not know, or gets wrong. | | `max_concurrent` | integer | — | Most requests sent to this upstream at the same time, from 1 to 1000. When it is full, a conversation that stays on it waits for a free slot and other requests go to the next upstream; see `failover.slot_wait_secs`. Unset: no limit. | | `disabled` | bool | `false` | Take the upstream out of routing and out of the model list, and keep its configuration. | @@ -565,6 +565,8 @@ offer the same model, `/v1/models` describes it by the first of them in |---|---|---|---| | `context_window` | integer | — | Context window: the most tokens a request can take in. Unset: the price table's. | | `max_output_tokens` | integer | — | The most tokens an answer can have. Unset: the price table's. | +| `reasoning` | bool | — | Whether the model reasons. The model list (`GET /v1/models`) carries it, so clients offer reasoning levels for it. Unset: the price table's; the list leaves it out when the price table does not say. | +| `image_input` | bool | — | Whether the model takes images as input. The model list carries it, as `input_modalities`. Unset: the price table's; the list leaves it out when the price table does not say. | ```yaml diff --git a/docs/config.zh-CN.md b/docs/config.zh-CN.md index 49507c7c..8279e362 100644 --- a/docs/config.zh-CN.md +++ b/docs/config.zh-CN.md @@ -300,7 +300,7 @@ clients: | `models_only` | 字符串列表 | — | 只使用这家的这些模型,写 ID 或通配。范围外的模型不出现在模型列表里,也不会路由到这家。不写:全部。写空列表会被拒绝,暂停使用请用 `disabled`。 | | `billing` | `per-token` \| `free` | `per-token` | `per-token`:费用为用量乘以所选价目表中的单价,订阅账号同样如此。`free`:费用记为 0。 | | `pricing` | 字符串 | — | `pricing.sheets` 中某张价目表的名字。不写:默认价目表。 | -| `model_specs` | 映射: 模型 ID → [`providers[].model_specs.*`](#cfg-providers-model_specs) | `{}` | 手写这家上游某些模型的上下文窗口和输出上限,按模型 ID 完全匹配。写了就优先于价目表,用于价目表里没有或写错的模型。 | +| `model_specs` | 映射: 模型 ID → [`providers[].model_specs.*`](#cfg-providers-model_specs) | `{}` | 手写这家上游某些模型的上下文窗口、输出上限、会不会推理、收不收图,按模型 ID 完全匹配。写了就优先于价目表,用于价目表里没有或写错的模型。 | | `max_concurrent` | 整数 | — | 同时发给这家的请求最多几个,取值 1 到 1000。满了的时候,留在这家的对话等空位,别的请求换下一家;等多久见 `failover.slot_wait_secs`。不写:不限。 | | `disabled` | 布尔 | `false` | 不参与路由,模型也不出现在模型列表里;配置原样保留。 | @@ -412,6 +412,8 @@ providers: |---|---|---|---| | `context_window` | 整数 | — | 上下文窗口,即一次请求最多输入多少 token。不写:取价目表的。 | | `max_output_tokens` | 整数 | — | 一次回答最多输出多少 token。不写:取价目表的。 | +| `reasoning` | 布尔 | — | 这个模型会不会推理。模型列表(`GET /v1/models`)带着它,客户端据此给出推理档位。不写:取价目表的;价目表也没写时,列表里不给这一项。 | +| `image_input` | 布尔 | — | 这个模型收不收图。模型列表里以 `input_modalities` 给出。不写:取价目表的;价目表也没写时,列表里不给这一项。 | ```yaml diff --git a/release-notes/0.64.0.md b/release-notes/0.64.0.md new file mode 100644 index 00000000..8dcf520a --- /dev/null +++ b/release-notes/0.64.0.md @@ -0,0 +1,22 @@ +This release tells clients more about each model. The model listing now says, for every model it knows, how much it can write in one answer, whether it reasons and whether it takes images, besides its context window. An upstream's model can have the last two set by hand, like the context window and output limit before. + +**Upgrade notes** + +- The control-plane protocol version (`CONTROL_API_VERSION`) goes from 39 to 40. ThinkWatch Lite connects only to a core with the same protocol version: ThinkWatch Lite 2026.10.6 includes 0.63.0 (protocol 39) and does not connect to 0.64.0. A server used with it stays on 0.63.0 until the app is updated to a release that includes 0.64.0; `sudo twcore upgrade --version 0.63.0 --restart` switches a server back. +- The request store's schema is unchanged: upgrading from 0.63.0 keeps the request history. +- The configuration format only gains fields (`providers[].model_specs.*.reasoning` and `.image_input`); every 0.63.0 configuration loads. +- Changed in the protocol: + - `ModelRow` gains `reasoning`, `reasoning_source`, `image_input` and `image_input_source` (`SpecSource`). + - `ModelSpecSave` gains `reasoning` and `image_input`. An entry is removed when all four fields are empty. A client written for protocol 39 drops the two new fields when it saves a model spec. +- Message codes, compared with 0.63.0: + - New: `config.model_spec_nothing_set`. + - Removed: `config.model_spec_empty`, replaced by `config.model_spec_nothing_set` because the sentence now names four fields. + +**What the model listing carries.** `GET /v1/models` adds three fields to each model in the OpenAI and Anthropic shapes, each only when it is known: +- `max_output_tokens`, the most tokens one answer can have; +- `supports_reasoning`, whether the model reasons; +- `input_modalities`, `["text", "image"]` or `["text"]`. + +The Gemini shape adds `thinking`. The values come from the same place as the context window: the upstream's hand-written model spec first, then the price table. When neither says, the field is left out rather than guessed. Clients then use their own defaults: a client told "does not reason" hides its reasoning levels, and that is not something to guess. The price table now keeps "not stated" apart from "no" for reasoning, and reads image support from the dataset's `supports_vision`. + +**Reasoning and image input set by hand.** `providers[].model_specs` takes `reasoning` and `image_input` (true or false) next to `context_window` and `max_output_tokens`, each overriding the price table on its own. `PUT /provider-model-spec` sets them. The speed test also uses the result: a model set by hand to reason gets room to reason in its probe request, as models the price table marks as reasoning already did.