feat(compute): 补 GLM-5.3 系列规格并标注 Ox Alpha 真身 (#81)

* feat(compute): 补 GLM-5.3 系列规格并标注 Ox Alpha 真身

Ox Alpha 是智谱 GLM-5.3-Flash 的匿名发布别名(canonical_slug
z-ai/glm-5.3-flash-20260826,2026-08-26 揭晓后已从 OpenRouter 列表下架)。

改动:
- model-specs/zhipu.json 新增 glm-5.3 与 glm-5.3-flash 正式规格。两者均
  强制思考(z.ai 文档:thinking.type 仅接受 enabled;OpenRouter models API:
  reasoning.mandatory=true,supported_efforts=[max,high,low]),故
  routing.reasoning.supportedModes 不含 off。按 #79 的教训只用 exact 匹配,
  避免 glm-5.3* 误吞 glm-5.3-flash。
- model-specs/stealth.json 的 ox-alpha 用 spec.extra.modelOrigin 标注真身。
  条目保留:云端算力可能仍以旧别名下发该模型。
- schemas/model-spec.schema.json 为 extra.thinkingOnly / thinkingDefault
  补正式定义与 description。此前这两个键无任何说明,客户端因此从未消费,
  用户选「关闭思考」即触发上游 400(见 desirecore#2307)。
- scripts/validate.mjs 新增静默失效键名巡检:extra 是开放对象,写错键名
  既不报错也无告警。当前巡出 17 处扁平 extra.reasoningEffort。刻意只告警
  不失败——存量取值需逐个核实各自接入面实际接受哪些 effort。

* docs(schema): 写明 extra.reasoning 与 thinkingOnly 的职责边界

provider schema 的 extra.reasoning 此前只有子字段 description、对象本身没有,
维护者看不出它与 model-spec 的 thinkingOnly 分别回答什么问题——issue
desirecore#2307 的误解正源于此。补上对象级说明:本键答「接入面接受哪些 effort
值」且只能写在 provider model;thinkingOnly 答「off 能不能用」、不声明深度档位。
This commit is contained in:
2026-08-27 15:21:25 +08:00
committed by GitHub
parent fa2f10dd0d
commit c46cf07d2f
7 changed files with 196 additions and 7 deletions

View File

@@ -105,7 +105,7 @@ describe('真实数据全量校验', () => {
if (spec.routing) routed.push(spec) if (spec.routing) routed.push(spec)
} }
} }
assert.equal(routed.length, 41) assert.equal(routed.length, 43)
assert.equal(routed.every((spec) => spec.routing.reasoning.supportedModes.includes(spec.routing.reasoning.defaultMode)), true) assert.equal(routed.every((spec) => spec.routing.reasoning.supportedModes.includes(spec.routing.reasoning.defaultMode)), true)
assert.equal(routed.every((spec) => Array.isArray(spec.spec.capabilities)), true) assert.equal(routed.every((spec) => Array.isArray(spec.spec.capabilities)), true)
}) })

View File

@@ -1,9 +1,9 @@
{ {
"description": "Stealth 系列模型规格。参数来源OpenRouter Models API。", "description": "Stealth 系列模型规格。参数来源OpenRouter Models API。这些是厂商匿名发布的临时别名,揭晓后用 spec.extra.modelOrigin 标注真身,条目本身保留——云端算力可能仍以旧别名下发该模型。",
"specs": [ "specs": [
{ {
"id": "ox-alpha", "id": "ox-alpha",
"displayName": "Ox Alpha", "displayName": "GLM-5.3-Flash (Ox Alpha)",
"family": "ox-alpha", "family": "ox-alpha",
"match": { "match": {
"exact": [ "exact": [
@@ -34,10 +34,15 @@
"defaultTemperature": 1, "defaultTemperature": 1,
"defaultTopP": 0.95, "defaultTopP": 0.95,
"supportsReasoning": true, "supportsReasoning": true,
"description": "Ox Alpha 多模态推理模型,面向编程、长程 Agent、复杂推理与生产工作负载", "description": "Ox Alpha 是智谱 GLM-5.3-Flash 的匿名发布别名2026-08-26 揭晓)。原生多模态推理模型,面向高效编码与长程 Agent 任务100 万上下文;强制思考,无法关闭",
"extra": { "extra": {
"thinkingDefault": true, "thinkingDefault": true,
"thinkingOnly": true "thinkingOnly": true,
"modelOrigin": {
"vendor": "zhipu",
"canonicalModelId": "glm-5.3-flash",
"family": "glm-5.3-flash"
}
}, },
"releasedAt": "2026-08-21" "releasedAt": "2026-08-21"
} }

View File

@@ -1,6 +1,115 @@
{ {
"description": "智谱 GLM 系列模型规格。参数来源config-center compute/providers/zhipu.json。glm-5 与 glm-5.1/glm-5-turbo/glm-5v-turbo 前缀相近,故各自仅用 exact 主键匹配,不用宽 pattern 以防误吞。", "description": "智谱 GLM 系列模型规格。参数来源config-center compute/providers/zhipu.json。glm-5 与 glm-5.1/glm-5-turbo/glm-5v-turbo 前缀相近,故各自仅用 exact 主键匹配,不用宽 pattern 以防误吞。",
"specs": [ "specs": [
{
"id": "glm-5.3",
"displayName": "GLM-5.3",
"family": "glm-5.3",
"match": {
"exact": [
"glm-5.3",
"z-ai/glm-5.3"
]
},
"spec": {
"contextWindow": 1048576,
"maxOutputTokens": 131072,
"capabilities": [
"chat",
"reasoning",
"deep_thinking",
"code",
"math",
"multilingual",
"tool_use",
"agent",
"long_context"
],
"serviceType": [
"chat"
],
"defaultTemperature": 1,
"defaultTopP": 0.95,
"supportsReasoning": true,
"description": "智谱 GLM-5.3 大推理模型,面向复杂软件工程与长程 Agent 任务100 万上下文;强制思考,无法关闭",
"extra": {
"thinkingDefault": true,
"thinkingOnly": true
},
"releasedAt": "2026-08-16"
},
"routing": {
"tier": "flagship",
"routingPriority": 50,
"eligibleForAgent": true,
"defaultReference": false,
"reasoning": {
"supportedModes": [
"auto",
"low",
"high",
"max"
],
"defaultMode": "high"
}
}
},
{
"id": "glm-5.3-flash",
"displayName": "GLM-5.3-Flash",
"family": "glm-5.3-flash",
"match": {
"exact": [
"glm-5.3-flash",
"z-ai/glm-5.3-flash"
]
},
"spec": {
"contextWindow": 1048576,
"maxOutputTokens": 131072,
"capabilities": [
"chat",
"reasoning",
"deep_thinking",
"code",
"tool_use",
"agent",
"long_context",
"vision",
"image_understanding",
"video_understanding"
],
"serviceType": [
"chat",
"reasoning",
"vision"
],
"defaultTemperature": 1,
"defaultTopP": 0.95,
"supportsReasoning": true,
"description": "智谱 GLM-5.3-Flash 原生多模态模型,混合稀疏与线性注意力,面向高效编码与长程 Agent 任务100 万上下文;强制思考,无法关闭",
"extra": {
"thinkingDefault": true,
"thinkingOnly": true
},
"releasedAt": "2026-08-26"
},
"routing": {
"tier": "balanced",
"routingPriority": 48,
"eligibleForAgent": true,
"defaultReference": false,
"reasoning": {
"supportedModes": [
"auto",
"low",
"high",
"max"
],
"defaultMode": "high"
}
}
},
{ {
"id": "glm-5.2", "id": "glm-5.2",
"displayName": "GLM-5.2", "displayName": "GLM-5.2",

View File

@@ -1,6 +1,6 @@
{ {
"version": "1.0.0", "version": "1.0.0",
"presetDataVersion": 94, "presetDataVersion": 95,
"updatedAt": "2026-08-25", "updatedAt": "2026-08-26",
"description": "DesireCore 官方配置中心" "description": "DesireCore 官方配置中心"
} }

View File

@@ -160,6 +160,16 @@
"type": "object", "type": "object",
"description": "模型内在附加配置(如 apiModelId 全名、推理预算等)。不含价格,也不得声明 Provider/接入面的 reasoning effort 能力。", "description": "模型内在附加配置(如 apiModelId 全名、推理预算等)。不含价格,也不得声明 Provider/接入面的 reasoning effort 能力。",
"not": { "required": ["reasoning"] }, "not": { "required": ["reasoning"] },
"properties": {
"thinkingOnly": {
"type": "boolean",
"description": "模型强制思考、无法关闭(如 GLM-5.3 系列 thinking.type 仅接受 enabled。true 时客户端把 off 从可选档位移除,并把内部 suppressThinking 降级为 auto避免发出必然被上游拒绝的关闭参数。只描述模型固有属性不声明接入面支持哪些深度档位——后者只能写在 provider model 的 extra.reasoning 中。"
},
"thinkingDefault": {
"type": "boolean",
"description": "模型默认开启思考。当前不改变客户端行为,仅供维护者参考。"
}
},
"additionalProperties": true "additionalProperties": true
}, },
"releasedAt": { "releasedAt": {

View File

@@ -256,6 +256,7 @@
"properties": { "properties": {
"reasoning": { "reasoning": {
"type": "object", "type": "object",
"description": "当前 Provider 接入面接受哪些 reasoning effort 值——精确档位矩阵。只能写在 provider model 的 extra 里:同一模型经不同网关支持的集合可能不同,因此 model-spec 的 spec.extra 里写本键会被客户端 applySpec 主动丢弃。想表达「这个模型根本关不掉思考」的用 model-spec 的 spec.extra.thinkingOnly那只回答 off 能否使用、不声明任何深度档位。",
"required": ["supportedEfforts"], "required": ["supportedEfforts"],
"properties": { "properties": {
"supportedEfforts": { "supportedEfforts": {

View File

@@ -105,6 +105,62 @@ export function validateFile(absPath, validators = loadSchemas()) {
return { ok: false, schemaKey, errors: validator.errors } return { ok: false, schemaKey, errors: validator.errors }
} }
/**
* 已知会静默失效的 extra 键名。
*
* `extra` 是开放对象:写错键名既不会被 schema 拒绝,也没有任何运行时告警——数据看着
* 在那儿客户端却永远读不到。desirecore#2307 就是这么来的model-spec 的
* extra.thinkingOnly 写下后零消费provider 侧另有 10 处扁平 extra.reasoningEffort
* 同样从未生效,直到用户撞上一个上游 400 才暴露。
*
* 刻意只警告、不失败:存量条目的正确取值必须逐个核实各自接入面实际接受哪些 effort
* 一次批量改写的风险远大于收益(声明过窄会削掉模型能力,漏写 none 会让原本能关思考
* 的模型失去该选项)。
*/
const SUSPICIOUS_EXTRA_KEYS = {
provider: {
reasoningEffort: '客户端只读嵌套的 extra.reasoning.supportedEfforts扁平写法永不生效',
defaultReasoningEffort: '同上,应并入 extra.reasoning.defaultEffort',
},
spec: {
reasoningEffort: 'reasoning effort 是接入面能力,只能声明在 provider model 的 extra.reasoning 中',
},
}
/** 巡检静默失效键名,返回告警列表(不影响退出码)。 */
function lintSuspiciousExtraKeys(targets) {
const warnings = []
for (const file of targets) {
const rel = relative(ROOT, file)
let data
try {
data = JSON.parse(readFileSync(file, 'utf8'))
} catch {
continue // 解析失败由 schema 校验负责报错
}
if (Array.isArray(data.models)) {
for (const model of data.models) {
for (const [key, hint] of Object.entries(SUSPICIOUS_EXTRA_KEYS.provider)) {
if (model?.extra && Object.hasOwn(model.extra, key)) {
warnings.push(`${rel} → models[${model.modelName}].extra.${key}${hint}`)
}
}
}
}
if (Array.isArray(data.specs)) {
for (const entry of data.specs) {
const extra = entry?.spec?.extra
for (const [key, hint] of Object.entries(SUSPICIOUS_EXTRA_KEYS.spec)) {
if (extra && Object.hasOwn(extra, key)) {
warnings.push(`${rel} → specs[${entry.id}].spec.extra.${key}${hint}`)
}
}
}
}
}
return warnings
}
function main() { function main() {
const args = process.argv.slice(2) const args = process.argv.slice(2)
const fileArgIdx = args.indexOf('--file') const fileArgIdx = args.indexOf('--file')
@@ -143,6 +199,14 @@ function main() {
} }
} }
const warnings = lintSuspiciousExtraKeys(targets)
if (warnings.length > 0) {
console.log()
console.log(`静默失效键名告警(${warnings.length} 处,不影响校验结果):`)
for (const warning of warnings) console.log(` warn ${warning}`)
console.log(' 这些键写在 extra 里不会报错,但客户端从不读取;修正前请逐个核实该接入面实际接受的 effort。')
}
console.log() console.log()
console.log(`Summary: ${validated} validated, ${skipped} skipped, ${failures} failed`) console.log(`Summary: ${validated} validated, ${skipped} skipped, ${failures} failed`)