Files
ragflow/conf/models/zhipu-ai.json

397 lines
7.2 KiB
JSON
Raw Permalink Normal View History

{
"name": "ZHIPU-AI",
"rank": 993,
"url": {
"default": "https://open.bigmodel.cn/api/paas/v4"
},
"url_suffix": {
"chat": "chat/completions",
"async_chat": "async/chat/completions",
"async_result": "async-result",
"embedding": "embeddings",
"rerank": "rerank",
"ocr": "layout_parsing",
"asr": "audio/transcriptions",
"tts": "audio/speech",
"files": "files",
"models": "models"
},
"class": "glm",
"models": [
{
"name": "glm-5.2",
"content_length": 1000000,
"max_output": 128000,
"model_types": [
"chat"
],
"thinking": {
"default_value": true,
"clear_thinking": true
},
"tools": {
"support": true
}
},
{
"name": "glm-5.1",
"content_length": 200000,
"max_output": 128000,
"model_types": [
"chat"
],
"thinking": {
"default_value": true,
"clear_thinking": true
},
"tools": {
"support": true
}
},
{
"name": "glm-5",
"content_length": 200000,
"max_output": 128000,
"model_types": [
"chat"
],
"thinking": {
"default_value": true,
"clear_thinking": true
},
"tools": {
"support": true
}
},
{
"name": "glm-5-turbo",
"content_length": 200000,
"max_output": 128000,
"model_types": [
"chat"
],
"thinking": {
"default_value": true,
"clear_thinking": true
},
"tools": {
"support": true
}
},
{
"name": "glm-5v-turbo",
"content_length": 200000,
"max_output": 128000,
"model_types": [
"chat"
],
"thinking": {
"default_value": true,
"clear_thinking": true
},
"tools": {
"support": true
}
},
{
"name": "glm-4.7",
"content_length": 200000,
"max_output": 128000,
"model_types": [
"chat"
],
"thinking": {
"default_value": true,
"clear_thinking": true
},
"tools": {
"support": true
}
},
{
"name": "glm-4.7-flashx",
"content_length": 200000,
"max_output": 128000,
"model_types": [
"chat"
],
"thinking": {
"default_value": true,
"clear_thinking": true
},
"tools": {
"support": true
}
},
{
"name": "glm-4.6",
"content_length": 200000,
"max_output": 128000,
"model_types": [
"chat"
],
"thinking": {
"default_value": true,
"clear_thinking": true
},
"tools": {
"support": true
}
},
{
"name": "glm-4.6v-Flash",
"content_length": 128000,
"max_output": 32768,
"model_types": [
"chat",
"vision"
],
"thinking": {
"default_value": true,
"clear_thinking": true
},
"tools": {
"support": true
}
},
{
"name": "glm-4.5",
"content_length": 128000,
"max_output": 96000,
"model_types": [
"chat"
],
"thinking": {
"default_value": true,
"clear_thinking": true
},
"tools": {
"support": true
}
},
{
"name": "glm-4.5-x",
"content_length": 128000,
"max_output": 96000,
"model_types": [
"chat"
],
"thinking": {
"default_value": true,
"clear_thinking": true
},
"tools": {
"support": true
}
},
{
"name": "glm-4.5-air",
"content_length": 128000,
"max_output": 96000,
"model_types": [
"chat"
],
"thinking": {
"default_value": true,
"clear_thinking": true
},
"tools": {
"support": true
}
},
{
"name": "glm-4.5-airx",
"content_length": 128000,
"max_output": 96000,
"model_types": [
"chat"
],
"thinking": {
"default_value": true,
"clear_thinking": true
},
"tools": {
"support": true
}
},
{
"name": "glm-4.5-flash",
"content_length": 128000,
"max_output": 96000,
"model_types": [
"chat"
],
"thinking": {
"default_value": true,
"clear_thinking": true
},
"tools": {
"support": true
}
},
{
"name": "glm-4.5v",
"content_length": 65536,
"max_output": 16384,
"model_types": [
"vision"
],
"thinking": {
"default_value": true,
"clear_thinking": true
},
"tools": {
"support": true
}
},
{
"name": "glm-4-plus",
"content_length": 128000,
"max_output": 4096,
"model_types": [
"chat"
],
"tools": {
"support": true
}
},
{
"name": "glm-4-0520",
"content_length": 128000,
"max_output": 4096,
"model_types": [
"chat"
],
"tools": {
"support": true
}
},
{
"name": "glm-4",
"content_length": 128000,
"max_output": 4096,
"model_types": [
"chat"
],
"tools": {
"support": true
}
},
{
"name": "glm-4-airx",
"content_length": 8192,
"max_output": 4096,
"model_types": [
"chat"
],
"tools": {
"support": true
}
},
{
"name": "glm-4-air",
"content_length": 128000,
"max_output": 16384,
"model_types": [
"chat"
],
"tools": {
"support": true
}
},
{
"name": "glm-4-flash",
"content_length": 128000,
"max_output": 16384,
"model_types": [
"chat"
],
"tools": {
"support": true
}
},
{
"name": "glm-4-flashx",
"content_length": 128000,
"max_output": 16384,
"model_types": [
"chat"
],
"tools": {
"support": true
}
},
{
"name": "glm-4-long",
"content_length": 1000000,
"max_output": 4096,
"model_types": [
"chat"
],
"tools": {
"support": true
}
},
{
"name": "glm-4v",
"content_length": 16384,
"max_output": 1024,
"model_types": [
"vision"
]
},
{
"name": "glm-4-9b",
"content_length": 128000,
"max_output": 16384,
"model_types": [
"chat"
],
"tools": {
"support": true
}
},
{
"name": "embedding-2",
"max_tokens": 512,
"model_types": [
"embedding"
feat: add batch_size to all embedding model configs (#17877) ## Summary Add a `batch_size` field to every embedding model entry in `conf/models/*.json`. The field represents the maximum number of text inputs that can be submitted to the embedding API in a single request. **75 embedding models across 30 config files** now carry a `batch_size`. Values were verified against each provider's official documentation (see the verification table at `Desktop/embedding_models_verified.md`). ## Distribution | batch_size | # models | Provider / Model | |---|---|---| | 1 | 2 | AWS Bedrock `amazon.titan-embed-text-v1/v2:0` — Bedrock `invoke` accepts a single input per call | | 10 | 3 | Aliyun `text-embedding-v3/v4`, Volcengine `doubao-embedding-vision-251215` | | 16 | 5 | BaiChuan `Baichuan-Text-Embedding`, Baidu Qianfan `embedding-v1`, Mistral `mistral-embed`, Replicate (x2) | | 32 | 10 | NVIDIA NIM (x3), SILICONFLOW (x2), PPIO (x3), GiteeAI `bge-m3`, HuaweiCloud `bge-m3` | | 50 | 4 | Tencent Hunyuan `kinfra` embeddings (x4) — `InputList.N` max 50 | | 96 | 8 | Cohere embed-v3/v4 (x5), Bedrock Cohere (x3) | | 100 | 3 | Google Gemini `text-embedding-004`, Upstage (x2) | | 512 | 5 | Zhipu GLM `embedding-2/3` (x2), Perplexity `pplx-embed` (x2), Astraflow `text-embedding-3-large` | | 1000 | 9 | Voyage AI (x9) — API reference max | | 1024 | 1 | DeepInfra `Qwen/Qwen3-Embedding-4B` | | 2048 | 15 | OpenAI (x3) + OpenAI-API-compatible proxies (CometAPI, n1n, Jiekou.AI, GreenPT, TogetherAI, NovitaAI) — OpenAI contract limit | | 16384 | 10 | Jina (x8), 302.AI, GiteeAI `jina-clip-v2` — no documented Jina batch limit, safe high cap | --------- Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-08-06 10:47:38 +08:00
],
"max_batch_size": 512
},
{
"name": "embedding-3",
"max_tokens": 512,
"model_types": [
"embedding"
feat: add batch_size to all embedding model configs (#17877) ## Summary Add a `batch_size` field to every embedding model entry in `conf/models/*.json`. The field represents the maximum number of text inputs that can be submitted to the embedding API in a single request. **75 embedding models across 30 config files** now carry a `batch_size`. Values were verified against each provider's official documentation (see the verification table at `Desktop/embedding_models_verified.md`). ## Distribution | batch_size | # models | Provider / Model | |---|---|---| | 1 | 2 | AWS Bedrock `amazon.titan-embed-text-v1/v2:0` — Bedrock `invoke` accepts a single input per call | | 10 | 3 | Aliyun `text-embedding-v3/v4`, Volcengine `doubao-embedding-vision-251215` | | 16 | 5 | BaiChuan `Baichuan-Text-Embedding`, Baidu Qianfan `embedding-v1`, Mistral `mistral-embed`, Replicate (x2) | | 32 | 10 | NVIDIA NIM (x3), SILICONFLOW (x2), PPIO (x3), GiteeAI `bge-m3`, HuaweiCloud `bge-m3` | | 50 | 4 | Tencent Hunyuan `kinfra` embeddings (x4) — `InputList.N` max 50 | | 96 | 8 | Cohere embed-v3/v4 (x5), Bedrock Cohere (x3) | | 100 | 3 | Google Gemini `text-embedding-004`, Upstage (x2) | | 512 | 5 | Zhipu GLM `embedding-2/3` (x2), Perplexity `pplx-embed` (x2), Astraflow `text-embedding-3-large` | | 1000 | 9 | Voyage AI (x9) — API reference max | | 1024 | 1 | DeepInfra `Qwen/Qwen3-Embedding-4B` | | 2048 | 15 | OpenAI (x3) + OpenAI-API-compatible proxies (CometAPI, n1n, Jiekou.AI, GreenPT, TogetherAI, NovitaAI) — OpenAI contract limit | | 16384 | 10 | Jina (x8), 302.AI, GiteeAI `jina-clip-v2` — no documented Jina batch limit, safe high cap | --------- Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-08-06 10:47:38 +08:00
],
"max_batch_size": 512
},
{
"name": "glm-asr-2512",
"max_tokens": 4000,
"model_types": [
"asr"
]
},
{
"name": "glm-tts",
"model_types": [
"tts"
]
},
{
"name": "glm-ocr",
"model_types": [
"ocr"
]
},
{
"name": "rerank",
"model_types": [
"rerank"
]
}
]
}