2026-04-29 17:05:08 +08:00
|
|
|
//
|
|
|
|
|
// Copyright 2026 The InfiniFlow Authors. All Rights Reserved.
|
|
|
|
|
//
|
|
|
|
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
|
|
|
// you may not use this file except in compliance with the License.
|
|
|
|
|
// You may obtain a copy of the License at
|
|
|
|
|
//
|
|
|
|
|
// http://www.apache.org/licenses/LICENSE-2.0
|
|
|
|
|
//
|
|
|
|
|
// Unless required by applicable law or agreed to in writing, software
|
|
|
|
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
|
|
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
|
|
|
// See the License for the specific language governing permissions and
|
|
|
|
|
// limitations under the License.
|
|
|
|
|
//
|
|
|
|
|
|
|
|
|
|
package models
|
|
|
|
|
|
|
|
|
|
import (
|
2026-04-30 16:30:14 +08:00
|
|
|
"bufio"
|
2026-04-29 17:05:08 +08:00
|
|
|
"bytes"
|
Go: implement Encode (embeddings) in vLLM driver (#14688)
### What problem does this PR solve?
The vLLM Go driver shipped with a stub \`Encode\` method that returned
\`not implemented\`, even though vLLM is one of the most common
production-grade self-hosted inference servers and exposes an
OpenAI-compatible embeddings endpoint at \`/v1/embeddings\`.
Users who self-host \`BAAI/bge-m3\`, \`Qwen3-Embedding-*\`,
\`NV-Embed-v2\`, or similar models on vLLM could not run an embedding
call through the Go layer. The existing \`ListModels\` already discovers
the loaded models, but the embedding path failed because \`Encode\` was
a stub.
### What this PR includes
- \`conf/models/vllm.json\`: add \`\"embedding\": \"embeddings\"\` under
\`url_suffix\` so the driver can build the URL from config.
- \`internal/entity/models/vllm.go\`: replace the \`Encode\` stub with a
real implementation. Adds a small local response
type that matches the OpenAI-compatible shape.
No factory change. No interface change.
### How the driver works
- Validate the model name. The API key is optional for self-hosted vLLM,
so the Authorization header is only set when both \`apiConfig\` and
\`ApiKey\` are non-nil and non-empty, the same pattern the recently
merged CheckConnection PR (#14614) uses.
- Resolve the region with a default fallback. Return a clear "missing
base URL" error when the user has not configured
the local access address yet.
- Use a per-call \`context.WithTimeout(30s)\` and
\`http.NewRequestWithContext\`, the same pattern the merged
Aliyun Encode (#14647) and in-flight Ollama Encode (#14664) use.
- Send \`{model, input: [texts]}\` in one request.
- Parse \`data[*].embedding\` and copy each slice into a \`[][]float64\`
indexed by \`data[*].index\`, so the output
order matches the input order.
- Handle both \`float64\` and \`float32\` element types.
- Empty input returns \`[][]float64{}\` with no HTTP call.
- Length mismatch between input and result, out-of-range index, and any
missing slot all return clear errors instead
of silent zero vectors.
### Type of change
- [x] New Feature (non-breaking change which adds functionality)
### How was this tested?
- \`go build ./internal/entity/models/...\` in a clean go 1.25 image
returns exit 0.
- The full method set on \`VllmModel\` still matches the \`ModelDriver\`
interface.
- Pattern parity with the merged Aliyun Encode (#14647), the in-flight
Ollama Encode (#14664), and the existing
SiliconFlow Encode.
Closes #14687
2026-05-11 06:09:17 +02:00
|
|
|
"context"
|
2026-04-29 17:05:08 +08:00
|
|
|
"encoding/json"
|
|
|
|
|
"fmt"
|
|
|
|
|
"io"
|
|
|
|
|
"net/http"
|
2026-05-06 10:41:58 +08:00
|
|
|
"ragflow/internal/common"
|
2026-04-29 17:05:08 +08:00
|
|
|
"strings"
|
|
|
|
|
"time"
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
// VllmModel implements ModelDriver for Vllm AI
|
|
|
|
|
type VllmModel struct {
|
|
|
|
|
BaseURL map[string]string
|
|
|
|
|
URLSuffix URLSuffix
|
|
|
|
|
httpClient *http.Client // Reusable HTTP client with connection pool
|
|
|
|
|
}
|
|
|
|
|
|
2026-05-12 17:17:44 +08:00
|
|
|
func (v *VllmModel) ParseFile() {
|
|
|
|
|
//TODO implement me
|
|
|
|
|
panic("implement me")
|
|
|
|
|
}
|
|
|
|
|
|
2026-04-29 17:05:08 +08:00
|
|
|
// NewVllmModel creates a new Vllm AI model instance
|
|
|
|
|
func NewVllmModel(baseURL map[string]string, urlSuffix URLSuffix) *VllmModel {
|
|
|
|
|
return &VllmModel{
|
|
|
|
|
BaseURL: baseURL,
|
|
|
|
|
URLSuffix: urlSuffix,
|
|
|
|
|
httpClient: &http.Client{
|
|
|
|
|
Timeout: 120 * time.Second,
|
|
|
|
|
Transport: &http.Transport{
|
|
|
|
|
MaxIdleConns: 100,
|
|
|
|
|
MaxIdleConnsPerHost: 10,
|
|
|
|
|
IdleConnTimeout: 90 * time.Second,
|
|
|
|
|
DisableCompression: false,
|
|
|
|
|
},
|
|
|
|
|
},
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (z *VllmModel) NewInstance(baseURL map[string]string) ModelDriver {
|
|
|
|
|
return &VllmModel{
|
|
|
|
|
BaseURL: baseURL,
|
|
|
|
|
URLSuffix: z.URLSuffix,
|
|
|
|
|
httpClient: &http.Client{
|
|
|
|
|
Timeout: 120 * time.Second,
|
|
|
|
|
Transport: &http.Transport{
|
|
|
|
|
MaxIdleConns: 100,
|
|
|
|
|
MaxIdleConnsPerHost: 10,
|
|
|
|
|
IdleConnTimeout: 90 * time.Second,
|
|
|
|
|
DisableCompression: false,
|
|
|
|
|
},
|
|
|
|
|
},
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (z *VllmModel) Name() string {
|
|
|
|
|
return "vllm"
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// ChatWithMessages sends multiple messages with roles and returns response
|
2026-04-30 19:33:57 +08:00
|
|
|
func (z *VllmModel) ChatWithMessages(modelName string, messages []Message, apiConfig *APIConfig, chatModelConfig *ChatConfig) (*ChatResponse, error) {
|
2026-04-30 15:25:01 +08:00
|
|
|
if len(messages) == 0 {
|
|
|
|
|
return nil, fmt.Errorf("messages is empty")
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
var region = "default"
|
2026-05-08 12:02:37 +08:00
|
|
|
if apiConfig != nil && apiConfig.Region != nil && *apiConfig.Region != "" {
|
2026-04-30 15:25:01 +08:00
|
|
|
region = *apiConfig.Region
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
url := fmt.Sprintf("%s/%s", z.BaseURL[region], z.URLSuffix.Chat)
|
|
|
|
|
|
|
|
|
|
// For qwen/glm models, use async chat endpoint
|
|
|
|
|
modelType := strings.Split(modelName, "-")[0]
|
|
|
|
|
if modelType == "qwen" || modelType == "glm" {
|
|
|
|
|
url = fmt.Sprintf("%s/%s", z.BaseURL[region], z.URLSuffix.AsyncChat)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Convert messages to API format
|
|
|
|
|
apiMessages := make([]map[string]interface{}, len(messages))
|
|
|
|
|
for i, msg := range messages {
|
|
|
|
|
apiMessages[i] = map[string]interface{}{
|
|
|
|
|
"role": msg.Role,
|
|
|
|
|
"content": msg.Content,
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Build request body
|
|
|
|
|
reqBody := map[string]interface{}{
|
|
|
|
|
"model": modelName,
|
2026-04-30 16:30:14 +08:00
|
|
|
"messages": apiMessages,
|
2026-04-30 15:25:01 +08:00
|
|
|
"stream": false,
|
|
|
|
|
"temperature": 1,
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if chatModelConfig != nil {
|
|
|
|
|
if chatModelConfig.Stream != nil {
|
|
|
|
|
reqBody["stream"] = *chatModelConfig.Stream
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if chatModelConfig.MaxTokens != nil {
|
|
|
|
|
reqBody["max_tokens"] = *chatModelConfig.MaxTokens
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if chatModelConfig.Temperature != nil {
|
|
|
|
|
reqBody["temperature"] = *chatModelConfig.Temperature
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if chatModelConfig.TopP != nil {
|
|
|
|
|
reqBody["top_p"] = *chatModelConfig.TopP
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if chatModelConfig.Stop != nil {
|
|
|
|
|
reqBody["stop"] = *chatModelConfig.Stop
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if chatModelConfig.Thinking != nil {
|
|
|
|
|
if *chatModelConfig.Thinking {
|
|
|
|
|
reqBody["thinking"] = map[string]interface{}{
|
|
|
|
|
"type": "enabled",
|
|
|
|
|
}
|
|
|
|
|
} else {
|
|
|
|
|
reqBody["thinking"] = map[string]interface{}{
|
|
|
|
|
"type": "disabled",
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
jsonData, err := json.Marshal(reqBody)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to marshal request: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
req, err := http.NewRequest("POST", url, bytes.NewBuffer(jsonData))
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to create request: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
req.Header.Set("Content-Type", "application/json")
|
|
|
|
|
req.Header.Set("Authorization", fmt.Sprintf("Bearer %s", *apiConfig.ApiKey))
|
|
|
|
|
|
|
|
|
|
resp, err := z.httpClient.Do(req)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to send request: %w", err)
|
|
|
|
|
}
|
|
|
|
|
defer resp.Body.Close()
|
|
|
|
|
|
|
|
|
|
body, err := io.ReadAll(resp.Body)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to read response: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if resp.StatusCode != http.StatusOK {
|
|
|
|
|
return nil, fmt.Errorf("API request failed with status %d: %s", resp.StatusCode, string(body))
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Parse response
|
|
|
|
|
var result map[string]interface{}
|
|
|
|
|
if err = json.Unmarshal(body, &result); err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to parse response: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
choices, ok := result["choices"].([]interface{})
|
|
|
|
|
if !ok || len(choices) == 0 {
|
|
|
|
|
return nil, fmt.Errorf("no choices in response")
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
firstChoice, ok := choices[0].(map[string]interface{})
|
|
|
|
|
if !ok {
|
|
|
|
|
return nil, fmt.Errorf("invalid choice format")
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
messageMap, ok := firstChoice["message"].(map[string]interface{})
|
|
|
|
|
if !ok {
|
|
|
|
|
return nil, fmt.Errorf("invalid message format")
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
content, ok := messageMap["content"].(string)
|
|
|
|
|
if !ok {
|
|
|
|
|
return nil, fmt.Errorf("invalid content format")
|
|
|
|
|
}
|
|
|
|
|
|
2026-05-08 12:02:37 +08:00
|
|
|
var reasonContent string
|
|
|
|
|
if chatModelConfig != nil && chatModelConfig.Thinking != nil && *chatModelConfig.Thinking {
|
|
|
|
|
reasonContent, ok = messageMap["reasoning_content"].(string)
|
|
|
|
|
if !ok {
|
|
|
|
|
return nil, fmt.Errorf("invalid content format")
|
|
|
|
|
}
|
|
|
|
|
if reasonContent != "" && reasonContent[0] == '\n' {
|
|
|
|
|
reasonContent = reasonContent[1:]
|
|
|
|
|
}
|
|
|
|
|
}
|
2026-04-30 15:25:01 +08:00
|
|
|
|
|
|
|
|
chatResponse := &ChatResponse{
|
2026-05-08 12:02:37 +08:00
|
|
|
Answer: &content,
|
|
|
|
|
ReasonContent: &reasonContent,
|
2026-04-30 15:25:01 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return chatResponse, nil
|
2026-04-29 17:05:08 +08:00
|
|
|
}
|
|
|
|
|
|
2026-04-30 19:33:57 +08:00
|
|
|
// ChatStreamlyWithSender sends messages and streams response via sender function (best performance, no channel)
|
|
|
|
|
func (z *VllmModel) ChatStreamlyWithSender(modelName string, messages []Message, apiConfig *APIConfig, modelConfig *ChatConfig, sender func(*string, *string) error) error {
|
|
|
|
|
if len(messages) == 0 {
|
|
|
|
|
return fmt.Errorf("messages is empty")
|
|
|
|
|
}
|
|
|
|
|
|
2026-04-30 16:30:14 +08:00
|
|
|
var region = "default"
|
2026-05-08 12:02:37 +08:00
|
|
|
if apiConfig != nil && apiConfig.Region != nil && *apiConfig.Region != "" {
|
2026-04-30 16:30:14 +08:00
|
|
|
region = *apiConfig.Region
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
url := fmt.Sprintf("%s/%s", z.BaseURL[region], z.URLSuffix.Chat)
|
2026-04-30 19:33:57 +08:00
|
|
|
modelType := strings.Split(modelName, "-")[0]
|
2026-04-30 16:30:14 +08:00
|
|
|
if modelType == "qwen" || modelType == "glm" {
|
|
|
|
|
url = fmt.Sprintf("%s/%s", z.BaseURL[region], z.URLSuffix.AsyncChat)
|
|
|
|
|
}
|
|
|
|
|
|
2026-04-30 19:33:57 +08:00
|
|
|
// Convert messages to API format (supporting multimodal content)
|
|
|
|
|
apiMessages := make([]map[string]interface{}, len(messages))
|
|
|
|
|
for i, msg := range messages {
|
|
|
|
|
apiMessages[i] = map[string]interface{}{
|
|
|
|
|
"role": msg.Role,
|
|
|
|
|
"content": msg.Content,
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-04-30 16:30:14 +08:00
|
|
|
// Build request body with streaming enabled
|
|
|
|
|
reqBody := map[string]interface{}{
|
2026-04-30 19:33:57 +08:00
|
|
|
"model": modelName,
|
|
|
|
|
"messages": apiMessages,
|
|
|
|
|
"stream": true,
|
2026-04-30 16:30:14 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if modelConfig.Stream != nil {
|
|
|
|
|
reqBody["stream"] = *modelConfig.Stream
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if modelConfig.MaxTokens != nil {
|
|
|
|
|
reqBody["max_tokens"] = *modelConfig.MaxTokens
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if modelConfig.Temperature != nil {
|
|
|
|
|
reqBody["temperature"] = *modelConfig.Temperature
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if modelConfig.DoSample != nil {
|
|
|
|
|
reqBody["do_sample"] = *modelConfig.DoSample
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if modelConfig.TopP != nil {
|
|
|
|
|
reqBody["top_p"] = *modelConfig.TopP
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if modelConfig.Stop != nil {
|
|
|
|
|
reqBody["stop"] = *modelConfig.Stop
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if modelConfig.Thinking != nil {
|
|
|
|
|
if *modelConfig.Thinking {
|
|
|
|
|
reqBody["thinking"] = map[string]interface{}{
|
|
|
|
|
"type": "enabled",
|
|
|
|
|
}
|
|
|
|
|
} else {
|
|
|
|
|
reqBody["thinking"] = map[string]interface{}{
|
|
|
|
|
"type": "disabled",
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
jsonData, err := json.Marshal(reqBody)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return fmt.Errorf("failed to marshal request: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
req, err := http.NewRequest("POST", url, bytes.NewBuffer(jsonData))
|
|
|
|
|
if err != nil {
|
|
|
|
|
return fmt.Errorf("failed to create request: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
req.Header.Set("Content-Type", "application/json")
|
|
|
|
|
req.Header.Set("Authorization", fmt.Sprintf("Bearer %s", *apiConfig.ApiKey))
|
|
|
|
|
|
|
|
|
|
resp, err := z.httpClient.Do(req)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return fmt.Errorf("failed to send request: %w", err)
|
|
|
|
|
}
|
|
|
|
|
defer resp.Body.Close()
|
|
|
|
|
|
|
|
|
|
if resp.StatusCode != http.StatusOK {
|
|
|
|
|
body, _ := io.ReadAll(resp.Body)
|
|
|
|
|
return fmt.Errorf("API request failed with status %d: %s", resp.StatusCode, string(body))
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// SSE parsing: read line by line
|
|
|
|
|
scanner := bufio.NewScanner(resp.Body)
|
|
|
|
|
for scanner.Scan() {
|
|
|
|
|
line := scanner.Text()
|
2026-05-06 10:41:58 +08:00
|
|
|
common.Info(line)
|
2026-04-30 16:30:14 +08:00
|
|
|
|
|
|
|
|
// SSE data line starts with "data:"
|
|
|
|
|
if !strings.HasPrefix(line, "data:") {
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Extract JSON after "data:"
|
|
|
|
|
data := strings.TrimSpace(line[5:])
|
|
|
|
|
|
|
|
|
|
// [DONE] marks the end of stream
|
|
|
|
|
if data == "[DONE]" {
|
|
|
|
|
break
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Parse the JSON event
|
|
|
|
|
var event map[string]interface{}
|
|
|
|
|
if err = json.Unmarshal([]byte(data), &event); err != nil {
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
choices, ok := event["choices"].([]interface{})
|
|
|
|
|
if !ok || len(choices) == 0 {
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
firstChoice, ok := choices[0].(map[string]interface{})
|
|
|
|
|
if !ok {
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
delta, ok := firstChoice["delta"].(map[string]interface{})
|
|
|
|
|
if !ok {
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
reasoningContent, ok := delta["reasoning_content"].(string)
|
|
|
|
|
if ok && reasoningContent != "" {
|
|
|
|
|
if err := sender(nil, &reasoningContent); err != nil {
|
|
|
|
|
return err
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
content, ok := delta["content"].(string)
|
|
|
|
|
if ok && content != "" {
|
|
|
|
|
if err := sender(&content, nil); err != nil {
|
|
|
|
|
return err
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
finishReason, ok := firstChoice["finish_reason"].(string)
|
|
|
|
|
if ok && finishReason != "" {
|
|
|
|
|
break
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Send [DONE] marker for OpenAI compatibility
|
|
|
|
|
endOfStream := "[DONE]"
|
|
|
|
|
if err = sender(&endOfStream, nil); err != nil {
|
|
|
|
|
return err
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return scanner.Err()
|
2026-04-29 17:05:08 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Encode encodes a list of texts into embeddings
|
Go: implement Encode (embeddings) in vLLM driver (#14688)
### What problem does this PR solve?
The vLLM Go driver shipped with a stub \`Encode\` method that returned
\`not implemented\`, even though vLLM is one of the most common
production-grade self-hosted inference servers and exposes an
OpenAI-compatible embeddings endpoint at \`/v1/embeddings\`.
Users who self-host \`BAAI/bge-m3\`, \`Qwen3-Embedding-*\`,
\`NV-Embed-v2\`, or similar models on vLLM could not run an embedding
call through the Go layer. The existing \`ListModels\` already discovers
the loaded models, but the embedding path failed because \`Encode\` was
a stub.
### What this PR includes
- \`conf/models/vllm.json\`: add \`\"embedding\": \"embeddings\"\` under
\`url_suffix\` so the driver can build the URL from config.
- \`internal/entity/models/vllm.go\`: replace the \`Encode\` stub with a
real implementation. Adds a small local response
type that matches the OpenAI-compatible shape.
No factory change. No interface change.
### How the driver works
- Validate the model name. The API key is optional for self-hosted vLLM,
so the Authorization header is only set when both \`apiConfig\` and
\`ApiKey\` are non-nil and non-empty, the same pattern the recently
merged CheckConnection PR (#14614) uses.
- Resolve the region with a default fallback. Return a clear "missing
base URL" error when the user has not configured
the local access address yet.
- Use a per-call \`context.WithTimeout(30s)\` and
\`http.NewRequestWithContext\`, the same pattern the merged
Aliyun Encode (#14647) and in-flight Ollama Encode (#14664) use.
- Send \`{model, input: [texts]}\` in one request.
- Parse \`data[*].embedding\` and copy each slice into a \`[][]float64\`
indexed by \`data[*].index\`, so the output
order matches the input order.
- Handle both \`float64\` and \`float32\` element types.
- Empty input returns \`[][]float64{}\` with no HTTP call.
- Length mismatch between input and result, out-of-range index, and any
missing slot all return clear errors instead
of silent zero vectors.
### Type of change
- [x] New Feature (non-breaking change which adds functionality)
### How was this tested?
- \`go build ./internal/entity/models/...\` in a clean go 1.25 image
returns exit 0.
- The full method set on \`VllmModel\` still matches the \`ModelDriver\`
interface.
- Pattern parity with the merged Aliyun Encode (#14647), the in-flight
Ollama Encode (#14664), and the existing
SiliconFlow Encode.
Closes #14687
2026-05-11 06:09:17 +02:00
|
|
|
type vllmEmbeddingResponse struct {
|
|
|
|
|
Data []struct {
|
2026-05-11 14:45:30 +08:00
|
|
|
Index int `json:"index"`
|
|
|
|
|
Embedding []float64 `json:"embedding"`
|
Go: implement Encode (embeddings) in vLLM driver (#14688)
### What problem does this PR solve?
The vLLM Go driver shipped with a stub \`Encode\` method that returned
\`not implemented\`, even though vLLM is one of the most common
production-grade self-hosted inference servers and exposes an
OpenAI-compatible embeddings endpoint at \`/v1/embeddings\`.
Users who self-host \`BAAI/bge-m3\`, \`Qwen3-Embedding-*\`,
\`NV-Embed-v2\`, or similar models on vLLM could not run an embedding
call through the Go layer. The existing \`ListModels\` already discovers
the loaded models, but the embedding path failed because \`Encode\` was
a stub.
### What this PR includes
- \`conf/models/vllm.json\`: add \`\"embedding\": \"embeddings\"\` under
\`url_suffix\` so the driver can build the URL from config.
- \`internal/entity/models/vllm.go\`: replace the \`Encode\` stub with a
real implementation. Adds a small local response
type that matches the OpenAI-compatible shape.
No factory change. No interface change.
### How the driver works
- Validate the model name. The API key is optional for self-hosted vLLM,
so the Authorization header is only set when both \`apiConfig\` and
\`ApiKey\` are non-nil and non-empty, the same pattern the recently
merged CheckConnection PR (#14614) uses.
- Resolve the region with a default fallback. Return a clear "missing
base URL" error when the user has not configured
the local access address yet.
- Use a per-call \`context.WithTimeout(30s)\` and
\`http.NewRequestWithContext\`, the same pattern the merged
Aliyun Encode (#14647) and in-flight Ollama Encode (#14664) use.
- Send \`{model, input: [texts]}\` in one request.
- Parse \`data[*].embedding\` and copy each slice into a \`[][]float64\`
indexed by \`data[*].index\`, so the output
order matches the input order.
- Handle both \`float64\` and \`float32\` element types.
- Empty input returns \`[][]float64{}\` with no HTTP call.
- Length mismatch between input and result, out-of-range index, and any
missing slot all return clear errors instead
of silent zero vectors.
### Type of change
- [x] New Feature (non-breaking change which adds functionality)
### How was this tested?
- \`go build ./internal/entity/models/...\` in a clean go 1.25 image
returns exit 0.
- The full method set on \`VllmModel\` still matches the \`ModelDriver\`
interface.
- Pattern parity with the merged Aliyun Encode (#14647), the in-flight
Ollama Encode (#14664), and the existing
SiliconFlow Encode.
Closes #14687
2026-05-11 06:09:17 +02:00
|
|
|
} `json:"data"`
|
|
|
|
|
}
|
|
|
|
|
|
2026-05-11 14:45:30 +08:00
|
|
|
// Embed embeds a list of texts into embeddings
|
|
|
|
|
func (z *VllmModel) Embed(modelName *string, texts []string, apiConfig *APIConfig, embeddingConfig *EmbeddingConfig) ([]EmbeddingData, error) {
|
Go: implement Encode (embeddings) in vLLM driver (#14688)
### What problem does this PR solve?
The vLLM Go driver shipped with a stub \`Encode\` method that returned
\`not implemented\`, even though vLLM is one of the most common
production-grade self-hosted inference servers and exposes an
OpenAI-compatible embeddings endpoint at \`/v1/embeddings\`.
Users who self-host \`BAAI/bge-m3\`, \`Qwen3-Embedding-*\`,
\`NV-Embed-v2\`, or similar models on vLLM could not run an embedding
call through the Go layer. The existing \`ListModels\` already discovers
the loaded models, but the embedding path failed because \`Encode\` was
a stub.
### What this PR includes
- \`conf/models/vllm.json\`: add \`\"embedding\": \"embeddings\"\` under
\`url_suffix\` so the driver can build the URL from config.
- \`internal/entity/models/vllm.go\`: replace the \`Encode\` stub with a
real implementation. Adds a small local response
type that matches the OpenAI-compatible shape.
No factory change. No interface change.
### How the driver works
- Validate the model name. The API key is optional for self-hosted vLLM,
so the Authorization header is only set when both \`apiConfig\` and
\`ApiKey\` are non-nil and non-empty, the same pattern the recently
merged CheckConnection PR (#14614) uses.
- Resolve the region with a default fallback. Return a clear "missing
base URL" error when the user has not configured
the local access address yet.
- Use a per-call \`context.WithTimeout(30s)\` and
\`http.NewRequestWithContext\`, the same pattern the merged
Aliyun Encode (#14647) and in-flight Ollama Encode (#14664) use.
- Send \`{model, input: [texts]}\` in one request.
- Parse \`data[*].embedding\` and copy each slice into a \`[][]float64\`
indexed by \`data[*].index\`, so the output
order matches the input order.
- Handle both \`float64\` and \`float32\` element types.
- Empty input returns \`[][]float64{}\` with no HTTP call.
- Length mismatch between input and result, out-of-range index, and any
missing slot all return clear errors instead
of silent zero vectors.
### Type of change
- [x] New Feature (non-breaking change which adds functionality)
### How was this tested?
- \`go build ./internal/entity/models/...\` in a clean go 1.25 image
returns exit 0.
- The full method set on \`VllmModel\` still matches the \`ModelDriver\`
interface.
- Pattern parity with the merged Aliyun Encode (#14647), the in-flight
Ollama Encode (#14664), and the existing
SiliconFlow Encode.
Closes #14687
2026-05-11 06:09:17 +02:00
|
|
|
if len(texts) == 0 {
|
2026-05-11 14:45:30 +08:00
|
|
|
return []EmbeddingData{}, nil
|
Go: implement Encode (embeddings) in vLLM driver (#14688)
### What problem does this PR solve?
The vLLM Go driver shipped with a stub \`Encode\` method that returned
\`not implemented\`, even though vLLM is one of the most common
production-grade self-hosted inference servers and exposes an
OpenAI-compatible embeddings endpoint at \`/v1/embeddings\`.
Users who self-host \`BAAI/bge-m3\`, \`Qwen3-Embedding-*\`,
\`NV-Embed-v2\`, or similar models on vLLM could not run an embedding
call through the Go layer. The existing \`ListModels\` already discovers
the loaded models, but the embedding path failed because \`Encode\` was
a stub.
### What this PR includes
- \`conf/models/vllm.json\`: add \`\"embedding\": \"embeddings\"\` under
\`url_suffix\` so the driver can build the URL from config.
- \`internal/entity/models/vllm.go\`: replace the \`Encode\` stub with a
real implementation. Adds a small local response
type that matches the OpenAI-compatible shape.
No factory change. No interface change.
### How the driver works
- Validate the model name. The API key is optional for self-hosted vLLM,
so the Authorization header is only set when both \`apiConfig\` and
\`ApiKey\` are non-nil and non-empty, the same pattern the recently
merged CheckConnection PR (#14614) uses.
- Resolve the region with a default fallback. Return a clear "missing
base URL" error when the user has not configured
the local access address yet.
- Use a per-call \`context.WithTimeout(30s)\` and
\`http.NewRequestWithContext\`, the same pattern the merged
Aliyun Encode (#14647) and in-flight Ollama Encode (#14664) use.
- Send \`{model, input: [texts]}\` in one request.
- Parse \`data[*].embedding\` and copy each slice into a \`[][]float64\`
indexed by \`data[*].index\`, so the output
order matches the input order.
- Handle both \`float64\` and \`float32\` element types.
- Empty input returns \`[][]float64{}\` with no HTTP call.
- Length mismatch between input and result, out-of-range index, and any
missing slot all return clear errors instead
of silent zero vectors.
### Type of change
- [x] New Feature (non-breaking change which adds functionality)
### How was this tested?
- \`go build ./internal/entity/models/...\` in a clean go 1.25 image
returns exit 0.
- The full method set on \`VllmModel\` still matches the \`ModelDriver\`
interface.
- Pattern parity with the merged Aliyun Encode (#14647), the in-flight
Ollama Encode (#14664), and the existing
SiliconFlow Encode.
Closes #14687
2026-05-11 06:09:17 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if modelName == nil || *modelName == "" {
|
|
|
|
|
return nil, fmt.Errorf("model name is required")
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
region := "default"
|
|
|
|
|
if apiConfig != nil && apiConfig.Region != nil && *apiConfig.Region != "" {
|
|
|
|
|
region = *apiConfig.Region
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
baseURL := z.BaseURL[region]
|
|
|
|
|
if baseURL == "" {
|
|
|
|
|
baseURL = z.BaseURL["default"]
|
|
|
|
|
}
|
|
|
|
|
if baseURL == "" {
|
|
|
|
|
return nil, fmt.Errorf("missing base URL: please configure the local access address for vLLM (e.g., http://127.0.0.1:8000/v1)")
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
url := fmt.Sprintf("%s/%s", strings.TrimSuffix(baseURL, "/"), z.URLSuffix.Embedding)
|
|
|
|
|
|
|
|
|
|
reqBody := map[string]interface{}{
|
|
|
|
|
"model": *modelName,
|
|
|
|
|
"input": texts,
|
|
|
|
|
}
|
|
|
|
|
if embeddingConfig != nil && embeddingConfig.Dimension > 0 {
|
|
|
|
|
reqBody["dimensions"] = embeddingConfig.Dimension
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
jsonData, err := json.Marshal(reqBody)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to marshal request: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
|
|
|
|
defer cancel()
|
|
|
|
|
|
|
|
|
|
req, err := http.NewRequestWithContext(ctx, "POST", url, bytes.NewBuffer(jsonData))
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to create request: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
req.Header.Set("Content-Type", "application/json")
|
|
|
|
|
if apiConfig != nil && apiConfig.ApiKey != nil && *apiConfig.ApiKey != "" {
|
|
|
|
|
req.Header.Set("Authorization", fmt.Sprintf("Bearer %s", *apiConfig.ApiKey))
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
resp, err := z.httpClient.Do(req)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to send request: %w", err)
|
|
|
|
|
}
|
|
|
|
|
defer resp.Body.Close()
|
|
|
|
|
|
|
|
|
|
body, err := io.ReadAll(resp.Body)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to read response: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if resp.StatusCode != http.StatusOK {
|
|
|
|
|
return nil, fmt.Errorf("vLLM embeddings API error: %s, body: %s", resp.Status, string(body))
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
var parsed vllmEmbeddingResponse
|
|
|
|
|
if err = json.Unmarshal(body, &parsed); err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to parse response: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
2026-05-11 14:45:30 +08:00
|
|
|
var embeddings []EmbeddingData
|
|
|
|
|
for _, dataElem := range parsed.Data {
|
|
|
|
|
var embeddingData EmbeddingData
|
|
|
|
|
embeddingData.Embedding = dataElem.Embedding
|
|
|
|
|
embeddingData.Index = dataElem.Index
|
|
|
|
|
embeddings = append(embeddings, embeddingData)
|
Go: implement Encode (embeddings) in vLLM driver (#14688)
### What problem does this PR solve?
The vLLM Go driver shipped with a stub \`Encode\` method that returned
\`not implemented\`, even though vLLM is one of the most common
production-grade self-hosted inference servers and exposes an
OpenAI-compatible embeddings endpoint at \`/v1/embeddings\`.
Users who self-host \`BAAI/bge-m3\`, \`Qwen3-Embedding-*\`,
\`NV-Embed-v2\`, or similar models on vLLM could not run an embedding
call through the Go layer. The existing \`ListModels\` already discovers
the loaded models, but the embedding path failed because \`Encode\` was
a stub.
### What this PR includes
- \`conf/models/vllm.json\`: add \`\"embedding\": \"embeddings\"\` under
\`url_suffix\` so the driver can build the URL from config.
- \`internal/entity/models/vllm.go\`: replace the \`Encode\` stub with a
real implementation. Adds a small local response
type that matches the OpenAI-compatible shape.
No factory change. No interface change.
### How the driver works
- Validate the model name. The API key is optional for self-hosted vLLM,
so the Authorization header is only set when both \`apiConfig\` and
\`ApiKey\` are non-nil and non-empty, the same pattern the recently
merged CheckConnection PR (#14614) uses.
- Resolve the region with a default fallback. Return a clear "missing
base URL" error when the user has not configured
the local access address yet.
- Use a per-call \`context.WithTimeout(30s)\` and
\`http.NewRequestWithContext\`, the same pattern the merged
Aliyun Encode (#14647) and in-flight Ollama Encode (#14664) use.
- Send \`{model, input: [texts]}\` in one request.
- Parse \`data[*].embedding\` and copy each slice into a \`[][]float64\`
indexed by \`data[*].index\`, so the output
order matches the input order.
- Handle both \`float64\` and \`float32\` element types.
- Empty input returns \`[][]float64{}\` with no HTTP call.
- Length mismatch between input and result, out-of-range index, and any
missing slot all return clear errors instead
of silent zero vectors.
### Type of change
- [x] New Feature (non-breaking change which adds functionality)
### How was this tested?
- \`go build ./internal/entity/models/...\` in a clean go 1.25 image
returns exit 0.
- The full method set on \`VllmModel\` still matches the \`ModelDriver\`
interface.
- Pattern parity with the merged Aliyun Encode (#14647), the in-flight
Ollama Encode (#14664), and the existing
SiliconFlow Encode.
Closes #14687
2026-05-11 06:09:17 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return embeddings, nil
|
2026-04-29 17:05:08 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (z *VllmModel) ListModels(apiConfig *APIConfig) ([]string, error) {
|
2026-04-30 16:30:14 +08:00
|
|
|
var region = "default"
|
|
|
|
|
|
fix(go): wire CheckConnection to ListModels in ollama, lm-studio, and vllm (#14614)
### What problem does this PR solve?
Three Go drivers had `CheckConnection` returning a hardcoded `no such
method` error, even though each one already has a working `ListModels`
that hits the configured base URL with the configured API key. So the
"Check connection" button in the model provider UI always failed for
these three providers, even when the underlying setup was fine.
Affected drivers:
- `internal/entity/models/ollama.go`
- `internal/entity/models/lmstudio.go`
- `internal/entity/models/vllm.go`
This is a real user-facing gap because Ollama and LM Studio are two of
the most popular local LLM runners, and vLLM is widely used for
self-hosted deployments.
### What this PR includes
For each of the three drivers, replace the stub with a small
implementation that calls `ListModels` and returns its error:
```go
func (o *OllamaModel) CheckConnection(apiConfig *APIConfig) error {
_, err := o.ListModels(apiConfig)
return err
}
```
This is the exact pattern that xai, moonshot, deepseek, aliyun, and
gitee already use for the same method.
No JSON change. No factory change. No interface change.
### Type of change
- [x] Bug Fix (non-breaking change which fixes an issue)
### How was this tested?
- `go build ./internal/entity/models/...` in a clean go 1.25 image (the
go.mod minimum) returns exit 0.
- The full ModelDriver interface still resolves on each driver
(NewInstance, Name, ChatWithMessages, ChatStreamlyWithSender, Encode,
Rerank, ListModels, Balance, CheckConnection).
- Pattern parity with the existing xai, moonshot, deepseek, aliyun, and
gitee CheckConnection methods.
Closes #14609
2026-05-08 06:00:10 +02:00
|
|
|
if apiConfig != nil && apiConfig.Region != nil && *apiConfig.Region != "" {
|
2026-04-30 16:30:14 +08:00
|
|
|
region = *apiConfig.Region
|
|
|
|
|
}
|
|
|
|
|
|
fix(go): wire CheckConnection to ListModels in ollama, lm-studio, and vllm (#14614)
### What problem does this PR solve?
Three Go drivers had `CheckConnection` returning a hardcoded `no such
method` error, even though each one already has a working `ListModels`
that hits the configured base URL with the configured API key. So the
"Check connection" button in the model provider UI always failed for
these three providers, even when the underlying setup was fine.
Affected drivers:
- `internal/entity/models/ollama.go`
- `internal/entity/models/lmstudio.go`
- `internal/entity/models/vllm.go`
This is a real user-facing gap because Ollama and LM Studio are two of
the most popular local LLM runners, and vLLM is widely used for
self-hosted deployments.
### What this PR includes
For each of the three drivers, replace the stub with a small
implementation that calls `ListModels` and returns its error:
```go
func (o *OllamaModel) CheckConnection(apiConfig *APIConfig) error {
_, err := o.ListModels(apiConfig)
return err
}
```
This is the exact pattern that xai, moonshot, deepseek, aliyun, and
gitee already use for the same method.
No JSON change. No factory change. No interface change.
### Type of change
- [x] Bug Fix (non-breaking change which fixes an issue)
### How was this tested?
- `go build ./internal/entity/models/...` in a clean go 1.25 image (the
go.mod minimum) returns exit 0.
- The full ModelDriver interface still resolves on each driver
(NewInstance, Name, ChatWithMessages, ChatStreamlyWithSender, Encode,
Rerank, ListModels, Balance, CheckConnection).
- Pattern parity with the existing xai, moonshot, deepseek, aliyun, and
gitee CheckConnection methods.
Closes #14609
2026-05-08 06:00:10 +02:00
|
|
|
baseURL := z.BaseURL[region]
|
|
|
|
|
if baseURL == "" {
|
|
|
|
|
baseURL = z.BaseURL["default"]
|
|
|
|
|
}
|
|
|
|
|
if baseURL == "" {
|
|
|
|
|
return nil, fmt.Errorf("missing base URL: please configure the local access address for vLLM (e.g., http://127.0.0.1:8000/v1)")
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
url := fmt.Sprintf("%s/%s", baseURL, z.URLSuffix.Models)
|
2026-04-30 16:30:14 +08:00
|
|
|
|
|
|
|
|
reqBody := map[string]interface{}{}
|
|
|
|
|
|
|
|
|
|
jsonData, err := json.Marshal(reqBody)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to marshal request: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
req, err := http.NewRequest("GET", url, bytes.NewBuffer(jsonData))
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to create request: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
req.Header.Set("Content-Type", "application/json")
|
fix(go): wire CheckConnection to ListModels in ollama, lm-studio, and vllm (#14614)
### What problem does this PR solve?
Three Go drivers had `CheckConnection` returning a hardcoded `no such
method` error, even though each one already has a working `ListModels`
that hits the configured base URL with the configured API key. So the
"Check connection" button in the model provider UI always failed for
these three providers, even when the underlying setup was fine.
Affected drivers:
- `internal/entity/models/ollama.go`
- `internal/entity/models/lmstudio.go`
- `internal/entity/models/vllm.go`
This is a real user-facing gap because Ollama and LM Studio are two of
the most popular local LLM runners, and vLLM is widely used for
self-hosted deployments.
### What this PR includes
For each of the three drivers, replace the stub with a small
implementation that calls `ListModels` and returns its error:
```go
func (o *OllamaModel) CheckConnection(apiConfig *APIConfig) error {
_, err := o.ListModels(apiConfig)
return err
}
```
This is the exact pattern that xai, moonshot, deepseek, aliyun, and
gitee already use for the same method.
No JSON change. No factory change. No interface change.
### Type of change
- [x] Bug Fix (non-breaking change which fixes an issue)
### How was this tested?
- `go build ./internal/entity/models/...` in a clean go 1.25 image (the
go.mod minimum) returns exit 0.
- The full ModelDriver interface still resolves on each driver
(NewInstance, Name, ChatWithMessages, ChatStreamlyWithSender, Encode,
Rerank, ListModels, Balance, CheckConnection).
- Pattern parity with the existing xai, moonshot, deepseek, aliyun, and
gitee CheckConnection methods.
Closes #14609
2026-05-08 06:00:10 +02:00
|
|
|
// vLLM is a local provider and the API key is optional. Only set
|
|
|
|
|
// the Authorization header when a non-empty key was supplied. This
|
|
|
|
|
// also avoids a nil-pointer dereference on apiConfig or ApiKey.
|
|
|
|
|
if apiConfig != nil && apiConfig.ApiKey != nil && *apiConfig.ApiKey != "" {
|
|
|
|
|
req.Header.Set("Authorization", fmt.Sprintf("Bearer %s", *apiConfig.ApiKey))
|
|
|
|
|
}
|
2026-04-30 16:30:14 +08:00
|
|
|
|
|
|
|
|
resp, err := z.httpClient.Do(req)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to send request: %w", err)
|
|
|
|
|
}
|
|
|
|
|
defer resp.Body.Close()
|
|
|
|
|
|
|
|
|
|
body, err := io.ReadAll(resp.Body)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to read response: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if resp.StatusCode != http.StatusOK {
|
|
|
|
|
return nil, fmt.Errorf("API request failed with status %d: %s", resp.StatusCode, string(body))
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Parse response
|
|
|
|
|
var result map[string]interface{}
|
|
|
|
|
if err = json.Unmarshal(body, &result); err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to parse response: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// convert result["data"] to []map[string]interface{}
|
|
|
|
|
models := make([]string, 0)
|
|
|
|
|
for _, model := range result["data"].([]interface{}) {
|
|
|
|
|
modelMap := model.(map[string]interface{})
|
|
|
|
|
modelName := modelMap["id"].(string)
|
|
|
|
|
models = append(models, modelName)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return models, nil
|
2026-04-29 17:05:08 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (z *VllmModel) Balance(apiConfig *APIConfig) (map[string]interface{}, error) {
|
|
|
|
|
return nil, fmt.Errorf("no such method")
|
|
|
|
|
}
|
|
|
|
|
|
2026-05-08 15:54:27 +08:00
|
|
|
// CheckConnection verifies that the configured vLLM base URL is reachable
|
2026-04-29 17:05:08 +08:00
|
|
|
func (z *VllmModel) CheckConnection(apiConfig *APIConfig) error {
|
2026-05-08 15:54:27 +08:00
|
|
|
_, err := z.ListModels(apiConfig)
|
|
|
|
|
return err
|
2026-04-29 17:05:08 +08:00
|
|
|
}
|
|
|
|
|
|
2026-05-09 17:41:54 +08:00
|
|
|
// Rerank calculates similarity scores between query and documents
|
|
|
|
|
func (z *VllmModel) Rerank(modelName *string, query string, documents []string, apiConfig *APIConfig, rerankConfig *RerankConfig) (*RerankResponse, error) {
|
2026-04-29 17:05:08 +08:00
|
|
|
return nil, fmt.Errorf("%s, Rerank not implemented", z.Name())
|
|
|
|
|
}
|
2026-05-12 17:17:44 +08:00
|
|
|
|
|
|
|
|
// TranscribeAudio transcribe audio
|
|
|
|
|
func (o *VllmModel) TranscribeAudio(modelName *string, file *string, apiConfig *APIConfig, asrConfig *ASRConfig) (*ASRResponse, error) {
|
|
|
|
|
return nil, fmt.Errorf("%s, no such method", o.Name())
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (z *VllmModel) TranscribeAudioWithSender(modelName *string, file *string, apiConfig *APIConfig, asrConfig *ASRConfig, sender func(*string, *string) error) error {
|
|
|
|
|
return fmt.Errorf("%s, no such method", z.Name())
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// AudioSpeech convert audio to text
|
|
|
|
|
func (o *VllmModel) AudioSpeech(modelName *string, audioContent *string, apiConfig *APIConfig, asrConfig *TTSConfig) (*TTSResponse, error) {
|
|
|
|
|
return nil, fmt.Errorf("%s, no such method", o.Name())
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (z *VllmModel) AudioSpeechWithSender(modelName *string, audioContent *string, apiConfig *APIConfig, ttsConfig *TTSConfig, sender func(*string, *string) error) error {
|
|
|
|
|
return fmt.Errorf("%s, no such method", z.Name())
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// OCRFile OCR file
|
2026-05-13 17:29:53 +08:00
|
|
|
func (m *VllmModel) OCRFile(modelName *string, content []byte, url *string, apiConfig *APIConfig, ocrConfig *OCRConfig) (*OCRResponse, error) {
|
2026-05-12 17:17:44 +08:00
|
|
|
return nil, fmt.Errorf("%s, no such method", m.Name())
|
|
|
|
|
}
|