// // Copyright 2026 The InfiniFlow Authors. All Rights Reserved. // // Licensed under the Apache License, Version 2.0 (the "License"); // you may not use this file except in compliance with the License. // You may obtain a copy of the License at // // http://www.apache.org/licenses/LICENSE-2.0 // // Unless required by applicable law or agreed to in writing, software // distributed under the License is distributed on an "AS IS" BASIS, // WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. // See the License for the specific language governing permissions and // limitations under the License. package component import ( "context" "ragflow/internal/dao" "sync" "testing" "ragflow/internal/entity" modelModule "ragflow/internal/entity/models" "ragflow/internal/utility" "gorm.io/gorm" ) // docxVisionFakeDriver satisfies modelModule.ModelDriver but never reaches the // network: docxVisionCaptureInvoker intercepts the call before the driver. type docxVisionFakeDriver struct { modelModule.ModelDriver } // docxVisionCaptureInvoker records the requested image and returns a fixed // description, mirroring the markdown-vision test's capture driver. type docxVisionCaptureInvoker struct { mu sync.Mutex images []string captured []modelModule.Message } func (c *docxVisionCaptureInvoker) invoke( ctx context.Context, driver modelModule.ModelDriver, modelName string, messages []modelModule.Message, apiConfig *modelModule.APIConfig, ) (*modelModule.ChatResponse, error) { c.mu.Lock() c.captured = append(c.captured, messages...) // Pull the data URI out of the second content part. if parts, ok := messages[0].Content.([]interface{}); ok && len(parts) >= 2 { if img, ok := parts[1].(map[string]any); ok { if url, ok := img["image_url"].(map[string]any); ok { if u, ok := url["url"].(string); ok { c.images = append(c.images, u) } } } } c.mu.Unlock() ans := "a diagram of a pipeline" return &modelModule.ChatResponse{Answer: &ans}, nil } // TestMaybeDispatchDOCXVision_EnhancesJSONImages verifies Diff 2.4: DOCX vision // enhancement must trigger on the JSON output path (like Python's // enhance_media_sections_with_vision in parser.py:_doc) and must NOT trigger on // the Markdown path. Image items with a non-empty `image` field get their VLM // description appended to `text`; table items (no image) and text items are // left untouched. func TestMaybeDispatchDOCXVision_EnhancesJSONImages(t *testing.T) { origResolver := resolveTenantModelByType origInvoker := visionChatInvoker origPrompt := docxVisionPromptBuilder defer func() { resolveTenantModelByType = origResolver visionChatInvoker = origInvoker docxVisionPromptBuilder = origPrompt }() resolveTenantModelByType = func(ctx context.Context, db *gorm.DB, tenantID string, modelType entity.ModelType) (modelModule.ModelDriver, string, *modelModule.APIConfig, int, error) { return &docxVisionFakeDriver{}, "docx-vision-model", &modelModule.APIConfig{}, 0, nil } invoker := &docxVisionCaptureInvoker{} visionChatInvoker = invoker.invoke docxVisionPromptBuilder = func(string, string) (string, error) { return "describe the figure", nil } dispatched := parserDispatchResult{ OutputFormat: "json", DocType: "docx", JSON: []map[string]any{ {"text": "Intro paragraph", "image": nil, "doc_type_kwd": "text"}, {"text": "", "image": "aGVsbG8taW1hZ2U=", "doc_type_kwd": "image"}, {"text": "