mirror of
https://github.com/infiniflow/ragflow.git
synced 2026-07-25 18:03:29 +08:00
## Summary Adds page-range parsing support to the Go-native pipeline path and introduces strict `parse_type` validation for both dataset and document update endpoints. ## What changed ### Pages range parsing - **`internal/utility/pdf_pages.go`** — `NormalizePDFPages`: normalizes raw page ranges (list of `[from,to]` 1-indexed inclusive ranges) into sorted, merged, deduplicated `[][]int`. Invalid ranges are dropped. - **`internal/ingestion/pipeline/pdf_pages.go`** — `NormalizeParserConfigPages`: walks any parser_config map and normalizes `"pages"` values under every component → filetype setup, so the persisted config always carries clean, merged ranges. - **`internal/deepdoc/parser/pdf/parser.go`** — integrates `resolvePagesToProcess` to filter parsed PDF pages by the configured ranges. - Pipeline integration (parser pages): `internal/parser/parser/pdf_parser_common.go`, `chunk_process.go`, plus associated e2e and unit tests. ### Parse type validation (shared logic) - **`internal/service/parser_mode.go`** (new) — `ValidateParseTypeMode`: shared function that validates `parse_type` (1=BuiltIn/parser_id, 2=Pipeline/pipeline_id) and ensures the corresponding field is present. Used by both dataset and document update endpoints. - **`internal/service/dataset/crud.go`** / `update.go` — replaces inline `isPipelineMode`/`isBuiltinMode` computation with the shared `service.ValidateParseTypeMode`. - **`internal/service/document/document_dataset_update.go`** — adds strict `parse_type` validation in `validateDatasetDocumentUpdate`, simplifies the reparse logic to a two-way switch (isBuiltin/isPipeline) now that parse_type is always valid. - **`internal/service/document/document.go`** — adds `ParseType` field to `UpdateDatasetDocumentRequest`. - **`internal/service/document/document_dataset_update.go`** — `updateDocumentParserConfig` fallback path when DSL loading fails. - **`internal/service/parser_mode_test.go`** (new) — test coverage for nil, invalid, and missing-field scenarios. ### Frontend - **`web/src/interfaces/request/document.ts`** — adds `parseType` to `IChangeParserRequestBody`. - **`web/src/hooks/use-document-request.ts`** — `useSetDocumentPipelineParser` sends `parse_type` in the PATCH payload. - **`web/src/pages/dataset/dataset/use-change-document-parser.ts`** — Go/Python branching for the document parser config dialog. - **`web/src/components/document-pipeline-dialog/use-document-pipeline-form.ts`** — `buildSubmitData` returns `parseType` (bugfix: was dropped from the return value). ### Test changes - **Removed**: 2 tests that verified the old "mutually exclusive" error (replaced by `ValidateParseTypeMode` coverage). - **Modified**: 6 tests across document and dataset packages to include `ParseType` in request structs. - **Added**: new e2e tests for pages parsing (`pages_e2e_test.go`, `pdf_parser_pages_e2e_test.go`) and unit tests for `NormalizePDFPages`, `NormalizeParserConfigPages`, `resolvePagesToProcess`. ## Backward compatibility - The `parse_type` field is **required** when `parser_id` or `pipeline_id` is sent. This changes the contract for both dataset and document PATCH endpoints, but aligns the Go backend with the existing frontend behavior (the frontend already sends `parse_type`). Callers that omit `parse_type` when updating parser/pipeline selections will receive a clear error message. - Existing callers that only update fields like `name`, `enabled`, or `meta_fields` are unaffected. - Test updates ensure all known call sites are compliant.
137 lines
4.3 KiB
Go
137 lines
4.3 KiB
Go
package service
|
|
|
|
import (
|
|
"encoding/json"
|
|
"errors"
|
|
"fmt"
|
|
"strings"
|
|
|
|
"ragflow/internal/dao"
|
|
"ragflow/internal/entity"
|
|
pipelinepkg "ragflow/internal/ingestion/pipeline"
|
|
)
|
|
|
|
// loadCanvasDSLJSON returns the DSL JSON for a custom canvas pipeline. The
|
|
// canvas's dsl column holds the same component-graph structure that built-in
|
|
// templates use, so it can be validated by the same schema extractor. It is a
|
|
// package-level function so both document and knowledge-base updates reuse it.
|
|
func loadCanvasDSLJSON(canvasID string) ([]byte, error) {
|
|
if strings.TrimSpace(canvasID) == "" {
|
|
return nil, fmt.Errorf("empty canvas id")
|
|
}
|
|
canvas, err := dao.NewUserCanvasDAO().GetByID(canvasID)
|
|
if err != nil {
|
|
if errors.Is(err, dao.ErrUserCanvasNotFound) {
|
|
return nil, fmt.Errorf("canvas %s not found", canvasID)
|
|
}
|
|
return nil, fmt.Errorf("load canvas %s: %w", canvasID, err)
|
|
}
|
|
if len(canvas.DSL) == 0 {
|
|
return nil, fmt.Errorf("canvas %s has no DSL", canvasID)
|
|
}
|
|
raw, err := json.Marshal(canvas.DSL)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("marshal canvas %s DSL: %w", canvasID, err)
|
|
}
|
|
return raw, nil
|
|
}
|
|
|
|
// LoadPipelineDSL loads the DSL JSON for a pipeline identified by parserID
|
|
// (built-in) or pipelineID (custom canvas). When both are provided, isPipeline
|
|
// selects which one to use.
|
|
func LoadPipelineDSL(isPipeline bool, parserID string, pipelineID *string) ([]byte, error) {
|
|
if isPipeline {
|
|
return loadCanvasDSLJSON(strings.TrimSpace(*pipelineID))
|
|
}
|
|
registry, err := pipelinepkg.DefaultRegistry()
|
|
if err != nil {
|
|
return nil, fmt.Errorf("builtin pipeline registry: %w", err)
|
|
}
|
|
if !registry.IsValid(parserID) {
|
|
return nil, fmt.Errorf("unknown builtin parser_id: %s", parserID)
|
|
}
|
|
dslStr, err := pipelinepkg.LoadBuiltinDSL(parserID)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("load builtin DSL for %q: %w", parserID, err)
|
|
}
|
|
return []byte(dslStr), nil
|
|
}
|
|
|
|
// ResolveComponentParamsDefaults loads the DSL for the target pipeline and
|
|
// returns the component params defaults as an entity.JSONMap {cpnID: {param: value}}.
|
|
// For builtin templates the DSL is loaded from the embedded registry; for custom
|
|
// canvas pipelines it is loaded from the canvas row in the database.
|
|
func ResolveComponentParamsDefaults(parserID string, pipelineID *string) (entity.JSONMap, error) {
|
|
isPipeline := pipelineID != nil && strings.TrimSpace(*pipelineID) != ""
|
|
var cp map[string]map[string]any
|
|
var err error
|
|
if isPipeline {
|
|
dslJSON, lerr := loadCanvasDSLJSON(strings.TrimSpace(*pipelineID))
|
|
if lerr != nil {
|
|
return nil, fmt.Errorf("load canvas DSL: %w", lerr)
|
|
}
|
|
cp, err = pipelinepkg.ComponentParamsDefaults(dslJSON)
|
|
} else {
|
|
registry, regErr := pipelinepkg.DefaultRegistry()
|
|
if regErr != nil {
|
|
return nil, fmt.Errorf("builtin registry: %w", regErr)
|
|
}
|
|
if !registry.IsValid(parserID) {
|
|
return nil, fmt.Errorf("unknown builtin parser_id: %q", parserID)
|
|
}
|
|
dslStr, dslErr := pipelinepkg.LoadBuiltinDSL(parserID)
|
|
if dslErr != nil {
|
|
return nil, fmt.Errorf("load builtin DSL: %w", dslErr)
|
|
}
|
|
cp, err = pipelinepkg.ComponentParamsDefaults([]byte(dslStr))
|
|
}
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
out := make(entity.JSONMap, len(cp))
|
|
for k, v := range cp {
|
|
out[k] = v
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// ValidateDatasetEmbeddingModels checks that all knowledge bases in the list
|
|
// either have an embedding model or none do, and that they all use the same model.
|
|
func ValidateDatasetEmbeddingModels(kbs []*entity.Knowledgebase) error {
|
|
embdIDs := make(map[string]struct{})
|
|
hasEmbd := false
|
|
noEmbd := false
|
|
for _, kb := range kbs {
|
|
if kb.EmbdID != "" {
|
|
hasEmbd = true
|
|
baseName := kb.EmbdID
|
|
if idx := strings.LastIndex(kb.EmbdID, "@"); idx > 0 {
|
|
baseName = kb.EmbdID[:idx]
|
|
// Strip the second-to-last @-segment too (instance name),
|
|
// matching Python's _base_model_name which uses rsplit("@", 2).
|
|
if idx2 := strings.LastIndex(baseName, "@"); idx2 > 0 {
|
|
baseName = baseName[:idx2]
|
|
}
|
|
}
|
|
embdIDs[baseName] = struct{}{}
|
|
} else {
|
|
noEmbd = true
|
|
}
|
|
}
|
|
if hasEmbd && noEmbd {
|
|
return fmt.Errorf("Cannot search across datasets where some have embedding models and others do not.")
|
|
}
|
|
if len(embdIDs) > 1 {
|
|
return fmt.Errorf("Datasets use different embedding models: %v", getEmbdIDs(kbs))
|
|
}
|
|
return nil
|
|
}
|
|
|
|
func getEmbdIDs(kbs []*entity.Knowledgebase) []string {
|
|
ids := make([]string, len(kbs))
|
|
for i, kb := range kbs {
|
|
ids[i] = kb.EmbdID
|
|
}
|
|
return ids
|
|
}
|