mirror of
https://github.com/infiniflow/ragflow.git
synced 2026-07-25 01:43:27 +08:00
## Summary Adds page-range parsing support to the Go-native pipeline path and introduces strict `parse_type` validation for both dataset and document update endpoints. ## What changed ### Pages range parsing - **`internal/utility/pdf_pages.go`** — `NormalizePDFPages`: normalizes raw page ranges (list of `[from,to]` 1-indexed inclusive ranges) into sorted, merged, deduplicated `[][]int`. Invalid ranges are dropped. - **`internal/ingestion/pipeline/pdf_pages.go`** — `NormalizeParserConfigPages`: walks any parser_config map and normalizes `"pages"` values under every component → filetype setup, so the persisted config always carries clean, merged ranges. - **`internal/deepdoc/parser/pdf/parser.go`** — integrates `resolvePagesToProcess` to filter parsed PDF pages by the configured ranges. - Pipeline integration (parser pages): `internal/parser/parser/pdf_parser_common.go`, `chunk_process.go`, plus associated e2e and unit tests. ### Parse type validation (shared logic) - **`internal/service/parser_mode.go`** (new) — `ValidateParseTypeMode`: shared function that validates `parse_type` (1=BuiltIn/parser_id, 2=Pipeline/pipeline_id) and ensures the corresponding field is present. Used by both dataset and document update endpoints. - **`internal/service/dataset/crud.go`** / `update.go` — replaces inline `isPipelineMode`/`isBuiltinMode` computation with the shared `service.ValidateParseTypeMode`. - **`internal/service/document/document_dataset_update.go`** — adds strict `parse_type` validation in `validateDatasetDocumentUpdate`, simplifies the reparse logic to a two-way switch (isBuiltin/isPipeline) now that parse_type is always valid. - **`internal/service/document/document.go`** — adds `ParseType` field to `UpdateDatasetDocumentRequest`. - **`internal/service/document/document_dataset_update.go`** — `updateDocumentParserConfig` fallback path when DSL loading fails. - **`internal/service/parser_mode_test.go`** (new) — test coverage for nil, invalid, and missing-field scenarios. ### Frontend - **`web/src/interfaces/request/document.ts`** — adds `parseType` to `IChangeParserRequestBody`. - **`web/src/hooks/use-document-request.ts`** — `useSetDocumentPipelineParser` sends `parse_type` in the PATCH payload. - **`web/src/pages/dataset/dataset/use-change-document-parser.ts`** — Go/Python branching for the document parser config dialog. - **`web/src/components/document-pipeline-dialog/use-document-pipeline-form.ts`** — `buildSubmitData` returns `parseType` (bugfix: was dropped from the return value). ### Test changes - **Removed**: 2 tests that verified the old "mutually exclusive" error (replaced by `ValidateParseTypeMode` coverage). - **Modified**: 6 tests across document and dataset packages to include `ParseType` in request structs. - **Added**: new e2e tests for pages parsing (`pages_e2e_test.go`, `pdf_parser_pages_e2e_test.go`) and unit tests for `NormalizePDFPages`, `NormalizeParserConfigPages`, `resolvePagesToProcess`. ## Backward compatibility - The `parse_type` field is **required** when `parser_id` or `pipeline_id` is sent. This changes the contract for both dataset and document PATCH endpoints, but aligns the Go backend with the existing frontend behavior (the frontend already sends `parse_type`). Callers that omit `parse_type` when updating parser/pipeline selections will receive a clear error message. - Existing callers that only update fields like `name`, `enabled`, or `meta_fields` are unaffected. - Test updates ensure all known call sites are compliant.
103 lines
2.8 KiB
Go
103 lines
2.8 KiB
Go
package utility
|
|
|
|
import (
|
|
"fmt"
|
|
"sort"
|
|
)
|
|
|
|
// NormalizePDFPages normalizes a raw "pages" value (list[list[int]], 1-indexed
|
|
// inclusive ranges, JSON-decoded as []any of []any of float64) into a sorted,
|
|
// merged, de-duplicated [][]int.
|
|
//
|
|
// Semantics (fail-fast):
|
|
// - nil or empty list → (nil, nil): "no value", callers treat as "parse all
|
|
// pages".
|
|
// - Any invalid range → (nil, error): the whole input is rejected; callers
|
|
// should surface the error and abort the request. No partial dropping.
|
|
// - All ranges valid → (normalized, nil).
|
|
//
|
|
// Validation per range [from, to]:
|
|
// - both values must be integers (int, int64, or integral float64);
|
|
// - from >= 1 (1-indexed);
|
|
// - from <= to.
|
|
//
|
|
// Surviving ranges are sorted by `from` then merged when overlapping or
|
|
// adjacent (next.from <= cur.to + 1).
|
|
func NormalizePDFPages(raw any) ([][]int, error) {
|
|
list, ok := raw.([]any)
|
|
if !ok {
|
|
// nil raw (no key / JSON null) is "no value"; a non-list raw is a type
|
|
// error. raw==nil falls through the !ok branch because nil does not
|
|
// satisfy []any.
|
|
if raw == nil {
|
|
return nil, nil
|
|
}
|
|
return nil, fmt.Errorf("pages must be a list of [from,to] ranges, got %T", raw)
|
|
}
|
|
if len(list) == 0 {
|
|
return nil, nil
|
|
}
|
|
|
|
ranges := make([][]int, 0, len(list))
|
|
for _, item := range list {
|
|
pair, ok := item.([]any)
|
|
if !ok || len(pair) != 2 {
|
|
return nil, fmt.Errorf("invalid page range %v: must be a [from,to] pair", item)
|
|
}
|
|
from, ok := toInt(pair[0])
|
|
if !ok {
|
|
return nil, fmt.Errorf("invalid page range [%v,%v]: from must be an integer", pair[0], pair[1])
|
|
}
|
|
to, ok := toInt(pair[1])
|
|
if !ok {
|
|
return nil, fmt.Errorf("invalid page range [%v,%v]: to must be an integer", pair[0], pair[1])
|
|
}
|
|
if from < 1 {
|
|
return nil, fmt.Errorf("invalid page range [%d,%d]: from must be >= 1", from, to)
|
|
}
|
|
if from > to {
|
|
return nil, fmt.Errorf("invalid page range [%d,%d]: from must be <= to", from, to)
|
|
}
|
|
ranges = append(ranges, []int{from, to})
|
|
}
|
|
|
|
sort.Slice(ranges, func(i, j int) bool {
|
|
if ranges[i][0] != ranges[j][0] {
|
|
return ranges[i][0] < ranges[j][0]
|
|
}
|
|
return ranges[i][1] < ranges[j][1]
|
|
})
|
|
|
|
merged := ranges[:1]
|
|
for _, r := range ranges[1:] {
|
|
last := merged[len(merged)-1]
|
|
if r[0] <= last[1]+1 { // overlap or adjacent
|
|
if r[1] > last[1] {
|
|
last[1] = r[1]
|
|
}
|
|
} else {
|
|
merged = append(merged, r)
|
|
}
|
|
}
|
|
return merged, nil
|
|
}
|
|
|
|
// toInt coerces a JSON-decoded numeric value to int. Accepts int, int64, and
|
|
// integral float64; rejects non-numeric or non-integral values.
|
|
func toInt(v any) (int, bool) {
|
|
switch x := v.(type) {
|
|
case int:
|
|
return x, true
|
|
case int64:
|
|
return int(x), true
|
|
case float64:
|
|
// Reject non-integral floats (e.g. 1.5) — pages must be integers.
|
|
if x != float64(int(x)) {
|
|
return 0, false
|
|
}
|
|
return int(x), true
|
|
default:
|
|
return 0, false
|
|
}
|
|
}
|