mirror of
https://github.com/infiniflow/ragflow.git
synced 2026-07-25 18:03:29 +08:00
## Summary Adds page-range parsing support to the Go-native pipeline path and introduces strict `parse_type` validation for both dataset and document update endpoints. ## What changed ### Pages range parsing - **`internal/utility/pdf_pages.go`** — `NormalizePDFPages`: normalizes raw page ranges (list of `[from,to]` 1-indexed inclusive ranges) into sorted, merged, deduplicated `[][]int`. Invalid ranges are dropped. - **`internal/ingestion/pipeline/pdf_pages.go`** — `NormalizeParserConfigPages`: walks any parser_config map and normalizes `"pages"` values under every component → filetype setup, so the persisted config always carries clean, merged ranges. - **`internal/deepdoc/parser/pdf/parser.go`** — integrates `resolvePagesToProcess` to filter parsed PDF pages by the configured ranges. - Pipeline integration (parser pages): `internal/parser/parser/pdf_parser_common.go`, `chunk_process.go`, plus associated e2e and unit tests. ### Parse type validation (shared logic) - **`internal/service/parser_mode.go`** (new) — `ValidateParseTypeMode`: shared function that validates `parse_type` (1=BuiltIn/parser_id, 2=Pipeline/pipeline_id) and ensures the corresponding field is present. Used by both dataset and document update endpoints. - **`internal/service/dataset/crud.go`** / `update.go` — replaces inline `isPipelineMode`/`isBuiltinMode` computation with the shared `service.ValidateParseTypeMode`. - **`internal/service/document/document_dataset_update.go`** — adds strict `parse_type` validation in `validateDatasetDocumentUpdate`, simplifies the reparse logic to a two-way switch (isBuiltin/isPipeline) now that parse_type is always valid. - **`internal/service/document/document.go`** — adds `ParseType` field to `UpdateDatasetDocumentRequest`. - **`internal/service/document/document_dataset_update.go`** — `updateDocumentParserConfig` fallback path when DSL loading fails. - **`internal/service/parser_mode_test.go`** (new) — test coverage for nil, invalid, and missing-field scenarios. ### Frontend - **`web/src/interfaces/request/document.ts`** — adds `parseType` to `IChangeParserRequestBody`. - **`web/src/hooks/use-document-request.ts`** — `useSetDocumentPipelineParser` sends `parse_type` in the PATCH payload. - **`web/src/pages/dataset/dataset/use-change-document-parser.ts`** — Go/Python branching for the document parser config dialog. - **`web/src/components/document-pipeline-dialog/use-document-pipeline-form.ts`** — `buildSubmitData` returns `parseType` (bugfix: was dropped from the return value). ### Test changes - **Removed**: 2 tests that verified the old "mutually exclusive" error (replaced by `ValidateParseTypeMode` coverage). - **Modified**: 6 tests across document and dataset packages to include `ParseType` in request structs. - **Added**: new e2e tests for pages parsing (`pages_e2e_test.go`, `pdf_parser_pages_e2e_test.go`) and unit tests for `NormalizePDFPages`, `NormalizeParserConfigPages`, `resolvePagesToProcess`. ## Backward compatibility - The `parse_type` field is **required** when `parser_id` or `pipeline_id` is sent. This changes the contract for both dataset and document PATCH endpoints, but aligns the Go backend with the existing frontend behavior (the frontend already sends `parse_type`). Callers that omit `parse_type` when updating parser/pipeline selections will receive a clear error message. - Existing callers that only update fields like `name`, `enabled`, or `meta_fields` are unaffected. - Test updates ensure all known call sites are compliant.
120 lines
4.1 KiB
Go
120 lines
4.1 KiB
Go
package pdf
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"reflect"
|
|
"strings"
|
|
"testing"
|
|
|
|
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
|
"ragflow/internal/utility"
|
|
)
|
|
|
|
// TestPagesEndToEnd_NormalizeThenParse verifies the full pages path: a
|
|
// frontend-submitted (JSON-decoded, possibly dirty) pages value is normalized
|
|
// by utility.NormalizePDFPages, fed into ParserConfig.Pages, and the deepdoc
|
|
// parser only processes the resulting page ranges.
|
|
//
|
|
// This ties together step 3 (API-layer normalization) and step 1 (deepdoc
|
|
// page filtering) end-to-end at the parser level.
|
|
func TestPagesEndToEnd_NormalizeThenParse(t *testing.T) {
|
|
// jsonPages builds a JSON-decoded-style []any of []any of float64, mimicking
|
|
// what arrives over the wire from the frontend.
|
|
jsonPages := func(ranges ...[2]float64) any {
|
|
out := make([]any, 0, len(ranges))
|
|
for _, r := range ranges {
|
|
out = append(out, []any{r[0], r[1]})
|
|
}
|
|
return out
|
|
}
|
|
|
|
t.Run("overlapping and unsorted ranges normalized then filtered", func(t *testing.T) {
|
|
// Frontend submitted [1,3],[2,5],[8,10] (overlap 1-3 & 2-5, unsorted).
|
|
raw := jsonPages([2]float64{1, 3}, [2]float64{2, 5}, [2]float64{8, 10})
|
|
normalized, err := utility.NormalizePDFPages(raw) // -> [[1,5],[8,10]]
|
|
if err != nil {
|
|
t.Fatalf("NormalizePDFPages: %v", err)
|
|
}
|
|
|
|
wantNorm := [][]int{{1, 5}, {8, 10}}
|
|
if !reflect.DeepEqual(normalized, wantNorm) {
|
|
t.Fatalf("NormalizePDFPages = %v, want %v", normalized, wantNorm)
|
|
}
|
|
|
|
cfg := pdf.DefaultParserConfig()
|
|
cfg.Pages = normalized
|
|
p := NewParser(cfg)
|
|
|
|
eng := makePageTaggedEngine(10)
|
|
result, err := p.ParseRaw(context.Background(), eng, &MockDocAnalyzer{Healthy: true})
|
|
if err != nil {
|
|
t.Fatalf("ParseRaw: %v", err)
|
|
}
|
|
|
|
// [1,5] -> 0-based 0..4 ; [8,10] -> 7..9
|
|
wantPages := map[int]struct{}{0: {}, 1: {}, 2: {}, 3: {}, 4: {}, 7: {}, 8: {}, 9: {}}
|
|
if got := pageHeightKeys(result.PageHeight); !reflect.DeepEqual(got, wantPages) {
|
|
t.Errorf("PageHeight keys = %v, want %v", got, wantPages)
|
|
}
|
|
|
|
// Sections must not carry text from skipped pages (5, 6).
|
|
combined := combineSectionText(result.Sections)
|
|
for _, pg := range []int{5, 6} {
|
|
if strings.Contains(combined, fmt.Sprintf("p%d", pg)) {
|
|
t.Errorf("expected p%d to be skipped; combined=%q", pg, combined)
|
|
}
|
|
}
|
|
})
|
|
|
|
t.Run("invalid range rejected during normalization (fail-fast)", func(t *testing.T) {
|
|
// [1,3],[3,1](from>to),[8,10] -> fail-fast rejects the whole input.
|
|
raw := jsonPages([2]float64{1, 3}, [2]float64{3, 1}, [2]float64{8, 10})
|
|
normalized, err := utility.NormalizePDFPages(raw)
|
|
if err == nil {
|
|
t.Fatalf("NormalizePDFPages(%v) expected error, got %v", raw, normalized)
|
|
}
|
|
if normalized != nil {
|
|
t.Fatalf("NormalizePDFPages(%v) = %v, want nil on error", raw, normalized)
|
|
}
|
|
})
|
|
|
|
t.Run("all invalid -> error (fail-fast, no parse-all fallback)", func(t *testing.T) {
|
|
// [3,1],[0,2] both invalid -> error. The request layer rejects this
|
|
// before it ever reaches the parser; normalize surfaces the error
|
|
// rather than silently degrading to "parse all pages".
|
|
raw := jsonPages([2]float64{3, 1}, [2]float64{0, 2})
|
|
normalized, err := utility.NormalizePDFPages(raw)
|
|
if err == nil {
|
|
t.Fatalf("NormalizePDFPages(%v) expected error, got %v", raw, normalized)
|
|
}
|
|
if normalized != nil {
|
|
t.Fatalf("NormalizePDFPages(%v) = %v, want nil on error", raw, normalized)
|
|
}
|
|
})
|
|
|
|
t.Run("range clamped to document page count", func(t *testing.T) {
|
|
// [1,1000000] on a 5-page doc -> clamped to all 5 pages.
|
|
raw := jsonPages([2]float64{1, 1000000})
|
|
normalized, err := utility.NormalizePDFPages(raw)
|
|
if err != nil {
|
|
t.Fatalf("NormalizePDFPages: %v", err)
|
|
}
|
|
|
|
cfg := pdf.DefaultParserConfig()
|
|
cfg.Pages = normalized
|
|
p := NewParser(cfg)
|
|
|
|
eng := makePageTaggedEngine(5)
|
|
result, err := p.ParseRaw(context.Background(), eng, &MockDocAnalyzer{Healthy: true})
|
|
if err != nil {
|
|
t.Fatalf("ParseRaw: %v", err)
|
|
}
|
|
|
|
wantPages := map[int]struct{}{0: {}, 1: {}, 2: {}, 3: {}, 4: {}}
|
|
if got := pageHeightKeys(result.PageHeight); !reflect.DeepEqual(got, wantPages) {
|
|
t.Errorf("PageHeight keys = %v, want all 5 pages", got)
|
|
}
|
|
})
|
|
}
|