Files
ragflow/internal/deepdoc/parser/pdf/text_dump_test.go

92 lines
2.1 KiB
Go
Raw Normal View History

//go:build cgo && manual
package pdf
import (
"context"
"os"
"path/filepath"
feat: parser pages range and parse type validation for dataset/document (#17293) ## Summary Adds page-range parsing support to the Go-native pipeline path and introduces strict `parse_type` validation for both dataset and document update endpoints. ## What changed ### Pages range parsing - **`internal/utility/pdf_pages.go`** — `NormalizePDFPages`: normalizes raw page ranges (list of `[from,to]` 1-indexed inclusive ranges) into sorted, merged, deduplicated `[][]int`. Invalid ranges are dropped. - **`internal/ingestion/pipeline/pdf_pages.go`** — `NormalizeParserConfigPages`: walks any parser_config map and normalizes `"pages"` values under every component → filetype setup, so the persisted config always carries clean, merged ranges. - **`internal/deepdoc/parser/pdf/parser.go`** — integrates `resolvePagesToProcess` to filter parsed PDF pages by the configured ranges. - Pipeline integration (parser pages): `internal/parser/parser/pdf_parser_common.go`, `chunk_process.go`, plus associated e2e and unit tests. ### Parse type validation (shared logic) - **`internal/service/parser_mode.go`** (new) — `ValidateParseTypeMode`: shared function that validates `parse_type` (1=BuiltIn/parser_id, 2=Pipeline/pipeline_id) and ensures the corresponding field is present. Used by both dataset and document update endpoints. - **`internal/service/dataset/crud.go`** / `update.go` — replaces inline `isPipelineMode`/`isBuiltinMode` computation with the shared `service.ValidateParseTypeMode`. - **`internal/service/document/document_dataset_update.go`** — adds strict `parse_type` validation in `validateDatasetDocumentUpdate`, simplifies the reparse logic to a two-way switch (isBuiltin/isPipeline) now that parse_type is always valid. - **`internal/service/document/document.go`** — adds `ParseType` field to `UpdateDatasetDocumentRequest`. - **`internal/service/document/document_dataset_update.go`** — `updateDocumentParserConfig` fallback path when DSL loading fails. - **`internal/service/parser_mode_test.go`** (new) — test coverage for nil, invalid, and missing-field scenarios. ### Frontend - **`web/src/interfaces/request/document.ts`** — adds `parseType` to `IChangeParserRequestBody`. - **`web/src/hooks/use-document-request.ts`** — `useSetDocumentPipelineParser` sends `parse_type` in the PATCH payload. - **`web/src/pages/dataset/dataset/use-change-document-parser.ts`** — Go/Python branching for the document parser config dialog. - **`web/src/components/document-pipeline-dialog/use-document-pipeline-form.ts`** — `buildSubmitData` returns `parseType` (bugfix: was dropped from the return value). ### Test changes - **Removed**: 2 tests that verified the old "mutually exclusive" error (replaced by `ValidateParseTypeMode` coverage). - **Modified**: 6 tests across document and dataset packages to include `ParseType` in request structs. - **Added**: new e2e tests for pages parsing (`pages_e2e_test.go`, `pdf_parser_pages_e2e_test.go`) and unit tests for `NormalizePDFPages`, `NormalizeParserConfigPages`, `resolvePagesToProcess`. ## Backward compatibility - The `parse_type` field is **required** when `parser_id` or `pipeline_id` is sent. This changes the contract for both dataset and document PATCH endpoints, but aligns the Go backend with the existing frontend behavior (the frontend already sends `parse_type`). Callers that omit `parse_type` when updating parser/pipeline selections will receive a clear error message. - Existing callers that only update fields like `name`, `enabled`, or `meta_fields` are unaffected. - Test updates ensure all known call sites are compliant.
2026-07-23 19:57:27 +08:00
"ragflow/internal/common"
pdf "ragflow/internal/deepdoc/parser/pdf/type"
"strings"
"testing"
)
// TestDumpTextOutput runs Parse on real PDFs and saves per-PDF text
// to testdata/output/go/noocr/text/{pdf}.txt. Set DUMP_COUNT env to limit first N PDFs.
func TestDumpTextOutput(t *testing.T) {
pdfDir := filepath.Join("testdata", "real_pdfs")
outDir := filepath.Join("testdata", "output", "go", "noocr", "text")
os.MkdirAll(outDir, 0755)
entries, err := os.ReadDir(pdfDir)
if err != nil {
t.Fatal(err)
}
count := len(entries)
if n := common.GetEnv(common.EnvDumpCount); n != "" {
c := 0
for _, ch := range n {
c = c*10 + int(ch-'0')
}
if c > 0 && c < count {
count = c
}
}
totalChars := 0
for i, e := range entries {
if i >= count {
break
}
if e.IsDir() || !strings.HasSuffix(strings.ToLower(e.Name()), ".pdf") {
continue
}
name := e.Name()
outPath := filepath.Join(outDir, name+".txt")
if _, err := os.Stat(outPath); err == nil {
data, _ := os.ReadFile(outPath)
n := len(data)
totalChars += n
t.Logf("[%d/%d] %s — SKIP (%d chars)", i+1, count, name, n)
continue
}
pdfPath := filepath.Join(pdfDir, name)
data, err := os.ReadFile(pdfPath)
if err != nil {
t.Logf("[%d/%d] %s — read error: %v", i+1, count, name, err)
continue
}
eng, err := NewEngine(data)
if err != nil {
t.Logf("[%d/%d] %s — engine error: %v", i+1, count, name, err)
continue
}
cfg := pdf.DefaultParserConfig()
p := NewParser(cfg)
result, err := p.ParseRaw(context.Background(), eng, &MockDocAnalyzer{Healthy: true})
eng.Close()
if err != nil {
t.Logf("[%d/%d] %s — parse error: %v", i+1, count, name, err)
continue
}
var sb strings.Builder
for _, s := range result.Sections {
sb.WriteString(s.Text)
sb.WriteByte('\n')
}
text := sb.String()
os.WriteFile(outPath, []byte(text), 0644)
totalChars += len(text)
t.Logf("[%d/%d] %s — %d chars", i+1, count, name, len(text))
}
t.Logf("Done. %d chars total. Output: %s/", totalChars, outDir)
}