2026-06-25 20:16:16 +08:00
|
|
|
|
//go:build cgo && manual
|
|
|
|
|
|
|
2026-07-02 09:46:33 +08:00
|
|
|
|
package pdf
|
2026-06-25 20:16:16 +08:00
|
|
|
|
|
|
|
|
|
|
import (
|
|
|
|
|
|
"context"
|
|
|
|
|
|
"os"
|
|
|
|
|
|
"path/filepath"
|
|
|
|
|
|
"sort"
|
|
|
|
|
|
"strings"
|
|
|
|
|
|
"testing"
|
2026-06-29 18:46:41 +08:00
|
|
|
|
|
feat: parser pages range and parse type validation for dataset/document (#17293)
## Summary
Adds page-range parsing support to the Go-native pipeline path and
introduces strict `parse_type` validation for both dataset and document
update endpoints.
## What changed
### Pages range parsing
- **`internal/utility/pdf_pages.go`** — `NormalizePDFPages`: normalizes
raw page ranges (list of `[from,to]` 1-indexed inclusive ranges) into
sorted, merged, deduplicated `[][]int`. Invalid ranges are dropped.
- **`internal/ingestion/pipeline/pdf_pages.go`** —
`NormalizeParserConfigPages`: walks any parser_config map and normalizes
`"pages"` values under every component → filetype setup, so the
persisted config always carries clean, merged ranges.
- **`internal/deepdoc/parser/pdf/parser.go`** — integrates
`resolvePagesToProcess` to filter parsed PDF pages by the configured
ranges.
- Pipeline integration (parser pages):
`internal/parser/parser/pdf_parser_common.go`, `chunk_process.go`, plus
associated e2e and unit tests.
### Parse type validation (shared logic)
- **`internal/service/parser_mode.go`** (new) — `ValidateParseTypeMode`:
shared function that validates `parse_type` (1=BuiltIn/parser_id,
2=Pipeline/pipeline_id) and ensures the corresponding field is present.
Used by both dataset and document update endpoints.
- **`internal/service/dataset/crud.go`** / `update.go` — replaces inline
`isPipelineMode`/`isBuiltinMode` computation with the shared
`service.ValidateParseTypeMode`.
- **`internal/service/document/document_dataset_update.go`** — adds
strict `parse_type` validation in `validateDatasetDocumentUpdate`,
simplifies the reparse logic to a two-way switch (isBuiltin/isPipeline)
now that parse_type is always valid.
- **`internal/service/document/document.go`** — adds `ParseType` field
to `UpdateDatasetDocumentRequest`.
- **`internal/service/document/document_dataset_update.go`** —
`updateDocumentParserConfig` fallback path when DSL loading fails.
- **`internal/service/parser_mode_test.go`** (new) — test coverage for
nil, invalid, and missing-field scenarios.
### Frontend
- **`web/src/interfaces/request/document.ts`** — adds `parseType` to
`IChangeParserRequestBody`.
- **`web/src/hooks/use-document-request.ts`** —
`useSetDocumentPipelineParser` sends `parse_type` in the PATCH payload.
- **`web/src/pages/dataset/dataset/use-change-document-parser.ts`** —
Go/Python branching for the document parser config dialog.
-
**`web/src/components/document-pipeline-dialog/use-document-pipeline-form.ts`**
— `buildSubmitData` returns `parseType` (bugfix: was dropped from the
return value).
### Test changes
- **Removed**: 2 tests that verified the old "mutually exclusive" error
(replaced by `ValidateParseTypeMode` coverage).
- **Modified**: 6 tests across document and dataset packages to include
`ParseType` in request structs.
- **Added**: new e2e tests for pages parsing (`pages_e2e_test.go`,
`pdf_parser_pages_e2e_test.go`) and unit tests for `NormalizePDFPages`,
`NormalizeParserConfigPages`, `resolvePagesToProcess`.
## Backward compatibility
- The `parse_type` field is **required** when `parser_id` or
`pipeline_id` is sent. This changes the contract for both dataset and
document PATCH endpoints, but aligns the Go backend with the existing
frontend behavior (the frontend already sends `parse_type`). Callers
that omit `parse_type` when updating parser/pipeline selections will
receive a clear error message.
- Existing callers that only update fields like `name`, `enabled`, or
`meta_fields` are unaffected.
- Test updates ensure all known call sites are compliant.
2026-07-23 19:57:27 +08:00
|
|
|
|
"ragflow/internal/common"
|
2026-06-29 18:46:41 +08:00
|
|
|
|
lyt "ragflow/internal/deepdoc/parser/pdf/layout"
|
|
|
|
|
|
"ragflow/internal/deepdoc/parser/pdf/tool"
|
|
|
|
|
|
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
2026-07-02 09:46:33 +08:00
|
|
|
|
util "ragflow/internal/deepdoc/parser/pdf/util"
|
2026-06-25 20:16:16 +08:00
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
// TestPipelineParity verifies Go pipeline logic equivalence with Python.
|
|
|
|
|
|
// It loads Python pdfplumber chars (from charspy/), runs the Go pipeline
|
|
|
|
|
|
// with Top-based sorting to match Python's ordering, and compares sections
|
|
|
|
|
|
// against Python's output/py/noocr/text/ output.
|
|
|
|
|
|
//
|
|
|
|
|
|
// CharSim must be 100% — if not, Go pipeline logic differs from Python's.
|
|
|
|
|
|
func TestPipelineParity(t *testing.T) {
|
|
|
|
|
|
charspyDir := filepath.Join("testdata", "charspy")
|
|
|
|
|
|
pyTextDir := filepath.Join("testdata", "output", "py", "noocr", "text")
|
|
|
|
|
|
|
|
|
|
|
|
entries, err := os.ReadDir(charspyDir)
|
|
|
|
|
|
if err != nil {
|
|
|
|
|
|
t.Skipf("charspy/ not found: %v", err)
|
|
|
|
|
|
}
|
|
|
|
|
|
|
2026-07-10 11:58:32 +08:00
|
|
|
|
filter := common.GetEnv(common.EnvBatchParityFilter)
|
2026-06-25 20:16:16 +08:00
|
|
|
|
|
|
|
|
|
|
total, passed := 0, 0
|
|
|
|
|
|
for _, e := range entries {
|
|
|
|
|
|
if e.IsDir() || !strings.HasSuffix(e.Name(), ".json") {
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
name := strings.TrimSuffix(e.Name(), ".json")
|
|
|
|
|
|
if filter != "" && !strings.Contains(e.Name(), filter) {
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// Load Python chars
|
|
|
|
|
|
jsonPath := filepath.Join(charspyDir, e.Name())
|
2026-06-29 18:46:41 +08:00
|
|
|
|
engine, err := tool.LoadPythonChars(jsonPath)
|
2026-06-25 20:16:16 +08:00
|
|
|
|
if err != nil {
|
2026-06-29 18:46:41 +08:00
|
|
|
|
t.Errorf("%s: tool.LoadPythonChars: %v", name, err)
|
2026-06-25 20:16:16 +08:00
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// Run Go pipeline (SKIP_OCR — no DeepDoc)
|
2026-06-29 18:46:41 +08:00
|
|
|
|
cfg := pdf.DefaultParserConfig()
|
2026-06-25 20:16:16 +08:00
|
|
|
|
cfg.SortByTop = true
|
2026-07-02 09:46:33 +08:00
|
|
|
|
mockAnalyzer := &MockDocAnalyzer{Healthy: true}
|
|
|
|
|
|
p := NewParser(cfg)
|
|
|
|
|
|
result, err := p.ParseRaw(context.Background(), engine, mockAnalyzer)
|
2026-06-25 20:16:16 +08:00
|
|
|
|
if err != nil {
|
|
|
|
|
|
t.Errorf("%s: Parse: %v", name, err)
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// Read Python sections
|
|
|
|
|
|
pyPath := filepath.Join(pyTextDir, name+".txt")
|
|
|
|
|
|
pyData, err := os.ReadFile(pyPath)
|
|
|
|
|
|
if err != nil {
|
|
|
|
|
|
t.Logf("%s: no Python reference at %s — skip", name, pyPath)
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// Build Go text
|
|
|
|
|
|
var goText strings.Builder
|
|
|
|
|
|
for _, s := range result.Sections {
|
|
|
|
|
|
goText.WriteString(s.Text)
|
|
|
|
|
|
goText.WriteByte('\n')
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// Compare
|
2026-06-29 18:46:41 +08:00
|
|
|
|
sim := tool.CharSimilarity(goText.String(), tool.StripMeta(string(pyData)))
|
2026-06-25 20:16:16 +08:00
|
|
|
|
total++
|
|
|
|
|
|
if sim >= 100.0 {
|
|
|
|
|
|
passed++
|
|
|
|
|
|
t.Logf("PASS %s: CharSim=%.1f%% boxes:%d->%d->%d->%d",
|
|
|
|
|
|
name, sim, result.Metrics.BoxesInitial, result.Metrics.BoxesTextMerge, result.Metrics.BoxesVertMerge, len(result.Sections))
|
|
|
|
|
|
} else {
|
|
|
|
|
|
t.Errorf("FAIL %s: CharSim=%.1f%% (must be 100%%) boxes:%d->%d->%d->%d",
|
|
|
|
|
|
name, sim, result.Metrics.BoxesInitial, result.Metrics.BoxesTextMerge, result.Metrics.BoxesVertMerge, len(result.Sections))
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
if total == 0 {
|
|
|
|
|
|
t.Skip("no charspy/ files found")
|
|
|
|
|
|
}
|
|
|
|
|
|
t.Logf("Pipeline parity: %d/%d passed", passed, total)
|
|
|
|
|
|
if passed < total {
|
|
|
|
|
|
t.Errorf("%d/%d parity tests failed — Go pipeline differs from Python", total-passed, total)
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// TestVMWhitespaceGapBridge reproduces the exact RAG PDF divergence
|
|
|
|
|
|
// with synthetic boxes. A whitespace box (width > 0, gap just below
|
|
|
|
|
|
// threshold) gets merged into a content box, extending its bottom by
|
|
|
|
|
|
// the whitespace height. This flips the next gap from reject to merge,
|
|
|
|
|
|
// creating a cascade that reduces the section count by 1.
|
|
|
|
|
|
//
|
|
|
|
|
|
// Go's whitespace pre-filter removes this box before VM, so the
|
|
|
|
|
|
// bottom extension never happens and the cascade fails to start.
|
|
|
|
|
|
func TestVMWhitespaceGapBridge(t *testing.T) {
|
|
|
|
|
|
// Coordinates extracted from RAG PDF charspy data, "服务体系" region.
|
2026-06-29 18:46:41 +08:00
|
|
|
|
boxes := []pdf.TextBox{
|
2026-06-25 20:16:16 +08:00
|
|
|
|
// Content A: merged result of 3 preceding lines
|
|
|
|
|
|
{X0: 37.6, X1: 491.0, Top: 339.35, Bottom: 382.39,
|
|
|
|
|
|
Text: "生成文本再用standard分词建立索引", PageNumber: 1},
|
|
|
|
|
|
// Whitespace: U+00A0 non-breaking space, has non-zero width
|
|
|
|
|
|
{X0: 37.6, X1: 40.3, Top: 396.39, Bottom: 406.79,
|
|
|
|
|
|
Text: " ", PageNumber: 1},
|
|
|
|
|
|
// Content B: would be rejected without whitespace gap bridge
|
|
|
|
|
|
{X0: 37.6, X1: 543.3, Top: 420.16, Bottom: 431.19,
|
|
|
|
|
|
Text: "直接用rag分词建立索引", PageNumber: 1},
|
|
|
|
|
|
// Content C: cascades after B merges
|
|
|
|
|
|
{X0: 37.6, X1: 526.4, Top: 436.16, Bottom: 447.20,
|
|
|
|
|
|
Text: "是在原文中并没有这样的文字", PageNumber: 1},
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
mh := 9.361 // RAG PDF char median
|
|
|
|
|
|
thr := mh * 1.5
|
|
|
|
|
|
|
|
|
|
|
|
// Run VM with whitespace PRESENT (Python-like, no pre-filter).
|
|
|
|
|
|
// Python's while/pop merges whitespace at b_ position into b
|
|
|
|
|
|
// (extending b.bottom), then compares same b against next content.
|
|
|
|
|
|
// We simulate this by letting whitespace through gap/xov checks
|
|
|
|
|
|
// and absorbing it into prev when the checks pass.
|
|
|
|
|
|
vWithWS := func() int {
|
2026-06-29 18:46:41 +08:00
|
|
|
|
bxs := make([]pdf.TextBox, len(boxes))
|
2026-06-25 20:16:16 +08:00
|
|
|
|
copy(bxs, boxes)
|
|
|
|
|
|
sort.Slice(bxs, func(i, j int) bool {
|
|
|
|
|
|
if bxs[i].Top != bxs[j].Top {
|
|
|
|
|
|
return bxs[i].Top < bxs[j].Top
|
|
|
|
|
|
}
|
|
|
|
|
|
return bxs[i].X0 < bxs[j].X0
|
|
|
|
|
|
})
|
2026-06-29 18:46:41 +08:00
|
|
|
|
out := make([]pdf.TextBox, 0, len(bxs))
|
2026-06-25 20:16:16 +08:00
|
|
|
|
for i := 0; i < len(bxs); i++ {
|
|
|
|
|
|
b := bxs[i]
|
|
|
|
|
|
isWS := strings.TrimSpace(b.Text) == ""
|
|
|
|
|
|
// Whitespace in b position (current box): pop (skip).
|
|
|
|
|
|
// In Python: bxs.pop(i); continue; i stays.
|
|
|
|
|
|
if isWS && len(out) == 0 {
|
|
|
|
|
|
continue // nothing to extend
|
|
|
|
|
|
}
|
|
|
|
|
|
if isWS && len(out) > 0 {
|
|
|
|
|
|
prev := &out[len(out)-1]
|
|
|
|
|
|
gap := b.Top - prev.Bottom
|
2026-07-02 09:46:33 +08:00
|
|
|
|
ov := util.OverlapX(prev, &b)
|
2026-06-25 20:16:16 +08:00
|
|
|
|
// Python: gap passes AND xov passes → whitespace merged
|
|
|
|
|
|
// into prev, extending bottom. i advances (Go for-loop).
|
|
|
|
|
|
if gap <= thr && ov >= 0.3 {
|
|
|
|
|
|
prev.Bottom = b.Bottom
|
|
|
|
|
|
}
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
if len(out) == 0 {
|
|
|
|
|
|
out = append(out, b)
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
prev := &out[len(out)-1]
|
|
|
|
|
|
if prev.LayoutNo != b.LayoutNo {
|
|
|
|
|
|
out = append(out, b)
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
gap := b.Top - prev.Bottom
|
2026-07-02 09:46:33 +08:00
|
|
|
|
ov := util.OverlapX(prev, &b)
|
2026-06-25 20:16:16 +08:00
|
|
|
|
if gap > thr {
|
|
|
|
|
|
out = append(out, b)
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
if ov < 0.3 {
|
|
|
|
|
|
out = append(out, b)
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
pt := strings.TrimSpace(prev.Text)
|
|
|
|
|
|
bt := strings.TrimSpace(b.Text)
|
|
|
|
|
|
prev.Text = strings.TrimSpace(strings.TrimRight(pt, " \t") + " " + strings.TrimLeft(bt, " \t"))
|
|
|
|
|
|
prev.Bottom = b.Bottom
|
|
|
|
|
|
if prev.X0 > b.X0 {
|
|
|
|
|
|
prev.X0 = b.X0
|
|
|
|
|
|
}
|
|
|
|
|
|
if prev.X1 < b.X1 {
|
|
|
|
|
|
prev.X1 = b.X1
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
return len(out)
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// Run VM with whitespace PRE-FILTERED (Go current behavior).
|
|
|
|
|
|
vNoWS := func() int {
|
2026-06-29 18:46:41 +08:00
|
|
|
|
bxs := make([]pdf.TextBox, 0, len(boxes))
|
2026-06-25 20:16:16 +08:00
|
|
|
|
for _, b := range boxes {
|
|
|
|
|
|
if strings.TrimSpace(b.Text) != "" {
|
|
|
|
|
|
bxs = append(bxs, b)
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
sort.Slice(bxs, func(i, j int) bool {
|
|
|
|
|
|
if bxs[i].Top != bxs[j].Top {
|
|
|
|
|
|
return bxs[i].Top < bxs[j].Top
|
|
|
|
|
|
}
|
|
|
|
|
|
return bxs[i].X0 < bxs[j].X0
|
|
|
|
|
|
})
|
2026-06-29 18:46:41 +08:00
|
|
|
|
out := make([]pdf.TextBox, 0, len(bxs))
|
2026-06-25 20:16:16 +08:00
|
|
|
|
for i := 0; i < len(bxs); i++ {
|
|
|
|
|
|
b := bxs[i]
|
|
|
|
|
|
if len(out) == 0 {
|
|
|
|
|
|
out = append(out, b)
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
prev := &out[len(out)-1]
|
|
|
|
|
|
if prev.LayoutNo != b.LayoutNo {
|
|
|
|
|
|
out = append(out, b)
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
gap := b.Top - prev.Bottom
|
2026-07-02 09:46:33 +08:00
|
|
|
|
ov := util.OverlapX(prev, &b)
|
2026-06-25 20:16:16 +08:00
|
|
|
|
if gap > thr {
|
|
|
|
|
|
out = append(out, b)
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
if ov < 0.3 {
|
|
|
|
|
|
out = append(out, b)
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
pt := strings.TrimSpace(prev.Text)
|
|
|
|
|
|
bt := strings.TrimSpace(b.Text)
|
|
|
|
|
|
prev.Text = strings.TrimSpace(strings.TrimRight(pt, " \t") + " " + strings.TrimLeft(bt, " \t"))
|
|
|
|
|
|
prev.Bottom = b.Bottom
|
|
|
|
|
|
if prev.X0 > b.X0 {
|
|
|
|
|
|
prev.X0 = b.X0
|
|
|
|
|
|
}
|
|
|
|
|
|
if prev.X1 < b.X1 {
|
|
|
|
|
|
prev.X1 = b.X1
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
return len(out)
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
nWS := vWithWS()
|
|
|
|
|
|
nNoWS := vNoWS()
|
|
|
|
|
|
t.Logf("With whitespace (Python-like): %d sections", nWS)
|
|
|
|
|
|
t.Logf("Without whitespace (Go pre-filter): %d sections", nNoWS)
|
|
|
|
|
|
t.Logf("Gap without bridge: 420.16 - 382.39 = %.2f > %.2f = REJECT", 420.16-382.39, thr)
|
|
|
|
|
|
t.Logf("Gap with bridge: 420.16 - 406.79 = %.2f < %.2f = MERGE", 420.16-406.79, thr)
|
|
|
|
|
|
|
|
|
|
|
|
// The manual vWithWS (Python-like) and vNoWS (old Go pre-filter) still
|
2026-07-02 09:46:33 +08:00
|
|
|
|
// differ — the mechanism is real. But production lyt.NaiveVerticalMerge now
|
2026-06-25 20:16:16 +08:00
|
|
|
|
// handles whitespace inline (gap bridge), matching Python.
|
|
|
|
|
|
if nWS == nNoWS {
|
|
|
|
|
|
t.Error("Manual implementations should differ — the gap bridge mechanism is real")
|
|
|
|
|
|
}
|
|
|
|
|
|
|
2026-07-02 09:46:33 +08:00
|
|
|
|
// Verify production lyt.NaiveVerticalMerge matches vWithWS (Python behavior).
|
2026-06-25 20:16:16 +08:00
|
|
|
|
mhMap := map[int]float64{1: mh}
|
|
|
|
|
|
mwMap := map[int]float64{1: 5}
|
2026-07-10 10:36:10 +08:00
|
|
|
|
vmResult := lyt.NaiveVerticalMerge(boxes, mhMap, mwMap, nil)
|
2026-07-02 09:46:33 +08:00
|
|
|
|
t.Logf("lyt.NaiveVerticalMerge (production): %d sections", len(vmResult))
|
2026-06-25 20:16:16 +08:00
|
|
|
|
if len(vmResult) != nWS {
|
2026-07-02 09:46:33 +08:00
|
|
|
|
t.Errorf("lyt.NaiveVerticalMerge produced %d sections, want %d (Python-like with gap bridge)", len(vmResult), nWS)
|
2026-06-25 20:16:16 +08:00
|
|
|
|
}
|
|
|
|
|
|
}
|