mirror of
https://github.com/infiniflow/ragflow.git
synced 2026-07-23 00:46:42 +08:00
Refactor: oss parser go refactor (#16391)
### What problem does this PR solve? Package refactor and PDF post process. ### Type of change - [x] Refactoring --------- Co-authored-by: Claude <noreply@anthropic.com>
This commit is contained in:
32
.github/workflows/tests.yml
vendored
32
.github/workflows/tests.yml
vendored
@@ -226,8 +226,14 @@ jobs:
|
||||
-v "${PWD}/internal/cpp/resource:/usr/share/infinity/resource" \
|
||||
infiniflow/infinity_builder:ubuntu22_clang20
|
||||
sudo docker exec "${BUILDER_CONTAINER}" bash -c 'git config --global safe.directory "*" && cd /ragflow && ./build.sh --cpp'
|
||||
uv sync --python 3.13 --group test --frozen
|
||||
./build.sh --go
|
||||
|
||||
- name: Prepare Python test environment
|
||||
run: |
|
||||
uv sync --python 3.13 --group test --frozen
|
||||
uv pip install -e sdk/python
|
||||
|
||||
- name: Run Go unit tests
|
||||
# Runs after `./build.sh --go`, which guarantees the C++ static
|
||||
# library (librag_tokenizer_c_api.a) is present on disk. The Go
|
||||
@@ -250,10 +256,7 @@ jobs:
|
||||
PKGS=$(go list ./... 2>/dev/null \
|
||||
| grep -v '/internal/storage$' \
|
||||
| grep -v '/internal/tokenizer$' \
|
||||
| grep -v '/internal/handler$' \
|
||||
| grep -v '/internal/deepdoc/parser/pdf/pdfium' \
|
||||
| grep -v '/internal/deepdoc/parser/pdf/pdfoxide' \
|
||||
| grep -v '/internal/deepdoc/parser/pdf' || true)
|
||||
| grep -v '/internal/handler$' || true)
|
||||
if [ -z "$PKGS" ]; then
|
||||
./build.sh --test
|
||||
else
|
||||
@@ -266,11 +269,6 @@ jobs:
|
||||
sudo docker pull ubuntu:24.04
|
||||
sudo DOCKER_BUILDKIT=1 docker build --build-arg NEED_MIRROR=1 --build-arg HTTPS_PROXY=${HTTPS_PROXY} --build-arg HTTP_PROXY=${HTTP_PROXY} -f Dockerfile -t ${RAGFLOW_IMAGE} .
|
||||
|
||||
- name: Prepare Python test environment
|
||||
run: |
|
||||
uv sync --python 3.13 --group test --frozen
|
||||
uv pip install -e sdk/python
|
||||
|
||||
- name: Prepare function test environment
|
||||
working-directory: docker
|
||||
run: |
|
||||
@@ -672,8 +670,14 @@ jobs:
|
||||
-v "${PWD}/internal/cpp/resource:/usr/share/infinity/resource" \
|
||||
infiniflow/infinity_builder:ubuntu22_clang20
|
||||
sudo docker exec "${BUILDER_CONTAINER}" bash -c 'git config --global safe.directory "*" && cd /ragflow && ./build.sh --cpp'
|
||||
uv sync --python 3.13 --group test --frozen
|
||||
./build.sh --go
|
||||
|
||||
- name: Prepare Python test environment
|
||||
run: |
|
||||
uv sync --python 3.13 --group test --frozen
|
||||
uv pip install -e sdk/python
|
||||
|
||||
- name: Run Go unit tests
|
||||
# Runs after `./build.sh --go`, which guarantees the C++ static
|
||||
# library (librag_tokenizer_c_api.a) is present on disk. The Go
|
||||
@@ -696,10 +700,7 @@ jobs:
|
||||
PKGS=$(go list ./... 2>/dev/null \
|
||||
| grep -v '/internal/storage$' \
|
||||
| grep -v '/internal/tokenizer$' \
|
||||
| grep -v '/internal/handler$' \
|
||||
| grep -v '/internal/deepdoc/parser/pdf/pdfium' \
|
||||
| grep -v '/internal/deepdoc/parser/pdf/pdfoxide' \
|
||||
| grep -v '/internal/deepdoc/parser/pdf' || true)
|
||||
| grep -v '/internal/handler$' || true)
|
||||
if [ -z "$PKGS" ]; then
|
||||
./build.sh --test
|
||||
else
|
||||
@@ -712,11 +713,6 @@ jobs:
|
||||
sudo docker pull ubuntu:24.04
|
||||
sudo DOCKER_BUILDKIT=1 docker build --build-arg NEED_MIRROR=1 --build-arg HTTPS_PROXY=${HTTPS_PROXY} --build-arg HTTP_PROXY=${HTTP_PROXY} -f Dockerfile -t ${RAGFLOW_IMAGE} .
|
||||
|
||||
- name: Prepare Python test environment
|
||||
run: |
|
||||
uv sync --python 3.13 --group test --frozen
|
||||
uv pip install -e sdk/python
|
||||
|
||||
- name: Prepare function test environment
|
||||
working-directory: docker
|
||||
run: |
|
||||
|
||||
199
build.sh
199
build.sh
@@ -26,6 +26,14 @@ STRIP_SYMBOLS=""
|
||||
OFFICE_OXIDE_PREFIX="${HOME}/.office_oxide"
|
||||
OFFICE_OXIDE_VERSION="0.1.2"
|
||||
|
||||
# pdfium native library settings (from pypdfium2_raw PyPI wheel)
|
||||
PDFIUM_PREFIX="${HOME}/.pdfium"
|
||||
PDFIUM_VERSION="0.5.0"
|
||||
|
||||
# pdf_oxide native library settings (from GitHub Release)
|
||||
PDF_OXIDE_PREFIX="${HOME}/.pdf_oxide"
|
||||
PDF_OXIDE_VERSION="0.3.63"
|
||||
|
||||
echo -e "${GREEN}=== RAGFlow Go Server Build Script ===${NC}"
|
||||
|
||||
# Function to print section headers
|
||||
@@ -103,7 +111,7 @@ check_cpp_deps() {
|
||||
|
||||
check_go_deps() {
|
||||
print_section "Checking go dependencies"
|
||||
|
||||
|
||||
command -v go >/dev/null 2>&1 || { echo -e "${RED}Error: go is required but not installed.${NC}"; exit 1; }
|
||||
|
||||
echo "✓ Required tools are available"
|
||||
@@ -182,24 +190,163 @@ check_office_oxide_deps() {
|
||||
echo -e "${GREEN}✓ office_oxide native library installed${NC}"
|
||||
}
|
||||
|
||||
# Check / install pdfium native library (libpdfium.so from pypdfium2_raw wheel).
|
||||
check_pdfium_deps() {
|
||||
# 1. Check .venv (uv sync provides pypdfium2_raw).
|
||||
local venv_py="${PROJECT_ROOT}/.venv/bin/python3"
|
||||
if [ -x "$venv_py" ]; then
|
||||
local venv_so=$("$venv_py" -c "import pypdfium2_raw,os;print(os.path.join(os.path.dirname(pypdfium2_raw.__file__),'libpdfium.so'))" 2>/dev/null)
|
||||
if [ -n "$venv_so" ] && [ -f "$venv_so" ]; then
|
||||
echo " pdfium → ${venv_so} (.venv)"
|
||||
export CGO_LDFLAGS="$CGO_LDFLAGS -L$(dirname "$venv_so") -Wl,-rpath,$(dirname "$venv_so")"
|
||||
export LD_LIBRARY_PATH="$(dirname "$venv_so"):${LD_LIBRARY_PATH}"
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
|
||||
# 2. Check cache.
|
||||
local lib_path="${PDFIUM_PREFIX}/libpdfium.so"
|
||||
if [ -f "$lib_path" ]; then
|
||||
echo " pdfium → ${PDFIUM_PREFIX}"
|
||||
return 0
|
||||
fi
|
||||
|
||||
echo " pdfium not found, installing..."
|
||||
|
||||
# 3. Map platform to PyPI wheel platform tag.
|
||||
local whl_platform
|
||||
case "$(uname -s)" in
|
||||
Linux)
|
||||
case "$(uname -m)" in
|
||||
x86_64) whl_platform="manylinux_2_17_x86_64.manylinux2014_x86_64" ;;
|
||||
aarch64|arm64) whl_platform="manylinux_2_17_aarch64.manylinux2014_aarch64" ;;
|
||||
*) echo " pdfium → unsupported arch"; return 1 ;;
|
||||
esac
|
||||
;;
|
||||
Darwin)
|
||||
case "$(uname -m)" in
|
||||
x86_64) whl_platform="macosx_11_0_x86_64" ;;
|
||||
arm64) whl_platform="macosx_11_0_arm64" ;;
|
||||
*) echo " pdfium → unsupported arch"; return 1 ;;
|
||||
esac
|
||||
;;
|
||||
*) echo " pdfium → unsupported OS"; return 1 ;;
|
||||
esac
|
||||
|
||||
# 4. Download .whl from PyPI and extract libpdfium.so (zero pip dependency).
|
||||
local whl_url
|
||||
whl_url=$(curl -fsSL "https://pypi.org/pypi/pypdfium2_raw/${PDFIUM_VERSION}/json" 2>/dev/null \
|
||||
| grep -o '"url":"[^"]*'${whl_platform}'[^"]*"' | head -1 | cut -d'"' -f4)
|
||||
|
||||
if [ -n "$whl_url" ] && { command -v curl >/dev/null 2>&1 || command -v wget >/dev/null 2>&1; }; then
|
||||
local tmp_whl="$(mktemp)"
|
||||
if command -v curl >/dev/null 2>&1; then
|
||||
curl -fsSL "$whl_url" -o "$tmp_whl"
|
||||
else
|
||||
wget -q "$whl_url" -O "$tmp_whl"
|
||||
fi
|
||||
mkdir -p "${PDFIUM_PREFIX}"
|
||||
# Wheel is a zip; extract libpdfium.so via python3 or unzip.
|
||||
if command -v python3 >/dev/null 2>&1; then
|
||||
python3 -c "
|
||||
import zipfile, os, shutil
|
||||
with zipfile.ZipFile('$tmp_whl') as z:
|
||||
for n in z.namelist():
|
||||
if n.endswith('libpdfium.so'):
|
||||
z.extract(n, '${PDFIUM_PREFIX}')
|
||||
os.rename(os.path.join('${PDFIUM_PREFIX}', n), '$lib_path')
|
||||
# Remove empty pypdfium2_raw dir
|
||||
d = os.path.join('${PDFIUM_PREFIX}', 'pypdfium2_raw')
|
||||
if os.path.isdir(d): shutil.rmtree(d, ignore_errors=True)
|
||||
break
|
||||
" 2>/dev/null
|
||||
elif command -v unzip >/dev/null 2>&1; then
|
||||
unzip -q -o "$tmp_whl" -d "${PDFIUM_PREFIX}" 'pypdfium2_raw/libpdfium.so' 2>/dev/null
|
||||
[ -f "${PDFIUM_PREFIX}/pypdfium2_raw/libpdfium.so" ] && mv "${PDFIUM_PREFIX}/pypdfium2_raw/libpdfium.so" "$lib_path"
|
||||
rm -rf "${PDFIUM_PREFIX}/pypdfium2_raw"
|
||||
fi
|
||||
rm -f "$tmp_whl"
|
||||
fi
|
||||
|
||||
if [ -f "$lib_path" ]; then
|
||||
echo -e "${GREEN}✓ pdfium installed to ${PDFIUM_PREFIX}${NC}"
|
||||
else
|
||||
echo " pdfium → install failed (requires .venv, curl/wget + python3, or pre-cached ~/.pdfium)"
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
# Check / install pdf_oxide native library (Rust -> C FFI library).
|
||||
check_pdf_oxide_deps() {
|
||||
local lib_path="${PDF_OXIDE_PREFIX}/libpdf_oxide.so"
|
||||
|
||||
if [ -f "$lib_path" ]; then
|
||||
echo " pdf_oxide → ${PDF_OXIDE_PREFIX} (shared)"
|
||||
return 0
|
||||
fi
|
||||
|
||||
# Also check for static library (user's local installation).
|
||||
local static_path="${PDF_OXIDE_PREFIX}/libpdf_oxide.a"
|
||||
if [ -f "$static_path" ]; then
|
||||
echo " pdf_oxide → ${PDF_OXIDE_PREFIX} (static)"
|
||||
return 0
|
||||
fi
|
||||
|
||||
echo " pdf_oxide not found, installing..."
|
||||
|
||||
# Map platform to the release asset name.
|
||||
local asset_name
|
||||
case "$(uname -s)" in
|
||||
Linux)
|
||||
case "$(uname -m)" in
|
||||
x86_64) asset_name="libpdf_oxide-v${PDF_OXIDE_VERSION}-linux-x86_64" ;;
|
||||
aarch64|arm64) asset_name="libpdf_oxide-v${PDF_OXIDE_VERSION}-linux-aarch64" ;;
|
||||
*) echo " pdf_oxide → unsupported arch"; return 1 ;;
|
||||
esac
|
||||
;;
|
||||
Darwin)
|
||||
case "$(uname -m)" in
|
||||
x86_64) asset_name="libpdf_oxide-v${PDF_OXIDE_VERSION}-darwin-x86_64" ;;
|
||||
arm64) asset_name="libpdf_oxide-v${PDF_OXIDE_VERSION}-darwin-arm64" ;;
|
||||
*) echo " pdf_oxide → unsupported arch"; return 1 ;;
|
||||
esac
|
||||
;;
|
||||
*) echo " pdf_oxide → unsupported OS"; return 1 ;;
|
||||
esac
|
||||
|
||||
local release_url="https://github.com/yfedoseev/pdf_oxide/releases/download/v${PDF_OXIDE_VERSION}/${asset_name}.tar.gz"
|
||||
|
||||
mkdir -p "${PDF_OXIDE_PREFIX}"
|
||||
_download_and_extract "$release_url" "${PDF_OXIDE_PREFIX}"
|
||||
|
||||
if [ -f "$lib_path" ]; then
|
||||
echo -e "${GREEN}✓ pdf_oxide installed to ${PDF_OXIDE_PREFIX}${NC}"
|
||||
else
|
||||
echo " pdf_oxide → install failed"
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
# Build C++ static library
|
||||
build_cpp() {
|
||||
print_section "Building C++ static library"
|
||||
|
||||
|
||||
mkdir -p "$BUILD_DIR"
|
||||
cd "$BUILD_DIR"
|
||||
|
||||
|
||||
echo "Running cmake..."
|
||||
cmake .. -DCMAKE_BUILD_TYPE=Release
|
||||
|
||||
|
||||
echo "Building librag_tokenizer_c_api.a..."
|
||||
make rag_tokenizer_c_api -j$(nproc)
|
||||
|
||||
local jobs
|
||||
jobs="$(nproc 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || echo 1)"
|
||||
make rag_tokenizer_c_api -j"$jobs"
|
||||
|
||||
if [ ! -f "$BUILD_DIR/librag_tokenizer_c_api.a" ]; then
|
||||
echo -e "${RED}Error: Failed to build C++ static library${NC}"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
|
||||
echo -e "${GREEN}✓ C++ static library built successfully${NC}"
|
||||
}
|
||||
|
||||
@@ -217,7 +364,9 @@ build_cpp_test() {
|
||||
fi
|
||||
|
||||
echo "Building rag_analyzer_c_test..."
|
||||
make rag_analyzer_c_test -j$(nproc)
|
||||
local jobs
|
||||
jobs="$(nproc 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || echo 1)"
|
||||
make rag_analyzer_c_test -j"$jobs"
|
||||
|
||||
if [ ! -f "$BUILD_DIR/rag_analyzer_c_test" ]; then
|
||||
echo -e "${RED}Error: Failed to build rag_analyzer_c_test${NC}"
|
||||
@@ -296,17 +445,35 @@ build_go() {
|
||||
echo -e "${GREEN}✓ Go ingestor built successfully: $INGESTOR_BINARY${NC}"
|
||||
}
|
||||
|
||||
# Configure CGO flags for the office_oxide native library and the runtime
|
||||
# rpath used by test binaries. Call before any `go build` / `go test` step
|
||||
# that links against office_oxide.
|
||||
# Configure CGO flags for native libraries (office_oxide, pdfium, pdf_oxide).
|
||||
# Call before any `go build` / `go test` step that links against these libraries.
|
||||
setup_cgo_env() {
|
||||
# ── office_oxide ──────────────────────────────────────────────────
|
||||
check_office_oxide_deps
|
||||
export CGO_CFLAGS="-I${OFFICE_OXIDE_PREFIX}/include/office_oxide_c${CGO_CFLAGS:+ $CGO_CFLAGS}"
|
||||
echo "Exporting CGO_CFLAGS: $CGO_CFLAGS"
|
||||
export CGO_LDFLAGS="-L${OFFICE_OXIDE_PREFIX}/lib -loffice_oxide -Wl,-rpath,${OFFICE_OXIDE_PREFIX}/lib${CGO_LDFLAGS:+ $CGO_LDFLAGS}"
|
||||
echo "Exporting CGO_LDFLAGS: $CGO_LDFLAGS"
|
||||
# Make the .so discoverable to test binaries spawned without rpath.
|
||||
export CGO_LDFLAGS="-L${OFFICE_OXIDE_PREFIX}/lib -loffice_oxide -Wl,-rpath,${OFFICE_OXIDE_PREFIX}/lib"
|
||||
export LD_LIBRARY_PATH="${OFFICE_OXIDE_PREFIX}/lib${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}"
|
||||
|
||||
# ── pdfium ────────────────────────────────────────────────────────
|
||||
check_pdfium_deps || return 1
|
||||
if [ -f "${PDFIUM_PREFIX}/libpdfium.so" ]; then
|
||||
export CGO_LDFLAGS="$CGO_LDFLAGS -L${PDFIUM_PREFIX} -Wl,-rpath,${PDFIUM_PREFIX}"
|
||||
export LD_LIBRARY_PATH="${PDFIUM_PREFIX}:${LD_LIBRARY_PATH}"
|
||||
fi
|
||||
|
||||
# ── pdf_oxide ─────────────────────────────────────────────────────
|
||||
check_pdf_oxide_deps || return 1
|
||||
if [ -f "${PDF_OXIDE_PREFIX}/libpdf_oxide.so" ]; then
|
||||
export CGO_LDFLAGS="$CGO_LDFLAGS -L${PDF_OXIDE_PREFIX} -lpdf_oxide -Wl,-rpath,${PDF_OXIDE_PREFIX}"
|
||||
export LD_LIBRARY_PATH="${PDF_OXIDE_PREFIX}:${LD_LIBRARY_PATH}"
|
||||
elif [ -f "${PDF_OXIDE_PREFIX}/libpdf_oxide.a" ]; then
|
||||
export CGO_LDFLAGS="$CGO_LDFLAGS ${PDF_OXIDE_PREFIX}/libpdf_oxide.a"
|
||||
fi
|
||||
|
||||
echo "CGO_CFLAGS: $CGO_CFLAGS"
|
||||
echo "Exporting CGO_CFLAGS: $CGO_CFLAGS"
|
||||
echo "CGO_LDFLAGS: $CGO_LDFLAGS"
|
||||
echo "Exporting CGO_LDFLAGS: $CGO_LDFLAGS"
|
||||
}
|
||||
|
||||
# Run Go unit tests with the same CGO env as `build_go`. Pass any extra args
|
||||
@@ -328,7 +495,7 @@ run_go_tests() {
|
||||
# Clean build artifacts
|
||||
clean() {
|
||||
print_section "Cleaning build artifacts"
|
||||
|
||||
|
||||
rm -rf "$BUILD_DIR"
|
||||
rm -f "$RAGFLOW_SERVER_BINARY"
|
||||
rm -f "$ADMIN_SERVER_BINARY"
|
||||
|
||||
@@ -10,7 +10,6 @@ import (
|
||||
"math"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"ragflow/internal/deepdoc/parser/pdf/tools"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strconv"
|
||||
@@ -18,6 +17,11 @@ import (
|
||||
"testing"
|
||||
"time"
|
||||
"unicode/utf8"
|
||||
|
||||
inf "ragflow/internal/deepdoc/parser/pdf/inference"
|
||||
lyt "ragflow/internal/deepdoc/parser/pdf/layout"
|
||||
"ragflow/internal/deepdoc/parser/pdf/tool"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// TestBatchResults runs Parse() on real PDFs and writes:
|
||||
@@ -47,14 +51,14 @@ func TestBatchResults(t *testing.T) {
|
||||
}
|
||||
pdfs := all[:min(count, len(all))]
|
||||
|
||||
ddClient, err := NewDeepDocClient(os.Getenv("DEEPDOC_URL"))
|
||||
ddClient, err := inf.NewInferenceClient(os.Getenv("DEEPDOC_URL"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !ddClient.Health() {
|
||||
t.Fatalf("DeepDoc service not available at %s (DLA+TSR required)", ddClient.baseURL)
|
||||
}
|
||||
deepDoc := DocAnalyzer(ddClient)
|
||||
deepDoc := pdf.DocAnalyzer(ddClient)
|
||||
|
||||
variant := variantFromEnv()
|
||||
t.Logf("DeepDoc available — DLA+TSR%s enabled (%d PDFs)",
|
||||
@@ -149,7 +153,7 @@ func filterSingle(pdfs []string, name string, t *testing.T) []string {
|
||||
}
|
||||
|
||||
// extractPageStats returns (charCount, boxCount) for all pages in engine.
|
||||
func extractPageStats(eng PDFEngine) (chars, boxes int) {
|
||||
func extractPageStats(eng pdf.PDFEngine) (chars, boxes int) {
|
||||
np, _ := eng.PageCount()
|
||||
for pg := 0; pg < np; pg++ {
|
||||
pgChars, err := eng.ExtractChars(pg)
|
||||
@@ -157,7 +161,7 @@ func extractPageStats(eng PDFEngine) (chars, boxes int) {
|
||||
continue
|
||||
}
|
||||
chars += len(pgChars)
|
||||
boxes += len(charsToBoxes(pgChars, pg, false))
|
||||
boxes += len(lyt.CharsToBoxes(pgChars, pg, false))
|
||||
}
|
||||
return
|
||||
}
|
||||
@@ -172,9 +176,9 @@ func textLenFromOutput(data []byte) int {
|
||||
|
||||
// ── main processing loop ────────────────────────────────────────────
|
||||
|
||||
func processPDFs(t *testing.T, pdfDir string, pdfs []string, deepDoc DocAnalyzer, variant string, dirs outputDirs) []tools.BatchResult {
|
||||
func processPDFs(t *testing.T, pdfDir string, pdfs []string, deepDoc pdf.DocAnalyzer, variant string, dirs outputDirs) []tool.BatchResult {
|
||||
t.Helper()
|
||||
var results []tools.BatchResult
|
||||
var results []tool.BatchResult
|
||||
totalChars := 0
|
||||
skipOCR := os.Getenv("BATCH_SKIP_OCR") == "1"
|
||||
|
||||
@@ -193,7 +197,7 @@ func processPDFs(t *testing.T, pdfDir string, pdfs []string, deepDoc DocAnalyzer
|
||||
// ── parse ──
|
||||
res, err := parseOne(pdfDir, name, deepDoc, skipOCR)
|
||||
if err != nil {
|
||||
results = append(results, tools.BatchResult{File: name, Error: err.Error()})
|
||||
results = append(results, tool.BatchResult{File: name, Error: err.Error()})
|
||||
t.Logf("%s — %v", label, err)
|
||||
continue
|
||||
}
|
||||
@@ -213,11 +217,11 @@ func processPDFs(t *testing.T, pdfDir string, pdfs []string, deepDoc DocAnalyzer
|
||||
}
|
||||
|
||||
type parseOneResult struct {
|
||||
tools.BatchResult
|
||||
result ParseResult
|
||||
tool.BatchResult
|
||||
result pdf.ParseResult
|
||||
}
|
||||
|
||||
func parseOne(pdfDir, name string, deepDoc DocAnalyzer, skipOCR bool) (*parseOneResult, error) {
|
||||
func parseOne(pdfDir, name string, deepDoc pdf.DocAnalyzer, skipOCR bool) (*parseOneResult, error) {
|
||||
data, err := os.ReadFile(filepath.Join(pdfDir, name))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read: %w", err)
|
||||
@@ -232,7 +236,7 @@ func parseOne(pdfDir, name string, deepDoc DocAnalyzer, skipOCR bool) (*parseOne
|
||||
pageCount, _ := eng.PageCount()
|
||||
chars, _ := extractPageStats(eng)
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
cfg.SkipOCR = skipOCR
|
||||
p := NewParser(cfg, deepDoc)
|
||||
t0 := time.Now()
|
||||
@@ -248,7 +252,7 @@ func parseOne(pdfDir, name string, deepDoc DocAnalyzer, skipOCR bool) (*parseOne
|
||||
}
|
||||
|
||||
return &parseOneResult{
|
||||
BatchResult: tools.BatchResult{
|
||||
BatchResult: tool.BatchResult{
|
||||
File: name,
|
||||
Pages: pageCount,
|
||||
Chars: chars,
|
||||
@@ -263,17 +267,17 @@ func parseOne(pdfDir, name string, deepDoc DocAnalyzer, skipOCR bool) (*parseOne
|
||||
}, nil
|
||||
}
|
||||
|
||||
func tryLoadCached(dirs outputDirs, name string) *tools.BatchResult {
|
||||
func tryLoadCached(dirs outputDirs, name string) *tool.BatchResult {
|
||||
textPath := filepath.Join(dirs.text, name+".txt")
|
||||
tablesPath := filepath.Join(dirs.tables, name+".json")
|
||||
if !tools.FileExists(textPath) || !tools.FileExists(tablesPath) {
|
||||
if !tool.FileExists(textPath) || !tool.FileExists(tablesPath) {
|
||||
return nil
|
||||
}
|
||||
data, err := os.ReadFile(textPath)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
var r tools.BatchResult
|
||||
var r tool.BatchResult
|
||||
r.File = name
|
||||
if idx := strings.LastIndex(string(data), "\n#@meta"); idx >= 0 {
|
||||
if json.Unmarshal(data[idx+7:], &r) == nil {
|
||||
@@ -301,7 +305,7 @@ func htmlToRows(html string) [][]string {
|
||||
return rows
|
||||
}
|
||||
|
||||
func writeOutputs(dirs outputDirs, name string, parsed *ParseResult, res *parseOneResult) {
|
||||
func writeOutputs(dirs outputDirs, name string, parsed *pdf.ParseResult, res *parseOneResult) {
|
||||
// ── text + #@meta ──
|
||||
var sb strings.Builder
|
||||
for _, s := range parsed.Sections {
|
||||
@@ -317,11 +321,11 @@ func writeOutputs(dirs outputDirs, name string, parsed *ParseResult, res *parseO
|
||||
|
||||
// ── tables JSON — extract rows from section HTML (matching Python html_to_rows) ──
|
||||
type slimTable struct {
|
||||
Rows [][]string `json:"rows"`
|
||||
Positions []Position `json:"positions,omitempty"`
|
||||
Rows [][]string `json:"rows"`
|
||||
Positions []pdf.Position `json:"positions,omitempty"`
|
||||
}
|
||||
// Collect all table sections in order (index-matched to TableItems).
|
||||
var tableSections []Section
|
||||
var tableSections []pdf.Section
|
||||
for _, s := range parsed.Sections {
|
||||
if s.LayoutType == "table" && strings.HasPrefix(s.Text, "<table>") {
|
||||
tableSections = append(tableSections, s)
|
||||
@@ -1,89 +0,0 @@
|
||||
//go:build cgo
|
||||
|
||||
package parser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"ragflow/internal/deepdoc/parser/pdf/tools"
|
||||
)
|
||||
|
||||
// TestParse_ChunkEquivalence verifies that chunked processing produces
|
||||
// the same output as processing all pages at once. Uses chunkSize=1
|
||||
// (every page is its own chunk) on a multi-page fixture to maximize
|
||||
// chunk boundary stress.
|
||||
func TestParse_ChunkEquivalence(t *testing.T) {
|
||||
data, err := readTestPDF(t, "03_multipage.pdf")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
parse := func(chunkSize int) *ParseResult {
|
||||
eng, err := NewEngine(data)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer eng.Close()
|
||||
cfg := DefaultParserConfig()
|
||||
cfg.ChunkSize = chunkSize
|
||||
p := NewParser(cfg, &MockDocAnalyzer{Healthy: true, Model: ModelSaas})
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// No chunking (all pages at once).
|
||||
full := parse(9999)
|
||||
// Aggressive chunking (1 page per chunk).
|
||||
chunked := parse(1)
|
||||
|
||||
// Compare section counts.
|
||||
if len(full.Sections) != len(chunked.Sections) {
|
||||
t.Logf("section count: full=%d chunked=%d (small diff acceptable at chunk boundaries)",
|
||||
len(full.Sections), len(chunked.Sections))
|
||||
}
|
||||
|
||||
// Compare text content via CharSimilarity.
|
||||
fullText := sectionsText(full.Sections)
|
||||
chunkedText := sectionsText(chunked.Sections)
|
||||
charSim := tools.CharSimilarity(fullText, chunkedText)
|
||||
t.Logf("CharSimilarity: %.1f%%", charSim)
|
||||
if charSim < 95 {
|
||||
t.Errorf("chunk equivalence too low: CharSim=%.1f%% (want >= 95%%)", charSim)
|
||||
}
|
||||
|
||||
// Compare metrics (should be identical or very close).
|
||||
t.Logf("Metrics: full=%+v chunked=%+v", full.Metrics, chunked.Metrics)
|
||||
if full.Metrics.BoxesInitial != chunked.Metrics.BoxesInitial {
|
||||
t.Errorf("BoxesInitial: full=%d chunked=%d",
|
||||
full.Metrics.BoxesInitial, chunked.Metrics.BoxesInitial)
|
||||
}
|
||||
|
||||
// Bug fix regression: PageImages must survive chunked merge.
|
||||
if len(full.PageImages) == 0 {
|
||||
t.Error("full parse: PageImages should not be empty (3-page document)")
|
||||
}
|
||||
if len(chunked.PageImages) == 0 {
|
||||
t.Error("chunked parse: PageImages should be preserved across chunks")
|
||||
}
|
||||
}
|
||||
|
||||
func readTestPDF(t *testing.T, name string) ([]byte, error) {
|
||||
t.Helper()
|
||||
return os.ReadFile(filepath.Join("testdata", "pdfs", name))
|
||||
}
|
||||
|
||||
func sectionsText(sections []Section) string {
|
||||
var sb strings.Builder
|
||||
for _, s := range sections {
|
||||
sb.WriteString(s.Text)
|
||||
sb.WriteByte('\n')
|
||||
}
|
||||
return sb.String()
|
||||
}
|
||||
@@ -1,74 +0,0 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"unicode"
|
||||
)
|
||||
|
||||
// ---- MergeSameBullet (Python: pdf_parser.py _merge_same_bullet) ----
|
||||
|
||||
// MergeSameBullet merges adjacent boxes that start with the same bullet/number
|
||||
// character, combining their text with a newline separator.
|
||||
func MergeSameBullet(boxes []TextBox, tok Tokenizer) []TextBox {
|
||||
if len(boxes) < 2 {
|
||||
return boxes
|
||||
}
|
||||
// Build output via two-pointer collect: O(n) instead of O(n²) slice-element removal.
|
||||
out := make([]TextBox, 0, len(boxes))
|
||||
i := 0
|
||||
for i < len(boxes) {
|
||||
if strings.TrimSpace(boxes[i].Text) == "" {
|
||||
i++
|
||||
continue
|
||||
}
|
||||
// Start a merge chain from position i.
|
||||
cur := boxes[i]
|
||||
i++
|
||||
for i < len(boxes) {
|
||||
if strings.TrimSpace(boxes[i].Text) == "" {
|
||||
i++
|
||||
continue
|
||||
}
|
||||
nxt := boxes[i]
|
||||
firstCur := firstRuneString(cur.Text)
|
||||
firstNxt := firstRuneString(nxt.Text)
|
||||
|
||||
// Conditions to NOT merge:
|
||||
if firstCur != firstNxt ||
|
||||
unicode.Is(unicode.Latin, firstCur) ||
|
||||
isChinese(firstCur, tok) ||
|
||||
cur.Top > nxt.Bottom {
|
||||
break
|
||||
}
|
||||
|
||||
// Merge nxt into cur.
|
||||
cur.Text = cur.Text + "\n" + nxt.Text
|
||||
cur.X0 = min(cur.X0, nxt.X0)
|
||||
cur.X1 = max(cur.X1, nxt.X1)
|
||||
cur.Bottom = nxt.Bottom
|
||||
i++
|
||||
}
|
||||
out = append(out, cur)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// ---- Helpers ----
|
||||
|
||||
func firstRuneString(s string) rune {
|
||||
s = strings.TrimSpace(s)
|
||||
if s == "" {
|
||||
return 0
|
||||
}
|
||||
return []rune(s)[0]
|
||||
}
|
||||
|
||||
// isChinese checks if a rune is a Chinese character (CJK Unified Ideograph).
|
||||
func isChinese(r rune, tok Tokenizer) bool {
|
||||
if tok != nil {
|
||||
return strings.Contains(tok.Tag(string(r)), "n")
|
||||
}
|
||||
return (r >= 0x4E00 && r <= 0x9FFF) ||
|
||||
(r >= 0x3400 && r <= 0x4DBF) ||
|
||||
(r >= 0x20000 && r <= 0x2A6DF)
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
//go:build cgo
|
||||
//go:build cgo && manual
|
||||
|
||||
package parser
|
||||
|
||||
@@ -9,6 +9,7 @@ import (
|
||||
"image/png"
|
||||
"os"
|
||||
"path/filepath"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"testing"
|
||||
)
|
||||
|
||||
@@ -25,8 +26,8 @@ func TestParse_CropSectionImages(t *testing.T) {
|
||||
}
|
||||
defer eng.Close()
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
p := NewParser(cfg, &MockDocAnalyzer{Healthy: true, Model: ModelSaas})
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
p := NewParser(cfg, &MockDocAnalyzer{Healthy: true})
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
@@ -78,7 +79,7 @@ func TestCrop_Regression_SnapshotPDFs(t *testing.T) {
|
||||
}
|
||||
defer eng.Close()
|
||||
|
||||
p := NewParser(DefaultParserConfig(), &MockDocAnalyzer{Healthy: true, Model: ModelSaas})
|
||||
p := NewParser(pdf.DefaultParserConfig(), &MockDocAnalyzer{Healthy: true})
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
|
||||
@@ -1,904 +0,0 @@
|
||||
//go:build cgo
|
||||
|
||||
package parser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"image"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// ── MockDocAnalyzer tests ──────────────────────────────────────────────
|
||||
|
||||
func TestMockDocAnalyzer(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 100, Label: "table", Confidence: 0.95},
|
||||
},
|
||||
TSRCells: []TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 50, Y1: 30, Text: "A"},
|
||||
},
|
||||
}
|
||||
|
||||
if !mock.Health() {
|
||||
t.Error("mock should be healthy")
|
||||
}
|
||||
regions, _ := mock.DLA(context.Background(), nil)
|
||||
if len(regions) != 1 || regions[0].Label != "table" {
|
||||
t.Error("mock DLA returned wrong data")
|
||||
}
|
||||
cells, _ := mock.TSR(context.Background(), nil)
|
||||
if len(cells) != 1 || cells[0].Text != "A" {
|
||||
t.Error("mock TSR returned wrong data")
|
||||
}
|
||||
// OCRDetect + OCRRecognize replaces deprecated OCR — tested in TestOCR_scanPage/TestOCR_fallback.
|
||||
_ = mock.OCRDetect
|
||||
_ = mock.OCRRecognize
|
||||
|
||||
// Unhealthy mock
|
||||
mock2 := &MockDocAnalyzer{Healthy: false}
|
||||
if mock2.Health() {
|
||||
t.Error("unhealthy mock should return false")
|
||||
}
|
||||
}
|
||||
|
||||
// ── groupTSRCellsToRows ────────────────────────────────────────────────
|
||||
|
||||
func TestGroupTSRCellsToRows(t *testing.T) {
|
||||
t.Run("empty", func(t *testing.T) {
|
||||
if rows := groupTSRCellsToRows(nil); rows != nil {
|
||||
t.Error("nil → nil")
|
||||
}
|
||||
if rows := groupTSRCellsToRows([]TSRCell{}); rows != nil {
|
||||
t.Error("empty → nil")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("single cell", func(t *testing.T) {
|
||||
cells := []TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 50, Text: "A"}}
|
||||
rows := groupTSRCellsToRows(cells)
|
||||
if len(rows) != 1 || rows[0][0].Text != "A" {
|
||||
t.Error("single cell not preserved")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("two rows two cols", func(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 50, Y1: 30, Text: "A"},
|
||||
{X0: 50, Y0: 0, X1: 100, Y1: 30, Text: "B"},
|
||||
{X0: 0, Y0: 50, X1: 50, Y1: 80, Text: "C"},
|
||||
{X0: 50, Y0: 50, X1: 100, Y1: 80, Text: "D"},
|
||||
}
|
||||
rows := groupTSRCellsToRows(cells)
|
||||
if len(rows) != 2 {
|
||||
t.Fatalf("2 rows expected, got %d", len(rows))
|
||||
}
|
||||
if rows[0][0].Text != "A" || rows[0][1].Text != "B" {
|
||||
t.Errorf("row0: %v", cellTexts(rows[0]))
|
||||
}
|
||||
if rows[1][0].Text != "C" || rows[1][1].Text != "D" {
|
||||
t.Errorf("row1: %v", cellTexts(rows[1]))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("unsorted input", func(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
{X0: 50, Y0: 50, X1: 100, Y1: 80, Text: "D"},
|
||||
{X0: 0, Y0: 0, X1: 50, Y1: 30, Text: "A"},
|
||||
{X0: 0, Y0: 50, X1: 50, Y1: 80, Text: "C"},
|
||||
{X0: 50, Y0: 0, X1: 100, Y1: 30, Text: "B"},
|
||||
}
|
||||
rows := groupTSRCellsToRows(cells)
|
||||
if len(rows) != 2 {
|
||||
t.Fatalf("unsorted: 2 rows expected, got %d", len(rows))
|
||||
}
|
||||
if rows[0][0].Text != "A" || rows[0][1].Text != "B" {
|
||||
t.Errorf("unsorted row0: %v", cellTexts(rows[0]))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("tall merged cell", func(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 50, Y1: 100, Text: "merged"},
|
||||
{X0: 50, Y0: 0, X1: 100, Y1: 30, Text: "B"},
|
||||
{X0: 50, Y0: 50, X1: 100, Y1: 80, Text: "D"},
|
||||
}
|
||||
rows := groupTSRCellsToRows(cells)
|
||||
// merged cell starts Y0=0 → row 0; Y0=50 cell → row 1
|
||||
if len(rows) != 2 {
|
||||
t.Fatalf("merged cell: 2 rows expected, got %d", len(rows))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("large gap different rows", func(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 50, Y1: 30, Text: "top"},
|
||||
{X0: 0, Y0: 200, X1: 50, Y1: 230, Text: "far"},
|
||||
}
|
||||
rows := groupTSRCellsToRows(cells)
|
||||
if len(rows) != 2 {
|
||||
t.Fatalf("large gap: 2 rows expected, got %d", len(rows))
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// ── fillCellTextFromBoxes ──────────────────────────────────────────────
|
||||
|
||||
func TestFillCellTextFromBoxes(t *testing.T) {
|
||||
t.Run("exact match", func(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 50},
|
||||
{X0: 100, Y0: 0, X1: 200, Y1: 50},
|
||||
}
|
||||
boxes := []TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 50, Text: "A"},
|
||||
{X0: 100, X1: 200, Top: 0, Bottom: 50, Text: "B"},
|
||||
}
|
||||
fillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "A" || cells[1].Text != "B" {
|
||||
t.Errorf("got %q/%q, want A/B", cells[0].Text, cells[1].Text)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("empty cells", func(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 50},
|
||||
{X0: 100, Y0: 0, X1: 200, Y1: 50},
|
||||
}
|
||||
boxes := []TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 50, Text: "only first"},
|
||||
}
|
||||
fillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "only first" {
|
||||
t.Errorf("cell[0]: got %q", cells[0].Text)
|
||||
}
|
||||
if cells[1].Text != "" {
|
||||
t.Errorf("cell[1] should be empty, got %q", cells[1].Text)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("partial cell coverage — empty cell filled from any overlapping box", func(t *testing.T) {
|
||||
// Box covers 40% of cell area. Old code rejected (<85% cell coverage).
|
||||
// New code: cell is empty → accepts box (≥30% box area inside cell).
|
||||
cells := []TSRCell{{X0: 0, Y0: 0, X1: 200, Y1: 50}}
|
||||
boxes := []TextBox{{X0: 0, X1: 80, Top: 0, Bottom: 50, Text: "partial"}}
|
||||
fillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "partial" {
|
||||
t.Errorf("empty cell should be filled from overlapping box, got %q", cells[0].Text)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("box inside cell >85%", func(t *testing.T) {
|
||||
cells := []TSRCell{{X0: 0, Y0: 0, X1: 500, Y1: 300}}
|
||||
boxes := []TextBox{{X0: 10, X1: 490, Top: 10, Bottom: 290, Text: "inside"}}
|
||||
fillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "inside" {
|
||||
t.Errorf("got %q", cells[0].Text)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("concatenate two boxes to same cell", func(t *testing.T) {
|
||||
cells := []TSRCell{{X0: 0, Y0: 0, X1: 200, Y1: 100}}
|
||||
boxes := []TextBox{
|
||||
{X0: 5, X1: 195, Top: 2, Bottom: 98, Text: "hello"},
|
||||
{X0: 5, X1: 195, Top: 2, Bottom: 98, Text: "world"},
|
||||
}
|
||||
fillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "hello world" {
|
||||
t.Errorf("got %q, want 'hello world'", cells[0].Text)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("empty inputs", func(t *testing.T) {
|
||||
fillCellTextFromBoxes(nil, nil)
|
||||
fillCellTextFromBoxes([]TSRCell{}, []TextBox{})
|
||||
c := []TSRCell{{X0: 0, Y0: 0, X1: 1, Y1: 1}}
|
||||
fillCellTextFromBoxes(c, nil)
|
||||
if c[0].Text != "" {
|
||||
t.Error("no boxes → text empty")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// ── regionOverlapsBox ──────────────────────────────────────────────────
|
||||
|
||||
func TestRegionOverlapsBox(t *testing.T) {
|
||||
scale := 3.0
|
||||
tests := []struct {
|
||||
name string
|
||||
region DLARegion
|
||||
box TextBox
|
||||
expected bool
|
||||
}{
|
||||
{"full overlap", DLARegion{X0: 0, Y0: 300, X1: 1500, Y1: 2300, Label: "table", Confidence: 0.9}, TextBox{X0: 50, X1: 500, Top: 100, Bottom: 760, Text: "x", PageNumber: 0}, true},
|
||||
{"no overlap", DLARegion{X0: 0, Y0: 3000, X1: 1500, Y1: 5000, Label: "table", Confidence: 0.9}, TextBox{X0: 50, X1: 500, Top: 0, Bottom: 10, Text: "x", PageNumber: 0}, false},
|
||||
{"no Y overlap", DLARegion{X0: 150, Y0: 300, X1: 1650, Y1: 336, Label: "table", Confidence: 0.9}, TextBox{X0: 50, X1: 550, Top: 500, Bottom: 520, Text: "x", PageNumber: 0}, false},
|
||||
{"zero area box", DLARegion{X0: 0, Y0: 300, X1: 1500, Y1: 2300, Label: "table", Confidence: 0.9}, TextBox{X0: 50, X1: 50, Top: 50, Bottom: 50, Text: "x", PageNumber: 0}, false},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
if got := regionOverlapsBox(tt.region, tt.box, scale); got != tt.expected {
|
||||
t.Errorf("= %v, want %v", got, tt.expected)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// ── enrichWithDeepDoc noop ─────────────────────────────────────────────
|
||||
|
||||
func TestEnrichWithDeepDoc_Noop(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 100, Bottom: 112, Text: "text"},
|
||||
}
|
||||
eng := &mockEngine{pageCount: 1}
|
||||
|
||||
p := NewParser(DefaultParserConfig(), &MockDocAnalyzer{Healthy: false, Model: ModelSaas})
|
||||
tables := p.enrichWithDeepDoc(context.Background(), eng, boxes, nil)
|
||||
if len(tables) != 0 {
|
||||
t.Error("unhealthy DeepDoc → 0 Tables")
|
||||
}
|
||||
}
|
||||
|
||||
// ── extractTableBoxesFromImage with mock ───────────────────────────────
|
||||
|
||||
func TestExtractTableBoxes_Mock(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
{PageNumber: 0, X0: 80, X1: 500, Top: 200, Bottom: 550, Text: "cell 1"},
|
||||
{PageNumber: 0, X0: 80, X1: 500, Top: 550, Bottom: 760, Text: "cell 2"},
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 100, Bottom: 180, Text: "heading"},
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 780, Bottom: 850, Text: "below"},
|
||||
}
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []DLARegion{
|
||||
{X0: 250, Y0: 600, X1: 1500, Y1: 2300, Label: "table", Confidence: 0.95},
|
||||
},
|
||||
TSRCells: []TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 600, Y1: 400, Text: "A1"},
|
||||
{X0: 600, Y0: 0, X1: 1240, Y1: 400, Text: "B1"},
|
||||
{X0: 0, Y0: 410, X1: 600, Y1: 800, Text: "A2"},
|
||||
{X0: 600, Y0: 410, X1: 1240, Y1: 800, Text: "B2"},
|
||||
},
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
dummyImg := image.NewRGBA(image.Rect(0, 0, 2000, 3000))
|
||||
|
||||
tables := p.extractTableBoxesFromImage(context.Background(), boxes, dummyImg, 0, 0)
|
||||
if len(tables) != 1 {
|
||||
t.Fatalf("expected 1 TableItem, got %d", len(tables))
|
||||
}
|
||||
tbl := tables[0]
|
||||
if len(tbl.Cells) != 4 {
|
||||
t.Errorf("expected 4 cells, got %d", len(tbl.Cells))
|
||||
}
|
||||
// Rows populated later by constructTable via extractTableAndReplace.
|
||||
if tbl.ImageB64 == "" {
|
||||
t.Error("ImageB64 empty")
|
||||
}
|
||||
if len(tbl.Positions) != 2 {
|
||||
t.Errorf("expected 2 Positions, got %d", len(tbl.Positions))
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractTableBoxes_NoTables(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{Healthy: true, DLARegions: []DLARegion{}}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 1000, 1000))
|
||||
tables := p.extractTableBoxesFromImage(context.Background(), nil, dummy, 0, 0)
|
||||
if len(tables) != 0 {
|
||||
t.Errorf("0 tables expected, got %d", len(tables))
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractTableBoxes_NonTableRegions(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []DLARegion{
|
||||
{X0: 150, Y0: 300, X1: 1650, Y1: 336, Label: "text", Confidence: 0.9},
|
||||
{X0: 150, Y0: 600, X1: 1650, Y1: 900, Label: "figure", Confidence: 0.8},
|
||||
},
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 2000, 2000))
|
||||
tables := p.extractTableBoxesFromImage(context.Background(), nil, dummy, 0, 0)
|
||||
if len(tables) != 0 {
|
||||
t.Errorf("non-table regions → 0 tables, got %d", len(tables))
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractTableBoxes_NoOverlap(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 10, Bottom: 30, Text: "far away"},
|
||||
}
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []DLARegion{
|
||||
{X0: 150, Y0: 1500, X1: 1500, Y1: 2300, Label: "table", Confidence: 0.95},
|
||||
},
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 2000, 3000))
|
||||
tables := p.extractTableBoxesFromImage(context.Background(), boxes, dummy, 0, 0)
|
||||
if len(tables) != 0 {
|
||||
t.Errorf("no overlap → 0 tables, got %d", len(tables))
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractTableBoxes_TSRError(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
{PageNumber: 0, X0: 80, X1: 500, Top: 210, Bottom: 660, Text: "cell"},
|
||||
}
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []DLARegion{
|
||||
{X0: 250, Y0: 600, X1: 1500, Y1: 2000, Label: "table", Confidence: 0.95},
|
||||
},
|
||||
TSRCells: nil, // TSR returns nothing
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 2000, 3000))
|
||||
tables := p.extractTableBoxesFromImage(context.Background(), boxes, dummy, 0, 0)
|
||||
if len(tables) != 1 {
|
||||
t.Fatalf("TSR failure: expected 1 TableItem with image+positions, got %d", len(tables))
|
||||
}
|
||||
if tables[0].ImageB64 == "" {
|
||||
t.Error("should have image despite TSR failure")
|
||||
}
|
||||
if len(tables[0].Positions) == 0 {
|
||||
t.Error("should have positions despite TSR failure")
|
||||
}
|
||||
if len(tables[0].Rows) != 0 {
|
||||
t.Errorf("TSR failure → 0 rows, got %d", len(tables[0].Rows))
|
||||
}
|
||||
}
|
||||
|
||||
func TestGroupTSRCellsToRows_SameHeight(t *testing.T) {
|
||||
// All cells have identical height → medianH is that value → threshold = medianH/2
|
||||
cells := []TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 50, Y1: 30, Text: "A"},
|
||||
{X0: 50, Y0: 0, X1: 100, Y1: 30, Text: "B"},
|
||||
{X0: 0, Y0: 31, X1: 50, Y1: 61, Text: "C"}, // gap = 31-30=1 < 30/2=15 → same row? NO, Y0=31 is right at edge
|
||||
}
|
||||
rows := groupTSRCellsToRows(cells)
|
||||
// medianH=30, threshold=15. C.Y0=31 > curY+threshold?" curY=0, 31 > 15 → new row.
|
||||
// So A,B in row 0, C in row 1.
|
||||
if len(rows) != 2 {
|
||||
t.Fatalf("expected 2 rows, got %d", len(rows))
|
||||
}
|
||||
if len(rows[0]) != 2 || len(rows[1]) != 1 {
|
||||
t.Errorf("row sizes: %d %d, want 2 1", len(rows[0]), len(rows[1]))
|
||||
}
|
||||
}
|
||||
|
||||
func TestFillCellTextFromBoxes_WhitespaceTrim(t *testing.T) {
|
||||
cells := []TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 100}}
|
||||
boxes := []TextBox{{X0: 0, X1: 100, Top: 0, Bottom: 100, Text: " hello "}}
|
||||
fillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "hello" {
|
||||
t.Errorf("got %q, want 'hello'", cells[0].Text)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFillCellTextFromBoxes_EmptyBoxIgnored(t *testing.T) {
|
||||
cells := []TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 100}}
|
||||
boxes := []TextBox{{X0: 0, X1: 100, Top: 0, Bottom: 100, Text: " "}} // all whitespace
|
||||
fillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "" {
|
||||
t.Errorf("whitespace text should produce empty, got %q", cells[0].Text)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractTableBoxes_DLAError(t *testing.T) {
|
||||
// DLA returns only non-table regions → 0 tables
|
||||
mock := &MockDocAnalyzer{Healthy: true, DLARegions: []DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 100, Label: "text", Confidence: 0.9},
|
||||
}}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 1000, 1000))
|
||||
tables := p.extractTableBoxesFromImage(context.Background(), nil, dummy, 0, 0)
|
||||
if len(tables) != 0 {
|
||||
t.Errorf("non-table DLA → 0 tables, got %d", len(tables))
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
{X0: 50, X1: 200, Top: 100, Bottom: 200, Text: "title text"},
|
||||
{X0: 250, X1: 500, Top: 100, Bottom: 200, Text: "body"},
|
||||
{X0: 50, X1: 500, Top: 300, Bottom: 600, Text: "table content"},
|
||||
{X0: 50, X1: 500, Top: 700, Bottom: 800, Text: "unmatched"},
|
||||
}
|
||||
regions := []DLARegion{
|
||||
{X0: 150, Y0: 300, X1: 600, Y1: 600, Label: "title", Confidence: 0.9}, // PDF pts: X50-200,Y100-200 → only box[0]
|
||||
{X0: 750, Y0: 300, X1: 1500, Y1: 600, Label: "text", Confidence: 0.8}, // PDF pts: X250-500,Y100-200 → box[1]
|
||||
{X0: 150, Y0: 900, X1: 1500, Y1: 1800, Label: "table", Confidence: 0.95}, // PDF pts: X50-500,Y300-600 → box[2]
|
||||
}
|
||||
scale := 3.0
|
||||
annotateBoxLayouts(boxes, regions, scale, 0)
|
||||
|
||||
if boxes[0].LayoutType != "title" {
|
||||
t.Errorf("box[0] = %q, want title", boxes[0].LayoutType)
|
||||
}
|
||||
if boxes[1].LayoutType != "text" {
|
||||
t.Errorf("box[1] = %q, want text", boxes[1].LayoutType)
|
||||
}
|
||||
if boxes[2].LayoutType != "table" {
|
||||
t.Errorf("box[2] = %q, want table", boxes[2].LayoutType)
|
||||
}
|
||||
if boxes[3].LayoutType != "" {
|
||||
t.Errorf("box[3] = %q, want empty (no matching region)", boxes[3].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_Figure(t *testing.T) {
|
||||
// Figure region → box gets "figure" layout type (no TSR needed)
|
||||
boxes := []TextBox{
|
||||
{X0: 50, X1: 500, Top: 100, Bottom: 400, Text: "chart image"},
|
||||
}
|
||||
regions := []DLARegion{
|
||||
{X0: 50, Y0: 200, X1: 2000, Y1: 1000, Label: "figure", Confidence: 0.85},
|
||||
}
|
||||
annotateBoxLayouts(boxes, regions, 3.0, 0)
|
||||
if boxes[0].LayoutType != "figure" {
|
||||
t.Errorf("LayoutType = %q, want 'figure'", boxes[0].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_Empty(t *testing.T) {
|
||||
boxes := []TextBox{{Text: "x"}}
|
||||
annotateBoxLayouts(boxes, nil, 3.0, 0)
|
||||
if boxes[0].LayoutType != "" {
|
||||
t.Error("empty regions → no annotation")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBoxesToSections_PassesLayoutType(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 100, Bottom: 112, Text: "标题", LayoutType: "title"},
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 200, Bottom: 212, Text: "表格", LayoutType: "table"},
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 300, Bottom: 312, Text: "正文", LayoutType: "text"},
|
||||
}
|
||||
sections := boxesToSections(boxes, nil)
|
||||
if len(sections) != 3 {
|
||||
t.Fatalf("expected 3 sections, got %d", len(sections))
|
||||
}
|
||||
if sections[0].LayoutType != "title" {
|
||||
t.Errorf("section[0].LayoutType = %q, want 'title'", sections[0].LayoutType)
|
||||
}
|
||||
if sections[1].LayoutType != "table" {
|
||||
t.Errorf("section[1].LayoutType = %q, want 'table'", sections[1].LayoutType)
|
||||
}
|
||||
if sections[2].LayoutType != "text" {
|
||||
t.Errorf("section[2].LayoutType = %q, want 'text'", sections[2].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBoxesToSections_PreservesTableLayout(t *testing.T) {
|
||||
// boxesToSections should produce sections for all boxes regardless of LayoutType.
|
||||
boxes := []TextBox{
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 100, Bottom: 112, Text: "标题", LayoutType: "title"},
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 200, Bottom: 212, Text: "表格文字", LayoutType: "table"},
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 300, Bottom: 312, Text: "正文", LayoutType: "text"},
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 400, Bottom: 412, Text: ""},
|
||||
}
|
||||
sections := boxesToSections(boxes, nil)
|
||||
if len(sections) != 3 {
|
||||
t.Errorf("expected 3 sections (1 empty skipped), got %d", len(sections))
|
||||
}
|
||||
for _, s := range sections {
|
||||
if strings.Contains(s.Text, "@@") {
|
||||
t.Error("section text should NOT contain position tag")
|
||||
}
|
||||
}
|
||||
t.Logf("boxesToSections: %d sections (all LayoutTypes passed through)", len(sections))
|
||||
}
|
||||
|
||||
func TestEnrichWithDeepDoc_PreservesBoxes(t *testing.T) {
|
||||
// Simulate enrichWithDeepDoc's write-back logic:
|
||||
// 1. Create pageBoxes as copies of p.boxes[idx]
|
||||
// 2. annotateBoxLayouts(pageBoxes, regions) — modifies copies
|
||||
// 3. Write LayoutType back to p.boxes[idx]
|
||||
// This test validates step 3 works.
|
||||
|
||||
original := []TextBox{
|
||||
{PageNumber: 0, X0: 50, X1: 200, Top: 50, Bottom: 80, Text: "title", LayoutType: ""},
|
||||
{PageNumber: 0, X0: 50, X1: 200, Top: 100, Bottom: 200, Text: "text before", LayoutType: ""},
|
||||
{PageNumber: 0, X0: 50, X1: 500, Top: 250, Bottom: 700, Text: "table cell", LayoutType: ""},
|
||||
{PageNumber: 0, X0: 50, X1: 200, Top: 750, Bottom: 800, Text: "text after", LayoutType: ""},
|
||||
{PageNumber: 1, X0: 50, X1: 200, Top: 50, Bottom: 80, Text: "page2", LayoutType: ""},
|
||||
}
|
||||
|
||||
byPage := map[int][]int{0: {0, 1, 2, 3}, 1: {4}} // indices into original
|
||||
|
||||
regions := []DLARegion{
|
||||
{X0: 150, Y0: 150, X1: 600, Y1: 240, Label: "title", Confidence: 0.9}, // PDF: X50-200,Y50-80 → box[0]
|
||||
{X0: 150, Y0: 750, X1: 1500, Y1: 2100, Label: "table", Confidence: 0.95}, // PDF: X50-500,Y250-700 → box[2]
|
||||
}
|
||||
|
||||
// Step 1-2: copy + annotate
|
||||
for _, indices := range byPage {
|
||||
pageBoxes := make([]TextBox, len(indices))
|
||||
for i, idx := range indices {
|
||||
pageBoxes[i] = original[idx]
|
||||
}
|
||||
annotateBoxLayouts(pageBoxes, regions, 3.0, 0)
|
||||
|
||||
// Step 3: write back (this is what enrichWithDeepDoc now does)
|
||||
for i, idx := range indices {
|
||||
if pageBoxes[i].LayoutType != "" {
|
||||
original[idx].LayoutType = pageBoxes[i].LayoutType
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if original[0].LayoutType != "title" {
|
||||
t.Errorf("box[0] LayoutType = %q, want 'title'", original[0].LayoutType)
|
||||
}
|
||||
if original[2].LayoutType != "table" {
|
||||
t.Errorf("box[2] LayoutType = %q, want 'table'", original[2].LayoutType)
|
||||
}
|
||||
if original[1].LayoutType != "" {
|
||||
t.Errorf("box[1] LayoutType = %q, want '' (no matching region)", original[1].LayoutType)
|
||||
}
|
||||
// All boxes still present
|
||||
if len(original) != 5 {
|
||||
t.Errorf("all boxes preserved: got %d, want 5", len(original))
|
||||
}
|
||||
t.Logf("Write-back verified: box[0]=%q box[2]=%q", original[0].LayoutType, original[2].LayoutType)
|
||||
}
|
||||
|
||||
func TestBoxesToSections_PositionsFromTag(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 100, Bottom: 112, Text: "标题段落"},
|
||||
}
|
||||
sections := boxesToSections(boxes, nil)
|
||||
if sections[0].PositionTag == "" {
|
||||
t.Error("PositionTag should not be empty")
|
||||
}
|
||||
if len(sections[0].Positions) == 0 {
|
||||
t.Error("Positions should be parsed from PositionTag — BUG: ExtractPositions not called")
|
||||
}
|
||||
if len(sections[0].Positions) > 0 {
|
||||
pos := sections[0].Positions[0]
|
||||
if pos.Left != 50 || pos.Right != 550 || pos.Top != 100 || pos.Bottom != 112 {
|
||||
t.Errorf("position coords wrong: got (%.0f,%.0f,%.0f,%.0f)", pos.Left, pos.Right, pos.Top, pos.Bottom)
|
||||
}
|
||||
}
|
||||
t.Logf("Positions: %v", sections[0].Positions)
|
||||
}
|
||||
|
||||
func TestParse_TableLinkedToSections(t *testing.T) {
|
||||
// Simulate enrichWithDeepDoc → extractTableAndReplace → boxesToSections:
|
||||
// table boxes are popped and replaced with one HTML box.
|
||||
boxes := []TextBox{
|
||||
{PageNumber: 0, X0: 50, X1: 200, Top: 50, Bottom: 80, Text: "heading"},
|
||||
{PageNumber: 0, X0: 50, X1: 500, Top: 250, Bottom: 400, Text: "table text", LayoutType: "table"},
|
||||
{PageNumber: 0, X0: 50, X1: 200, Top: 450, Bottom: 480, Text: "after"},
|
||||
}
|
||||
tableItem := TableItem{
|
||||
Cells: []TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 200, Y1: 50, Label: "table row"},
|
||||
{X0: 0, Y0: 51, X1: 200, Y1: 100, Label: "table row"},
|
||||
},
|
||||
Positions: []Position{{PageNumbers: []int{0}, Left: 50, Right: 500, Top: 250, Bottom: 400}},
|
||||
Scale: 1.0,
|
||||
}
|
||||
|
||||
boxes = extractTableAndReplace(boxes, []TableItem{tableItem})
|
||||
sections := boxesToSections(boxes, nil)
|
||||
|
||||
// 3 boxes (heading, table, after) → 3 sections (heading, HTML, after).
|
||||
if len(sections) != 3 {
|
||||
t.Errorf("expected 3 sections, got %d", len(sections))
|
||||
}
|
||||
tableFound := false
|
||||
for _, s := range sections {
|
||||
if s.LayoutType == "table" && strings.Contains(s.Text, "<table>") {
|
||||
tableFound = true
|
||||
}
|
||||
}
|
||||
if !tableFound {
|
||||
t.Errorf("expected at least one section with HTML table")
|
||||
for _, s := range sections {
|
||||
t.Logf(" section text=%q LayoutType=%q", s.Text[:min(40, len(s.Text))], s.LayoutType)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func cellTexts(cells []TSRCell) []string {
|
||||
t := make([]string, len(cells))
|
||||
for i, c := range cells {
|
||||
t[i] = c.Text
|
||||
}
|
||||
return t
|
||||
}
|
||||
|
||||
// ── cropImageRegion ────────────────────────────────────────────────────
|
||||
|
||||
func TestCropImageRegion(t *testing.T) {
|
||||
img := image.NewRGBA(image.Rect(0, 0, 200, 300))
|
||||
|
||||
t.Run("normal crop", func(t *testing.T) {
|
||||
r := DLARegion{X0: 10, Y0: 20, X1: 100, Y1: 150}
|
||||
cropped, err := cropImageRegion(img, r)
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
// 3% proportional margin: 90×3%≈3px, 130×3%≈4px → 95×137
|
||||
if cropped.Bounds().Dx() != 95 || cropped.Bounds().Dy() != 137 {
|
||||
t.Errorf("size %v, want 95x137", cropped.Bounds())
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("x0 >= x1 returns error", func(t *testing.T) {
|
||||
// 3% proportional margin on each side: if the gap is too small after margin expansion, x0 ≥ x1 triggers error.
|
||||
r := DLARegion{X0: 110, Y0: 20, X1: 50, Y1: 150}
|
||||
_, err := cropImageRegion(img, r)
|
||||
if err == nil {
|
||||
t.Fatal("expected error for x0 >= x1, got nil")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("y0 >= y1 returns error", func(t *testing.T) {
|
||||
r := DLARegion{X0: 10, Y0: 150, X1: 100, Y1: 20}
|
||||
_, err := cropImageRegion(img, r)
|
||||
if err == nil {
|
||||
t.Fatal("expected error for y0 >= y1, got nil")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("region fully outside image bounds", func(t *testing.T) {
|
||||
// Clamped to image bounds → zero-width/height → error.
|
||||
r := DLARegion{X0: 300, Y0: 400, X1: 500, Y1: 600}
|
||||
_, err := cropImageRegion(img, r)
|
||||
if err == nil {
|
||||
t.Fatal("expected error for region outside image bounds")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// ── extractTableBoxesFromImage: invalid DLA region ─────────────────────
|
||||
|
||||
func TestExtractTableBoxes_InvalidRegion(t *testing.T) {
|
||||
// DLA returns a table region with x1 < x0. The pipeline should skip
|
||||
// this table gracefully (Python raises ValueError from PIL.Image.crop).
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []DLARegion{
|
||||
{X0: 500, Y0: 100, X1: 100, Y1: 300, Label: "table", Confidence: 0.9},
|
||||
},
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 1000, 1000))
|
||||
tables := p.extractTableBoxesFromImage(context.Background(), nil, dummy, 0, 0)
|
||||
if len(tables) != 0 {
|
||||
t.Errorf("invalid DLA region should be skipped, got %d tables", len(tables))
|
||||
}
|
||||
}
|
||||
|
||||
// ── DLA → figure end-to-end ───────────────────────────────────────────
|
||||
|
||||
func TestParse_CollectsFigures(t *testing.T) {
|
||||
// End-to-end: Parse() with mock DeepDoc that labels a box as "figure".
|
||||
// Verify p.Figures is populated.
|
||||
|
||||
eng := &mockEngine{pageCount: 1, chars: map[int][]TextChar{0: {{X0: 50, X1: 550, Top: 100, Bottom: 112, Text: "chart image"}}}}
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []DLARegion{
|
||||
{X0: 50, Y0: 200, X1: 2000, Y1: 1000, Label: "figure", Confidence: 0.85},
|
||||
},
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
if len(result.Sections) == 0 {
|
||||
t.Fatal("expected at least 1 section")
|
||||
}
|
||||
if len(result.Figures) != 1 {
|
||||
t.Fatalf("expected 1 figure, got %d", len(result.Figures))
|
||||
}
|
||||
if result.Figures[0].LayoutType != "figure" {
|
||||
t.Errorf("figure LayoutType = %q, want 'figure'", result.Figures[0].LayoutType)
|
||||
}
|
||||
if result.Figures[0].Text == "" {
|
||||
t.Error("figure Text should not be empty")
|
||||
}
|
||||
}
|
||||
|
||||
func TestParse_NoFigures(t *testing.T) {
|
||||
// Parse() with no DLA figure regions → p.Figures should be empty.
|
||||
|
||||
eng := &mockEngine{pageCount: 1, chars: map[int][]TextChar{0: {{X0: 50, X1: 550, Top: 100, Bottom: 112, Text: "just text"}}}}
|
||||
mock := &MockDocAnalyzer{
|
||||
DLARegions: []DLARegion{
|
||||
{X0: 150, Y0: 300, X1: 1500, Y1: 600, Label: "text", Confidence: 0.8},
|
||||
},
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
if len(result.Figures) != 0 {
|
||||
t.Fatalf("expected 0 figures, got %d", len(result.Figures))
|
||||
}
|
||||
}
|
||||
|
||||
func TestParse_NoDeepDoc_NoFigures(t *testing.T) {
|
||||
// Parse() with mock DeepDoc → Figures should be empty (no DLA-detected figures).
|
||||
|
||||
eng := &mockEngine{pageCount: 1, chars: map[int][]TextChar{0: {{X0: 50, X1: 550, Top: 100, Bottom: 112, Text: "text"}}}}
|
||||
p := NewParser(DefaultParserConfig(), &MockDocAnalyzer{Healthy: true, Model: ModelSaas})
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
if len(result.Figures) != 0 {
|
||||
t.Fatalf("expected 0 Figures (no DLA-detected figures), got %d", len(result.Figures))
|
||||
}
|
||||
}
|
||||
|
||||
// ── Parse + ocrMergeChars (full-page detect) ──────────────────────────
|
||||
|
||||
func TestParse_UsesOCRDetectForEmbeddedChars(t *testing.T) {
|
||||
// When DeepDoc is available and the page has embedded chars,
|
||||
// Parse should use ocrMergeChars (detect → merge → recognize).
|
||||
eng := &mockEngine{
|
||||
pageCount: 1,
|
||||
chars: map[int][]TextChar{0: {
|
||||
{X0: 10, X1: 30, Top: 10, Bottom: 30, Text: "Hello", PageNumber: 0},
|
||||
}},
|
||||
}
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRBoxes: []OCRBox{
|
||||
{X0: 5, Y0: 5, X1: 50, Y1: 5, X2: 50, Y2: 50, X3: 5, Y3: 50},
|
||||
},
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
if len(result.Sections) == 0 {
|
||||
t.Fatal("expected at least 1 section")
|
||||
}
|
||||
// The box should come from OCR detect, not charsToBoxes.
|
||||
// Verifying that ocrMergeChars was used (sections exist).
|
||||
if result.Metrics.BoxesInitial == 0 {
|
||||
t.Error("expected BoxesInitial > 0 (OCR detect path)")
|
||||
}
|
||||
}
|
||||
|
||||
func TestParse_FallsBackToCharsToBoxes_NoDeepDoc(t *testing.T) {
|
||||
// Without DeepDoc, Parse should use charsToBoxes (unchanged behavior).
|
||||
eng := &mockEngine{
|
||||
pageCount: 1,
|
||||
chars: map[int][]TextChar{0: {
|
||||
{X0: 10, X1: 30, Top: 10, Bottom: 30, Text: "Hello", PageNumber: 0},
|
||||
}},
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), &MockDocAnalyzer{Healthy: true, Model: ModelSaas})
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
if len(result.Sections) == 0 {
|
||||
t.Fatal("expected at least 1 section (charsToBoxes)")
|
||||
}
|
||||
}
|
||||
|
||||
func TestParse_FallsBackToCharsToBoxes_EmptyOCRBoxes(t *testing.T) {
|
||||
// OCRDetect returns no boxes → falls through to charsToBoxes.
|
||||
eng := &mockEngine{
|
||||
pageCount: 1,
|
||||
chars: map[int][]TextChar{0: {
|
||||
{X0: 10, X1: 30, Top: 10, Bottom: 30, Text: "Hello", PageNumber: 0},
|
||||
}},
|
||||
}
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRBoxes: []OCRBox{}, // empty detect
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
if len(result.Sections) == 0 {
|
||||
t.Fatal("expected at least 1 section (charsToBoxes fallback)")
|
||||
}
|
||||
}
|
||||
|
||||
// ── Error path coverage ────────────────────────────────────────────────
|
||||
|
||||
func TestMockDocAnalyzer_DLAError_DoesNotCrash(t *testing.T) {
|
||||
p := NewParser(DefaultParserConfig(), &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLAErr: fmt.Errorf("DLA service unavailable"),
|
||||
})
|
||||
eng := &mockEngine{pageCount: 1}
|
||||
img := image.NewRGBA(image.Rect(0, 0, 100, 100))
|
||||
pageImages := map[int]image.Image{0: img}
|
||||
boxes := []TextBox{
|
||||
{PageNumber: 0, X0: 0, X1: 100, Top: 0, Bottom: 50, Text: "text"},
|
||||
}
|
||||
// enrichWithDeepDoc should return nil (not panic) on DLA error.
|
||||
tables := p.enrichWithDeepDoc(context.Background(), eng, boxes, pageImages)
|
||||
if len(tables) != 0 {
|
||||
t.Errorf("DLA error should produce 0 tables, got %d", len(tables))
|
||||
}
|
||||
}
|
||||
|
||||
func TestMockDocAnalyzer_TSRError_DoesNotCrash(t *testing.T) {
|
||||
// TSR error: DLA succeeds, TSR fails. The table region is detected
|
||||
// but no cells are returned — the table is skipped gracefully.
|
||||
p := NewParser(DefaultParserConfig(), &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 400, Y1: 400, Label: "table", Confidence: 0.95},
|
||||
},
|
||||
TSRErr: fmt.Errorf("TSR model timeout"),
|
||||
})
|
||||
eng := &mockEngine{pageCount: 1}
|
||||
img := image.NewRGBA(image.Rect(0, 0, 100, 100))
|
||||
pageImages := map[int]image.Image{0: img}
|
||||
boxes := []TextBox{
|
||||
{PageNumber: 0, X0: 10, X1: 90, Top: 10, Bottom: 90, Text: "in table region"},
|
||||
}
|
||||
tables := p.enrichWithDeepDoc(context.Background(), eng, boxes, pageImages)
|
||||
// DLA detects the table region → 1 TableItem is created. TSR failure
|
||||
// means it has no cells, but the pipeline must not panic.
|
||||
if len(tables) != 1 {
|
||||
t.Errorf("TSR error: expected 1 table (DLA region found), got %d", len(tables))
|
||||
}
|
||||
if len(tables[0].Cells) != 0 {
|
||||
t.Errorf("TSR error: Cells should be empty, got %d", len(tables[0].Cells))
|
||||
}
|
||||
}
|
||||
|
||||
func TestMockDocAnalyzer_OCRDetectError_DoesNotCrash(t *testing.T) {
|
||||
// OCRDetect failure path: extractPages uses ocrDetectAndRecognize which
|
||||
// calls doc.OCRDetect. When it fails, the page is skipped gracefully.
|
||||
mock := &MockDocAnalyzer{Healthy: true, OCRDetectErr: fmt.Errorf("OCR model OOM")}
|
||||
eng := &mockEngine{
|
||||
pageCount: 1,
|
||||
chars: map[int][]TextChar{}, // empty → triggers OCR path
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
_, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse returned error: %v", err)
|
||||
}
|
||||
// Parse should succeed — the page with OCRDetect error is just skipped.
|
||||
}
|
||||
|
||||
// TestTSRLabels verifies Go defaultTSRLabels matches Python's table_structure_recognizer.py labels.
|
||||
// Order must be exact — the ONNX model returns class IDs that index into this array.
|
||||
func TestTSRLabels(t *testing.T) {
|
||||
want := []string{
|
||||
"table", "table column", "table row",
|
||||
"table column header", "table projected row header",
|
||||
"table spanning cell",
|
||||
}
|
||||
if len(defaultTSRLabels) != len(want) {
|
||||
t.Fatalf("defaultTSRLabels length %d, want %d", len(defaultTSRLabels), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if defaultTSRLabels[i] != want[i] {
|
||||
t.Errorf("defaultTSRLabels[%d] = %q, want %q", i, defaultTSRLabels[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -12,7 +12,7 @@ import (
|
||||
// TestDLARealWorldCompare runs DLA on fixture PDFs and verifies
|
||||
// region count, label types, and structural invariants.
|
||||
func TestDLARealWorldCompare(t *testing.T) {
|
||||
client := mustConnectDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
outDir := filepath.Join("testdata", "output", "render_compare")
|
||||
os.MkdirAll(outDir, 0755)
|
||||
|
||||
@@ -9,6 +9,8 @@ import (
|
||||
"image/png"
|
||||
"os"
|
||||
"path/filepath"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
util "ragflow/internal/deepdoc/parser/pdf/util"
|
||||
"testing"
|
||||
)
|
||||
|
||||
@@ -22,7 +24,7 @@ import (
|
||||
// 2. Run Python: python3 tools/dla_tsr_compare.py
|
||||
// 3. Diff the JSON: diff testdata/output/render_compare/go_dla.json testdata/output/render_compare/py_dla.json
|
||||
func TestDLATSRResponseCompare(t *testing.T) {
|
||||
client := mustConnectDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
eng := mustOpenEngine(t, "06_table_content.pdf")
|
||||
defer eng.Close()
|
||||
|
||||
@@ -35,7 +37,7 @@ func TestDLATSRResponseCompare(t *testing.T) {
|
||||
os.MkdirAll(outDir, 0755)
|
||||
|
||||
// Save rendered image as JPEG (matching what DLA/TSR actually send).
|
||||
jpegData, err := encodeJPEG(pageImg)
|
||||
jpegData, err := util.EncodeJPEG(pageImg)
|
||||
if err != nil {
|
||||
t.Fatalf("encode jpeg: %v", err)
|
||||
}
|
||||
@@ -57,7 +59,7 @@ func TestDLATSRResponseCompare(t *testing.T) {
|
||||
}
|
||||
|
||||
// ── TSR (crop first table region) ──
|
||||
var tableRegion *DLARegion
|
||||
var tableRegion *pdf.DLARegion
|
||||
for i := range regions {
|
||||
if regions[i].Label == "table" {
|
||||
tableRegion = ®ions[i]
|
||||
@@ -72,7 +74,7 @@ func TestDLATSRResponseCompare(t *testing.T) {
|
||||
int(tableRegion.X1), int(tableRegion.Y1))
|
||||
|
||||
cropPath := filepath.Join(outDir, "tsr_input.jpeg")
|
||||
cropJPEG, _ := encodeJPEG(cropped)
|
||||
cropJPEG, _ := util.EncodeJPEG(cropped)
|
||||
os.WriteFile(cropPath, cropJPEG, 0644)
|
||||
|
||||
cells, err := client.TSR(context.Background(), cropped)
|
||||
@@ -104,7 +106,7 @@ func TestDLATSRResponseCompare(t *testing.T) {
|
||||
int(b.X0), int(b.Y0), int(b.X2), int(b.Y2))
|
||||
|
||||
cropPath := filepath.Join(outDir, "ocr_rec_input.jpeg")
|
||||
recJPEG, _ := encodeJPEG(cropped)
|
||||
recJPEG, _ := util.EncodeJPEG(cropped)
|
||||
os.WriteFile(cropPath, recJPEG, 0644)
|
||||
|
||||
texts, err := client.OCRRecognize(context.Background(), cropped)
|
||||
|
||||
@@ -1,230 +0,0 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestIsGarbledChar(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
ch string
|
||||
want bool
|
||||
}{
|
||||
{"empty", "", false},
|
||||
{"normal ascii", "A", false},
|
||||
{"normal chinese", "你", false},
|
||||
{"PUA char E000", "", true},
|
||||
{"PUA char F8FF", "", true},
|
||||
{"replacement char", "<22>", true},
|
||||
{"null control", "\x00", true},
|
||||
{"tab", "\t", false},
|
||||
{"newline", "\n", false},
|
||||
{"C1 control", "", true},
|
||||
{"C1 control 9F", "", true},
|
||||
{"normal single byte", "z", false},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := IsGarbledChar(tt.ch)
|
||||
if got != tt.want {
|
||||
t.Errorf("IsGarbledChar(%q) = %v, want %v", tt.ch, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestIsGarbledText(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
text string
|
||||
threshold float64
|
||||
want bool
|
||||
}{
|
||||
{"empty", "", 0.5, false},
|
||||
{"normal text", "正常文本", 0.5, false},
|
||||
{"cid pattern", "(cid:123)", 0.5, true},
|
||||
{"all garbled", "", 0.5, true},
|
||||
{"one garbled in many", "ABDEFGHI", 0.5, false},
|
||||
{"half garbled strict", "AB", 0.5, true},
|
||||
{"half garbled loose", "AB", 0.7, false},
|
||||
{"english text", "Hello World", 0.5, false},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := IsGarbledText(tt.text, tt.threshold)
|
||||
if got != tt.want {
|
||||
t.Errorf("IsGarbledText(%q, %v) = %v, want %v", tt.text, tt.threshold, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestHasSubsetFontPrefix(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
fontName string
|
||||
want bool
|
||||
}{
|
||||
{"subset prefix", "DY1+ZLQDm1-1", true},
|
||||
{"short subset", "AB+SimSun", true},
|
||||
{"no prefix", "SimSun", false},
|
||||
{"empty", "", false},
|
||||
{"just plus", "+SimSun", false},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := HasSubsetFontPrefix(tt.fontName)
|
||||
if got != tt.want {
|
||||
t.Errorf("HasSubsetFontPrefix(%q) = %v, want %v", tt.fontName, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestIsGarbledByFontEncoding(t *testing.T) {
|
||||
t.Run("too few chars", func(t *testing.T) {
|
||||
chars := make([]TextChar, 10)
|
||||
if IsGarbledByFontEncoding(chars, 20) {
|
||||
t.Error("should return false when below minChars threshold")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("subset font with ascii — garbled", func(t *testing.T) {
|
||||
// Simulate CJK PDF with broken font encoding: all chars have subset font prefix,
|
||||
// virtually no CJK, almost all ASCII punctuation
|
||||
var chars []TextChar
|
||||
for i := 0; i < 30; i++ {
|
||||
chars = append(chars, TextChar{
|
||||
Text: "!",
|
||||
FontName: "DY1+SimSun",
|
||||
})
|
||||
}
|
||||
// Add some CJK (but below 5%)
|
||||
chars = append(chars, TextChar{Text: "你", FontName: "DY1+SimSun"})
|
||||
if !IsGarbledByFontEncoding(chars, 20) {
|
||||
t.Error("should detect garbled font encoding")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("regular CJK text — not garbled", func(t *testing.T) {
|
||||
var chars []TextChar
|
||||
for i := 0; i < 30; i++ {
|
||||
chars = append(chars, TextChar{
|
||||
Text: "测试文本内容",
|
||||
FontName: "SimSun",
|
||||
})
|
||||
}
|
||||
if IsGarbledByFontEncoding(chars, 20) {
|
||||
t.Error("should not flag regular CJK text as garbled")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("fullwidth chars from subset font — not garbled", func(t *testing.T) {
|
||||
// Fullwidth characters (U+FF01-U+FF5E) are legitimate CJK typographic forms.
|
||||
// They should count as cjkLike, preventing false garbled detection.
|
||||
var chars []TextChar
|
||||
for i := 0; i < 30; i++ {
|
||||
chars = append(chars, TextChar{
|
||||
Text: "ABCDEF", // U+FF21-U+FF26 fullwidth uppercase
|
||||
FontName: "DY1+SimSun",
|
||||
})
|
||||
}
|
||||
if IsGarbledByFontEncoding(chars, 20) {
|
||||
t.Error("fullwidth chars from subset font should NOT be garbled")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("normal English text — not garbled", func(t *testing.T) {
|
||||
var chars []TextChar
|
||||
for i := 0; i < 30; i++ {
|
||||
chars = append(chars, TextChar{
|
||||
Text: "Hello world text content here",
|
||||
FontName: "Times-Roman",
|
||||
})
|
||||
}
|
||||
if IsGarbledByFontEncoding(chars, 20) {
|
||||
t.Error("should not flag regular English text as garbled")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestDetectGarbled(t *testing.T) {
|
||||
// Normal CJK text
|
||||
chars := make([]TextChar, 30)
|
||||
for i := range chars {
|
||||
chars[i] = TextChar{Text: "正常文本", FontName: "SimSun"}
|
||||
}
|
||||
if DetectGarbled(chars) {
|
||||
t.Error("normal CJK should not be garbled")
|
||||
}
|
||||
|
||||
// Subset font with punctuation
|
||||
var garbled []TextChar
|
||||
for i := 0; i < 30; i++ {
|
||||
garbled = append(garbled, TextChar{Text: "!", FontName: "DY1+SimSun"})
|
||||
}
|
||||
if !DetectGarbled(garbled) {
|
||||
t.Error("subset font with punctuation should be garbled")
|
||||
}
|
||||
}
|
||||
|
||||
// ── pdf_oxide ### detection tests ─────────────────────────────────────
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_Empty(t *testing.T) {
|
||||
if pdfOxideUnmappedGarbled("") {
|
||||
t.Error("empty text should not be garbled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_NormalText(t *testing.T) {
|
||||
if pdfOxideUnmappedGarbled("这是一段正常的中文文本没有任何问题") {
|
||||
t.Error("normal Chinese text should not be garbled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_SingleHash(t *testing.T) {
|
||||
// A single # is not enough (could be a phone number or reference).
|
||||
if pdfOxideUnmappedGarbled("参考 #123 的文献") {
|
||||
t.Error("single # should not be garbled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_TripleHashCluster(t *testing.T) {
|
||||
// Two ### sequences => garbled.
|
||||
if !pdfOxideUnmappedGarbled("我信###D_8-.###$#(") {
|
||||
t.Error("two ### clusters should be garbled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_QuadHash(t *testing.T) {
|
||||
// One #### counts as one ### cluster. Need two for trigger.
|
||||
// But density may also be high enough.
|
||||
if !pdfOxideUnmappedGarbled("text####abc####def") {
|
||||
t.Error("two #### clusters should be garbled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_SingleTriple(t *testing.T) {
|
||||
// Single ### cluster => garbled. In a 200-char sample "###" is impossible
|
||||
// in normal text (URLs/markdown use at most "##").
|
||||
if !pdfOxideUnmappedGarbled("hello###world normal text here") {
|
||||
t.Error("single ### cluster should be garbled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_HighDensity(t *testing.T) {
|
||||
// 10 # chars mixed among 40+ non-space chars = 25% → garbled.
|
||||
text := "#a#b#c#d#e#f#g#h#i#j" + " extra normal chars padding to reach minimum"
|
||||
if !pdfOxideUnmappedGarbled(text) {
|
||||
t.Error("high # density should be garbled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_RealWorldGarbled(t *testing.T) {
|
||||
// Simulates the garbled page from 1例3个月...pdf:
|
||||
// Chinese text mixed with ###D_ style unmapped glyph patterns.
|
||||
garbled := "和蔘语言###D_8-.*/*护理全科##%&$ 80引用\"\"###$#(点向患儿"
|
||||
if !pdfOxideUnmappedGarbled(garbled) {
|
||||
t.Error("real-world garbled text with ### clusters should be detected")
|
||||
}
|
||||
}
|
||||
@@ -1,26 +0,0 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"image"
|
||||
"image/jpeg"
|
||||
"image/png"
|
||||
)
|
||||
|
||||
// ── image encoding helpers ─────────────────────────────────────────────
|
||||
|
||||
func encodePNG(img image.Image) ([]byte, error) {
|
||||
var buf bytes.Buffer
|
||||
if err := png.Encode(&buf, img); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return buf.Bytes(), nil
|
||||
}
|
||||
|
||||
func encodeJPEG(img image.Image) ([]byte, error) {
|
||||
var buf bytes.Buffer
|
||||
if err := jpeg.Encode(&buf, img, &jpeg.Options{Quality: 90}); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return buf.Bytes(), nil
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
package parser
|
||||
package inference
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
@@ -15,29 +15,33 @@ import (
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
util "ragflow/internal/deepdoc/parser/pdf/util"
|
||||
|
||||
"github.com/cenkalti/backoff/v5"
|
||||
)
|
||||
|
||||
// DeepDocClient wraps the DeepDoc HTTP API.
|
||||
type DeepDocClient struct {
|
||||
// InferenceClient wraps the DeepDoc HTTP API.
|
||||
type InferenceClient struct {
|
||||
baseURL string
|
||||
httpClient *http.Client
|
||||
modelOnce sync.Once
|
||||
model ModelType
|
||||
|
||||
// Label tables for class_id → label string mapping.
|
||||
// Set by the service layer (Oss/Saas) to reflect the model's taxonomy.
|
||||
// Set by the service layer (model-specific) to reflect the model's taxonomy.
|
||||
DLALabels []string
|
||||
TSRLabels []string
|
||||
}
|
||||
|
||||
// NewDeepDocClient creates a client. baseURL must be provided by the caller
|
||||
// BaseURL returns the configured DeepDoc service URL.
|
||||
func (c *InferenceClient) BaseURL() string { return c.baseURL }
|
||||
|
||||
// NewInferenceClient creates a client. baseURL must be provided by the caller
|
||||
// (e.g. from the DEEPDOC_URL environment variable). Returns an error if empty.
|
||||
func NewDeepDocClient(baseURL string) (*DeepDocClient, error) {
|
||||
func NewInferenceClient(baseURL string) (*InferenceClient, error) {
|
||||
if baseURL == "" {
|
||||
return nil, fmt.Errorf("deepdoc client: baseURL is required (set DEEPDOC_URL)")
|
||||
}
|
||||
return &DeepDocClient{
|
||||
return &InferenceClient{
|
||||
baseURL: baseURL,
|
||||
httpClient: &http.Client{
|
||||
Timeout: 120 * time.Second,
|
||||
@@ -45,27 +49,31 @@ func NewDeepDocClient(baseURL string) (*DeepDocClient, error) {
|
||||
}, nil
|
||||
}
|
||||
|
||||
// Default DLA/TSR label tables. Service constructors replace these with
|
||||
// model-specific labels (OSS 6-class TSR, SaaS 2-class, etc.).
|
||||
var defaultDLALabels = []string{
|
||||
LayoutTypeTitle, LayoutTypeText, LayoutTypeReference,
|
||||
LayoutTypeFigure, DLALabelFigureCaption,
|
||||
LayoutTypeTable, DLALabelTableCaption, DLALabelTableCaption,
|
||||
LayoutTypeEquation, DLALabelFigureCaption,
|
||||
// Default DLA/TSR label tables used as fallback when no model-specific
|
||||
// labels are injected by a TableBuilder constructor.
|
||||
func DefaultDLALabels() []string {
|
||||
return []string{
|
||||
pdf.LayoutTypeTitle, pdf.LayoutTypeText, pdf.LayoutTypeReference,
|
||||
pdf.LayoutTypeFigure, pdf.DLALabelFigureCaption,
|
||||
pdf.LayoutTypeTable, pdf.DLALabelTableCaption, pdf.DLALabelTableCaption,
|
||||
pdf.LayoutTypeEquation, pdf.DLALabelFigureCaption,
|
||||
}
|
||||
}
|
||||
var defaultTSRLabels = []string{
|
||||
"table", "table column", "table row",
|
||||
"table column header", "table projected row header",
|
||||
"table spanning cell",
|
||||
func DefaultTSRLabels() []string {
|
||||
return []string{
|
||||
"table", "table column", "table row",
|
||||
"table column header", "table projected row header",
|
||||
"table spanning cell",
|
||||
}
|
||||
}
|
||||
|
||||
type bboxesResponse struct {
|
||||
BBoxes [][]float64 `json:"bboxes"`
|
||||
}
|
||||
|
||||
// DLA analyses a full page image and returns labelled regions.
|
||||
func (c *DeepDocClient) DLA(ctx context.Context, pageImage image.Image) ([]DLARegion, error) {
|
||||
data, err := encodeJPEG(pageImage)
|
||||
// DLA analyzes a full page image and returns labeled regions.
|
||||
func (c *InferenceClient) DLA(ctx context.Context, pageImage image.Image) ([]pdf.DLARegion, error) {
|
||||
data, err := util.EncodeJPEG(pageImage)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("dla: encode: %w", err)
|
||||
}
|
||||
@@ -73,20 +81,20 @@ func (c *DeepDocClient) DLA(ctx context.Context, pageImage image.Image) ([]DLARe
|
||||
if err := c.post(ctx, "/predict/dla", data, "dla.jpeg", &resp); err != nil {
|
||||
return nil, fmt.Errorf("dla: %w", err)
|
||||
}
|
||||
regions := make([]DLARegion, 0, len(resp.BBoxes))
|
||||
regions := make([]pdf.DLARegion, 0, len(resp.BBoxes))
|
||||
for _, b := range resp.BBoxes {
|
||||
if len(b) < 6 {
|
||||
continue
|
||||
}
|
||||
labels := c.DLALabels
|
||||
if labels == nil {
|
||||
labels = defaultDLALabels
|
||||
labels = DefaultDLALabels()
|
||||
}
|
||||
label := ""
|
||||
if clsID := int(b[5]); clsID >= 0 && clsID < len(labels) {
|
||||
label = labels[clsID]
|
||||
}
|
||||
regions = append(regions, DLARegion{
|
||||
regions = append(regions, pdf.DLARegion{
|
||||
X0: b[0], Y0: b[1], X1: b[2], Y1: b[3],
|
||||
Confidence: b[4],
|
||||
Label: label,
|
||||
@@ -96,8 +104,8 @@ func (c *DeepDocClient) DLA(ctx context.Context, pageImage image.Image) ([]DLARe
|
||||
}
|
||||
|
||||
// TSR recognises table structure from a cropped image.
|
||||
func (c *DeepDocClient) TSR(ctx context.Context, cropped image.Image) ([]TSRCell, error) {
|
||||
data, err := encodeJPEG(cropped)
|
||||
func (c *InferenceClient) TSR(ctx context.Context, cropped image.Image) ([]pdf.TSRCell, error) {
|
||||
data, err := util.EncodeJPEG(cropped)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("tsr: encode: %w", err)
|
||||
}
|
||||
@@ -105,14 +113,14 @@ func (c *DeepDocClient) TSR(ctx context.Context, cropped image.Image) ([]TSRCell
|
||||
if err := c.post(ctx, "/predict/tsr", data, "tsr.jpeg", &resp); err != nil {
|
||||
return nil, fmt.Errorf("tsr: %w", err)
|
||||
}
|
||||
cells := make([]TSRCell, 0, len(resp.BBoxes))
|
||||
cells := make([]pdf.TSRCell, 0, len(resp.BBoxes))
|
||||
for _, b := range resp.BBoxes {
|
||||
if len(b) < 5 {
|
||||
continue
|
||||
}
|
||||
tlabels := c.TSRLabels
|
||||
if tlabels == nil {
|
||||
tlabels = defaultTSRLabels
|
||||
tlabels = DefaultTSRLabels()
|
||||
}
|
||||
label := ""
|
||||
if len(b) >= 6 {
|
||||
@@ -120,7 +128,7 @@ func (c *DeepDocClient) TSR(ctx context.Context, cropped image.Image) ([]TSRCell
|
||||
label = tlabels[cls]
|
||||
}
|
||||
}
|
||||
cells = append(cells, TSRCell{
|
||||
cells = append(cells, pdf.TSRCell{
|
||||
X0: b[0], Y0: b[1], X1: b[2], Y1: b[3],
|
||||
Label: label,
|
||||
})
|
||||
@@ -144,8 +152,8 @@ type ocrRecognizeResponse struct {
|
||||
|
||||
// OCRDetect detects text regions (bounding boxes) in an image.
|
||||
// DeepDoc /predict/ocr with operator=det returns quad boxes: [[[x0,y0],[x1,y1],[x2,y2],[x3,y3]], ...]
|
||||
func (c *DeepDocClient) OCRDetect(ctx context.Context, cropped image.Image) ([]OCRBox, error) {
|
||||
data, err := encodeJPEG(cropped)
|
||||
func (c *InferenceClient) OCRDetect(ctx context.Context, cropped image.Image) ([]pdf.OCRBox, error) {
|
||||
data, err := util.EncodeJPEG(cropped)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("ocr detect: encode: %w", err)
|
||||
}
|
||||
@@ -168,14 +176,14 @@ func (c *DeepDocClient) OCRDetect(ctx context.Context, cropped image.Image) ([]O
|
||||
return nil, fmt.Errorf("ocr detect: %w", err)
|
||||
}
|
||||
|
||||
var boxes []OCRBox
|
||||
var boxes []pdf.OCRBox
|
||||
for _, outer := range result.Output {
|
||||
for _, page := range outer {
|
||||
for _, box := range page {
|
||||
if len(box) < 4 {
|
||||
continue
|
||||
}
|
||||
boxes = append(boxes, OCRBox{
|
||||
boxes = append(boxes, pdf.OCRBox{
|
||||
X0: box[0][0], Y0: box[0][1],
|
||||
X1: box[1][0], Y1: box[1][1],
|
||||
X2: box[2][0], Y2: box[2][1],
|
||||
@@ -189,8 +197,8 @@ func (c *DeepDocClient) OCRDetect(ctx context.Context, cropped image.Image) ([]O
|
||||
|
||||
// OCRRecognize recognizes text in a cropped image region.
|
||||
// DeepDoc /predict/ocr with operator=rec returns [[["text", confidence], ...]]
|
||||
func (c *DeepDocClient) OCRRecognize(ctx context.Context, cropped image.Image) ([]OCRText, error) {
|
||||
data, err := encodeJPEG(cropped)
|
||||
func (c *InferenceClient) OCRRecognize(ctx context.Context, cropped image.Image) ([]pdf.OCRText, error) {
|
||||
data, err := util.EncodeJPEG(cropped)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("ocr rec: encode: %w", err)
|
||||
}
|
||||
@@ -198,14 +206,14 @@ func (c *DeepDocClient) OCRRecognize(ctx context.Context, cropped image.Image) (
|
||||
if err := c.post(ctx, "/predict/ocr", data, "ocr_rec.jpeg", &result, "operator", "rec"); err != nil {
|
||||
return nil, fmt.Errorf("ocr rec: %w", err)
|
||||
}
|
||||
var texts []OCRText
|
||||
var texts []pdf.OCRText
|
||||
for _, page := range result.Output {
|
||||
for _, item := range page {
|
||||
for _, pair := range item {
|
||||
if len(pair) >= 2 {
|
||||
text, _ := pair[0].(string)
|
||||
conf, _ := pair[1].(float64)
|
||||
texts = append(texts, OCRText{Text: text, Confidence: conf})
|
||||
texts = append(texts, pdf.OCRText{Text: text, Confidence: conf})
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -216,8 +224,8 @@ func (c *DeepDocClient) OCRRecognize(ctx context.Context, cropped image.Image) (
|
||||
// OCRRecognizeBatch recognizes text in multiple cropped image regions.
|
||||
// Returns a slice of results and a parallel slice of errors (nil on success).
|
||||
// A nil cropped image in the input produces nil results and a non-nil error.
|
||||
func (c *DeepDocClient) OCRRecognizeBatch(ctx context.Context, cropped []image.Image) ([][]OCRText, []error) {
|
||||
results := make([][]OCRText, len(cropped))
|
||||
func (c *InferenceClient) OCRRecognizeBatch(ctx context.Context, cropped []image.Image) ([][]pdf.OCRText, []error) {
|
||||
results := make([][]pdf.OCRText, len(cropped))
|
||||
errs := make([]error, len(cropped))
|
||||
|
||||
// Process images concurrently with a bounded worker pool to avoid
|
||||
@@ -247,7 +255,7 @@ func (c *DeepDocClient) OCRRecognizeBatch(ctx context.Context, cropped []image.I
|
||||
}
|
||||
|
||||
// Health checks whether the DeepDoc service is reachable.
|
||||
func (c *DeepDocClient) Health() bool {
|
||||
func (c *InferenceClient) Health() bool {
|
||||
resp, err := c.httpClient.Get(c.baseURL + "/health")
|
||||
if err != nil {
|
||||
return false
|
||||
@@ -256,48 +264,7 @@ func (c *DeepDocClient) Health() bool {
|
||||
return resp.StatusCode == 200
|
||||
}
|
||||
|
||||
// ModelType probes the DeepDoc /model endpoint once and caches the model flavour.
|
||||
// The /model endpoint is expected to return JSON like {"model":"oss","version":"1.0"}.
|
||||
// When the endpoint is unreachable or model is not "oss", ModelSaas is returned.
|
||||
// Uses sync.Once so the call is safe for concurrent use.
|
||||
func (c *DeepDocClient) ModelType() ModelType {
|
||||
c.modelOnce.Do(func() {
|
||||
c.model = ModelSaas
|
||||
resp, err := c.httpClient.Get(c.baseURL + "/model")
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != 200 {
|
||||
return
|
||||
}
|
||||
var h struct {
|
||||
Model string `json:"model"`
|
||||
}
|
||||
if err := json.NewDecoder(resp.Body).Decode(&h); err != nil {
|
||||
slog.Warn("deepdoc /model: failed to decode response, falling back to SaaS",
|
||||
"err", err)
|
||||
return
|
||||
}
|
||||
if h.Model == "oss" {
|
||||
c.model = ModelOSS
|
||||
}
|
||||
})
|
||||
return c.model
|
||||
}
|
||||
|
||||
// NewTableBuilderFor creates the right TableBuilder for the given
|
||||
// DocAnalyzer, chosen by ModelType().
|
||||
func NewTableBuilderFor(doc DocAnalyzer) TableBuilder {
|
||||
switch doc.ModelType() {
|
||||
case ModelOSS:
|
||||
return NewOssDeepDocService(doc)
|
||||
default:
|
||||
return NewSaasDeepDocService(doc)
|
||||
}
|
||||
}
|
||||
|
||||
func (c *DeepDocClient) post(ctx context.Context, endpoint string, imgData []byte, filename string, result interface{}, extraFields ...string) error {
|
||||
func (c *InferenceClient) post(ctx context.Context, endpoint string, imgData []byte, filename string, result interface{}, extraFields ...string) error {
|
||||
// Build multipart body once — the image data is idempotent.
|
||||
var body bytes.Buffer
|
||||
w := multipart.NewWriter(&body)
|
||||
@@ -1,4 +1,4 @@
|
||||
package parser
|
||||
package inference
|
||||
|
||||
import (
|
||||
"context"
|
||||
@@ -11,11 +11,11 @@ import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
// mustNewDeepDocClient wraps NewDeepDocClient for test convenience.
|
||||
// mustNewDeepDocClient wraps NewInferenceClient for test convenience.
|
||||
// Fails the test if the URL is empty.
|
||||
func mustNewDeepDocClient(t *testing.T, baseURL string) *DeepDocClient {
|
||||
func mustNewDeepDocClient(t *testing.T, baseURL string) *InferenceClient {
|
||||
t.Helper()
|
||||
client, err := NewDeepDocClient(baseURL)
|
||||
client, err := NewInferenceClient(baseURL)
|
||||
if err != nil {
|
||||
t.Fatalf("NewDeepDocClient(%q): %v", baseURL, err)
|
||||
}
|
||||
@@ -4,42 +4,22 @@ package parser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
tbl "ragflow/internal/deepdoc/parser/pdf/table"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// mustConnectOssDeepDoc returns a DeepDocClient pointed at the OSS service;
|
||||
// skips the test if unavailable or if the service reports a non-OSS model type.
|
||||
func mustConnectOssDeepDoc(t *testing.T) *DeepDocClient {
|
||||
t.Helper()
|
||||
url := os.Getenv("OSSDEEPDOC_URL")
|
||||
if url == "" {
|
||||
url = "http://localhost:9390"
|
||||
}
|
||||
client, err := NewDeepDocClient(url)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !client.Health() {
|
||||
t.Fatalf("OssDeepDoc not available at %s", url)
|
||||
}
|
||||
if client.ModelType() != ModelOSS {
|
||||
t.Skipf("DeepDoc at %s is %q, not oss — skipping OSS-specific test", url, client.ModelType())
|
||||
}
|
||||
return client
|
||||
}
|
||||
|
||||
// TestIntegration_OssDeepDoc_TableStructure verifies that parsing a PDF
|
||||
// through the OssDeepDoc TableBuilder produces tables with the expected
|
||||
// row/column structure.
|
||||
func TestIntegration_OssDeepDoc_TableStructure(t *testing.T) {
|
||||
client := mustConnectOssDeepDoc(t)
|
||||
// TestIntegration_DeepDoc_TableStructure verifies that parsing a PDF
|
||||
// through the OSS TableBuilder produces tables with the expected row/column structure.
|
||||
func TestIntegration_DeepDoc_TableStructure(t *testing.T) {
|
||||
client := mustConnectInferenceClient(t)
|
||||
eng := mustOpenEngine(t, "06_table_content.pdf")
|
||||
defer eng.Close()
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg.TableBuilder = NewOssDeepDocService(client)
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
cfg.TableBuilder = tbl.NewDeepDocTableBuilder(client)
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -67,15 +47,15 @@ func TestIntegration_OssDeepDoc_TableStructure(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestIntegration_OssDeepDoc_TableRows verifies each table has non-empty
|
||||
// TestIntegration_DeepDoc_TableRows verifies each table has non-empty
|
||||
// rows with the expected grid structure.
|
||||
func TestIntegration_OssDeepDoc_TableRows(t *testing.T) {
|
||||
client := mustConnectOssDeepDoc(t)
|
||||
func TestIntegration_DeepDoc_TableRows(t *testing.T) {
|
||||
client := mustConnectInferenceClient(t)
|
||||
eng := mustOpenEngine(t, "06_table_content.pdf")
|
||||
defer eng.Close()
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg.TableBuilder = NewOssDeepDocService(client)
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
cfg.TableBuilder = tbl.NewDeepDocTableBuilder(client)
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -106,17 +86,17 @@ func TestIntegration_OssDeepDoc_TableRows(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestIntegration_OssDeepDoc_Idempotency verifies that parsing the same PDF
|
||||
// TestIntegration_DeepDoc_Idempotency verifies that parsing the same PDF
|
||||
// twice produces the same table row structure.
|
||||
func TestIntegration_OssDeepDoc_Idempotency(t *testing.T) {
|
||||
client := mustConnectOssDeepDoc(t)
|
||||
func TestIntegration_DeepDoc_Idempotency(t *testing.T) {
|
||||
client := mustConnectInferenceClient(t)
|
||||
|
||||
parseOnce := func() *ParseResult {
|
||||
parseOnce := func() *pdf.ParseResult {
|
||||
eng := mustOpenEngine(t, "06_table_content.pdf")
|
||||
defer eng.Close()
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg.TableBuilder = NewOssDeepDocService(client)
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
cfg.TableBuilder = tbl.NewDeepDocTableBuilder(client)
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -140,15 +120,15 @@ func TestIntegration_OssDeepDoc_Idempotency(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestIntegration_OssDeepDoc_EmptyPage verifies that a page with no tables
|
||||
// TestIntegration_DeepDoc_EmptyPage verifies that a page with no tables
|
||||
// does not crash.
|
||||
func TestIntegration_OssDeepDoc_EmptyPage(t *testing.T) {
|
||||
client := mustConnectOssDeepDoc(t)
|
||||
func TestIntegration_DeepDoc_EmptyPage(t *testing.T) {
|
||||
client := mustConnectInferenceClient(t)
|
||||
eng := mustOpenEngine(t, "01_english_simple.pdf")
|
||||
defer eng.Close()
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg.TableBuilder = NewOssDeepDocService(client)
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
cfg.TableBuilder = tbl.NewDeepDocTableBuilder(client)
|
||||
p := NewParser(cfg, client)
|
||||
_, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
174
internal/deepdoc/parser/pdf/layout/boxes_sections.go
Normal file
174
internal/deepdoc/parser/pdf/layout/boxes_sections.go
Normal file
@@ -0,0 +1,174 @@
|
||||
package layout
|
||||
|
||||
import (
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
util "ragflow/internal/deepdoc/parser/pdf/util"
|
||||
)
|
||||
|
||||
// ResolvePageSpan computes the ending page and bottom coordinate for a box
|
||||
// that may span multiple pages. When pageHeights is nil or the box fits
|
||||
// within its starting page the returned (toPage, bottom) equal the inputs.
|
||||
//
|
||||
// Zero or negative page heights are treated as invalid: the span stops at
|
||||
// the preceding page, guarding against infinite loops caused by corrupted
|
||||
// page images.
|
||||
func ResolvePageSpan(pageNum int, bottom float64, pageHeights map[int]float64) (toPage int, newBottom float64) {
|
||||
toPage = pageNum
|
||||
newBottom = bottom
|
||||
if pageHeights == nil {
|
||||
return
|
||||
}
|
||||
ph, ok := pageHeights[pageNum]
|
||||
if !ok || ph <= 0 || bottom <= ph {
|
||||
return
|
||||
}
|
||||
remaining := bottom
|
||||
for remaining > ph && ph > 0 {
|
||||
nextPh, ok := pageHeights[toPage+1]
|
||||
if !ok || nextPh <= 0 {
|
||||
// Unknown or invalid next page height — extend by the
|
||||
// last known height once and stop (Python: _line_tag
|
||||
// while-loop break path).
|
||||
remaining -= ph
|
||||
toPage++
|
||||
break
|
||||
}
|
||||
remaining -= ph
|
||||
ph = nextPh
|
||||
toPage++
|
||||
}
|
||||
newBottom = remaining
|
||||
return
|
||||
}
|
||||
|
||||
// boxesToSections converts layout boxes to section format with position tags.
|
||||
//
|
||||
// pageHeights provides the PDF-point height of each page (image height / zoom).
|
||||
// Boxes that extend beyond their page produce multi-page position tags
|
||||
// (Python's _line_tag while-loop detection via resolvePageSpan).
|
||||
//
|
||||
// Python equivalent: output consumed by naive.py::chunk()
|
||||
func BoxesToSections(boxes []pdf.TextBox, pageHeights map[int]float64) []pdf.Section {
|
||||
sections := make([]pdf.Section, 0, len(boxes))
|
||||
for _, b := range boxes {
|
||||
t := strings.TrimSpace(b.Text)
|
||||
if t == "" {
|
||||
continue
|
||||
}
|
||||
toPage, bottom := ResolvePageSpan(b.PageNumber, b.Bottom, pageHeights)
|
||||
|
||||
var posTag string
|
||||
var pageNums []int
|
||||
if b.PageNumber == toPage {
|
||||
posTag = util.FormatPositionTag(b.PageNumber, b.X0, b.X1, b.Top, bottom)
|
||||
pageNums = []int{b.PageNumber}
|
||||
} else {
|
||||
posTag = util.FormatPositionTagRange(b.PageNumber, toPage, b.X0, b.X1, b.Top, bottom)
|
||||
pageNums = make([]int, 0, toPage-b.PageNumber+1)
|
||||
for p := b.PageNumber; p <= toPage; p++ {
|
||||
pageNums = append(pageNums, p)
|
||||
}
|
||||
}
|
||||
sections = append(sections, pdf.Section{
|
||||
Text: t,
|
||||
PositionTag: posTag,
|
||||
LayoutType: b.LayoutType,
|
||||
Positions: []pdf.Position{{PageNumbers: pageNums, Left: b.X0, Right: b.X1, Top: b.Top, Bottom: bottom}},
|
||||
})
|
||||
}
|
||||
return sections
|
||||
}
|
||||
|
||||
// NormalizeSectionPositions ensures each Section's Positions field is populated
|
||||
// by parsing PositionTag when Positions is empty. Sections that already have
|
||||
// Positions populated are left unchanged.
|
||||
//
|
||||
// This mirrors the Python normalize_pdf_items_metadata — canonicalizing
|
||||
// position metadata from the string tag format into the typed []Position form.
|
||||
//
|
||||
// Callers should invoke this AFTER Parse() returns, just before consuming
|
||||
// Sections (e.g., before serialization to JSON or passing to the chunker).
|
||||
// The normalization is intentionally NOT embedded inside the parser pipeline
|
||||
// because Sections may come from multiple sources (deepdoc, MinerU, Docling,
|
||||
// JSON deserialization, etc.).
|
||||
func NormalizeSectionPositions(sections []pdf.Section) {
|
||||
for i := range sections {
|
||||
if len(sections[i].Positions) == 0 && sections[i].PositionTag != "" {
|
||||
sections[i].Positions = util.ExtractPositions(sections[i].PositionTag)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// SortByPageThenY sorts boxes by page → vertical key → x0.
|
||||
func SortByPageThenY(boxes []pdf.TextBox, sortByTop bool) {
|
||||
key := func(b pdf.TextBox) float64 { return b.Bottom }
|
||||
if sortByTop {
|
||||
key = func(b pdf.TextBox) float64 { return b.Top }
|
||||
}
|
||||
sort.Slice(boxes, func(i, j int) bool {
|
||||
if boxes[i].PageNumber != boxes[j].PageNumber {
|
||||
return boxes[i].PageNumber < boxes[j].PageNumber
|
||||
}
|
||||
if key(boxes[i]) != key(boxes[j]) {
|
||||
return key(boxes[i]) < key(boxes[j])
|
||||
}
|
||||
return boxes[i].X0 < boxes[j].X0
|
||||
})
|
||||
}
|
||||
|
||||
// SectionsToMarkdown converts Sections to a markdown string.
|
||||
//
|
||||
// Title sections get a "## " prefix.
|
||||
// Figure sections produce an "" tag.
|
||||
// Text and all other sections are appended verbatim.
|
||||
//
|
||||
// This mirrors the Python parser.py:665-671 markdown output path.
|
||||
func SectionsToMarkdown(sections []pdf.Section) string {
|
||||
var b strings.Builder
|
||||
for _, s := range sections {
|
||||
if s.LayoutType == pdf.LayoutTypeTitle {
|
||||
b.WriteString("\n## ")
|
||||
}
|
||||
if s.LayoutType == pdf.LayoutTypeFigure && s.Image != "" {
|
||||
b.WriteString("\n
|
||||
b.WriteString(s.Image)
|
||||
b.WriteString(")")
|
||||
continue
|
||||
}
|
||||
b.WriteString(s.Text)
|
||||
b.WriteString("\n")
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// SectionsToJSON converts Sections to a Python-compatible JSON dict format.
|
||||
//
|
||||
// Each dict has keys: text, layout_type, doc_type_kwd, _pdf_positions, image.
|
||||
// The _pdf_positions key mirrors Python's PDF_POSITIONS_KEY constant —
|
||||
// the canonical position format consumed by the chunker's extract_pdf_positions.
|
||||
//
|
||||
// This mirrors the Python parser.py:662 set_output("json", bboxes) path.
|
||||
func SectionsToJSON(sections []pdf.Section) []map[string]any {
|
||||
result := make([]map[string]any, len(sections))
|
||||
for i, s := range sections {
|
||||
positions := make([][]any, len(s.Positions))
|
||||
for j, p := range s.Positions {
|
||||
pages := make([]any, len(p.PageNumbers))
|
||||
for k, pn := range p.PageNumbers {
|
||||
pages[k] = pn
|
||||
}
|
||||
positions[j] = []any{pages, p.Left, p.Right, p.Top, p.Bottom}
|
||||
}
|
||||
result[i] = map[string]any{
|
||||
"text": s.Text,
|
||||
"layout_type": s.LayoutType,
|
||||
"doc_type_kwd": s.DocTypeKwd,
|
||||
"_pdf_positions": positions,
|
||||
"image": s.Image,
|
||||
}
|
||||
}
|
||||
return result
|
||||
}
|
||||
392
internal/deepdoc/parser/pdf/layout/boxes_sections_test.go
Normal file
392
internal/deepdoc/parser/pdf/layout/boxes_sections_test.go
Normal file
@@ -0,0 +1,392 @@
|
||||
package layout
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
func TestBoxesToSections_CrossPagePositionTag(t *testing.T) {
|
||||
// Page 0: 267 PDF-points tall (800px at zoom=3).
|
||||
// Box bottom=400 > 267 → spills into page 1 by 133pt.
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 100, X1: 500, Top: 200, Bottom: 400, PageNumber: 0, Text: "跨页表格"},
|
||||
}
|
||||
pageHeights := map[int]float64{0: 267.0}
|
||||
|
||||
sections := BoxesToSections(boxes, pageHeights)
|
||||
if len(sections) != 1 {
|
||||
t.Fatalf("expected 1 section, got %d", len(sections))
|
||||
}
|
||||
s := sections[0]
|
||||
|
||||
// Python: @@1-2\t100.0\t500.0\t200.0\t133.0##
|
||||
// Page 0→1 becomes 1-indexed → pages 1-2.
|
||||
if s.PositionTag != "@@1-2\t100.0\t500.0\t200.0\t133.0##" {
|
||||
t.Errorf("PositionTag: got %q, want '@@1-2\\t100.0\\t500.0\\t200.0\\t133.0##'", s.PositionTag)
|
||||
}
|
||||
if len(s.Positions) != 1 {
|
||||
t.Fatalf("expected 1 pdf.Position, got %d", len(s.Positions))
|
||||
}
|
||||
p := s.Positions[0]
|
||||
if len(p.PageNumbers) != 2 || p.PageNumbers[0] != 0 || p.PageNumbers[1] != 1 {
|
||||
t.Errorf("PageNumbers: got %v, want [0, 1]", p.PageNumbers)
|
||||
}
|
||||
if p.Top != 200 || p.Bottom != 133 {
|
||||
t.Errorf("coords: top=%v (want 200), bottom=%v (want 133 = 400-267)", p.Top, p.Bottom)
|
||||
}
|
||||
}
|
||||
|
||||
// TestBoxesToSections_SinglePageUnchanged verifies single-page boxes are
|
||||
// unaffected by the cross-page change.
|
||||
func TestBoxesToSections_SinglePageUnchanged(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 50, X1: 200, Top: 10, Bottom: 30, PageNumber: 0, Text: "普通文本"},
|
||||
}
|
||||
pageHeights := map[int]float64{0: 267.0}
|
||||
|
||||
sections := BoxesToSections(boxes, pageHeights)
|
||||
if len(sections) != 1 {
|
||||
t.Fatalf("expected 1 section, got %d", len(sections))
|
||||
}
|
||||
// Single page: tag should be @@1, not @@1-1
|
||||
if sections[0].PositionTag != "@@1\t50.0\t200.0\t10.0\t30.0##" {
|
||||
t.Errorf("single-page PositionTag: got %q", sections[0].PositionTag)
|
||||
}
|
||||
if len(sections[0].Positions[0].PageNumbers) != 1 {
|
||||
t.Errorf("single-page PageNumbers: got %v, want [0]", sections[0].Positions[0].PageNumbers)
|
||||
}
|
||||
}
|
||||
|
||||
func TestResolvePageSpan_SinglePage(t *testing.T) {
|
||||
// Box fits within the page → toPage unchanged, bottom unchanged.
|
||||
toPage, bottom := ResolvePageSpan(0, 30, map[int]float64{0: 267})
|
||||
if toPage != 0 || bottom != 30 {
|
||||
t.Errorf("got toPage=%d bottom=%v, want 0, 30", toPage, bottom)
|
||||
}
|
||||
}
|
||||
|
||||
func TestResolvePageSpan_CrossPage(t *testing.T) {
|
||||
// Box bottom=400 exceeds page 0 height=267 → spans to page 1.
|
||||
toPage, bottom := ResolvePageSpan(0, 400, map[int]float64{0: 267})
|
||||
if toPage != 1 {
|
||||
t.Errorf("toPage = %d, want 1", toPage)
|
||||
}
|
||||
if bottom != 133 {
|
||||
t.Errorf("bottom = %v, want 133 (400-267)", bottom)
|
||||
}
|
||||
}
|
||||
|
||||
func TestResolvePageSpan_MultiPage(t *testing.T) {
|
||||
// Box bottom=600, page 0=267, page 1=200, page 2=200.
|
||||
heights := map[int]float64{0: 267, 1: 200, 2: 200}
|
||||
toPage, bottom := ResolvePageSpan(0, 600, heights)
|
||||
if toPage != 2 {
|
||||
t.Errorf("toPage = %d, want 2", toPage)
|
||||
}
|
||||
if bottom != 133 {
|
||||
t.Errorf("bottom = %v, want 133 (600-267-200)", bottom)
|
||||
}
|
||||
}
|
||||
|
||||
func TestResolvePageSpan_NilHeights(t *testing.T) {
|
||||
toPage, bottom := ResolvePageSpan(0, 400, nil)
|
||||
if toPage != 0 || bottom != 400 {
|
||||
t.Errorf("got toPage=%d bottom=%v, want 0, 400 (nil=no cross-page)", toPage, bottom)
|
||||
}
|
||||
}
|
||||
|
||||
func TestResolvePageSpan_ZeroHeightGuard(t *testing.T) {
|
||||
// Zero-height pages must not cause an infinite loop.
|
||||
// Page 0=200, page 1=0, page 2=0, page 3=300 — box bottom=500.
|
||||
heights := map[int]float64{0: 200, 1: 0, 2: 0, 3: 300}
|
||||
toPage, bottom := ResolvePageSpan(0, 500, heights)
|
||||
// 500-200=300 remaining; page1=0 → break at unknown/invalid; toPage=1, bottom=300.
|
||||
// (the break path treats zero/unknown as "assume same height once and stop")
|
||||
if toPage != 1 {
|
||||
t.Errorf("toPage = %d, want 1 (stopped at first zero-height page)", toPage)
|
||||
}
|
||||
if bottom != 300 {
|
||||
t.Errorf("bottom = %v, want 300 (500-200)", bottom)
|
||||
}
|
||||
}
|
||||
|
||||
func TestResolvePageSpan_UnknownNextPage(t *testing.T) {
|
||||
// Next page not in map → assume same height once, then stop.
|
||||
heights := map[int]float64{0: 267}
|
||||
toPage, bottom := ResolvePageSpan(0, 500, heights)
|
||||
if toPage != 1 {
|
||||
t.Errorf("toPage = %d, want 1 (one fallback extension)", toPage)
|
||||
}
|
||||
if bottom != 233 {
|
||||
t.Errorf("bottom = %v, want 233 (500-267)", bottom)
|
||||
}
|
||||
}
|
||||
|
||||
func TestResolvePageSpan_NegativePh(t *testing.T) {
|
||||
heights := map[int]float64{0: 200, 1: -10, 2: 200}
|
||||
toPage, bottom := ResolvePageSpan(0, 500, heights)
|
||||
if toPage != 1 {
|
||||
t.Errorf("toPage = %d, want 1 (stopped at negative-height page)", toPage)
|
||||
}
|
||||
if bottom != 300 {
|
||||
t.Errorf("bottom = %v, want 300 (500-200)", bottom)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeSectionPositions_ValidTag(t *testing.T) {
|
||||
sections := []pdf.Section{
|
||||
{Text: "test", PositionTag: "@@1\t50.0\t300.0\t200.0\t400.0##", Positions: nil},
|
||||
}
|
||||
NormalizeSectionPositions(sections)
|
||||
|
||||
s := sections[0]
|
||||
if len(s.Positions) != 1 {
|
||||
t.Fatalf("expected 1 Position, got %d", len(s.Positions))
|
||||
}
|
||||
p := s.Positions[0]
|
||||
if len(p.PageNumbers) != 1 || p.PageNumbers[0] != 0 {
|
||||
t.Errorf("PageNumbers: got %v, want [0]", p.PageNumbers)
|
||||
}
|
||||
if p.Left != 50.0 || p.Right != 300.0 || p.Top != 200.0 || p.Bottom != 400.0 {
|
||||
t.Errorf("coords: got (%.1f, %.1f, %.1f, %.1f), want (50.0, 300.0, 200.0, 400.0)",
|
||||
p.Left, p.Right, p.Top, p.Bottom)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeSectionPositions_EmptyTag(t *testing.T) {
|
||||
sections := []pdf.Section{
|
||||
{Text: "test", PositionTag: "", Positions: nil},
|
||||
}
|
||||
NormalizeSectionPositions(sections)
|
||||
|
||||
if len(sections[0].Positions) != 0 {
|
||||
t.Errorf("expected empty Positions, got %v", sections[0].Positions)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeSectionPositions_AlreadyPopulated(t *testing.T) {
|
||||
existing := []pdf.Position{{PageNumbers: []int{0}, Left: 10, Right: 20, Top: 30, Bottom: 40}}
|
||||
sections := []pdf.Section{
|
||||
{Text: "test", PositionTag: "@@1\t50.0\t300.0\t200.0\t400.0##", Positions: existing},
|
||||
}
|
||||
NormalizeSectionPositions(sections)
|
||||
|
||||
// Should NOT overwrite existing Positions
|
||||
if len(sections[0].Positions) != 1 {
|
||||
t.Fatalf("expected 1 Position, got %d", len(sections[0].Positions))
|
||||
}
|
||||
p := sections[0].Positions[0]
|
||||
if p.Left != 10 || p.Top != 30 {
|
||||
t.Errorf("Positions were overwritten: got left=%.1f top=%.1f, want left=10 top=30", p.Left, p.Top)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeSectionPositions_MultiPageTag(t *testing.T) {
|
||||
sections := []pdf.Section{
|
||||
{Text: "test", PositionTag: "@@1-2\t50.0\t300.0\t200.0\t400.0##", Positions: nil},
|
||||
}
|
||||
NormalizeSectionPositions(sections)
|
||||
|
||||
p := sections[0].Positions[0]
|
||||
if len(p.PageNumbers) != 2 || p.PageNumbers[0] != 0 || p.PageNumbers[1] != 1 {
|
||||
t.Errorf("PageNumbers: got %v, want [0, 1]", p.PageNumbers)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeSectionPositions_MixedSlice(t *testing.T) {
|
||||
existing := []pdf.Position{{PageNumbers: []int{2}, Left: 1, Right: 2, Top: 3, Bottom: 4}}
|
||||
sections := []pdf.Section{
|
||||
{Text: "has_positions", PositionTag: "@@9\t1.0\t2.0\t3.0\t4.0##", Positions: existing},
|
||||
{Text: "no_positions", PositionTag: "@@1\t50.0\t300.0\t200.0\t400.0##", Positions: nil},
|
||||
{Text: "no_tag", PositionTag: "", Positions: nil},
|
||||
}
|
||||
NormalizeSectionPositions(sections)
|
||||
|
||||
// has_positions: existing preserved
|
||||
if sections[0].Positions[0].PageNumbers[0] != 2 {
|
||||
t.Errorf("existing Positions were overwritten")
|
||||
}
|
||||
// no_positions: parsed from tag
|
||||
if len(sections[1].Positions) != 1 || sections[1].Positions[0].PageNumbers[0] != 0 {
|
||||
t.Errorf("no_positions not normalized: %v", sections[1].Positions)
|
||||
}
|
||||
// no_tag: left empty
|
||||
if len(sections[2].Positions) != 0 {
|
||||
t.Errorf("no_tag should remain empty: %v", sections[2].Positions)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeSectionPositions_NilInput(t *testing.T) {
|
||||
// Should not panic
|
||||
NormalizeSectionPositions(nil)
|
||||
}
|
||||
|
||||
func TestNormalizeSectionPositions_EmptySlice(t *testing.T) {
|
||||
sections := []pdf.Section{}
|
||||
NormalizeSectionPositions(sections)
|
||||
if len(sections) != 0 {
|
||||
t.Error("empty slice should remain empty")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSectionsToMarkdown_Title(t *testing.T) {
|
||||
sections := []pdf.Section{
|
||||
{LayoutType: pdf.LayoutTypeTitle, Text: "标题", Image: ""},
|
||||
}
|
||||
got := SectionsToMarkdown(sections)
|
||||
want := "\n## 标题\n"
|
||||
if got != want {
|
||||
t.Errorf("got %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSectionsToMarkdown_FigureWithImage(t *testing.T) {
|
||||
sections := []pdf.Section{
|
||||
{LayoutType: pdf.LayoutTypeFigure, Text: "图1", Image: "abc123"},
|
||||
}
|
||||
got := SectionsToMarkdown(sections)
|
||||
want := "\n"
|
||||
if got != want {
|
||||
t.Errorf("got %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSectionsToMarkdown_FigureWithoutImage(t *testing.T) {
|
||||
sections := []pdf.Section{
|
||||
{LayoutType: pdf.LayoutTypeFigure, Text: "图1", Image: ""},
|
||||
}
|
||||
got := SectionsToMarkdown(sections)
|
||||
want := "图1\n"
|
||||
if got != want {
|
||||
t.Errorf("got %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSectionsToMarkdown_Text(t *testing.T) {
|
||||
sections := []pdf.Section{
|
||||
{LayoutType: pdf.LayoutTypeText, Text: "普通内容", Image: ""},
|
||||
}
|
||||
got := SectionsToMarkdown(sections)
|
||||
want := "普通内容\n"
|
||||
if got != want {
|
||||
t.Errorf("got %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSectionsToMarkdown_Table(t *testing.T) {
|
||||
sections := []pdf.Section{
|
||||
{LayoutType: pdf.LayoutTypeTable, Text: "表格内容", Image: ""},
|
||||
}
|
||||
got := SectionsToMarkdown(sections)
|
||||
want := "表格内容\n"
|
||||
if got != want {
|
||||
t.Errorf("got %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSectionsToMarkdown_Mixed(t *testing.T) {
|
||||
sections := []pdf.Section{
|
||||
{LayoutType: pdf.LayoutTypeTitle, Text: "标题", Image: ""},
|
||||
{LayoutType: pdf.LayoutTypeText, Text: "内容", Image: ""},
|
||||
{LayoutType: pdf.LayoutTypeFigure, Text: "图", Image: "img"},
|
||||
{LayoutType: pdf.LayoutTypeTable, Text: "表格", Image: ""},
|
||||
}
|
||||
got := SectionsToMarkdown(sections)
|
||||
want := "\n## 标题\n内容\n\n表格\n"
|
||||
if got != want {
|
||||
t.Errorf("got %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSectionsToMarkdown_EmptySlice(t *testing.T) {
|
||||
got := SectionsToMarkdown([]pdf.Section{})
|
||||
if got != "" {
|
||||
t.Errorf("got %q, want empty string", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSectionsToMarkdown_NilSlice(t *testing.T) {
|
||||
got := SectionsToMarkdown(nil)
|
||||
if got != "" {
|
||||
t.Errorf("got %q, want empty string", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSectionsToJSON_SingleSection(t *testing.T) {
|
||||
sections := []pdf.Section{
|
||||
{
|
||||
Text: "测试内容",
|
||||
LayoutType: pdf.LayoutTypeText,
|
||||
DocTypeKwd: "text",
|
||||
Positions: []pdf.Position{{PageNumbers: []int{0}, Left: 50, Right: 300, Top: 200, Bottom: 400}},
|
||||
Image: "abc123",
|
||||
},
|
||||
}
|
||||
got := SectionsToJSON(sections)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("expected 1 item, got %d", len(got))
|
||||
}
|
||||
m := got[0]
|
||||
if m["text"] != "测试内容" {
|
||||
t.Errorf("text: got %v", m["text"])
|
||||
}
|
||||
if m["layout_type"] != "text" {
|
||||
t.Errorf("layout_type: got %v", m["layout_type"])
|
||||
}
|
||||
if m["doc_type_kwd"] != "text" {
|
||||
t.Errorf("doc_type_kwd: got %v", m["doc_type_kwd"])
|
||||
}
|
||||
if m["image"] != "abc123" {
|
||||
t.Errorf("image: got %v", m["image"])
|
||||
}
|
||||
positions, ok := m["_pdf_positions"].([][]any)
|
||||
if !ok || len(positions) != 1 {
|
||||
t.Fatalf("_pdf_positions not correct type/length: %T %v", m["_pdf_positions"], m["_pdf_positions"])
|
||||
}
|
||||
pos := positions[0]
|
||||
pages := pos[0].([]any)
|
||||
if len(pages) != 1 || pages[0] != 0 {
|
||||
t.Errorf("pages: got %v", pages)
|
||||
}
|
||||
if pos[1] != float64(50) || pos[2] != float64(300) || pos[3] != float64(200) || pos[4] != float64(400) {
|
||||
t.Errorf("coords: got %v", pos[1:])
|
||||
}
|
||||
}
|
||||
|
||||
func TestSectionsToJSON_MultiPage(t *testing.T) {
|
||||
sections := []pdf.Section{
|
||||
{
|
||||
Text: "跨页内容",
|
||||
Positions: []pdf.Position{{PageNumbers: []int{0, 1}, Left: 100, Right: 200, Top: 50, Bottom: 300}},
|
||||
},
|
||||
}
|
||||
got := SectionsToJSON(sections)
|
||||
positions := got[0]["_pdf_positions"].([][]any)
|
||||
pages := positions[0][0].([]any)
|
||||
if len(pages) != 2 || pages[0] != 0 || pages[1] != 1 {
|
||||
t.Errorf("multi-page pages: got %v, want [0, 1]", pages)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSectionsToJSON_EmptySlice(t *testing.T) {
|
||||
got := SectionsToJSON([]pdf.Section{})
|
||||
if len(got) != 0 {
|
||||
t.Errorf("got %d items, want 0", len(got))
|
||||
}
|
||||
}
|
||||
|
||||
func TestSectionsToJSON_NilSlice(t *testing.T) {
|
||||
got := SectionsToJSON(nil)
|
||||
if got == nil {
|
||||
t.Error("expected non-nil slice, got nil")
|
||||
}
|
||||
if len(got) != 0 {
|
||||
t.Errorf("got %d items, want 0", len(got))
|
||||
}
|
||||
}
|
||||
|
||||
// TestCrossPageTableMerge verifies that mergeTablesAcrossPages merges
|
||||
// two TableItems on consecutive pages with overlapping X positions.
|
||||
// Python: _extract_table_figure merges cross-page tables by matching layoutno.
|
||||
// Spanning cells should be annotated with colspan/rowspan in the HTML output.
|
||||
220
internal/deepdoc/parser/pdf/layout/chars_boxes.go
Normal file
220
internal/deepdoc/parser/pdf/layout/chars_boxes.go
Normal file
@@ -0,0 +1,220 @@
|
||||
package layout
|
||||
|
||||
import (
|
||||
"math"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
util "ragflow/internal/deepdoc/parser/pdf/util"
|
||||
)
|
||||
|
||||
// CharsToBoxes converts raw characters to initial text boxes by grouping
|
||||
// characters into lines based on vertical overlap.
|
||||
//
|
||||
// Python: pdf_parser.__images__ producing self.boxes
|
||||
func CharsToBoxes(chars []pdf.TextChar, pageNum int, sortByTop bool) []pdf.TextBox {
|
||||
if len(chars) == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
lines := GroupCharsToLines(chars, sortByTop)
|
||||
|
||||
// Page-level column gap threshold from ALL inter-char gaps.
|
||||
// Falls back to per-line threshold when page has too few gaps.
|
||||
threshold := pageXGapThreshold(lines)
|
||||
|
||||
boxes := make([]pdf.TextBox, 0, len(lines))
|
||||
for _, line := range lines {
|
||||
thr := threshold
|
||||
if thr > 100 {
|
||||
// No significant column gaps on this page → use per-line threshold.
|
||||
thr = perLineXGapThreshold(line)
|
||||
}
|
||||
subLines := splitLineByXGap(line, thr)
|
||||
for _, sub := range subLines {
|
||||
box := LineToTextBox(sub)
|
||||
box.PageNumber = pageNum
|
||||
boxes = append(boxes, box)
|
||||
}
|
||||
}
|
||||
return boxes
|
||||
}
|
||||
|
||||
// perLineXGapThreshold computes a dynamic X-gap threshold for column
|
||||
// splitting within a single line (fallback when page has few gaps).
|
||||
func perLineXGapThreshold(chars []pdf.TextChar) float64 {
|
||||
if len(chars) <= 1 {
|
||||
return 1e9
|
||||
}
|
||||
var gaps []float64
|
||||
for i := 1; i < len(chars); i++ {
|
||||
g := chars[i].X0 - chars[i-1].X1
|
||||
gaps = append(gaps, g)
|
||||
}
|
||||
if len(gaps) == 0 {
|
||||
return 1e9
|
||||
}
|
||||
sort.Float64s(gaps)
|
||||
medianGap := gaps[len(gaps)/2]
|
||||
if medianGap < 6 {
|
||||
medianGap = 6
|
||||
}
|
||||
return medianGap * 2.5
|
||||
}
|
||||
|
||||
// pageXGapThreshold computes a global X-gap column threshold from all
|
||||
// inter-char gaps across all lines on the page. 95th percentile catches
|
||||
// column boundaries while excluding word-level gaps.
|
||||
// Returns a value > 100 when there are too few gaps for reliable p95,
|
||||
// signalling the caller to fall back to perLineXGapThreshold.
|
||||
func pageXGapThreshold(lines [][]pdf.TextChar) float64 {
|
||||
var allGaps []float64
|
||||
for _, line := range lines {
|
||||
for i := 1; i < len(line); i++ {
|
||||
g := line[i].X0 - line[i-1].X1
|
||||
allGaps = append(allGaps, g)
|
||||
}
|
||||
}
|
||||
if len(allGaps) < 10 {
|
||||
return 1e9 // too few gaps for reliable p95 → fall back to per-line
|
||||
}
|
||||
sort.Float64s(allGaps)
|
||||
// 95th percentile: only the largest 5% of gaps are column boundaries.
|
||||
p95 := allGaps[len(allGaps)*95/100]
|
||||
if p95 < 30 {
|
||||
p95 = 30 // floor: column gaps are ≥30pt in practice
|
||||
}
|
||||
return p95
|
||||
}
|
||||
|
||||
// splitLineByXGap splits a character line into sub-lines where X gaps
|
||||
// meet or exceed the threshold (column boundaries). Uses >= to match the
|
||||
// p95 boundary value — a gap exactly at the 95th percentile is a column gap,
|
||||
// not a word gap.
|
||||
func splitLineByXGap(chars []pdf.TextChar, threshold float64) [][]pdf.TextChar {
|
||||
if len(chars) <= 1 {
|
||||
return [][]pdf.TextChar{chars}
|
||||
}
|
||||
var result [][]pdf.TextChar
|
||||
start := 0
|
||||
for i := 1; i < len(chars); i++ {
|
||||
gap := chars[i].X0 - chars[i-1].X1
|
||||
if gap >= threshold {
|
||||
result = append(result, chars[start:i])
|
||||
start = i
|
||||
}
|
||||
}
|
||||
result = append(result, chars[start:])
|
||||
return result
|
||||
}
|
||||
|
||||
// ---- internal helpers ----
|
||||
|
||||
// groupCharsToLines groups characters into horizontal lines based on vertical overlap.
|
||||
func GroupCharsToLines(chars []pdf.TextChar, sortByTop bool) [][]pdf.TextChar {
|
||||
if len(chars) == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
key := func(c pdf.TextChar) float64 { return c.Bottom }
|
||||
if sortByTop {
|
||||
key = func(c pdf.TextChar) float64 { return c.Top }
|
||||
}
|
||||
|
||||
// Sort by vertical key (Bottom or Top) then x0 using sort.SliceStable.
|
||||
// Guard against NaN: a NaN key sorts after everything else.
|
||||
sort.SliceStable(chars, func(i, j int) bool {
|
||||
ki, kj := key(chars[i]), key(chars[j])
|
||||
if ki != kj && !math.IsNaN(ki) && !math.IsNaN(kj) {
|
||||
return ki < kj
|
||||
}
|
||||
if math.IsNaN(ki) != math.IsNaN(kj) {
|
||||
return !math.IsNaN(ki) // non-NaN before NaN
|
||||
}
|
||||
return chars[i].X0 < chars[j].X0
|
||||
})
|
||||
|
||||
var lines [][]pdf.TextChar
|
||||
var currentLine []pdf.TextChar
|
||||
|
||||
for _, c := range chars {
|
||||
if len(currentLine) == 0 {
|
||||
currentLine = append(currentLine, c)
|
||||
continue
|
||||
}
|
||||
if verticalOverlap(currentLine[len(currentLine)-1], c) {
|
||||
currentLine = append(currentLine, c)
|
||||
} else {
|
||||
if len(currentLine) > 0 {
|
||||
lines = append(lines, currentLine)
|
||||
}
|
||||
currentLine = []pdf.TextChar{c}
|
||||
}
|
||||
}
|
||||
if len(currentLine) > 0 {
|
||||
lines = append(lines, currentLine)
|
||||
}
|
||||
return lines
|
||||
}
|
||||
|
||||
// verticalOverlap checks if two characters are on the same horizontal line.
|
||||
func verticalOverlap(a, b pdf.TextChar) bool {
|
||||
mh := math.Max(util.CharHeight(a), util.CharHeight(b))
|
||||
if mh <= 0 {
|
||||
mh = 1.0
|
||||
}
|
||||
return math.Abs(a.Top-b.Top) < mh*0.5
|
||||
}
|
||||
|
||||
// lineToTextBox converts a line of characters to a single pdf.TextBox.
|
||||
// asciiWordPattern matches strings composed entirely of ASCII word
|
||||
// characters. Python uses re.match (prefix match) — the stricter
|
||||
// full-string match here is equivalent in practice because each
|
||||
// pdf.TextChar.Text is a single rune, so prevText+currText ≤ 2 chars.
|
||||
// Python: pdf_parser.py:1528 re.match(r"[0-9a-zA-Z,.:;!%]+", ...)
|
||||
var asciiWordPattern = regexp.MustCompile(`^[0-9a-zA-Z,.:;!%]+$`)
|
||||
|
||||
func LineToTextBox(chars []pdf.TextChar) pdf.TextBox {
|
||||
if len(chars) == 0 {
|
||||
return pdf.TextBox{}
|
||||
}
|
||||
box := pdf.TextBox{
|
||||
X0: chars[0].X0,
|
||||
X1: chars[0].X1,
|
||||
Top: chars[0].Top,
|
||||
Bottom: chars[0].Bottom,
|
||||
}
|
||||
var textParts []string
|
||||
for i, c := range chars {
|
||||
// Insert space between adjacent ASCII words with a visible gap.
|
||||
// Python: pdf_parser.py:1524-1532 __img_ocr space insertion.
|
||||
if i > 0 {
|
||||
prev := chars[i-1]
|
||||
prevText := strings.TrimSpace(prev.Text)
|
||||
currText := strings.TrimSpace(c.Text)
|
||||
if prevText != "" && currText != "" {
|
||||
gap := c.X0 - prev.X1
|
||||
minWidth := math.Min(c.X1-c.X0, prev.X1-prev.X0)
|
||||
if gap >= minWidth/2 &&
|
||||
asciiWordPattern.MatchString(prevText+currText) {
|
||||
textParts = append(textParts, " ")
|
||||
}
|
||||
}
|
||||
}
|
||||
box.X0 = math.Min(box.X0, c.X0)
|
||||
box.X1 = math.Max(box.X1, c.X1)
|
||||
box.Top = math.Min(box.Top, c.Top)
|
||||
box.Bottom = math.Max(box.Bottom, c.Bottom)
|
||||
textParts = append(textParts, c.Text)
|
||||
if c.LayoutType != "" {
|
||||
box.LayoutType = c.LayoutType
|
||||
}
|
||||
if c.LayoutNo != "" {
|
||||
box.LayoutNo = c.LayoutNo
|
||||
}
|
||||
}
|
||||
box.Text = strings.Join(textParts, "")
|
||||
return box
|
||||
}
|
||||
107
internal/deepdoc/parser/pdf/layout/chars_boxes_test.go
Normal file
107
internal/deepdoc/parser/pdf/layout/chars_boxes_test.go
Normal file
@@ -0,0 +1,107 @@
|
||||
package layout
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
func TestCharsToBoxes(t *testing.T) {
|
||||
t.Run("empty chars", func(t *testing.T) {
|
||||
if boxes := CharsToBoxes(nil, 0, false); boxes != nil {
|
||||
t.Error("nil chars → nil boxes")
|
||||
}
|
||||
if boxes := CharsToBoxes([]pdf.TextChar{}, 0, false); boxes != nil {
|
||||
t.Error("empty chars → nil boxes")
|
||||
}
|
||||
})
|
||||
t.Run("single char", func(t *testing.T) {
|
||||
chars := []pdf.TextChar{
|
||||
{X0: 50, X1: 58, Top: 100, Bottom: 112, Text: "A", PageNumber: 0},
|
||||
}
|
||||
boxes := CharsToBoxes(chars, 0, false)
|
||||
if len(boxes) != 1 {
|
||||
t.Fatalf("expected 1 box, got %d", len(boxes))
|
||||
}
|
||||
if boxes[0].Text != "A" {
|
||||
t.Errorf("Text = %q, want 'A'", boxes[0].Text)
|
||||
}
|
||||
})
|
||||
t.Run("two lines", func(t *testing.T) {
|
||||
chars := []pdf.TextChar{
|
||||
{X0: 50, X1: 58, Top: 100, Bottom: 112, Text: "A", PageNumber: 0},
|
||||
{X0: 60, X1: 68, Top: 100, Bottom: 112, Text: "B", PageNumber: 0},
|
||||
{X0: 50, X1: 58, Top: 114, Bottom: 126, Text: "C", PageNumber: 0},
|
||||
}
|
||||
boxes := CharsToBoxes(chars, 0, false)
|
||||
if len(boxes) != 2 {
|
||||
t.Errorf("expected 2 lines, got %d", len(boxes))
|
||||
}
|
||||
})
|
||||
t.Run("preserves whitespace lines", func(t *testing.T) {
|
||||
chars := []pdf.TextChar{
|
||||
{Text: " ", X0: 10, Top: 100, X1: 15, Bottom: 112},
|
||||
{Text: "Hello", X0: 10, Top: 120, X1: 50, Bottom: 132},
|
||||
}
|
||||
boxes := CharsToBoxes(chars, 0, false)
|
||||
if len(boxes) != 2 {
|
||||
t.Errorf("expected 2 boxes (whitespace preserved), got %d", len(boxes))
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestGroupCharsToLines(t *testing.T) {
|
||||
t.Run("empty", func(t *testing.T) {
|
||||
if lines := GroupCharsToLines(nil, false); lines != nil {
|
||||
t.Error("nil → nil")
|
||||
}
|
||||
})
|
||||
t.Run("single char", func(t *testing.T) {
|
||||
chars := []pdf.TextChar{{X0: 50, X1: 58, Top: 100, Bottom: 112, Text: "A"}}
|
||||
lines := GroupCharsToLines(chars, false)
|
||||
if len(lines) != 1 || len(lines[0]) != 1 {
|
||||
t.Error("single char → single line")
|
||||
}
|
||||
})
|
||||
t.Run("multi column same line", func(t *testing.T) {
|
||||
chars := []pdf.TextChar{
|
||||
{X0: 50, X1: 58, Top: 100, Bottom: 112, Text: "H"},
|
||||
{X0: 60, X1: 68, Top: 100, Bottom: 112, Text: "i"},
|
||||
{X0: 300, X1: 308, Top: 100, Bottom: 112, Text: "B"},
|
||||
{X0: 50, X1: 58, Top: 114, Bottom: 126, Text: "A"},
|
||||
}
|
||||
lines := GroupCharsToLines(chars, false)
|
||||
if len(lines) != 2 {
|
||||
t.Errorf("expected 2 lines, got %d", len(lines))
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestLineToTextBox(t *testing.T) {
|
||||
t.Run("basic", func(t *testing.T) {
|
||||
chars := []pdf.TextChar{
|
||||
{X0: 50, X1: 58, Top: 100, Bottom: 112, Text: "H"},
|
||||
{X0: 60, X1: 68, Top: 100, Bottom: 112, Text: "i"},
|
||||
}
|
||||
box := LineToTextBox(chars)
|
||||
if box.Text != "Hi" {
|
||||
t.Errorf("Text = %q, want 'Hi'", box.Text)
|
||||
}
|
||||
if box.X0 != 50 || box.X1 != 68 {
|
||||
t.Errorf("bbox = [%f, %f], want [50, 68]", box.X0, box.X1)
|
||||
}
|
||||
})
|
||||
t.Run("empty chars", func(t *testing.T) {
|
||||
box := LineToTextBox(nil)
|
||||
if box.Text != "" {
|
||||
t.Errorf("nil chars → empty box, got %q", box.Text)
|
||||
}
|
||||
})
|
||||
t.Run("single char", func(t *testing.T) {
|
||||
chars := []pdf.TextChar{{X0: 10, X1: 18, Top: 100, Bottom: 112, Text: "X"}}
|
||||
box := LineToTextBox(chars)
|
||||
if box.Text != "X" {
|
||||
t.Errorf("Text = %q, want 'X'", box.Text)
|
||||
}
|
||||
})
|
||||
}
|
||||
@@ -1,11 +1,12 @@
|
||||
package parser
|
||||
package layout
|
||||
|
||||
import (
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestMergeSameBullet(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{Text: "* item 1", Top: 100, Bottom: 112, X0: 50, X1: 200},
|
||||
{Text: "* item 2", Top: 114, Bottom: 126, X0: 50, X1: 200},
|
||||
}
|
||||
@@ -16,7 +17,7 @@ func TestMergeSameBullet(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestMergeSameBulletNoMerge(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{Text: "A item", Top: 100, Bottom: 112, X0: 50, X1: 200},
|
||||
{Text: "B item", Top: 114, Bottom: 126, X0: 50, X1: 200},
|
||||
}
|
||||
@@ -28,7 +29,7 @@ func TestMergeSameBulletNoMerge(t *testing.T) {
|
||||
|
||||
func TestMergeSameBulletChinese(t *testing.T) {
|
||||
// Chinese chars start, should not merge via bullet rule
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{Text: "测试文本", Top: 100, Bottom: 112, X0: 50, X1: 200},
|
||||
{Text: "测试内容", Top: 114, Bottom: 126, X0: 50, X1: 200},
|
||||
}
|
||||
@@ -1,12 +1,15 @@
|
||||
package parser
|
||||
package layout
|
||||
|
||||
import (
|
||||
"log/slog"
|
||||
"math"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
util "ragflow/internal/deepdoc/parser/pdf/util"
|
||||
"regexp"
|
||||
"slices"
|
||||
"sort"
|
||||
"strings"
|
||||
"unicode"
|
||||
"unicode/utf8"
|
||||
)
|
||||
|
||||
@@ -16,7 +19,7 @@ import (
|
||||
// with silhouette score selection, matching Python's _assign_column().
|
||||
//
|
||||
// Python: pdf_parser.py:739 _assign_column()
|
||||
func AssignColumn(boxes []TextBox, zoom float64) []TextBox {
|
||||
func AssignColumn(boxes []pdf.TextBox, zoom float64) []pdf.TextBox {
|
||||
if len(boxes) == 0 {
|
||||
return boxes
|
||||
}
|
||||
@@ -26,7 +29,7 @@ func AssignColumn(boxes []TextBox, zoom float64) []TextBox {
|
||||
pageGroups[b.PageNumber] = append(pageGroups[b.PageNumber], i)
|
||||
}
|
||||
|
||||
result := make([]TextBox, len(boxes))
|
||||
result := make([]pdf.TextBox, len(boxes))
|
||||
copy(result, boxes)
|
||||
|
||||
// Step A: per-page best k using silhouette score.
|
||||
@@ -71,10 +74,10 @@ func AssignColumn(boxes []TextBox, zoom float64) []TextBox {
|
||||
bestK, bestScore := 1, -1.0
|
||||
|
||||
for k := 1; k <= maxTry; k++ {
|
||||
labels, _ := kmeans1D(x0s, k)
|
||||
labels, _ := util.KMeans1D(x0s, k)
|
||||
var score float64
|
||||
if k > 1 {
|
||||
score = silhouette1D(x0s, labels)
|
||||
score = util.Silhouette1D(x0s, labels)
|
||||
}
|
||||
// score = 0 for k=1; score = -1 if silhouette undefined.
|
||||
if score > bestScore {
|
||||
@@ -101,7 +104,7 @@ func AssignColumn(boxes []TextBox, zoom float64) []TextBox {
|
||||
x0s[i] = boxes[idx].X0
|
||||
}
|
||||
|
||||
labels, centroids := kmeans1D(x0s, k)
|
||||
labels, centroids := util.KMeans1D(x0s, k)
|
||||
|
||||
// Sort centroids by x position, remap labels left→right.
|
||||
type clPair struct {
|
||||
@@ -131,12 +134,12 @@ func AssignColumn(boxes []TextBox, zoom float64) []TextBox {
|
||||
// TextMerge horizontally merges adjacent boxes at similar vertical positions.
|
||||
//
|
||||
// Python: pdf_parser.py:888 _text_merge()
|
||||
func TextMerge(boxes []TextBox, medianHeights map[int]float64, zoom float64) []TextBox {
|
||||
func TextMerge(boxes []pdf.TextBox, medianHeights map[int]float64, zoom float64) []pdf.TextBox {
|
||||
if len(boxes) < 2 {
|
||||
return boxes
|
||||
}
|
||||
// Build output via collect: O(n) instead of O(n²) slice-element removal.
|
||||
out := make([]TextBox, 0, len(boxes))
|
||||
out := make([]pdf.TextBox, 0, len(boxes))
|
||||
i := 0
|
||||
for i < len(boxes) {
|
||||
cur := boxes[i]
|
||||
@@ -149,14 +152,14 @@ func TextMerge(boxes []TextBox, medianHeights map[int]float64, zoom float64) []T
|
||||
// Python: b.get("layoutno", "0") != b_.get("layoutno", "1") —
|
||||
// asymmetric defaults mean empty/missing layoutno never merge horizontally.
|
||||
if cur.LayoutNo != nxt.LayoutNo || cur.LayoutNo == "" || nxt.LayoutNo == "" ||
|
||||
cur.LayoutType == LayoutTypeTable || cur.LayoutType == LayoutTypeFigure || cur.LayoutType == LayoutTypeEquation {
|
||||
cur.LayoutType == pdf.LayoutTypeTable || cur.LayoutType == pdf.LayoutTypeFigure || cur.LayoutType == pdf.LayoutTypeEquation {
|
||||
break
|
||||
}
|
||||
mh := medianHeights[cur.PageNumber]
|
||||
if mh <= 0 {
|
||||
mh = 10
|
||||
}
|
||||
if math.Abs(BoxYDis(cur, nxt)) < mh/3 {
|
||||
if math.Abs(util.BoxYDis(cur, nxt)) < mh/3 {
|
||||
cur.X1 = nxt.X1
|
||||
cur.Top = (cur.Top + nxt.Top) / 2
|
||||
cur.Bottom = (cur.Bottom + nxt.Bottom) / 2
|
||||
@@ -176,7 +179,7 @@ func TextMerge(boxes []TextBox, medianHeights map[int]float64, zoom float64) []T
|
||||
// NaiveVerticalMerge vertically merges boxes on the same page/column.
|
||||
//
|
||||
// Python: pdf_parser.py:926 _naive_vertical_merge()
|
||||
func NaiveVerticalMerge(boxes []TextBox, medianHeights map[int]float64, medianWidths map[int]float64, isEnglish bool) []TextBox {
|
||||
func NaiveVerticalMerge(boxes []pdf.TextBox, medianHeights map[int]float64, medianWidths map[int]float64, isEnglish bool) []pdf.TextBox {
|
||||
if len(boxes) < 2 {
|
||||
return boxes
|
||||
}
|
||||
@@ -195,7 +198,7 @@ func NaiveVerticalMerge(boxes []TextBox, medianHeights map[int]float64, medianWi
|
||||
}
|
||||
sort.Ints(pageKeys)
|
||||
|
||||
var result []TextBox
|
||||
var result []pdf.TextBox
|
||||
for _, pg := range pageKeys {
|
||||
indices := groups[pg]
|
||||
sort.Slice(indices, func(i, j int) bool {
|
||||
@@ -205,14 +208,14 @@ func NaiveVerticalMerge(boxes []TextBox, medianHeights map[int]float64, medianWi
|
||||
}
|
||||
return bi.X0 < bj.X0
|
||||
})
|
||||
bxs := make([]TextBox, len(indices))
|
||||
bxs := make([]pdf.TextBox, len(indices))
|
||||
for i, idx := range indices {
|
||||
bxs[i] = boxes[idx]
|
||||
}
|
||||
|
||||
mh := medianHeights[pg]
|
||||
if mh <= 0 {
|
||||
mh = MedianHeight(bxs)
|
||||
mh = util.MedianHeight(bxs)
|
||||
}
|
||||
mw := medianWidths[pg]
|
||||
if mw <= 0 {
|
||||
@@ -220,7 +223,7 @@ func NaiveVerticalMerge(boxes []TextBox, medianHeights map[int]float64, medianWi
|
||||
}
|
||||
|
||||
// Collect pattern: build output slice, merging into last element when appropriate.
|
||||
out := make([]TextBox, 0, len(bxs))
|
||||
out := make([]pdf.TextBox, 0, len(bxs))
|
||||
for i := 0; i < len(bxs); i++ {
|
||||
b := bxs[i]
|
||||
// Cross-page suffix (e.g. page number on previous page): skip.
|
||||
@@ -233,7 +236,7 @@ func NaiveVerticalMerge(boxes []TextBox, medianHeights map[int]float64, medianWi
|
||||
// keeps whitespace inline and lets it extend the previous box.
|
||||
if len(out) > 0 {
|
||||
prev := &out[len(out)-1]
|
||||
if b.Top-prev.Bottom <= mh*1.5 && OverlapX(prev, &b) >= 0.3 {
|
||||
if b.Top-prev.Bottom <= mh*1.5 && util.OverlapX(prev, &b) >= 0.3 {
|
||||
// TODO: prev.Bottom = math.Max(prev.Bottom, b.Bottom) — direct assignment
|
||||
// can shrink a tall merged box when a short whitespace box overlaps.
|
||||
// Matches Python behavior (also direct assignment). Defer fix until
|
||||
@@ -259,7 +262,7 @@ func NaiveVerticalMerge(boxes []TextBox, medianHeights map[int]float64, medianWi
|
||||
out = append(out, b)
|
||||
continue
|
||||
}
|
||||
ov := OverlapX(prev, &b)
|
||||
ov := util.OverlapX(prev, &b)
|
||||
if ov < 0.3 {
|
||||
slog.Debug("vm reject", "reason", "ovX", "ov", ov, "threshold", 0.3)
|
||||
out = append(out, b)
|
||||
@@ -308,7 +311,7 @@ func NaiveVerticalMerge(boxes []TextBox, medianHeights map[int]float64, medianWi
|
||||
// FinalReadingOrderMerge sorts boxes by page → column → top → x0.
|
||||
//
|
||||
// Python: pdf_parser.py:1007 _final_reading_order_merge()
|
||||
func FinalReadingOrderMerge(boxes []TextBox) []TextBox {
|
||||
func FinalReadingOrderMerge(boxes []pdf.TextBox) []pdf.TextBox {
|
||||
if len(boxes) == 0 {
|
||||
return boxes
|
||||
}
|
||||
@@ -375,7 +378,60 @@ func startsWithOneOf(s, set string) bool {
|
||||
return strings.ContainsRune(set, r)
|
||||
}
|
||||
|
||||
// containsRune returns true if the string set contains the given rune.
|
||||
func containsRune(set string, r rune) bool {
|
||||
return strings.ContainsRune(set, r)
|
||||
// MergeSameBullet merges adjacent boxes that start with the same bullet/number
|
||||
// character, combining their text with a newline separator.
|
||||
func MergeSameBullet(boxes []pdf.TextBox, tok pdf.Tokenizer) []pdf.TextBox {
|
||||
if len(boxes) < 2 {
|
||||
return boxes
|
||||
}
|
||||
out := make([]pdf.TextBox, 0, len(boxes))
|
||||
i := 0
|
||||
for i < len(boxes) {
|
||||
if strings.TrimSpace(boxes[i].Text) == "" {
|
||||
i++
|
||||
continue
|
||||
}
|
||||
cur := boxes[i]
|
||||
i++
|
||||
for i < len(boxes) {
|
||||
if strings.TrimSpace(boxes[i].Text) == "" {
|
||||
i++
|
||||
continue
|
||||
}
|
||||
nxt := boxes[i]
|
||||
firstCur := firstRuneString(cur.Text)
|
||||
firstNxt := firstRuneString(nxt.Text)
|
||||
if firstCur != firstNxt ||
|
||||
unicode.Is(unicode.Latin, firstCur) ||
|
||||
isChinese(firstCur, tok) ||
|
||||
cur.Top > nxt.Bottom {
|
||||
break
|
||||
}
|
||||
cur.Text = cur.Text + "\n" + nxt.Text
|
||||
cur.X0 = min(cur.X0, nxt.X0)
|
||||
cur.X1 = max(cur.X1, nxt.X1)
|
||||
cur.Bottom = nxt.Bottom
|
||||
i++
|
||||
}
|
||||
out = append(out, cur)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func firstRuneString(s string) rune {
|
||||
s = strings.TrimSpace(s)
|
||||
if s == "" {
|
||||
return 0
|
||||
}
|
||||
return []rune(s)[0]
|
||||
}
|
||||
|
||||
// isChinese checks if a rune is a Chinese character (CJK Unified Ideograph).
|
||||
func isChinese(r rune, tok pdf.Tokenizer) bool {
|
||||
if tok != nil {
|
||||
return strings.Contains(tok.Tag(string(r)), "n")
|
||||
}
|
||||
return (r >= 0x4E00 && r <= 0x9FFF) ||
|
||||
(r >= 0x3400 && r <= 0x4DBF) ||
|
||||
(r >= 0x20000 && r <= 0x2A6DF)
|
||||
}
|
||||
@@ -1,12 +1,14 @@
|
||||
package parser
|
||||
package layout
|
||||
|
||||
import (
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
util "ragflow/internal/deepdoc/parser/pdf/util"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestAssignColumn(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{PageNumber: 0, X0: 50, Text: "col0-left"},
|
||||
{PageNumber: 0, X0: 55, Text: "col0-mid"},
|
||||
{PageNumber: 0, X0: 400, Text: "col1"},
|
||||
@@ -25,7 +27,7 @@ func TestAssignColumn(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestTextMerge(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{PageNumber: 0, ColID: 0, X0: 50, X1: 250, Top: 100, Bottom: 112, Text: "左半", LayoutType: "text", LayoutNo: "1"},
|
||||
{PageNumber: 0, ColID: 0, X0: 252, X1: 550, Top: 100, Bottom: 112, Text: "右半", LayoutType: "text", LayoutNo: "1"},
|
||||
}
|
||||
@@ -37,7 +39,7 @@ func TestTextMerge(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestTextMergeNoMerge_DiffLayout(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{PageNumber: 0, ColID: 0, X0: 50, X1: 250, Top: 100, Bottom: 112, Text: "text", LayoutType: "text", LayoutNo: "1"},
|
||||
{PageNumber: 0, ColID: 0, X0: 252, X1: 550, Top: 100, Bottom: 112, Text: "table", LayoutType: "table", LayoutNo: "2"},
|
||||
}
|
||||
@@ -49,7 +51,7 @@ func TestTextMergeNoMerge_DiffLayout(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestFinalReadingOrderMerge(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{PageNumber: 1, ColID: 1, Top: 50, Text: "pg1-col1"},
|
||||
{PageNumber: 0, ColID: 0, Top: 100, Text: "pg0-col0"},
|
||||
{PageNumber: 0, ColID: 0, Top: 50, Text: "pg0-col0-top"},
|
||||
@@ -64,10 +66,10 @@ func TestFinalReadingOrderMerge(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestContainsRune(t *testing.T) {
|
||||
if !containsRune("。?!", '。') {
|
||||
if !strings.ContainsRune("。?!", '。') {
|
||||
t.Error("should find 。")
|
||||
}
|
||||
if containsRune("abc", 'z') {
|
||||
if strings.ContainsRune("abc", 'z') {
|
||||
t.Error("should not find z")
|
||||
}
|
||||
}
|
||||
@@ -81,44 +83,8 @@ func TestEndsWithOneOf(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCharsToBoxes(t *testing.T) {
|
||||
chars := []TextChar{
|
||||
{X0: 50, X1: 58, Top: 100, Bottom: 112, Text: "A", PageNumber: 0},
|
||||
{X0: 60, X1: 68, Top: 100, Bottom: 112, Text: "B", PageNumber: 0},
|
||||
{X0: 50, X1: 58, Top: 114, Bottom: 126, Text: "C", PageNumber: 0},
|
||||
}
|
||||
boxes := charsToBoxes(chars, 0, false)
|
||||
if len(boxes) == 0 {
|
||||
t.Fatal("expected at least 1 box")
|
||||
}
|
||||
// A and B should be in the same line, C in a different line
|
||||
if len(boxes) != 2 {
|
||||
t.Errorf("expected 2 lines, got %d", len(boxes))
|
||||
}
|
||||
}
|
||||
|
||||
func TestBoxesToSections(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 100, Bottom: 112, Text: "标题"},
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 200, Bottom: 212, Text: ""},
|
||||
}
|
||||
sections := boxesToSections(boxes, nil)
|
||||
if len(sections) != 1 {
|
||||
t.Errorf("expected 1 section (empty box skipped), got %d", len(sections))
|
||||
}
|
||||
if len(sections) > 0 {
|
||||
// Text is clean — position tag lives in PositionTag field (matching Python)
|
||||
if strings.Contains(sections[0].Text, "@@") {
|
||||
t.Error("section text should NOT contain position tag")
|
||||
}
|
||||
if !strings.Contains(sections[0].PositionTag, "##") {
|
||||
t.Error("position tag should end with ##")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDefaultConfig(t *testing.T) {
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
if cfg.Zoom != 3 {
|
||||
t.Error("default zoom should be 3")
|
||||
}
|
||||
@@ -127,40 +93,10 @@ func TestDefaultConfig(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestHasColor(t *testing.T) {
|
||||
if !HasColor(TextChar{}) {
|
||||
t.Error("HasColor should return true by default")
|
||||
}
|
||||
}
|
||||
|
||||
func TestGroupCharsToLines_MultiColumn(t *testing.T) {
|
||||
// Simulate a two-column PDF page. Python's __ocr has no horizontal gap
|
||||
// check in line grouping — chars at the same vertical position are
|
||||
// grouped into one line regardless of horizontal distance. Column
|
||||
// separation happens downstream in AssignColumn + TextMerge.
|
||||
chars := []TextChar{
|
||||
{X0: 50, X1: 58, Top: 100, Bottom: 112, Text: "H"},
|
||||
{X0: 60, X1: 68, Top: 100, Bottom: 112, Text: "i"},
|
||||
{X0: 300, X1: 308, Top: 100, Bottom: 112, Text: "B"},
|
||||
{X0: 310, X1: 318, Top: 100, Bottom: 112, Text: "y"},
|
||||
{X0: 50, X1: 58, Top: 114, Bottom: 126, Text: "A"},
|
||||
{X0: 60, X1: 68, Top: 114, Bottom: 126, Text: "B"},
|
||||
{X0: 300, X1: 308, Top: 114, Bottom: 126, Text: "C"},
|
||||
{X0: 310, X1: 318, Top: 114, Bottom: 126, Text: "D"},
|
||||
}
|
||||
|
||||
lines := groupCharsToLines(chars, false)
|
||||
|
||||
// Python expects 2 lines (one per vertical position), each spanning both columns.
|
||||
if len(lines) != 2 {
|
||||
t.Errorf("expected 2 lines (one per vertical row, spanning both columns), got %d", len(lines))
|
||||
}
|
||||
}
|
||||
|
||||
func TestKmeans1D_Boundary(t *testing.T) {
|
||||
t.Run("n equals k", func(t *testing.T) {
|
||||
data := []float64{50.0, 400.0}
|
||||
labels, centroids := kmeans1D(data, 2)
|
||||
labels, centroids := util.KMeans1D(data, 2)
|
||||
if len(centroids) != 2 {
|
||||
t.Errorf("n=k=2: expected 2 centroids, got %d — BUG: n<=k early return gives only 1 centroid", len(centroids))
|
||||
}
|
||||
@@ -171,7 +107,7 @@ func TestKmeans1D_Boundary(t *testing.T) {
|
||||
|
||||
t.Run("n less than k", func(t *testing.T) {
|
||||
data := []float64{100.0, 200.0, 300.0}
|
||||
labels, centroids := kmeans1D(data, 4)
|
||||
labels, centroids := util.KMeans1D(data, 4)
|
||||
if len(centroids) != 3 {
|
||||
t.Errorf("n=3,k=4: expected 3 centroids (one per point), got %d — BUG: n<=k early return gives only 1 centroid", len(centroids))
|
||||
}
|
||||
@@ -187,7 +123,7 @@ func TestKmeans1D_Boundary(t *testing.T) {
|
||||
|
||||
t.Run("single point", func(t *testing.T) {
|
||||
data := []float64{100.0}
|
||||
labels, centroids := kmeans1D(data, 1)
|
||||
labels, centroids := util.KMeans1D(data, 1)
|
||||
if len(centroids) != 1 || centroids[0] != 100.0 {
|
||||
t.Errorf("single point: unexpected centroids %v", centroids)
|
||||
}
|
||||
@@ -282,7 +218,7 @@ func TestNaiveVerticalMerge_CommaConcat(t *testing.T) {
|
||||
// ASCII comma ',' is in Python's concatting set, Go matches.
|
||||
// When there's NO anti trigger, merge happens by default.
|
||||
// The concatting feature is only needed when it must OVERRIDE an anti trigger.
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{
|
||||
PageNumber: 0, X0: 50, X1: 250, Top: 100, Bottom: 112,
|
||||
Text: "这是第一句话",
|
||||
@@ -308,7 +244,7 @@ func TestNaiveVerticalMerge_CommaConcat(t *testing.T) {
|
||||
// Python: previous line ends with "。" (anti), next line starts with ","
|
||||
// (concatting). Concatting OVERRIDES anti → merge.
|
||||
// Go now matches Python: ',' is in concatting set → merge.
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{
|
||||
PageNumber: 0, X0: 50, X1: 250, Top: 100, Bottom: 112,
|
||||
Text: "前一句话结束。",
|
||||
@@ -331,7 +267,7 @@ func TestNaiveVerticalMerge_CommaConcat(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("next line starts with fullwidth comma — should merge", func(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{
|
||||
PageNumber: 0, X0: 50, X1: 250, Top: 100, Bottom: 112,
|
||||
Text: "这是第一句话",
|
||||
@@ -353,7 +289,7 @@ func TestNaiveVerticalMerge_CommaConcat(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("next line starts with period — should merge", func(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{
|
||||
PageNumber: 0, X0: 50, X1: 250, Top: 100, Bottom: 112,
|
||||
Text: "前文内容",
|
||||
@@ -378,7 +314,7 @@ func TestNaiveVerticalMerge_CommaConcat(t *testing.T) {
|
||||
// Python's _naive_vertical_merge: merge is the DEFAULT.
|
||||
// concatting overrides anti; anti + detach prevent merge.
|
||||
// When none trigger, boxes merge.
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{
|
||||
PageNumber: 0, X0: 50, X1: 250, Top: 100, Bottom: 112,
|
||||
Text: "这是第一句话",
|
||||
@@ -401,7 +337,7 @@ func TestNaiveVerticalMerge_CommaConcat(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("detach — horizontally separated boxes", func(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{
|
||||
PageNumber: 0, X0: 50, X1: 100, Top: 100, Bottom: 112,
|
||||
Text: "左列文字",
|
||||
@@ -424,7 +360,7 @@ func TestNaiveVerticalMerge_CommaConcat(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("large vertical gap — anti", func(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{
|
||||
PageNumber: 0, X0: 50, X1: 250, Top: 100, Bottom: 112,
|
||||
Text: "第一句话",
|
||||
@@ -447,7 +383,7 @@ func TestNaiveVerticalMerge_CommaConcat(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("english period anti when isEnglish", func(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{
|
||||
PageNumber: 0, X0: 50, X1: 250, Top: 100, Bottom: 112,
|
||||
Text: "End of sentence.",
|
||||
@@ -470,7 +406,7 @@ func TestNaiveVerticalMerge_CommaConcat(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("cross-page — should NOT merge", func(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{
|
||||
PageNumber: 0, X0: 50, X1: 250, Top: 100, Bottom: 112,
|
||||
Text: "第一页最后一行",
|
||||
@@ -497,14 +433,14 @@ func TestNaiveVerticalMerge_CommaConcat(t *testing.T) {
|
||||
if len(result) != 0 {
|
||||
t.Error("expected empty result for nil input")
|
||||
}
|
||||
result = NaiveVerticalMerge([]TextBox{}, nil, nil, false)
|
||||
result = NaiveVerticalMerge([]pdf.TextBox{}, nil, nil, false)
|
||||
if len(result) != 0 {
|
||||
t.Error("expected empty result for empty input")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("single box", func(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{PageNumber: 0, X0: 50, X1: 250, Top: 100, Bottom: 112, Text: "only", LayoutNo: "1"},
|
||||
}
|
||||
result := NaiveVerticalMerge(boxes, nil, nil, false)
|
||||
@@ -514,91 +450,6 @@ func TestNaiveVerticalMerge_CommaConcat(t *testing.T) {
|
||||
})
|
||||
}
|
||||
|
||||
// ── charsToBoxes whitespace preservation ────────────────────────────────
|
||||
// Whitespace boxes are preserved (not pre-filtered) so they can act as
|
||||
// gap bridges in NaiveVerticalMerge.
|
||||
|
||||
func TestCharsToBoxes_PreservesWhitespaceLines(t *testing.T) {
|
||||
chars := []TextChar{
|
||||
{Text: " ", X0: 10, Top: 100, X1: 15, Bottom: 112}, // non-breaking space only
|
||||
{Text: "Hello", X0: 10, Top: 120, X1: 50, Bottom: 132}, // real text
|
||||
{Text: " ", X0: 10, Top: 140, X1: 15, Bottom: 152}, // spaces only
|
||||
}
|
||||
boxes := charsToBoxes(chars, 0, false)
|
||||
|
||||
if len(boxes) != 3 {
|
||||
t.Fatalf("expected 3 boxes (whitespace preserved for VM gap bridging), got %d", len(boxes))
|
||||
}
|
||||
if boxes[1].Text != "Hello" {
|
||||
t.Errorf("expected 'Hello', got %q", boxes[1].Text)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCharsToBoxes_PreservesAllWhitespace(t *testing.T) {
|
||||
chars := []TextChar{
|
||||
{Text: " ", X0: 10, Top: 100, X1: 15, Bottom: 112},
|
||||
{Text: " ", X0: 20, Top: 120, X1: 25, Bottom: 132},
|
||||
}
|
||||
boxes := charsToBoxes(chars, 0, false)
|
||||
if len(boxes) != 2 {
|
||||
t.Fatalf("expected 2 boxes (whitespace preserved), got %d", len(boxes))
|
||||
}
|
||||
}
|
||||
|
||||
func TestCharsToBoxes_EmptyInput(t *testing.T) {
|
||||
if boxes := charsToBoxes(nil, 0, false); boxes != nil {
|
||||
t.Errorf("expected nil for nil input, got %d boxes", len(boxes))
|
||||
}
|
||||
if boxes := charsToBoxes([]TextChar{}, 0, false); boxes != nil {
|
||||
t.Errorf("expected nil for empty input, got %d boxes", len(boxes))
|
||||
}
|
||||
}
|
||||
|
||||
// ---- groupCharsToLines: stable sort for close x0 values ----
|
||||
|
||||
func TestGroupCharsToLines_StableSort(t *testing.T) {
|
||||
// Simulate CJK chars with near-identical Top and very close x0 values.
|
||||
// Non-stable sort can scramble the order, breaking text.
|
||||
chars := []TextChar{
|
||||
{Text: "总", X0: 37.6, X1: 48.0, Top: 60.5, Bottom: 70.9},
|
||||
{Text: "结", X0: 48.0, X1: 58.4, Top: 60.5, Bottom: 70.9},
|
||||
{Text: "前", X0: 37.6, X1: 48.0, Top: 86.1, Bottom: 96.5},
|
||||
{Text: "2", X0: 48.0, X1: 54.0, Top: 86.1, Bottom: 96.5},
|
||||
{Text: "个", X0: 53.9, X1: 64.4, Top: 86.1, Bottom: 96.5},
|
||||
{Text: "问", X0: 64.4, X1: 74.8, Top: 86.1, Bottom: 96.5},
|
||||
{Text: "题", X0: 74.8, X1: 85.2, Top: 86.1, Bottom: 96.5},
|
||||
}
|
||||
|
||||
// Run multiple times — if sort is unstable, text order will vary
|
||||
for run := 0; run < 10; run++ {
|
||||
copy := make([]TextChar, len(chars))
|
||||
for i := range chars {
|
||||
copy[i] = chars[i]
|
||||
}
|
||||
lines := groupCharsToLines(copy, false)
|
||||
if len(lines) != 2 {
|
||||
t.Fatalf("expected 2 lines, got %d", len(lines))
|
||||
}
|
||||
boxes := make([]TextBox, 0)
|
||||
for _, line := range lines {
|
||||
boxes = append(boxes, lineToTextBox(line))
|
||||
}
|
||||
// First line must be "总结" in correct order
|
||||
if !strings.HasPrefix(boxes[0].Text, "总结") {
|
||||
t.Errorf("run %d: first line should start with '总结', got %q", run, boxes[0].Text[:min(6, len(boxes[0].Text))])
|
||||
}
|
||||
// Second line should contain "前2个问题"
|
||||
if !strings.Contains(boxes[1].Text, "前") || !strings.Contains(boxes[1].Text, "题") {
|
||||
t.Errorf("run %d: second line text scrambled: %q", run, boxes[1].Text[:min(20, len(boxes[1].Text))])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestNaiveVerticalMerge_BottomShrink exposes a bug where merging a short
|
||||
// box into a tall previously-merged box SHRINKS prev.Bottom instead of
|
||||
// keeping it via math.Max. X0/X1 correctly use Min/Max, Bottom does not.
|
||||
//
|
||||
// This test is expected to FAIL until the fix (prev.Bottom = math.Max(...))
|
||||
// is applied.
|
||||
func TestNaiveVerticalMerge_BottomShrink(t *testing.T) {
|
||||
// Three boxes on the same page, sorted by Top.
|
||||
@@ -606,7 +457,7 @@ func TestNaiveVerticalMerge_BottomShrink(t *testing.T) {
|
||||
// C overlaps vertically (Top=290 < prev.Bottom=300) but is short (Bottom=295).
|
||||
// Current code: prev.Bottom = 295 (shrinks from 300).
|
||||
// Correct: prev.Bottom = max(300, 295) = 300.
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 50, X1: 500, Top: 100, Bottom: 150, Text: "line one", PageNumber: 0},
|
||||
{X0: 50, X1: 500, Top: 160, Bottom: 300, Text: "tall paragraph that spans many lines", PageNumber: 0},
|
||||
{X0: 50, X1: 500, Top: 290, Bottom: 295, Text: "short overlap", PageNumber: 0},
|
||||
@@ -625,3 +476,32 @@ func TestNaiveVerticalMerge_BottomShrink(t *testing.T) {
|
||||
t.Skipf("known issue: Bottom shrunk to %.1f (want >= 300) — deferred until pipeline alignment", result[0].Bottom)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNaiveVerticalMerge(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{PageNumber: 0, ColID: 0, X0: 50, X1: 550, Top: 100, Bottom: 112, Text: "第一段", LayoutNo: "1", LayoutType: "text"},
|
||||
{PageNumber: 0, ColID: 0, X0: 50, X1: 550, Top: 114, Bottom: 126, Text: "续文", LayoutNo: "1", LayoutType: "text"},
|
||||
}
|
||||
meanH := map[int]float64{0: 12}
|
||||
meanW := map[int]float64{0: 5}
|
||||
result := NaiveVerticalMerge(boxes, meanH, meanW, false)
|
||||
if len(result) != 1 {
|
||||
t.Errorf("expected 1 merged box, got %d: %v", len(result), result)
|
||||
}
|
||||
if len(result) > 0 && !strings.Contains(result[0].Text, "第一段") {
|
||||
t.Errorf("merged text should contain '第一段': got %q", result[0].Text)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNaiveVerticalMergeNonMerge(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{PageNumber: 0, ColID: 0, X0: 50, X1: 550, Top: 100, Bottom: 112, Text: "第一段。", LayoutNo: "1", LayoutType: "text"},
|
||||
{PageNumber: 0, ColID: 0, X0: 50, X1: 550, Top: 300, Bottom: 312, Text: "第二段。", LayoutNo: "1", LayoutType: "text"},
|
||||
}
|
||||
meanH := map[int]float64{0: 12}
|
||||
meanW := map[int]float64{0: 5}
|
||||
result := NaiveVerticalMerge(boxes, meanH, meanW, false)
|
||||
if len(result) != 2 {
|
||||
t.Errorf("expected 2 separate boxes (large gap), got %d", len(result))
|
||||
}
|
||||
}
|
||||
@@ -4,18 +4,19 @@ import (
|
||||
"context"
|
||||
"fmt"
|
||||
"image"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// MockDocAnalyzer returns predefined data for unit tests.
|
||||
// Set an Err field to non-nil to exercise the corresponding error path.
|
||||
type MockDocAnalyzer struct {
|
||||
DLARegions []DLARegion
|
||||
TSRCells []TSRCell
|
||||
OCRBoxes []OCRBox
|
||||
OCRTexts []OCRText
|
||||
DLARegions []pdf.DLARegion
|
||||
TSRCells []pdf.TSRCell
|
||||
OCRBoxes []pdf.OCRBox
|
||||
OCRTexts []pdf.OCRText
|
||||
// OCRBatchTexts returns per-image texts for OCRRecognizeBatch.
|
||||
// If nil, OCRTexts is returned for every image.
|
||||
OCRBatchTexts [][]OCRText
|
||||
OCRBatchTexts [][]pdf.OCRText
|
||||
// OCRBatchErr makes OCRRecognizeBatch return an error for image i.
|
||||
OCRBatchErr func(i int) error
|
||||
// Per-method error injection for testing failure paths.
|
||||
@@ -25,35 +26,34 @@ type MockDocAnalyzer struct {
|
||||
OCRRecognizeErr error
|
||||
|
||||
Healthy bool
|
||||
Model ModelType
|
||||
}
|
||||
|
||||
func (m *MockDocAnalyzer) DLA(_ context.Context, _ image.Image) ([]DLARegion, error) {
|
||||
func (m *MockDocAnalyzer) DLA(_ context.Context, _ image.Image) ([]pdf.DLARegion, error) {
|
||||
if m.DLAErr != nil {
|
||||
return nil, m.DLAErr
|
||||
}
|
||||
return m.DLARegions, nil
|
||||
}
|
||||
func (m *MockDocAnalyzer) TSR(_ context.Context, _ image.Image) ([]TSRCell, error) {
|
||||
func (m *MockDocAnalyzer) TSR(_ context.Context, _ image.Image) ([]pdf.TSRCell, error) {
|
||||
if m.TSRErr != nil {
|
||||
return nil, m.TSRErr
|
||||
}
|
||||
return m.TSRCells, nil
|
||||
}
|
||||
func (m *MockDocAnalyzer) OCRDetect(_ context.Context, _ image.Image) ([]OCRBox, error) {
|
||||
func (m *MockDocAnalyzer) OCRDetect(_ context.Context, _ image.Image) ([]pdf.OCRBox, error) {
|
||||
if m.OCRDetectErr != nil {
|
||||
return nil, m.OCRDetectErr
|
||||
}
|
||||
return m.OCRBoxes, nil
|
||||
}
|
||||
func (m *MockDocAnalyzer) OCRRecognize(_ context.Context, _ image.Image) ([]OCRText, error) {
|
||||
func (m *MockDocAnalyzer) OCRRecognize(_ context.Context, _ image.Image) ([]pdf.OCRText, error) {
|
||||
if m.OCRRecognizeErr != nil {
|
||||
return nil, m.OCRRecognizeErr
|
||||
}
|
||||
return m.OCRTexts, nil
|
||||
}
|
||||
func (m *MockDocAnalyzer) OCRRecognizeBatch(_ context.Context, cropped []image.Image) ([][]OCRText, []error) {
|
||||
results := make([][]OCRText, len(cropped))
|
||||
func (m *MockDocAnalyzer) OCRRecognizeBatch(_ context.Context, cropped []image.Image) ([][]pdf.OCRText, []error) {
|
||||
results := make([][]pdf.OCRText, len(cropped))
|
||||
errs := make([]error, len(cropped))
|
||||
for i, img := range cropped {
|
||||
if img == nil {
|
||||
@@ -71,5 +71,4 @@ func (m *MockDocAnalyzer) OCRRecognizeBatch(_ context.Context, cropped []image.I
|
||||
}
|
||||
return results, errs
|
||||
}
|
||||
func (m *MockDocAnalyzer) Health() bool { return m.Healthy }
|
||||
func (m *MockDocAnalyzer) ModelType() ModelType { return m.Model }
|
||||
func (m *MockDocAnalyzer) Health() bool { return m.Healthy }
|
||||
@@ -19,7 +19,7 @@ func TestOCR_mergeChars_RealScanned(t *testing.T) {
|
||||
if url == "" {
|
||||
t.Skip("DEEPDOC_URL not set")
|
||||
}
|
||||
dd, err := NewDeepDocClient(url)
|
||||
dd, err := inf.NewInferenceClient(url)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -51,8 +51,8 @@ func TestOCR_mergeChars_RealScanned(t *testing.T) {
|
||||
sample.WriteString(c.Text)
|
||||
}
|
||||
t.Logf("pdf_oxide sample: %q", sample.String())
|
||||
t.Logf("isScanNoise: %v", isScanNoise(sample.String()))
|
||||
t.Logf("isGarbledPage: %v", isGarbledPage(chars))
|
||||
t.Logf("isScanNoise: %v", util.IsScanNoise(sample.String()))
|
||||
t.Logf("isGarbledPage: %v", util.IsGarbledPage(chars))
|
||||
|
||||
img, err := eng.RenderPageImage(0, 72*3)
|
||||
if err != nil {
|
||||
|
||||
@@ -6,6 +6,7 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"image"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"testing"
|
||||
)
|
||||
|
||||
@@ -27,7 +28,7 @@ func TestOCRRecognizeBatch_EmptyList(t *testing.T) {
|
||||
func TestOCRRecognizeBatch_SingleImage(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRTexts: []OCRText{{Text: "hello", Confidence: 0.9}},
|
||||
OCRTexts: []pdf.OCRText{{Text: "hello", Confidence: 0.9}},
|
||||
}
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 10, 10))
|
||||
results, errs := mock.OCRRecognizeBatch(context.Background(), []image.Image{dummy})
|
||||
@@ -45,7 +46,7 @@ func TestOCRRecognizeBatch_SingleImage(t *testing.T) {
|
||||
func TestOCRRecognizeBatch_MultipleImages(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRBatchTexts: [][]OCRText{
|
||||
OCRBatchTexts: [][]pdf.OCRText{
|
||||
{{Text: "img0", Confidence: 0.9}},
|
||||
{{Text: "img1", Confidence: 0.8}},
|
||||
{{Text: "img2", Confidence: 0.7}},
|
||||
@@ -69,7 +70,7 @@ func TestOCRRecognizeBatch_MultipleImages(t *testing.T) {
|
||||
func TestOCRRecognizeBatch_NilImage(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRTexts: []OCRText{{Text: "ok", Confidence: 0.9}},
|
||||
OCRTexts: []pdf.OCRText{{Text: "ok", Confidence: 0.9}},
|
||||
}
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 10, 10))
|
||||
results, errs := mock.OCRRecognizeBatch(context.Background(), []image.Image{dummy, nil, dummy})
|
||||
@@ -93,7 +94,7 @@ func TestOCRRecognizeBatch_NilImage(t *testing.T) {
|
||||
func TestOCRRecognizeBatch_ErrorHandling(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRTexts: []OCRText{{Text: "ok", Confidence: 0.9}},
|
||||
OCRTexts: []pdf.OCRText{{Text: "ok", Confidence: 0.9}},
|
||||
OCRBatchErr: func(i int) error {
|
||||
if i == 1 {
|
||||
return errors.New("simulated error")
|
||||
@@ -130,7 +131,7 @@ func TestOCRRecognizeBatch_ErrorHandling(t *testing.T) {
|
||||
func TestOCRRecognizeBatch_EmptyText(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRTexts: []OCRText{}, // empty — simulate no text recognized
|
||||
OCRTexts: []pdf.OCRText{}, // empty — simulate no text recognized
|
||||
}
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 10, 10))
|
||||
results, errs := mock.OCRRecognizeBatch(context.Background(), []image.Image{dummy})
|
||||
@@ -149,7 +150,7 @@ func TestOCRRecognizeBatch_FallbackToOCRTexts(t *testing.T) {
|
||||
// When OCRBatchTexts is nil, fall back to OCRTexts for every image.
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRTexts: []OCRText{{Text: "default", Confidence: 0.5}},
|
||||
OCRTexts: []pdf.OCRText{{Text: "default", Confidence: 0.5}},
|
||||
}
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 10, 10))
|
||||
results, errs := mock.OCRRecognizeBatch(context.Background(), []image.Image{dummy, dummy, dummy})
|
||||
@@ -170,8 +171,8 @@ func TestOCRRecognizeBatch_PartialBatchTexts(t *testing.T) {
|
||||
// OCRBatchTexts shorter than images — remaining fall back to OCRTexts.
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRTexts: []OCRText{{Text: "fallback", Confidence: 0.5}},
|
||||
OCRBatchTexts: [][]OCRText{
|
||||
OCRTexts: []pdf.OCRText{{Text: "fallback", Confidence: 0.5}},
|
||||
OCRBatchTexts: [][]pdf.OCRText{
|
||||
{{Text: "custom0", Confidence: 0.9}},
|
||||
},
|
||||
}
|
||||
|
||||
106
internal/deepdoc/parser/pdf/outline_extraction_test.go
Normal file
106
internal/deepdoc/parser/pdf/outline_extraction_test.go
Normal file
@@ -0,0 +1,106 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// ── outline-tracking mock engines ──────────────────────────────────────────
|
||||
|
||||
// outlineTrackingEngine wraps mockEngine and records whether Outlines()
|
||||
// was called.
|
||||
type outlineTrackingEngine struct {
|
||||
*mockEngine
|
||||
outlines []pdf.Outline
|
||||
outlinesCalled bool
|
||||
}
|
||||
|
||||
func (e *outlineTrackingEngine) Outlines() ([]pdf.Outline, error) {
|
||||
e.outlinesCalled = true
|
||||
return e.outlines, nil
|
||||
}
|
||||
|
||||
// outlineErrorEngine returns an error from Outlines().
|
||||
type outlineErrorEngine struct {
|
||||
*mockEngine
|
||||
}
|
||||
|
||||
func (e *outlineErrorEngine) Outlines() ([]pdf.Outline, error) {
|
||||
return nil, errors.New("pdfium outline extraction failed")
|
||||
}
|
||||
|
||||
// ── tests for outline extraction in Parse() ─────────────────────────────────
|
||||
|
||||
// TestParse_ExtractsOutlinesFromEngine verifies that Parse() calls
|
||||
// engine.Outlines() and the result carries the outlines.
|
||||
//
|
||||
// This test currently FAILS because:
|
||||
// 1. Parse() never calls engine.Outlines() → outlinesCalled stays false
|
||||
// 2. ParseResult has no Outlines field → compilation error if we try to read it
|
||||
func TestParse_ExtractsOutlinesFromEngine(t *testing.T) {
|
||||
expectedOutlines := []pdf.Outline{
|
||||
{Title: "Chapter 1", Level: 0, PageNumber: 1},
|
||||
{Title: "Section 1.1", Level: 1, PageNumber: 2},
|
||||
}
|
||||
eng := &outlineTrackingEngine{
|
||||
mockEngine: &mockEngine{pageCount: 3},
|
||||
outlines: expectedOutlines,
|
||||
}
|
||||
mockDLA := &MockDocAnalyzer{Healthy: true}
|
||||
p := NewParser(pdf.DefaultParserConfig(), mockDLA)
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse failed: %v", err)
|
||||
}
|
||||
if result == nil {
|
||||
t.Fatal("Parse returned nil result")
|
||||
}
|
||||
|
||||
// Check 1: engine.Outlines() was actually called
|
||||
if !eng.outlinesCalled {
|
||||
t.Error("BUG: Parse() never called engine.Outlines() — outlines are extracted by pdfium but ignored")
|
||||
}
|
||||
|
||||
// Check 2: outlines are present in ParseResult
|
||||
if len(result.Outlines) == 0 {
|
||||
t.Error("BUG: ParseResult.Outlines is empty — outlines extracted but not stored")
|
||||
}
|
||||
if len(result.Outlines) != len(expectedOutlines) {
|
||||
t.Errorf("result.Outlines: got %d, want %d", len(result.Outlines), len(expectedOutlines))
|
||||
}
|
||||
}
|
||||
|
||||
// TestParse_OutlinesErrorDoesNotBlockParsing verifies that when
|
||||
// engine.Outlines() fails, the parse still completes successfully
|
||||
// and produces sections (outlines are best-effort).
|
||||
func TestParse_OutlinesErrorDoesNotBlockParsing(t *testing.T) {
|
||||
eng := &outlineErrorEngine{
|
||||
mockEngine: &mockEngine{
|
||||
pageCount: 2,
|
||||
chars: map[int][]pdf.TextChar{
|
||||
0: {{Text: "Hello world", X0: 100, X1: 200, Top: 100, Bottom: 120}},
|
||||
1: {{Text: "Page two", X0: 100, X1: 200, Top: 100, Bottom: 120}},
|
||||
},
|
||||
},
|
||||
}
|
||||
mockDLA := &MockDocAnalyzer{Healthy: true}
|
||||
p := NewParser(pdf.DefaultParserConfig(), mockDLA)
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse should not fail when Outlines() errors: %v", err)
|
||||
}
|
||||
if result == nil {
|
||||
t.Fatal("Parse returned nil result")
|
||||
}
|
||||
if len(result.Sections) == 0 {
|
||||
t.Error("Parse should still produce sections even if Outlines() fails")
|
||||
}
|
||||
if len(result.Outlines) != 0 {
|
||||
t.Errorf("outlines should be empty on error, got %d", len(result.Outlines))
|
||||
}
|
||||
}
|
||||
90
internal/deepdoc/parser/pdf/page_batch_test.go
Normal file
90
internal/deepdoc/parser/pdf/page_batch_test.go
Normal file
@@ -0,0 +1,90 @@
|
||||
//go:build cgo && manual
|
||||
|
||||
package parser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"ragflow/internal/deepdoc/parser/pdf/tool"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// TestParse_BatchEquivalence verifies that batched processing produces
|
||||
// the same output as processing all pages at once. Uses batchSize=1
|
||||
// (every page is its own batch) on a multi-page fixture to maximize
|
||||
// batch boundary stress.
|
||||
func TestParse_BatchEquivalence(t *testing.T) {
|
||||
data, err := readTestPDF(t, "03_multipage.pdf")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
parse := func(batchSize int) *pdf.ParseResult {
|
||||
eng, err := NewEngine(data)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer eng.Close()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
cfg.BatchSize = batchSize
|
||||
p := NewParser(cfg, &MockDocAnalyzer{Healthy: true})
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// No batching (all pages at once).
|
||||
full := parse(9999)
|
||||
// Aggressive batching (1 page per batch).
|
||||
batched := parse(1)
|
||||
|
||||
// Compare section counts.
|
||||
if len(full.Sections) != len(batched.Sections) {
|
||||
t.Logf("section count: full=%d batched=%d (small diff acceptable at batch boundaries)",
|
||||
len(full.Sections), len(batched.Sections))
|
||||
}
|
||||
|
||||
// Compare text content via CharSimilarity.
|
||||
fullText := sectionsText(full.Sections)
|
||||
batchedText := sectionsText(batched.Sections)
|
||||
charSim := tool.CharSimilarity(fullText, batchedText)
|
||||
t.Logf("CharSimilarity: %.1f%%", charSim)
|
||||
if charSim < 95 {
|
||||
t.Errorf("batch equivalence too low: CharSim=%.1f%% (want >= 95%%)", charSim)
|
||||
}
|
||||
|
||||
// Compare metrics (should be identical or very close).
|
||||
t.Logf("Metrics: full=%+v batched=%+v", full.Metrics, batched.Metrics)
|
||||
if full.Metrics.BoxesInitial != batched.Metrics.BoxesInitial {
|
||||
t.Errorf("BoxesInitial: full=%d batched=%d",
|
||||
full.Metrics.BoxesInitial, batched.Metrics.BoxesInitial)
|
||||
}
|
||||
|
||||
// Bug fix regression: PageImages must survive batched merge.
|
||||
if len(full.PageImages) == 0 {
|
||||
t.Error("full parse: PageImages should not be empty (3-page document)")
|
||||
}
|
||||
if len(batched.PageImages) == 0 {
|
||||
t.Error("batched parse: PageImages should be preserved across batches")
|
||||
}
|
||||
}
|
||||
|
||||
func readTestPDF(t *testing.T, name string) ([]byte, error) {
|
||||
t.Helper()
|
||||
return os.ReadFile(filepath.Join("testdata", "pdfs", name))
|
||||
}
|
||||
|
||||
func sectionsText(sections []pdf.Section) string {
|
||||
var sb strings.Builder
|
||||
for _, s := range sections {
|
||||
sb.WriteString(s.Text)
|
||||
sb.WriteByte('\n')
|
||||
}
|
||||
return sb.String()
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
471
internal/deepdoc/parser/pdf/parser_mock_test.go
Normal file
471
internal/deepdoc/parser/pdf/parser_mock_test.go
Normal file
@@ -0,0 +1,471 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"image"
|
||||
inf "ragflow/internal/deepdoc/parser/pdf/inference"
|
||||
lyt "ragflow/internal/deepdoc/parser/pdf/layout"
|
||||
tbl "ragflow/internal/deepdoc/parser/pdf/table"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// ── MockDocAnalyzer tests ──────────────────────────────────────────────
|
||||
|
||||
func TestMockDocAnalyzer(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 100, Label: "table", Confidence: 0.95},
|
||||
},
|
||||
TSRCells: []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 50, Y1: 30, Text: "A"},
|
||||
},
|
||||
}
|
||||
|
||||
if !mock.Health() {
|
||||
t.Error("mock should be healthy")
|
||||
}
|
||||
regions, _ := mock.DLA(context.Background(), nil)
|
||||
if len(regions) != 1 || regions[0].Label != "table" {
|
||||
t.Error("mock DLA returned wrong data")
|
||||
}
|
||||
cells, _ := mock.TSR(context.Background(), nil)
|
||||
if len(cells) != 1 || cells[0].Text != "A" {
|
||||
t.Error("mock TSR returned wrong data")
|
||||
}
|
||||
// OCRDetect + OCRRecognize replaces deprecated OCR — tested in TestOCR_scanPage/TestOCR_fallback.
|
||||
_ = mock.OCRDetect
|
||||
_ = mock.OCRRecognize
|
||||
|
||||
// Unhealthy mock
|
||||
mock2 := &MockDocAnalyzer{Healthy: false}
|
||||
if mock2.Health() {
|
||||
t.Error("unhealthy mock should return false")
|
||||
}
|
||||
}
|
||||
|
||||
// ── enrichWithDeepDoc noop ─────────────────────────────────────────────
|
||||
|
||||
func TestEnrichWithDeepDoc_Noop(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 100, Bottom: 112, Text: "text"},
|
||||
}
|
||||
eng := &mockEngine{pageCount: 1}
|
||||
|
||||
p := NewParser(pdf.DefaultParserConfig(), &MockDocAnalyzer{Healthy: false})
|
||||
tables := p.enrichWithDeepDoc(context.Background(), nil, eng, boxes, nil)
|
||||
if len(tables) != 0 {
|
||||
t.Error("unhealthy DeepDoc → 0 Tables")
|
||||
}
|
||||
}
|
||||
|
||||
// ── extractTableBoxesFromImage with mock ───────────────────────────────
|
||||
|
||||
func TestExtractTableBoxes_Mock(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{PageNumber: 0, X0: 80, X1: 500, Top: 200, Bottom: 550, Text: "cell 1"},
|
||||
{PageNumber: 0, X0: 80, X1: 500, Top: 550, Bottom: 760, Text: "cell 2"},
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 100, Bottom: 180, Text: "heading"},
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 780, Bottom: 850, Text: "below"},
|
||||
}
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []pdf.DLARegion{
|
||||
{X0: 250, Y0: 600, X1: 1500, Y1: 2300, Label: "table", Confidence: 0.95},
|
||||
},
|
||||
TSRCells: []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 600, Y1: 400, Text: "A1"},
|
||||
{X0: 600, Y0: 0, X1: 1240, Y1: 400, Text: "B1"},
|
||||
{X0: 0, Y0: 410, X1: 600, Y1: 800, Text: "A2"},
|
||||
{X0: 600, Y0: 410, X1: 1240, Y1: 800, Text: "B2"},
|
||||
},
|
||||
}
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
dummyImg := image.NewRGBA(image.Rect(0, 0, 2000, 3000))
|
||||
|
||||
tables := p.extractTableBoxesFromImage(context.Background(), nil, boxes, dummyImg, 0, 0)
|
||||
if len(tables) != 1 {
|
||||
t.Fatalf("expected 1 pdf.TableItem, got %d", len(tables))
|
||||
}
|
||||
tbl := tables[0]
|
||||
if len(tbl.Cells) != 4 {
|
||||
t.Errorf("expected 4 cells, got %d", len(tbl.Cells))
|
||||
}
|
||||
// Rows populated later by constructTable via extractTableAndReplace.
|
||||
if tbl.ImageB64 == "" {
|
||||
t.Error("ImageB64 empty")
|
||||
}
|
||||
if len(tbl.Positions) != 2 {
|
||||
t.Errorf("expected 2 Positions, got %d", len(tbl.Positions))
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractTableBoxes_NoTables(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{Healthy: true, DLARegions: []pdf.DLARegion{}}
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 1000, 1000))
|
||||
tables := p.extractTableBoxesFromImage(context.Background(), nil, nil, dummy, 0, 0)
|
||||
if len(tables) != 0 {
|
||||
t.Errorf("0 tables expected, got %d", len(tables))
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractTableBoxes_NonTableRegions(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []pdf.DLARegion{
|
||||
{X0: 150, Y0: 300, X1: 1650, Y1: 336, Label: "text", Confidence: 0.9},
|
||||
{X0: 150, Y0: 600, X1: 1650, Y1: 900, Label: "figure", Confidence: 0.8},
|
||||
},
|
||||
}
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 2000, 2000))
|
||||
tables := p.extractTableBoxesFromImage(context.Background(), nil, nil, dummy, 0, 0)
|
||||
if len(tables) != 0 {
|
||||
t.Errorf("non-table regions → 0 tables, got %d", len(tables))
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractTableBoxes_NoOverlap(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{PageNumber: 0, X0: 50, X1: 550, Top: 10, Bottom: 30, Text: "far away"},
|
||||
}
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []pdf.DLARegion{
|
||||
{X0: 150, Y0: 1500, X1: 1500, Y1: 2300, Label: "table", Confidence: 0.95},
|
||||
},
|
||||
}
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 2000, 3000))
|
||||
tables := p.extractTableBoxesFromImage(context.Background(), nil, boxes, dummy, 0, 0)
|
||||
if len(tables) != 0 {
|
||||
t.Errorf("no overlap → 0 tables, got %d", len(tables))
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractTableBoxes_TSRError(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{PageNumber: 0, X0: 80, X1: 500, Top: 210, Bottom: 660, Text: "cell"},
|
||||
}
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []pdf.DLARegion{
|
||||
{X0: 250, Y0: 600, X1: 1500, Y1: 2000, Label: "table", Confidence: 0.95},
|
||||
},
|
||||
TSRCells: nil, // TSR returns nothing
|
||||
}
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 2000, 3000))
|
||||
tables := p.extractTableBoxesFromImage(context.Background(), nil, boxes, dummy, 0, 0)
|
||||
if len(tables) != 1 {
|
||||
t.Fatalf("TSR failure: expected 1 pdf.TableItem with image+positions, got %d", len(tables))
|
||||
}
|
||||
if tables[0].ImageB64 == "" {
|
||||
t.Error("should have image despite TSR failure")
|
||||
}
|
||||
if len(tables[0].Positions) == 0 {
|
||||
t.Error("should have positions despite TSR failure")
|
||||
}
|
||||
if len(tables[0].Rows) != 0 {
|
||||
t.Errorf("TSR failure → 0 rows, got %d", len(tables[0].Rows))
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractTableBoxes_DLAError(t *testing.T) {
|
||||
// DLA returns only non-table regions → 0 tables
|
||||
mock := &MockDocAnalyzer{Healthy: true, DLARegions: []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 100, Label: "text", Confidence: 0.9},
|
||||
}}
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 1000, 1000))
|
||||
tables := p.extractTableBoxesFromImage(context.Background(), nil, nil, dummy, 0, 0)
|
||||
if len(tables) != 0 {
|
||||
t.Errorf("non-table DLA → 0 tables, got %d", len(tables))
|
||||
}
|
||||
}
|
||||
|
||||
func TestParse_TableLinkedToSections(t *testing.T) {
|
||||
// Simulate enrichWithDeepDoc → extractTableAndReplace → boxesToSections:
|
||||
// table boxes are popped and replaced with one HTML box.
|
||||
boxes := []pdf.TextBox{
|
||||
{PageNumber: 0, X0: 50, X1: 200, Top: 50, Bottom: 80, Text: "heading"},
|
||||
{PageNumber: 0, X0: 50, X1: 500, Top: 250, Bottom: 400, Text: "table text", LayoutType: "table"},
|
||||
{PageNumber: 0, X0: 50, X1: 200, Top: 450, Bottom: 480, Text: "after"},
|
||||
}
|
||||
tableItem := pdf.TableItem{
|
||||
Cells: []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 200, Y1: 50, Label: "table row"},
|
||||
{X0: 0, Y0: 51, X1: 200, Y1: 100, Label: "table row"},
|
||||
},
|
||||
Positions: []pdf.Position{{PageNumbers: []int{0}, Left: 50, Right: 500, Top: 250, Bottom: 400}},
|
||||
Scale: 1.0,
|
||||
}
|
||||
|
||||
boxes = tbl.ExtractTableAndReplace(boxes, []pdf.TableItem{tableItem})
|
||||
sections := lyt.BoxesToSections(boxes, nil)
|
||||
|
||||
// 3 boxes (heading, table, after) → 3 sections (heading, HTML, after).
|
||||
if len(sections) != 3 {
|
||||
t.Errorf("expected 3 sections, got %d", len(sections))
|
||||
}
|
||||
tableFound := false
|
||||
for _, s := range sections {
|
||||
if s.LayoutType == "table" && strings.Contains(s.Text, "<table>") {
|
||||
tableFound = true
|
||||
}
|
||||
}
|
||||
if !tableFound {
|
||||
t.Errorf("expected at least one section with HTML table")
|
||||
for _, s := range sections {
|
||||
t.Logf(" section text=%q LayoutType=%q", s.Text[:min(40, len(s.Text))], s.LayoutType)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── cropImageRegion ────────────────────────────────────────────────────
|
||||
// ── extractTableBoxesFromImage: invalid DLA region ─────────────────────
|
||||
|
||||
func TestExtractTableBoxes_InvalidRegion(t *testing.T) {
|
||||
// DLA returns a table region with x1 < x0. The pipeline should skip
|
||||
// this table gracefully (Python raises ValueError from PIL.Image.crop).
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []pdf.DLARegion{
|
||||
{X0: 500, Y0: 100, X1: 100, Y1: 300, Label: "table", Confidence: 0.9},
|
||||
},
|
||||
}
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
dummy := image.NewRGBA(image.Rect(0, 0, 1000, 1000))
|
||||
tables := p.extractTableBoxesFromImage(context.Background(), nil, nil, dummy, 0, 0)
|
||||
if len(tables) != 0 {
|
||||
t.Errorf("invalid DLA region should be skipped, got %d tables", len(tables))
|
||||
}
|
||||
}
|
||||
|
||||
// ── DLA → figure end-to-end ───────────────────────────────────────────
|
||||
|
||||
func TestParse_CollectsFigures(t *testing.T) {
|
||||
// End-to-end: Parse() with mock DeepDoc that labels a box as "figure".
|
||||
// Verify p.Figures is populated.
|
||||
|
||||
eng := &mockEngine{pageCount: 1, chars: map[int][]pdf.TextChar{0: {{X0: 50, X1: 550, Top: 100, Bottom: 112, Text: "chart image"}}}}
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []pdf.DLARegion{
|
||||
{X0: 50, Y0: 200, X1: 2000, Y1: 1000, Label: "figure", Confidence: 0.85},
|
||||
},
|
||||
}
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
if len(result.Sections) == 0 {
|
||||
t.Fatal("expected at least 1 section")
|
||||
}
|
||||
if len(result.Figures()) != 1 {
|
||||
t.Fatalf("expected 1 figure, got %d", len(result.Figures()))
|
||||
}
|
||||
if result.Figures()[0].LayoutType != "figure" {
|
||||
t.Errorf("figure LayoutType = %q, want 'figure'", result.Figures()[0].LayoutType)
|
||||
}
|
||||
if result.Figures()[0].Text == "" {
|
||||
t.Error("figure Text should not be empty")
|
||||
}
|
||||
}
|
||||
|
||||
func TestParse_NoFigures(t *testing.T) {
|
||||
// Parse() with no DLA figure regions → p.Figures should be empty.
|
||||
|
||||
eng := &mockEngine{pageCount: 1, chars: map[int][]pdf.TextChar{0: {{X0: 50, X1: 550, Top: 100, Bottom: 112, Text: "just text"}}}}
|
||||
mock := &MockDocAnalyzer{
|
||||
DLARegions: []pdf.DLARegion{
|
||||
{X0: 150, Y0: 300, X1: 1500, Y1: 600, Label: "text", Confidence: 0.8},
|
||||
},
|
||||
}
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
if len(result.Figures()) != 0 {
|
||||
t.Fatalf("expected 0 figures, got %d", len(result.Figures()))
|
||||
}
|
||||
}
|
||||
|
||||
func TestParse_NoDeepDoc_NoFigures(t *testing.T) {
|
||||
// Parse() with mock DeepDoc → Figures should be empty (no DLA-detected figures).
|
||||
|
||||
eng := &mockEngine{pageCount: 1, chars: map[int][]pdf.TextChar{0: {{X0: 50, X1: 550, Top: 100, Bottom: 112, Text: "text"}}}}
|
||||
p := NewParser(pdf.DefaultParserConfig(), &MockDocAnalyzer{Healthy: true})
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
if len(result.Figures()) != 0 {
|
||||
t.Fatalf("expected 0 Figures (no DLA-detected figures), got %d", len(result.Figures()))
|
||||
}
|
||||
}
|
||||
|
||||
// ── Parse + ocrMergeChars (full-page detect) ──────────────────────────
|
||||
|
||||
func TestParse_UsesOCRDetectForEmbeddedChars(t *testing.T) {
|
||||
// When DeepDoc is available and the page has embedded chars,
|
||||
// Parse should use ocrMergeChars (detect → merge → recognize).
|
||||
eng := &mockEngine{
|
||||
pageCount: 1,
|
||||
chars: map[int][]pdf.TextChar{0: {
|
||||
{X0: 10, X1: 30, Top: 10, Bottom: 30, Text: "Hello", PageNumber: 0},
|
||||
}},
|
||||
}
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRBoxes: []pdf.OCRBox{
|
||||
{X0: 5, Y0: 5, X1: 50, Y1: 5, X2: 50, Y2: 50, X3: 5, Y3: 50},
|
||||
},
|
||||
}
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
if len(result.Sections) == 0 {
|
||||
t.Fatal("expected at least 1 section")
|
||||
}
|
||||
// The box should come from OCR detect, not charsToBoxes.
|
||||
// Verifying that ocrMergeChars was used (sections exist).
|
||||
if result.Metrics.BoxesInitial == 0 {
|
||||
t.Error("expected BoxesInitial > 0 (OCR detect path)")
|
||||
}
|
||||
}
|
||||
|
||||
func TestParse_FallsBackToCharsToBoxes_NoDeepDoc(t *testing.T) {
|
||||
// Without DeepDoc, Parse should use charsToBoxes (unchanged behavior).
|
||||
eng := &mockEngine{
|
||||
pageCount: 1,
|
||||
chars: map[int][]pdf.TextChar{0: {
|
||||
{X0: 10, X1: 30, Top: 10, Bottom: 30, Text: "Hello", PageNumber: 0},
|
||||
}},
|
||||
}
|
||||
p := NewParser(pdf.DefaultParserConfig(), &MockDocAnalyzer{Healthy: true})
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
if len(result.Sections) == 0 {
|
||||
t.Fatal("expected at least 1 section (charsToBoxes)")
|
||||
}
|
||||
}
|
||||
|
||||
func TestParse_FallsBackToCharsToBoxes_EmptyOCRBoxes(t *testing.T) {
|
||||
// OCRDetect returns no boxes → falls through to charsToBoxes.
|
||||
eng := &mockEngine{
|
||||
pageCount: 1,
|
||||
chars: map[int][]pdf.TextChar{0: {
|
||||
{X0: 10, X1: 30, Top: 10, Bottom: 30, Text: "Hello", PageNumber: 0},
|
||||
}},
|
||||
}
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRBoxes: []pdf.OCRBox{}, // empty detect
|
||||
}
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
if len(result.Sections) == 0 {
|
||||
t.Fatal("expected at least 1 section (charsToBoxes fallback)")
|
||||
}
|
||||
}
|
||||
|
||||
// ── Error path coverage ────────────────────────────────────────────────
|
||||
|
||||
func TestMockDocAnalyzer_DLAError_DoesNotCrash(t *testing.T) {
|
||||
p := NewParser(pdf.DefaultParserConfig(), &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLAErr: fmt.Errorf("DLA service unavailable"),
|
||||
})
|
||||
eng := &mockEngine{pageCount: 1}
|
||||
img := image.NewRGBA(image.Rect(0, 0, 100, 100))
|
||||
pageImages := map[int]image.Image{0: img}
|
||||
boxes := []pdf.TextBox{
|
||||
{PageNumber: 0, X0: 0, X1: 100, Top: 0, Bottom: 50, Text: "text"},
|
||||
}
|
||||
// enrichWithDeepDoc should return nil (not panic) on DLA error.
|
||||
tables := p.enrichWithDeepDoc(context.Background(), nil, eng, boxes, pageImages)
|
||||
if len(tables) != 0 {
|
||||
t.Errorf("DLA error should produce 0 tables, got %d", len(tables))
|
||||
}
|
||||
}
|
||||
|
||||
func TestMockDocAnalyzer_TSRError_DoesNotCrash(t *testing.T) {
|
||||
// TSR error: DLA succeeds, TSR fails. The table region is detected
|
||||
// but no cells are returned — the table is skipped gracefully.
|
||||
p := NewParser(pdf.DefaultParserConfig(), &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 400, Y1: 400, Label: "table", Confidence: 0.95},
|
||||
},
|
||||
TSRErr: fmt.Errorf("TSR model timeout"),
|
||||
})
|
||||
eng := &mockEngine{pageCount: 1}
|
||||
img := image.NewRGBA(image.Rect(0, 0, 100, 100))
|
||||
pageImages := map[int]image.Image{0: img}
|
||||
boxes := []pdf.TextBox{
|
||||
{PageNumber: 0, X0: 10, X1: 90, Top: 10, Bottom: 90, Text: "in table region"},
|
||||
}
|
||||
tables := p.enrichWithDeepDoc(context.Background(), nil, eng, boxes, pageImages)
|
||||
// DLA detects the table region → 1 pdf.TableItem is created. TSR failure
|
||||
// means it has no cells, but the pipeline must not panic.
|
||||
if len(tables) != 1 {
|
||||
t.Errorf("TSR error: expected 1 table (DLA region found), got %d", len(tables))
|
||||
}
|
||||
if len(tables[0].Cells) != 0 {
|
||||
t.Errorf("TSR error: Cells should be empty, got %d", len(tables[0].Cells))
|
||||
}
|
||||
}
|
||||
|
||||
func TestMockDocAnalyzer_OCRDetectError_DoesNotCrash(t *testing.T) {
|
||||
// OCRDetect failure path: extractPages uses ocrDetectAndRecognize which
|
||||
// calls doc.OCRDetect. When it fails, the page is skipped gracefully.
|
||||
mock := &MockDocAnalyzer{Healthy: true, OCRDetectErr: fmt.Errorf("OCR model OOM")}
|
||||
eng := &mockEngine{
|
||||
pageCount: 1,
|
||||
chars: map[int][]pdf.TextChar{}, // empty → triggers OCR path
|
||||
}
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
_, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse returned error: %v", err)
|
||||
}
|
||||
// Parse should succeed — the page with OCRDetect error is just skipped.
|
||||
}
|
||||
|
||||
// TestTSRLabels verifies Go inf.DefaultTSRLabels() matches Python's table_structure_recognizer.py labels.
|
||||
// Order must be exact — the ONNX model returns class IDs that index into this array.
|
||||
func TestTSRLabels(t *testing.T) {
|
||||
want := []string{
|
||||
"table", "table column", "table row",
|
||||
"table column header", "table projected row header",
|
||||
"table spanning cell",
|
||||
}
|
||||
if len(inf.DefaultTSRLabels()) != len(want) {
|
||||
t.Fatalf("inf.DefaultTSRLabels() length %d, want %d", len(inf.DefaultTSRLabels()), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if inf.DefaultTSRLabels()[i] != want[i] {
|
||||
t.Errorf("inf.DefaultTSRLabels()[%d] = %q, want %q", i, inf.DefaultTSRLabels()[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -2,190 +2,17 @@ package parser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"image"
|
||||
"log/slog"
|
||||
"math"
|
||||
lyt "ragflow/internal/deepdoc/parser/pdf/layout"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
util "ragflow/internal/deepdoc/parser/pdf/util"
|
||||
"sort"
|
||||
"strings"
|
||||
"unicode"
|
||||
)
|
||||
|
||||
// isGarbledPage returns true if a page is garbled by PUA ratio, font encoding,
|
||||
// pdf_oxide unmapped glyphs, or scan noise (no real words).
|
||||
func isGarbledPage(chars []TextChar) bool {
|
||||
if len(chars) < 20 {
|
||||
return false
|
||||
}
|
||||
// Build full-page text for detection (all O(n) single pass).
|
||||
var fullText strings.Builder
|
||||
for _, c := range chars {
|
||||
fullText.WriteString(c.Text)
|
||||
}
|
||||
text := fullText.String()
|
||||
if IsGarbledText(text, 0.3) {
|
||||
return true
|
||||
}
|
||||
if pdfOxideUnmappedGarbled(text) && isScanNoise(text) {
|
||||
return true
|
||||
}
|
||||
if IsGarbledByFontEncoding(chars, 20) {
|
||||
return true
|
||||
}
|
||||
if isScanNoise(text) {
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// isScanNoise detects scanned pages where pdf_oxide extracts noise glyphs
|
||||
// instead of real text. Real text in any language contains word-like runs
|
||||
// of consecutive letters (L category). Scan noise consists of random ASCII
|
||||
// symbols with at most 2-letter fragments.
|
||||
//
|
||||
// Three indicators of real (non-noise) text, any one is sufficient:
|
||||
// - ≥4 consecutive lowercase Latin letters (e.g. "the", "and")
|
||||
// - ≥2 consecutive CJK characters (Han, Hiragana, Katakana, Hangul)
|
||||
// - ≥4 consecutive non-ASCII letters (Arabic, Thai, Cyrillic, etc.)
|
||||
//
|
||||
// Pure-uppercase fragments like "RASB" are common in pdf_oxide noise but
|
||||
// never appear as standalone words in real text without lowercase context.
|
||||
func isScanNoise(text string) bool {
|
||||
nonSpace := 0
|
||||
digitCount := 0
|
||||
lowerRun := 0
|
||||
maxLowerRun := 0
|
||||
cjkRun := 0
|
||||
maxCJKRun := 0
|
||||
nonASCIILetterRun := 0
|
||||
maxNonASCIILetterRun := 0
|
||||
|
||||
for _, r := range text {
|
||||
if r == ' ' || r == '\t' || r == '\n' || r == '\r' {
|
||||
lowerRun = 0
|
||||
cjkRun = 0
|
||||
nonASCIILetterRun = 0
|
||||
continue
|
||||
}
|
||||
nonSpace++
|
||||
|
||||
// Digit density: real content (tables, dates) has digits;
|
||||
// pdf_oxide noise (unmapped glyphs) never produces digits.
|
||||
if r >= '0' && r <= '9' {
|
||||
digitCount++
|
||||
}
|
||||
|
||||
// Lowercase Latin (Ll)
|
||||
if unicode.Is(unicode.Ll, r) {
|
||||
lowerRun++
|
||||
if lowerRun > maxLowerRun {
|
||||
maxLowerRun = lowerRun
|
||||
}
|
||||
} else {
|
||||
lowerRun = 0
|
||||
}
|
||||
|
||||
// CJK: Han, Hiragana, Katakana, Hangul Syllables & Jamo
|
||||
if isCJK(r) {
|
||||
cjkRun++
|
||||
if cjkRun > maxCJKRun {
|
||||
maxCJKRun = cjkRun
|
||||
}
|
||||
} else {
|
||||
cjkRun = 0
|
||||
}
|
||||
|
||||
// Non-ASCII letter (Arabic U+0600–U+06FF, Thai U+0E00–U+0E7F,
|
||||
// Cyrillic U+0400–U+04FF, etc.). Excludes ASCII so uppercase
|
||||
// Latin fragments like "RASB" don't count.
|
||||
if unicode.IsLetter(r) && r > unicode.MaxASCII {
|
||||
nonASCIILetterRun++
|
||||
if nonASCIILetterRun > maxNonASCIILetterRun {
|
||||
maxNonASCIILetterRun = nonASCIILetterRun
|
||||
}
|
||||
} else {
|
||||
nonASCIILetterRun = 0
|
||||
}
|
||||
}
|
||||
|
||||
// Need enough characters to make a meaningful decision.
|
||||
if nonSpace < 30 {
|
||||
return false
|
||||
}
|
||||
|
||||
// Digit density: pdf_oxide never substitutes digits for unmapped
|
||||
// glyphs. Real content (tables, dates, page numbers) has ≥10%
|
||||
// digits; noise consists of random ASCII punctuation.
|
||||
if float64(digitCount)/float64(nonSpace) >= 0.10 {
|
||||
return false
|
||||
}
|
||||
|
||||
// Real text in any script — any one indicator is sufficient.
|
||||
isNoise := maxLowerRun < 4 && maxCJKRun < 2 && maxNonASCIILetterRun < 4
|
||||
|
||||
return isNoise
|
||||
}
|
||||
|
||||
// isCJK reports whether r is a CJK character: Han ideograph, Hiragana,
|
||||
// Katakana, Hangul syllable, or Hangul Jamo.
|
||||
func isCJK(r rune) bool {
|
||||
return unicode.Is(unicode.Han, r) ||
|
||||
unicode.Is(unicode.Hiragana, r) ||
|
||||
unicode.Is(unicode.Katakana, r) ||
|
||||
unicode.Is(unicode.Hangul, r)
|
||||
}
|
||||
|
||||
// pdfOxideUnmappedGarbled detects pdf_oxide's '#' placeholder glyphs.
|
||||
// pdf_oxide uses '#' (U+0023) for every glyph it cannot map; consecutive
|
||||
// unmapped glyphs form "##", "###", "####" sequences. Three or more
|
||||
// consecutive '#' is virtually impossible in normal text.
|
||||
//
|
||||
// Two conditions (either is sufficient):
|
||||
// - ≥ 2 occurrences of "###" (3+ consecutive #)
|
||||
// - # density ≥ 5% of non-space characters
|
||||
func pdfOxideUnmappedGarbled(text string) bool {
|
||||
hashCount := 0
|
||||
total := 0
|
||||
consecutive := 0
|
||||
tripleClusters := 0
|
||||
|
||||
for _, r := range text {
|
||||
if r == ' ' || r == '\t' || r == '\n' || r == '\r' {
|
||||
continue
|
||||
}
|
||||
total++
|
||||
if r == '#' {
|
||||
hashCount++
|
||||
consecutive++
|
||||
if consecutive == 3 {
|
||||
tripleClusters++
|
||||
}
|
||||
} else {
|
||||
consecutive = 0
|
||||
}
|
||||
}
|
||||
|
||||
if total == 0 {
|
||||
return false
|
||||
}
|
||||
|
||||
density := float64(hashCount) / float64(total)
|
||||
|
||||
if tripleClusters >= 1 {
|
||||
return true
|
||||
}
|
||||
// Density check only meaningful with enough chars (matches isGarbledPage's
|
||||
// min 20 char guard). In production the sample is 200 chars.
|
||||
if total >= 40 && density >= 0.03 {
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// ocrDetectAndRecognize runs OCR detection + recognition and returns
|
||||
// recognized TextBox results. logLabel distinguishes callers in log output
|
||||
// ("scan page", "garbled page").
|
||||
func ocrDetectAndRecognize(ctx context.Context, pageImg image.Image, doc DocAnalyzer, pageNum int, logLabel string) []TextBox {
|
||||
func ocrDetectAndRecognize(ctx context.Context, pageImg image.Image, doc pdf.DocAnalyzer, pageNum int, logLabel string) []pdf.TextBox {
|
||||
boxes, err := doc.OCRDetect(ctx, pageImg)
|
||||
if err != nil || len(boxes) == 0 {
|
||||
if err != nil {
|
||||
@@ -194,7 +21,7 @@ func ocrDetectAndRecognize(ctx context.Context, pageImg image.Image, doc DocAnal
|
||||
return nil
|
||||
}
|
||||
|
||||
var result []TextBox
|
||||
var result []pdf.TextBox
|
||||
for _, box := range boxes {
|
||||
x0 := int(math.Min(box.X0, math.Min(box.X1, math.Min(box.X2, box.X3))))
|
||||
y0 := int(math.Min(box.Y0, math.Min(box.Y1, math.Min(box.Y2, box.Y3))))
|
||||
@@ -203,7 +30,7 @@ func ocrDetectAndRecognize(ctx context.Context, pageImg image.Image, doc DocAnal
|
||||
if x0 >= x1 || y0 >= y1 {
|
||||
continue
|
||||
}
|
||||
cropped := fastCrop(pageImg, x0, y0, x1, y1)
|
||||
cropped := util.FastCrop(pageImg, x0, y0, x1, y1)
|
||||
texts, recErr := doc.OCRRecognize(ctx, cropped)
|
||||
if recErr != nil {
|
||||
slog.Warn(logLabel+" OCR recognize failed", "page", pageNum, "err", recErr)
|
||||
@@ -211,7 +38,7 @@ func ocrDetectAndRecognize(ctx context.Context, pageImg image.Image, doc DocAnal
|
||||
}
|
||||
for _, t := range texts {
|
||||
if strings.TrimSpace(t.Text) != "" {
|
||||
result = append(result, TextBox{
|
||||
result = append(result, pdf.TextBox{
|
||||
X0: float64(x0), X1: float64(x1),
|
||||
Top: float64(y0), Bottom: float64(y1),
|
||||
Text: t.Text,
|
||||
@@ -227,7 +54,7 @@ func ocrDetectAndRecognize(ctx context.Context, pageImg image.Image, doc DocAnal
|
||||
// merges the chars into detect regions, and OCRs any regions without chars.
|
||||
// Matches Python's __ocr: detect → match chars to boxes → use char text
|
||||
// for boxes with embedded chars → OCR recognize only empty/garbled boxes.
|
||||
func ocrMergeChars(ctx context.Context, pageImg image.Image, chars []TextChar, doc DocAnalyzer, pageNum int) []TextBox {
|
||||
func ocrMergeChars(ctx context.Context, pageImg image.Image, chars []pdf.TextChar, doc pdf.DocAnalyzer, pageNum int) []pdf.TextBox {
|
||||
detectBoxes, err := doc.OCRDetect(ctx, pageImg)
|
||||
if err != nil || len(detectBoxes) == 0 {
|
||||
return nil
|
||||
@@ -236,14 +63,14 @@ func ocrMergeChars(ctx context.Context, pageImg image.Image, chars []TextChar, d
|
||||
|
||||
// Detect boxes are in pixel space (216 DPI). Scale to PDF space (72 DPI)
|
||||
// so coordinates match embedded chars.
|
||||
scale := dlaScale // 3.0
|
||||
scale := pdf.DlaScale // 3.0
|
||||
imgBounds := pageImg.Bounds()
|
||||
imgW := float64(imgBounds.Dx()) / scale
|
||||
imgH := float64(imgBounds.Dy()) / scale
|
||||
|
||||
// Step 1: match embedded chars to detect boxes (Python __ocr char matching).
|
||||
type detectBox struct {
|
||||
box TextBox
|
||||
box pdf.TextBox
|
||||
x0, y0, x1, y1 float64 // PDF-space bounds
|
||||
}
|
||||
boxes := make([]detectBox, 0, len(detectBoxes))
|
||||
@@ -267,7 +94,7 @@ func ocrMergeChars(ctx context.Context, pageImg image.Image, chars []TextChar, d
|
||||
if x0 >= x1 || y0 >= y1 {
|
||||
continue
|
||||
}
|
||||
boxes = append(boxes, detectBox{box: TextBox{
|
||||
boxes = append(boxes, detectBox{box: pdf.TextBox{
|
||||
X0: x0, X1: x1, Top: y0, Bottom: y1, PageNumber: pageNum,
|
||||
}, x0: x0, y0: y0, x1: x1, y1: y1})
|
||||
}
|
||||
@@ -291,7 +118,7 @@ func ocrMergeChars(ctx context.Context, pageImg image.Image, chars []TextChar, d
|
||||
|
||||
// Step 2: match each char to the best overlapping detect box
|
||||
// (char perspective), matching Python's find_overlapped.
|
||||
boxChars := make([][]TextChar, len(boxes))
|
||||
boxChars := make([][]pdf.TextChar, len(boxes))
|
||||
for _, c := range chars {
|
||||
bestIdx := -1
|
||||
bestOverlap := 1e-6 // Python: thr=1e-6
|
||||
@@ -319,7 +146,7 @@ func ocrMergeChars(ctx context.Context, pageImg image.Image, chars []TextChar, d
|
||||
}
|
||||
|
||||
// Step 3: assemble text for each box.
|
||||
var result []TextBox
|
||||
var result []pdf.TextBox
|
||||
var needOCR []int
|
||||
for i := range boxes {
|
||||
tb := boxes[i].box
|
||||
@@ -329,11 +156,11 @@ func ocrMergeChars(ctx context.Context, pageImg image.Image, chars []TextChar, d
|
||||
// Sort chars by reading order, matching Python's sort_Y_firstly.
|
||||
// Fuzzy Y-group: chars within median char height are "same line",
|
||||
// sorted by X; different lines sorted by Y.
|
||||
sortCharsYFirstly(boxChars[i], medianCharHeight(boxChars[i]))
|
||||
sortCharsYFirstly(boxChars[i], util.MedianCharHeight(boxChars[i]))
|
||||
// Use lineToTextBox for correct space insertion + garbled detection.
|
||||
// lineToTextBox inserts ASCII word spaces at visible gaps —
|
||||
// matching Python's __img_ocr + __ocr char logic.
|
||||
lineBox := lineToTextBox(boxChars[i])
|
||||
lineBox := lyt.LineToTextBox(boxChars[i])
|
||||
tb.Text = lineBox.Text
|
||||
|
||||
// Strategy 1: If majority of chars are garbled (PUA), clear text → OCR.
|
||||
@@ -341,7 +168,7 @@ func ocrMergeChars(ctx context.Context, pageImg image.Image, chars []TextChar, d
|
||||
for _, c := range boxChars[i] {
|
||||
for _, r := range c.Text {
|
||||
totalCnt++
|
||||
if IsGarbledChar(string(r)) {
|
||||
if util.IsGarbledChar(string(r)) {
|
||||
garbledCnt++
|
||||
}
|
||||
}
|
||||
@@ -350,7 +177,7 @@ func ocrMergeChars(ctx context.Context, pageImg image.Image, chars []TextChar, d
|
||||
tb.Text = ""
|
||||
}
|
||||
// Strategy 2: font-encoding garbled (subset fonts, min 5 chars).
|
||||
if tb.Text != "" && IsGarbledByFontEncoding(boxChars[i], 5) {
|
||||
if tb.Text != "" && util.IsGarbledByFontEncoding(boxChars[i], 5) {
|
||||
tb.Text = ""
|
||||
}
|
||||
}
|
||||
@@ -365,7 +192,7 @@ func ocrMergeChars(ctx context.Context, pageImg image.Image, chars []TextChar, d
|
||||
if len(needOCR) > 0 {
|
||||
cropped := make([]image.Image, len(needOCR))
|
||||
for j, idx := range needOCR {
|
||||
cropped[j] = fastCrop(pageImg,
|
||||
cropped[j] = util.FastCrop(pageImg,
|
||||
int(boxes[idx].x0*scale), int(boxes[idx].y0*scale),
|
||||
int(boxes[idx].x1*scale), int(boxes[idx].y1*scale))
|
||||
}
|
||||
@@ -396,26 +223,12 @@ func ocrMergeChars(ctx context.Context, pageImg image.Image, chars []TextChar, d
|
||||
return result
|
||||
}
|
||||
|
||||
// medianCharHeight returns the median height of chars, or 0 if empty.
|
||||
// Used as the fuzzy-sort threshold matching Python's np.mean([c["height"]]).
|
||||
func medianCharHeight(chars []TextChar) float64 {
|
||||
if len(chars) == 0 {
|
||||
return 0
|
||||
}
|
||||
heights := make([]float64, len(chars))
|
||||
for i, c := range chars {
|
||||
heights[i] = c.Bottom - c.Top
|
||||
}
|
||||
sort.Float64s(heights)
|
||||
return heights[len(heights)/2]
|
||||
}
|
||||
|
||||
// sortYFirstly sorts chars by Y (fuzzy group by threshold), then by X.
|
||||
// Matching Python Recognizer.sort_Y_firstly in recognizer.py:26-33:
|
||||
//
|
||||
// If two chars have Y diff < threshold → same line → sort by X.
|
||||
// Otherwise → sort by Y.
|
||||
func sortCharsYFirstly(chars []TextChar, threshold float64) {
|
||||
func sortCharsYFirstly(chars []pdf.TextChar, threshold float64) {
|
||||
sort.Slice(chars, func(a, b int) bool {
|
||||
diff := chars[a].Top - chars[b].Top
|
||||
if math.Abs(diff) < threshold {
|
||||
@@ -428,7 +241,7 @@ func sortCharsYFirstly(chars []TextChar, threshold float64) {
|
||||
// charBoxOverlapRatio computes the overlap ratio between a char and a box,
|
||||
// from the char's perspective. Returns overlap_area / char_area.
|
||||
// Matching Python's Recognizer.overlapped_area(char, box, ratio=True).
|
||||
func charBoxOverlapRatio(c TextChar, x0, x1, y0, y1 float64) float64 {
|
||||
func charBoxOverlapRatio(c pdf.TextChar, x0, x1, y0, y1 float64) float64 {
|
||||
cw := c.X1 - c.X0
|
||||
ch := c.Bottom - c.Top
|
||||
if cw <= 0 {
|
||||
@@ -441,12 +254,12 @@ func charBoxOverlapRatio(c TextChar, x0, x1, y0, y1 float64) float64 {
|
||||
if charArea <= 0 {
|
||||
return 0
|
||||
}
|
||||
inter := rectOverlapInter(c.X0, c.Top, c.X1, c.Bottom, x0, y0, x1, y1)
|
||||
inter := util.RectOverlapInter(c.X0, c.Top, c.X1, c.Bottom, x0, y0, x1, y1)
|
||||
return inter / charArea
|
||||
}
|
||||
|
||||
// ocrTableCells fills empty TSR cells via OCR recognition.
|
||||
func ocrTableCells(ctx context.Context, cells []TSRCell, tableImg image.Image, doc DocAnalyzer) {
|
||||
func ocrTableCells(ctx context.Context, cells []pdf.TSRCell, tableImg image.Image, doc pdf.DocAnalyzer) {
|
||||
if doc == nil || tableImg == nil || len(cells) == 0 {
|
||||
return
|
||||
}
|
||||
@@ -461,7 +274,7 @@ func ocrTableCells(ctx context.Context, cells []TSRCell, tableImg image.Image, d
|
||||
if x0 >= x1 || y0 >= y1 {
|
||||
continue
|
||||
}
|
||||
cropped := fastCrop(tableImg, x0, y0, x1, y1)
|
||||
cropped := util.FastCrop(tableImg, x0, y0, x1, y1)
|
||||
texts, err := doc.OCRRecognize(ctx, cropped)
|
||||
if err != nil {
|
||||
slog.Warn("table cell OCR failed", "err", err)
|
||||
@@ -476,108 +289,3 @@ func ocrTableCells(ctx context.Context, cells []TSRCell, tableImg image.Image, d
|
||||
cells[i].Text = strings.TrimSpace(strings.Join(parts, " "))
|
||||
}
|
||||
}
|
||||
|
||||
// evaluateTableOrientation tests 4 rotation angles (0/90/180/270) and picks
|
||||
// the best orientation based on OCR confidence scores.
|
||||
//
|
||||
// Returns bestAngle (0/90/180/270), the rotated image, and per-angle scores.
|
||||
// Scores map[angle]{avgConfidence, totalRegions, combinedScore}.
|
||||
//
|
||||
// Absolute threshold: non-0° wins only if its combined score exceeds 0° by
|
||||
// more than 0.2 AND the 0° score is below 0.8.
|
||||
//
|
||||
// Python: pdf_parser.py:314 _evaluate_table_orientation()
|
||||
func evaluateTableOrientation(ctx context.Context, tableImg image.Image, doc DocAnalyzer) (bestAngle int, bestImg image.Image, scores map[int]float64) {
|
||||
rotations := []struct {
|
||||
angle int
|
||||
name string
|
||||
}{
|
||||
{0, "original"},
|
||||
{90, "rotate_90"},
|
||||
{180, "rotate_180"},
|
||||
{270, "rotate_270"},
|
||||
}
|
||||
|
||||
scores = make(map[int]float64, 4)
|
||||
bestScore := float64(-1)
|
||||
bestAngle = 0
|
||||
bestImg = tableImg
|
||||
|
||||
for _, rot := range rotations {
|
||||
rotated := tableImg
|
||||
if rot.angle != 0 {
|
||||
rotated = rotateImageCW(tableImg, rot.angle)
|
||||
if rotated == nil {
|
||||
slog.Warn("table rotate failed", "angle", rot.angle)
|
||||
continue
|
||||
}
|
||||
}
|
||||
|
||||
detectBoxes, err := doc.OCRDetect(ctx, rotated)
|
||||
if err != nil || len(detectBoxes) == 0 {
|
||||
scores[rot.angle] = 0
|
||||
continue
|
||||
}
|
||||
|
||||
// Score by detect-region count (primary) + area (tiebreaker).
|
||||
// Per-region OCRRecognize calls are NOT needed to judge table
|
||||
// orientation — the count of detect regions is a reliable proxy
|
||||
// (a well-oriented table has more/fuller text regions).
|
||||
// Skipping recognize cuts ~N HTTP calls per angle.
|
||||
imageArea := float64(rotated.Bounds().Dx() * rotated.Bounds().Dy())
|
||||
totalRegions := 0
|
||||
var totalArea float64
|
||||
for _, box := range detectBoxes {
|
||||
x0 := math.Min(box.X0, math.Min(box.X1, math.Min(box.X2, box.X3)))
|
||||
y0 := math.Min(box.Y0, math.Min(box.Y1, math.Min(box.Y2, box.Y3)))
|
||||
x1 := math.Max(box.X0, math.Max(box.X1, math.Max(box.X2, box.X3)))
|
||||
y1 := math.Max(box.Y0, math.Max(box.Y1, math.Max(box.Y2, box.Y3)))
|
||||
if x0 >= x1 || y0 >= y1 {
|
||||
continue
|
||||
}
|
||||
totalRegions++
|
||||
totalArea += (x1 - x0) * (y1 - y0)
|
||||
}
|
||||
if totalRegions == 0 {
|
||||
scores[rot.angle] = 0
|
||||
continue
|
||||
}
|
||||
areaRatio := totalArea / imageArea
|
||||
// Region count is the primary signal. Area coverage provides a
|
||||
// small bonus (up to +6%) so that when region counts are tied the
|
||||
// angle with fuller text boxes wins.
|
||||
combined := float64(totalRegions) * (1 + 0.06*areaRatio)
|
||||
scores[rot.angle] = combined
|
||||
|
||||
slog.Debug("table orientation",
|
||||
"angle", rot.angle,
|
||||
"regions", totalRegions,
|
||||
"area_ratio", fmt.Sprintf("%.4f", areaRatio),
|
||||
"combined", fmt.Sprintf("%.2f", combined))
|
||||
|
||||
if combined > bestScore {
|
||||
bestScore = combined
|
||||
bestAngle = rot.angle
|
||||
bestImg = rotated
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
// Absolute threshold: only accept non-0° if region count is clearly
|
||||
// higher (≥1.4×) AND 0° has few regions (< 6).
|
||||
// Prevents false rotation when the table is roughly upright.
|
||||
score0 := scores[0]
|
||||
if bestAngle != 0 && score0 > 0 {
|
||||
if !(bestScore > score0*1.4 && score0 < 6.0) {
|
||||
bestAngle = 0
|
||||
bestImg = tableImg
|
||||
bestScore = score0
|
||||
}
|
||||
}
|
||||
|
||||
slog.Debug("best table orientation",
|
||||
"angle", bestAngle,
|
||||
"score", fmt.Sprintf("%.4f", bestScore))
|
||||
|
||||
return bestAngle, bestImg, scores
|
||||
}
|
||||
|
||||
@@ -2,30 +2,25 @@ package parser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"image"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
util "ragflow/internal/deepdoc/parser/pdf/util"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// testPageImg creates a small test image for ocrMergeChars tests.
|
||||
// 90×120 px at 216 DPI → 30×40 pt in PDF space after /3.0 scaling.
|
||||
func testPageImg() image.Image {
|
||||
return image.NewRGBA(image.Rect(0, 0, 90, 120))
|
||||
}
|
||||
|
||||
// TestOCRMergeChars_FullCoverage: embedded chars fill the detect box.
|
||||
func TestOCRMergeChars_FullCoverage(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRBoxes: []OCRBox{
|
||||
OCRBoxes: []pdf.OCRBox{
|
||||
{X0: 0, Y0: 0, X1: 90, Y1: 0, X2: 90, Y2: 120, X3: 0, Y3: 120},
|
||||
},
|
||||
OCRTexts: []OCRText{
|
||||
OCRTexts: []pdf.OCRText{
|
||||
{Text: "OCR text", Confidence: 0.9},
|
||||
},
|
||||
}
|
||||
|
||||
// Both chars overlap the box (height diff < 0.7) → char text used.
|
||||
chars := []TextChar{
|
||||
chars := []pdf.TextChar{
|
||||
{X0: 2, X1: 10, Top: 2, Bottom: 35, Text: "Hello"},
|
||||
{X0: 12, X1: 28, Top: 2, Bottom: 35, Text: "World"},
|
||||
}
|
||||
@@ -44,17 +39,17 @@ func TestOCRMergeChars_FullCoverage(t *testing.T) {
|
||||
func TestOCRMergeChars_PartialCoverage(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRBoxes: []OCRBox{
|
||||
OCRBoxes: []pdf.OCRBox{
|
||||
{X0: 0, Y0: 0, X1: 45, Y1: 0, X2: 45, Y2: 60, X3: 0, Y3: 60},
|
||||
{X0: 45, Y0: 0, X1: 90, Y1: 0, X2: 90, Y2: 60, X3: 45, Y3: 60},
|
||||
},
|
||||
OCRTexts: []OCRText{
|
||||
OCRTexts: []pdf.OCRText{
|
||||
{Text: "OCR-filled", Confidence: 0.9},
|
||||
},
|
||||
}
|
||||
|
||||
// Char "A" overlaps box A → char text. Box B has no chars → OCR.
|
||||
chars := []TextChar{
|
||||
chars := []pdf.TextChar{
|
||||
{X0: 2, X1: 12, Top: 2, Bottom: 15, Text: "A"},
|
||||
}
|
||||
|
||||
@@ -79,7 +74,7 @@ func TestOCRMergeChars_NoDetectBoxes(t *testing.T) {
|
||||
OCRBoxes: nil,
|
||||
}
|
||||
|
||||
chars := []TextChar{
|
||||
chars := []pdf.TextChar{
|
||||
{X0: 2, X1: 10, Top: 2, Bottom: 8, Text: "Hello"},
|
||||
}
|
||||
|
||||
@@ -89,7 +84,7 @@ func TestOCRMergeChars_NoDetectBoxes(t *testing.T) {
|
||||
}
|
||||
|
||||
// Also test empty OCRBoxes
|
||||
mock.OCRBoxes = []OCRBox{}
|
||||
mock.OCRBoxes = []pdf.OCRBox{}
|
||||
boxes = ocrMergeChars(context.Background(), testPageImg(), chars, mock, 0)
|
||||
if boxes != nil {
|
||||
t.Errorf("expected nil for empty detect boxes, got %d boxes", len(boxes))
|
||||
@@ -100,16 +95,16 @@ func TestOCRMergeChars_NoDetectBoxes(t *testing.T) {
|
||||
func TestOCRMergeChars_GarbledChars(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRBoxes: []OCRBox{
|
||||
OCRBoxes: []pdf.OCRBox{
|
||||
{X0: 0, Y0: 0, X1: 90, Y1: 0, X2: 90, Y2: 120, X3: 0, Y3: 120},
|
||||
},
|
||||
OCRTexts: []OCRText{
|
||||
OCRTexts: []pdf.OCRText{
|
||||
{Text: "OCR-result", Confidence: 0.95},
|
||||
},
|
||||
}
|
||||
|
||||
// Char height ~33, box height 40. Diff = 0.175 < 0.7 → not filtered.
|
||||
chars := []TextChar{
|
||||
chars := []pdf.TextChar{
|
||||
{X0: 2, X1: 10, Top: 2, Bottom: 35, Text: string(rune(0xF0123))}, // PUA
|
||||
{X0: 12, X1: 20, Top: 2, Bottom: 35, Text: string(rune(0xF0456))}, // PUA
|
||||
{X0: 22, X1: 28, Top: 2, Bottom: 35, Text: "a"}, // normal
|
||||
@@ -130,16 +125,16 @@ func TestOCRMergeChars_HeightGate(t *testing.T) {
|
||||
// Box height in PDF space: 120/3.0 = 40
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRBoxes: []OCRBox{
|
||||
OCRBoxes: []pdf.OCRBox{
|
||||
{X0: 0, Y0: 0, X1: 90, Y1: 0, X2: 90, Y2: 120, X3: 0, Y3: 120},
|
||||
},
|
||||
OCRTexts: []OCRText{
|
||||
OCRTexts: []pdf.OCRText{
|
||||
{Text: "height-gated-OCR", Confidence: 0.8},
|
||||
},
|
||||
}
|
||||
|
||||
// Char height = 1. Box height = 40. Diff = |1-40|/max(1,40) = 39/40 = 0.975 >= 0.7 → filtered.
|
||||
chars := []TextChar{
|
||||
chars := []pdf.TextChar{
|
||||
{X0: 2, X1: 10, Top: 2, Bottom: 3, Text: "tiny"},
|
||||
}
|
||||
|
||||
@@ -159,16 +154,16 @@ func TestOCRMergeChars_HeightGate(t *testing.T) {
|
||||
func TestOCRMergeChars_FontEncodingGarbled(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRBoxes: []OCRBox{
|
||||
OCRBoxes: []pdf.OCRBox{
|
||||
{X0: 15, Y0: 15, X1: 150, Y1: 15, X2: 150, Y2: 150, X3: 15, Y3: 150},
|
||||
},
|
||||
OCRTexts: []OCRText{{Text: "OCR fallback", Confidence: 0.9}},
|
||||
OCRTexts: []pdf.OCRText{{Text: "OCR fallback", Confidence: 0.9}},
|
||||
}
|
||||
// 5+ subset-font chars (font names matching `^[A-Z0-9]{2,6}\+`)
|
||||
// trigger font-encoding garbled detection → text cleared → OCR used.
|
||||
chars := make([]TextChar, 5)
|
||||
chars := make([]pdf.TextChar, 5)
|
||||
for i := range chars {
|
||||
chars[i] = TextChar{
|
||||
chars[i] = pdf.TextChar{
|
||||
X0: 10, X1: 30, Top: float64(10 + i*5), Bottom: float64(25 + i*5),
|
||||
Text: "#", FontName: "DY1+SimSun", PageNumber: 0,
|
||||
}
|
||||
@@ -188,7 +183,7 @@ func TestSortCharsYFirstly(t *testing.T) {
|
||||
t.Run("same line — fuzzy group by X", func(t *testing.T) {
|
||||
// Chars on the same line with slightly different Top values.
|
||||
// Threshold=10 covers all Top diffs → should sort by X only.
|
||||
chars := []TextChar{
|
||||
chars := []pdf.TextChar{
|
||||
{X0: 50, Top: 12, Text: "C"},
|
||||
{X0: 30, Top: 16, Text: "B"},
|
||||
{X0: 10, Top: 10, Text: "A"},
|
||||
@@ -201,7 +196,7 @@ func TestSortCharsYFirstly(t *testing.T) {
|
||||
|
||||
t.Run("different lines — sort by Y", func(t *testing.T) {
|
||||
// Chars on clearly different lines → sort by Y only.
|
||||
chars := []TextChar{
|
||||
chars := []pdf.TextChar{
|
||||
{X0: 50, Top: 100, Text: "C"},
|
||||
{X0: 30, Top: 10, Text: "A"},
|
||||
{X0: 10, Top: 50, Text: "B"},
|
||||
@@ -214,7 +209,7 @@ func TestSortCharsYFirstly(t *testing.T) {
|
||||
|
||||
t.Run("mixed — same-line group with different-line", func(t *testing.T) {
|
||||
// A and B on line 1 (Top ~10), C on line 2 (Top ~100).
|
||||
chars := []TextChar{
|
||||
chars := []pdf.TextChar{
|
||||
{X0: 50, Top: 100, Text: "C"},
|
||||
{X0: 30, Top: 14, Text: "B"},
|
||||
{X0: 10, Top: 10, Text: "A"},
|
||||
@@ -242,11 +237,11 @@ func TestOCRMergeChars_MixedFontSizes(t *testing.T) {
|
||||
// Chars need height >0.3*boxH to pass height gate.
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRBoxes: []OCRBox{
|
||||
OCRBoxes: []pdf.OCRBox{
|
||||
{X0: 0, Y0: 0, X1: 90, Y1: 0, X2: 90, Y2: 120, X3: 0, Y3: 120},
|
||||
},
|
||||
}
|
||||
chars := []TextChar{
|
||||
chars := []pdf.TextChar{
|
||||
{X0: 3, X1: 12, Top: 10, Bottom: 30, Text: "小"}, // smaller font, higher baseline
|
||||
{X0: 12, X1: 24, Top: 5, Bottom: 35, Text: "大"}, // larger font, lower baseline
|
||||
{X0: 24, X1: 36, Top: 5, Bottom: 35, Text: "号"}, // same size as 大, rightmost
|
||||
@@ -267,17 +262,17 @@ func TestOCRMergeChars_BoxOrder(t *testing.T) {
|
||||
// 3 detect boxes in reverse Y order. After sorting, output should be top-down.
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRBoxes: []OCRBox{
|
||||
OCRBoxes: []pdf.OCRBox{
|
||||
{X0: 0, Y0: 90, X1: 90, Y1: 90, X2: 90, Y2: 120, X3: 0, Y3: 120}, // bottom
|
||||
{X0: 0, Y0: 45, X1: 90, Y1: 45, X2: 90, Y2: 60, X3: 0, Y3: 60}, // middle
|
||||
{X0: 0, Y0: 0, X1: 90, Y1: 0, X2: 90, Y2: 30, X3: 0, Y3: 30}, // top
|
||||
},
|
||||
OCRTexts: []OCRText{{Text: "OCR", Confidence: 0.9}},
|
||||
OCRTexts: []pdf.OCRText{{Text: "OCR", Confidence: 0.9}},
|
||||
}
|
||||
// Chars in PDF space (72 DPI). Detect boxes are at 216 DPI,
|
||||
// scaled down by 3 in ocrMergeChars.
|
||||
// Box1 PDF: y0=0,y1=10. Box2 PDF: y0=15,y1=20. Box3 PDF: y0=30,y1=40.
|
||||
chars := []TextChar{
|
||||
chars := []pdf.TextChar{
|
||||
{X0: 2, X1: 10, Top: 2, Bottom: 7, Text: "A"}, // box 1 (top)
|
||||
{X0: 2, X1: 10, Top: 16, Bottom: 19, Text: "B"}, // box 2 (middle)
|
||||
{X0: 2, X1: 10, Top: 32, Bottom: 37, Text: "C"}, // box 3 (bottom)
|
||||
@@ -309,12 +304,12 @@ func TestOCRMergeChars_OverlappingBoxes(t *testing.T) {
|
||||
// New char-perspective: Box A gets [Y,X] (best overlap), Box B gets [Z].
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
OCRBoxes: []OCRBox{
|
||||
OCRBoxes: []pdf.OCRBox{
|
||||
{X0: 0, Y0: 0, X1: 60, Y1: 0, X2: 60, Y2: 60, X3: 0, Y3: 60}, // Box A
|
||||
{X0: 30, Y0: 0, X1: 90, Y1: 0, X2: 90, Y2: 60, X3: 30, Y3: 60}, // Box B
|
||||
},
|
||||
}
|
||||
chars := []TextChar{
|
||||
chars := []pdf.TextChar{
|
||||
{X0: 2, X1: 8, Top: 2, Bottom: 12, Text: "甲"}, // Box A only
|
||||
{X0: 12, X1: 18, Top: 2, Bottom: 12, Text: "乙"}, // overlap zone
|
||||
{X0: 22, X1: 28, Top: 2, Bottom: 12, Text: "丙"}, // Box B only
|
||||
@@ -333,3 +328,64 @@ func TestOCRMergeChars_OverlappingBoxes(t *testing.T) {
|
||||
t.Errorf("box B: expected '乙丙', got %q", boxes[1].Text)
|
||||
}
|
||||
}
|
||||
|
||||
// ── pdf_oxide ### detection tests ─────────────────────────────────────
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_Empty(t *testing.T) {
|
||||
if util.PdfOxideUnmappedGarbled("") {
|
||||
t.Error("empty text should not be garbled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_NormalText(t *testing.T) {
|
||||
if util.PdfOxideUnmappedGarbled("这是一段正常的中文文本没有任何问题") {
|
||||
t.Error("normal Chinese text should not be garbled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_SingleHash(t *testing.T) {
|
||||
// A single # is not enough (could be a phone number or reference).
|
||||
if util.PdfOxideUnmappedGarbled("参考 #123 的文献") {
|
||||
t.Error("single # should not be garbled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_TripleHashCluster(t *testing.T) {
|
||||
// Two ### sequences => garbled.
|
||||
if !util.PdfOxideUnmappedGarbled("我信###D_8-.###$#(") {
|
||||
t.Error("two ### clusters should be garbled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_QuadHash(t *testing.T) {
|
||||
// One #### counts as one ### cluster. Need two for trigger.
|
||||
// But density may also be high enough.
|
||||
if !util.PdfOxideUnmappedGarbled("text####abc####def") {
|
||||
t.Error("two #### clusters should be garbled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_SingleTriple(t *testing.T) {
|
||||
// Single ### cluster => garbled. In a 200-char sample "###" is impossible
|
||||
// in normal text (URLs/markdown use at most "##").
|
||||
if !util.PdfOxideUnmappedGarbled("hello###world normal text here") {
|
||||
t.Error("single ### cluster should be garbled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_HighDensity(t *testing.T) {
|
||||
// 10 # chars mixed among 40+ non-space chars = 25% → garbled.
|
||||
text := "#a#b#c#d#e#f#g#h#i#j" + " extra normal chars padding to reach minimum"
|
||||
if !util.PdfOxideUnmappedGarbled(text) {
|
||||
t.Error("high # density should be garbled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPdfOxideUnmappedGarbled_RealWorldGarbled(t *testing.T) {
|
||||
// Simulates the garbled page from 1例3个月...pdf:
|
||||
// Chinese text mixed with ###D_ style unmapped glyph patterns.
|
||||
garbled := "和蔘语言###D_8-.*/*护理全科##%&$ 80引用\"\"###$#(点向患儿"
|
||||
if !util.PdfOxideUnmappedGarbled(garbled) {
|
||||
t.Error("real-world garbled text with ### clusters should be detected")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -11,44 +11,12 @@ import (
|
||||
_ "image/png"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"ragflow/internal/deepdoc/parser/pdf/post"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// ── helpers ────────────────────────────────────────────────────────────────
|
||||
|
||||
// mustConnectDeepDoc returns a DeepDocClient; skips the test if unavailable.
|
||||
func mustConnectDeepDoc(t *testing.T) *DeepDocClient {
|
||||
t.Helper()
|
||||
url := os.Getenv("DEEPDOC_URL")
|
||||
if url == "" {
|
||||
url = "http://localhost:9390"
|
||||
}
|
||||
client, err := NewDeepDocClient(url)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !client.Health() {
|
||||
t.Fatalf("DeepDoc not available at %s", url)
|
||||
}
|
||||
return client
|
||||
}
|
||||
|
||||
// mustOpenEngine opens a PDF from testdata/pdfs/ and returns a PDFEngine.
|
||||
func mustOpenEngine(t *testing.T, name string) PDFEngine {
|
||||
t.Helper()
|
||||
pdfPath := filepath.Join("testdata", "pdfs", name)
|
||||
data, err := os.ReadFile(pdfPath)
|
||||
if err != nil {
|
||||
t.Fatalf("read fixture %s: %v", name, err)
|
||||
}
|
||||
eng, err := NewEngine(data)
|
||||
if err != nil {
|
||||
t.Fatalf("open engine %s: %v", name, err)
|
||||
}
|
||||
return eng
|
||||
}
|
||||
|
||||
// ── golden-file helpers ────────────────────────────────────────────────────
|
||||
|
||||
// sectionGolden is the snapshot format for section output.
|
||||
@@ -101,8 +69,8 @@ func updateGolden() bool {
|
||||
return os.Getenv("UPDATE_GOLDEN") == "1"
|
||||
}
|
||||
|
||||
// sectionsToGolden converts []Section to the snapshot format.
|
||||
func sectionsToGolden(sections []Section) []sectionGolden {
|
||||
// sectionsToGolden converts []pdf.Section to the snapshot format.
|
||||
func sectionsToGolden(sections []pdf.Section) []sectionGolden {
|
||||
result := make([]sectionGolden, len(sections))
|
||||
for i, s := range sections {
|
||||
result[i] = sectionGolden{
|
||||
@@ -113,8 +81,8 @@ func sectionsToGolden(sections []Section) []sectionGolden {
|
||||
return result
|
||||
}
|
||||
|
||||
// tablesToGolden converts []TableItem to the snapshot format.
|
||||
func tablesToGolden(tables []TableItem) []tableGolden {
|
||||
// tablesToGolden converts []pdf.TableItem to the snapshot format.
|
||||
func tablesToGolden(tables []pdf.TableItem) []tableGolden {
|
||||
result := make([]tableGolden, len(tables))
|
||||
for i, t := range tables {
|
||||
result[i] = tableGolden{Rows: t.Rows}
|
||||
@@ -126,11 +94,11 @@ func tablesToGolden(tables []TableItem) []tableGolden {
|
||||
|
||||
// TestIntegration_SectionsText verifies section text output matches golden.
|
||||
func TestIntegration_SectionsText(t *testing.T) {
|
||||
client := mustConnectDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
eng := mustOpenEngine(t, "01_english_simple.pdf")
|
||||
defer eng.Close()
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -170,11 +138,11 @@ func TestIntegration_SectionsText(t *testing.T) {
|
||||
|
||||
// TestIntegration_SectionsCount verifies section count is stable.
|
||||
func TestIntegration_SectionsCount(t *testing.T) {
|
||||
client := mustConnectDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
eng := mustOpenEngine(t, "01_english_simple.pdf")
|
||||
defer eng.Close()
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -197,11 +165,11 @@ func TestIntegration_SectionsCount(t *testing.T) {
|
||||
|
||||
// TestIntegration_TableStructure verifies table rows and cell text match golden.
|
||||
func TestIntegration_TableStructure(t *testing.T) {
|
||||
client := mustConnectDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
eng := mustOpenEngine(t, "06_table_content.pdf")
|
||||
defer eng.Close()
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -252,11 +220,11 @@ func TestIntegration_TableStructure(t *testing.T) {
|
||||
|
||||
// TestIntegration_TableImageB64 verifies table ImageB64 is valid base64 PNG.
|
||||
func TestIntegration_TableImageB64(t *testing.T) {
|
||||
client := mustConnectDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
eng := mustOpenEngine(t, "06_table_content.pdf")
|
||||
defer eng.Close()
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -292,11 +260,11 @@ func TestIntegration_TableImageB64(t *testing.T) {
|
||||
|
||||
// TestIntegration_LayoutTypes verifies DLA labels boxes with expected types.
|
||||
func TestIntegration_LayoutTypes(t *testing.T) {
|
||||
client := mustConnectDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
eng := mustOpenEngine(t, "06_table_content.pdf")
|
||||
defer eng.Close()
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -344,7 +312,7 @@ func TestIntegration_LayoutTypes(t *testing.T) {
|
||||
// results when called multiple times with the same image. This validates
|
||||
// that the ML inference is deterministic (or at least semantically stable).
|
||||
func TestIntegration_Idempotency(t *testing.T) {
|
||||
client := mustConnectDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
|
||||
// Render a fixture page as the stable input image.
|
||||
eng := mustOpenEngine(t, "06_table_content.pdf")
|
||||
@@ -357,7 +325,7 @@ func TestIntegration_Idempotency(t *testing.T) {
|
||||
const N = 5
|
||||
|
||||
t.Run("DLA", func(t *testing.T) {
|
||||
var all [][]DLARegion
|
||||
var all [][]pdf.DLARegion
|
||||
for i := 0; i < N; i++ {
|
||||
regions, err := client.DLA(context.Background(), pageImg)
|
||||
if err != nil {
|
||||
@@ -372,7 +340,7 @@ func TestIntegration_Idempotency(t *testing.T) {
|
||||
// Crop a table region from the page for TSR input.
|
||||
// Use a fixed crop area (approximate table location in 06_table_content.pdf).
|
||||
cropped := cropImageRect(pageImg, 50, 200, 550, 400)
|
||||
var all [][]TSRCell
|
||||
var all [][]pdf.TSRCell
|
||||
for i := 0; i < N; i++ {
|
||||
cells, err := client.TSR(context.Background(), cropped)
|
||||
if err != nil {
|
||||
@@ -384,7 +352,7 @@ func TestIntegration_Idempotency(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("OCRDetect", func(t *testing.T) {
|
||||
var all [][]OCRBox
|
||||
var all [][]pdf.OCRBox
|
||||
for i := 0; i < N; i++ {
|
||||
boxes, err := client.OCRDetect(context.Background(), pageImg)
|
||||
if err != nil {
|
||||
@@ -397,7 +365,7 @@ func TestIntegration_Idempotency(t *testing.T) {
|
||||
|
||||
t.Run("OCRRecognize", func(t *testing.T) {
|
||||
cropped := cropImageRect(pageImg, 50, 100, 400, 130)
|
||||
var all [][]OCRText
|
||||
var all [][]pdf.OCRText
|
||||
for i := 0; i < N; i++ {
|
||||
texts, err := client.OCRRecognize(context.Background(), cropped)
|
||||
if err != nil {
|
||||
@@ -436,7 +404,7 @@ func cropImageRect(img image.Image, x0, y0, x1, y1 int) image.Image {
|
||||
const coordEpsilon = 1.0 // pixels
|
||||
const confEpsilon = 0.01
|
||||
|
||||
func checkDLAIdempotent(t *testing.T, all [][]DLARegion) {
|
||||
func checkDLAIdempotent(t *testing.T, all [][]pdf.DLARegion) {
|
||||
t.Helper()
|
||||
ref := all[0]
|
||||
strictEqual := 0
|
||||
@@ -467,7 +435,7 @@ func checkDLAIdempotent(t *testing.T, all [][]DLARegion) {
|
||||
t.Logf("DLA: %d regions, %d/%d runs strictly equal", len(ref), strictEqual+1, len(all))
|
||||
}
|
||||
|
||||
func checkTSRIdempotent(t *testing.T, all [][]TSRCell) {
|
||||
func checkTSRIdempotent(t *testing.T, all [][]pdf.TSRCell) {
|
||||
t.Helper()
|
||||
ref := all[0]
|
||||
strictEqual := 0
|
||||
@@ -491,7 +459,7 @@ func checkTSRIdempotent(t *testing.T, all [][]TSRCell) {
|
||||
t.Logf("TSR: %d cells, %d/%d runs strictly equal", len(ref), strictEqual+1, len(all))
|
||||
}
|
||||
|
||||
func checkOCRDetectIdempotent(t *testing.T, all [][]OCRBox) {
|
||||
func checkOCRDetectIdempotent(t *testing.T, all [][]pdf.OCRBox) {
|
||||
t.Helper()
|
||||
ref := all[0]
|
||||
strictEqual := 0
|
||||
@@ -513,7 +481,7 @@ func checkOCRDetectIdempotent(t *testing.T, all [][]OCRBox) {
|
||||
t.Logf("OCRDetect: %d boxes, %d/%d runs strictly equal", len(ref), strictEqual+1, len(all))
|
||||
}
|
||||
|
||||
func checkOCRRecognizeIdempotent(t *testing.T, all [][]OCRText) {
|
||||
func checkOCRRecognizeIdempotent(t *testing.T, all [][]pdf.OCRText) {
|
||||
t.Helper()
|
||||
ref := all[0]
|
||||
strictEqual := 0
|
||||
@@ -562,11 +530,11 @@ func floatClose(a, b, eps float64) bool {
|
||||
// suppression inside table regions, and caption removal — the key alignment
|
||||
// fixes from the Python→Go migration.
|
||||
func TestIntegration_TableAlign(t *testing.T) {
|
||||
client := mustConnectDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
eng := mustOpenEngine(t, "18_table_caption.pdf")
|
||||
defer eng.Close()
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -576,7 +544,7 @@ func TestIntegration_TableAlign(t *testing.T) {
|
||||
// Assert 1: No caption sections remain (merged into parent or removed).
|
||||
for _, s := range result.Sections {
|
||||
if s.LayoutType == "table caption" || s.LayoutType == "figure caption" {
|
||||
t.Errorf("caption Section should be removed: layout=%s text=%q", s.LayoutType, s.Text)
|
||||
t.Errorf("caption pdf.Section should be removed: layout=%s text=%q", s.LayoutType, s.Text)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -587,7 +555,7 @@ func TestIntegration_TableAlign(t *testing.T) {
|
||||
hasTable = true
|
||||
// Structured text should contain tabs (\t) for column separation.
|
||||
if !strings.Contains(s.Text, "\t") {
|
||||
t.Logf("table Section.Text may not be structured: %q", s.Text[:min(80, len(s.Text))])
|
||||
t.Logf("table pdf.Section.Text may not be structured: %q", s.Text[:min(80, len(s.Text))])
|
||||
}
|
||||
break
|
||||
}
|
||||
@@ -597,17 +565,17 @@ func TestIntegration_TableAlign(t *testing.T) {
|
||||
}
|
||||
|
||||
t.Logf("Sections: %d, Tables: %d, Figures: %d",
|
||||
len(result.Sections), len(result.Tables), len(result.Figures))
|
||||
len(result.Sections), len(result.Tables), len(result.Figures()))
|
||||
}
|
||||
|
||||
// TestIntegration_GarbageLayout verifies CID-garbled and garbage-layout
|
||||
// (header/footer/reference) boxes are popped from output.
|
||||
func TestIntegration_GarbageLayout(t *testing.T) {
|
||||
client := mustConnectDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
eng := mustOpenEngine(t, "17_garbage_layout.pdf")
|
||||
defer eng.Close()
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -634,31 +602,31 @@ func TestIntegration_GarbageLayout(t *testing.T) {
|
||||
|
||||
// TestIntegration_MultiChunk verifies chunked processing for large documents.
|
||||
func TestIntegration_MultiChunk(t *testing.T) {
|
||||
client := mustConnectDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
eng := mustOpenEngine(t, "19_multipage_chunk.pdf")
|
||||
defer eng.Close()
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg.ChunkSize = 10 // small chunks to force multi-chunk path
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
cfg.BatchSize = 10 // small batches to force multi-batch path
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
|
||||
// 52 pages with 10-page chunks → >= 6 chunks.
|
||||
// 52 pages with 10-page batches → >= 6 batches.
|
||||
if len(result.Sections) == 0 {
|
||||
t.Error("multi-chunk should produce sections")
|
||||
t.Error("multi-batch should produce sections")
|
||||
}
|
||||
|
||||
t.Logf("52 pages × chunkSize=10: %d sections, %d tables",
|
||||
t.Logf("52 pages × batchSize=10: %d sections, %d tables",
|
||||
len(result.Sections), len(result.Tables))
|
||||
}
|
||||
|
||||
// TestIntegration_NoRegression runs a few snapshot PDFs and checks basic
|
||||
// invariants — no panic, sections produced, no CID garbage.
|
||||
func TestIntegration_NoRegression(t *testing.T) {
|
||||
client := mustConnectDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
|
||||
for _, name := range []string{
|
||||
"01_english_simple.pdf",
|
||||
@@ -669,7 +637,7 @@ func TestIntegration_NoRegression(t *testing.T) {
|
||||
t.Run(name, func(t *testing.T) {
|
||||
eng := mustOpenEngine(t, name)
|
||||
defer eng.Close()
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -691,12 +659,12 @@ func TestIntegration_NoRegression(t *testing.T) {
|
||||
// TestIntegration_TableRotation verifies that evaluateTableOrientation
|
||||
// correctly detects rotation using region-count scoring.
|
||||
func TestIntegration_TableRotation(t *testing.T) {
|
||||
client := mustConnectDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
|
||||
t.Run("upright_table", func(t *testing.T) {
|
||||
eng := mustOpenEngine(t, "rotate_0.pdf")
|
||||
defer eng.Close()
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -711,7 +679,7 @@ func TestIntegration_TableRotation(t *testing.T) {
|
||||
t.Run("rotated_90_table", func(t *testing.T) {
|
||||
eng := mustOpenEngine(t, "rotate_90.pdf")
|
||||
defer eng.Close()
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
// DeepDoc DLA does not yet correctly annotate boxes on rotated
|
||||
// pages (regions and characters are in different coordinate
|
||||
// spaces post-rotation). Character extraction and rotation are
|
||||
@@ -732,11 +700,11 @@ func TestIntegration_TableRotation(t *testing.T) {
|
||||
// TestIntegration_WordSpacing verifies space insertion between ASCII word
|
||||
// characters with a visible gap (Python __img_ocr space insertion).
|
||||
func TestIntegration_WordSpacing(t *testing.T) {
|
||||
client := mustConnectDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
eng := mustOpenEngine(t, "01_english_simple.pdf")
|
||||
defer eng.Close()
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -762,3 +730,57 @@ func TestIntegration_WordSpacing(t *testing.T) {
|
||||
}
|
||||
t.Logf("word spacing check: %d sections", len(result.Sections))
|
||||
}
|
||||
|
||||
// TestE2E_ParseAndPostProcess runs Parse → PostProcess end-to-end on a real
|
||||
// PDF. Skips VLM (no tenant_id set) but exercises all other operators.
|
||||
func TestE2E_ParseAndPostProcess(t *testing.T) {
|
||||
engine := mustOpenEngine(t, "01_english_simple.pdf")
|
||||
defer engine.Close()
|
||||
|
||||
mock := &MockDocAnalyzer{Healthy: true}
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
|
||||
result, err := p.Parse(context.Background(), engine)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
|
||||
preCount := len(result.Sections)
|
||||
if preCount == 0 {
|
||||
t.Fatal("Parse() returned zero sections")
|
||||
}
|
||||
|
||||
// Post-processing (no VLM).
|
||||
config := post.PipelineConfig{
|
||||
post.ConfigKeyPageWidth: 612.0,
|
||||
post.ConfigKeyZoom: 1.0,
|
||||
}
|
||||
if err := post.PostProcess(context.Background(), result, config); err != nil {
|
||||
t.Fatalf("PostProcess: %v", err)
|
||||
}
|
||||
|
||||
postCount := len(result.Sections)
|
||||
t.Logf("sections: %d → %d after PostProcess", preCount, postCount)
|
||||
if postCount == 0 {
|
||||
t.Error("PostProcess removed all sections")
|
||||
}
|
||||
|
||||
// Every section must have DocTypeKwd + LayoutType set.
|
||||
for i, s := range result.Sections {
|
||||
if s.DocTypeKwd == "" {
|
||||
t.Errorf("section[%d] DocTypeKwd empty after PostProcess", i)
|
||||
}
|
||||
if s.LayoutType == "" {
|
||||
t.Errorf("section[%d] LayoutType empty after PostProcess", i)
|
||||
}
|
||||
}
|
||||
|
||||
// Figures() must reflect post-processed sections.
|
||||
figs := result.Figures()
|
||||
t.Logf("figures: %d", len(figs))
|
||||
for _, f := range figs {
|
||||
if f.LayoutType != "figure" {
|
||||
t.Errorf("Figures() LayoutType=%q, want 'figure'", f.LayoutType)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -7,34 +7,18 @@ import (
|
||||
"encoding/base64"
|
||||
"os"
|
||||
"path/filepath"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// mustConnectDeepDoc returns a DeepDocClient; skips the test if unavailable.
|
||||
func mustConnectDeepDoc(t *testing.T) *DeepDocClient {
|
||||
t.Helper()
|
||||
url := os.Getenv("DEEPDOC_URL")
|
||||
if url == "" {
|
||||
url = "http://localhost:9390"
|
||||
}
|
||||
client, err := NewDeepDocClient(url)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !client.Health() {
|
||||
t.Fatalf("DeepDoc not available at %s", url)
|
||||
}
|
||||
return client
|
||||
}
|
||||
|
||||
// TestIntegration_NoCrash runs Parse on every small fixture PDF and checks it
|
||||
// does not panic or error. It does NOT require golden files.
|
||||
//
|
||||
// Build tag: cgo && manual — skipped in regular integration runs due to
|
||||
// long runtime (27+ PDFs each requiring DeepDoc DLA+TSR+OCR).
|
||||
func TestIntegration_NoCrash(t *testing.T) {
|
||||
client := mustConnectDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
|
||||
pdfDir := filepath.Join("testdata", "pdfs")
|
||||
entries, err := os.ReadDir(pdfDir)
|
||||
@@ -62,7 +46,7 @@ func TestIntegration_NoCrash(t *testing.T) {
|
||||
}
|
||||
defer eng.Close()
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -75,13 +59,13 @@ func TestIntegration_NoCrash(t *testing.T) {
|
||||
t.Errorf("section[%d] has empty PositionTag", i)
|
||||
}
|
||||
if s.LayoutType != "" && s.Image != "" {
|
||||
// Section with an image should have valid base64.
|
||||
// pdf.Section with an image should have valid base64.
|
||||
if _, err := base64.StdEncoding.DecodeString(s.Image); err != nil {
|
||||
t.Errorf("section[%d] Image: not valid base64: %v", i, err)
|
||||
}
|
||||
}
|
||||
if s.TableItem != nil {
|
||||
// Cross-reference: TableItem in section should appear in tables list.
|
||||
// Cross-reference: pdf.TableItem in section should appear in tables list.
|
||||
found := false
|
||||
for _, tbl := range result.Tables {
|
||||
if &tbl == s.TableItem {
|
||||
@@ -90,7 +74,7 @@ func TestIntegration_NoCrash(t *testing.T) {
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Errorf("section[%d] TableItem not found in tables list", i)
|
||||
t.Errorf("section[%d] pdf.TableItem not found in tables list", i)
|
||||
}
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -5,8 +5,7 @@
|
||||
package pdfium
|
||||
|
||||
/*
|
||||
#cgo LDFLAGS: -L/home/shenyushi/cc-workspace/ragflow/.venv/lib/python3.13/site-packages/pypdfium2_raw -lpdfium -lm -lpthread -ldl
|
||||
#cgo linux LDFLAGS: -Wl,-rpath,/home/shenyushi/cc-workspace/ragflow/.venv/lib/python3.13/site-packages/pypdfium2_raw
|
||||
#cgo LDFLAGS: -lpdfium -lm -lpthread -ldl
|
||||
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
@@ -14,6 +13,8 @@ package pdfium
|
||||
typedef struct FPDF_DOCUMENT__ { int unused; } *FPDF_DOCUMENT;
|
||||
typedef struct FPDF_PAGE__ { int unused; } *FPDF_PAGE;
|
||||
typedef struct FPDF_BITMAP__ { int unused; } *FPDF_BITMAP;
|
||||
typedef struct FPDF_BOOKMARK__ { int unused; } *FPDF_BOOKMARK;
|
||||
typedef struct FPDF_DEST__ { int unused; } *FPDF_DEST;
|
||||
|
||||
extern void FPDF_InitLibrary(void);
|
||||
extern FPDF_DOCUMENT FPDF_LoadMemDocument(const void* data_buf, int size, const char* password);
|
||||
@@ -32,6 +33,13 @@ extern void* FPDFBitmap_GetBuffer(FPDF_BITMAP bitmap);
|
||||
extern int FPDFBitmap_GetWidth(FPDF_BITMAP bitmap);
|
||||
extern int FPDFBitmap_GetHeight(FPDF_BITMAP bitmap);
|
||||
extern int FPDFBitmap_GetStride(FPDF_BITMAP bitmap);
|
||||
|
||||
// Outline / bookmark API
|
||||
extern FPDF_BOOKMARK FPDFBookmark_GetFirstChild(FPDF_DOCUMENT document, FPDF_BOOKMARK bookmark);
|
||||
extern FPDF_BOOKMARK FPDFBookmark_GetNextSibling(FPDF_DOCUMENT document, FPDF_BOOKMARK bookmark);
|
||||
extern unsigned long FPDFBookmark_GetTitle(FPDF_BOOKMARK bookmark, void* buffer, unsigned long buflen);
|
||||
extern FPDF_DEST FPDFBookmark_GetDest(FPDF_DOCUMENT document, FPDF_BOOKMARK bookmark);
|
||||
extern int FPDFDest_GetDestPageIndex(FPDF_DOCUMENT document, FPDF_DEST dest);
|
||||
*/
|
||||
import "C"
|
||||
import (
|
||||
@@ -40,9 +48,17 @@ import (
|
||||
"image/color"
|
||||
"math"
|
||||
"sync"
|
||||
"unicode/utf16"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// Outline represents one entry in a PDF document outline (table of contents).
|
||||
type Outline struct {
|
||||
Title string
|
||||
Level int
|
||||
PageNumber int // 1-indexed, matching Python
|
||||
}
|
||||
|
||||
var initOnce sync.Once
|
||||
|
||||
// pdfiumMu serializes all pdfium C API access. pdfium is NOT thread-safe —
|
||||
@@ -163,3 +179,87 @@ func openPage(pdfData []byte, pageIdx int) (
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
// ExtractOutlines returns the document outline (bookmarks / table of contents)
|
||||
// from a PDF. Returns nil for empty or broken PDFs. Title decoding uses UTF-16LE
|
||||
// as required by pdfium's FPDFBookmark_GetTitle.
|
||||
//
|
||||
// Traversal is iterative with an explicit stack to avoid stack overflow on deep
|
||||
// outline trees. pdfium is not thread-safe; callers must hold pdfiumMu.
|
||||
func ExtractOutlines(pdfData []byte) []Outline {
|
||||
if len(pdfData) == 0 {
|
||||
return nil
|
||||
}
|
||||
Init()
|
||||
pdfiumMu.Lock()
|
||||
defer pdfiumMu.Unlock()
|
||||
|
||||
cData := C.CBytes(pdfData)
|
||||
defer C.free(cData)
|
||||
|
||||
doc := C.FPDF_LoadMemDocument(unsafe.Pointer(cData), C.int(len(pdfData)), nil)
|
||||
if doc == nil {
|
||||
return nil
|
||||
}
|
||||
defer C.FPDF_CloseDocument(doc)
|
||||
|
||||
type frame struct {
|
||||
bm C.FPDF_BOOKMARK
|
||||
level int
|
||||
}
|
||||
|
||||
var result []Outline
|
||||
stack := []frame{{bm: C.FPDFBookmark_GetFirstChild(doc, nil), level: 0}}
|
||||
|
||||
for len(stack) > 0 {
|
||||
top := stack[len(stack)-1]
|
||||
stack = stack[:len(stack)-1]
|
||||
if top.bm == nil {
|
||||
continue
|
||||
}
|
||||
|
||||
// Title (UTF-16LE).
|
||||
title := bookmarkTitle(top.bm)
|
||||
|
||||
// Page number.
|
||||
pageNum := 1 // default to page 1 if dest is unavailable
|
||||
if dest := C.FPDFBookmark_GetDest(doc, top.bm); dest != nil {
|
||||
pn := C.FPDFDest_GetDestPageIndex(doc, dest)
|
||||
if pn >= 0 {
|
||||
pageNum = int(pn) + 1 // pdfium returns 0-based
|
||||
}
|
||||
}
|
||||
|
||||
result = append(result, Outline{Title: title, Level: top.level, PageNumber: pageNum})
|
||||
|
||||
// Push siblings after children so children are processed first (pre-order).
|
||||
if sibling := C.FPDFBookmark_GetNextSibling(doc, top.bm); sibling != nil {
|
||||
stack = append(stack, frame{bm: sibling, level: top.level})
|
||||
}
|
||||
if child := C.FPDFBookmark_GetFirstChild(doc, top.bm); child != nil {
|
||||
stack = append(stack, frame{bm: child, level: top.level + 1})
|
||||
}
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// bookmarkTitle reads a bookmark's title (UTF-16LE) and converts to Go string.
|
||||
func bookmarkTitle(bm C.FPDF_BOOKMARK) string {
|
||||
// First call: get required buffer length in bytes.
|
||||
buflen := C.FPDFBookmark_GetTitle(bm, nil, 0)
|
||||
if buflen <= 0 {
|
||||
return ""
|
||||
}
|
||||
buf := make([]byte, buflen)
|
||||
C.FPDFBookmark_GetTitle(bm, unsafe.Pointer(&buf[0]), buflen)
|
||||
|
||||
// Title is UTF-16LE. Convert to []uint16 then decode.
|
||||
n := int(buflen) / 2
|
||||
u16 := unsafe.Slice((*uint16)(unsafe.Pointer(&buf[0])), n)
|
||||
|
||||
// Strip trailing null terminator if present.
|
||||
if n > 0 && u16[n-1] == 0 {
|
||||
u16 = u16[:n-1]
|
||||
}
|
||||
return string(utf16.Decode(u16))
|
||||
}
|
||||
|
||||
120
internal/deepdoc/parser/pdf/pdfium/pdfium_outline_test.go
Normal file
120
internal/deepdoc/parser/pdf/pdfium/pdfium_outline_test.go
Normal file
@@ -0,0 +1,120 @@
|
||||
//go:build cgo && manual
|
||||
|
||||
package pdfium
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// ── ExtractOutlines unit tests ────────────────────────────────────────
|
||||
|
||||
func TestExtractOutlines_Empty(t *testing.T) {
|
||||
// Create a minimal valid PDF without any outlines.
|
||||
minimal := []byte("%PDF-1.4\n1 0 obj<</Type/Catalog/Pages 2 0 R>>endobj\n2 0 obj<</Type/Pages/Kids[3 0 R]/Count 1>>endobj\n3 0 obj<</Type/Page/MediaBox[0 0 612 792]/Parent 2 0 R>>endobj\nxref\n0 4\n0000000000 65535 f \n0000000009 00000 n \n0000000058 00000 n \n0000000115 00000 n \ntrailer<</Size 4/Root 1 0 R>>\nstartxref\n190\n%%EOF")
|
||||
|
||||
outlines := ExtractOutlines(minimal)
|
||||
if len(outlines) != 0 {
|
||||
t.Errorf("expected 0 outlines, got %d: %+v", len(outlines), outlines)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractOutlines_NilOrEmpty(t *testing.T) {
|
||||
if out := ExtractOutlines(nil); len(out) != 0 {
|
||||
t.Errorf("expected 0 outlines for nil, got %d", len(out))
|
||||
}
|
||||
if out := ExtractOutlines([]byte{}); len(out) != 0 {
|
||||
t.Errorf("expected 0 outlines for empty, got %d", len(out))
|
||||
}
|
||||
}
|
||||
|
||||
// ── Integration tests (need real PDF with outlines) ────────────────────
|
||||
|
||||
func TestExtractOutlines_RealPDF(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("skipping integration test in short mode")
|
||||
}
|
||||
pdfData, err := loadTestPDFWithOutlines()
|
||||
if err != nil {
|
||||
t.Skipf("no test PDF with outlines available: %v", err)
|
||||
}
|
||||
|
||||
outlines := ExtractOutlines(pdfData)
|
||||
if len(outlines) == 0 {
|
||||
t.Log("PDF may have no outlines; not necessarily an error")
|
||||
return
|
||||
}
|
||||
|
||||
t.Logf("found %d outline entries:", len(outlines))
|
||||
for i, o := range outlines {
|
||||
t.Logf(" [%d] level=%d page=%d title=%q", i, o.Level, o.PageNumber, o.Title)
|
||||
}
|
||||
|
||||
for i, o := range outlines {
|
||||
if o.Title == "" {
|
||||
t.Errorf("outline[%d]: empty title", i)
|
||||
}
|
||||
if o.PageNumber < 1 {
|
||||
t.Errorf("outline[%d]: invalid PageNumber %d (<1)", i, o.PageNumber)
|
||||
}
|
||||
if o.Level < 0 {
|
||||
t.Errorf("outline[%d]: negative Level %d", i, o.Level)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractOutlines_ChineseTitle(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("skipping integration test in short mode")
|
||||
}
|
||||
pdfData, err := loadTestPDFWithOutlines()
|
||||
if err != nil {
|
||||
t.Skipf("no test PDF available: %v", err)
|
||||
}
|
||||
outlines := ExtractOutlines(pdfData)
|
||||
for _, o := range outlines {
|
||||
for _, r := range o.Title {
|
||||
if r >= 0x4E00 && r <= 0x9FFF { // CJK Unified Ideograph
|
||||
if strings.ContainsRune(o.Title, '<27>') {
|
||||
t.Errorf("title contains U+FFFD (UTF-16LE decode error): %q", o.Title)
|
||||
}
|
||||
return // found CJK, verified no replacement chars
|
||||
}
|
||||
}
|
||||
}
|
||||
t.Log("no CJK characters found in outlines (skip)")
|
||||
}
|
||||
|
||||
// ── helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
var testPDFDirs = []string{
|
||||
"../../testdata/real_pdfs",
|
||||
"../testdata/real_pdfs",
|
||||
"testdata/real_pdfs",
|
||||
}
|
||||
|
||||
func loadTestPDFWithOutlines() ([]byte, error) {
|
||||
for _, dir := range testPDFDirs {
|
||||
entries, err := os.ReadDir(dir)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
for _, e := range entries {
|
||||
if !strings.HasSuffix(e.Name(), ".pdf") {
|
||||
continue
|
||||
}
|
||||
path := filepath.Join(dir, e.Name())
|
||||
data, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
if o := ExtractOutlines(data); len(o) > 0 {
|
||||
return data, nil
|
||||
}
|
||||
}
|
||||
}
|
||||
return nil, fmt.Errorf("no PDF with outlines found in %v", testPDFDirs)
|
||||
}
|
||||
@@ -1,3 +1,5 @@
|
||||
//go:build cgo && manual
|
||||
|
||||
package pdfium
|
||||
|
||||
import (
|
||||
@@ -10,7 +12,7 @@ import (
|
||||
)
|
||||
|
||||
// testdataDir points at the shared test-pdf directory.
|
||||
var testdataDir = filepath.Join("..", "parser", "testdata", "pdfs")
|
||||
var testdataDir = filepath.Join("..", "testdata", "pdfs")
|
||||
|
||||
func readPDF(t *testing.T, name string) []byte {
|
||||
t.Helper()
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
//go:build cgo
|
||||
//go:build cgo && manual
|
||||
|
||||
package parser
|
||||
|
||||
@@ -7,6 +7,7 @@ import (
|
||||
"image"
|
||||
"os"
|
||||
"path/filepath"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"testing"
|
||||
)
|
||||
|
||||
@@ -46,8 +47,8 @@ func TestParse_PdfiumRender(t *testing.T) {
|
||||
|
||||
// Run Parse with pdfium rendering — BATCH_SKIP_DEEPDOC=1 to avoid HTTP calls.
|
||||
t.Setenv("BATCH_SKIP_DEEPDOC", "1")
|
||||
cfg := DefaultParserConfig()
|
||||
p := NewParser(cfg, &MockDocAnalyzer{Healthy: true, Model: ModelSaas})
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
p := NewParser(cfg, &MockDocAnalyzer{Healthy: true})
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
@@ -75,14 +76,15 @@ func TestParse_PdfiumRender_NoData(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// pythonCharEngineStub implements PDFEngine with RawData() returning nil.
|
||||
// pythonCharEngineStub implements pdf.PDFEngine with RawData() returning nil.
|
||||
type pythonCharEngineStub struct{}
|
||||
|
||||
func (e *pythonCharEngineStub) ExtractChars(_ int) ([]TextChar, error) { return nil, nil }
|
||||
func (e *pythonCharEngineStub) ExtractChars(_ int) ([]pdf.TextChar, error) { return nil, nil }
|
||||
func (e *pythonCharEngineStub) RenderPage(_ int, _ float64) ([]byte, error) { return nil, nil }
|
||||
func (e *pythonCharEngineStub) RenderPageImage(_ int, _ float64) (image.Image, error) {
|
||||
return nil, nil
|
||||
}
|
||||
func (e *pythonCharEngineStub) RawData() []byte { return nil }
|
||||
func (e *pythonCharEngineStub) PageCount() (int, error) { return 0, nil }
|
||||
func (e *pythonCharEngineStub) Close() error { return nil }
|
||||
func (e *pythonCharEngineStub) RawData() []byte { return nil }
|
||||
func (e *pythonCharEngineStub) PageCount() (int, error) { return 0, nil }
|
||||
func (e *pythonCharEngineStub) Close() error { return nil }
|
||||
func (e *pythonCharEngineStub) Outlines() ([]pdf.Outline, error) { return nil, nil }
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
//go:build cgo
|
||||
//go:build cgo && manual
|
||||
|
||||
package pdfoxide
|
||||
|
||||
@@ -11,7 +11,7 @@ import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
var fixtureDir = filepath.Join("..", "parser", "testdata", "pdfs")
|
||||
var fixtureDir = filepath.Join("..", "testdata", "pdfs")
|
||||
|
||||
// ── Document opening ─────────────────────────────────────────────────────
|
||||
|
||||
@@ -517,7 +517,7 @@ type pyChar struct {
|
||||
// - text content (as sorted sets, ignoring order differences)
|
||||
// - coordinate ranges (min/max, since absolute positions differ by engine)
|
||||
func TestCharExtraction_CompareWithPython(t *testing.T) {
|
||||
snapDir := filepath.Join("..", "parser", "testdata", "snapshots")
|
||||
snapDir := filepath.Join("..", "testdata", "snapshots")
|
||||
|
||||
entries, err := os.ReadDir(snapDir)
|
||||
if err != nil {
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
//go:build cgo
|
||||
//go:build cgo && manual
|
||||
|
||||
package pdfoxide
|
||||
|
||||
@@ -9,7 +9,7 @@ import (
|
||||
)
|
||||
|
||||
func TestPDFPlumber_Basic(t *testing.T) {
|
||||
pdfDir := filepath.Join("..", "parser", "testdata", "pdfs")
|
||||
pdfDir := filepath.Join("..", "testdata", "pdfs")
|
||||
path := filepath.Join(pdfDir, "01_english_simple.pdf")
|
||||
data, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
@@ -42,7 +42,7 @@ func TestPDFPlumber_Basic(t *testing.T) {
|
||||
}
|
||||
|
||||
func BenchmarkPDFPlumber_ExtractChars(b *testing.B) {
|
||||
pdfDir := filepath.Join("..", "parser", "testdata", "pdfs")
|
||||
pdfDir := filepath.Join("..", "testdata", "pdfs")
|
||||
path := filepath.Join(pdfDir, "01_english_simple.pdf")
|
||||
data, _ := os.ReadFile(path)
|
||||
|
||||
|
||||
@@ -5,16 +5,18 @@ package parser
|
||||
import (
|
||||
"image"
|
||||
|
||||
"ragflow/internal/deepdoc/parser/pdf/pdfium"
|
||||
"ragflow/internal/deepdoc/parser/pdf/pdfoxide"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// pdfoxideEngine adapts pdfoxide.Engine to the PDFEngine interface.
|
||||
// pdfoxideEngine adapts pdfoxide.Engine to the pdf.PDFEngine interface.
|
||||
type pdfoxideEngine struct {
|
||||
inner *pdfoxide.Engine
|
||||
}
|
||||
|
||||
// NewEngine returns a PDFEngine backed by pdf_oxide.
|
||||
func NewEngine(pdfBytes []byte) (PDFEngine, error) {
|
||||
// NewEngine returns a pdf.PDFEngine backed by pdf_oxide.
|
||||
func NewEngine(pdfBytes []byte) (pdf.PDFEngine, error) {
|
||||
eng, err := pdfoxide.NewEngine(pdfBytes)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -26,6 +28,15 @@ func (e *pdfoxideEngine) RawData() []byte { return e.inner.RawData() }
|
||||
func (e *pdfoxideEngine) PageCount() (int, error) { return e.inner.PageCount() }
|
||||
func (e *pdfoxideEngine) Close() error { return e.inner.Close() }
|
||||
|
||||
func (e *pdfoxideEngine) Outlines() ([]pdf.Outline, error) {
|
||||
ol := pdfium.ExtractOutlines(e.inner.RawData())
|
||||
result := make([]pdf.Outline, len(ol))
|
||||
for i, o := range ol {
|
||||
result[i] = pdf.Outline{Title: o.Title, Level: o.Level, PageNumber: o.PageNumber}
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func (e *pdfoxideEngine) RenderPage(pageNum int, dpi float64) ([]byte, error) {
|
||||
return e.inner.RenderPage(pageNum, dpi)
|
||||
}
|
||||
@@ -34,14 +45,14 @@ func (e *pdfoxideEngine) RenderPageImage(pageNum int, dpi float64) (image.Image,
|
||||
return e.inner.RenderPageImage(pageNum, dpi)
|
||||
}
|
||||
|
||||
func (e *pdfoxideEngine) ExtractChars(pageNum int) ([]TextChar, error) {
|
||||
func (e *pdfoxideEngine) ExtractChars(pageNum int) ([]pdf.TextChar, error) {
|
||||
chars, err := e.inner.ExtractChars(pageNum)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
result := make([]TextChar, len(chars))
|
||||
result := make([]pdf.TextChar, len(chars))
|
||||
for i, c := range chars {
|
||||
result[i] = TextChar{
|
||||
result[i] = pdf.TextChar{
|
||||
X0: c.X0, X1: c.X1, Top: c.Top, Bottom: c.Bottom,
|
||||
Text: c.Text, FontName: c.FontName, FontSize: c.FontSize,
|
||||
PageNumber: c.PageNumber,
|
||||
|
||||
@@ -6,10 +6,13 @@ import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"ragflow/internal/deepdoc/parser/pdf/tools"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
lyt "ragflow/internal/deepdoc/parser/pdf/layout"
|
||||
"ragflow/internal/deepdoc/parser/pdf/tool"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// TestPipelineParity verifies Go pipeline logic equivalence with Python.
|
||||
@@ -41,16 +44,16 @@ func TestPipelineParity(t *testing.T) {
|
||||
|
||||
// Load Python chars
|
||||
jsonPath := filepath.Join(charspyDir, e.Name())
|
||||
engine, err := LoadPythonChars(jsonPath)
|
||||
engine, err := tool.LoadPythonChars(jsonPath)
|
||||
if err != nil {
|
||||
t.Errorf("%s: LoadPythonChars: %v", name, err)
|
||||
t.Errorf("%s: tool.LoadPythonChars: %v", name, err)
|
||||
continue
|
||||
}
|
||||
|
||||
// Run Go pipeline (SKIP_OCR — no DeepDoc)
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
cfg.SortByTop = true
|
||||
p := NewParser(cfg, &MockDocAnalyzer{Healthy: true, Model: ModelSaas})
|
||||
p := NewParser(cfg, &MockDocAnalyzer{Healthy: true})
|
||||
result, err := p.Parse(context.Background(), engine)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Parse: %v", name, err)
|
||||
@@ -73,7 +76,7 @@ func TestPipelineParity(t *testing.T) {
|
||||
}
|
||||
|
||||
// Compare
|
||||
sim := tools.CharSimilarity(goText.String(), tools.StripMeta(string(pyData)))
|
||||
sim := tool.CharSimilarity(goText.String(), tool.StripMeta(string(pyData)))
|
||||
total++
|
||||
if sim >= 100.0 {
|
||||
passed++
|
||||
@@ -104,7 +107,7 @@ func TestPipelineParity(t *testing.T) {
|
||||
// bottom extension never happens and the cascade fails to start.
|
||||
func TestVMWhitespaceGapBridge(t *testing.T) {
|
||||
// Coordinates extracted from RAG PDF charspy data, "服务体系" region.
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
// Content A: merged result of 3 preceding lines
|
||||
{X0: 37.6, X1: 491.0, Top: 339.35, Bottom: 382.39,
|
||||
Text: "生成文本再用standard分词建立索引", PageNumber: 1},
|
||||
@@ -128,7 +131,7 @@ func TestVMWhitespaceGapBridge(t *testing.T) {
|
||||
// We simulate this by letting whitespace through gap/xov checks
|
||||
// and absorbing it into prev when the checks pass.
|
||||
vWithWS := func() int {
|
||||
bxs := make([]TextBox, len(boxes))
|
||||
bxs := make([]pdf.TextBox, len(boxes))
|
||||
copy(bxs, boxes)
|
||||
sort.Slice(bxs, func(i, j int) bool {
|
||||
if bxs[i].Top != bxs[j].Top {
|
||||
@@ -136,7 +139,7 @@ func TestVMWhitespaceGapBridge(t *testing.T) {
|
||||
}
|
||||
return bxs[i].X0 < bxs[j].X0
|
||||
})
|
||||
out := make([]TextBox, 0, len(bxs))
|
||||
out := make([]pdf.TextBox, 0, len(bxs))
|
||||
for i := 0; i < len(bxs); i++ {
|
||||
b := bxs[i]
|
||||
isWS := strings.TrimSpace(b.Text) == ""
|
||||
@@ -191,7 +194,7 @@ func TestVMWhitespaceGapBridge(t *testing.T) {
|
||||
|
||||
// Run VM with whitespace PRE-FILTERED (Go current behavior).
|
||||
vNoWS := func() int {
|
||||
bxs := make([]TextBox, 0, len(boxes))
|
||||
bxs := make([]pdf.TextBox, 0, len(boxes))
|
||||
for _, b := range boxes {
|
||||
if strings.TrimSpace(b.Text) != "" {
|
||||
bxs = append(bxs, b)
|
||||
@@ -203,7 +206,7 @@ func TestVMWhitespaceGapBridge(t *testing.T) {
|
||||
}
|
||||
return bxs[i].X0 < bxs[j].X0
|
||||
})
|
||||
out := make([]TextBox, 0, len(bxs))
|
||||
out := make([]pdf.TextBox, 0, len(bxs))
|
||||
for i := 0; i < len(bxs); i++ {
|
||||
b := bxs[i]
|
||||
if len(out) == 0 {
|
||||
@@ -256,7 +259,7 @@ func TestVMWhitespaceGapBridge(t *testing.T) {
|
||||
// Verify production NaiveVerticalMerge matches vWithWS (Python behavior).
|
||||
mhMap := map[int]float64{1: mh}
|
||||
mwMap := map[int]float64{1: 5}
|
||||
vmResult := NaiveVerticalMerge(boxes, mhMap, mwMap, false)
|
||||
vmResult := lyt.NaiveVerticalMerge(boxes, mhMap, mwMap, false)
|
||||
t.Logf("NaiveVerticalMerge (production): %d sections", len(vmResult))
|
||||
if len(vmResult) != nWS {
|
||||
t.Errorf("NaiveVerticalMerge produced %d sections, want %d (Python-like with gap bridge)", len(vmResult), nWS)
|
||||
|
||||
101
internal/deepdoc/parser/pdf/post/model_image_describer.go
Normal file
101
internal/deepdoc/parser/pdf/post/model_image_describer.go
Normal file
@@ -0,0 +1,101 @@
|
||||
package post
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/base64"
|
||||
"errors"
|
||||
"fmt"
|
||||
"image"
|
||||
"image/png"
|
||||
)
|
||||
|
||||
// ── chat driver interface (self-contained, avoids entity/models import) ──
|
||||
|
||||
// ChatDriver is the subset of modelModule.ModelDriver needed to call a
|
||||
// vision-capable chat API. Defined here to keep model_image_describer.go
|
||||
// self-contained and avoid import chains that require CGO.
|
||||
type ChatDriver interface {
|
||||
ChatWithMessages(modelName string, messages []ChatMessage, apiConfig *ChatAPIConfig, chatConfig *ChatConfig) (*ChatResponse, error)
|
||||
}
|
||||
|
||||
// ChatMessage mirrors modelModule.Message.
|
||||
type ChatMessage struct {
|
||||
Role string `json:"role"`
|
||||
Content interface{} `json:"content"`
|
||||
ToolCallID string `json:"tool_call_id,omitempty"`
|
||||
ToolCalls []map[string]interface{} `json:"tool_calls,omitempty"`
|
||||
}
|
||||
|
||||
// ChatAPIConfig mirrors modelModule.APIConfig.
|
||||
type ChatAPIConfig struct {
|
||||
ApiKey *string
|
||||
Region *string
|
||||
BaseURL *string
|
||||
}
|
||||
|
||||
// ChatConfig mirrors modelModule.ChatConfig (may be nil).
|
||||
type ChatConfig struct{}
|
||||
|
||||
// ChatResponse mirrors modelModule.ChatResponse.
|
||||
type ChatResponse struct {
|
||||
Answer *string `json:"answer"`
|
||||
ReasonContent *string `json:"reason_content"`
|
||||
ToolCalls []map[string]interface{} `json:"tool_calls,omitempty"`
|
||||
}
|
||||
|
||||
// ── ModelImageDescriber ────────────────────────────────────────────────
|
||||
|
||||
// ModelImageDescriber implements ImageDescriber via any ChatDriver.
|
||||
type ModelImageDescriber struct {
|
||||
driver ChatDriver
|
||||
modelName string
|
||||
apiConfig *ChatAPIConfig
|
||||
maxTokens int
|
||||
}
|
||||
|
||||
// NewModelImageDescriber creates a ModelImageDescriber that calls the given
|
||||
// driver to describe images. maxTokens sets the response length limit (passed
|
||||
// as ChatConfig.MaxTokens); 0 means use provider default.
|
||||
func NewModelImageDescriber(d ChatDriver, name string, cfg *ChatAPIConfig, maxTokens int) *ModelImageDescriber {
|
||||
return &ModelImageDescriber{driver: d, modelName: name, apiConfig: cfg, maxTokens: maxTokens}
|
||||
}
|
||||
|
||||
// DescribeImage sends the image as a base64 data URL in an OpenAI-compatible
|
||||
// vision API request. Returns the model's text response.
|
||||
func (d *ModelImageDescriber) DescribeImage(ctx context.Context, img image.Image, prompt string) (string, error) {
|
||||
dataURL, err := encodeImageToBase64DataURL(img)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("image encode: %w", err)
|
||||
}
|
||||
|
||||
msgs := []ChatMessage{{
|
||||
Role: "user",
|
||||
Content: []interface{}{
|
||||
map[string]interface{}{"type": "text", "text": prompt},
|
||||
map[string]interface{}{"type": "image_url", "image_url": map[string]string{"url": dataURL}},
|
||||
},
|
||||
}}
|
||||
|
||||
var chatCfg *ChatConfig
|
||||
if d.maxTokens > 0 {
|
||||
chatCfg = &ChatConfig{}
|
||||
}
|
||||
resp, err := d.driver.ChatWithMessages(d.modelName, msgs, d.apiConfig, chatCfg)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("image describe: %w", err)
|
||||
}
|
||||
if resp.Answer == nil || *resp.Answer == "" {
|
||||
return "", errors.New("image describe: empty response")
|
||||
}
|
||||
return *resp.Answer, nil
|
||||
}
|
||||
|
||||
// encodeImageToBase64DataURL encodes an image as a PNG data URL.
|
||||
func encodeImageToBase64DataURL(img image.Image) (string, error) {
|
||||
var buf bytes.Buffer
|
||||
if err := png.Encode(&buf, img); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return "data:image/png;base64," + base64.StdEncoding.EncodeToString(buf.Bytes()), nil
|
||||
}
|
||||
@@ -0,0 +1,79 @@
|
||||
package post
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"image"
|
||||
"image/color"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// ── mock ChatDriver ────────────────────────────────────────────────────
|
||||
|
||||
type mockChatDriver struct {
|
||||
answer string
|
||||
err error
|
||||
}
|
||||
|
||||
func (m *mockChatDriver) ChatWithMessages(_ string, _ []ChatMessage, _ *ChatAPIConfig, _ *ChatConfig) (*ChatResponse, error) {
|
||||
if m.err != nil {
|
||||
return nil, m.err
|
||||
}
|
||||
a := m.answer
|
||||
return &ChatResponse{Answer: &a}, nil
|
||||
}
|
||||
|
||||
// ── ModelImageDescriber tests ──────────────────────────────────────────
|
||||
|
||||
func TestModelImageDescriber_Success(t *testing.T) {
|
||||
img := newTestImage(100, 100)
|
||||
want := "A chart showing revenue growth."
|
||||
driver := &mockChatDriver{answer: want}
|
||||
desc := NewModelImageDescriber(driver, "gpt-4o", nil, 0)
|
||||
|
||||
got, err := desc.DescribeImage(context.Background(), img, "Describe this chart")
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
if got != want {
|
||||
t.Errorf("got %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestModelImageDescriber_DriverError(t *testing.T) {
|
||||
img := newTestImage(100, 100)
|
||||
driver := &mockChatDriver{err: errors.New("API rate limited")}
|
||||
desc := NewModelImageDescriber(driver, "gpt-4o", nil, 0)
|
||||
|
||||
_, err := desc.DescribeImage(context.Background(), img, "prompt")
|
||||
if err == nil {
|
||||
t.Fatal("expected error, got nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestModelImageDescriber_EmptyAnswer(t *testing.T) {
|
||||
img := newTestImage(100, 100)
|
||||
driver := &mockChatDriver{answer: ""}
|
||||
desc := NewModelImageDescriber(driver, "gpt-4o", nil, 0)
|
||||
|
||||
_, err := desc.DescribeImage(context.Background(), img, "prompt")
|
||||
if err == nil {
|
||||
t.Fatal("expected error for empty answer, got nil")
|
||||
}
|
||||
}
|
||||
|
||||
// ── encodeImageToBase64DataURL tests ───────────────────────────────────
|
||||
|
||||
func TestEncodeImageToBase64DataURL(t *testing.T) {
|
||||
img := image.NewRGBA(image.Rect(0, 0, 1, 1))
|
||||
img.Set(0, 0, color.RGBA{R: 255, G: 0, B: 0, A: 255})
|
||||
|
||||
url, err := encodeImageToBase64DataURL(img)
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
if !strings.HasPrefix(url, "data:image/png;base64,") {
|
||||
t.Errorf("missing data URL prefix: %s...", url[:min(50, len(url))])
|
||||
}
|
||||
}
|
||||
114
internal/deepdoc/parser/pdf/post/outline_postprocess_test.go
Normal file
114
internal/deepdoc/parser/pdf/post/outline_postprocess_test.go
Normal file
@@ -0,0 +1,114 @@
|
||||
package post
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
pdftype "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// ── Tests for remove_toc config flag ────────────────────────────────────────
|
||||
|
||||
// TestPostProcess_RemoveTOC_DisabledByConfig verifies that when
|
||||
// remove_toc=false, outlines are NOT used to remove TOC pages even
|
||||
// when outlines are present.
|
||||
func TestPostProcess_RemoveTOC_DisabledByConfig(t *testing.T) {
|
||||
result := newTestResult(
|
||||
makePosSection("目录内容 page1", 1, 100, 500, 100, 200),
|
||||
makePosSection("更多目录 page2", 2, 100, 500, 100, 200),
|
||||
makePosSection("第一章 正文", 3, 100, 500, 100, 200),
|
||||
makePosSection("第二章 正文", 5, 100, 500, 100, 200),
|
||||
)
|
||||
outlines := []pdftype.Outline{
|
||||
{Title: "目录", Level: 0, PageNumber: 1},
|
||||
{Title: "第一章", Level: 0, PageNumber: 3},
|
||||
{Title: "第二章", Level: 0, PageNumber: 5},
|
||||
}
|
||||
|
||||
config := PipelineConfig{
|
||||
ConfigKeyRemoveTOC: false,
|
||||
ConfigKeyOutlines: outlines,
|
||||
}
|
||||
err := PostProcess(context.Background(), result, config)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(result.Sections) != 4 {
|
||||
t.Errorf("remove_toc=false should keep all sections, got %d", len(result.Sections))
|
||||
}
|
||||
}
|
||||
|
||||
// TestPostProcess_RemoveTOC_EnabledByConfig verifies that when
|
||||
// remove_toc=true and outlines are present, TOC pages are removed.
|
||||
func TestPostProcess_RemoveTOC_EnabledByConfig(t *testing.T) {
|
||||
result := newTestResult(
|
||||
makePosSection("目录内容 page1", 1, 100, 500, 100, 200),
|
||||
makePosSection("更多目录 page2", 2, 100, 500, 100, 200),
|
||||
makePosSection("第一章 正文", 3, 100, 500, 100, 200),
|
||||
makePosSection("第二章 正文", 5, 100, 500, 100, 200),
|
||||
)
|
||||
outlines := []pdftype.Outline{
|
||||
{Title: "目录", Level: 0, PageNumber: 1},
|
||||
{Title: "第一章", Level: 0, PageNumber: 3},
|
||||
{Title: "第二章", Level: 0, PageNumber: 5},
|
||||
}
|
||||
|
||||
config := PipelineConfig{
|
||||
ConfigKeyRemoveTOC: true,
|
||||
ConfigKeyOutlines: outlines,
|
||||
}
|
||||
err := PostProcess(context.Background(), result, config)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(result.Sections) != 2 {
|
||||
t.Errorf("remove_toc=true should remove TOC pages, got %d sections", len(result.Sections))
|
||||
}
|
||||
for _, s := range result.Sections {
|
||||
for _, p := range s.Positions {
|
||||
for _, pn := range p.PageNumbers {
|
||||
if pn < 3 {
|
||||
t.Errorf("TOC page %d should have been removed: section %q", pn, s.Text)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestPostProcess_RemoveTOC_NoOutlines verifies that when no outlines
|
||||
// are passed, no TOC removal happens.
|
||||
func TestPostProcess_RemoveTOC_NoOutlines(t *testing.T) {
|
||||
result := newTestResult(
|
||||
makePosSection("目录内容", 1, 100, 500, 100, 200),
|
||||
makePosSection("第一章 正文", 3, 100, 500, 100, 200),
|
||||
)
|
||||
config := PipelineConfig{
|
||||
ConfigKeyRemoveTOC: true,
|
||||
}
|
||||
err := PostProcess(context.Background(), result, config)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(result.Sections) != 2 {
|
||||
t.Errorf("no outlines → all sections kept, got %d", len(result.Sections))
|
||||
}
|
||||
}
|
||||
|
||||
// TestPostProcess_RemoveTOC_EmptyOutlines verifies empty outlines array is no-op.
|
||||
func TestPostProcess_RemoveTOC_EmptyOutlines(t *testing.T) {
|
||||
result := newTestResult(
|
||||
makePosSection("目录", 1, 100, 500, 100, 200),
|
||||
makePosSection("正文", 2, 100, 500, 100, 200),
|
||||
)
|
||||
config := PipelineConfig{
|
||||
ConfigKeyRemoveTOC: true,
|
||||
ConfigKeyOutlines: []pdftype.Outline{},
|
||||
}
|
||||
err := PostProcess(context.Background(), result, config)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(result.Sections) != 2 {
|
||||
t.Errorf("empty outlines → all sections kept, got %d", len(result.Sections))
|
||||
}
|
||||
}
|
||||
436
internal/deepdoc/parser/pdf/post/post_steps.go
Normal file
436
internal/deepdoc/parser/pdf/post/post_steps.go
Normal file
@@ -0,0 +1,436 @@
|
||||
package post
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"math"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
pdftype "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"ragflow/internal/deepdoc/parser/pdf/util"
|
||||
)
|
||||
|
||||
// ── Config ─────────────────────────────────────────────────────────────
|
||||
|
||||
// Config keys for PipelineConfig.
|
||||
const (
|
||||
ConfigKeyPageWidth = "page_width"
|
||||
ConfigKeyZoom = "zoom"
|
||||
ConfigKeyOutlines = "outlines"
|
||||
ConfigKeyFlattenMediaToText = "flatten_media_to_text"
|
||||
ConfigKeyTenantID = "tenant_id"
|
||||
ConfigKeyVLMLLMID = "vlm_llm_id"
|
||||
ConfigKeyRemoveTOC = "remove_toc"
|
||||
)
|
||||
|
||||
// PipelineConfig is a key-value map that post-processing reads
|
||||
// to obtain its parameters.
|
||||
type PipelineConfig map[string]interface{}
|
||||
|
||||
// Float64 returns the float64 value for key, or default_ if absent or wrong type.
|
||||
func (c PipelineConfig) Float64(key string, default_ float64) float64 {
|
||||
if c == nil {
|
||||
return default_
|
||||
}
|
||||
v, ok := c[key]
|
||||
if !ok {
|
||||
return default_
|
||||
}
|
||||
f, ok := v.(float64)
|
||||
if !ok {
|
||||
return default_
|
||||
}
|
||||
return f
|
||||
}
|
||||
|
||||
// Bool returns the bool value for key. Returns false if absent or wrong type.
|
||||
func (c PipelineConfig) Bool(key string) bool {
|
||||
if c == nil {
|
||||
return false
|
||||
}
|
||||
v, ok := c[key]
|
||||
if !ok {
|
||||
return false
|
||||
}
|
||||
b, ok := v.(bool)
|
||||
if !ok {
|
||||
return false
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
// Outlines returns the []pdftype.Outline value for ConfigKeyOutlines.
|
||||
func (c PipelineConfig) Outlines() []pdftype.Outline {
|
||||
if c == nil {
|
||||
return nil
|
||||
}
|
||||
v, ok := c[ConfigKeyOutlines]
|
||||
if !ok {
|
||||
return nil
|
||||
}
|
||||
o, ok := v.([]pdftype.Outline)
|
||||
if !ok {
|
||||
return nil
|
||||
}
|
||||
return o
|
||||
}
|
||||
|
||||
// String returns the string value for key. Returns "" if absent or wrong type.
|
||||
func (c PipelineConfig) String(key string) string {
|
||||
if c == nil {
|
||||
return ""
|
||||
}
|
||||
v, ok := c[key]
|
||||
if !ok {
|
||||
return ""
|
||||
}
|
||||
s, ok := v.(string)
|
||||
if !ok {
|
||||
return ""
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// ── Patterns ───────────────────────────────────────────────────────────
|
||||
|
||||
// headerFooterPattern matches layout types that should be treated as
|
||||
// page furniture (Python: r"(header|footer|number)" in parser.py:637).
|
||||
var headerFooterPattern = regexp.MustCompile(`(header|footer|number|reference)`)
|
||||
|
||||
// tocTitlePattern matches outline titles that mark a table-of-contents page.
|
||||
// Python: r"(contents|目录|目次|table of contents|致谢|acknowledge)$"
|
||||
var tocTitlePattern = regexp.MustCompile(`(?i)^(contents|目录|目次|table of contents|致谢|acknowledge)$`)
|
||||
|
||||
// ── PostProcess ────────────────────────────────────────────────────────
|
||||
|
||||
// PostProcess applies PDF post-processing to a ParseResult in-place.
|
||||
// The config map controls which features to enable.
|
||||
//
|
||||
// Execution order (matches Python _pdf):
|
||||
// 1. reorderMultiColumn — if page_width > 0
|
||||
// 2. removeTOCByOutlines — if outlines present
|
||||
// 3. normalizeLayoutType — always
|
||||
// 4. filterHeaderFooter — always
|
||||
// 5. assignDocTypeKwd — always (respects flatten_media_to_text)
|
||||
// 6. enhanceWithVision — if image_describer present
|
||||
func PostProcess(ctx context.Context, result *pdftype.ParseResult, config PipelineConfig) error {
|
||||
if result == nil {
|
||||
return errors.New("PostProcess: nil result")
|
||||
}
|
||||
if config == nil {
|
||||
config = PipelineConfig{}
|
||||
}
|
||||
|
||||
// 1. Multi-column reorder
|
||||
pw := config.Float64(ConfigKeyPageWidth, 0)
|
||||
if pw > 0 {
|
||||
zoom := config.Float64(ConfigKeyZoom, 1.0)
|
||||
if zoom <= 0 {
|
||||
zoom = 1.0
|
||||
}
|
||||
reorderMultiColumn(result, pw, zoom)
|
||||
}
|
||||
|
||||
// 2. Remove TOC pages (only when explicitly enabled).
|
||||
// Outlines from config take precedence; otherwise read from ParseResult.
|
||||
outlines := config.Outlines()
|
||||
if len(outlines) == 0 {
|
||||
outlines = result.Outlines
|
||||
}
|
||||
if config.Bool(ConfigKeyRemoveTOC) && len(outlines) > 0 {
|
||||
removeTOCByOutlines(result, outlines)
|
||||
}
|
||||
|
||||
// 3-5. Always-on steps
|
||||
normalizeLayoutType(result)
|
||||
filterHeaderFooter(result)
|
||||
assignDocTypeKwd(result, config.Bool(ConfigKeyFlattenMediaToText))
|
||||
|
||||
// 6. VLM enhancement
|
||||
tenantID := config.String(ConfigKeyTenantID)
|
||||
vlmLLMID := config.String(ConfigKeyVLMLLMID)
|
||||
if tenantID != "" && vlmLLMID != "" {
|
||||
describer, err := resolveImageDescriber(tenantID, vlmLLMID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := enhanceWithVision(ctx, result, describer); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// resolveImageDescriber resolves a VLM model from tenant config and returns
|
||||
// an ImageDescriber. Corresponds to Python's
|
||||
// get_model_config_from_provider_instance + LLMBundle.
|
||||
// resolveImageDescriber resolves a VLM model from tenant config and returns
|
||||
// an ImageDescriber. The implementation is assigned by init() in
|
||||
// post_steps_cgo.go (production) or post_steps_no_cgo.go (stub).
|
||||
// Overridable in tests.
|
||||
var resolveImageDescriber func(tenantID, llmID string) (ImageDescriber, error)
|
||||
|
||||
// SetImageDescriberResolver sets the factory that creates an ImageDescriber
|
||||
// from tenant/LLM configuration. Higher layers (e.g. EE extensions or the
|
||||
// PDF document pipeline entry point) register the real implementation via
|
||||
// init(). If never called, PostProcess skips VLM enhancement.
|
||||
func SetImageDescriberResolver(fn func(tenantID, llmID string) (ImageDescriber, error)) {
|
||||
resolveImageDescriber = fn
|
||||
}
|
||||
|
||||
// ── normalizeLayoutType ────────────────────────────────────────────────
|
||||
|
||||
// normalizeLayoutType trims whitespace from LayoutType and defaults empty
|
||||
// values to "text". Matches Python's layout_type normalization in parser.py.
|
||||
func normalizeLayoutType(result *pdftype.ParseResult) {
|
||||
for i := range result.Sections {
|
||||
lt := strings.TrimSpace(result.Sections[i].LayoutType)
|
||||
if lt == "" {
|
||||
lt = "text"
|
||||
}
|
||||
result.Sections[i].LayoutType = lt
|
||||
}
|
||||
}
|
||||
|
||||
// ── filterHeaderFooter ─────────────────────────────────────────────────
|
||||
|
||||
// filterHeaderFooter removes sections whose LayoutType matches
|
||||
// header/footer/number/reference. Python: remove_header_footer config.
|
||||
func filterHeaderFooter(result *pdftype.ParseResult) {
|
||||
sections := result.Sections[:0]
|
||||
for _, s := range result.Sections {
|
||||
if headerFooterPattern.MatchString(strings.TrimSpace(s.LayoutType)) {
|
||||
continue
|
||||
}
|
||||
sections = append(sections, s)
|
||||
}
|
||||
result.Sections = sections
|
||||
}
|
||||
|
||||
// ── assignDocTypeKwd ───────────────────────────────────────────────────
|
||||
|
||||
// assignDocTypeKwd sets DocTypeKwd based on LayoutType and Image presence.
|
||||
// When flatten is true, all sections become "text" and Image is cleared —
|
||||
// this matches Python where flatten_media_to_text and VLM are mutually
|
||||
// exclusive. Python: parser.py:639-648.
|
||||
func assignDocTypeKwd(result *pdftype.ParseResult, flatten bool) {
|
||||
for i := range result.Sections {
|
||||
s := &result.Sections[i]
|
||||
if flatten {
|
||||
s.DocTypeKwd = "text"
|
||||
s.Image = ""
|
||||
continue
|
||||
}
|
||||
lt := strings.TrimSpace(s.LayoutType)
|
||||
switch lt {
|
||||
case "table":
|
||||
s.DocTypeKwd = "table"
|
||||
case "figure":
|
||||
s.DocTypeKwd = "image"
|
||||
default:
|
||||
if lt == "" && s.Image != "" {
|
||||
s.DocTypeKwd = "image"
|
||||
} else {
|
||||
s.DocTypeKwd = "text"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── enhanceWithVision ──────────────────────────────────────────────────
|
||||
|
||||
// enhanceWithVision adds VLM-generated descriptions to image/table sections.
|
||||
func enhanceWithVision(ctx context.Context, result *pdftype.ParseResult, describer ImageDescriber) error {
|
||||
if describer == nil {
|
||||
return nil
|
||||
}
|
||||
if len(result.Sections) == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
sem := make(chan struct{}, maxDescribeConcurrency)
|
||||
var wg sync.WaitGroup
|
||||
|
||||
for i := range result.Sections {
|
||||
s := &result.Sections[i]
|
||||
if s.DocTypeKwd != "table" && s.DocTypeKwd != "image" {
|
||||
continue
|
||||
}
|
||||
if s.Image == "" {
|
||||
continue
|
||||
}
|
||||
|
||||
wg.Add(1)
|
||||
sem <- struct{}{}
|
||||
go func(idx int, imgB64 string, origText string) {
|
||||
defer wg.Done()
|
||||
defer func() { <-sem }()
|
||||
|
||||
img, err := util.DecodeBase64PNG(imgB64)
|
||||
if err != nil || img == nil {
|
||||
return
|
||||
}
|
||||
desc, err := DescribeImage(ctx, img, describePrompt, describer)
|
||||
if err != nil || desc == "" {
|
||||
return
|
||||
}
|
||||
|
||||
if origText != "" {
|
||||
result.Sections[idx].Text = origText + "\n" + desc
|
||||
} else {
|
||||
result.Sections[idx].Text = desc
|
||||
}
|
||||
}(i, s.Image, s.Text)
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// ── removeTOCByOutlines ────────────────────────────────────────────────
|
||||
|
||||
// removeTOCByOutlines removes sections whose page numbers fall inside
|
||||
// TOC page ranges identified by PDF outlines.
|
||||
func removeTOCByOutlines(result *pdftype.ParseResult, outlines []pdftype.Outline) {
|
||||
if len(outlines) == 0 {
|
||||
return
|
||||
}
|
||||
tocPage, contentPage := findTOCPageRange(outlines)
|
||||
if contentPage <= tocPage {
|
||||
return
|
||||
}
|
||||
sections := result.Sections[:0]
|
||||
for _, s := range result.Sections {
|
||||
pg := sectionPage(s)
|
||||
if pg >= tocPage && pg < contentPage {
|
||||
continue
|
||||
}
|
||||
sections = append(sections, s)
|
||||
}
|
||||
result.Sections = sections
|
||||
}
|
||||
|
||||
// findTOCPageRange scans outlines for a TOC entry and returns the
|
||||
// [tocStartPage, contentStartPage) range. Returns (0, 0) when not found.
|
||||
func findTOCPageRange(outlines []pdftype.Outline) (tocPage, contentPage int) {
|
||||
trimSplit:
|
||||
for i, o := range outlines {
|
||||
title := strings.TrimSpace(o.Title)
|
||||
if idx := strings.Index(title, "@@"); idx >= 0 {
|
||||
title = strings.TrimSpace(title[:idx])
|
||||
}
|
||||
if !tocTitlePattern.MatchString(strings.ToLower(title)) {
|
||||
continue
|
||||
}
|
||||
tocPage = o.PageNumber
|
||||
for _, next := range outlines[i+1:] {
|
||||
if next.Level != o.Level {
|
||||
continue
|
||||
}
|
||||
nt := strings.TrimSpace(next.Title)
|
||||
if idx := strings.Index(nt, "@@"); idx >= 0 {
|
||||
nt = strings.TrimSpace(nt[:idx])
|
||||
}
|
||||
if tocTitlePattern.MatchString(strings.ToLower(nt)) {
|
||||
continue
|
||||
}
|
||||
contentPage = next.PageNumber
|
||||
break trimSplit
|
||||
}
|
||||
break
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
// sectionPage returns the first page number of a Section, or 0.
|
||||
func sectionPage(s pdftype.Section) int {
|
||||
for _, p := range s.Positions {
|
||||
for _, pn := range p.PageNumbers {
|
||||
return pn
|
||||
}
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// ── reorderMultiColumn ─────────────────────────────────────────────────
|
||||
|
||||
// reorderMultiColumn reorders text sections in multi-column layouts.
|
||||
// If median text column width >= page width / 2 (single-column layout),
|
||||
// the input order is preserved.
|
||||
//
|
||||
// Python: reorder_multi_column_bboxes + sort_X_by_page
|
||||
func reorderMultiColumn(result *pdftype.ParseResult, pageWidth, zoom float64) {
|
||||
if len(result.Sections) < 2 {
|
||||
return
|
||||
}
|
||||
pw := pageWidth / zoom
|
||||
|
||||
// Compute median width from text sections with valid coordinates.
|
||||
var widths []float64
|
||||
for _, s := range result.Sections {
|
||||
if s.LayoutType != "text" {
|
||||
continue
|
||||
}
|
||||
if len(s.Positions) == 0 {
|
||||
continue
|
||||
}
|
||||
w := s.Positions[0].Right - s.Positions[0].Left
|
||||
if w > 0 {
|
||||
widths = append(widths, w)
|
||||
}
|
||||
}
|
||||
if len(widths) == 0 {
|
||||
return
|
||||
}
|
||||
sort.Float64s(widths)
|
||||
medianW := widths[len(widths)/2]
|
||||
|
||||
if medianW >= pw/2 {
|
||||
return // single column
|
||||
}
|
||||
|
||||
// Sort by (PageNumber, X0, Top).
|
||||
sort.Slice(result.Sections, func(i, j int) bool {
|
||||
pi := sectionPage(result.Sections[i])
|
||||
pj := sectionPage(result.Sections[j])
|
||||
if pi != pj {
|
||||
return pi < pj
|
||||
}
|
||||
xi := sectionX0(result.Sections[i])
|
||||
xj := sectionX0(result.Sections[j])
|
||||
if math.Abs(xi-xj) > 1e-6 {
|
||||
return xi < xj
|
||||
}
|
||||
return sectionTop(result.Sections[i]) < sectionTop(result.Sections[j])
|
||||
})
|
||||
|
||||
threshold := medianW / 2
|
||||
// Correct same-page sections with nearly-same X0 but inverted Top.
|
||||
for i := len(result.Sections) - 1; i >= 1; i-- {
|
||||
for j := i - 1; j >= 0; j-- {
|
||||
if math.Abs(sectionX0(result.Sections[j+1])-sectionX0(result.Sections[j])) < threshold &&
|
||||
sectionTop(result.Sections[j+1]) < sectionTop(result.Sections[j]) &&
|
||||
sectionPage(result.Sections[j+1]) == sectionPage(result.Sections[j]) {
|
||||
result.Sections[j], result.Sections[j+1] = result.Sections[j+1], result.Sections[j]
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func sectionX0(s pdftype.Section) float64 {
|
||||
for _, p := range s.Positions {
|
||||
return p.Left
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func sectionTop(s pdftype.Section) float64 {
|
||||
for _, p := range s.Positions {
|
||||
return p.Top
|
||||
}
|
||||
return 0
|
||||
}
|
||||
434
internal/deepdoc/parser/pdf/post/post_steps_test.go
Normal file
434
internal/deepdoc/parser/pdf/post/post_steps_test.go
Normal file
@@ -0,0 +1,434 @@
|
||||
package post
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
pdftype "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// ── helpers ──────────────────────────────────────────────────────────────
|
||||
|
||||
// dummyBase64PNG is a valid 50×50 red pixel PNG, base64-encoded.
|
||||
const dummyBase64PNG = "iVBORw0KGgoAAAANSUhEUgAAADIAAAAyCAIAAACRXR/mAAAAUElEQVR4nOzOsREAEAAAMefsvzILaL6iSCbI2uNH83XgTqvQKrQKrUKr0Cq0Cq1Cq9AqtAqtQqvQKrQKrUKr0Cq0Cq1Cq9AqtAqt4gQAAP//miQBZqrF+JAAAAAASUVORK5CYII="
|
||||
|
||||
func newTestResult(sections ...pdftype.Section) *pdftype.ParseResult {
|
||||
return &pdftype.ParseResult{Sections: sections}
|
||||
}
|
||||
|
||||
func makePosSection(text string, page int, x0, x1, top, bottom float64) pdftype.Section {
|
||||
return pdftype.Section{
|
||||
Text: text,
|
||||
LayoutType: "text",
|
||||
Positions: []pdftype.Position{{PageNumbers: []int{page}, Left: x0, Right: x1, Top: top, Bottom: bottom}},
|
||||
}
|
||||
}
|
||||
|
||||
// ── normalizeLayoutType ────────────────────────────────────────────────
|
||||
|
||||
func TestNormalizeLayoutType(t *testing.T) {
|
||||
result := newTestResult(
|
||||
pdftype.Section{Text: "a", LayoutType: ""},
|
||||
pdftype.Section{Text: "b", LayoutType: " "},
|
||||
pdftype.Section{Text: "c", LayoutType: "table"},
|
||||
pdftype.Section{Text: "d", LayoutType: " figure "},
|
||||
pdftype.Section{Text: "e", LayoutType: "text"},
|
||||
)
|
||||
normalizeLayoutType(result)
|
||||
want := []string{"text", "text", "table", "figure", "text"}
|
||||
for i, s := range result.Sections {
|
||||
if s.LayoutType != want[i] {
|
||||
t.Errorf("Sections[%d]: got %q, want %q", i, s.LayoutType, want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── filterHeaderFooter ─────────────────────────────────────────────────
|
||||
|
||||
func TestFilterHeaderFooter(t *testing.T) {
|
||||
result := newTestResult(
|
||||
pdftype.Section{Text: "Page 1", LayoutType: "header"},
|
||||
pdftype.Section{Text: "Chapter 1", LayoutType: "text"},
|
||||
pdftype.Section{LayoutType: "footer"},
|
||||
pdftype.Section{LayoutType: "number"},
|
||||
pdftype.Section{Text: "Body", LayoutType: "text"},
|
||||
pdftype.Section{Text: "reference item", LayoutType: "reference"},
|
||||
)
|
||||
filterHeaderFooter(result)
|
||||
if len(result.Sections) != 2 {
|
||||
t.Fatalf("expected 2 sections, got %d: %+v", len(result.Sections), result.Sections)
|
||||
}
|
||||
if result.Sections[0].Text != "Chapter 1" || result.Sections[1].Text != "Body" {
|
||||
t.Errorf("wrong sections kept: %+v", result.Sections)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFilterHeaderFooter_Empty(t *testing.T) {
|
||||
result := newTestResult()
|
||||
filterHeaderFooter(result)
|
||||
if len(result.Sections) != 0 {
|
||||
t.Error("expected empty result")
|
||||
}
|
||||
}
|
||||
|
||||
// ── assignDocTypeKwd ───────────────────────────────────────────────────
|
||||
|
||||
func TestAssignDocTypeKwd_Normal(t *testing.T) {
|
||||
result := newTestResult(
|
||||
pdftype.Section{Text: "a", LayoutType: "table"},
|
||||
pdftype.Section{Text: "b", LayoutType: "figure"},
|
||||
pdftype.Section{Text: "c", LayoutType: "equation"},
|
||||
pdftype.Section{Text: "d", LayoutType: "", Image: dummyBase64PNG},
|
||||
pdftype.Section{Text: "e", LayoutType: "text"},
|
||||
pdftype.Section{Text: "f", LayoutType: ""},
|
||||
)
|
||||
assignDocTypeKwd(result, false)
|
||||
want := []string{"table", "image", "text", "image", "text", "text"}
|
||||
for i, s := range result.Sections {
|
||||
if s.DocTypeKwd != want[i] {
|
||||
t.Errorf("Sections[%d]: got %q, want %q", i, s.DocTypeKwd, want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssignDocTypeKwd_Flatten(t *testing.T) {
|
||||
result := newTestResult(
|
||||
pdftype.Section{Text: "a", LayoutType: "table", DocTypeKwd: "table", Image: dummyBase64PNG},
|
||||
pdftype.Section{Text: "b", LayoutType: "figure", DocTypeKwd: "image", Image: dummyBase64PNG},
|
||||
pdftype.Section{Text: "c", LayoutType: "text", DocTypeKwd: "text"},
|
||||
)
|
||||
assignDocTypeKwd(result, true)
|
||||
for _, s := range result.Sections {
|
||||
if s.DocTypeKwd != "text" {
|
||||
t.Errorf("expected all 'text', got %q", s.DocTypeKwd)
|
||||
}
|
||||
if s.Image != "" {
|
||||
t.Error("flatten should clear Image to prevent VLM enhancement")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── enhanceWithVision ──────────────────────────────────────────────────
|
||||
|
||||
func TestEnhanceWithVision_NoOp(t *testing.T) {
|
||||
result := newTestResult(
|
||||
pdftype.Section{Text: "original", Image: dummyBase64PNG, DocTypeKwd: "table"},
|
||||
)
|
||||
_ = enhanceWithVision(context.Background(), result, nil)
|
||||
if result.Sections[0].Text != "original" {
|
||||
t.Errorf("text changed when describer is nil: %q", result.Sections[0].Text)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEnhanceWithVision_Success(t *testing.T) {
|
||||
want := "A table showing Q1 revenue."
|
||||
desc := &mockImageDescriber{describe: want}
|
||||
|
||||
result := newTestResult(
|
||||
pdftype.Section{Text: "", Image: dummyBase64PNG, DocTypeKwd: "table"},
|
||||
)
|
||||
if err := enhanceWithVision(context.Background(), result, desc); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if result.Sections[0].Text != want {
|
||||
t.Errorf("text not enhanced: got %q", result.Sections[0].Text)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEnhanceWithVision_SkipText(t *testing.T) {
|
||||
desc := &mockImageDescriber{describe: "should not be called"}
|
||||
|
||||
result := newTestResult(
|
||||
pdftype.Section{Text: "plain text", DocTypeKwd: "text", Image: ""},
|
||||
)
|
||||
if err := enhanceWithVision(context.Background(), result, desc); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if result.Sections[0].Text != "plain text" {
|
||||
t.Errorf("text changed: %q", result.Sections[0].Text)
|
||||
}
|
||||
}
|
||||
|
||||
// ── removeTOCByOutlines ────────────────────────────────────────────────
|
||||
|
||||
func TestRemoveTOCByOutlines_Removes(t *testing.T) {
|
||||
outlines := []pdftype.Outline{
|
||||
{Title: "Chapter 1 Introduction", Level: 0, PageNumber: 1},
|
||||
{Title: "目录", Level: 0, PageNumber: 3},
|
||||
{Title: "Chapter 2 Methods", Level: 0, PageNumber: 5},
|
||||
}
|
||||
result := newTestResult(
|
||||
makePosSection("s1", 1, 50, 550, 100, 120),
|
||||
makePosSection("s2", 2, 50, 550, 100, 120),
|
||||
makePosSection("toc1", 3, 50, 550, 100, 120),
|
||||
makePosSection("toc2", 4, 50, 550, 100, 120),
|
||||
makePosSection("body1", 5, 50, 550, 100, 120),
|
||||
makePosSection("body2", 6, 50, 550, 100, 120),
|
||||
)
|
||||
removeTOCByOutlines(result, outlines)
|
||||
if len(result.Sections) != 4 {
|
||||
t.Fatalf("expected 4 sections, got %d", len(result.Sections))
|
||||
}
|
||||
if result.Sections[0].Text != "s1" || result.Sections[1].Text != "s2" {
|
||||
t.Error("pre-TOC pages should be kept")
|
||||
}
|
||||
if result.Sections[2].Text != "body1" || result.Sections[3].Text != "body2" {
|
||||
t.Error("post-TOC pages should be kept")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRemoveTOCByOutlines_NoMatch(t *testing.T) {
|
||||
outlines := []pdftype.Outline{
|
||||
{Title: "1. Introduction", Level: 0, PageNumber: 1},
|
||||
{Title: "2. Background", Level: 0, PageNumber: 3},
|
||||
}
|
||||
result := newTestResult(
|
||||
makePosSection("s1", 1, 50, 550, 100, 120),
|
||||
makePosSection("s2", 2, 50, 550, 100, 120),
|
||||
)
|
||||
removeTOCByOutlines(result, outlines)
|
||||
if len(result.Sections) != 2 {
|
||||
t.Errorf("expected 2 sections, got %d (no TOC should mean no removal)", len(result.Sections))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRemoveTOCByOutlines_NilOutlines(t *testing.T) {
|
||||
result := newTestResult(makePosSection("a", 1, 50, 550, 100, 120))
|
||||
removeTOCByOutlines(result, nil)
|
||||
if len(result.Sections) != 1 {
|
||||
t.Errorf("nil outlines should be no-op: got %d sections", len(result.Sections))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRemoveTOCByOutlines_EmptyOutlines(t *testing.T) {
|
||||
result := newTestResult(makePosSection("a", 1, 50, 550, 100, 120))
|
||||
removeTOCByOutlines(result, []pdftype.Outline{})
|
||||
if len(result.Sections) != 1 {
|
||||
t.Errorf("empty outlines should be no-op: got %d sections", len(result.Sections))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRemoveTOCByOutlines_NoNext(t *testing.T) {
|
||||
outlines := []pdftype.Outline{
|
||||
{Title: "目录", Level: 0, PageNumber: 2},
|
||||
}
|
||||
result := newTestResult(
|
||||
makePosSection("toc", 2, 50, 550, 100, 120),
|
||||
makePosSection("body", 3, 50, 550, 100, 120),
|
||||
)
|
||||
removeTOCByOutlines(result, outlines)
|
||||
if len(result.Sections) != 2 {
|
||||
t.Errorf("no next outline → keep all sections: got %d", len(result.Sections))
|
||||
}
|
||||
}
|
||||
|
||||
// ── reorderMultiColumn ─────────────────────────────────────────────────
|
||||
|
||||
func TestReorderMultiColumn_SingleCol(t *testing.T) {
|
||||
result := newTestResult(
|
||||
makePosSection("B", 0, 50, 550, 200, 220),
|
||||
makePosSection("A", 0, 50, 550, 100, 120),
|
||||
)
|
||||
reorderMultiColumn(result, 600.0, 1.0)
|
||||
// medianW=500 >= 300 → single col, order preserved
|
||||
if result.Sections[0].Text != "B" {
|
||||
t.Fatal("single column should preserve original order")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReorderMultiColumn_MultiCol(t *testing.T) {
|
||||
result := newTestResult(
|
||||
makePosSection("B", 0, 300, 500, 100, 120),
|
||||
makePosSection("A", 0, 50, 250, 100, 120),
|
||||
)
|
||||
reorderMultiColumn(result, 600.0, 1.0)
|
||||
if result.Sections[0].Positions[0].Left > result.Sections[1].Positions[0].Left {
|
||||
t.Log("multi-column: sections reordered")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReorderMultiColumn_Empty(t *testing.T) {
|
||||
result := newTestResult()
|
||||
reorderMultiColumn(result, 600.0, 1.0)
|
||||
if len(result.Sections) != 0 {
|
||||
t.Error("empty sections should remain empty")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReorderMultiColumn_NoText(t *testing.T) {
|
||||
result := newTestResult(
|
||||
pdftype.Section{Text: "t1", LayoutType: "table", Positions: []pdftype.Position{{PageNumbers: []int{0}, Left: 300, Right: 500, Top: 100, Bottom: 120}}},
|
||||
pdftype.Section{Text: "t2", LayoutType: "table", Positions: []pdftype.Position{{PageNumbers: []int{0}, Left: 50, Right: 250, Top: 100, Bottom: 120}}},
|
||||
)
|
||||
reorderMultiColumn(result, 600.0, 1.0)
|
||||
if len(result.Sections) != 2 {
|
||||
t.Fatal("expected 2 sections")
|
||||
}
|
||||
}
|
||||
|
||||
// ── PostProcess integration ────────────────────────────────────────────
|
||||
|
||||
func TestPostProcess_FullPipeline(t *testing.T) {
|
||||
// Simulates post-processing after Parse(): all features enabled.
|
||||
result := newTestResult(
|
||||
// Page 1: TOC — should be removed
|
||||
pdftype.Section{Text: "目录", LayoutType: "text", Positions: []pdftype.Position{{PageNumbers: []int{1}, Left: 50, Right: 550, Top: 100, Bottom: 120}}},
|
||||
pdftype.Section{Text: "Chapter 1 ... 1", LayoutType: "text", Positions: []pdftype.Position{{PageNumbers: []int{1}, Left: 50, Right: 550, Top: 120, Bottom: 140}}},
|
||||
// Page 1: header — should be removed
|
||||
pdftype.Section{Text: "Page 1", LayoutType: "header", Positions: []pdftype.Position{{PageNumbers: []int{1}, Left: 500, Right: 550, Top: 10, Bottom: 20}}},
|
||||
// Page 3: actual content
|
||||
pdftype.Section{Text: "Introduction text", LayoutType: "", Positions: []pdftype.Position{{PageNumbers: []int{3}, Left: 50, Right: 550, Top: 100, Bottom: 120}}},
|
||||
pdftype.Section{Text: "Row1 Col1 Row1 Col2", LayoutType: "table", Positions: []pdftype.Position{{PageNumbers: []int{3}, Left: 50, Right: 550, Top: 200, Bottom: 300}}, Image: dummyBase64PNG},
|
||||
pdftype.Section{Text: "Chart description", LayoutType: "figure", Positions: []pdftype.Position{{PageNumbers: []int{3}, Left: 50, Right: 550, Top: 300, Bottom: 400}}, Image: dummyBase64PNG},
|
||||
// Page 4: footer — should be removed
|
||||
pdftype.Section{Text: "Confidential", LayoutType: "footer", Positions: []pdftype.Position{{PageNumbers: []int{4}, Left: 50, Right: 550, Top: 700, Bottom: 720}}},
|
||||
)
|
||||
|
||||
outlines := []pdftype.Outline{
|
||||
{Title: "目录", Level: 0, PageNumber: 1},
|
||||
{Title: "Chapter 1 Introduction", Level: 0, PageNumber: 3},
|
||||
}
|
||||
|
||||
wantVLM := "This table shows quarterly revenue data with 2 columns."
|
||||
describer := &mockImageDescriber{describe: wantVLM}
|
||||
|
||||
// First pass: non-VLM steps through PostProcess
|
||||
config := PipelineConfig{
|
||||
ConfigKeyPageWidth: 600.0,
|
||||
ConfigKeyZoom: 1.0,
|
||||
ConfigKeyOutlines: outlines,
|
||||
ConfigKeyRemoveTOC: true,
|
||||
}
|
||||
if err := PostProcess(context.Background(), result, config); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// Then: VLM enhancement through internal function (with mock)
|
||||
if err := enhanceWithVision(context.Background(), result, describer); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// Then: flatten
|
||||
if err := PostProcess(context.Background(), result, PipelineConfig{
|
||||
ConfigKeyFlattenMediaToText: true,
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Verify
|
||||
if len(result.Sections) != 3 {
|
||||
t.Fatalf("expected 3 sections after filtering, got %d: %+v", len(result.Sections), result.Sections)
|
||||
}
|
||||
for i, s := range result.Sections {
|
||||
if s.DocTypeKwd != "text" {
|
||||
t.Errorf("section[%d] DocTypeKwd = %q, want 'text'", i, s.DocTypeKwd)
|
||||
}
|
||||
if s.LayoutType == "header" || s.LayoutType == "footer" {
|
||||
t.Errorf("section[%d] LayoutType = %q, should have been filtered out", i, s.LayoutType)
|
||||
}
|
||||
}
|
||||
// Table section should have enhanced text
|
||||
found := false
|
||||
for _, s := range result.Sections {
|
||||
if s.LayoutType == "table" {
|
||||
found = true
|
||||
if s.Text != "Row1 Col1 Row1 Col2\n"+wantVLM {
|
||||
t.Errorf("table text not enhanced: %q", s.Text)
|
||||
}
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Error("table section missing from result")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPostProcess_Minimal(t *testing.T) {
|
||||
result := newTestResult(
|
||||
pdftype.Section{Text: "Hello", LayoutType: ""},
|
||||
pdftype.Section{Text: "World", LayoutType: " "},
|
||||
)
|
||||
if err := PostProcess(context.Background(), result, nil); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(result.Sections) != 2 {
|
||||
t.Fatalf("expected 2 sections, got %d", len(result.Sections))
|
||||
}
|
||||
if result.Sections[0].LayoutType != "text" || result.Sections[1].LayoutType != "text" {
|
||||
t.Error("layout not normalized")
|
||||
}
|
||||
if result.Sections[0].DocTypeKwd != "text" || result.Sections[1].DocTypeKwd != "text" {
|
||||
t.Error("doc_type_kwd not assigned")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPostProcess_NilResult(t *testing.T) {
|
||||
if err := PostProcess(context.Background(), nil, nil); err == nil {
|
||||
t.Error("expected error for nil result")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPostProcess_EmptySections(t *testing.T) {
|
||||
result := newTestResult()
|
||||
if err := PostProcess(context.Background(), result, nil); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(result.Sections) != 0 {
|
||||
t.Error("empty should remain empty")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPostProcess_FiguresLazy(t *testing.T) {
|
||||
result := newTestResult(
|
||||
pdftype.Section{Text: "Fig1", LayoutType: "figure"},
|
||||
pdftype.Section{Text: "Body", LayoutType: "text"},
|
||||
pdftype.Section{Text: "Fig2", LayoutType: "figure"},
|
||||
)
|
||||
if err := PostProcess(context.Background(), result, nil); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
figs := result.Figures()
|
||||
if len(figs) != 2 {
|
||||
t.Fatalf("expected 2 figures, got %d", len(figs))
|
||||
}
|
||||
if figs[0].Text != "Fig1" || figs[1].Text != "Fig2" {
|
||||
t.Errorf("wrong figures: %+v", figs)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPostProcess_FilterOnly(t *testing.T) {
|
||||
result := newTestResult(
|
||||
pdftype.Section{Text: "Header", LayoutType: "header"},
|
||||
pdftype.Section{Text: "Second", LayoutType: "text"},
|
||||
pdftype.Section{Text: "First", LayoutType: "text"},
|
||||
)
|
||||
if err := PostProcess(context.Background(), result, nil); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(result.Sections) != 2 {
|
||||
t.Fatalf("expected 2 sections after filtering, got %d", len(result.Sections))
|
||||
}
|
||||
figs := result.Figures()
|
||||
if len(figs) != 0 {
|
||||
t.Errorf("expected 0 figures, got %d", len(figs))
|
||||
}
|
||||
}
|
||||
|
||||
func TestPostProcess_ReorderOnly(t *testing.T) {
|
||||
result := newTestResult(
|
||||
makePosSection("B", 0, 300, 500, 100, 120),
|
||||
makePosSection("A", 0, 50, 250, 100, 120),
|
||||
)
|
||||
config := PipelineConfig{
|
||||
ConfigKeyPageWidth: 600.0,
|
||||
ConfigKeyZoom: 1.0,
|
||||
}
|
||||
// Remove the outlines key since we don't need it
|
||||
if err := PostProcess(context.Background(), result, config); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(result.Sections) != 2 {
|
||||
t.Fatal("expected 2 sections")
|
||||
}
|
||||
// Should be reordered: col 1 leftmost: A then B
|
||||
if result.Sections[0].Positions[0].Left > result.Sections[1].Positions[0].Left {
|
||||
t.Log("multi-column: sections reordered left-to-right")
|
||||
}
|
||||
}
|
||||
98
internal/deepdoc/parser/pdf/post/vision_describe.go
Normal file
98
internal/deepdoc/parser/pdf/post/vision_describe.go
Normal file
@@ -0,0 +1,98 @@
|
||||
package post
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"image"
|
||||
)
|
||||
|
||||
// ImageDescriber describes an image using a vision language model.
|
||||
type ImageDescriber interface {
|
||||
DescribeImage(ctx context.Context, img image.Image, prompt string) (string, error)
|
||||
}
|
||||
|
||||
// maxDescribeConcurrency limits how many concurrent VLM calls are in flight.
|
||||
const maxDescribeConcurrency = 10
|
||||
|
||||
// minImageSide is the minimum width or height (in pixels) for an image
|
||||
// to be sent to a VLM. Tiny crops fail provider image-size limits.
|
||||
const minImageSide = 11
|
||||
|
||||
// describePrompt is the default prompt for image/table description.
|
||||
// Python: vision_llm_figure_describe_prompt.md
|
||||
const describePrompt = `## ROLE
|
||||
|
||||
You are an expert visual data analyst.
|
||||
|
||||
## GOAL
|
||||
|
||||
Analyze the image and produce a textual representation strictly based on what is visible in the image.
|
||||
|
||||
## DECISION RULE (CRITICAL)
|
||||
|
||||
First, determine whether the image contains an explicit visual data representation with enumerable data units forming a coherent dataset.
|
||||
|
||||
## OUTPUT RULES (STRICT)
|
||||
|
||||
- Produce output in exactly one of the two modes defined below.
|
||||
- Do NOT mention, label, or reference the modes in the output.
|
||||
- Do NOT combine content from both modes.
|
||||
- Do NOT explain or justify the choice of mode.
|
||||
- Do NOT add any headings, titles, or commentary beyond what the mode requires.
|
||||
|
||||
---
|
||||
|
||||
## MODE 1: STRUCTURED VISUAL DATA OUTPUT
|
||||
|
||||
(Use only if the image contains enumerable data units forming a coherent dataset.)
|
||||
|
||||
Output only the following fields, in list form:
|
||||
- Visual Type:
|
||||
- Title:
|
||||
- Axes / Legends / Labels:
|
||||
- Data Points:
|
||||
- Captions / Annotations:
|
||||
|
||||
---
|
||||
|
||||
## MODE 2: GENERAL FIGURE CONTENT
|
||||
|
||||
(Use only if the image does NOT contain enumerable data units.)
|
||||
|
||||
Write the content directly, starting from the first sentence.
|
||||
Do NOT add any introductory labels, titles, headings, or prefixes.
|
||||
|
||||
Requirements:
|
||||
- Describe visible regions and components in a stable order (e.g., top-to-bottom, left-to-right).
|
||||
- Explicitly name interface elements or visual objects exactly as they appear.
|
||||
- Transcribe all visible text verbatim; do not paraphrase, summarize, or reinterpret labels.
|
||||
- Describe spatial grouping, containment, and alignment of elements.
|
||||
- Do NOT interpret intent, behavior, workflows, gameplay rules, or processes.
|
||||
- Avoid narrative or stylistic language unless it is a dominant and functional visual element.
|
||||
|
||||
Use concise, information-dense sentences.
|
||||
Do not use bullet lists or structured fields in this mode.`
|
||||
|
||||
// DescribeImage calls the VLM to produce a natural-language description of
|
||||
// the given image. Returns the description text or an error.
|
||||
//
|
||||
// Images smaller than minImageSide in either dimension are silently skipped
|
||||
// (returning an empty string and no error), matching Python's behavior.
|
||||
func DescribeImage(ctx context.Context, img image.Image, prompt string, client ImageDescriber) (string, error) {
|
||||
if img == nil {
|
||||
return "", errors.New("DescribeImage: nil image")
|
||||
}
|
||||
b := img.Bounds()
|
||||
if b.Dx() == 0 || b.Dy() == 0 {
|
||||
return "", errors.New("DescribeImage: empty image (0x0)")
|
||||
}
|
||||
if b.Dx() < minImageSide || b.Dy() < minImageSide {
|
||||
return "", nil // skip tiny crops, Python compatible
|
||||
}
|
||||
|
||||
if err := ctx.Err(); err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
return client.DescribeImage(ctx, img, prompt)
|
||||
}
|
||||
112
internal/deepdoc/parser/pdf/post/vision_describe_test.go
Normal file
112
internal/deepdoc/parser/pdf/post/vision_describe_test.go
Normal file
@@ -0,0 +1,112 @@
|
||||
package post
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"image"
|
||||
"image/color"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// ── mock image describer ───────────────────────────────────────────────
|
||||
|
||||
type mockImageDescriber struct {
|
||||
describe string
|
||||
err error
|
||||
}
|
||||
|
||||
func (m *mockImageDescriber) DescribeImage(_ context.Context, _ image.Image, _ string) (string, error) {
|
||||
return m.describe, m.err
|
||||
}
|
||||
|
||||
// ── DescribeImage tests ────────────────────────────────────────────────
|
||||
|
||||
func TestDescribeImage_Success(t *testing.T) {
|
||||
img := newTestImage(100, 100)
|
||||
want := "This is a bar chart showing quarterly revenue."
|
||||
client := &mockImageDescriber{describe: want}
|
||||
|
||||
got, err := DescribeImage(context.Background(), img, "Describe this image", client)
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
if got != want {
|
||||
t.Errorf("DescribeImage() = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDescribeImage_VLMError(t *testing.T) {
|
||||
img := newTestImage(100, 100)
|
||||
client := &mockImageDescriber{err: errors.New("VLM timeout")}
|
||||
|
||||
got, err := DescribeImage(context.Background(), img, "Describe this image", client)
|
||||
if err == nil {
|
||||
t.Fatal("expected error, got nil")
|
||||
}
|
||||
if got != "" {
|
||||
t.Errorf("expected empty string on error, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDescribeImage_CanceledContext(t *testing.T) {
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel() // cancel immediately
|
||||
img := newTestImage(100, 100)
|
||||
client := &mockImageDescriber{describe: "should not be reached"}
|
||||
|
||||
got, err := DescribeImage(ctx, img, "prompt", client)
|
||||
if err == nil {
|
||||
t.Fatal("expected context error, got nil")
|
||||
}
|
||||
if got != "" {
|
||||
t.Errorf("expected empty string, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDescribeImage_NilImage(t *testing.T) {
|
||||
client := &mockImageDescriber{describe: "should not be reached"}
|
||||
|
||||
got, err := DescribeImage(context.Background(), nil, "prompt", client)
|
||||
if err == nil {
|
||||
t.Fatal("expected error for nil image, got nil")
|
||||
}
|
||||
if got != "" {
|
||||
t.Errorf("expected empty string, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDescribeImage_EmptyImage(t *testing.T) {
|
||||
img := newTestImage(0, 0)
|
||||
client := &mockImageDescriber{describe: "should not be reached"}
|
||||
|
||||
_, err := DescribeImage(context.Background(), img, "prompt", client)
|
||||
if err == nil {
|
||||
t.Fatal("expected error for empty image, got nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDescribeImage_TinyImage(t *testing.T) {
|
||||
img := newTestImage(5, 5) // below minSide=11
|
||||
client := &mockImageDescriber{describe: "should not be reached"}
|
||||
|
||||
got, err := DescribeImage(context.Background(), img, "prompt", client)
|
||||
if err != nil {
|
||||
t.Fatal("tiny images should be silently skipped, not error")
|
||||
}
|
||||
if got != "" {
|
||||
t.Errorf("expected empty string for tiny image, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// ── helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
func newTestImage(w, h int) image.Image {
|
||||
img := image.NewRGBA(image.Rect(0, 0, w, h))
|
||||
// Fill with a recognizable pattern.
|
||||
for y := 0; y < h; y++ {
|
||||
for x := 0; x < w; x++ {
|
||||
img.Set(x, y, color.RGBA{R: uint8(x % 256), G: uint8(y % 256), B: 128, A: 255})
|
||||
}
|
||||
}
|
||||
return img
|
||||
}
|
||||
@@ -2,6 +2,7 @@ package parser
|
||||
|
||||
import (
|
||||
"image"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"reflect"
|
||||
)
|
||||
|
||||
@@ -12,13 +13,13 @@ import (
|
||||
var renderFn = fallbackRender
|
||||
|
||||
// renderPageToImage renders a page at 216 DPI for downstream DLA/TSR/OCR.
|
||||
func renderPageToImage(engine PDFEngine, pageNum int) (image.Image, error) {
|
||||
func renderPageToImage(engine pdf.PDFEngine, pageNum int) (image.Image, error) {
|
||||
return renderFn(engine, pageNum)
|
||||
}
|
||||
|
||||
// fallbackRender uses the engine's own RenderPageImage (no C dependency).
|
||||
func fallbackRender(engine PDFEngine, pageNum int) (image.Image, error) {
|
||||
img, err := engine.RenderPageImage(pageNum, dlaDPI)
|
||||
func fallbackRender(engine pdf.PDFEngine, pageNum int) (image.Image, error) {
|
||||
img, err := engine.RenderPageImage(pageNum, pdf.DlaDPI)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
@@ -6,12 +6,13 @@ import (
|
||||
"image"
|
||||
|
||||
"ragflow/internal/deepdoc/parser/pdf/pdfium"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// pdfiumRender uses the pdfium C library for higher-quality rasterisation
|
||||
// (AA, hinting) which is essential for downstream OCR/DLA accuracy on
|
||||
// scanned or low-quality PDFs.
|
||||
func pdfiumRender(engine PDFEngine, pageNum int) (image.Image, error) {
|
||||
func pdfiumRender(engine pdf.PDFEngine, pageNum int) (image.Image, error) {
|
||||
raw := engine.RawData()
|
||||
if raw == nil {
|
||||
// PythonCharEngine and mocks don't carry PDF bytes —
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
//go:build cgo
|
||||
//go:build cgo && manual
|
||||
|
||||
package parser
|
||||
|
||||
@@ -10,15 +10,16 @@ import (
|
||||
"sort"
|
||||
"testing"
|
||||
|
||||
lyt "ragflow/internal/deepdoc/parser/pdf/layout"
|
||||
"ragflow/internal/deepdoc/parser/pdf/pdfium"
|
||||
"ragflow/internal/deepdoc/parser/pdf/pdfoxide"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// ── helpers ──────────────────────────────────────────────────────────────
|
||||
|
||||
// pdfiumPtSize returns post-rotation page dimensions via pdfium.
|
||||
// pdfiumPtSize returns post-rotation page dimensions via pdfium.
|
||||
func pdfiumPtSize(eng PDFEngine, file string, t *testing.T) (w, h float64) {
|
||||
func pdfiumPtSize(eng pdf.PDFEngine, file string, t *testing.T) (w, h float64) {
|
||||
t.Helper()
|
||||
raw := eng.RawData()
|
||||
if raw == nil {
|
||||
@@ -38,7 +39,7 @@ func pdfiumPtSize(eng PDFEngine, file string, t *testing.T) (w, h float64) {
|
||||
// openPDF reads a PDF fixture from dir/name, opens it via pdfoxide, and
|
||||
// returns both the engine and document. The document is closed via t.Cleanup.
|
||||
// Missing or corrupt fixtures cause a hard failure (t.Fatal).
|
||||
func openPDF(t *testing.T, dir, name string) (PDFEngine, *pdfoxide.Document) {
|
||||
func openPDF(t *testing.T, dir, name string) (pdf.PDFEngine, *pdfoxide.Document) {
|
||||
t.Helper()
|
||||
data, err := os.ReadFile(filepath.Join(dir, name))
|
||||
if err != nil {
|
||||
@@ -56,7 +57,7 @@ func openPDF(t *testing.T, dir, name string) (PDFEngine, *pdfoxide.Document) {
|
||||
return eng, doc
|
||||
}
|
||||
|
||||
func openRotatePDF(t *testing.T, name string) (PDFEngine, *pdfoxide.Document) {
|
||||
func openRotatePDF(t *testing.T, name string) (pdf.PDFEngine, *pdfoxide.Document) {
|
||||
t.Helper()
|
||||
return openPDF(t, "testdata/pdfs", name)
|
||||
}
|
||||
@@ -180,7 +181,7 @@ func TestRotation_SameLinePreserved(t *testing.T) {
|
||||
tolerance = 15.0 // char widths vary ~10-13pts on same line
|
||||
}
|
||||
|
||||
lines := groupCharsToLines(chars, false)
|
||||
lines := lyt.GroupCharsToLines(chars, false)
|
||||
violations := 0
|
||||
for li, line := range lines {
|
||||
if len(line) <= 1 {
|
||||
@@ -330,7 +331,7 @@ func TestRotation_RenderAlignment(t *testing.T) {
|
||||
const dpi = 216.0
|
||||
const scale = dpi / 72.0
|
||||
|
||||
identityMap := func(c TextChar, _, _ float64) (px0, py0, px1, py1 int) {
|
||||
identityMap := func(c pdf.TextChar, _, _ float64) (px0, py0, px1, py1 int) {
|
||||
return int(math.Round(c.X0 * scale)),
|
||||
int(math.Round(c.Top * scale)),
|
||||
int(math.Round(c.X1 * scale)),
|
||||
@@ -568,7 +569,7 @@ func TestRotation_DocumentPageSize(t *testing.T) {
|
||||
|
||||
// ── bboxDarkPixelHitRate helper ─────────────────────────────────────────
|
||||
|
||||
func bboxDarkPixelHitRate(t *testing.T, chars []TextChar, img *image.RGBA, scale float64) (hit, checked int) {
|
||||
func bboxDarkPixelHitRate(t *testing.T, chars []pdf.TextChar, img *image.RGBA, scale float64) (hit, checked int) {
|
||||
t.Helper()
|
||||
imgW, imgH := img.Bounds().Dx(), img.Bounds().Dy()
|
||||
n, step := len(chars), max(1, len(chars)/min(50, len(chars)))
|
||||
|
||||
@@ -1,153 +0,0 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"image"
|
||||
"regexp"
|
||||
"sort"
|
||||
)
|
||||
|
||||
// SaaS model label taxonomies.
|
||||
// DLA: 10 classes with duplicates (matching SaaS Docker TSR endpoint).
|
||||
var saasDLALabels = []string{
|
||||
LayoutTypeTitle, LayoutTypeText, LayoutTypeReference,
|
||||
LayoutTypeFigure, DLALabelFigureCaption,
|
||||
LayoutTypeTable, DLALabelTableCaption, DLALabelTableCaption,
|
||||
LayoutTypeEquation, DLALabelFigureCaption,
|
||||
}
|
||||
|
||||
// TSR: 2-class separator lines (v=vertical, h=horizontal).
|
||||
var saasTSRLabels = []string{"v", "h"}
|
||||
|
||||
// DeepDoc label regexes — compiled once at package init.
|
||||
// These match the TSR label taxonomy returned by the Python DeepDoc
|
||||
// table structure recognition service.
|
||||
var (
|
||||
reHeader = regexp.MustCompile(`.*header$`)
|
||||
reRowHdr = regexp.MustCompile(`table$|.* (row|header)`)
|
||||
// "table$" catches the default TSR label "table" (class 0), matching
|
||||
// Python's behavior which uses all cells regardless of label.
|
||||
reSpan = regexp.MustCompile(`.*spanning`)
|
||||
reColumn = regexp.MustCompile(`table column$`)
|
||||
)
|
||||
|
||||
// gatherTSR filters cells by label regex pattern.
|
||||
func gatherTSR(cells []TSRCell, re *regexp.Regexp) []TSRCell {
|
||||
var result []TSRCell
|
||||
for _, c := range cells {
|
||||
if re.MatchString(c.Label) {
|
||||
result = append(result, c)
|
||||
}
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// SaasDeepDocService implements TableBuilder and DocAnalyzer using the
|
||||
// Python DeepDoc TSR service.
|
||||
type SaasDeepDocService struct {
|
||||
doc DocAnalyzer
|
||||
}
|
||||
|
||||
// NewSaasDeepDocService creates a service backed by the SaaS DeepDoc service.
|
||||
// If doc is a *DeepDocClient, its DLALabels/TSRLabels are set to the SaaS
|
||||
// taxonomy.
|
||||
func NewSaasDeepDocService(doc DocAnalyzer) *SaasDeepDocService {
|
||||
if c, ok := doc.(*DeepDocClient); ok {
|
||||
c.DLALabels = saasDLALabels
|
||||
c.TSRLabels = saasTSRLabels
|
||||
}
|
||||
return &SaasDeepDocService{doc: doc}
|
||||
}
|
||||
|
||||
func (b *SaasDeepDocService) Name() string { return "deepdoc" }
|
||||
|
||||
func (b *SaasDeepDocService) DetectCells(ctx context.Context, cropped image.Image) ([]TSRCell, error) {
|
||||
return b.doc.TSR(ctx, cropped)
|
||||
}
|
||||
|
||||
func (b *SaasDeepDocService) GroupCells(cells []TSRCell) [][]TSRCell {
|
||||
return groupTSRCellsToRowsLabeled(cells)
|
||||
}
|
||||
|
||||
// groupTSRCellsToRowsLabeled groups TSR cells into rows using labels
|
||||
// (header, row, spanning) instead of just Y proximity. Matching Python's
|
||||
// gather-based approach.
|
||||
func groupTSRCellsToRowsLabeled(cells []TSRCell) [][]TSRCell {
|
||||
rows := gatherTSR(cells, reRowHdr)
|
||||
spans := gatherTSR(cells, reSpan)
|
||||
clmns := gatherTSR(cells, reColumn)
|
||||
|
||||
if len(rows) == 0 && len(spans) == 0 {
|
||||
return groupTSRCellsToRows(cells)
|
||||
}
|
||||
|
||||
sortYFirstly(rows, 10)
|
||||
sortXFirstly(clmns, 10)
|
||||
|
||||
var grouped [][]TSRCell
|
||||
var curRow []TSRCell
|
||||
curY := 0.0
|
||||
rowThreshold := 0.0
|
||||
if len(rows) > 0 {
|
||||
heights := make([]float64, len(rows))
|
||||
for i, r := range rows {
|
||||
heights[i] = r.Y1 - r.Y0
|
||||
}
|
||||
sort.Float64s(heights)
|
||||
rowThreshold = heights[len(heights)/2] * 0.5
|
||||
if rowThreshold <= 0 {
|
||||
rowThreshold = 10
|
||||
}
|
||||
}
|
||||
|
||||
for _, c := range rows {
|
||||
if len(curRow) == 0 {
|
||||
curRow = append(curRow, c)
|
||||
curY = c.Y0
|
||||
continue
|
||||
}
|
||||
if c.Y0-curY > rowThreshold {
|
||||
grouped = append(grouped, curRow)
|
||||
curRow = []TSRCell{c}
|
||||
curY = c.Y0
|
||||
} else {
|
||||
curRow = append(curRow, c)
|
||||
}
|
||||
}
|
||||
if len(curRow) > 0 {
|
||||
grouped = append(grouped, curRow)
|
||||
}
|
||||
|
||||
for _, s := range spans {
|
||||
for ri, row := range grouped {
|
||||
if len(row) > 0 && s.Y0 <= row[0].Y1 && s.Y1 >= row[0].Y0 {
|
||||
grouped[ri] = append(grouped[ri], s)
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for _, row := range grouped {
|
||||
sortXFirstly(row, 10)
|
||||
}
|
||||
|
||||
maxCols := 0
|
||||
for _, row := range grouped {
|
||||
if len(row) > maxCols {
|
||||
maxCols = len(row)
|
||||
}
|
||||
}
|
||||
for i := range grouped {
|
||||
if len(grouped[i]) == 0 {
|
||||
continue // no real cells → cannot derive valid coordinates for padding
|
||||
}
|
||||
for len(grouped[i]) < maxCols {
|
||||
lastX := grouped[i][len(grouped[i])-1].X1 + 10
|
||||
rowY0 := grouped[i][0].Y0
|
||||
rowY1 := grouped[i][0].Y1
|
||||
grouped[i] = append(grouped[i], TSRCell{X0: lastX, X1: lastX + 1, Y0: rowY0, Y1: rowY1})
|
||||
}
|
||||
}
|
||||
|
||||
return grouped
|
||||
}
|
||||
@@ -1,111 +0,0 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestSaasDeepDocService_GroupCells(t *testing.T) {
|
||||
b := &SaasDeepDocService{}
|
||||
|
||||
t.Run("labels group into rows", func(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "H1", Label: "table column header"},
|
||||
{X0: 100, Y0: 0, X1: 200, Y1: 30, Text: "H2", Label: "table column header"},
|
||||
{X0: 0, Y0: 35, X1: 100, Y1: 65, Text: "A1", Label: "table row"},
|
||||
{X0: 100, Y0: 35, X1: 200, Y1: 65, Text: "B1", Label: "table row"},
|
||||
{X0: 0, Y0: 70, X1: 100, Y1: 100, Text: "A2", Label: "table row"},
|
||||
{X0: 100, Y0: 70, X1: 200, Y1: 100, Text: "B2", Label: "table row"},
|
||||
}
|
||||
grid := b.GroupCells(cells)
|
||||
if len(grid) != 3 {
|
||||
t.Fatalf("expected 3 rows, got %d", len(grid))
|
||||
}
|
||||
if len(grid[0]) != 2 || len(grid[1]) != 2 || len(grid[2]) != 2 {
|
||||
t.Errorf("expected 2 cols per row, got %d/%d/%d",
|
||||
len(grid[0]), len(grid[1]), len(grid[2]))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("spanning cell added to row", func(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 200, Y1: 30, Text: "H1", Label: "table column header"},
|
||||
{X0: 100, Y0: 0, X1: 200, Y1: 30, Text: "H2", Label: "table column header"},
|
||||
{X0: 0, Y0: 0, X1: 200, Y1: 30, Text: "Span", Label: "table spanning cell"},
|
||||
{X0: 0, Y0: 35, X1: 100, Y1: 65, Text: "D1", Label: "table row"},
|
||||
{X0: 100, Y0: 35, X1: 200, Y1: 65, Text: "D2", Label: "table row"},
|
||||
}
|
||||
grid := b.GroupCells(cells)
|
||||
if len(grid) != 2 {
|
||||
t.Fatalf("expected 2 rows (header + data), got %d", len(grid))
|
||||
}
|
||||
if len(grid[0]) < 3 {
|
||||
t.Errorf("expected row 0 to contain 2 headers + spanning = 3 cells, got %d", len(grid[0]))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("fallback to Y-proximity when no labels match", func(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "C1", Label: "unknown"},
|
||||
{X0: 100, Y0: 0, X1: 200, Y1: 30, Text: "C2", Label: "unknown"},
|
||||
{X0: 0, Y0: 50, X1: 100, Y1: 80, Text: "D1", Label: "unknown"},
|
||||
{X0: 100, Y0: 50, X1: 200, Y1: 80, Text: "D2", Label: "unknown"},
|
||||
}
|
||||
grid := b.GroupCells(cells)
|
||||
if len(grid) != 2 {
|
||||
t.Fatalf("expected 2 rows from Y-proximity fallback, got %d", len(grid))
|
||||
}
|
||||
if len(grid[0]) != 2 || len(grid[1]) != 2 {
|
||||
t.Errorf("expected 2 cols per row, got %d/%d", len(grid[0]), len(grid[1]))
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestSaasDeepDocService_Name(t *testing.T) {
|
||||
b := &SaasDeepDocService{}
|
||||
if b.Name() != "deepdoc" {
|
||||
t.Errorf("expected 'deepdoc', got %q", b.Name())
|
||||
}
|
||||
}
|
||||
|
||||
func TestGatherTSR(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
{Label: "table row", Text: "A"},
|
||||
{Label: "table column header", Text: "H"},
|
||||
{Label: "table row", Text: "B"},
|
||||
}
|
||||
result := gatherTSR(cells, reRowHdr)
|
||||
if len(result) < 2 {
|
||||
t.Errorf("expected at least 2 matching cells, got %d", len(result))
|
||||
}
|
||||
for _, c := range result {
|
||||
if !strings.Contains("ABH", c.Text[:1]) {
|
||||
t.Errorf("unexpected cell in result: %+v", c)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestGroupTSRCellsToRowsLabeled_NoZeroHeightPhantomCells(t *testing.T) {
|
||||
// Row0: 1 row cell + 1 spanning cell → 2 cells.
|
||||
// Row1: 1 row cell → 1 cell. maxCols=2 → Row1 padded.
|
||||
// The padded cell must have valid height from the real cell.
|
||||
cells := []TSRCell{
|
||||
{Label: "table row", X0: 0, Y0: 0, X1: 100, Y1: 20},
|
||||
{Label: "table spanning cell", X0: 120, Y0: 0, X1: 200, Y1: 20},
|
||||
{Label: "table row", X0: 0, Y0: 100, X1: 100, Y1: 120},
|
||||
}
|
||||
result := groupTSRCellsToRowsLabeled(cells)
|
||||
if len(result) != 2 {
|
||||
t.Fatalf("expected 2 rows, got %d", len(result))
|
||||
}
|
||||
if len(result[0]) != 2 {
|
||||
t.Fatalf("row 0: expected 2 cells, got %d", len(result[0]))
|
||||
}
|
||||
if len(result[1]) != 2 {
|
||||
t.Fatalf("row 1: expected 2 cells (padded), got %d", len(result[1]))
|
||||
}
|
||||
phantom := result[1][1]
|
||||
if phantom.Y1 <= phantom.Y0 {
|
||||
t.Errorf("phantom cell has zero height: Y0=%v Y1=%v", phantom.Y0, phantom.Y1)
|
||||
}
|
||||
}
|
||||
@@ -7,52 +7,18 @@ import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// mustConnectOssDeepDoc returns a DeepDocClient pointed at the OSS service.
|
||||
func mustConnectOssDeepDoc(t *testing.T) *DeepDocClient {
|
||||
t.Helper()
|
||||
url := os.Getenv("OSSDEEPDOC_URL")
|
||||
if url == "" {
|
||||
url = "http://localhost:9390"
|
||||
}
|
||||
client, err := NewDeepDocClient(url)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !client.Health() {
|
||||
t.Fatalf("OssDeepDoc not available at %s", url)
|
||||
}
|
||||
if client.ModelType() != ModelOSS {
|
||||
t.Skipf("DeepDoc at %s is %q, not oss — skipping OSS-specific test", url, client.ModelType())
|
||||
}
|
||||
return client
|
||||
}
|
||||
|
||||
// mustOpenEngine opens a PDF from testdata/pdfs/ and returns a PDFEngine.
|
||||
func mustOpenEngine(t *testing.T, name string) PDFEngine {
|
||||
t.Helper()
|
||||
pdfPath := filepath.Join("testdata", "pdfs", name)
|
||||
data, err := os.ReadFile(pdfPath)
|
||||
if err != nil {
|
||||
t.Fatalf("read fixture %s: %v", name, err)
|
||||
}
|
||||
eng, err := NewEngine(data)
|
||||
if err != nil {
|
||||
t.Fatalf("open engine %s: %v", name, err)
|
||||
}
|
||||
return eng
|
||||
}
|
||||
|
||||
// TestScanAllPDFs iterates over all PDFs in testdata/pdfs/, parses each
|
||||
// with OssDeepDoc TSR, and prints a summary. Run with:
|
||||
//
|
||||
// CGO_ENABLED=1 CGO_LDFLAGS="..." go test -tags=manual -run TestScanAllPDFs -v -count=1
|
||||
func TestScanAllPDFs(t *testing.T) {
|
||||
client := mustConnectOssDeepDoc(t)
|
||||
client := mustConnectInferenceClient(t)
|
||||
|
||||
pdfDir := filepath.Join("testdata", "pdfs")
|
||||
entries, err := os.ReadDir(pdfDir)
|
||||
@@ -76,8 +42,8 @@ func TestScanAllPDFs(t *testing.T) {
|
||||
fmt.Printf("\n── %s %s\n", name, strings.Repeat("─", maxint(1, 68-len(name))))
|
||||
|
||||
eng := mustOpenEngine(t, name)
|
||||
cfg := DefaultParserConfig()
|
||||
cfg.TableBuilder = NewOssDeepDocService(client)
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
cfg.TableBuilder = NewDeepDocTableBuildService(client)
|
||||
p := NewParser(cfg, client)
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
eng.Close()
|
||||
|
||||
@@ -8,6 +8,8 @@ import (
|
||||
"math"
|
||||
"os"
|
||||
"path/filepath"
|
||||
lyt "ragflow/internal/deepdoc/parser/pdf/layout"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
@@ -42,12 +44,12 @@ func TestSnapshotStageComparison(t *testing.T) {
|
||||
t.Logf(" Sample boxes (page 0): %d", len(s1.SampleBoxesPage0))
|
||||
t.Logf(" Text merge: %d -> %d boxes", s4.BoxesBefore, s4.BoxesAfter)
|
||||
|
||||
// Convert sample boxes to Go TextBox format
|
||||
// Convert sample boxes to Go pdf.TextBox format
|
||||
goBoxes := snapshotBoxesToGo(s1.SampleBoxesPage0)
|
||||
|
||||
// Run Go TextMerge with default params
|
||||
meanH := map[int]float64{0: avg(s1.MeanHeight)}
|
||||
merged := TextMerge(goBoxes, meanH, 3)
|
||||
merged := lyt.TextMerge(goBoxes, meanH, 3)
|
||||
|
||||
// Compare counts
|
||||
if len(merged) > 0 {
|
||||
@@ -59,7 +61,7 @@ func TestSnapshotStageComparison(t *testing.T) {
|
||||
|
||||
// Run Go NaiveVerticalMerge
|
||||
meanW := map[int]float64{0: avg(s1.MeanWidth)}
|
||||
vm := NaiveVerticalMerge(merged, meanH, meanW, s1.IsEnglish)
|
||||
vm := lyt.NaiveVerticalMerge(merged, meanH, meanW, s1.IsEnglish)
|
||||
if s6, ok := snap.Stages["_naive_vertical_merge"]; ok {
|
||||
t.Logf(" Go VerticalMerge: %d -> %d boxes (Python: %d->%d)",
|
||||
len(merged), len(vm), s6.BoxesBefore, s6.BoxesAfter)
|
||||
@@ -73,7 +75,7 @@ func TestSnapshotStageComparison(t *testing.T) {
|
||||
}
|
||||
|
||||
// Run Go boxesToSections
|
||||
sections := boxesToSections(vm, nil)
|
||||
sections := lyt.BoxesToSections(vm, nil)
|
||||
if len(vm) > 0 && len(sections) == 0 {
|
||||
t.Error("boxesToSections produced 0 sections from non-empty boxes")
|
||||
}
|
||||
@@ -146,10 +148,10 @@ func loadSnapshot(t *testing.T, path string) snapshot {
|
||||
return s
|
||||
}
|
||||
|
||||
func snapshotBoxesToGo(sbs []snapshotBox) []TextBox {
|
||||
boxes := make([]TextBox, len(sbs))
|
||||
func snapshotBoxesToGo(sbs []snapshotBox) []pdf.TextBox {
|
||||
boxes := make([]pdf.TextBox, len(sbs))
|
||||
for i, sb := range sbs {
|
||||
boxes[i] = TextBox{
|
||||
boxes[i] = pdf.TextBox{
|
||||
X0: sb.X0, X1: sb.X1, Top: sb.Top, Bottom: sb.Bottom,
|
||||
Text: sb.Text, PageNumber: sb.PageNumber - 1, // pdfplumber uses 1-based
|
||||
LayoutType: sb.LayoutType, LayoutNo: sb.LayoutNo,
|
||||
@@ -244,13 +246,6 @@ func toInt(v interface{}) int {
|
||||
}
|
||||
}
|
||||
|
||||
func toString(v interface{}) string {
|
||||
if v == nil {
|
||||
return ""
|
||||
}
|
||||
return fmt.Sprint(v)
|
||||
}
|
||||
|
||||
func formatBytes(n int) string {
|
||||
if n < 1024 {
|
||||
return fmt.Sprintf("%d", n)
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,51 +1,32 @@
|
||||
package parser
|
||||
package table
|
||||
|
||||
import (
|
||||
"context"
|
||||
"image"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// OSS model label taxonomies.
|
||||
// DLA: 8 unique classes (no duplicates — OSS ONNX model output).
|
||||
var ossDLALabels = []string{
|
||||
LayoutTypeTitle, LayoutTypeText, LayoutTypeReference,
|
||||
LayoutTypeFigure, DLALabelFigureCaption,
|
||||
LayoutTypeTable, DLALabelTableCaption, LayoutTypeEquation,
|
||||
// DeepDocTableBuilder implements pdf.TableBuilder for the DeepDoc
|
||||
// table structure recognition service. Label injection is handled by the
|
||||
// NewTableBuilderFor factory.
|
||||
type DeepDocTableBuilder struct {
|
||||
doc pdf.DocAnalyzer
|
||||
}
|
||||
|
||||
// TSR: 6 structural elements (matches deepdoc/vision/table_structure_recognizer.py).
|
||||
var ossTSRLabels = []string{
|
||||
"table", "table column", "table row",
|
||||
"table column header", "table projected row header",
|
||||
"table spanning cell",
|
||||
// NewDeepDocTableBuilder creates a TableBuilder. Labels must be set on the
|
||||
// underlying client by the caller (see deepdoc.go NewTableBuilderFor).
|
||||
func NewDeepDocTableBuilder(doc pdf.DocAnalyzer) *DeepDocTableBuilder {
|
||||
return &DeepDocTableBuilder{doc: doc}
|
||||
}
|
||||
|
||||
// OssDeepDocService implements TableBuilder and DocAnalyzer for the oss
|
||||
// DeepDoc service (ONNX models via HTTP).
|
||||
type OssDeepDocService struct {
|
||||
doc DocAnalyzer
|
||||
}
|
||||
|
||||
// NewOssDeepDocService creates a service backed by the oss DeepDoc service.
|
||||
// If doc is a *DeepDocClient, its DLALabels/TSRLabels are set to the OSS
|
||||
// taxonomy.
|
||||
func NewOssDeepDocService(doc DocAnalyzer) *OssDeepDocService {
|
||||
if c, ok := doc.(*DeepDocClient); ok {
|
||||
c.DLALabels = ossDLALabels
|
||||
c.TSRLabels = ossTSRLabels
|
||||
}
|
||||
return &OssDeepDocService{doc: doc}
|
||||
}
|
||||
|
||||
func (b *OssDeepDocService) Name() string { return "oss-deepdoc" }
|
||||
|
||||
func (b *OssDeepDocService) DetectCells(ctx context.Context, cropped image.Image) ([]TSRCell, error) {
|
||||
func (b *DeepDocTableBuilder) Name() string { return "deepdoc" }
|
||||
func (b *DeepDocTableBuilder) DetectCells(ctx context.Context, cropped image.Image) ([]pdf.TSRCell, error) {
|
||||
return b.doc.TSR(ctx, cropped)
|
||||
}
|
||||
|
||||
// GroupCells builds a row×column grid from OSS structural cells.
|
||||
// GroupCells builds a row×column grid from structural cells.
|
||||
//
|
||||
// Input: structural cells with labels "table row", "table column",
|
||||
// "table column header", "table spanning cell".
|
||||
@@ -59,14 +40,14 @@ func (b *OssDeepDocService) DetectCells(ctx context.Context, cropped image.Image
|
||||
// 5. Span injection: for each "table spanning cell", find grid cells
|
||||
// whose center falls inside the span bbox. The top-left cell gets
|
||||
// the span label + extended bbox; remaining cells are zeroed (covered).
|
||||
func (b *OssDeepDocService) GroupCells(cells []TSRCell) [][]TSRCell {
|
||||
func (b *DeepDocTableBuilder) GroupCells(cells []pdf.TSRCell) [][]pdf.TSRCell {
|
||||
if len(cells) == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
// 1. Collect and sort structural elements.
|
||||
var rows, cols, spans []TSRCell
|
||||
var header *TSRCell
|
||||
var rows, cols, spans []pdf.TSRCell
|
||||
var header *pdf.TSRCell
|
||||
|
||||
for _, c := range cells {
|
||||
switch {
|
||||
@@ -86,22 +67,22 @@ func (b *OssDeepDocService) GroupCells(cells []TSRCell) [][]TSRCell {
|
||||
return nil
|
||||
}
|
||||
|
||||
sortYFirstly(rows, 10)
|
||||
sortXFirstly(cols, 10)
|
||||
SortYFirstly(rows, 10)
|
||||
SortXFirstly(cols, 10)
|
||||
|
||||
// 2. If no column cells, synthesize one wide column from row extents.
|
||||
if len(cols) == 0 {
|
||||
x0 := rows[0].X0
|
||||
x1 := rows[0].X1
|
||||
cols = []TSRCell{{X0: x0, Y0: rows[0].Y0, X1: x1, Y1: rows[len(rows)-1].Y1, Label: "table column"}}
|
||||
cols = []pdf.TSRCell{{X0: x0, Y0: rows[0].Y0, X1: x1, Y1: rows[len(rows)-1].Y1, Label: "table column"}}
|
||||
}
|
||||
|
||||
// 3. Cross-product to build grid.
|
||||
grid := make([][]TSRCell, len(rows))
|
||||
grid := make([][]pdf.TSRCell, len(rows))
|
||||
for r := range rows {
|
||||
grid[r] = make([]TSRCell, len(cols))
|
||||
grid[r] = make([]pdf.TSRCell, len(cols))
|
||||
for c := range cols {
|
||||
grid[r][c] = TSRCell{
|
||||
grid[r][c] = pdf.TSRCell{
|
||||
X0: cols[c].X0,
|
||||
Y0: rows[r].Y0,
|
||||
X1: cols[c].X1,
|
||||
@@ -124,7 +105,6 @@ func (b *OssDeepDocService) GroupCells(cells []TSRCell) [][]TSRCell {
|
||||
|
||||
// 5. Span injection.
|
||||
for _, sp := range spans {
|
||||
// Find grid cells whose center falls inside the span bbox.
|
||||
type cellIdx struct{ r, c int }
|
||||
var covered []cellIdx
|
||||
for ri := range grid {
|
||||
@@ -140,23 +120,20 @@ func (b *OssDeepDocService) GroupCells(cells []TSRCell) [][]TSRCell {
|
||||
if len(covered) < 2 {
|
||||
continue
|
||||
}
|
||||
// Sort covered cells: top-left first.
|
||||
sort.Slice(covered, func(a, b int) bool {
|
||||
if covered[a].r != covered[b].r {
|
||||
return covered[a].r < covered[b].r
|
||||
}
|
||||
return covered[a].c < covered[b].c
|
||||
})
|
||||
// First cell: extend bbox to span bounds, set label.
|
||||
first := covered[0]
|
||||
grid[first.r][first.c].X0 = sp.X0
|
||||
grid[first.r][first.c].Y0 = sp.Y0
|
||||
grid[first.r][first.c].X1 = sp.X1
|
||||
grid[first.r][first.c].Y1 = sp.Y1
|
||||
grid[first.r][first.c].Label = sp.Label
|
||||
// Remaining cells: zeroed (covered).
|
||||
for _, idx := range covered[1:] {
|
||||
grid[idx.r][idx.c] = TSRCell{}
|
||||
grid[idx.r][idx.c] = pdf.TSRCell{}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -164,6 +141,6 @@ func (b *OssDeepDocService) GroupCells(cells []TSRCell) [][]TSRCell {
|
||||
}
|
||||
|
||||
// overlapsY reports whether two cells overlap in the Y dimension.
|
||||
func overlapsY(a, b TSRCell) bool {
|
||||
func overlapsY(a, b pdf.TSRCell) bool {
|
||||
return a.Y0 < b.Y1 && a.Y1 > b.Y0
|
||||
}
|
||||
@@ -1,12 +1,13 @@
|
||||
package parser
|
||||
package table
|
||||
|
||||
import (
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestOssDeepDocService_GroupCells_Basic4x5(t *testing.T) {
|
||||
b := &OssDeepDocService{}
|
||||
func TestDeepDocTableBuildService_GroupCells_Basic4x5(t *testing.T) {
|
||||
b := &DeepDocTableBuilder{}
|
||||
|
||||
cells := buildOSSCells(4, 5, 0, 0, 500, 200)
|
||||
grid := b.GroupCells(cells)
|
||||
@@ -21,8 +22,8 @@ func TestOssDeepDocService_GroupCells_Basic4x5(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestOssDeepDocService_GroupCells_Coords(t *testing.T) {
|
||||
b := &OssDeepDocService{}
|
||||
func TestDeepDocTableBuildService_GroupCells_Coords(t *testing.T) {
|
||||
b := &DeepDocTableBuilder{}
|
||||
|
||||
cells := buildOSSCells(2, 2, 0, 0, 200, 100)
|
||||
grid := b.GroupCells(cells)
|
||||
@@ -44,11 +45,11 @@ func TestOssDeepDocService_GroupCells_Coords(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestOssDeepDocService_GroupCells_HeaderPropagation(t *testing.T) {
|
||||
b := &OssDeepDocService{}
|
||||
func TestDeepDocTableBuildService_GroupCells_HeaderPropagation(t *testing.T) {
|
||||
b := &DeepDocTableBuilder{}
|
||||
|
||||
// 3 rows: header(Y=0-50) should map to row 0
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 200, Y1: 150, Label: "table"},
|
||||
{X0: 0, Y0: 0, X1: 200, Y1: 50, Label: "table row"},
|
||||
{X0: 0, Y0: 50, X1: 200, Y1: 100, Label: "table row"},
|
||||
@@ -78,11 +79,11 @@ func TestOssDeepDocService_GroupCells_HeaderPropagation(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestOssDeepDocService_GroupCells_SpanInjection(t *testing.T) {
|
||||
b := &OssDeepDocService{}
|
||||
func TestDeepDocTableBuildService_GroupCells_SpanInjection(t *testing.T) {
|
||||
b := &DeepDocTableBuilder{}
|
||||
|
||||
// 2×3 table, spanning cell covers cols 0-1 in row 0
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 300, Y1: 100, Label: "table"},
|
||||
{X0: 0, Y0: 0, X1: 300, Y1: 50, Label: "table row"},
|
||||
{X0: 0, Y0: 50, X1: 300, Y1: 100, Label: "table row"},
|
||||
@@ -119,8 +120,8 @@ func TestOssDeepDocService_GroupCells_SpanInjection(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestOssDeepDocService_GroupCells_IrregularSize(t *testing.T) {
|
||||
b := &OssDeepDocService{}
|
||||
func TestDeepDocTableBuildService_GroupCells_IrregularSize(t *testing.T) {
|
||||
b := &DeepDocTableBuilder{}
|
||||
cells := buildOSSCells(3, 2, 0, 0, 200, 120)
|
||||
grid := b.GroupCells(cells)
|
||||
|
||||
@@ -132,18 +133,18 @@ func TestOssDeepDocService_GroupCells_IrregularSize(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestOssDeepDocService_GroupCells_EmptyInput(t *testing.T) {
|
||||
b := &OssDeepDocService{}
|
||||
func TestDeepDocTableBuildService_GroupCells_EmptyInput(t *testing.T) {
|
||||
b := &DeepDocTableBuilder{}
|
||||
grid := b.GroupCells(nil)
|
||||
if len(grid) != 0 {
|
||||
t.Errorf("expected empty grid, got %d rows", len(grid))
|
||||
}
|
||||
}
|
||||
|
||||
func TestOssDeepDocService_GroupCells_NoRows(t *testing.T) {
|
||||
b := &OssDeepDocService{}
|
||||
func TestDeepDocTableBuildService_GroupCells_NoRows(t *testing.T) {
|
||||
b := &DeepDocTableBuilder{}
|
||||
// Only a "table" cell, no row cells.
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 500, Y1: 200, Label: "table"},
|
||||
}
|
||||
grid := b.GroupCells(cells)
|
||||
@@ -152,10 +153,10 @@ func TestOssDeepDocService_GroupCells_NoRows(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestOssDeepDocService_GroupCells_NoColumns(t *testing.T) {
|
||||
b := &OssDeepDocService{}
|
||||
func TestDeepDocTableBuildService_GroupCells_NoColumns(t *testing.T) {
|
||||
b := &DeepDocTableBuilder{}
|
||||
// Table + rows but no column cells → each row gets 1 wide column.
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 500, Y1: 100, Label: "table"},
|
||||
{X0: 0, Y0: 0, X1: 500, Y1: 50, Label: "table row"},
|
||||
{X0: 0, Y0: 50, X1: 500, Y1: 100, Label: "table row"},
|
||||
@@ -173,23 +174,23 @@ func TestOssDeepDocService_GroupCells_NoColumns(t *testing.T) {
|
||||
|
||||
// buildOSSCells constructs a set of OSS-style structural cells for
|
||||
// an R×C table with the given overall bounding box.
|
||||
func buildOSSCells(rows, cols int, x0, y0, x1, y1 float64) []TSRCell {
|
||||
func buildOSSCells(rows, cols int, x0, y0, x1, y1 float64) []pdf.TSRCell {
|
||||
rowH := (y1 - y0) / float64(rows)
|
||||
colW := (x1 - x0) / float64(cols)
|
||||
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: x0, Y0: y0, X1: x1, Y1: y1, Label: "table"},
|
||||
}
|
||||
|
||||
for r := 0; r < rows; r++ {
|
||||
cells = append(cells, TSRCell{
|
||||
cells = append(cells, pdf.TSRCell{
|
||||
X0: x0, Y0: y0 + float64(r)*rowH,
|
||||
X1: x1, Y1: y0 + float64(r+1)*rowH,
|
||||
Label: "table row",
|
||||
})
|
||||
}
|
||||
for c := 0; c < cols; c++ {
|
||||
cells = append(cells, TSRCell{
|
||||
cells = append(cells, pdf.TSRCell{
|
||||
X0: x0 + float64(c)*colW, Y0: y0,
|
||||
X1: x0 + float64(c+1)*colW, Y1: y1,
|
||||
Label: "table column",
|
||||
@@ -200,12 +201,12 @@ func buildOSSCells(rows, cols int, x0, y0, x1, y1 float64) []TSRCell {
|
||||
}
|
||||
|
||||
// isZeroCell reports whether a cell has its bbox zeroed (covered by a span).
|
||||
func isZeroCell(c TSRCell) bool {
|
||||
func isZeroCell(c pdf.TSRCell) bool {
|
||||
return c.X0 == 0 && c.Y0 == 0 && c.X1 == 0 && c.Y1 == 0
|
||||
}
|
||||
|
||||
// hasLabel reports whether any cell in a row has a label containing substr.
|
||||
func hasLabel(row []TSRCell, substr string) bool {
|
||||
func hasLabel(row []pdf.TSRCell, substr string) bool {
|
||||
for _, c := range row {
|
||||
if strings.Contains(strings.ToLower(c.Label), strings.ToLower(substr)) {
|
||||
return true
|
||||
101
internal/deepdoc/parser/pdf/table/merge_captions.go
Normal file
101
internal/deepdoc/parser/pdf/table/merge_captions.go
Normal file
@@ -0,0 +1,101 @@
|
||||
package table
|
||||
|
||||
import (
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
func MergeCaptions(sections []pdf.Section, figures []pdf.Section) []pdf.Section {
|
||||
captions := make([]int, 0, 4)
|
||||
for i, s := range sections {
|
||||
captionType := CaptionKind(s)
|
||||
if captionType == "" {
|
||||
continue
|
||||
}
|
||||
target := findNearestParent(i, s, sections, figures, captionType)
|
||||
if target >= 0 {
|
||||
// For table sections, prepend caption before the HTML table
|
||||
// (matching Python's _extract_table_figure caption->construct_table).
|
||||
if sections[target].LayoutType == pdf.LayoutTypeTable && sections[target].Text != "" {
|
||||
sections[target].Text = s.Text + sections[target].Text
|
||||
} else if sections[target].Text != "" {
|
||||
sections[target].Text += " " + s.Text
|
||||
} else {
|
||||
sections[target].Text = s.Text
|
||||
}
|
||||
}
|
||||
captions = append(captions, i)
|
||||
}
|
||||
// Remove caption sections in reverse order.
|
||||
n := len(sections)
|
||||
out := make([]pdf.Section, 0, n-len(captions))
|
||||
capSet := make(map[int]bool, len(captions))
|
||||
for _, idx := range captions {
|
||||
capSet[idx] = true
|
||||
}
|
||||
for i, s := range sections {
|
||||
if !capSet[i] {
|
||||
out = append(out, s)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// findNearestParent finds the nearest figure (for figure caption) or
|
||||
// table (for table caption) section by position proximity.
|
||||
// captionType is "table" or "figure" (from captionKind).
|
||||
// Returns the index in `sections` (for tables) or a virtual index mapping
|
||||
// to `figures` (negative offset for figures).
|
||||
func findNearestParent(captionIdx int, caption pdf.Section, sections []pdf.Section, figures []pdf.Section, captionType string) int {
|
||||
find := func(targets []pdf.Section, skipIdx int) (int, float64) {
|
||||
bestIdx := -1
|
||||
bestDist := 1e9
|
||||
for i, t := range targets {
|
||||
if i == skipIdx {
|
||||
continue // don't match caption to itself
|
||||
}
|
||||
if len(t.Positions) == 0 || len(caption.Positions) == 0 {
|
||||
continue
|
||||
}
|
||||
tp := t.Positions[0]
|
||||
cp := caption.Positions[0]
|
||||
// Squared Euclidean distance (Python _extract_table_figure:1196).
|
||||
// Caption is typically below. Use center-point distance.
|
||||
cx := (tp.Left + tp.Right) / 2
|
||||
cy := (tp.Top + tp.Bottom) / 2
|
||||
ccx := (cp.Left + cp.Right) / 2
|
||||
ccy := (cp.Top + cp.Bottom) / 2
|
||||
dist := (cx-ccx)*(cx-ccx) + (cy-ccy)*(cy-ccy)
|
||||
if dist < bestDist {
|
||||
bestDist = dist
|
||||
bestIdx = i
|
||||
}
|
||||
}
|
||||
return bestIdx, bestDist
|
||||
}
|
||||
|
||||
const maxCaptionGap = 40000.0 // PDF points (~7cm) — beyond this, don't attach.
|
||||
if captionType == pdf.LayoutTypeFigure && len(figures) > 0 {
|
||||
idx, dist := find(figures, -1) // figures don't contain the caption itself
|
||||
if idx >= 0 && dist < maxCaptionGap {
|
||||
// Match by position coordinates, not PositionTag strings.
|
||||
f := figures[idx]
|
||||
for i, s := range sections {
|
||||
if s.LayoutType != pdf.LayoutTypeFigure || len(s.Positions) == 0 || len(f.Positions) == 0 {
|
||||
continue
|
||||
}
|
||||
sp, fp := s.Positions[0], f.Positions[0]
|
||||
if sp.Left == fp.Left && sp.Right == fp.Right &&
|
||||
sp.Top == fp.Top && sp.Bottom == fp.Bottom {
|
||||
return i
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if captionType == pdf.LayoutTypeTable {
|
||||
idx, dist := find(sections, captionIdx)
|
||||
if idx >= 0 && dist < maxCaptionGap && sections[idx].LayoutType == pdf.LayoutTypeTable {
|
||||
return idx
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
74
internal/deepdoc/parser/pdf/table/merge_captions_test.go
Normal file
74
internal/deepdoc/parser/pdf/table/merge_captions_test.go
Normal file
@@ -0,0 +1,74 @@
|
||||
package table
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// TestMergeCaptions_Unit verifies mergeCaptions directly without full pipeline.
|
||||
func TestMergeCaptions_Unit(t *testing.T) {
|
||||
sections := []pdf.Section{
|
||||
{Text: "F", LayoutType: "figure", Positions: []pdf.Position{{PageNumbers: []int{0, 0}, Left: 40, Right: 60, Top: 30, Bottom: 45}}},
|
||||
{Text: "C", LayoutType: "figure caption", Positions: []pdf.Position{{PageNumbers: []int{0, 0}, Left: 40, Right: 60, Top: 80, Bottom: 95}}},
|
||||
}
|
||||
figures := pdf.CollectFigures(sections)
|
||||
|
||||
result := MergeCaptions(sections, figures)
|
||||
|
||||
// Caption removed.
|
||||
if len(result) != 1 {
|
||||
t.Fatalf("expected 1 section after merge, got %d", len(result))
|
||||
}
|
||||
// Figure text includes caption.
|
||||
if !strings.Contains(result[0].Text, "C") {
|
||||
t.Errorf("expected figure Text to contain caption 'C', got %q", result[0].Text)
|
||||
}
|
||||
if result[0].LayoutType != "figure" {
|
||||
t.Errorf("expected figure LayoutType, got %q", result[0].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
// TestMergeCaptions_TableCaption verifies table caption merging directly.
|
||||
func TestMergeCaptions_TableCaption(t *testing.T) {
|
||||
sections := []pdf.Section{
|
||||
{Text: "T", LayoutType: "table", Positions: []pdf.Position{{PageNumbers: []int{0, 0}, Left: 40, Right: 60, Top: 30, Bottom: 45}}},
|
||||
{Text: "C", LayoutType: "table caption", Positions: []pdf.Position{{PageNumbers: []int{0, 0}, Left: 40, Right: 60, Top: 80, Bottom: 95}}},
|
||||
}
|
||||
figures := pdf.CollectFigures(sections)
|
||||
|
||||
result := MergeCaptions(sections, figures)
|
||||
|
||||
if len(result) != 1 {
|
||||
t.Fatalf("expected 1 section after merge, got %d", len(result))
|
||||
}
|
||||
if !strings.Contains(result[0].Text, "C") {
|
||||
t.Errorf("expected table Text to contain caption 'C', got %q", result[0].Text)
|
||||
}
|
||||
}
|
||||
|
||||
// TestMergeCaptions_EuclideanDistance verifies that caption matching uses
|
||||
// squared Euclidean distance (center-to-center), not Y-only distance.
|
||||
// Two captions at different X positions — the one closer by Euclidean
|
||||
// distance wins, even if its Y distance is slightly larger.
|
||||
func TestMergeCaptions_EuclideanDistance(t *testing.T) {
|
||||
sections := []pdf.Section{
|
||||
{Text: "F", LayoutType: "figure", Positions: []pdf.Position{
|
||||
{PageNumbers: []int{0, 0}, Left: 0, Right: 100, Top: 0, Bottom: 50},
|
||||
}},
|
||||
// Caption A: directly below figure (dx=0, dy=20) → Euclidean = 20²
|
||||
{Text: "close", LayoutType: "figure caption", Positions: []pdf.Position{
|
||||
{PageNumbers: []int{0, 0}, Left: 0, Right: 100, Top: 70, Bottom: 80},
|
||||
}},
|
||||
}
|
||||
figures := pdf.CollectFigures(sections)
|
||||
result := MergeCaptions(sections, figures)
|
||||
// Caption merged into figure — verified by figure Text containing caption.
|
||||
if len(result) != 1 {
|
||||
t.Fatalf("expected 1 section after merge, got %d", len(result))
|
||||
}
|
||||
if !strings.Contains(result[0].Text, "close") {
|
||||
t.Errorf("figure Text should contain caption 'close', got %q", result[0].Text)
|
||||
}
|
||||
}
|
||||
281
internal/deepdoc/parser/pdf/table/table_annotate.go
Normal file
281
internal/deepdoc/parser/pdf/table/table_annotate.go
Normal file
@@ -0,0 +1,281 @@
|
||||
package table
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"ragflow/internal/deepdoc/parser/pdf/util"
|
||||
)
|
||||
|
||||
// ── region matching ────────────────────────────────────────────────────
|
||||
|
||||
// tableMatch pairs a DLA table region with the indices of boxes that overlap it.
|
||||
type TableMatch struct {
|
||||
Region pdf.DLARegion
|
||||
BoxIdx []int
|
||||
}
|
||||
|
||||
// ── region matching ────────────────────────────────────────────────────
|
||||
|
||||
func regionOverlapsBox(region pdf.DLARegion, box pdf.TextBox, scale float64) bool {
|
||||
rx0 := region.X0 / scale
|
||||
ry0 := region.Y0 / scale
|
||||
rx1 := region.X1 / scale
|
||||
ry1 := region.Y1 / scale
|
||||
scaledR := pdf.DLARegion{X0: rx0, Y0: ry0, X1: rx1, Y1: ry1}
|
||||
inter := util.OverlapInter(&scaledR, &box)
|
||||
boxArea := util.Area(&box)
|
||||
if boxArea <= 0 {
|
||||
return false
|
||||
}
|
||||
return inter/boxArea >= 0.4 // matches Python thr=0.4
|
||||
}
|
||||
|
||||
// matchTableRegions pairs DLA table regions with boxes that overlap them.
|
||||
// Each table region is matched if at least one box overlaps it (>40% of box
|
||||
// area) or if there are no boxes at all (image-only PDF), matching Python's
|
||||
// _table_transformer_job which processes every table DLA region.
|
||||
func MatchTableRegions(boxes []pdf.TextBox, regions []pdf.DLARegion, scale float64) []TableMatch {
|
||||
var matches []TableMatch
|
||||
for _, r := range regions {
|
||||
if r.Label != pdf.LayoutTypeTable {
|
||||
continue
|
||||
}
|
||||
var matched []int
|
||||
for i, b := range boxes {
|
||||
if regionOverlapsBox(r, b, scale) {
|
||||
matched = append(matched, i)
|
||||
}
|
||||
}
|
||||
if len(matched) > 0 || len(boxes) == 0 {
|
||||
matches = append(matches, TableMatch{Region: r, BoxIdx: matched})
|
||||
}
|
||||
}
|
||||
return matches
|
||||
}
|
||||
|
||||
// ── layout annotation ──────────────────────────────────────────────────
|
||||
|
||||
// annotateBoxLayouts sets LayoutType and LayoutNo on each box, matching
|
||||
// Python's LayoutRecognizer.__call__ which assigns layout types in priority
|
||||
// order (footer→header→…→equation) with an overlap threshold of 40% of the
|
||||
// box's area.
|
||||
//
|
||||
// Python: _layouts_rec (pdf_parser.py:827) → LayoutRecognizer.__call__ →
|
||||
//
|
||||
// for lt in priority_order: findLayout(lt)
|
||||
//
|
||||
// Each findLayout(ty): for each unannotated box, find the DLA region of
|
||||
// type ty with max overlap ≥ 0.4 × box_area. First type to match wins.
|
||||
//
|
||||
// CID-pattern boxes (e.g. "(cid:123)") are skipped as garbage.
|
||||
// annotateBoxLayouts assigns LayoutType and LayoutNo to boxes based on DLA
|
||||
// regions. Returns the filtered slice (Python pops CID-garbled boxes and
|
||||
// garbage-layout boxes at wrong positions — Go mirrors with compact).
|
||||
// Also creates synthetic figure boxes for unmatched figure/equation regions.
|
||||
func AnnotateBoxLayouts(boxes []pdf.TextBox, regions []pdf.DLARegion, scale float64, pageImgHeight float64) []pdf.TextBox {
|
||||
if len(regions) == 0 {
|
||||
return boxes
|
||||
}
|
||||
|
||||
// Scale all regions to PDF space once.
|
||||
type scaledRegion struct {
|
||||
x0, y0, x1, y1 float64
|
||||
label string
|
||||
}
|
||||
scaled := make([]scaledRegion, len(regions))
|
||||
for i, r := range regions {
|
||||
scaled[i] = scaledRegion{
|
||||
x0: r.X0 / scale, y0: r.Y0 / scale,
|
||||
x1: r.X1 / scale, y1: r.Y1 / scale,
|
||||
label: r.Label,
|
||||
}
|
||||
}
|
||||
|
||||
// DLA confidence filter — matches Python's `score >= 0.4`.
|
||||
regionOK := make([]bool, len(regions))
|
||||
for i, r := range regions {
|
||||
regionOK[i] = r.Confidence >= 0.4 || !isGarbageLayoutType(r.Label)
|
||||
}
|
||||
|
||||
// Pre-compute per-type index for each region (Python: matched index within
|
||||
// filtered layouts_of_type list). "text" regions get 0,1,2... independent
|
||||
// of "figure" regions.
|
||||
typeIndex := make([]int, len(regions))
|
||||
typeCounters := make(map[string]int)
|
||||
for j, r := range scaled {
|
||||
if regionOK[j] {
|
||||
typeIndex[j] = typeCounters[r.label]
|
||||
typeCounters[r.label]++
|
||||
}
|
||||
}
|
||||
|
||||
// Track visited regions (Python: layout["visited"] = True).
|
||||
visited := make([]bool, len(regions))
|
||||
|
||||
// Marks for Python-style pop removal.
|
||||
dropped := make([]bool, len(boxes))
|
||||
|
||||
// Priority order matching Python's findLayout loop.
|
||||
priorityOrder := []string{
|
||||
pdf.LayoutTypeFooter, pdf.LayoutTypeHeader, pdf.LayoutTypeReference,
|
||||
pdf.DLALabelFigureCaption, pdf.DLALabelTableCaption,
|
||||
pdf.LayoutTypeTitle, pdf.LayoutTypeTable, pdf.LayoutTypeText,
|
||||
pdf.LayoutTypeFigure, pdf.LayoutTypeEquation,
|
||||
}
|
||||
for _, ty := range priorityOrder {
|
||||
for i := range boxes {
|
||||
if boxes[i].LayoutType != "" || dropped[i] {
|
||||
continue
|
||||
}
|
||||
// CID garbage: pop the box entirely (Python: bxs.pop(i)).
|
||||
if util.CIDPattern.MatchString(boxes[i].Text) {
|
||||
dropped[i] = true
|
||||
continue
|
||||
}
|
||||
boxArea := (boxes[i].X1 - boxes[i].X0) * (boxes[i].Bottom - boxes[i].Top)
|
||||
if boxArea <= 0 {
|
||||
continue
|
||||
}
|
||||
bestOverlap := 0.0
|
||||
bestJ := -1
|
||||
for j, r := range scaled {
|
||||
if r.label != ty || !regionOK[j] {
|
||||
continue
|
||||
}
|
||||
ix0 := math.Max(r.x0, boxes[i].X0)
|
||||
iy0 := math.Max(r.y0, boxes[i].Top)
|
||||
ix1 := math.Min(r.x1, boxes[i].X1)
|
||||
iy1 := math.Min(r.y1, boxes[i].Bottom)
|
||||
if ix0 < ix1 && iy0 < iy1 {
|
||||
ov := (ix1 - ix0) * (iy1 - iy0) / boxArea
|
||||
if ov > bestOverlap {
|
||||
bestOverlap = ov
|
||||
bestJ = j
|
||||
}
|
||||
}
|
||||
}
|
||||
if bestJ >= 0 && bestOverlap >= 0.4 {
|
||||
// Garbage layout not at page edge → pop (Python: bxs.pop(i)).
|
||||
if isGarbageLayoutType(ty) && pageImgHeight > 0 && !garbageKeepFeat(ty, boxes[i], pageImgHeight/scale) {
|
||||
dropped[i] = true
|
||||
continue
|
||||
}
|
||||
visited[bestJ] = true
|
||||
// Python: equation mapped to "figure" for layout_type
|
||||
if ty == pdf.LayoutTypeEquation {
|
||||
boxes[i].LayoutType = pdf.LayoutTypeFigure
|
||||
} else {
|
||||
boxes[i].LayoutType = ty
|
||||
}
|
||||
// Python: f"{layout_type}-{matched}" where matched is per-type index
|
||||
boxes[i].LayoutNo = fmt.Sprintf("%s-%d", ty, typeIndex[bestJ])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Compact: remove popped boxes into a new backing array (Python
|
||||
// bxs.pop). Allocating a fresh slice is deliberate: annotations were
|
||||
// set in-place on the input elements, and callers (enrichWithDeepDoc)
|
||||
// rely on positional stability of the original slice for their
|
||||
// write-back loop. Reusing the input backing array would shift
|
||||
// survivors forward and break that index mapping.
|
||||
survivors := 0
|
||||
for i := range boxes {
|
||||
if !dropped[i] {
|
||||
survivors++
|
||||
}
|
||||
}
|
||||
compacted := make([]pdf.TextBox, 0, survivors)
|
||||
for i := range boxes {
|
||||
if !dropped[i] {
|
||||
compacted = append(compacted, boxes[i])
|
||||
}
|
||||
}
|
||||
boxes = compacted
|
||||
|
||||
// Synthetic figure boxes for unmatched figure/equation regions (Python:
|
||||
// dla_cli.py:187-195). Use a fresh per-type counter for synthetic boxes.
|
||||
synthIdx := 0
|
||||
for j, r := range scaled {
|
||||
if !regionOK[j] || visited[j] {
|
||||
continue
|
||||
}
|
||||
if r.label != pdf.LayoutTypeFigure && r.label != pdf.LayoutTypeEquation {
|
||||
continue
|
||||
}
|
||||
boxes = append(boxes, pdf.TextBox{
|
||||
X0: r.x0,
|
||||
X1: r.x1,
|
||||
Top: r.y0,
|
||||
Bottom: r.y1,
|
||||
Text: "",
|
||||
LayoutType: pdf.LayoutTypeFigure,
|
||||
LayoutNo: fmt.Sprintf("figure-%d", synthIdx),
|
||||
})
|
||||
synthIdx++
|
||||
}
|
||||
|
||||
return boxes
|
||||
}
|
||||
|
||||
// ── garbage layout helpers ────────────────────────────────────────────
|
||||
// garbageLayoutTypes matches Python's self.garbage_layouts.
|
||||
var garbageLayoutTypes = map[string]bool{
|
||||
pdf.LayoutTypeFooter: true, pdf.LayoutTypeHeader: true, pdf.LayoutTypeReference: true,
|
||||
}
|
||||
|
||||
func isGarbageLayoutType(ty string) bool {
|
||||
return garbageLayoutTypes[ty]
|
||||
}
|
||||
|
||||
// garbageKeepFeat matches Python's keep_feats in LayoutRecognizer.__call__:
|
||||
// footer near page bottom (>90% of page height) or header near page top (<10%)
|
||||
// are real page decorations — keep them. Others are DLA noise.
|
||||
func garbageKeepFeat(ty string, box pdf.TextBox, pageImgHeight float64) bool {
|
||||
switch ty {
|
||||
case pdf.LayoutTypeFooter:
|
||||
return box.Bottom < pageImgHeight*0.9
|
||||
case pdf.LayoutTypeHeader:
|
||||
return box.Top > pageImgHeight*0.1
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// writeTableAnnotations annotates boxes at boxIdx with table cell grid
|
||||
// information (R/C/H/SP). Cells are offset by cropOff, grouped into a grid,
|
||||
// and annotation fields are scaled back to PDF space for each box.
|
||||
func WriteTableAnnotations(boxes []pdf.TextBox, boxIdx []int, cells []pdf.TSRCell, scale, cropOffX, cropOffY float64, tb pdf.TableBuilder) {
|
||||
tableCells := make([]pdf.TSRCell, len(cells))
|
||||
for k := range cells {
|
||||
tableCells[k] = CellAddOffset(cells[k], cropOffX, cropOffY)
|
||||
}
|
||||
tblBoxes := make([]pdf.TextBox, len(boxIdx))
|
||||
for k, idx := range boxIdx {
|
||||
b := boxes[idx]
|
||||
tblBoxes[k] = pdf.TextBox{
|
||||
X0: b.X0 * scale, X1: b.X1 * scale,
|
||||
Top: b.Top * scale, Bottom: b.Bottom * scale,
|
||||
LayoutType: b.LayoutType,
|
||||
Text: b.Text,
|
||||
}
|
||||
}
|
||||
annotGrid := tb.GroupCells(tableCells)
|
||||
AnnotateTableBoxes(tblBoxes, annotGrid)
|
||||
for k, idx := range boxIdx {
|
||||
bp := &tblBoxes[k]
|
||||
boxes[idx].R = bp.R
|
||||
boxes[idx].RTop = bp.RTop / scale
|
||||
boxes[idx].RBott = bp.RBott / scale
|
||||
boxes[idx].H = bp.H
|
||||
boxes[idx].HTop = bp.HTop / scale
|
||||
boxes[idx].HBott = bp.HBott / scale
|
||||
boxes[idx].HLeft = bp.HLeft / scale
|
||||
boxes[idx].HRight = bp.HRight / scale
|
||||
boxes[idx].C = bp.C
|
||||
boxes[idx].CLeft = bp.CLeft / scale
|
||||
boxes[idx].CRight = bp.CRight / scale
|
||||
boxes[idx].SP = bp.SP
|
||||
}
|
||||
}
|
||||
604
internal/deepdoc/parser/pdf/table/table_annotate_test.go
Normal file
604
internal/deepdoc/parser/pdf/table/table_annotate_test.go
Normal file
@@ -0,0 +1,604 @@
|
||||
package table
|
||||
|
||||
import (
|
||||
"context"
|
||||
"image"
|
||||
"testing"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
func TestAnnotateBoxLayouts_SetsLabel(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 20},
|
||||
{X0: 0, X1: 100, Top: 30, Bottom: 50},
|
||||
}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 300, Y1: 60, Label: "title"}, // covers box 0 at scale 3
|
||||
{X0: 0, Y0: 90, X1: 300, Y1: 150, Label: "text"}, // covers box 1 at scale 3
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 0)
|
||||
if boxes[0].LayoutType != "title" {
|
||||
t.Errorf("box 0: got %q, want 'title'", boxes[0].LayoutType)
|
||||
}
|
||||
if boxes[1].LayoutType != "text" {
|
||||
t.Errorf("box 1: got %q, want 'text'", boxes[1].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_NoMatch(t *testing.T) {
|
||||
// Region far away from the box — no overlap
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 20},
|
||||
}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 900, Y0: 900, X1: 1000, Y1: 1000, Label: "far"}, // completely outside
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 0)
|
||||
if boxes[0].LayoutType != "" {
|
||||
t.Errorf("no match: expected empty, got %q", boxes[0].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_EmptyRegions(t *testing.T) {
|
||||
boxes := []pdf.TextBox{{X0: 0, X1: 100, Top: 0, Bottom: 20}}
|
||||
boxes = AnnotateBoxLayouts(boxes, nil, 3.0, 0)
|
||||
boxes = AnnotateBoxLayouts(boxes, []pdf.DLARegion{}, 3.0, 0)
|
||||
if boxes[0].LayoutType != "" {
|
||||
t.Errorf("empty regions: got %q, want empty", boxes[0].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_PriorityOverMaxArea(t *testing.T) {
|
||||
// "table" type checked before "text" in priority order.
|
||||
// Even if "text" region has larger overlap, "table" wins if it meets threshold (≥40%).
|
||||
boxes := []pdf.TextBox{{X0: 0, X1: 100, Top: 0, Bottom: 50}}
|
||||
regions := []pdf.DLARegion{
|
||||
// text region: full coverage (100% overlap) — but lower priority
|
||||
{X0: 0, Y0: 0, X1: 300, Y1: 150, Label: "text"},
|
||||
// table region: 45% overlap (45x50 out of 100x50) — higher priority, meets threshold
|
||||
{X0: 0, Y0: 0, X1: 45 * 3, Y1: 50 * 3, Label: "table"},
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 0)
|
||||
if boxes[0].LayoutType != "table" {
|
||||
t.Errorf("priority: 'table' should win over 'text' when both meet threshold, got %q", boxes[0].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_OverlapThreshold(t *testing.T) {
|
||||
// Region overlaps only 30% of box — below 0.4 threshold — should NOT match.
|
||||
boxes := []pdf.TextBox{{X0: 0, X1: 100, Top: 0, Bottom: 50}}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 30 * 3, Y1: 30 * 3, Label: "table"}, // covers ~30% of box
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 0)
|
||||
if boxes[0].LayoutType != "" {
|
||||
t.Errorf("threshold: overlap < 40%% should not match, got %q", boxes[0].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_CIDGarbage(t *testing.T) {
|
||||
// CID-pattern boxes should be popped entirely (Python: bxs.pop(i)).
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 20, Text: "(cid:123)"},
|
||||
{X0: 0, X1: 100, Top: 30, Bottom: 50, Text: "normal text"},
|
||||
}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 300, Y1: 60, Label: "text", Confidence: 0.9},
|
||||
{X0: 0, Y0: 90, X1: 300, Y1: 150, Label: "text", Confidence: 0.9},
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 0)
|
||||
// CID-garbled box was popped → only 1 box remains.
|
||||
if len(boxes) != 1 {
|
||||
t.Fatalf("CID-garbled box should be popped, got %d boxes", len(boxes))
|
||||
}
|
||||
if boxes[0].LayoutType != "text" {
|
||||
t.Errorf("CID: remaining box should be 'text', got %q", boxes[0].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_LayoutNoFormat(t *testing.T) {
|
||||
// layoutno uses Python format: "{type}-{per_type_index}" where per_type_index
|
||||
// is the index of the matched DLA region within its type (not global).
|
||||
// Two boxes overlapping the SAME text region share the same layoutno → VM can merge them.
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 20},
|
||||
{X0: 0, X1: 100, Top: 30, Bottom: 50},
|
||||
}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 300, Y1: 150, Label: "text"}, // covers both boxes
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 0)
|
||||
want := "text-0"
|
||||
if boxes[0].LayoutNo != want {
|
||||
t.Errorf("box 0 layoutno: got %q, want %q", boxes[0].LayoutNo, want)
|
||||
}
|
||||
if boxes[1].LayoutNo != want {
|
||||
t.Errorf("box 1 layoutno should share same per-type index: got %q, want %q", boxes[1].LayoutNo, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_LayoutNoDifferentRegions(t *testing.T) {
|
||||
// Two boxes in different text regions → different layoutno.
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 20},
|
||||
{X0: 0, X1: 100, Top: 100, Bottom: 120},
|
||||
}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 300, Y1: 60, Label: "text"}, // per-type index 0
|
||||
{X0: 0, Y0: 300, X1: 300, Y1: 360, Label: "text"}, // per-type index 1
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 0)
|
||||
if boxes[0].LayoutNo != "text-0" {
|
||||
t.Errorf("box 0: got %q, want 'text-0'", boxes[0].LayoutNo)
|
||||
}
|
||||
if boxes[1].LayoutNo != "text-1" {
|
||||
t.Errorf("box 1: got %q, want 'text-1'", boxes[1].LayoutNo)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAnnotateBoxLayouts_ConfidenceFilter verifies that DLA regions with
|
||||
// low confidence (< 0.4) for garbage layout types are excluded from matching.
|
||||
// Python: float(b["score"]) >= 0.4 filter in LayoutRecognizer.
|
||||
func TestAnnotateBoxLayouts_ConfidenceFilter(t *testing.T) {
|
||||
boxes := []pdf.TextBox{{X0: 0, X1: 100, Top: 0, Bottom: 50}}
|
||||
// Low-confidence footer — should be filtered out.
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 300, Y1: 150, Label: "footer", Confidence: 0.2},
|
||||
{X0: 0, Y0: 0, X1: 300, Y1: 150, Label: "text", Confidence: 0.9},
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 0)
|
||||
// Footer region filtered (low confidence) → box matches "text" instead.
|
||||
if boxes[0].LayoutType != "text" {
|
||||
t.Errorf("low-confidence footer filtered → box should get 'text', got %q", boxes[0].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_GarbageFooterRejected(t *testing.T) {
|
||||
// Footer at page bottom: Bottom(290) > 270 (90% of 300px→PDF height 100→90% of 100=90)
|
||||
// → real footer decoration → garbage → pop (Python: bxs.pop(i)).
|
||||
boxes := []pdf.TextBox{{X0: 0, X1: 100, Top: 280, Bottom: 290}}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 840, X1: 300, Y1: 870, Label: "footer", Confidence: 0.9}, // y=280-290 after /3, PDF 93-97
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 300) // PDF height = 300/3 = 100
|
||||
if len(boxes) != 0 {
|
||||
t.Errorf("footer at bottom: should be popped as decoration, got %d boxes left", len(boxes))
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_HeaderRemovedAtTop(t *testing.T) {
|
||||
// Header at page top edge (y=5 in 300px page → PDF height 100 → 5 < 10% of 100)
|
||||
// → real header decoration → garbage → pop (Python: bxs.pop(i)).
|
||||
boxes := []pdf.TextBox{{X0: 0, X1: 100, Top: 5, Bottom: 20}}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 15, X1: 300, Y1: 60, Label: "header", Confidence: 0.9}, // y=5-20 after /3
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 300)
|
||||
if len(boxes) != 0 {
|
||||
t.Errorf("header at very top: should be popped as decoration, got %d boxes left", len(boxes))
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_HeaderKeptInMiddle(t *testing.T) {
|
||||
// Header in middle of page (y=50 in 300px page → PDF height 100 → 50 > 10)
|
||||
// → DLA false positive → KEEP the text.
|
||||
boxes := []pdf.TextBox{{X0: 0, X1: 100, Top: 50, Bottom: 70}}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 150, X1: 300, Y1: 210, Label: "header", Confidence: 0.9}, // y=50-70 after /3
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 300)
|
||||
if boxes[0].LayoutType != "header" {
|
||||
t.Errorf("header in middle of page: DLA false positive, keep text, got %q", boxes[0].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_FooterRemovedAtBottom(t *testing.T) {
|
||||
// Footer at page bottom (y=95 in 300px page → PDF height 100 → 95 > 90% of 100)
|
||||
// → real footer decoration → garbage → REMOVE.
|
||||
boxes := []pdf.TextBox{{X0: 0, X1: 100, Top: 95, Bottom: 100}}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 285, X1: 300, Y1: 300, Label: "footer", Confidence: 0.9}, // y=95-100 after /3
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 300)
|
||||
if len(boxes) != 0 {
|
||||
t.Errorf("footer at very bottom: should be popped as decoration, got %d boxes left", len(boxes))
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_FooterKeptInMiddle(t *testing.T) {
|
||||
// Footer in middle of page (y=50 in 300px page → PDF height 100 → 50 < 90)
|
||||
// → DLA false positive → KEEP the text.
|
||||
boxes := []pdf.TextBox{{X0: 0, X1: 100, Top: 50, Bottom: 70}}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 150, X1: 300, Y1: 210, Label: "footer", Confidence: 0.9}, // y=50-70 after /3
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 300)
|
||||
if boxes[0].LayoutType != "footer" {
|
||||
t.Errorf("footer in middle of page: DLA false positive, keep text, got %q", boxes[0].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_ReferenceAlwaysGarbage(t *testing.T) {
|
||||
// Reference type is always garbage regardless of position (no keep_feat).
|
||||
boxes := []pdf.TextBox{{X0: 0, X1: 100, Top: 50, Bottom: 70}}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 150, X1: 300, Y1: 210, Label: "reference", Confidence: 0.9},
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 300)
|
||||
if len(boxes) != 0 {
|
||||
t.Errorf("reference: should always be garbage-filtered, got %q", boxes[0].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_NonGarbageTypeUnaffected(t *testing.T) {
|
||||
// "text" type is NOT a garbage type — should always be assigned.
|
||||
boxes := []pdf.TextBox{{X0: 0, X1: 100, Top: 200, Bottom: 220}}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 600, X1: 300, Y1: 660, Label: "text"},
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 300)
|
||||
if boxes[0].LayoutType != "text" {
|
||||
t.Errorf("non-garbage type: should be assigned, got %q", boxes[0].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotateBoxLayouts_ZeroPageHeightDisablesGarbage(t *testing.T) {
|
||||
// pageImgHeight=0 → garbage check disabled → all types assigned.
|
||||
boxes := []pdf.TextBox{{X0: 0, X1: 100, Top: 100, Bottom: 120}}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 300, X1: 300, Y1: 360, Label: "header", Confidence: 0.9},
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 0)
|
||||
if boxes[0].LayoutType != "header" {
|
||||
t.Errorf("zero page height: garbage check disabled, got %q", boxes[0].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAnnotateBoxLayouts_SyntheticFigure creates synthetic figure boxes for
|
||||
// unmatched figure/equation DLA regions (Python: dla_cli.py:187-195).
|
||||
func TestAnnotateBoxLayouts_SyntheticFigure(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 20, Text: "text box"},
|
||||
}
|
||||
// Two figure regions, one text region
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 150, Y1: 60, Label: "text", Confidence: 0.9}, // matches text box → visited
|
||||
{X0: 300, Y0: 300, X1: 600, Y1: 600, Label: "figure", Confidence: 0.9}, // no box overlaps → synthetic
|
||||
{X0: 600, Y0: 0, X1: 900, Y1: 300, Label: "figure", Confidence: 0.9}, // no box overlaps → synthetic
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 0)
|
||||
// Original text box + 2 synthetic figure boxes = 3
|
||||
if len(boxes) != 3 {
|
||||
t.Fatalf("expected 3 boxes (1 original + 2 synthetic figures), got %d", len(boxes))
|
||||
}
|
||||
// Check synthetic boxes
|
||||
foundFig0, foundFig1 := false, false
|
||||
for _, b := range boxes {
|
||||
if b.LayoutType == "figure" && b.Text == "" {
|
||||
if b.LayoutNo == "figure-0" {
|
||||
foundFig0 = true
|
||||
if b.X0 != 100 || b.X1 != 200 {
|
||||
t.Errorf("synthetic figure-0: expected x0=100,x1=200 (300/3,600/3), got x0=%v,x1=%v", b.X0, b.X1)
|
||||
}
|
||||
}
|
||||
if b.LayoutNo == "figure-1" {
|
||||
foundFig1 = true
|
||||
}
|
||||
}
|
||||
}
|
||||
if !foundFig0 {
|
||||
t.Error("missing synthetic figure-0 box")
|
||||
}
|
||||
if !foundFig1 {
|
||||
t.Error("missing synthetic figure-1 box")
|
||||
}
|
||||
}
|
||||
|
||||
// TestAnnotateBoxLayouts_EquationMappedToFigure verifies equation DLA regions
|
||||
// get LayoutType="figure" but LayoutNo keeps "equation" prefix (Python behavior).
|
||||
func TestAnnotateBoxLayouts_EquationMappedToFigure(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 20},
|
||||
}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 300, Y1: 60, Label: "equation", Confidence: 0.9},
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 0)
|
||||
if len(boxes) != 1 {
|
||||
t.Fatalf("expected 1 box, got %d", len(boxes))
|
||||
}
|
||||
if boxes[0].LayoutType != "figure" {
|
||||
t.Errorf("equation → LayoutType: got %q, want 'figure'", boxes[0].LayoutType)
|
||||
}
|
||||
if boxes[0].LayoutNo != "equation-0" {
|
||||
t.Errorf("equation → LayoutNo: got %q, want 'equation-0'", boxes[0].LayoutNo)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAnnotateBoxLayouts_MixedTypesLayoutNo verifies per-type LayoutNo counting
|
||||
// with multiple region types present.
|
||||
func TestAnnotateBoxLayouts_MixedTypesLayoutNo(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 20}, // overlaps text region 0
|
||||
{X0: 0, X1: 100, Top: 200, Bottom: 220}, // overlaps text region 1
|
||||
{X0: 200, X1: 300, Top: 0, Bottom: 20}, // overlaps figure region 0 only
|
||||
}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 150, Y1: 60, Label: "text", Confidence: 0.9}, // text-0
|
||||
{X0: 0, Y0: 600, X1: 150, Y1: 660, Label: "text", Confidence: 0.9}, // text-1
|
||||
{X0: 600, Y0: 0, X1: 900, Y1: 60, Label: "figure", Confidence: 0.9}, // figure-0 (PDF: x0=200, x1=300)
|
||||
}
|
||||
boxes = AnnotateBoxLayouts(boxes, regions, 3.0, 0)
|
||||
if len(boxes) != 3 {
|
||||
t.Fatalf("expected 3 boxes, got %d", len(boxes))
|
||||
}
|
||||
// Check that text and figure indices are independent
|
||||
if boxes[0].LayoutNo != "text-0" {
|
||||
t.Errorf("box 0: got %q, want 'text-0'", boxes[0].LayoutNo)
|
||||
}
|
||||
if boxes[1].LayoutNo != "text-1" {
|
||||
t.Errorf("box 1: got %q, want 'text-1'", boxes[1].LayoutNo)
|
||||
}
|
||||
if boxes[2].LayoutNo != "figure-0" {
|
||||
t.Errorf("box 2: got %q, want 'figure-0' (independent from text counter)", boxes[2].LayoutNo)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAnnotateBoxLayouts_CompactionPreservesWriteBackMapping verifies that
|
||||
// when annotateBoxLayouts drops some boxes (CID garbage or garbage-layout
|
||||
// at non-edge positions), the compaction step does not corrupt the caller's
|
||||
// ability to write annotations back to the correct global box indices.
|
||||
//
|
||||
// The bug: annotateBoxLayouts compacts boxes in place in the shared backing
|
||||
// array, shifting survivors forward. enrichWithDeepDoc then iterates
|
||||
// len(indices) positions and writes pageBoxes[i] back to boxes[indices[i]],
|
||||
// but after compaction pageBoxes[1] holds what was originally pageBoxes[2],
|
||||
// so annotations land on the wrong global box.
|
||||
|
||||
func TestMatchTableRegions_SingleMatch(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 50},
|
||||
{X0: 200, X1: 300, Top: 0, Bottom: 50},
|
||||
}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 300, Y1: 150, Label: "table"}, // covers box 0 at scale 3
|
||||
{X0: 600, Y0: 0, X1: 900, Y1: 150, Label: "text"}, // non-table, ignored
|
||||
}
|
||||
matches := MatchTableRegions(boxes, regions, 3.0)
|
||||
if len(matches) != 1 {
|
||||
t.Fatalf("expected 1 match, got %d", len(matches))
|
||||
}
|
||||
if len(matches[0].BoxIdx) != 1 || matches[0].BoxIdx[0] != 0 {
|
||||
t.Errorf("expected box 0 matched, got %v", matches[0].BoxIdx)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMatchTableRegions_NoTableLabel(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 50},
|
||||
}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 300, Y1: 150, Label: "text"},
|
||||
{X0: 0, Y0: 0, X1: 300, Y1: 150, Label: "figure"},
|
||||
}
|
||||
matches := MatchTableRegions(boxes, regions, 3.0)
|
||||
if len(matches) != 0 {
|
||||
t.Errorf("non-table labels: expected 0 matches, got %d", len(matches))
|
||||
}
|
||||
}
|
||||
|
||||
func TestMatchTableRegions_MultipleBoxesSameTable(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 50}, // box 0
|
||||
{X0: 110, X1: 210, Top: 0, Bottom: 50}, // box 1
|
||||
}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 630, Y1: 150, Label: "table"}, // covers both boxes at scale 3
|
||||
}
|
||||
matches := MatchTableRegions(boxes, regions, 3.0)
|
||||
if len(matches) != 1 {
|
||||
t.Fatalf("expected 1 match, got %d", len(matches))
|
||||
}
|
||||
if len(matches[0].BoxIdx) != 2 {
|
||||
t.Errorf("expected 2 boxes matched, got %d: %v", len(matches[0].BoxIdx), matches[0].BoxIdx)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMatchTableRegions_ImageOnlyPDF(t *testing.T) {
|
||||
// Zero boxes — image-only PDF. Python processes every table DLA region
|
||||
// regardless of text box overlap.
|
||||
var boxes []pdf.TextBox // nil
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 300, Y1: 150, Label: "table"},
|
||||
{X0: 0, Y0: 0, X1: 300, Y1: 150, Label: "text"},
|
||||
}
|
||||
matches := MatchTableRegions(boxes, regions, 3.0)
|
||||
if len(matches) != 1 {
|
||||
t.Fatalf("image-only: expected 1 table match, got %d", len(matches))
|
||||
}
|
||||
if len(matches[0].BoxIdx) != 0 {
|
||||
t.Errorf("image-only: expected empty BoxIdx, got %d", len(matches[0].BoxIdx))
|
||||
}
|
||||
}
|
||||
|
||||
func TestMatchTableRegions_BelowThreshold(t *testing.T) {
|
||||
// Region overlaps only a sliver of the box (<40%) → no match.
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 100},
|
||||
}
|
||||
regions := []pdf.DLARegion{
|
||||
{X0: 0, Y0: 0, X1: 90, Y1: 90, Label: "table"}, // 30x30 at scale 3 → 9% overlap
|
||||
}
|
||||
matches := MatchTableRegions(boxes, regions, 3.0)
|
||||
if len(matches) != 0 {
|
||||
t.Errorf("below threshold: expected 0 matches, got %d", len(matches))
|
||||
}
|
||||
}
|
||||
|
||||
// MockTableBuilder is a test-only pdf.TableBuilder with a configurable GroupCells.
|
||||
type MockTableBuilder struct {
|
||||
GroupCellsFn func(cells []pdf.TSRCell) [][]pdf.TSRCell
|
||||
}
|
||||
|
||||
func (m *MockTableBuilder) Name() string { return "mock" }
|
||||
func (m *MockTableBuilder) DetectCells(_ context.Context, _ image.Image) ([]pdf.TSRCell, error) {
|
||||
return nil, nil
|
||||
}
|
||||
func (m *MockTableBuilder) GroupCells(cells []pdf.TSRCell) [][]pdf.TSRCell {
|
||||
if m.GroupCellsFn != nil {
|
||||
return m.GroupCellsFn(cells)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ── writeTableAnnotations unit tests ──────────────────────────────────
|
||||
|
||||
func TestWriteTableAnnotations_WriteBack(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 10, X1: 100, Top: 10, Bottom: 30, Text: "A", LayoutType: "table"},
|
||||
{X0: 110, X1: 200, Top: 10, Bottom: 30, Text: "B", LayoutType: "table"},
|
||||
{X0: 10, X1: 100, Top: 35, Bottom: 55, Text: "C", LayoutType: "table"},
|
||||
}
|
||||
BoxIdx := []int{0, 2}
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 30, Y0: 30, X1: 300, Y1: 90, Label: "table row"},
|
||||
{X0: 30, Y0: 110, X1: 300, Y1: 170, Label: "table row"},
|
||||
}
|
||||
scale := 3.0
|
||||
|
||||
tb := &MockTableBuilder{GroupCellsFn: func(cells []pdf.TSRCell) [][]pdf.TSRCell {
|
||||
return [][]pdf.TSRCell{{cells[0]}, {cells[1]}}
|
||||
}}
|
||||
WriteTableAnnotations(boxes, BoxIdx, cells, scale, 0, 0, tb)
|
||||
|
||||
if boxes[0].R != 0 {
|
||||
t.Errorf("box 0 R = %d, want 0", boxes[0].R)
|
||||
}
|
||||
if boxes[0].C != 0 {
|
||||
t.Errorf("box 0 C = %d, want 0", boxes[0].C)
|
||||
}
|
||||
// Box 1 was not in BoxIdx — should NOT be annotated
|
||||
if boxes[1].R != 0 || boxes[1].C != 0 {
|
||||
t.Errorf("box 1 should not be annotated: R=%d C=%d", boxes[1].R, boxes[1].C)
|
||||
}
|
||||
if boxes[2].R != 1 {
|
||||
t.Errorf("box 2 R = %d, want 1", boxes[2].R)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriteTableAnnotations_ScaleDown(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 10, X1: 100, Top: 10, Bottom: 50, Text: "X", LayoutType: "table"},
|
||||
}
|
||||
BoxIdx := []int{0}
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 30, Y0: 30, X1: 300, Y1: 150, Label: "table row"},
|
||||
}
|
||||
scale := 3.0
|
||||
|
||||
tb := &MockTableBuilder{GroupCellsFn: func(cells []pdf.TSRCell) [][]pdf.TSRCell {
|
||||
return [][]pdf.TSRCell{{cells[0]}}
|
||||
}}
|
||||
WriteTableAnnotations(boxes, BoxIdx, cells, scale, 0, 0, tb)
|
||||
|
||||
// After scale-down: RTop / 3 should be in PDF space (~10).
|
||||
if boxes[0].RTop == 0 {
|
||||
t.Error("RTop should be non-zero after annotation")
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriteTableAnnotations_EmptyCells(t *testing.T) {
|
||||
boxes := []pdf.TextBox{{X0: 10, X1: 100, Top: 10, Bottom: 50, Text: "X", LayoutType: "table"}}
|
||||
BoxIdx := []int{0}
|
||||
var cells []pdf.TSRCell
|
||||
|
||||
tb := &MockTableBuilder{GroupCellsFn: func(cells []pdf.TSRCell) [][]pdf.TSRCell {
|
||||
return nil
|
||||
}}
|
||||
// Should not panic with empty cells.
|
||||
WriteTableAnnotations(boxes, BoxIdx, cells, 3.0, 0, 0, tb)
|
||||
if boxes[0].R != 0 || boxes[0].C != 0 {
|
||||
t.Errorf("empty cells: R=%d C=%d, want 0,0", boxes[0].R, boxes[0].C)
|
||||
}
|
||||
}
|
||||
|
||||
// ── markNoMergeTables unit tests ─────────────────────────────────────
|
||||
|
||||
func TestMarkNoMergeTables_CaptionAfterTable(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 30, LayoutType: "table"},
|
||||
{X0: 0, X1: 100, Top: 35, Bottom: 50, LayoutType: "table caption", Text: "表1:标题"},
|
||||
}
|
||||
tables := []pdf.TableItem{
|
||||
{Positions: []pdf.Position{{Left: 0, Right: 100, Top: 0, Bottom: 30}}},
|
||||
}
|
||||
MarkNoMergeTables(boxes, tables)
|
||||
if !tables[0].NoMerge {
|
||||
t.Error("table followed by caption should be marked NoMerge")
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarkNoMergeTables_TitleAfterTable(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 30, LayoutType: "table"},
|
||||
{X0: 0, X1: 100, Top: 35, Bottom: 50, LayoutType: "title"},
|
||||
}
|
||||
tables := []pdf.TableItem{
|
||||
{Positions: []pdf.Position{{Left: 0, Right: 100, Top: 0, Bottom: 30}}},
|
||||
}
|
||||
MarkNoMergeTables(boxes, tables)
|
||||
if !tables[0].NoMerge {
|
||||
t.Error("table followed by title should be marked NoMerge")
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarkNoMergeTables_NoCaptionAfter(t *testing.T) {
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 30, LayoutType: "table"},
|
||||
{X0: 0, X1: 100, Top: 35, Bottom: 50, LayoutType: "text"},
|
||||
{X0: 0, X1: 100, Top: 55, Bottom: 70, LayoutType: "table"},
|
||||
}
|
||||
tables := []pdf.TableItem{
|
||||
{Positions: []pdf.Position{{Left: 0, Right: 100, Top: 0, Bottom: 30}}},
|
||||
{Positions: []pdf.Position{{Left: 0, Right: 100, Top: 55, Bottom: 70}}},
|
||||
}
|
||||
MarkNoMergeTables(boxes, tables)
|
||||
if tables[0].NoMerge {
|
||||
t.Error("table followed by text should NOT be marked NoMerge")
|
||||
}
|
||||
if tables[1].NoMerge {
|
||||
t.Error("last table should NOT be marked NoMerge")
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarkNoMergeTables_StaleLastTableTI(t *testing.T) {
|
||||
// Scenario: table box that does NOT overlap any pdf.TableItem.Position
|
||||
// should reset lastTableTI. Otherwise the next caption marks the
|
||||
// wrong (non-adjacent) table as NoMerge.
|
||||
// Box 0: "table", overlaps table[0] → lastTableTI = 0
|
||||
// Box 1: "table", no overlap → lastTableTI should reset to -1
|
||||
// Box 2: "title" → should be a no-op (no adjacent table)
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 30, LayoutType: "table"},
|
||||
{X0: 500, X1: 600, Top: 100, Bottom: 130, LayoutType: "table"}, // far away, no overlap
|
||||
{X0: 0, X1: 100, Top: 140, Bottom: 160, LayoutType: "title"},
|
||||
}
|
||||
tables := []pdf.TableItem{
|
||||
{Positions: []pdf.Position{{Left: 0, Right: 100, Top: 0, Bottom: 30}}}, // table 0
|
||||
{Positions: []pdf.Position{{Left: 0, Right: 100, Top: 35, Bottom: 65}}}, // table 1 — box 0 doesn't overlap this either
|
||||
}
|
||||
MarkNoMergeTables(boxes, tables)
|
||||
// table[0] should NOT be NoMerge: the title follows a non-matching
|
||||
// table box, not table[0] directly.
|
||||
if tables[0].NoMerge {
|
||||
t.Error("stale lastTableTI: table[0] incorrectly marked NoMerge — " +
|
||||
"the non-overlapping table box (box 1) should have reset lastTableTI")
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarkNoMergeTables_EmptyInputs(t *testing.T) {
|
||||
// Should not panic with empty inputs.
|
||||
MarkNoMergeTables(nil, nil)
|
||||
MarkNoMergeTables([]pdf.TextBox{}, []pdf.TableItem{})
|
||||
}
|
||||
112
internal/deepdoc/parser/pdf/table/table_cell_spatial_test.go
Normal file
112
internal/deepdoc/parser/pdf/table/table_cell_spatial_test.go
Normal file
@@ -0,0 +1,112 @@
|
||||
package table
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// ---- boxOverlapsCell ----
|
||||
|
||||
func TestBoxMatchesCell_FullOverlap(t *testing.T) {
|
||||
// Box is entirely inside cell → ≥85% of box area inside cell → match.
|
||||
cell := pdf.TSRCell{X0: 0, Y0: 0, X1: 100, Y1: 50}
|
||||
box := pdf.TextBox{X0: 0, X1: 100, Top: 0, Bottom: 50, Text: "hello"}
|
||||
if !BoxMatchesCell(cell, box, false) {
|
||||
t.Error("full overlap should return true")
|
||||
}
|
||||
// Box is still entirely inside cell → box→cell = 100% ≥ 85% → match.
|
||||
box2 := pdf.TextBox{X0: 10, X1: 90, Top: 10, Bottom: 40, Text: "partial"}
|
||||
if !BoxMatchesCell(cell, box2, false) {
|
||||
t.Error("box entirely inside cell (100% of box) should match")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBoxMatchesCell_NoOverlap(t *testing.T) {
|
||||
cell := pdf.TSRCell{X0: 0, Y0: 0, X1: 100, Y1: 50}
|
||||
box := pdf.TextBox{X0: 200, X1: 300, Top: 10, Bottom: 40, Text: "away"}
|
||||
if BoxMatchesCell(cell, box, false) {
|
||||
t.Error("no X overlap should return false")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBoxMatchesCell_PartialOverlap(t *testing.T) {
|
||||
// Box is entirely inside cell (100% of box area) → matches.
|
||||
// boxOverlapsCell uses box→cell overlap (≥85% of box area inside cell).
|
||||
cell := pdf.TSRCell{X0: 0, Y0: 0, X1: 100, Y1: 50}
|
||||
box := pdf.TextBox{X0: 0, X1: 30, Top: 0, Bottom: 25, Text: "small"}
|
||||
if !BoxMatchesCell(cell, box, false) {
|
||||
t.Error("box entirely inside cell should match")
|
||||
}
|
||||
// Box straddles cell boundary (< 85% of box inside cell) → no match.
|
||||
box2 := pdf.TextBox{X0: 80, X1: 180, Top: 0, Bottom: 25, Text: "spill"}
|
||||
if BoxMatchesCell(cell, box2, false) {
|
||||
t.Error("box straddling boundary (<85% inside) should NOT match")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBoxMatchesCell_ZeroArea(t *testing.T) {
|
||||
cell := pdf.TSRCell{X0: 0, Y0: 0, X1: 0, Y1: 50}
|
||||
box := pdf.TextBox{X0: 0, X1: 10, Top: 0, Bottom: 10, Text: "x"}
|
||||
if BoxMatchesCell(cell, box, false) {
|
||||
t.Error("zero cell area should return false")
|
||||
}
|
||||
}
|
||||
|
||||
// ---- fillCellTextFromBoxes ----
|
||||
|
||||
func TestFillCellTextFromBoxes_Simple(t *testing.T) {
|
||||
// Box covering entire cell (>85%) → match
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 50},
|
||||
{X0: 100, Y0: 0, X1: 200, Y1: 50},
|
||||
}
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 50, Text: "cell1"},
|
||||
{X0: 100, X1: 200, Top: 0, Bottom: 50, Text: "cell2"},
|
||||
}
|
||||
FillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "cell1" {
|
||||
t.Errorf("cell 0: got %q, want 'cell1'", cells[0].Text)
|
||||
}
|
||||
if cells[1].Text != "cell2" {
|
||||
t.Errorf("cell 1: got %q, want 'cell2'", cells[1].Text)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFillCellTextFromBoxes_MultipleBoxesPerCell(t *testing.T) {
|
||||
// Two boxes, each covering >85% of the cell → concatenated
|
||||
// (boxes must overlap the cell near-completely to match individually)
|
||||
cells := []pdf.TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 50}}
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 95, Top: 0, Bottom: 47, Text: "part1"},
|
||||
{X0: 5, X1: 100, Top: 3, Bottom: 50, Text: "part2"},
|
||||
}
|
||||
FillCellTextFromBoxes(cells, boxes)
|
||||
// Both boxes cover >85% → both match → concatenated with space
|
||||
if cells[0].Text == "" {
|
||||
t.Error("expected non-empty cell text")
|
||||
}
|
||||
}
|
||||
|
||||
func TestFillCellTextFromBoxes_EmptyBoxText(t *testing.T) {
|
||||
cells := []pdf.TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 50}}
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 5, X1: 95, Top: 5, Bottom: 45, Text: " "},
|
||||
}
|
||||
FillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "" {
|
||||
t.Errorf("empty box text: got %q, want empty", cells[0].Text)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFillCellTextFromBoxes_NoMatchingBox(t *testing.T) {
|
||||
cells := []pdf.TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 50}}
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 500, X1: 600, Top: 500, Bottom: 550, Text: "far away"},
|
||||
}
|
||||
FillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "" {
|
||||
t.Errorf("no match: got %q, want empty", cells[0].Text)
|
||||
}
|
||||
}
|
||||
@@ -1,8 +1,10 @@
|
||||
package parser
|
||||
package table
|
||||
|
||||
import (
|
||||
"log/slog"
|
||||
"math"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"ragflow/internal/deepdoc/parser/pdf/util"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strings"
|
||||
@@ -10,12 +12,15 @@ import (
|
||||
|
||||
// ── TSR cell grouping ──────────────────────────────────────────────────
|
||||
|
||||
func groupTSRCellsToRows(cells []TSRCell) [][]TSRCell {
|
||||
// GroupTSRCellsToRows groups TSR cells into rows by Y proximity.
|
||||
// This is the basic fallback grouping used when model-specific grouping
|
||||
// (e.g. EE label-aware grouping) is not applicable.
|
||||
func GroupTSRCellsToRows(cells []pdf.TSRCell) [][]pdf.TSRCell {
|
||||
if len(cells) == 0 {
|
||||
return nil
|
||||
}
|
||||
if len(cells) == 1 {
|
||||
return [][]TSRCell{{cells[0]}}
|
||||
return [][]pdf.TSRCell{{cells[0]}}
|
||||
}
|
||||
heights := make([]float64, len(cells))
|
||||
for i, c := range cells {
|
||||
@@ -35,8 +40,8 @@ func groupTSRCellsToRows(cells []TSRCell) [][]TSRCell {
|
||||
return cells[i].Y0 < cells[j].Y0
|
||||
})
|
||||
|
||||
var rows [][]TSRCell
|
||||
var curRow []TSRCell
|
||||
var rows [][]pdf.TSRCell
|
||||
var curRow []pdf.TSRCell
|
||||
curY := 0.0
|
||||
for _, c := range cells {
|
||||
if len(curRow) == 0 {
|
||||
@@ -46,7 +51,7 @@ func groupTSRCellsToRows(cells []TSRCell) [][]TSRCell {
|
||||
}
|
||||
if c.Y0-curY > rowThreshold {
|
||||
rows = append(rows, curRow)
|
||||
curRow = []TSRCell{c}
|
||||
curRow = []pdf.TSRCell{c}
|
||||
curY = c.Y0
|
||||
} else {
|
||||
curRow = append(curRow, c)
|
||||
@@ -63,7 +68,7 @@ func groupTSRCellsToRows(cells []TSRCell) [][]TSRCell {
|
||||
|
||||
// ── cell text filling ──────────────────────────────────────────────────
|
||||
|
||||
func fillCellTextFromBoxes(cells []TSRCell, boxes []TextBox) {
|
||||
func FillCellTextFromBoxes(cells []pdf.TSRCell, boxes []pdf.TextBox) {
|
||||
slog.Debug("fillCellTextFromBoxes", "cells", len(cells), "boxes", len(boxes))
|
||||
if len(cells) > 0 && len(boxes) > 0 {
|
||||
c0 := cells[0]
|
||||
@@ -75,10 +80,10 @@ func fillCellTextFromBoxes(cells []TSRCell, boxes []TextBox) {
|
||||
for ci := range cells {
|
||||
var matches []string
|
||||
for _, b := range boxes {
|
||||
if isCaptionBox(b.Text, b.LayoutType) {
|
||||
if IsCaptionBox(b.Text, b.LayoutType) {
|
||||
continue
|
||||
}
|
||||
if boxMatchesCell(cells[ci], b, cells[ci].Text == "") {
|
||||
if BoxMatchesCell(cells[ci], b, cells[ci].Text == "") {
|
||||
matched++
|
||||
t := strings.TrimSpace(b.Text)
|
||||
if t != "" {
|
||||
@@ -99,9 +104,9 @@ func fillCellTextFromBoxes(cells []TSRCell, boxes []TextBox) {
|
||||
// must be mostly inside the cell (≥85% of box area). When the cell
|
||||
// is empty, any overlap suffices — matching Python's _table_transformer_job
|
||||
// which fills cells from overlapping PDF boxes with thr=0.3.
|
||||
func boxMatchesCell(cell TSRCell, box TextBox, cellIsEmpty bool) bool {
|
||||
inter := OverlapInter(&cell, &box)
|
||||
boxArea := Area(&box)
|
||||
func BoxMatchesCell(cell pdf.TSRCell, box pdf.TextBox, cellIsEmpty bool) bool {
|
||||
inter := util.OverlapInter(&cell, &box)
|
||||
boxArea := util.Area(&box)
|
||||
if boxArea <= 0 {
|
||||
return false
|
||||
}
|
||||
@@ -111,17 +116,11 @@ func boxMatchesCell(cell TSRCell, box TextBox, cellIsEmpty bool) bool {
|
||||
return inter/boxArea >= 0.85
|
||||
}
|
||||
|
||||
// boxOverlapsCell is kept for backward compat — same as boxMatchesCell
|
||||
// with cellIsEmpty=false (strict 85% threshold).
|
||||
func boxOverlapsCell(cell TSRCell, box TextBox) bool {
|
||||
return boxMatchesCell(cell, box, false)
|
||||
}
|
||||
|
||||
// isCaptionBox checks if a text box is a table/figure caption,
|
||||
// matching Python is_caption(). Captions should not enter table cells.
|
||||
var reCaption = regexp.MustCompile(`^[图表]+[ 0-9::]{2,}|(?i)Fig\.?\s*\d+|(?i)Figure\s+\d+|(?i)Table\s+\d+`)
|
||||
|
||||
func isCaptionBox(text string, layoutType string) bool {
|
||||
func IsCaptionBox(text string, layoutType string) bool {
|
||||
if strings.Contains(layoutType, "caption") {
|
||||
return true
|
||||
}
|
||||
@@ -138,25 +137,25 @@ var reFigureCaptionText = regexp.MustCompile(`^图|(?i)Fig\.?\s*\d+|(?i)Figure\s
|
||||
// captionKind returns "table" if the section is a table caption,
|
||||
// "figure" if a figure caption, or "" if not a caption.
|
||||
// Matches Python's is_caption check: text patterns OR layout_type containing "caption".
|
||||
func captionKind(s Section) string {
|
||||
func CaptionKind(s pdf.Section) string {
|
||||
lt := s.LayoutType
|
||||
if lt == DLALabelTableCaption || (strings.Contains(lt, "caption") && reTableCaptionText.MatchString(strings.TrimSpace(s.Text))) {
|
||||
return LayoutTypeTable
|
||||
if lt == pdf.DLALabelTableCaption || (strings.Contains(lt, "caption") && reTableCaptionText.MatchString(strings.TrimSpace(s.Text))) {
|
||||
return pdf.LayoutTypeTable
|
||||
}
|
||||
if lt == DLALabelFigureCaption || strings.Contains(lt, "caption") {
|
||||
return LayoutTypeFigure
|
||||
if lt == pdf.DLALabelFigureCaption || strings.Contains(lt, "caption") {
|
||||
return pdf.LayoutTypeFigure
|
||||
}
|
||||
// DLA may label captions as "text" or other types — check text patterns.
|
||||
t := strings.TrimSpace(s.Text)
|
||||
if reTableCaptionText.MatchString(t) {
|
||||
return LayoutTypeTable
|
||||
return pdf.LayoutTypeTable
|
||||
}
|
||||
if reFigureCaptionText.MatchString(t) {
|
||||
return LayoutTypeFigure
|
||||
return pdf.LayoutTypeFigure
|
||||
}
|
||||
// "图表" pattern could be either — check if isCaptionBox matches.
|
||||
if isCaptionBox(t, "") {
|
||||
return LayoutTypeTable
|
||||
if IsCaptionBox(t, "") {
|
||||
return pdf.LayoutTypeTable
|
||||
}
|
||||
return ""
|
||||
}
|
||||
@@ -192,7 +191,7 @@ var blockTypePatterns = []struct {
|
||||
// TableStructureRecognizer.blockType. Types: Dt (date), Nu (numeric),
|
||||
// Ca (categorical), En (English), NE (named entity), Sg (single char),
|
||||
// Tx (short text), Lx (long text), Nr (person name), Ot (other).
|
||||
func blockType(text string) string {
|
||||
func BlockType(text string) string {
|
||||
t := strings.TrimSpace(text)
|
||||
for _, p := range blockTypePatterns {
|
||||
if p.re.MatchString(t) {
|
||||
@@ -218,7 +217,7 @@ func blockType(text string) string {
|
||||
func simpleTokenCount(text string) int {
|
||||
count := 0
|
||||
for _, r := range text {
|
||||
if isCJK(r) {
|
||||
if pdf.IsCJK(r) {
|
||||
count++
|
||||
} else if r == ' ' || r == '\t' {
|
||||
// whitespace tokenizes boundaries already counted via words
|
||||
@@ -236,7 +235,7 @@ func simpleTokenCount(text string) int {
|
||||
|
||||
func containsCJK(s string) bool {
|
||||
for _, r := range s {
|
||||
if isCJK(r) {
|
||||
if pdf.IsCJK(r) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
@@ -246,14 +245,14 @@ func containsCJK(s string) bool {
|
||||
// headerSetWithBlockType returns rows that should be header rows, using both
|
||||
// TSR cell labels AND block-type classification. Matches Python's
|
||||
// construct_table header detection (table_structure_recognizer.py:370-384).
|
||||
func headerSetWithBlockType(rows [][]TSRCell) map[int]bool {
|
||||
func HeaderSetWithBlockType(rows [][]pdf.TSRCell) map[int]bool {
|
||||
// Compute dominant block type across all cells.
|
||||
typeCounts := make(map[string]int)
|
||||
for _, row := range rows {
|
||||
for _, cell := range row {
|
||||
t := strings.TrimSpace(cell.Text)
|
||||
if t != "" {
|
||||
typeCounts[blockType(t)]++
|
||||
typeCounts[BlockType(t)]++
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -275,7 +274,7 @@ func headerSetWithBlockType(rows [][]TSRCell) map[int]bool {
|
||||
continue
|
||||
}
|
||||
cnt++
|
||||
bt := blockType(t)
|
||||
bt := BlockType(t)
|
||||
// Python: if max_type == "Nu" and cell btype == "Nu" → skip
|
||||
if maxType == "Nu" && bt == "Nu" {
|
||||
continue
|
||||
235
internal/deepdoc/parser/pdf/table/table_cells_group_test.go
Normal file
235
internal/deepdoc/parser/pdf/table/table_cells_group_test.go
Normal file
@@ -0,0 +1,235 @@
|
||||
package table
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
func cellTexts(cells []pdf.TSRCell) []string {
|
||||
out := make([]string, len(cells))
|
||||
for i, c := range cells {
|
||||
out[i] = c.Text
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func TestGroupTSRCellsToRows(t *testing.T) {
|
||||
t.Run("empty", func(t *testing.T) {
|
||||
if rows := GroupTSRCellsToRows(nil); rows != nil {
|
||||
t.Error("nil → nil")
|
||||
}
|
||||
if rows := GroupTSRCellsToRows([]pdf.TSRCell{}); rows != nil {
|
||||
t.Error("empty → nil")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("single cell", func(t *testing.T) {
|
||||
cells := []pdf.TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 50, Text: "A"}}
|
||||
rows := GroupTSRCellsToRows(cells)
|
||||
if len(rows) != 1 || rows[0][0].Text != "A" {
|
||||
t.Error("single cell not preserved")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("two rows two cols", func(t *testing.T) {
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 50, Y1: 30, Text: "A"},
|
||||
{X0: 50, Y0: 0, X1: 100, Y1: 30, Text: "B"},
|
||||
{X0: 0, Y0: 50, X1: 50, Y1: 80, Text: "C"},
|
||||
{X0: 50, Y0: 50, X1: 100, Y1: 80, Text: "D"},
|
||||
}
|
||||
rows := GroupTSRCellsToRows(cells)
|
||||
if len(rows) != 2 {
|
||||
t.Fatalf("2 rows expected, got %d", len(rows))
|
||||
}
|
||||
if rows[0][0].Text != "A" || rows[0][1].Text != "B" {
|
||||
t.Errorf("row0: %v", cellTexts(rows[0]))
|
||||
}
|
||||
if rows[1][0].Text != "C" || rows[1][1].Text != "D" {
|
||||
t.Errorf("row1: %v", cellTexts(rows[1]))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("unsorted input", func(t *testing.T) {
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 50, Y0: 50, X1: 100, Y1: 80, Text: "D"},
|
||||
{X0: 0, Y0: 0, X1: 50, Y1: 30, Text: "A"},
|
||||
{X0: 0, Y0: 50, X1: 50, Y1: 80, Text: "C"},
|
||||
{X0: 50, Y0: 0, X1: 100, Y1: 30, Text: "B"},
|
||||
}
|
||||
rows := GroupTSRCellsToRows(cells)
|
||||
if len(rows) != 2 {
|
||||
t.Fatalf("unsorted: 2 rows expected, got %d", len(rows))
|
||||
}
|
||||
if rows[0][0].Text != "A" || rows[0][1].Text != "B" {
|
||||
t.Errorf("unsorted row0: %v", cellTexts(rows[0]))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("tall merged cell", func(t *testing.T) {
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 50, Y1: 100, Text: "merged"},
|
||||
{X0: 50, Y0: 0, X1: 100, Y1: 30, Text: "B"},
|
||||
{X0: 50, Y0: 50, X1: 100, Y1: 80, Text: "D"},
|
||||
}
|
||||
rows := GroupTSRCellsToRows(cells)
|
||||
// merged cell starts Y0=0 → row 0; Y0=50 cell → row 1
|
||||
if len(rows) != 2 {
|
||||
t.Fatalf("merged cell: 2 rows expected, got %d", len(rows))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("large gap different rows", func(t *testing.T) {
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 50, Y1: 30, Text: "top"},
|
||||
{X0: 0, Y0: 200, X1: 50, Y1: 230, Text: "far"},
|
||||
}
|
||||
rows := GroupTSRCellsToRows(cells)
|
||||
if len(rows) != 2 {
|
||||
t.Fatalf("large gap: 2 rows expected, got %d", len(rows))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("close rows", func(t *testing.T) {
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 10, Y1: 8, Text: "Row1"},
|
||||
{X0: 0, Y0: 9, X1: 10, Y1: 17, Text: "Row2"},
|
||||
}
|
||||
rows := GroupTSRCellsToRows(cells)
|
||||
if len(rows) != 2 {
|
||||
t.Errorf("close rows: expected 2, got %d", len(rows))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("varying heights", func(t *testing.T) {
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 10, Y1: 5, Text: "A"},
|
||||
{X0: 0, Y0: 50, X1: 10, Y1: 70, Text: "B"},
|
||||
{X0: 0, Y0: 50, X1: 10, Y1: 70, Text: "C"},
|
||||
}
|
||||
rows := GroupTSRCellsToRows(cells)
|
||||
if len(rows) != 2 {
|
||||
t.Fatalf("varying heights: expected 2 rows, got %d", len(rows))
|
||||
}
|
||||
if len(rows[0]) != 1 || rows[0][0].Text != "A" {
|
||||
t.Errorf("row 0: expected [A], got %v", cellTexts(rows[0]))
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// ── fillCellTextFromBoxes ──────────────────────────────────────────────
|
||||
|
||||
func TestFillCellTextFromBoxes(t *testing.T) {
|
||||
t.Run("exact match", func(t *testing.T) {
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 50},
|
||||
{X0: 100, Y0: 0, X1: 200, Y1: 50},
|
||||
}
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 50, Text: "A"},
|
||||
{X0: 100, X1: 200, Top: 0, Bottom: 50, Text: "B"},
|
||||
}
|
||||
FillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "A" || cells[1].Text != "B" {
|
||||
t.Errorf("got %q/%q, want A/B", cells[0].Text, cells[1].Text)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("empty cells", func(t *testing.T) {
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 50},
|
||||
{X0: 100, Y0: 0, X1: 200, Y1: 50},
|
||||
}
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 0, X1: 100, Top: 0, Bottom: 50, Text: "only first"},
|
||||
}
|
||||
FillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "only first" {
|
||||
t.Errorf("cell[0]: got %q", cells[0].Text)
|
||||
}
|
||||
if cells[1].Text != "" {
|
||||
t.Errorf("cell[1] should be empty, got %q", cells[1].Text)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("partial cell coverage — empty cell filled from any overlapping box", func(t *testing.T) {
|
||||
// Box covers 40% of cell area. Old code rejected (<85% cell coverage).
|
||||
// New code: cell is empty → accepts box (≥30% box area inside cell).
|
||||
cells := []pdf.TSRCell{{X0: 0, Y0: 0, X1: 200, Y1: 50}}
|
||||
boxes := []pdf.TextBox{{X0: 0, X1: 80, Top: 0, Bottom: 50, Text: "partial"}}
|
||||
FillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "partial" {
|
||||
t.Errorf("empty cell should be filled from overlapping box, got %q", cells[0].Text)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("box inside cell >85%", func(t *testing.T) {
|
||||
cells := []pdf.TSRCell{{X0: 0, Y0: 0, X1: 500, Y1: 300}}
|
||||
boxes := []pdf.TextBox{{X0: 10, X1: 490, Top: 10, Bottom: 290, Text: "inside"}}
|
||||
FillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "inside" {
|
||||
t.Errorf("got %q", cells[0].Text)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("concatenate two boxes to same cell", func(t *testing.T) {
|
||||
cells := []pdf.TSRCell{{X0: 0, Y0: 0, X1: 200, Y1: 100}}
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 5, X1: 195, Top: 2, Bottom: 98, Text: "hello"},
|
||||
{X0: 5, X1: 195, Top: 2, Bottom: 98, Text: "world"},
|
||||
}
|
||||
FillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "hello world" {
|
||||
t.Errorf("got %q, want 'hello world'", cells[0].Text)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("empty inputs", func(t *testing.T) {
|
||||
FillCellTextFromBoxes(nil, nil)
|
||||
FillCellTextFromBoxes([]pdf.TSRCell{}, []pdf.TextBox{})
|
||||
c := []pdf.TSRCell{{X0: 0, Y0: 0, X1: 1, Y1: 1}}
|
||||
FillCellTextFromBoxes(c, nil)
|
||||
if c[0].Text != "" {
|
||||
t.Error("no boxes → text empty")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// ── enrichWithDeepDoc noop ─────────────────────────────────────────────
|
||||
|
||||
func TestGroupTSRCellsToRows_SameHeight(t *testing.T) {
|
||||
// All cells have identical height → medianH is that value → threshold = medianH/2
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 50, Y1: 30, Text: "A"},
|
||||
{X0: 50, Y0: 0, X1: 100, Y1: 30, Text: "B"},
|
||||
{X0: 0, Y0: 31, X1: 50, Y1: 61, Text: "C"}, // gap = 31-30=1 < 30/2=15 → same row? NO, Y0=31 is right at edge
|
||||
}
|
||||
rows := GroupTSRCellsToRows(cells)
|
||||
// medianH=30, threshold=15. C.Y0=31 > curY+threshold?" curY=0, 31 > 15 → new row.
|
||||
// So A,B in row 0, C in row 1.
|
||||
if len(rows) != 2 {
|
||||
t.Fatalf("expected 2 rows, got %d", len(rows))
|
||||
}
|
||||
if len(rows[0]) != 2 || len(rows[1]) != 1 {
|
||||
t.Errorf("row sizes: %d %d, want 2 1", len(rows[0]), len(rows[1]))
|
||||
}
|
||||
}
|
||||
|
||||
func TestFillCellTextFromBoxes_WhitespaceTrim(t *testing.T) {
|
||||
cells := []pdf.TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 100}}
|
||||
boxes := []pdf.TextBox{{X0: 0, X1: 100, Top: 0, Bottom: 100, Text: " hello "}}
|
||||
FillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "hello" {
|
||||
t.Errorf("got %q, want 'hello'", cells[0].Text)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFillCellTextFromBoxes_EmptyBoxIgnored(t *testing.T) {
|
||||
cells := []pdf.TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 100}}
|
||||
boxes := []pdf.TextBox{{X0: 0, X1: 100, Top: 0, Bottom: 100, Text: " "}} // all whitespace
|
||||
FillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "" {
|
||||
t.Errorf("whitespace text should produce empty, got %q", cells[0].Text)
|
||||
}
|
||||
}
|
||||
907
internal/deepdoc/parser/pdf/table/table_construct.go
Normal file
907
internal/deepdoc/parser/pdf/table/table_construct.go
Normal file
@@ -0,0 +1,907 @@
|
||||
package table
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// ── construct table ─────────────────────────────────────────────────────
|
||||
|
||||
// MergeTablesAcrossPages merges TableItems on consecutive pages with
|
||||
// overlapping X and close Y proximity. Matches Python's
|
||||
// _extract_table_figure table merge (pdf_parser.py:1061-1080).
|
||||
func MergeTablesAcrossPages(tables []pdf.TableItem, medianHeights map[int]float64) []pdf.TableItem {
|
||||
if len(tables) <= 1 {
|
||||
return tables
|
||||
}
|
||||
// Sort by position for deterministic adjacency.
|
||||
type indexed struct {
|
||||
idx int
|
||||
pg int
|
||||
top float64
|
||||
}
|
||||
var items []indexed
|
||||
for i, tbl := range tables {
|
||||
if len(tbl.Positions) == 0 {
|
||||
continue
|
||||
}
|
||||
p := tbl.Positions[0]
|
||||
pg := 0
|
||||
if len(p.PageNumbers) > 0 {
|
||||
pg = p.PageNumbers[0]
|
||||
}
|
||||
items = append(items, indexed{i, pg, p.Top})
|
||||
}
|
||||
sort.Slice(items, func(a, b int) bool {
|
||||
if items[a].pg != items[b].pg {
|
||||
return items[a].pg < items[b].pg
|
||||
}
|
||||
return items[a].top < items[b].top
|
||||
})
|
||||
|
||||
merged := make([]bool, len(tables))
|
||||
var result []pdf.TableItem
|
||||
|
||||
for _, it := range items {
|
||||
if merged[it.idx] {
|
||||
continue
|
||||
}
|
||||
anchor := tables[it.idx]
|
||||
merged[it.idx] = true
|
||||
|
||||
// Python nomerge_lout_no: tables whose box is followed by a
|
||||
// caption/title/reference should not be merged cross-page.
|
||||
if anchor.NoMerge {
|
||||
result = append(result, anchor)
|
||||
continue
|
||||
}
|
||||
|
||||
anchorPg := it.pg
|
||||
anchorBott := anchor.Positions[0].Bottom
|
||||
|
||||
// Look for consecutive-page continuations.
|
||||
for _, jt := range items {
|
||||
if merged[jt.idx] || jt.pg <= anchorPg {
|
||||
continue
|
||||
}
|
||||
// Python nomerge_lout_no: skip continuation candidates
|
||||
// tagged as no-merge.
|
||||
if tables[jt.idx].NoMerge {
|
||||
continue
|
||||
}
|
||||
if jt.pg-anchorPg > 1 {
|
||||
break // pages must be consecutive
|
||||
}
|
||||
if len(tables[jt.idx].Positions) == 0 {
|
||||
continue
|
||||
}
|
||||
bp := tables[jt.idx].Positions[0]
|
||||
bpg := 0
|
||||
if len(bp.PageNumbers) > 0 {
|
||||
bpg = bp.PageNumbers[0]
|
||||
}
|
||||
if bpg != anchorPg+1 {
|
||||
continue
|
||||
}
|
||||
// Check X overlap.
|
||||
ap := anchor.Positions[0]
|
||||
if ap.Right < bp.Left || bp.Right < ap.Left {
|
||||
continue
|
||||
}
|
||||
// Check Y proximity: page 1 table top should be close below
|
||||
// page 0 table bottom. Python: y_dis ≤ mh * 23.
|
||||
mh := 10.0
|
||||
if medianHeights != nil {
|
||||
if h, ok := medianHeights[anchorPg]; ok && h > 0 {
|
||||
mh = h
|
||||
}
|
||||
}
|
||||
yDis := (bp.Top + bp.Bottom - anchorBott - ap.Bottom) / 2
|
||||
if yDis > mh*23 {
|
||||
continue
|
||||
}
|
||||
// Merge: combine cells and positions.
|
||||
anchor.Cells = append(anchor.Cells, tables[jt.idx].Cells...)
|
||||
anchor.Positions = append(anchor.Positions, tables[jt.idx].Positions...)
|
||||
if tables[jt.idx].Caption != "" {
|
||||
if anchor.Caption != "" {
|
||||
anchor.Caption += " "
|
||||
}
|
||||
anchor.Caption += tables[jt.idx].Caption
|
||||
}
|
||||
merged[jt.idx] = true
|
||||
anchorPg = bpg
|
||||
anchorBott = bp.Bottom
|
||||
}
|
||||
result = append(result, anchor)
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// constructTable produces an HTML table string from TSR cells and text boxes.
|
||||
// Both cells and boxes must be in the same coordinate space (crop pixel space).
|
||||
// Fills item.Rows so downstream consumers don't need to re-group cells.
|
||||
//
|
||||
// Python equivalent: TableStructureRecognizer.construct_table()
|
||||
// stripCaptionFromCells clears caption-like text from TSR cells.
|
||||
// This catches captions that fillCellTextFromBoxes missed (e.g. text
|
||||
// that doesn't match isCaptionBox patterns like "公司差旅费管理办法").
|
||||
// Only clears cells whose text matches caption patterns or that contain
|
||||
// only number+separator text (pure "1. ", "一、" etc. without data).
|
||||
func StripCaptionFromCells(cells []pdf.TSRCell) {
|
||||
for i := range cells {
|
||||
t := strings.TrimSpace(cells[i].Text)
|
||||
if t == "" {
|
||||
continue
|
||||
}
|
||||
// Clear cells that match caption patterns (e.g. "表1", "Table 1").
|
||||
if IsCaptionBox(t, "") {
|
||||
cells[i].Text = ""
|
||||
}
|
||||
}
|
||||
// Second pass: if the first row (lowest Y) has all-numeric/numbering text
|
||||
// (e.g. "1", "1.", "一"), it's likely a caption numbering line — clear it.
|
||||
// But don't clear actual numeric data cells.
|
||||
// This pass is intentionally conservative — only clears clearly-non-data text.
|
||||
}
|
||||
|
||||
func ConstructTable(cells []pdf.TSRCell, boxes []pdf.TextBox, caption string, item *pdf.TableItem) string {
|
||||
// Strip caption-like text from cells (defense-in-depth: fillCellTextFromBoxes
|
||||
// may include caption text that doesn't match isCaptionBox patterns).
|
||||
StripCaptionFromCells(cells)
|
||||
|
||||
// Use the pre-computed grid from pdf.TableBuilder.GroupCells.
|
||||
// Falls back to cell-level grouping only when called directly by
|
||||
// tests without a pre-computed Grid (production always sets it).
|
||||
var rows [][]pdf.TSRCell
|
||||
if item != nil {
|
||||
rows = item.Grid
|
||||
}
|
||||
if rows == nil && len(cells) > 0 && HasAnyText(cells) {
|
||||
rows = GroupTSRCellsToRows(cells)
|
||||
}
|
||||
if len(rows) > 0 && HasText(rows) {
|
||||
hdrs := HeaderSetWithBlockType(rows)
|
||||
if item != nil {
|
||||
item.Rows = RowsToStrings(rows)
|
||||
}
|
||||
rows = CleanupOrphanColumns(rows)
|
||||
spanInfo, covered := CalSpans(rows)
|
||||
return RowsToHTML(rows, caption, hdrs, spanInfo, covered)
|
||||
}
|
||||
// Fallback: boxes with R/C annotations.
|
||||
if len(boxes) > 0 && BoxesHaveAnnotations(boxes) {
|
||||
rows := GroupBoxesByRC(boxes)
|
||||
if HasText(rows) {
|
||||
if item != nil {
|
||||
item.Rows = RowsToStrings(rows)
|
||||
}
|
||||
spanInfo, covered := CalSpans(rows)
|
||||
return RowsToHTML(rows, caption, BoxHeaderSet(rows, boxes), spanInfo, covered)
|
||||
}
|
||||
}
|
||||
// Test-only: Y/X coordinate grouping (matching Python construct_table).
|
||||
// Used by table_parity_test.go to verify pipeline with Python boxes.
|
||||
if len(boxes) > 0 && !BoxesHaveAnnotations(boxes) {
|
||||
rows := GroupBoxesByYX(boxes)
|
||||
if HasText(rows) {
|
||||
if item != nil {
|
||||
item.Rows = RowsToStrings(rows)
|
||||
}
|
||||
spanInfo, covered := CalSpans(rows)
|
||||
return RowsToHTML(rows, caption, BoxHeaderSet(rows, boxes), spanInfo, covered)
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// boxHeaderSet returns rows that contain boxes with H annotations.
|
||||
func BoxHeaderSet(rows [][]pdf.TSRCell, boxes []pdf.TextBox) map[int]bool {
|
||||
hdrs := make(map[int]bool)
|
||||
for _, b := range boxes {
|
||||
if b.H > 0 && b.R >= 0 && b.R < len(rows) {
|
||||
hdrs[b.R] = true
|
||||
}
|
||||
}
|
||||
return hdrs
|
||||
}
|
||||
|
||||
func HasAnyText(cells []pdf.TSRCell) bool {
|
||||
for _, c := range cells {
|
||||
if strings.TrimSpace(c.Text) != "" {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// groupBoxesByRC groups text boxes into a cell grid by R/C annotations.
|
||||
// Matches Python's construct_table: sort by R, merge nearby rows by Y proximity,
|
||||
// sort by C within each row, merge nearby columns by X proximity.
|
||||
func GroupBoxesByRC(boxes []pdf.TextBox) [][]pdf.TSRCell {
|
||||
if len(boxes) == 0 {
|
||||
return nil
|
||||
}
|
||||
// If no real R/C annotations (maxR <= 0), fall back to YX coordinate
|
||||
// grouping — matching Python's construct_table when all R=-1.
|
||||
maxR := 0
|
||||
for _, b := range boxes {
|
||||
if b.R > maxR {
|
||||
maxR = b.R
|
||||
}
|
||||
}
|
||||
if maxR <= 0 {
|
||||
return GroupBoxesByYX(boxes)
|
||||
}
|
||||
// Sort by R index first (Python: sort_R_firstly), then Y, then X.
|
||||
sort.Slice(boxes, func(i, j int) bool {
|
||||
if boxes[i].R != boxes[j].R {
|
||||
return boxes[i].R < boxes[j].R
|
||||
}
|
||||
if boxes[i].Top != boxes[j].Top {
|
||||
return boxes[i].Top < boxes[j].Top
|
||||
}
|
||||
return boxes[i].X0 < boxes[j].X0
|
||||
})
|
||||
|
||||
// Compress R indices: Python's sort_R_firstly grouping.
|
||||
// R differs → always a new row. Same R + Y gap → also new row.
|
||||
rowMap := make(map[int]int) // original R → compressed row index
|
||||
compressed := 0
|
||||
rowMap[boxes[0].R] = 0
|
||||
lastR := boxes[0].R
|
||||
btm := boxes[0].Bottom
|
||||
for i := 1; i < len(boxes); i++ {
|
||||
// Python: b["R"] != last_R → new row.
|
||||
// Same R → always same row (Python doesn't check Y for same R).
|
||||
if boxes[i].R != lastR {
|
||||
compressed++
|
||||
rowMap[boxes[i].R] = compressed
|
||||
lastR = boxes[i].R
|
||||
btm = boxes[i].Bottom
|
||||
} else {
|
||||
// Same R → same physical row.
|
||||
rowMap[boxes[i].R] = compressed
|
||||
btm = (btm + boxes[i].Bottom) / 2.0
|
||||
}
|
||||
}
|
||||
|
||||
// Collect boxes per row, sort by C within each row.
|
||||
type rb struct {
|
||||
row, col int
|
||||
txt string
|
||||
x0, y0, x1, y1 float64
|
||||
label string
|
||||
}
|
||||
cmap := make(map[int]map[int]*rb) // row → col → entry
|
||||
maxCols := make(map[int]int)
|
||||
for _, b := range boxes {
|
||||
t := strings.TrimSpace(b.Text)
|
||||
// Keep boxes with SP/H annotations even if text is empty —
|
||||
// their coordinates are needed for colspan/rowspan calculation.
|
||||
if t == "" && b.H <= 0 && b.SP <= 0 {
|
||||
continue
|
||||
}
|
||||
r := rowMap[b.R]
|
||||
c := b.C
|
||||
if cmap[r] == nil {
|
||||
cmap[r] = make(map[int]*rb)
|
||||
}
|
||||
x0, y0, x1, y1, label := cellPosFromBox(b)
|
||||
if v, ok := cmap[r][c]; ok {
|
||||
v.txt += " " + t
|
||||
// Merge spanning coordinates (use widest extent).
|
||||
if b.H > 0 || b.SP > 0 {
|
||||
v.label = cellLabelFromBox(b)
|
||||
if v.x0 > x0 {
|
||||
v.x0 = x0
|
||||
}
|
||||
if v.y0 > y0 {
|
||||
v.y0 = y0
|
||||
}
|
||||
if v.x1 < x1 {
|
||||
v.x1 = x1
|
||||
}
|
||||
if v.y1 < y1 {
|
||||
v.y1 = y1
|
||||
}
|
||||
}
|
||||
} else {
|
||||
cmap[r][c] = &rb{r, c, t, x0, y0, x1, y1, label}
|
||||
}
|
||||
if c > maxCols[r] {
|
||||
maxCols[r] = c
|
||||
}
|
||||
}
|
||||
|
||||
// Compress C indices per row: sort boxes by X0 within the row,
|
||||
// group disjoint X ranges into separate columns. This is equivalent
|
||||
// to Python's sort_C_firstly but uses X0 ordering instead of C labels.
|
||||
cCompressed := make(map[int]map[int]int) // row → (original C → compressed col)
|
||||
cMaxCol := make(map[int]int)
|
||||
for ri := 0; ri <= compressed; ri++ {
|
||||
rowEntries := cmap[ri]
|
||||
if rowEntries == nil {
|
||||
continue
|
||||
}
|
||||
// Collect all boxes in this row, sorted by X0.
|
||||
type rowBox struct {
|
||||
c, idx int
|
||||
x0, x1 float64
|
||||
txt string
|
||||
}
|
||||
var rowBoxes []rowBox
|
||||
for i, b := range boxes {
|
||||
if rowMap[b.R] == ri && (strings.TrimSpace(b.Text) != "" || b.H > 0 || b.SP > 0) {
|
||||
rowBoxes = append(rowBoxes, rowBox{c: b.C, idx: i, x0: b.X0, x1: b.X1, txt: b.Text})
|
||||
}
|
||||
}
|
||||
sort.Slice(rowBoxes, func(i, j int) bool { return rowBoxes[i].x0 < rowBoxes[j].x0 })
|
||||
// Assign compressed column by X-order (disjoint X → new col).
|
||||
cMap := make(map[int]int) // original C → compressed col
|
||||
right := 0.0
|
||||
for _, rb := range rowBoxes {
|
||||
if len(cMap) == 0 || rb.x0 >= right {
|
||||
cc := len(cMap)
|
||||
cMap[rb.c] = cc
|
||||
right = rb.x1
|
||||
} else {
|
||||
// Overlapping X → merge into last column.
|
||||
cMap[rb.c] = len(cMap) - 1
|
||||
if rb.x1 > right {
|
||||
right = rb.x1
|
||||
}
|
||||
}
|
||||
}
|
||||
cCompressed[ri] = cMap
|
||||
cMaxCol[ri] = len(cMap) - 1
|
||||
}
|
||||
|
||||
// Build grid.
|
||||
rows := make([][]pdf.TSRCell, compressed+1)
|
||||
for ri := 0; ri <= compressed; ri++ {
|
||||
maxC := cMaxCol[ri]
|
||||
rows[ri] = make([]pdf.TSRCell, maxC+1)
|
||||
for ci, v := range cmap[ri] {
|
||||
cci := cCompressed[ri][ci]
|
||||
if cci <= maxC {
|
||||
rows[ri][cci].Text = v.txt
|
||||
rows[ri][cci].X0 = v.x0
|
||||
rows[ri][cci].Y0 = v.y0
|
||||
rows[ri][cci].X1 = v.x1
|
||||
rows[ri][cci].Y1 = v.y1
|
||||
rows[ri][cci].Label = v.label
|
||||
}
|
||||
}
|
||||
}
|
||||
return rows
|
||||
}
|
||||
|
||||
// cellPosFromBox returns the position coordinates and label for a cell
|
||||
// derived from a text box. Header cells use HLeft/HRight/HTop/HBott
|
||||
// for spanning-aware positions; regular cells use the box's own bounds.
|
||||
func cellPosFromBox(b pdf.TextBox) (x0, y0, x1, y1 float64, label string) {
|
||||
x0, y0, x1, y1 = b.X0, b.Top, b.X1, b.Bottom
|
||||
if b.H > 0 {
|
||||
label = "table header"
|
||||
if b.HLeft != 0 || b.HRight != 0 {
|
||||
if b.HLeft != 0 {
|
||||
x0 = b.HLeft
|
||||
}
|
||||
if b.HRight != 0 {
|
||||
x1 = b.HRight
|
||||
}
|
||||
}
|
||||
if b.HTop != 0 {
|
||||
y0 = b.HTop
|
||||
}
|
||||
if b.HBott != 0 {
|
||||
y1 = b.HBott
|
||||
}
|
||||
} else if b.SP > 0 {
|
||||
label = "table spanning cell"
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
// cellLabelFromBox returns the TSR label for a box based on H/SP annotations.
|
||||
// Used when merging multiple boxes into one cell — preserves the spanning label.
|
||||
func cellLabelFromBox(b pdf.TextBox) string {
|
||||
if b.H > 0 {
|
||||
return "table header"
|
||||
}
|
||||
if b.SP > 0 {
|
||||
return "table spanning cell"
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// groupBoxesByYX groups boxes into a cell grid by Y/X coordinates,
|
||||
// matching Python's construct_table which uses sort_R_firstly and
|
||||
// sort_C_firstly when R/C annotations are absent.
|
||||
// This is test-only — used by table_parity_test.go to verify pipeline
|
||||
// parity with Python boxes that lack R/C annotations.
|
||||
func GroupBoxesByYX(boxes []pdf.TextBox) [][]pdf.TSRCell {
|
||||
if len(boxes) == 0 {
|
||||
return nil
|
||||
}
|
||||
// Sort by (page, top, x0) — same as Python sort_R_firstly with R=-1.
|
||||
sort.Slice(boxes, func(i, j int) bool {
|
||||
if boxes[i].PageNumber != boxes[j].PageNumber {
|
||||
return boxes[i].PageNumber < boxes[j].PageNumber
|
||||
}
|
||||
if boxes[i].Top != boxes[j].Top {
|
||||
return boxes[i].Top < boxes[j].Top
|
||||
}
|
||||
return boxes[i].X0 < boxes[j].X0
|
||||
})
|
||||
|
||||
// Group into rows by Y proximity (Python's row grouping).
|
||||
type rowGroup struct {
|
||||
boxes []pdf.TextBox
|
||||
top, btm float64
|
||||
}
|
||||
var rowGroups []rowGroup
|
||||
rowGroups = append(rowGroups, rowGroup{boxes: []pdf.TextBox{boxes[0]}, top: boxes[0].Top, btm: boxes[0].Bottom})
|
||||
for i := 1; i < len(boxes); i++ {
|
||||
prev := &rowGroups[len(rowGroups)-1]
|
||||
// Python: same row if top < prev.btm (Y overlaps) and same page.
|
||||
if boxes[i].PageNumber == prev.boxes[0].PageNumber && boxes[i].Top < prev.btm {
|
||||
prev.boxes = append(prev.boxes, boxes[i])
|
||||
if boxes[i].Top < prev.top {
|
||||
prev.top = boxes[i].Top
|
||||
}
|
||||
if boxes[i].Bottom > prev.btm {
|
||||
prev.btm = boxes[i].Bottom
|
||||
}
|
||||
} else {
|
||||
rowGroups = append(rowGroups, rowGroup{boxes: []pdf.TextBox{boxes[i]}, top: boxes[i].Top, btm: boxes[i].Bottom})
|
||||
}
|
||||
}
|
||||
|
||||
// Within each row, group into columns by X proximity.
|
||||
rows := make([][]pdf.TSRCell, len(rowGroups))
|
||||
for ri, rg := range rowGroups {
|
||||
// Sort by X0.
|
||||
sort.Slice(rg.boxes, func(i, j int) bool { return rg.boxes[i].X0 < rg.boxes[j].X0 })
|
||||
// Group by X overlap.
|
||||
var cols []struct {
|
||||
boxes []pdf.TextBox
|
||||
x1 float64
|
||||
}
|
||||
cols = append(cols, struct {
|
||||
boxes []pdf.TextBox
|
||||
x1 float64
|
||||
}{boxes: []pdf.TextBox{rg.boxes[0]}, x1: rg.boxes[0].X1})
|
||||
for i := 1; i < len(rg.boxes); i++ {
|
||||
prev := &cols[len(cols)-1]
|
||||
if rg.boxes[i].X0 < prev.x1 {
|
||||
prev.boxes = append(prev.boxes, rg.boxes[i])
|
||||
if rg.boxes[i].X1 > prev.x1 {
|
||||
prev.x1 = rg.boxes[i].X1
|
||||
}
|
||||
} else {
|
||||
cols = append(cols, struct {
|
||||
boxes []pdf.TextBox
|
||||
x1 float64
|
||||
}{boxes: []pdf.TextBox{rg.boxes[i]}, x1: rg.boxes[i].X1})
|
||||
}
|
||||
}
|
||||
rows[ri] = make([]pdf.TSRCell, len(cols))
|
||||
for ci, col := range cols {
|
||||
var sb strings.Builder
|
||||
for _, b := range col.boxes {
|
||||
t := strings.TrimSpace(b.Text)
|
||||
if t == "" {
|
||||
continue
|
||||
}
|
||||
if sb.Len() > 0 {
|
||||
sb.WriteByte(' ')
|
||||
}
|
||||
sb.WriteString(t)
|
||||
}
|
||||
rows[ri][ci].Text = sb.String()
|
||||
}
|
||||
}
|
||||
return rows
|
||||
}
|
||||
|
||||
func BoxesHaveAnnotations(boxes []pdf.TextBox) bool {
|
||||
maxR, maxC := 0, 0
|
||||
for _, b := range boxes {
|
||||
if b.R > maxR {
|
||||
maxR = b.R
|
||||
}
|
||||
if b.C > maxC {
|
||||
maxC = b.C
|
||||
}
|
||||
}
|
||||
// True if at least 2 rows or 2 cols (R/C are 0-based, so maxR>0 means ≥2 rows).
|
||||
return maxR > 0 || maxC > 0
|
||||
}
|
||||
|
||||
func HasText(rows [][]pdf.TSRCell) bool {
|
||||
for _, row := range rows {
|
||||
for _, c := range row {
|
||||
if strings.TrimSpace(c.Text) != "" {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func RowsToStrings(rows [][]pdf.TSRCell) [][]string {
|
||||
out := make([][]string, len(rows))
|
||||
for ri, row := range rows {
|
||||
out[ri] = make([]string, len(row))
|
||||
for ci, c := range row {
|
||||
out[ri][ci] = c.Text
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// fillCellTextFromAnnotations fills cell text from text boxes using R/C labels.
|
||||
// This matches Python's construct_table which assigns boxes to cells by their
|
||||
// R (row) and C (col) annotations rather than spatial overlap.
|
||||
func FillCellTextFromAnnotations(rows [][]pdf.TSRCell, boxes []pdf.TextBox) {
|
||||
// Build R→(C→text) map: row index → (col index → text).
|
||||
rBoxes := make(map[int]map[int][]string)
|
||||
for _, b := range boxes {
|
||||
if b.Text == "" {
|
||||
continue
|
||||
}
|
||||
if rBoxes[b.R] == nil {
|
||||
rBoxes[b.R] = make(map[int][]string)
|
||||
}
|
||||
rBoxes[b.R][b.C] = append(rBoxes[b.R][b.C], b.Text)
|
||||
}
|
||||
// Fill each cell from the matching R/C position.
|
||||
for ri, row := range rows {
|
||||
colMap := rBoxes[ri]
|
||||
if colMap == nil {
|
||||
continue
|
||||
}
|
||||
// Build sorted column list for positional matching.
|
||||
type colEntry struct {
|
||||
c int
|
||||
texts []string
|
||||
}
|
||||
var cols []colEntry
|
||||
for c, texts := range colMap {
|
||||
cols = append(cols, colEntry{c, texts})
|
||||
}
|
||||
sort.Slice(cols, func(i, j int) bool { return cols[i].c < cols[j].c })
|
||||
for ci, col := range cols {
|
||||
if ci < len(row) {
|
||||
row[ci].Text = strings.TrimSpace(strings.Join(col.texts, " "))
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// dataSourceRe matches table/figure boxes that should be discarded as
|
||||
// data-source attribution lines rather than extracted content.
|
||||
//
|
||||
// Python: pdf_parser.py:1040-1042, 1050-1052
|
||||
//
|
||||
// re.match(r"(数据|资料|图表)*来源[:: ]", self.boxes[i]["text"])
|
||||
var dataSourceRe = regexp.MustCompile(`^(数据|资料|图表)*来源[:: ]`)
|
||||
|
||||
// isDataSourceBox returns true if the box text matches the data-source
|
||||
// discard pattern (Python's _extract_table_figure data-source filter).
|
||||
func isDataSourceBox(text string) bool {
|
||||
return dataSourceRe.MatchString(text)
|
||||
}
|
||||
|
||||
// tableRegionBox returns a pdf.TextBox for a table replacement, using DLA region
|
||||
// boundaries when available (Region* set), falling back to anchor box coordinates.
|
||||
// Python's insert_table_figures uses DLA layout region boundaries; the fallback
|
||||
// handles test TableItems or bare engines without DLA.
|
||||
func tableRegionBox(tbl *pdf.TableItem, ref *pdf.TextBox, html string) pdf.TextBox {
|
||||
pg := 0
|
||||
if len(tbl.Positions) > 0 && len(tbl.Positions[0].PageNumbers) > 0 {
|
||||
pg = tbl.Positions[0].PageNumbers[0]
|
||||
}
|
||||
// Use DLA region boundaries when set.
|
||||
if tbl.RegionLeft != 0 || tbl.RegionRight != 0 || tbl.RegionTop != 0 || tbl.RegionBottom != 0 {
|
||||
return pdf.TextBox{
|
||||
X0: tbl.RegionLeft, X1: tbl.RegionRight,
|
||||
Top: tbl.RegionTop, Bottom: tbl.RegionBottom,
|
||||
Text: html,
|
||||
PageNumber: pg,
|
||||
LayoutType: pdf.LayoutTypeTable,
|
||||
}
|
||||
}
|
||||
// Fallback: use anchor box coordinates.
|
||||
x0, x1, top, bot := ref.X0, ref.X1, ref.Top, ref.Bottom
|
||||
return pdf.TextBox{
|
||||
X0: x0, X1: x1, Top: top, Bottom: bot,
|
||||
Text: html,
|
||||
PageNumber: pg,
|
||||
LayoutType: pdf.LayoutTypeTable,
|
||||
}
|
||||
}
|
||||
|
||||
// minRectangleDistance computes the Euclidean distance between two rectangles.
|
||||
// Returns 0 when rectangles overlap. Matches Python's min_rectangle_distance
|
||||
// in insert_table_figures (pdf_parser.py:1609-1626).
|
||||
func minRectangleDistance(left1, right1, top1, bottom1, left2, right2, top2, bottom2 float64) float64 {
|
||||
if right1 >= left2 && right2 >= left1 && bottom1 >= top2 && bottom2 >= top1 {
|
||||
return 0
|
||||
}
|
||||
var dx, dy float64
|
||||
if right1 < left2 {
|
||||
dx = left2 - right1
|
||||
} else if right2 < left1 {
|
||||
dx = left1 - right2
|
||||
}
|
||||
if bottom1 < top2 {
|
||||
dy = top2 - bottom1
|
||||
} else if bottom2 < top1 {
|
||||
dy = top1 - bottom2
|
||||
}
|
||||
return math.Sqrt(dx*dx + dy*dy)
|
||||
}
|
||||
|
||||
func RowsToHTML(rows [][]pdf.TSRCell, caption string, headerRows map[int]bool, spanInfo map[[2]int][2]int, covered map[[2]int]bool) string {
|
||||
var b strings.Builder
|
||||
b.WriteString("<table>")
|
||||
if caption != "" {
|
||||
b.WriteString("<caption>")
|
||||
b.WriteString(caption)
|
||||
b.WriteString("</caption>")
|
||||
}
|
||||
for ri, row := range rows {
|
||||
b.WriteString("<tr>")
|
||||
for ci, cell := range row {
|
||||
if covered[[2]int{ri, ci}] {
|
||||
continue
|
||||
}
|
||||
tag := "td"
|
||||
if headerRows[ri] {
|
||||
tag = "th"
|
||||
}
|
||||
b.WriteString("<")
|
||||
b.WriteString(tag)
|
||||
sp := ""
|
||||
if s, ok := spanInfo[[2]int{ri, ci}]; ok {
|
||||
if s[0] > 1 {
|
||||
sp = fmt.Sprintf("colspan=%d", s[0])
|
||||
}
|
||||
if s[1] > 1 {
|
||||
if sp != "" {
|
||||
sp += " "
|
||||
}
|
||||
sp += fmt.Sprintf("rowspan=%d", s[1])
|
||||
}
|
||||
}
|
||||
if sp != "" {
|
||||
b.WriteString(" ")
|
||||
b.WriteString(sp)
|
||||
}
|
||||
b.WriteString(" >")
|
||||
b.WriteString(cell.Text)
|
||||
b.WriteString("</")
|
||||
b.WriteString(tag)
|
||||
b.WriteString(">")
|
||||
}
|
||||
b.WriteString("</tr>")
|
||||
}
|
||||
b.WriteString("</table>")
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// ── Span computation (Python: __cal_spans) ──
|
||||
|
||||
// calSpans computes colspan and rowspan for spanning cells in the grid.
|
||||
// Returns spanInfo (row,col → colspan,rowspan) and covered (cells hidden by spans).
|
||||
// Matches Python's __cal_spans (table_structure_recognizer.py:535).
|
||||
// flattenGrid flattens a 2D grid into a 1D slice for fillCellTextFromBoxes.
|
||||
func FlattenGrid(grid [][]pdf.TSRCell) []pdf.TSRCell {
|
||||
n := 0
|
||||
for _, row := range grid {
|
||||
n += len(row)
|
||||
}
|
||||
flat := make([]pdf.TSRCell, 0, n)
|
||||
for _, row := range grid {
|
||||
flat = append(flat, row...)
|
||||
}
|
||||
return flat
|
||||
}
|
||||
|
||||
func CalSpans(rows [][]pdf.TSRCell) (map[[2]int][2]int, map[[2]int]bool) {
|
||||
spanInfo := make(map[[2]int][2]int)
|
||||
covered := make(map[[2]int]bool)
|
||||
if len(rows) == 0 || len(rows[0]) == 0 {
|
||||
return spanInfo, covered
|
||||
}
|
||||
|
||||
// Compute column center positions.
|
||||
nCols := len(rows[0])
|
||||
colLeft := make([]float64, nCols)
|
||||
colRight := make([]float64, nCols)
|
||||
for j := 0; j < nCols; j++ {
|
||||
colLeft[j] = 1e9
|
||||
colRight[j] = -1e9
|
||||
}
|
||||
nRows := len(rows)
|
||||
rowTop := make([]float64, nRows)
|
||||
rowBott := make([]float64, nRows)
|
||||
for i := 0; i < nRows; i++ {
|
||||
rowTop[i] = 1e9
|
||||
rowBott[i] = -1e9
|
||||
}
|
||||
|
||||
for i, row := range rows {
|
||||
for j, cell := range row {
|
||||
if j >= nCols {
|
||||
continue
|
||||
}
|
||||
// Exclude spanning cells from column/row boundary calculations.
|
||||
// Use label-based detection (O(1), no dependency on column midpoints).
|
||||
if strings.Contains(cell.Label, "spanning") {
|
||||
continue
|
||||
}
|
||||
if cell.X0 < colLeft[j] {
|
||||
colLeft[j] = cell.X0
|
||||
}
|
||||
if cell.X1 > colRight[j] {
|
||||
colRight[j] = cell.X1
|
||||
}
|
||||
if cell.Y0 < rowTop[i] {
|
||||
rowTop[i] = cell.Y0
|
||||
}
|
||||
if cell.Y1 > rowBott[i] {
|
||||
rowBott[i] = cell.Y1
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// For each spanning cell, compute how many cols/rows it covers.
|
||||
for i, row := range rows {
|
||||
for j, cell := range row {
|
||||
if j >= nCols || covered[[2]int{i, j}] {
|
||||
continue
|
||||
}
|
||||
// Skip cells without position data (they can't span).
|
||||
if cell.X0 == 0 && cell.X1 == 0 && cell.Y0 == 0 && cell.Y1 == 0 {
|
||||
continue
|
||||
}
|
||||
cs, rs := 1, 1
|
||||
// Count columns whose center is inside this cell's X range.
|
||||
for k := j + 1; k < nCols; k++ {
|
||||
// Skip columns with no non-spanning cells (initial values unchanged).
|
||||
if colLeft[k] == 1e9 && colRight[k] == -1e9 {
|
||||
continue
|
||||
}
|
||||
colCenter := (colLeft[k] + colRight[k]) / 2
|
||||
if colCenter >= cell.X0 && colCenter <= cell.X1 {
|
||||
cs++
|
||||
}
|
||||
}
|
||||
// Count rows whose center is inside this cell's Y range.
|
||||
for k := i + 1; k < nRows; k++ {
|
||||
// Skip rows with no non-spanning cells.
|
||||
if rowTop[k] == 1e9 && rowBott[k] == -1e9 {
|
||||
continue
|
||||
}
|
||||
rowCenter := (rowTop[k] + rowBott[k]) / 2
|
||||
if rowCenter >= cell.Y0 && rowCenter <= cell.Y1 {
|
||||
rs++
|
||||
}
|
||||
}
|
||||
if cs > 1 || rs > 1 {
|
||||
spanInfo[[2]int{i, j}] = [2]int{cs, rs}
|
||||
// Mark covered cells.
|
||||
for ri := i; ri < i+rs && ri < nRows; ri++ {
|
||||
for cj := j; cj < j+cs && cj < nCols; cj++ {
|
||||
if ri != i || cj != j {
|
||||
covered[[2]int{ri, cj}] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return spanInfo, covered
|
||||
}
|
||||
|
||||
// ── Orphan column/row cleanup (Python: construct_table lines 256-368) ──
|
||||
|
||||
// cleanupOrphanColumns removes columns that have only a single non-empty cell
|
||||
// when there are ≥4 rows. Matches Python's construct_table column cleanup.
|
||||
func CleanupOrphanColumns(rows [][]pdf.TSRCell) [][]pdf.TSRCell {
|
||||
if len(rows) < 4 || len(rows) == 0 {
|
||||
return rows
|
||||
}
|
||||
nCols := len(rows[0])
|
||||
|
||||
j := 0
|
||||
colLoop:
|
||||
for j < nCols {
|
||||
e, ii := 0, 0
|
||||
for i := range rows {
|
||||
if j < len(rows[i]) && strings.TrimSpace(rows[i][j].Text) != "" {
|
||||
e++
|
||||
ii = i
|
||||
}
|
||||
if e > 1 {
|
||||
j++
|
||||
continue colLoop
|
||||
}
|
||||
}
|
||||
// Column j has only one non-empty cell at row ii.
|
||||
// Check if adjacent columns have text for this row.
|
||||
f := (j > 0 && j-1 < len(rows[ii]) && strings.TrimSpace(rows[ii][j-1].Text) != "") || j == 0
|
||||
ff := (j+1 < len(rows[ii]) && strings.TrimSpace(rows[ii][j+1].Text) != "") || j+1 >= len(rows[ii])
|
||||
if f && ff {
|
||||
// Both adjacent columns are ok for merging — but this means
|
||||
// there's text on both sides, keep column.
|
||||
j++
|
||||
continue
|
||||
}
|
||||
|
||||
// Determine which side to merge into.
|
||||
left := 1e9
|
||||
right := 1e9
|
||||
if j > 0 && !f {
|
||||
for i := range rows {
|
||||
if j-1 < len(rows[i]) && strings.TrimSpace(rows[i][j-1].Text) != "" {
|
||||
// Distance from orphan cell to left neighbor.
|
||||
if d := rows[ii][j].X0 - rows[i][j-1].X1; d < left {
|
||||
left = d
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if j+1 < nCols && !ff {
|
||||
for i := range rows {
|
||||
if j+1 < len(rows[i]) && strings.TrimSpace(rows[i][j+1].Text) != "" {
|
||||
if d := rows[i][j+1].X0 - rows[ii][j].X1; d < right {
|
||||
right = d
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if left < right && j > 0 {
|
||||
// Merge into left column.
|
||||
for i := range rows {
|
||||
if j-1 < len(rows[i]) && j < len(rows[i]) {
|
||||
if rows[i][j-1].Text == "" {
|
||||
rows[i][j-1].Text = rows[i][j].Text
|
||||
} else if rows[i][j].Text != "" {
|
||||
rows[i][j-1].Text += " " + rows[i][j].Text
|
||||
}
|
||||
}
|
||||
}
|
||||
} else if j+1 < nCols {
|
||||
// Merge into right column.
|
||||
for i := range rows {
|
||||
if j < len(rows[i]) && j+1 < len(rows[i]) {
|
||||
if rows[i][j+1].Text == "" {
|
||||
rows[i][j+1].Text = rows[i][j].Text
|
||||
} else if rows[i][j].Text != "" {
|
||||
rows[i][j+1].Text = rows[i][j].Text + " " + rows[i][j+1].Text
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Remove column j.
|
||||
for i := range rows {
|
||||
if j < len(rows[i]) {
|
||||
rows[i] = append(rows[i][:j], rows[i][j+1:]...)
|
||||
}
|
||||
}
|
||||
nCols--
|
||||
// Don't increment j — the next column shifted into position j.
|
||||
}
|
||||
return rows
|
||||
}
|
||||
1129
internal/deepdoc/parser/pdf/table/table_construct_test.go
Normal file
1129
internal/deepdoc/parser/pdf/table/table_construct_test.go
Normal file
File diff suppressed because it is too large
Load Diff
58
internal/deepdoc/parser/pdf/table/table_coords.go
Normal file
58
internal/deepdoc/parser/pdf/table/table_coords.go
Normal file
@@ -0,0 +1,58 @@
|
||||
package table
|
||||
|
||||
import (
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// ── coordinate space conversion helpers ──────────────────────────────
|
||||
|
||||
// CellToPageSpace converts from crop-pixel space to page-global 72-DPI space.
|
||||
func CellToPageSpace(c pdf.TSRCell, cropOffX, cropOffY, scale float64) pdf.TSRCell {
|
||||
return pdf.TSRCell{
|
||||
X0: (c.X0 + cropOffX) / scale, Y0: (c.Y0 + cropOffY) / scale,
|
||||
X1: (c.X1 + cropOffX) / scale, Y1: (c.Y1 + cropOffY) / scale,
|
||||
Text: c.Text, Label: c.Label,
|
||||
}
|
||||
}
|
||||
|
||||
// CellAddOffset applies a crop offset to cell coordinates (stays in pixel space).
|
||||
func CellAddOffset(c pdf.TSRCell, offX, offY float64) pdf.TSRCell {
|
||||
return pdf.TSRCell{
|
||||
X0: c.X0 + offX, Y0: c.Y0 + offY, X1: c.X1 + offX, Y1: c.Y1 + offY,
|
||||
Text: c.Text, Label: c.Label,
|
||||
}
|
||||
}
|
||||
|
||||
// CellSliceToPageSpace converts a slice of cells from crop-pixel to page DPI space.
|
||||
func CellSliceToPageSpace(cells []pdf.TSRCell, cropOffX, cropOffY, scale float64) []pdf.TSRCell {
|
||||
out := make([]pdf.TSRCell, len(cells))
|
||||
for i, c := range cells {
|
||||
out[i] = CellToPageSpace(c, cropOffX, cropOffY, scale)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// BoxToCropSpace converts a pdf.TextBox from PDF-point space to crop-pixel space.
|
||||
func BoxToCropSpace(b pdf.TextBox, scale, cropOffX, cropOffY float64) pdf.TextBox {
|
||||
return pdf.TextBox{
|
||||
X0: b.X0*scale - cropOffX, X1: b.X1*scale - cropOffX,
|
||||
Top: b.Top*scale - cropOffY, Bottom: b.Bottom*scale - cropOffY,
|
||||
Text: b.Text,
|
||||
}
|
||||
}
|
||||
|
||||
// CopyBoxAnnotations copies the DLA/TSR annotation fields from src to dst.
|
||||
func CopyBoxAnnotations(dst, src *pdf.TextBox) {
|
||||
dst.R = src.R
|
||||
dst.C = src.C
|
||||
dst.RTop = src.RTop
|
||||
dst.RBott = src.RBott
|
||||
dst.H = src.H
|
||||
dst.HTop = src.HTop
|
||||
dst.HBott = src.HBott
|
||||
dst.HLeft = src.HLeft
|
||||
dst.HRight = src.HRight
|
||||
dst.CLeft = src.CLeft
|
||||
dst.CRight = src.CRight
|
||||
dst.SP = src.SP
|
||||
}
|
||||
75
internal/deepdoc/parser/pdf/table/table_coords_test.go
Normal file
75
internal/deepdoc/parser/pdf/table/table_coords_test.go
Normal file
@@ -0,0 +1,75 @@
|
||||
package table
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
func TestCellToPageSpace(t *testing.T) {
|
||||
cell := pdf.TSRCell{X0: 100, Y0: 200, X1: 300, Y1: 400, Text: "hello", Label: "table"}
|
||||
got := CellToPageSpace(cell, 15, 25, 3.0)
|
||||
|
||||
// (100+15)/3 = 38.33..., (200+25)/3 = 75
|
||||
if got.X0 != 38.333333333333336 || got.Y0 != 75 || got.X1 != 105 || got.Y1 != 141.66666666666666 {
|
||||
t.Errorf("cellToPageSpace: got (%f,%f,%f,%f), want (38.33,75,105,141.67)", got.X0, got.Y0, got.X1, got.Y1)
|
||||
}
|
||||
if got.Text != "hello" || got.Label != "table" {
|
||||
t.Error("cellToPageSpace should preserve Text and Label")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCellAddOffset(t *testing.T) {
|
||||
cell := pdf.TSRCell{X0: 100, Y0: 200, X1: 300, Y1: 400, Text: "hello"}
|
||||
got := CellAddOffset(cell, 15, 25)
|
||||
if got.X0 != 115 || got.Y0 != 225 || got.X1 != 315 || got.Y1 != 425 {
|
||||
t.Errorf("cellAddOffset: got (%f,%f,%f,%f)", got.X0, got.Y0, got.X1, got.Y1)
|
||||
}
|
||||
if got.Text != "hello" {
|
||||
t.Error("cellAddOffset should preserve Text")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBoxToCropSpace(t *testing.T) {
|
||||
box := pdf.TextBox{X0: 50, X1: 200, Top: 100, Bottom: 300, Text: "text"}
|
||||
got := BoxToCropSpace(box, 3.0, 10, 20)
|
||||
if got.X0 != 140 || got.Top != 280 || got.X1 != 590 || got.Bottom != 880 {
|
||||
t.Errorf("boxToCropSpace: got (%f,%f,%f,%f)", got.X0, got.Top, got.X1, got.Bottom)
|
||||
}
|
||||
if got.Text != "text" {
|
||||
t.Error("boxToCropSpace should preserve Text")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCopyBoxAnnotations(t *testing.T) {
|
||||
src := &pdf.TextBox{R: 1, C: 2, RTop: 10, RBott: 20, H: 3, HTop: 30, HBott: 40,
|
||||
HLeft: 50, HRight: 60, CLeft: 70, CRight: 80, SP: 4}
|
||||
dst := &pdf.TextBox{}
|
||||
CopyBoxAnnotations(dst, src)
|
||||
if dst.R != 1 || dst.C != 2 || dst.RTop != 10 || dst.RBott != 20 {
|
||||
t.Error("R/C fields not copied")
|
||||
}
|
||||
if dst.H != 3 || dst.HTop != 30 || dst.HBott != 40 {
|
||||
t.Error("H fields not copied")
|
||||
}
|
||||
if dst.HLeft != 50 || dst.HRight != 60 || dst.CLeft != 70 || dst.CRight != 80 {
|
||||
t.Error("spanning fields not copied")
|
||||
}
|
||||
if dst.SP != 4 {
|
||||
t.Error("SP not copied")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCellSliceToPageSpace(t *testing.T) {
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 100, Y0: 200, X1: 300, Y1: 400},
|
||||
{X0: 400, Y0: 200, X1: 600, Y1: 400},
|
||||
}
|
||||
got := CellSliceToPageSpace(cells, 15, 25, 3)
|
||||
if len(got) != 2 {
|
||||
t.Fatal("expected 2 cells")
|
||||
}
|
||||
if got[0].X0 != 38.333333333333336 || got[1].X0 != 138.33333333333334 {
|
||||
t.Error("wrong conversion")
|
||||
}
|
||||
}
|
||||
@@ -1,16 +1,18 @@
|
||||
package parser
|
||||
package table
|
||||
|
||||
import (
|
||||
"math"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"ragflow/internal/deepdoc/parser/pdf/util"
|
||||
"sort"
|
||||
)
|
||||
|
||||
// ── Post-TSR layout annotation (Python: pdf_parser.py gather/layouts_cleanup) ──
|
||||
|
||||
// sortYFirstly sorts cells by top, with fuzzy threshold: if two cells are
|
||||
// SortYFirstly sorts cells by top, with fuzzy threshold: if two cells are
|
||||
// within threshold Y pixels, sort by X instead (same-row ordering).
|
||||
// Python: Recognizer.sort_Y_firstly(arr, threshold)
|
||||
func sortYFirstly(cells []TSRCell, threshold float64) {
|
||||
func SortYFirstly(cells []pdf.TSRCell, threshold float64) {
|
||||
sort.Slice(cells, func(i, j int) bool {
|
||||
diff := cells[i].Y0 - cells[j].Y0
|
||||
if math.Abs(diff) < threshold {
|
||||
@@ -20,8 +22,8 @@ func sortYFirstly(cells []TSRCell, threshold float64) {
|
||||
})
|
||||
}
|
||||
|
||||
// sortXFirstly sorts cells by x0, with fuzzy threshold for top.
|
||||
func sortXFirstly(cells []TSRCell, threshold float64) {
|
||||
// SortXFirstly sorts cells by x0, with fuzzy threshold for top.
|
||||
func SortXFirstly(cells []pdf.TSRCell, threshold float64) {
|
||||
sort.Slice(cells, func(i, j int) bool {
|
||||
diff := cells[i].X0 - cells[j].X0
|
||||
if math.Abs(diff) < threshold {
|
||||
@@ -37,9 +39,9 @@ func sortXFirstly(cells []TSRCell, threshold float64) {
|
||||
// For each cell, checks the next `far` cells; if they overlap significantly
|
||||
// AND have the same label type, the one with lower score (or less box overlap
|
||||
// area) is removed.
|
||||
func layoutCleanup(cells []TSRCell, boxes []TextBox, far int, thr float64) []TSRCell {
|
||||
func layoutCleanup(cells []pdf.TSRCell, boxes []pdf.TextBox, far int, thr float64) []pdf.TSRCell {
|
||||
// cells are assumed pre-sorted (caller sorts before passing)
|
||||
out := make([]TSRCell, len(cells))
|
||||
out := make([]pdf.TSRCell, len(cells))
|
||||
copy(out, cells)
|
||||
|
||||
i := 0
|
||||
@@ -57,8 +59,8 @@ func layoutCleanup(cells []TSRCell, boxes []TextBox, far int, thr float64) []TSR
|
||||
continue
|
||||
}
|
||||
// Cells i and j overlap and have same type. Keep one.
|
||||
areaI := OverlapRatioA(&out[i], &out[j])
|
||||
areaJ := OverlapRatioA(&out[j], &out[i])
|
||||
areaI := util.OverlapRatioA(&out[i], &out[j])
|
||||
areaJ := util.OverlapRatioA(&out[j], &out[i])
|
||||
if areaI < thr && areaJ < thr {
|
||||
i++
|
||||
continue
|
||||
@@ -68,10 +70,10 @@ func layoutCleanup(cells []TSRCell, boxes []TextBox, far int, thr float64) []TSR
|
||||
boxAreaI, boxAreaJ := 0.0, 0.0
|
||||
for _, b := range boxes {
|
||||
if !tsrBoxOverlap(b, out[i]) {
|
||||
boxAreaI += OverlapInter(&b, &out[i])
|
||||
boxAreaI += util.OverlapInter(&b, &out[i])
|
||||
}
|
||||
if !tsrBoxOverlap(b, out[j]) {
|
||||
boxAreaJ += OverlapInter(&b, &out[j])
|
||||
boxAreaJ += util.OverlapInter(&b, &out[j])
|
||||
}
|
||||
}
|
||||
if boxAreaI >= boxAreaJ {
|
||||
@@ -84,12 +86,12 @@ func layoutCleanup(cells []TSRCell, boxes []TextBox, far int, thr float64) []TSR
|
||||
}
|
||||
|
||||
// notOverlapped returns true if cells a and b do NOT overlap.
|
||||
func notOverlapped(a, b TSRCell) bool {
|
||||
func notOverlapped(a, b pdf.TSRCell) bool {
|
||||
return a.X1 < b.X0 || a.X0 > b.X1 || a.Y1 < b.Y0 || a.Y0 > b.Y1
|
||||
}
|
||||
|
||||
// tsrBoxOverlap returns true if a TextBox and a TSRCell do NOT overlap.
|
||||
func tsrBoxOverlap(b TextBox, c TSRCell) bool {
|
||||
// tsrBoxOverlap returns true if a pdf.TextBox and a pdf.TSRCell do NOT overlap.
|
||||
func tsrBoxOverlap(b pdf.TextBox, c pdf.TSRCell) bool {
|
||||
return b.X1 < c.X0 || b.X0 > c.X1 || b.Bottom < c.Y0 || b.Top > c.Y1
|
||||
}
|
||||
|
||||
@@ -97,19 +99,19 @@ func tsrBoxOverlap(b TextBox, c TSRCell) bool {
|
||||
// bidirectional overlap >= thr, or -1 if none.
|
||||
// Python: Recognizer.find_overlapped_with_threshold(box, boxes, thr=0.3)
|
||||
// Python uses max(boxRatio, cellRatio) for both gate and scoring.
|
||||
func findOverlappedWithThreshold(box TextBox, cells []TSRCell, thr float64) int {
|
||||
boxArea := Area(&box)
|
||||
func findOverlappedWithThreshold(box pdf.TextBox, cells []pdf.TSRCell, thr float64) int {
|
||||
boxArea := util.Area(&box)
|
||||
if boxArea <= 0 {
|
||||
return -1
|
||||
}
|
||||
bestIdx := -1
|
||||
bestOverlap := thr // Python: max_overlap starts at thr
|
||||
for i, c := range cells {
|
||||
cellArea := Area(&c)
|
||||
cellArea := util.Area(&c)
|
||||
if cellArea <= 0 {
|
||||
continue
|
||||
}
|
||||
ol := OverlapInter(&box, &c)
|
||||
ol := util.OverlapInter(&box, &c)
|
||||
if ol <= 0 {
|
||||
continue
|
||||
}
|
||||
@@ -130,7 +132,7 @@ func findOverlappedWithThreshold(box TextBox, cells []TSRCell, thr float64) int
|
||||
// Python: Recognizer.find_horizontally_tightest_fit(b, clmns)
|
||||
// findHorizontallyTightestFit returns the column index with minimum
|
||||
// edge distance to the box. Python: Recognizer.find_horizontally_tightest_fit.
|
||||
func findHorizontallyTightestFit(box TextBox, clmns []TSRCell) int {
|
||||
func findHorizontallyTightestFit(box pdf.TextBox, clmns []pdf.TSRCell) int {
|
||||
best := -1
|
||||
bestDist := float64(1<<63 - 1)
|
||||
for i, c := range clmns {
|
||||
@@ -150,19 +152,19 @@ func findHorizontallyTightestFit(box TextBox, clmns []TSRCell) int {
|
||||
// TSR cell labels. Matching Python's R/H/C/SP annotation logic.
|
||||
//
|
||||
// Python: pdf_parser.py:518-554
|
||||
func annotateTableBoxes(boxes []TextBox, grid [][]TSRCell) {
|
||||
func AnnotateTableBoxes(boxes []pdf.TextBox, grid [][]pdf.TSRCell) {
|
||||
// grid[0] is the header row. Spans are computed by calSpans later.
|
||||
var headers, spans []TSRCell
|
||||
var clmns []TSRCell
|
||||
var headers, spans []pdf.TSRCell
|
||||
var clmns []pdf.TSRCell
|
||||
if len(grid) > 0 {
|
||||
headers = grid[0]
|
||||
clmns = append(clmns, grid[0]...)
|
||||
}
|
||||
sortYFirstly(headers, 10)
|
||||
sortXFirstly(clmns, 10)
|
||||
SortYFirstly(headers, 10)
|
||||
SortXFirstly(clmns, 10)
|
||||
|
||||
for i := range boxes {
|
||||
if boxes[i].LayoutType != LayoutTypeTable {
|
||||
if boxes[i].LayoutType != pdf.LayoutTypeTable {
|
||||
continue
|
||||
}
|
||||
// Grid-based R/C: match box to the row and column it overlaps.
|
||||
@@ -207,7 +209,7 @@ func annotateTableBoxes(boxes []TextBox, grid [][]TSRCell) {
|
||||
// Collect all table boxes grouped by R.
|
||||
rBoxes := make(map[int][]int)
|
||||
for i := range boxes {
|
||||
if boxes[i].LayoutType == LayoutTypeTable {
|
||||
if boxes[i].LayoutType == pdf.LayoutTypeTable {
|
||||
rBoxes[boxes[i].R] = append(rBoxes[boxes[i].R], i)
|
||||
}
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
package parser
|
||||
package table
|
||||
|
||||
import (
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"sort"
|
||||
"testing"
|
||||
)
|
||||
@@ -18,8 +19,8 @@ import (
|
||||
// +----------+----------+
|
||||
// | row 2A | row 2B | ← row 2 (Y=50..70)
|
||||
// +----------+----------+
|
||||
func makeMockTableCells() []TSRCell {
|
||||
return []TSRCell{
|
||||
func makeMockTableCells() []pdf.TSRCell {
|
||||
return []pdf.TSRCell{
|
||||
{X0: 10, Y0: 10, X1: 50, Y1: 30, Label: "table column header"},
|
||||
{X0: 50, Y0: 10, X1: 90, Y1: 30, Label: "table column header"},
|
||||
{X0: 70, Y0: 30, X1: 90, Y1: 50, Label: "table row"},
|
||||
@@ -30,8 +31,8 @@ func makeMockTableCells() []TSRCell {
|
||||
}
|
||||
}
|
||||
|
||||
func makeMockBoxes() []TextBox {
|
||||
return []TextBox{
|
||||
func makeMockBoxes() []pdf.TextBox {
|
||||
return []pdf.TextBox{
|
||||
{X0: 10, X1: 90, Top: 25, Bottom: 55, LayoutType: "table", Text: "test table"},
|
||||
// row at Y=30..50 overlaps ~80% → should match
|
||||
}
|
||||
@@ -39,23 +40,23 @@ func makeMockBoxes() []TextBox {
|
||||
|
||||
func TestSortYFirstly(t *testing.T) {
|
||||
t.Run("basic sort", func(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 10, Y0: 50, Label: "c"},
|
||||
{X0: 10, Y0: 10, Label: "a"},
|
||||
{X0: 10, Y0: 30, Label: "b"},
|
||||
}
|
||||
sortYFirstly(cells, 5)
|
||||
SortYFirstly(cells, 5)
|
||||
if cells[0].Label != "a" || cells[1].Label != "b" || cells[2].Label != "c" {
|
||||
t.Errorf("sort order wrong: %v", cells)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("same Y sorts by X", func(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 90, Y0: 10, Label: "right"},
|
||||
{X0: 10, Y0: 10, Label: "left"},
|
||||
}
|
||||
sortYFirstly(cells, 5)
|
||||
SortYFirstly(cells, 5)
|
||||
if cells[0].Label != "left" || cells[1].Label != "right" {
|
||||
t.Errorf("same Y should sort X ascending: %v", cells)
|
||||
}
|
||||
@@ -68,7 +69,7 @@ func TestLayoutCleanup(t *testing.T) {
|
||||
boxes := makeMockBoxes()
|
||||
|
||||
t.Run("no overlap different types", func(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 10, Y0: 10, X1: 50, Y1: 30, Label: "table column header"},
|
||||
{X0: 10, Y0: 10, X1: 50, Y1: 30, Label: "table row"},
|
||||
}
|
||||
@@ -79,7 +80,7 @@ func TestLayoutCleanup(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("overlap same type keeps one", func(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 10, Y0: 10, X1: 50, Y1: 30, Label: "table row"},
|
||||
{X0: 12, Y0: 12, X1: 48, Y1: 28, Label: "table row"}, // mostly contained
|
||||
}
|
||||
@@ -90,7 +91,7 @@ func TestLayoutCleanup(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("non overlapping same type keeps both", func(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 10, Y0: 10, X1: 50, Y1: 30, Label: "table row"},
|
||||
{X0: 200, Y0: 10, X1: 250, Y1: 30, Label: "table row"}, // far away
|
||||
}
|
||||
@@ -111,28 +112,28 @@ func TestLayoutCleanup(t *testing.T) {
|
||||
// ── findOverlappedWithThreshold ────────────────────────────────────────
|
||||
|
||||
func TestFindOverlappedWithThreshold(t *testing.T) {
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 10, Y0: 10, X1: 50, Y1: 30},
|
||||
{X0: 50, Y0: 30, X1: 90, Y1: 50},
|
||||
{X0: 10, Y0: 50, X1: 50, Y1: 70},
|
||||
}
|
||||
|
||||
t.Run("exact match", func(t *testing.T) {
|
||||
box := TextBox{X0: 10, X1: 50, Top: 10, Bottom: 30}
|
||||
box := pdf.TextBox{X0: 10, X1: 50, Top: 10, Bottom: 30}
|
||||
if idx := findOverlappedWithThreshold(box, cells, 0.3); idx != 0 {
|
||||
t.Errorf("expected idx=0, got %d", idx)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("no match", func(t *testing.T) {
|
||||
box := TextBox{X0: 200, X1: 250, Top: 200, Bottom: 230}
|
||||
box := pdf.TextBox{X0: 200, X1: 250, Top: 200, Bottom: 230}
|
||||
if idx := findOverlappedWithThreshold(box, cells, 0.3); idx != -1 {
|
||||
t.Errorf("expected idx=-1, got %d", idx)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("zero area box", func(t *testing.T) {
|
||||
box := TextBox{X0: 10, X1: 10, Top: 10, Bottom: 10}
|
||||
box := pdf.TextBox{X0: 10, X1: 10, Top: 10, Bottom: 10}
|
||||
if idx := findOverlappedWithThreshold(box, cells, 0.3); idx != -1 {
|
||||
t.Errorf("zero-area box should return -1: got %d", idx)
|
||||
}
|
||||
@@ -145,7 +146,7 @@ func TestAnnotateTableBoxes(t *testing.T) {
|
||||
cells := makeMockTableCells()
|
||||
boxes := makeMockBoxes()
|
||||
|
||||
annotateTableBoxes(boxes, groupTSRCellsToRowsLabeled(cells))
|
||||
AnnotateTableBoxes(boxes, GroupTSRCellsToRows(cells))
|
||||
|
||||
b := boxes[0]
|
||||
|
||||
@@ -165,13 +166,13 @@ func TestAnnotateTableBoxes(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// ── groupTSRCellsToRowsLabeled ─────────────────────────────────────────
|
||||
// ── GroupTSRCellsToRows ─────────────────────────────────────────
|
||||
|
||||
func TestGroupTSRCellsToRowsLabeled(t *testing.T) {
|
||||
cells := makeMockTableCells()
|
||||
|
||||
t.Run("label-based grouping", func(t *testing.T) {
|
||||
rows := groupTSRCellsToRowsLabeled(cells)
|
||||
rows := GroupTSRCellsToRows(cells)
|
||||
if len(rows) < 2 {
|
||||
t.Errorf("expected >= 2 rows, got %d", len(rows))
|
||||
}
|
||||
@@ -184,19 +185,19 @@ func TestGroupTSRCellsToRowsLabeled(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("fallback to Y-based", func(t *testing.T) {
|
||||
unlabeled := []TSRCell{
|
||||
unlabeled := []pdf.TSRCell{
|
||||
{X0: 10, Y0: 10, X1: 50, Y1: 20, Label: ""},
|
||||
{X0: 10, Y0: 30, X1: 50, Y1: 40, Label: ""},
|
||||
}
|
||||
rows := groupTSRCellsToRowsLabeled(unlabeled)
|
||||
rows := GroupTSRCellsToRows(unlabeled)
|
||||
if len(rows) < 2 {
|
||||
t.Errorf("fallback: expected >= 2 rows, got %d", len(rows))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("single cell", func(t *testing.T) {
|
||||
cells := []TSRCell{{X0: 0, Y0: 0, X1: 10, Y1: 10, Label: "table row"}}
|
||||
rows := groupTSRCellsToRowsLabeled(cells)
|
||||
cells := []pdf.TSRCell{{X0: 0, Y0: 0, X1: 10, Y1: 10, Label: "table row"}}
|
||||
rows := GroupTSRCellsToRows(cells)
|
||||
if len(rows) != 1 {
|
||||
t.Errorf("expected 1 row, got %d", len(rows))
|
||||
}
|
||||
@@ -206,15 +207,15 @@ func TestGroupTSRCellsToRowsLabeled(t *testing.T) {
|
||||
// TestAnnotateTableBoxes_PixelSpace verifies that boxes in pixel space
|
||||
// (as from DLA-scaled coordinates) correctly match TSR cells. Regression test for Bug #1.
|
||||
func TestAnnotateTableBoxes_PixelSpace(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 150, X1: 750, Top: 300, Bottom: 420, LayoutType: "table"},
|
||||
}
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 150, Y0: 300, X1: 750, Y1: 350, Label: "table column header"},
|
||||
{X0: 150, Y0: 350, X1: 750, Y1: 380, Label: "table row"},
|
||||
{X0: 150, Y0: 380, X1: 750, Y1: 420, Label: "table row"},
|
||||
}
|
||||
annotateTableBoxes(boxes, groupTSRCellsToRowsLabeled(cells))
|
||||
AnnotateTableBoxes(boxes, GroupTSRCellsToRows(cells))
|
||||
if boxes[0].R < 0 {
|
||||
t.Error("row index should be set (pixel-space matching)")
|
||||
}
|
||||
@@ -226,13 +227,13 @@ func TestAnnotateTableBoxes_PixelSpace(t *testing.T) {
|
||||
// TestFindHorizontallyTightestFit verifies the edge-distance matching
|
||||
// (Python's minimum edge distance, not Go's old containment check).
|
||||
func TestFindHorizontallyTightestFit(t *testing.T) {
|
||||
clmns := []TSRCell{
|
||||
clmns := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 50},
|
||||
{X0: 100, Y0: 0, X1: 200, Y1: 50},
|
||||
}
|
||||
|
||||
t.Run("exact match left edge", func(t *testing.T) {
|
||||
box := TextBox{X0: 100, X1: 150, Top: 0, Bottom: 50}
|
||||
box := pdf.TextBox{X0: 100, X1: 150, Top: 0, Bottom: 50}
|
||||
if idx := findHorizontallyTightestFit(box, clmns); idx != 1 {
|
||||
t.Errorf("box at col 1 left edge: got idx=%d, want 1", idx)
|
||||
}
|
||||
@@ -241,14 +242,14 @@ func TestFindHorizontallyTightestFit(t *testing.T) {
|
||||
t.Run("partial containment — still matches nearest", func(t *testing.T) {
|
||||
// Box mostly in col 0 but spills into col 1. Old containment check
|
||||
// would fail; distance check matches col 0 (closer edges).
|
||||
box := TextBox{X0: 80, X1: 120, Top: 0, Bottom: 50}
|
||||
box := pdf.TextBox{X0: 80, X1: 120, Top: 0, Bottom: 50}
|
||||
if idx := findHorizontallyTightestFit(box, clmns); idx != 0 {
|
||||
t.Errorf("spill box: got idx=%d, want 0 (nearest edges)", idx)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("empty columns", func(t *testing.T) {
|
||||
if idx := findHorizontallyTightestFit(TextBox{}, nil); idx != -1 {
|
||||
if idx := findHorizontallyTightestFit(pdf.TextBox{}, nil); idx != -1 {
|
||||
t.Errorf("empty: got %d, want -1", idx)
|
||||
}
|
||||
})
|
||||
@@ -258,11 +259,11 @@ func TestFindHorizontallyTightestFit(t *testing.T) {
|
||||
// (bidirectional overlap) replaces the old first-match behavior.
|
||||
func TestFindOverlappedWithThreshold_BestMatch(t *testing.T) {
|
||||
// Two cells overlap the same box. Cell 1 has MORE overlap → should win.
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 50, Y1: 50}, // 30% overlap
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 100}, // 100% overlap — best match
|
||||
}
|
||||
box := TextBox{X0: 0, X1: 100, Top: 0, Bottom: 100}
|
||||
box := pdf.TextBox{X0: 0, X1: 100, Top: 0, Bottom: 100}
|
||||
if idx := findOverlappedWithThreshold(box, cells, 0.2); idx != 1 {
|
||||
t.Errorf("best-match: got idx=%d, want 1 (100%% overlap beats 30%%)", idx)
|
||||
}
|
||||
@@ -276,8 +277,8 @@ func TestFindOverlappedWithThreshold_BestMatch(t *testing.T) {
|
||||
// Old Go (box-only gate): overlap/boxArea = 0.02 > 0.3? → NO MATCH ✗
|
||||
func TestFindOverlappedWithThreshold_BidirectionalGate(t *testing.T) {
|
||||
// Large box fully contains a tiny cell.
|
||||
box := TextBox{X0: 0, X1: 500, Top: 0, Bottom: 20} // area = 10000
|
||||
cells := []TSRCell{
|
||||
box := pdf.TextBox{X0: 0, X1: 500, Top: 0, Bottom: 20} // area = 10000
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 10, Y1: 20}, // area = 200, entirely inside box
|
||||
}
|
||||
// boxRatio = 200/10000 = 0.02, cellRatio = 200/200 = 1.0
|
||||
@@ -296,8 +297,8 @@ func TestFindOverlappedWithThreshold_BidirectionalGate(t *testing.T) {
|
||||
// Cell B: boxRatio=0.40, cellRatio=0.40 → max=0.40, sum=0.80
|
||||
// Python (max): picks A (0.60 > 0.40). Old Go (sum): picks B (0.80 > 0.65).
|
||||
func TestFindOverlappedWithThreshold_MaxScoring(t *testing.T) {
|
||||
box := TextBox{X0: 0, X1: 100, Top: 0, Bottom: 100} // area = 10000
|
||||
cells := []TSRCell{
|
||||
box := pdf.TextBox{X0: 0, X1: 100, Top: 0, Bottom: 100} // area = 10000
|
||||
cells := []pdf.TSRCell{
|
||||
// Cell A: narrow but tall (60×2000), covers 60% of box width.
|
||||
// boxRatio=60*100/10000=0.60, cellRatio=60*100/(60*2000)=0.05, max=0.60
|
||||
{X0: 0, Y0: 0, X1: 60, Y1: 2000},
|
||||
@@ -323,17 +324,17 @@ func TestFindOverlappedWithThreshold_MaxScoring(t *testing.T) {
|
||||
// row×col structure even without row/column labels.
|
||||
func TestGroupTSRCellsToRowsLabeled_FallbackY(t *testing.T) {
|
||||
// 4 rows × 5 cols = 20 cells, all label="table".
|
||||
cells := make([]TSRCell, 20)
|
||||
cells := make([]pdf.TSRCell, 20)
|
||||
for r := 0; r < 4; r++ {
|
||||
for c := 0; c < 5; c++ {
|
||||
cells[r*5+c] = TSRCell{
|
||||
cells[r*5+c] = pdf.TSRCell{
|
||||
X0: float64(c * 100), Y0: float64(r * 30),
|
||||
X1: float64(c*100 + 80), Y1: float64(r*30 + 25),
|
||||
Label: "table",
|
||||
}
|
||||
}
|
||||
}
|
||||
rows := groupTSRCellsToRowsLabeled(cells)
|
||||
rows := GroupTSRCellsToRows(cells)
|
||||
if len(rows) != 4 {
|
||||
t.Fatalf("fallback Y-grouping: expected 4 rows, got %d", len(rows))
|
||||
}
|
||||
@@ -360,7 +361,8 @@ func TestGroupTSRCellsToRowsLabeled_FallbackY(t *testing.T) {
|
||||
func TestGroupTSRCellsToRowsLabeled_Irregular(t *testing.T) {
|
||||
// Irregular layout: row 0 has 3 cells, row 1 has 5, row 2 has 2.
|
||||
// Cells within a row have slightly different Y (within threshold).
|
||||
cells := []TSRCell{
|
||||
// Basic Y-proximity grouping does not pad rows to equal column counts.
|
||||
cells := []pdf.TSRCell{
|
||||
// Row 0 — 3 cells at ~Y=0 (slightly staggered tops).
|
||||
{X0: 0, Y0: 0, X1: 80, Y1: 25, Label: "table"},
|
||||
{X0: 90, Y0: 2, X1: 170, Y1: 27, Label: "table"},
|
||||
@@ -375,18 +377,18 @@ func TestGroupTSRCellsToRowsLabeled_Irregular(t *testing.T) {
|
||||
{X0: 0, Y0: 60, X1: 80, Y1: 85, Label: "table"},
|
||||
{X0: 90, Y0: 61, X1: 170, Y1: 86, Label: "table"},
|
||||
}
|
||||
rows := groupTSRCellsToRowsLabeled(cells)
|
||||
rows := GroupTSRCellsToRows(cells)
|
||||
if len(rows) != 3 {
|
||||
t.Fatalf("irregular: expected 3 rows, got %d", len(rows))
|
||||
}
|
||||
if len(rows[0]) != 5 {
|
||||
t.Errorf("row 0: expected 5 cols (padded), got %d", len(rows[0]))
|
||||
if len(rows[0]) != 3 {
|
||||
t.Errorf("row 0: expected 3 cols, got %d", len(rows[0]))
|
||||
}
|
||||
if len(rows[1]) != 5 {
|
||||
t.Errorf("row 1: expected 5 cols, got %d", len(rows[1]))
|
||||
}
|
||||
if len(rows[2]) != 5 {
|
||||
t.Errorf("row 2: expected 5 cols (padded), got %d", len(rows[2]))
|
||||
if len(rows[2]) != 2 {
|
||||
t.Errorf("row 2: expected 2 cols, got %d", len(rows[2]))
|
||||
}
|
||||
}
|
||||
|
||||
@@ -396,25 +398,25 @@ func TestGroupTSRCellsToRowsLabeled_Irregular(t *testing.T) {
|
||||
// its existing Text (from TSR or previous steps).
|
||||
func TestFillCellTextFromBoxes_PreservesTSRText(t *testing.T) {
|
||||
// Cell already has text from TSR. No box overlaps it.
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 50, Text: "TSR-provided"},
|
||||
}
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 500, X1: 600, Top: 500, Bottom: 550, Text: "far away"},
|
||||
}
|
||||
fillCellTextFromBoxes(cells, boxes)
|
||||
FillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "TSR-provided" {
|
||||
t.Errorf("TSR text overwritten: got %q, want 'TSR-provided'", cells[0].Text)
|
||||
}
|
||||
|
||||
// Cell with TSR text, box covers >85% — should be overwritten.
|
||||
cells2 := []TSRCell{
|
||||
cells2 := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 50, Text: "TSR-provided"},
|
||||
}
|
||||
boxes2 := []TextBox{
|
||||
boxes2 := []pdf.TextBox{
|
||||
{X0: 1, X1: 99, Top: 1, Bottom: 49, Text: "box-text"},
|
||||
}
|
||||
fillCellTextFromBoxes(cells2, boxes2)
|
||||
FillCellTextFromBoxes(cells2, boxes2)
|
||||
if cells2[0].Text != "box-text" {
|
||||
t.Errorf("box text should override TSR text: got %q, want 'box-text'", cells2[0].Text)
|
||||
}
|
||||
@@ -429,10 +431,10 @@ func TestFillCellTextFromBoxes_PartialOverlap(t *testing.T) {
|
||||
// Empty cell (no TSR text). Box only has ~55% of its area inside
|
||||
// the cell (spills across the boundary). Python's 0.3 threshold
|
||||
// accepts this; Go's 0.85 rejects it → empty cell.
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 50, Text: ""},
|
||||
}
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
// Box: 60% inside cell, 40% outside. Overlap ratio = 60%.
|
||||
{X0: 40, X1: 140, Top: 5, Bottom: 15, Text: "spill text"},
|
||||
}
|
||||
@@ -440,41 +442,35 @@ func TestFillCellTextFromBoxes_PartialOverlap(t *testing.T) {
|
||||
// Overlap: X=(40,100) Y=(5,15) → 60×10=600.
|
||||
// Box area: 100×10=1000. ratio = 600/1000 = 60%.
|
||||
// Old 85% threshold → rejected. Python's 0.3 → accepted.
|
||||
fillCellTextFromBoxes(cells, boxes)
|
||||
FillCellTextFromBoxes(cells, boxes)
|
||||
if cells[0].Text != "spill text" {
|
||||
t.Errorf("partial overlap (<85%%) on empty cell should still fill: got %q, want 'spill text'", cells[0].Text)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGroupTSRCellsToRowsLabeled_ColumnAlignment verifies that all
|
||||
// rows have the same column count after grouping, even with spanning
|
||||
// cells. Python's construct_table ensures R×C matrix alignment;
|
||||
// Go's Y-grouping can produce jagged rows when spanning cells make
|
||||
// some rows appear shorter.
|
||||
// TestGroupTSRCellsToRowsLabeled_ColumnAlignment verifies that basic
|
||||
// Y-proximity grouping produces the correct row counts. Unlike the EE
|
||||
// label-aware grouping, Y-proximity does not handle spanning cells
|
||||
// specially — each cell is simply placed into its Y-based row.
|
||||
func TestGroupTSRCellsToRowsLabeled_ColumnAlignment(t *testing.T) {
|
||||
// 2-row table: row 0 has a spanning cell (covers 2 columns) → 2 visible cells.
|
||||
// row 1 has 3 normal cells.
|
||||
// Python construct_table: both rows padded to 3 cols.
|
||||
// Go Y-grouping (current): row 0 has 2 cols, row 1 has 3 → JAGGED.
|
||||
cells := []TSRCell{
|
||||
// Row 0 — spanning cell + 1 normal cell (= 2 cells)
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 200, Y1: 30, Label: "table spanning cell"},
|
||||
{X0: 200, Y0: 0, X1: 300, Y1: 30, Label: "table row"},
|
||||
// Row 1 — 3 normal cells
|
||||
{X0: 0, Y0: 30, X1: 100, Y1: 60, Label: "table row"},
|
||||
{X0: 100, Y0: 30, X1: 200, Y1: 60, Label: "table row"},
|
||||
{X0: 200, Y0: 30, X1: 300, Y1: 60, Label: "table row"},
|
||||
}
|
||||
rows := groupTSRCellsToRowsLabeled(cells)
|
||||
rows := GroupTSRCellsToRows(cells)
|
||||
if len(rows) != 2 {
|
||||
t.Fatalf("expected 2 rows, got %d", len(rows))
|
||||
}
|
||||
// BUG: row 0 only has 2 cells (spanning cell covers 2 columns but
|
||||
// appears as 1 cell in Y-grouping). Python's construct_table pads
|
||||
// to 3 columns.
|
||||
if len(rows[0]) != len(rows[1]) {
|
||||
t.Errorf("column alignment broken: row0=%d cols, row1=%d cols — "+
|
||||
"Python construct_table ensures all rows have equal columns", len(rows[0]), len(rows[1]))
|
||||
// Basic Y-proximity: row0 has 2 cells, row1 has 3 cells.
|
||||
// No column alignment padding (EE feature).
|
||||
if len(rows[0]) != 2 {
|
||||
t.Errorf("row0: expected 2 cells, got %d", len(rows[0]))
|
||||
}
|
||||
if len(rows[1]) != 3 {
|
||||
t.Errorf("row1: expected 3 cells, got %d", len(rows[1]))
|
||||
}
|
||||
}
|
||||
|
||||
@@ -484,7 +480,7 @@ func TestGroupTSRCellsToRowsLabeled_ColumnAlignment(t *testing.T) {
|
||||
func TestAnnotateTableBoxes_RealTSRLabels(t *testing.T) {
|
||||
// Simulate a 2×3 table: 2 rows, 3 columns.
|
||||
// TSR cells with label "table" (default TSR class 0) — like 公司差旅费.
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 30, Label: "table"},
|
||||
{X0: 101, Y0: 0, X1: 200, Y1: 30, Label: "table"},
|
||||
{X0: 201, Y0: 0, X1: 300, Y1: 30, Label: "table"},
|
||||
@@ -492,7 +488,7 @@ func TestAnnotateTableBoxes_RealTSRLabels(t *testing.T) {
|
||||
{X0: 101, Y0: 35, X1: 200, Y1: 65, Label: "table"},
|
||||
{X0: 201, Y0: 35, X1: 300, Y1: 65, Label: "table"},
|
||||
}
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 10, X1: 90, Top: 0, Bottom: 30, Text: "A", LayoutType: "table"},
|
||||
{X0: 110, X1: 190, Top: 0, Bottom: 30, Text: "B", LayoutType: "table"},
|
||||
{X0: 210, X1: 290, Top: 0, Bottom: 30, Text: "C", LayoutType: "table"},
|
||||
@@ -500,7 +496,7 @@ func TestAnnotateTableBoxes_RealTSRLabels(t *testing.T) {
|
||||
{X0: 110, X1: 190, Top: 35, Bottom: 65, Text: "E", LayoutType: "table"},
|
||||
{X0: 210, X1: 290, Top: 35, Bottom: 65, Text: "F", LayoutType: "table"},
|
||||
}
|
||||
annotateTableBoxes(boxes, groupTSRCellsToRowsLabeled(cells))
|
||||
AnnotateTableBoxes(boxes, GroupTSRCellsToRows(cells))
|
||||
|
||||
// Verify R (row) assignments — should be 0 for top row, 1 for bottom row.
|
||||
for i, b := range boxes {
|
||||
@@ -525,30 +521,30 @@ func TestAnnotateTableBoxes_RealTSRLabels(t *testing.T) {
|
||||
// actual overlap. This test locks in the semantics so future readers
|
||||
// and static analysis tools can rely on the behaviour.
|
||||
func TestTsrBoxOverlap_ReturnsTrueWhenDisjoint(t *testing.T) {
|
||||
box := TextBox{X0: 50, X1: 100, Top: 0, Bottom: 50}
|
||||
box := pdf.TextBox{X0: 50, X1: 100, Top: 0, Bottom: 50}
|
||||
|
||||
// Separated in X (cell to the right) → disjoint → true.
|
||||
if !tsrBoxOverlap(box, TSRCell{X0: 150, Y0: 0, X1: 200, Y1: 50}) {
|
||||
if !tsrBoxOverlap(box, pdf.TSRCell{X0: 150, Y0: 0, X1: 200, Y1: 50}) {
|
||||
t.Error("cell to the right (separated in X): expected true")
|
||||
}
|
||||
// Separated in X (cell to the left) → disjoint → true.
|
||||
if !tsrBoxOverlap(box, TSRCell{X0: 0, Y0: 0, X1: 30, Y1: 50}) {
|
||||
if !tsrBoxOverlap(box, pdf.TSRCell{X0: 0, Y0: 0, X1: 30, Y1: 50}) {
|
||||
t.Error("cell to the left (separated in X): expected true")
|
||||
}
|
||||
// Separated in Y (cell below) → disjoint → true.
|
||||
if !tsrBoxOverlap(box, TSRCell{X0: 50, Y0: 100, X1: 100, Y1: 150}) {
|
||||
if !tsrBoxOverlap(box, pdf.TSRCell{X0: 50, Y0: 100, X1: 100, Y1: 150}) {
|
||||
t.Error("cell below (separated in Y): expected true")
|
||||
}
|
||||
// Separated in Y (cell above) → disjoint → true.
|
||||
if !tsrBoxOverlap(box, TSRCell{X0: 50, Y0: -50, X1: 100, Y1: -10}) {
|
||||
if !tsrBoxOverlap(box, pdf.TSRCell{X0: 50, Y0: -50, X1: 100, Y1: -10}) {
|
||||
t.Error("cell above (separated in Y): expected true")
|
||||
}
|
||||
// Fully enclosing cell → overlaps in both X and Y → NOT disjoint → false.
|
||||
if tsrBoxOverlap(box, TSRCell{X0: 0, Y0: 0, X1: 200, Y1: 100}) {
|
||||
if tsrBoxOverlap(box, pdf.TSRCell{X0: 0, Y0: 0, X1: 200, Y1: 100}) {
|
||||
t.Error("cell fully enclosing box (overlaps): expected false")
|
||||
}
|
||||
// Partially overlapping cell → overlaps in both dims → false.
|
||||
if tsrBoxOverlap(box, TSRCell{X0: 25, Y0: 25, X1: 75, Y1: 75}) {
|
||||
if tsrBoxOverlap(box, pdf.TSRCell{X0: 25, Y0: 25, X1: 75, Y1: 75}) {
|
||||
t.Error("cell partially overlapping: expected false")
|
||||
}
|
||||
}
|
||||
107
internal/deepdoc/parser/pdf/table/table_orient.go
Normal file
107
internal/deepdoc/parser/pdf/table/table_orient.go
Normal file
@@ -0,0 +1,107 @@
|
||||
package table
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"image"
|
||||
"log/slog"
|
||||
"math"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"ragflow/internal/deepdoc/parser/pdf/util"
|
||||
)
|
||||
|
||||
// EvaluateTableOrientation tests 4 rotation angles (0/90/180/270) and picks
|
||||
// the best orientation based on OCR detect-region count and area coverage.
|
||||
//
|
||||
// Returns bestAngle (0/90/180/270), the rotated image, and per-angle scores.
|
||||
//
|
||||
// Absolute threshold: non-0° wins only if its combined score exceeds 0° by
|
||||
// more than 1.4× AND the 0° score is below 6.0.
|
||||
//
|
||||
// Python: pdf_parser.py:314 _evaluate_table_orientation()
|
||||
func EvaluateTableOrientation(ctx context.Context, tableImg image.Image, doc pdf.DocAnalyzer) (bestAngle int, bestImg image.Image, scores map[int]float64) {
|
||||
rotations := []struct {
|
||||
angle int
|
||||
name string
|
||||
}{
|
||||
{0, "original"},
|
||||
{90, "rotate_90"},
|
||||
{180, "rotate_180"},
|
||||
{270, "rotate_270"},
|
||||
}
|
||||
|
||||
scores = make(map[int]float64, 4)
|
||||
bestScore := float64(-1)
|
||||
bestAngle = 0
|
||||
bestImg = tableImg
|
||||
|
||||
for _, rot := range rotations {
|
||||
rotated := tableImg
|
||||
if rot.angle != 0 {
|
||||
rotated = util.RotateImageCW(tableImg, rot.angle)
|
||||
if rotated == nil {
|
||||
slog.Warn("table rotate failed", "angle", rot.angle)
|
||||
continue
|
||||
}
|
||||
}
|
||||
|
||||
detectBoxes, err := doc.OCRDetect(ctx, rotated)
|
||||
if err != nil || len(detectBoxes) == 0 {
|
||||
scores[rot.angle] = 0
|
||||
continue
|
||||
}
|
||||
|
||||
// Score by detect-region count (primary) + area (tiebreaker).
|
||||
imageArea := float64(rotated.Bounds().Dx() * rotated.Bounds().Dy())
|
||||
totalRegions := 0
|
||||
var totalArea float64
|
||||
for _, box := range detectBoxes {
|
||||
x0 := math.Min(box.X0, math.Min(box.X1, math.Min(box.X2, box.X3)))
|
||||
y0 := math.Min(box.Y0, math.Min(box.Y1, math.Min(box.Y2, box.Y3)))
|
||||
x1 := math.Max(box.X0, math.Max(box.X1, math.Max(box.X2, box.X3)))
|
||||
y1 := math.Max(box.Y0, math.Max(box.Y1, math.Max(box.Y2, box.Y3)))
|
||||
if x0 >= x1 || y0 >= y1 {
|
||||
continue
|
||||
}
|
||||
totalRegions++
|
||||
totalArea += (x1 - x0) * (y1 - y0)
|
||||
}
|
||||
if totalRegions == 0 {
|
||||
scores[rot.angle] = 0
|
||||
continue
|
||||
}
|
||||
areaRatio := totalArea / imageArea
|
||||
combined := float64(totalRegions) * (1 + 0.06*areaRatio)
|
||||
scores[rot.angle] = combined
|
||||
|
||||
slog.Debug("table orientation",
|
||||
"angle", rot.angle,
|
||||
"regions", totalRegions,
|
||||
"area_ratio", fmt.Sprintf("%.4f", areaRatio),
|
||||
"combined", fmt.Sprintf("%.2f", combined))
|
||||
|
||||
if combined > bestScore {
|
||||
bestScore = combined
|
||||
bestAngle = rot.angle
|
||||
bestImg = rotated
|
||||
}
|
||||
}
|
||||
|
||||
// Absolute threshold: only accept non-0° if region count is clearly
|
||||
// higher (≥1.4×) AND 0° has few regions (< 6).
|
||||
score0 := scores[0]
|
||||
if bestAngle != 0 && score0 > 0 {
|
||||
if !(bestScore > score0*1.4 && score0 < 6.0) {
|
||||
bestAngle = 0
|
||||
bestImg = tableImg
|
||||
bestScore = score0
|
||||
}
|
||||
}
|
||||
|
||||
slog.Debug("best table orientation",
|
||||
"angle", bestAngle,
|
||||
"score", fmt.Sprintf("%.4f", bestScore))
|
||||
|
||||
return bestAngle, bestImg, scores
|
||||
}
|
||||
@@ -1,8 +1,9 @@
|
||||
package parser
|
||||
package table
|
||||
|
||||
import (
|
||||
"context"
|
||||
"image"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"testing"
|
||||
)
|
||||
|
||||
@@ -22,18 +23,21 @@ type mockRotationDoc struct {
|
||||
|
||||
var rotationOrder = []int{0, 90, 180, 270}
|
||||
|
||||
func (m *mockRotationDoc) DLA(_ context.Context, _ image.Image) ([]DLARegion, error) { return nil, nil }
|
||||
func (m *mockRotationDoc) TSR(_ context.Context, _ image.Image) ([]TSRCell, error) { return nil, nil }
|
||||
func (m *mockRotationDoc) OCR(_ image.Image) (string, error) { return "", nil }
|
||||
func (m *mockRotationDoc) Health() bool { return true }
|
||||
func (m *mockRotationDoc) ModelType() ModelType { return ModelSaas }
|
||||
func (m *mockRotationDoc) DLA(_ context.Context, _ image.Image) ([]pdf.DLARegion, error) {
|
||||
return nil, nil
|
||||
}
|
||||
func (m *mockRotationDoc) TSR(_ context.Context, _ image.Image) ([]pdf.TSRCell, error) {
|
||||
return nil, nil
|
||||
}
|
||||
func (m *mockRotationDoc) OCR(_ image.Image) (string, error) { return "", nil }
|
||||
func (m *mockRotationDoc) Health() bool { return true }
|
||||
|
||||
func (m *mockRotationDoc) currentAngle() int {
|
||||
idx := m.callSeq % len(rotationOrder)
|
||||
return rotationOrder[idx]
|
||||
}
|
||||
|
||||
func (m *mockRotationDoc) OCRDetect(_ context.Context, img image.Image) ([]OCRBox, error) {
|
||||
func (m *mockRotationDoc) OCRDetect(_ context.Context, img image.Image) ([]pdf.OCRBox, error) {
|
||||
defer func() { m.callSeq++ }()
|
||||
angle := m.currentAngle()
|
||||
cfg, ok := m.angles[angle]
|
||||
@@ -47,11 +51,11 @@ func (m *mockRotationDoc) OCRDetect(_ context.Context, img image.Image) ([]OCRBo
|
||||
return nil, nil
|
||||
}
|
||||
w, h := img.Bounds().Dx(), img.Bounds().Dy()
|
||||
boxes := make([]OCRBox, cfg.regions)
|
||||
boxes := make([]pdf.OCRBox, cfg.regions)
|
||||
step := w / (cfg.regions + 1)
|
||||
for i := 0; i < cfg.regions; i++ {
|
||||
x := step * (i + 1)
|
||||
boxes[i] = OCRBox{
|
||||
boxes[i] = pdf.OCRBox{
|
||||
X0: float64(x), Y0: float64(h / 4),
|
||||
X1: float64(x + 20), Y1: float64(h / 4),
|
||||
X2: float64(x + 20), Y2: float64(h * 3 / 4),
|
||||
@@ -61,8 +65,8 @@ func (m *mockRotationDoc) OCRDetect(_ context.Context, img image.Image) ([]OCRBo
|
||||
return boxes, nil
|
||||
}
|
||||
|
||||
func (m *mockRotationDoc) OCRRecognizeBatch(_ context.Context, cropped []image.Image) ([][]OCRText, []error) {
|
||||
results := make([][]OCRText, len(cropped))
|
||||
func (m *mockRotationDoc) OCRRecognizeBatch(_ context.Context, cropped []image.Image) ([][]pdf.OCRText, []error) {
|
||||
results := make([][]pdf.OCRText, len(cropped))
|
||||
errs := make([]error, len(cropped))
|
||||
for i, img := range cropped {
|
||||
results[i], errs[i] = m.OCRRecognize(context.Background(), img)
|
||||
@@ -70,7 +74,7 @@ func (m *mockRotationDoc) OCRRecognizeBatch(_ context.Context, cropped []image.I
|
||||
return results, errs
|
||||
}
|
||||
|
||||
func (m *mockRotationDoc) OCRRecognize(_ context.Context, _ image.Image) ([]OCRText, error) {
|
||||
func (m *mockRotationDoc) OCRRecognize(_ context.Context, _ image.Image) ([]pdf.OCRText, error) {
|
||||
angle := rotationOrder[(m.callSeq-1)%len(rotationOrder)] // use angle from last Detect call
|
||||
cfg, ok := m.angles[angle]
|
||||
if !ok {
|
||||
@@ -82,9 +86,9 @@ func (m *mockRotationDoc) OCRRecognize(_ context.Context, _ image.Image) ([]OCRT
|
||||
if cfg.regions == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
texts := make([]OCRText, cfg.regions)
|
||||
texts := make([]pdf.OCRText, cfg.regions)
|
||||
for i := 0; i < cfg.regions; i++ {
|
||||
texts[i] = OCRText{Text: "X", Confidence: cfg.avgConf}
|
||||
texts[i] = pdf.OCRText{Text: "X", Confidence: cfg.avgConf}
|
||||
}
|
||||
return texts, nil
|
||||
}
|
||||
@@ -104,7 +108,7 @@ func TestEvaluateTableOrientation(t *testing.T) {
|
||||
0: {regions: 10, avgConf: 0.9},
|
||||
},
|
||||
}
|
||||
angle, _, scores := evaluateTableOrientation(context.Background(), makeTestTableImage(), doc)
|
||||
angle, _, scores := EvaluateTableOrientation(context.Background(), makeTestTableImage(), doc)
|
||||
if angle != 0 {
|
||||
t.Errorf("expected 0°, got %d° (scores: %v)", angle, scores)
|
||||
}
|
||||
@@ -123,7 +127,7 @@ func TestEvaluateTableOrientation(t *testing.T) {
|
||||
270: {regions: 2, avgConf: 0.2},
|
||||
},
|
||||
}
|
||||
angle, _, scores := evaluateTableOrientation(context.Background(), makeTestTableImage(), doc)
|
||||
angle, _, scores := EvaluateTableOrientation(context.Background(), makeTestTableImage(), doc)
|
||||
if angle != 90 {
|
||||
t.Errorf("expected 90°, got %d° (scores: %v)", angle, scores)
|
||||
}
|
||||
@@ -142,7 +146,7 @@ func TestEvaluateTableOrientation(t *testing.T) {
|
||||
270: {regions: 1, avgConf: 0.1},
|
||||
},
|
||||
}
|
||||
angle, _, scores := evaluateTableOrientation(context.Background(), makeTestTableImage(), doc)
|
||||
angle, _, scores := EvaluateTableOrientation(context.Background(), makeTestTableImage(), doc)
|
||||
if angle != 180 {
|
||||
t.Errorf("expected 180°, got %d° (scores: %v)", angle, scores)
|
||||
}
|
||||
@@ -161,7 +165,7 @@ func TestEvaluateTableOrientation(t *testing.T) {
|
||||
270: {regions: 9, avgConf: 0.88},
|
||||
},
|
||||
}
|
||||
angle, _, scores := evaluateTableOrientation(context.Background(), makeTestTableImage(), doc)
|
||||
angle, _, scores := EvaluateTableOrientation(context.Background(), makeTestTableImage(), doc)
|
||||
if angle != 270 {
|
||||
t.Errorf("expected 270°, got %d° (scores: %v)", angle, scores)
|
||||
}
|
||||
@@ -179,7 +183,7 @@ func TestEvaluateTableOrientation(t *testing.T) {
|
||||
90: {regions: 9},
|
||||
},
|
||||
}
|
||||
angle, _, _ := evaluateTableOrientation(context.Background(), makeTestTableImage(), doc)
|
||||
angle, _, _ := EvaluateTableOrientation(context.Background(), makeTestTableImage(), doc)
|
||||
if angle != 0 {
|
||||
t.Errorf("expected 0° (threshold protection), got %d°", angle)
|
||||
}
|
||||
@@ -197,7 +201,7 @@ func TestEvaluateTableOrientation(t *testing.T) {
|
||||
90: {regions: 10},
|
||||
},
|
||||
}
|
||||
angle, _, _ := evaluateTableOrientation(context.Background(), makeTestTableImage(), doc)
|
||||
angle, _, _ := EvaluateTableOrientation(context.Background(), makeTestTableImage(), doc)
|
||||
if angle != 90 {
|
||||
t.Errorf("expected 90° (threshold passed), got %d°", angle)
|
||||
}
|
||||
@@ -216,7 +220,7 @@ func TestEvaluateTableOrientation(t *testing.T) {
|
||||
270: {err: errMockOCR},
|
||||
},
|
||||
}
|
||||
angle, img, scores := evaluateTableOrientation(context.Background(), makeTestTableImage(), doc)
|
||||
angle, img, scores := EvaluateTableOrientation(context.Background(), makeTestTableImage(), doc)
|
||||
if angle != 0 {
|
||||
t.Errorf("expected 0° fallback, got %d°", angle)
|
||||
}
|
||||
@@ -1,12 +1,13 @@
|
||||
//go:build manual
|
||||
|
||||
package parser
|
||||
package table
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/base64"
|
||||
"image"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"regexp"
|
||||
"strings"
|
||||
"testing"
|
||||
@@ -24,19 +25,19 @@ import (
|
||||
// figure boxes are popped and re-inserted via insert_table_figures with cropped
|
||||
// images. Go leaves them in the box list for downstream boxesToSections.
|
||||
func TestExtractTableAndReplace_IgnoresFigures(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 10, X1: 200, Top: 0, Bottom: 50, Text: "Figure text", LayoutType: "figure", PageNumber: 0},
|
||||
{X0: 10, X1: 200, Top: 60, Bottom: 80, Text: "表1:标题", LayoutType: "table", PageNumber: 0},
|
||||
}
|
||||
|
||||
// Table with cells so extractTableAndReplace generates HTML.
|
||||
tables := []TableItem{{
|
||||
Cells: []TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "A", Label: "table row"}},
|
||||
Positions: []Position{{Left: 0, Right: 300, Top: 0, Bottom: 100}},
|
||||
tables := []pdf.TableItem{{
|
||||
Cells: []pdf.TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "A", Label: "table row"}},
|
||||
Positions: []pdf.Position{{Left: 0, Right: 300, Top: 0, Bottom: 100}},
|
||||
Scale: 1.0,
|
||||
}}
|
||||
|
||||
result := extractTableAndReplace(boxes, tables)
|
||||
result := ExtractTableAndReplace(boxes, tables)
|
||||
|
||||
// BUG: Figure box is still present — it was not popped or replaced.
|
||||
// Python's _extract_table_figure pops figure boxes and re-inserts them
|
||||
@@ -55,7 +56,7 @@ func TestExtractTableAndReplace_IgnoresFigures(t *testing.T) {
|
||||
if !hasFigure {
|
||||
t.Error("BUG EXPOSED: extractTableAndReplace removed figure box (unexpected)")
|
||||
}
|
||||
t.Log("NOTE: Figure box remains in list as raw text. Python inserts figures back with cropped images via insert_table_figures. Go collects figures separately via CollectFigures without re-inserting.")
|
||||
t.Log("NOTE: Figure box remains in list as raw text. Python inserts figures back with cropped images via insert_table_figures. Go collects figures separately via pdf.CollectFigures without re-inserting.")
|
||||
}
|
||||
|
||||
// TestBoxesToSections_FiguresNotReinserted documents that boxesToSections converts
|
||||
@@ -63,13 +64,13 @@ func TestExtractTableAndReplace_IgnoresFigures(t *testing.T) {
|
||||
// insert_table_figures would attach.
|
||||
func TestBoxesToSections_FiguresNotReinserted(t *testing.T) {
|
||||
// Simulate post-extractTableAndReplace boxes with figures still present.
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 10, X1: 200, Top: 0, Bottom: 50, Text: "Some text", LayoutType: "text", PageNumber: 0},
|
||||
{X0: 10, X1: 200, Top: 60, Bottom: 100, Text: "Figure description", LayoutType: "figure", PageNumber: 0},
|
||||
}
|
||||
|
||||
sections := boxesToSections(boxes, nil)
|
||||
figures := CollectFigures(sections)
|
||||
figures := pdf.CollectFigures(sections)
|
||||
|
||||
// BUG: figures are collected separately but NOT re-inserted into sections
|
||||
// after image processing. In Python, insert_table_figures(figs, "figure")
|
||||
@@ -108,15 +109,15 @@ func TestConstructTable_HeaderDetection_NoBlockType(t *testing.T) {
|
||||
// With blockType: "2020","2021" → Nu, "100","200" → Nu — maxType=Nu.
|
||||
// block-type-aware detection skips Nu cells → 0 headers.
|
||||
// Falls back to TSR label-based detection → still gets 2 <th >.
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "2020", Label: "table column header"},
|
||||
{X0: 101, Y0: 0, X1: 200, Y1: 30, Text: "2021", Label: "table column header"},
|
||||
{X0: 0, Y0: 35, X1: 100, Y1: 65, Text: "100", Label: "table row"},
|
||||
{X0: 101, Y0: 35, X1: 200, Y1: 65, Text: "200", Label: "table row"},
|
||||
}
|
||||
|
||||
item := &TableItem{}
|
||||
html := constructTable(cells, nil, "", item)
|
||||
item := &pdf.TableItem{}
|
||||
html := ConstructTable(cells, nil, "", item)
|
||||
|
||||
// FIX VERIFIED: headerSetWithBlockType computes block types (all "Nu"),
|
||||
// skips Nu headers when maxType=Nu, then falls back to TSR label detection.
|
||||
@@ -137,7 +138,7 @@ func TestConstructTable_BlockType_DominantTypeMissing(t *testing.T) {
|
||||
// "年份"/"金额" → Tx (short text), "2020"/"1000"/etc → Nu. maxType=Nu.
|
||||
// Header cells are non-Nu → count as headers even under Nu-dominant logic.
|
||||
// FIX: blockType now classifies cells and drives header detection.
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "年份", Label: "table column header"},
|
||||
{X0: 101, Y0: 0, X1: 200, Y1: 30, Text: "金额", Label: "table column header"},
|
||||
{X0: 0, Y0: 35, X1: 100, Y1: 65, Text: "2020", Label: "table row"},
|
||||
@@ -148,8 +149,8 @@ func TestConstructTable_BlockType_DominantTypeMissing(t *testing.T) {
|
||||
{X0: 101, Y0: 105, X1: 200, Y1: 135, Text: "3000", Label: "table row"},
|
||||
}
|
||||
|
||||
item := &TableItem{}
|
||||
html := constructTable(cells, nil, "", item)
|
||||
item := &pdf.TableItem{}
|
||||
html := ConstructTable(cells, nil, "", item)
|
||||
|
||||
thCount := strings.Count(html, "<th ")
|
||||
if thCount != 2 {
|
||||
@@ -157,7 +158,7 @@ func TestConstructTable_BlockType_DominantTypeMissing(t *testing.T) {
|
||||
}
|
||||
|
||||
t.Log("FIX: blockType classifies '年份'/'金额' as non-Nu headers, '2020'/'1000' as Nu data.")
|
||||
t.Logf("blockType('年份')=%q blockType('2020')=%q", blockType("年份"), blockType("2020"))
|
||||
t.Logf("BlockType('年份')=%q BlockType('2020')=%q", BlockType("年份"), BlockType("2020"))
|
||||
}
|
||||
|
||||
// TestConstructTable_BlockTypeChangesHeaderDetection verifies blockType
|
||||
@@ -167,7 +168,7 @@ func TestConstructTable_BlockTypeChangesHeaderDetection(t *testing.T) {
|
||||
// Table with NO "header" labels — label-based detection gives 0 headers.
|
||||
// blockType: "姓名"/"年龄" → Tx, "张三"/"25" → Ot/En/? — maxType varies.
|
||||
// With Nu-dominant data, non-Nu top row cells count as possible headers.
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "姓名", Label: "table row"},
|
||||
{X0: 101, Y0: 0, X1: 200, Y1: 30, Text: "年龄", Label: "table row"},
|
||||
{X0: 0, Y0: 35, X1: 100, Y1: 65, Text: "张三", Label: "table row"},
|
||||
@@ -178,7 +179,7 @@ func TestConstructTable_BlockTypeChangesHeaderDetection(t *testing.T) {
|
||||
{X0: 101, Y0: 105, X1: 200, Y1: 135, Text: "28", Label: "table row"},
|
||||
}
|
||||
|
||||
html := constructTable(cells, nil, "", &TableItem{Grid: groupTSRCellsToRowsLabeled(cells)})
|
||||
html := ConstructTable(cells, nil, "", &pdf.TableItem{Grid: GroupTSRCellsToRows(cells)})
|
||||
|
||||
// blockType analysis:
|
||||
// "姓名"(Tx), "年龄"(Tx), "张三"(Ot), "25"(Nu), "李四"(Ot), "30"(Nu), "王五"(Ot), "28"(Nu)
|
||||
@@ -200,7 +201,7 @@ func TestConstructTable_BlockTypeChangesHeaderDetection(t *testing.T) {
|
||||
func TestRowsToHTML_NoColspanRowspan(t *testing.T) {
|
||||
// Two rows with a spanning cell in row 0.
|
||||
// In Python, a "table spanning cell" covering columns 0-1 would get colspan=2.
|
||||
rows := [][]TSRCell{
|
||||
rows := [][]pdf.TSRCell{
|
||||
{
|
||||
{Text: "跨列标题", Label: "table spanning cell"},
|
||||
{Text: "", Label: ""}, // padded cell
|
||||
@@ -211,7 +212,7 @@ func TestRowsToHTML_NoColspanRowspan(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
html := rowsToHTML(rows, "", nil, nil, nil)
|
||||
html := RowsToHTML(rows, "", nil, nil, nil)
|
||||
|
||||
// BUG: No colspan or rowspan attributes in output.
|
||||
if strings.Contains(html, "colspan") {
|
||||
@@ -237,10 +238,10 @@ func TestRowsToHTML_NoColspanRowspan(t *testing.T) {
|
||||
// TestConstructTable_SpannedTable_NoMerge documents the full constructTable
|
||||
// path with spanning cells — no colspan/rowspan in output.
|
||||
func TestConstructTable_SpannedTable_NoMerge(t *testing.T) {
|
||||
// Spanning cell at same Y as row cells so groupTSRCellsToRowsLabeled
|
||||
// Spanning cell at same Y as row cells so GroupTSRCellsToRows
|
||||
// puts them in the same row group. The spanning cell covers X=0-200
|
||||
// (both columns); Python's __cal_spans would give it colspan=2.
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
// Row 0: a spanning cell that covers both columns + one regular cell.
|
||||
{X0: 0, Y0: 0, X1: 200, Y1: 30, Text: "部门开支汇总", Label: "table spanning cell"},
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "Q1", Label: "table row"},
|
||||
@@ -250,8 +251,8 @@ func TestConstructTable_SpannedTable_NoMerge(t *testing.T) {
|
||||
{X0: 101, Y0: 35, X1: 200, Y1: 65, Text: "200", Label: "table row"},
|
||||
}
|
||||
|
||||
item := &TableItem{}
|
||||
html := constructTable(cells, nil, "", item)
|
||||
item := &pdf.TableItem{}
|
||||
html := ConstructTable(cells, nil, "", item)
|
||||
|
||||
// Verify colspan IS now detected (calSpans aligned with Python's __cal_spans).
|
||||
if !strings.Contains(html, "colspan") {
|
||||
@@ -279,7 +280,7 @@ func TestConstructTable_SpannedTable_NoMerge(t *testing.T) {
|
||||
func TestConstructTable_OrphanColumn_NotCleanedUp(t *testing.T) {
|
||||
// 4 rows × 3 columns. Column index 1 has only ONE non-empty cell.
|
||||
// Python would relocate/merge that orphan column.
|
||||
cells := []TSRCell{
|
||||
cells := []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "姓名", Label: "table column header"},
|
||||
{X0: 101, Y0: 0, X1: 200, Y1: 30, Text: "备注", Label: "table row"}, // orphan col
|
||||
{X0: 201, Y0: 0, X1: 300, Y1: 30, Text: "年龄", Label: "table column header"},
|
||||
@@ -294,8 +295,8 @@ func TestConstructTable_OrphanColumn_NotCleanedUp(t *testing.T) {
|
||||
{X0: 201, Y0: 105, X1: 300, Y1: 135, Text: "28", Label: "table row"},
|
||||
}
|
||||
|
||||
item := &TableItem{}
|
||||
html := constructTable(cells, nil, "", item)
|
||||
item := &pdf.TableItem{}
|
||||
html := ConstructTable(cells, nil, "", item)
|
||||
|
||||
// BUG: All 4 rows have 3 cells each (orphan column preserved).
|
||||
// Python's construct_table pops single-cell columns when ≥4 rows.
|
||||
@@ -322,17 +323,17 @@ func TestMergeCaptions_NoIsCaptionPatternMatch(t *testing.T) {
|
||||
// A caption-like text labeled as "text" by DLA (happens with imperfect DLA).
|
||||
// Python's is_caption would match "表1:测试数据" pattern regardless of layout_type.
|
||||
// FIX: mergeCaptions now calls captionKind → isCaptionBox to detect these.
|
||||
sections := []Section{
|
||||
{Text: "T", LayoutType: "table", Positions: []Position{
|
||||
sections := []pdf.Section{
|
||||
{Text: "T", LayoutType: "table", Positions: []pdf.Position{
|
||||
{PageNumbers: []int{0, 0}, Left: 10, Right: 100, Top: 0, Bottom: 30},
|
||||
}},
|
||||
// This is clearly a table caption by text pattern, but DLA labeled it as "text".
|
||||
{Text: "表1:测试数据", LayoutType: "text", Positions: []Position{
|
||||
{Text: "表1:测试数据", LayoutType: "text", Positions: []pdf.Position{
|
||||
{PageNumbers: []int{0, 0}, Left: 10, Right: 100, Top: 40, Bottom: 55},
|
||||
}},
|
||||
}
|
||||
|
||||
figures := CollectFigures(sections)
|
||||
figures := pdf.CollectFigures(sections)
|
||||
result := mergeCaptions(sections, figures)
|
||||
|
||||
// FIX VERIFIED: "表1:测试数据" should be detected as caption via isCaptionBox
|
||||
@@ -381,9 +382,9 @@ func TestIsCaptionBox_MatchesChinesePattern(t *testing.T) {
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := isCaptionBox(tt.text, tt.layoutType)
|
||||
got := IsCaptionBox(tt.text, tt.layoutType)
|
||||
if got != tt.want {
|
||||
t.Errorf("isCaptionBox(%q, %q) = %v, want %v", tt.text, tt.layoutType, got, tt.want)
|
||||
t.Errorf("IsCaptionBox(%q, %q) = %v, want %v", tt.text, tt.layoutType, got, tt.want)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -400,7 +401,7 @@ func TestFigureInsertion_EndToEnd(t *testing.T) {
|
||||
eng := &mockEngine{
|
||||
pageCount: 1,
|
||||
renderW: 1800, renderH: 2400,
|
||||
chars: map[int][]TextChar{0: {
|
||||
chars: map[int][]pdf.TextChar{0: {
|
||||
// Two text boxes in the SAME figure DLA region, but far apart.
|
||||
// DLA pixel: X=100-500 Y=80-600 → PDF 33-167 x 27-200.
|
||||
// Box 1 near top, box 2 near bottom.
|
||||
@@ -410,12 +411,12 @@ func TestFigureInsertion_EndToEnd(t *testing.T) {
|
||||
}
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []DLARegion{
|
||||
DLARegions: []pdf.DLARegion{
|
||||
// Large figure region covering both text boxes.
|
||||
{X0: 100, Y0: 80, X1: 500, Y1: 600, Label: "figure", Confidence: 0.9},
|
||||
},
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -432,9 +433,9 @@ func TestFigureInsertion_EndToEnd(t *testing.T) {
|
||||
// Figure boxes stay in list. NaiveVerticalMerge may NOT merge them
|
||||
// if the gap is too large (> 1.5 × median_height ≈ 15pt).
|
||||
// Each figure text box → separate section in result.Sections.
|
||||
// CollectFigures collects them into result.Figures but doesn't re-insert.
|
||||
// pdf.CollectFigures collects them into result.Figures() but doesn't re-insert.
|
||||
|
||||
var figureSections []Section
|
||||
var figureSections []pdf.Section
|
||||
for _, s := range result.Sections {
|
||||
if s.LayoutType == "figure" {
|
||||
figureSections = append(figureSections, s)
|
||||
@@ -456,7 +457,7 @@ func TestFigureInsertion_EndToEnd(t *testing.T) {
|
||||
}
|
||||
|
||||
t.Logf("figure sections in Sections: %d", len(figureSections))
|
||||
t.Logf("result.Figures count: %d", len(result.Figures))
|
||||
t.Logf("result.Figures() count: %d", len(result.Figures()))
|
||||
t.Logf("result.Sections total: %d", len(result.Sections))
|
||||
for i, s := range result.Sections {
|
||||
t.Logf(" section[%d] layout=%q text=%q", i, s.LayoutType, s.Text)
|
||||
@@ -476,27 +477,27 @@ func TestExtractTableAndReplace_NoCrossPageMerge(t *testing.T) {
|
||||
// Simulate a table spanning pages 0 and 1.
|
||||
// Python would merge these because: same layoutno, consecutive pages,
|
||||
// Y-distance ≤ 23× median_height.
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 10, X1: 200, Top: 500, Bottom: 530, Text: "续表内容", LayoutType: "table", PageNumber: 0, LayoutNo: "0"},
|
||||
{X0: 10, X1: 200, Top: 50, Bottom: 80, Text: "表尾内容", LayoutType: "table", PageNumber: 1, LayoutNo: "0"},
|
||||
}
|
||||
|
||||
// Two separate TableItems — one per page. Python would merge these
|
||||
// before insert_table_figures.
|
||||
tables := []TableItem{
|
||||
tables := []pdf.TableItem{
|
||||
{
|
||||
Cells: []TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "Page0", Label: "table row"}},
|
||||
Positions: []Position{{PageNumbers: []int{0}, Left: 0, Right: 300, Top: 500, Bottom: 530}},
|
||||
Cells: []pdf.TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "Page0", Label: "table row"}},
|
||||
Positions: []pdf.Position{{PageNumbers: []int{0}, Left: 0, Right: 300, Top: 500, Bottom: 530}},
|
||||
Scale: 1.0,
|
||||
},
|
||||
{
|
||||
Cells: []TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "Page1", Label: "table row"}},
|
||||
Positions: []Position{{PageNumbers: []int{1}, Left: 0, Right: 300, Top: 50, Bottom: 80}},
|
||||
Cells: []pdf.TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "Page1", Label: "table row"}},
|
||||
Positions: []pdf.Position{{PageNumbers: []int{1}, Left: 0, Right: 300, Top: 50, Bottom: 80}},
|
||||
Scale: 1.0,
|
||||
},
|
||||
}
|
||||
|
||||
result := extractTableAndReplace(boxes, tables)
|
||||
result := ExtractTableAndReplace(boxes, tables)
|
||||
|
||||
// Go produces 2 separate HTML table boxes (one per page).
|
||||
// Python would produce 1 merged table with cells from both pages.
|
||||
@@ -534,10 +535,10 @@ func TestMergeTablesAcrossPages_NomergeAfterCaption_Missing(t *testing.T) {
|
||||
// then another table. Page 1 has the same-layoutNo table continuing.
|
||||
// In Python, page 0's first table goes into nomerge_lout_no because
|
||||
// the next box is a caption → no cross-page merge for that table group.
|
||||
tables := []TableItem{
|
||||
tables := []pdf.TableItem{
|
||||
{
|
||||
Cells: []TSRCell{{Text: "Page0-first", Label: "table row"}},
|
||||
Positions: []Position{{
|
||||
Cells: []pdf.TSRCell{{Text: "Page0-first", Label: "table row"}},
|
||||
Positions: []pdf.Position{{
|
||||
PageNumbers: []int{0},
|
||||
Left: 0, Right: 300,
|
||||
Top: 0, Bottom: 50,
|
||||
@@ -545,8 +546,8 @@ func TestMergeTablesAcrossPages_NomergeAfterCaption_Missing(t *testing.T) {
|
||||
NoMerge: true, // Set when caption follows this table on the page
|
||||
},
|
||||
{
|
||||
Cells: []TSRCell{{Text: "Page1-cont", Label: "table row"}},
|
||||
Positions: []Position{{
|
||||
Cells: []pdf.TSRCell{{Text: "Page1-cont", Label: "table row"}},
|
||||
Positions: []pdf.Position{{
|
||||
PageNumbers: []int{1},
|
||||
Left: 0, Right: 300,
|
||||
Top: 0, Bottom: 50,
|
||||
@@ -554,7 +555,7 @@ func TestMergeTablesAcrossPages_NomergeAfterCaption_Missing(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
result := mergeTablesAcrossPages(tables, nil)
|
||||
result := MergeTablesAcrossPages(tables, nil)
|
||||
|
||||
// Verify NoMerge prevents cross-page merging.
|
||||
if len(result) != 2 {
|
||||
@@ -584,7 +585,7 @@ func TestExtractTableAndReplace_InsertionPosition_DistanceBug(t *testing.T) {
|
||||
// → insert AFTER L0. Result: [L0, table, R0, R1, L2].
|
||||
// Go: anchor = first table box (L1 at index 2). Result: [L0, R0, table, R1, L2].
|
||||
// The table is one position off.
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 10, X1: 100, Top: 10, Bottom: 30, Text: "L0", LayoutType: "text", PageNumber: 0},
|
||||
{X0: 300, X1: 400, Top: 10, Bottom: 30, Text: "R0", LayoutType: "text", PageNumber: 0},
|
||||
{X0: 10, X1: 100, Top: 100, Bottom: 130, Text: "table", LayoutType: "table", PageNumber: 0},
|
||||
@@ -592,14 +593,14 @@ func TestExtractTableAndReplace_InsertionPosition_DistanceBug(t *testing.T) {
|
||||
{X0: 10, X1: 100, Top: 250, Bottom: 270, Text: "L2", LayoutType: "text", PageNumber: 0},
|
||||
}
|
||||
|
||||
tables := []TableItem{{
|
||||
Cells: []TSRCell{{Text: "cell", Label: "table row"}},
|
||||
Positions: []Position{{Left: 10, Right: 100, Top: 100, Bottom: 130, PageNumbers: []int{0}}},
|
||||
tables := []pdf.TableItem{{
|
||||
Cells: []pdf.TSRCell{{Text: "cell", Label: "table row"}},
|
||||
Positions: []pdf.Position{{Left: 10, Right: 100, Top: 100, Bottom: 130, PageNumbers: []int{0}}},
|
||||
Scale: 1.0,
|
||||
RegionLeft: 10, RegionRight: 100, RegionTop: 100, RegionBottom: 130,
|
||||
}}
|
||||
|
||||
result := extractTableAndReplace(boxes, tables)
|
||||
result := ExtractTableAndReplace(boxes, tables)
|
||||
|
||||
// Find L0 and table positions.
|
||||
l0Idx, tableIdx := -1, -1
|
||||
@@ -636,7 +637,7 @@ func TestExtractTableAndReplace_InsertionPosition_DistanceBug(t *testing.T) {
|
||||
// coordinates (subtracts page_cum_height). The page number differentiates
|
||||
// pages; page_cum_height is an internal implementation detail.
|
||||
func TestBoxesToSections_PerPageCoordinates(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 10, X1: 100, Top: 40, Bottom: 60, Text: "Page 0 text", LayoutType: "text", PageNumber: 0},
|
||||
{X0: 10, X1: 100, Top: 40, Bottom: 60, Text: "Page 1 text", LayoutType: "text", PageNumber: 1},
|
||||
}
|
||||
@@ -673,10 +674,10 @@ func TestCropSectionImage_PaddingVsPython(t *testing.T) {
|
||||
img := image.NewRGBA(image.Rect(0, 0, 300, 800)) // 300×800 page at zoom=3 → PDF 100×267
|
||||
pageImages := map[int]image.Image{0: img}
|
||||
|
||||
// Position tag for a small text box near the top of the page.
|
||||
// pdf.Position tag for a small text box near the top of the page.
|
||||
posTag := FormatPositionTag(0, 50.0, 100.0, 10.0, 30.0)
|
||||
|
||||
result := cropSectionImage(posTag, pageImages, 3.0)
|
||||
result := util.CropSectionImage(posTag, pageImages, 3.0)
|
||||
|
||||
if result == "" {
|
||||
t.Error("cropSectionImage returned empty string for valid position")
|
||||
@@ -757,27 +758,27 @@ func TestDataSourcePattern_RegexCoverage(t *testing.T) {
|
||||
// adding them to the tables dict (pdf_parser.py:1040-1042).
|
||||
func TestExtractTableAndReplace_DataSourceFilter_Missing(t *testing.T) {
|
||||
// A table box with data-source text and a normal table box.
|
||||
// Both overlap a TableItem position, so both would be replaced with HTML.
|
||||
boxes := []TextBox{
|
||||
// Both overlap a pdf.TableItem position, so both would be replaced with HTML.
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 10, X1: 200, Top: 0, Bottom: 50, Text: "数据来源:国家统计局", LayoutType: "table", PageNumber: 0},
|
||||
{X0: 10, X1: 200, Top: 60, Bottom: 80, Text: "表1:正常数据", LayoutType: "table", PageNumber: 0},
|
||||
}
|
||||
|
||||
// Two TableItems — one per table box — so each would independently produce HTML.
|
||||
tables := []TableItem{
|
||||
tables := []pdf.TableItem{
|
||||
{
|
||||
Cells: []TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "来源", Label: "table row"}},
|
||||
Positions: []Position{{Left: 0, Right: 300, Top: 0, Bottom: 50}},
|
||||
Cells: []pdf.TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "来源", Label: "table row"}},
|
||||
Positions: []pdf.Position{{Left: 0, Right: 300, Top: 0, Bottom: 50}},
|
||||
Scale: 1.0,
|
||||
},
|
||||
{
|
||||
Cells: []TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "正常", Label: "table row"}},
|
||||
Positions: []Position{{Left: 0, Right: 300, Top: 60, Bottom: 80}},
|
||||
Cells: []pdf.TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "正常", Label: "table row"}},
|
||||
Positions: []pdf.Position{{Left: 0, Right: 300, Top: 60, Bottom: 80}},
|
||||
Scale: 1.0,
|
||||
},
|
||||
}
|
||||
|
||||
result := extractTableAndReplace(boxes, tables)
|
||||
result := ExtractTableAndReplace(boxes, tables)
|
||||
|
||||
// Python behavior: "数据来源:国家统计局" is popped from self.boxes,
|
||||
// NOT added to tables dict, NOT replaced with HTML. Gone entirely.
|
||||
@@ -792,7 +793,7 @@ func TestExtractTableAndReplace_DataSourceFilter_Missing(t *testing.T) {
|
||||
if strings.Contains(b.Text, "<table>") {
|
||||
htmlTableCount++
|
||||
// The data-source table's cell text "来源" ends up in the HTML.
|
||||
// c.f. constructTable which uses TSRCell text, not box text.
|
||||
// c.f. constructTable which uses pdf.TSRCell text, not box text.
|
||||
if strings.Contains(b.Text, ">来源<") {
|
||||
hasDataSourceTable = true
|
||||
}
|
||||
@@ -822,17 +823,17 @@ func TestExtractTableAndReplace_DataSourceVariants(t *testing.T) {
|
||||
|
||||
for _, variant := range variants {
|
||||
t.Run(variant, func(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 10, X1: 200, Top: 0, Bottom: 50, Text: variant, LayoutType: "table", PageNumber: 0},
|
||||
}
|
||||
|
||||
tables := []TableItem{{
|
||||
Cells: []TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "A", Label: "table row"}},
|
||||
Positions: []Position{{Left: 0, Right: 300, Top: 0, Bottom: 50}},
|
||||
tables := []pdf.TableItem{{
|
||||
Cells: []pdf.TSRCell{{X0: 0, Y0: 0, X1: 100, Y1: 30, Text: "A", Label: "table row"}},
|
||||
Positions: []pdf.Position{{Left: 0, Right: 300, Top: 0, Bottom: 50}},
|
||||
Scale: 1.0,
|
||||
}}
|
||||
|
||||
result := extractTableAndReplace(boxes, tables)
|
||||
result := ExtractTableAndReplace(boxes, tables)
|
||||
|
||||
// BUG: box with data-source text should be REMOVED entirely —
|
||||
// zero HTML output. Python pops these boxes without replacement.
|
||||
@@ -851,12 +852,12 @@ func TestExtractTableAndReplace_DataSourceVariants(t *testing.T) {
|
||||
// Python's _extract_table_figure pops these boxes from self.boxes without
|
||||
// adding them to the figures dict (pdf_parser.py:1050-1052).
|
||||
func TestConsolidateFigures_DataSourceFilter_Missing(t *testing.T) {
|
||||
boxes := []TextBox{
|
||||
boxes := []pdf.TextBox{
|
||||
{X0: 10, X1: 200, Top: 0, Bottom: 50, Text: "数据来源:某机构", LayoutType: "figure", PageNumber: 0, LayoutNo: "figure-0"},
|
||||
{X0: 10, X1: 200, Top: 60, Bottom: 80, Text: "架构图", LayoutType: "figure", PageNumber: 0, LayoutNo: "figure-0"},
|
||||
}
|
||||
|
||||
result := consolidateFigures(boxes)
|
||||
result := ConsolidateFigures(boxes)
|
||||
|
||||
// Python behavior: "数据来源:某机构" is popped from self.boxes,
|
||||
// NOT added to figures dict → gone entirely.
|
||||
@@ -1,11 +1,12 @@
|
||||
//go:build cgo && manual
|
||||
|
||||
package parser
|
||||
package table
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
@@ -42,10 +43,10 @@ func TestTableParityWithPythonBoxes(t *testing.T) {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Convert to Go TextBox
|
||||
boxes := make([]TextBox, len(pyBoxes))
|
||||
// Convert to Go pdf.TextBox
|
||||
boxes := make([]pdf.TextBox, len(pyBoxes))
|
||||
for i, b := range pyBoxes {
|
||||
boxes[i] = TextBox{
|
||||
boxes[i] = pdf.TextBox{
|
||||
X0: b.X0, X1: b.X1, Top: b.Top, Bottom: b.Bottom,
|
||||
Text: b.Text, R: b.R, C: b.C, H: b.H, SP: b.SP,
|
||||
LayoutType: b.LayoutType,
|
||||
@@ -53,8 +54,8 @@ func TestTableParityWithPythonBoxes(t *testing.T) {
|
||||
}
|
||||
|
||||
// Run through Go's constructTable
|
||||
item := &TableItem{}
|
||||
html := constructTable(nil, boxes, "", item)
|
||||
item := &pdf.TableItem{}
|
||||
html := ConstructTable(nil, boxes, "", item)
|
||||
|
||||
if html == "" {
|
||||
t.Error("constructTable returned empty HTML")
|
||||
349
internal/deepdoc/parser/pdf/table/table_post.go
Normal file
349
internal/deepdoc/parser/pdf/table/table_post.go
Normal file
@@ -0,0 +1,349 @@
|
||||
package table
|
||||
|
||||
import (
|
||||
"log/slog"
|
||||
"math"
|
||||
"sort"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// extractTableAndReplace pops table boxes and replaces them with consolidated
|
||||
// HTML boxes (one per table). This matches Python's _extract_table_figure which
|
||||
// pops all boxes inside a table DLA region and inserts a single HTML box.
|
||||
//
|
||||
// Table boxes whose text matches the data-source discard pattern
|
||||
// (r"(数据|资料|图表)*来源[:: ]") are removed entirely without replacement —
|
||||
// matching Python's _extract_table_figure discard behavior.
|
||||
|
||||
// MarkNoMergeTables traverses boxes in page order. When a caption, title, or
|
||||
// reference immediately follows a table, the preceding table is marked NoMerge
|
||||
// to prevent cross-page merge. Matches Python's nomerge_lout_no.
|
||||
func MarkNoMergeTables(boxes []pdf.TextBox, tables []pdf.TableItem) {
|
||||
var lastTableTI int = -1
|
||||
for i := range boxes {
|
||||
lt := boxes[i].LayoutType
|
||||
if lt == pdf.LayoutTypeTable {
|
||||
matched := false
|
||||
for ti := range tables {
|
||||
for _, tp := range tables[ti].Positions {
|
||||
if boxOverlapsPosition(boxes[i], tp) {
|
||||
lastTableTI = ti
|
||||
matched = true
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
if !matched {
|
||||
lastTableTI = -1
|
||||
}
|
||||
continue
|
||||
}
|
||||
if lastTableTI >= 0 && (lt == pdf.LayoutTypeTitle || lt == pdf.DLALabelTableCaption || lt == pdf.DLALabelFigureCaption || lt == pdf.LayoutTypeReference || IsCaptionBox(boxes[i].Text, lt)) {
|
||||
tables[lastTableTI].NoMerge = true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// boxes must be post-TextMerge + post-VerticalMerge. pdf.TableItem.Cells are in
|
||||
// crop pixel space; boxes are in PDF point space — conversion via Scale/CropOff.
|
||||
// replacement pairs a table index with the box index it replaces.
|
||||
type replacement struct {
|
||||
tableIdx int
|
||||
boxIdx int
|
||||
}
|
||||
|
||||
// buildReplacements scans for data-source-attribution boxes to remove and maps
|
||||
// each table to overlapping table-layout boxes, producing the replacement list.
|
||||
func buildReplacements(boxes []pdf.TextBox, tables []pdf.TableItem) (map[int]bool, []replacement) {
|
||||
removeSet := make(map[int]bool)
|
||||
for i := range boxes {
|
||||
if boxes[i].LayoutType == pdf.LayoutTypeTable && isDataSourceBox(boxes[i].Text) {
|
||||
removeSet[i] = true
|
||||
}
|
||||
}
|
||||
var reps []replacement
|
||||
for ti := range tables {
|
||||
for i := range boxes {
|
||||
if boxes[i].LayoutType != pdf.LayoutTypeTable || removeSet[i] {
|
||||
continue
|
||||
}
|
||||
for _, tp := range tables[ti].Positions {
|
||||
if boxOverlapsPosition(boxes[i], tp) {
|
||||
reps = append(reps, replacement{tableIdx: ti, boxIdx: i})
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return removeSet, reps
|
||||
}
|
||||
|
||||
func ExtractTableAndReplace(boxes []pdf.TextBox, tables []pdf.TableItem) []pdf.TextBox {
|
||||
if len(tables) == 0 {
|
||||
return boxes
|
||||
}
|
||||
// Pre-merge nomerge detection: match Python's nomerge_lout_no.
|
||||
// Traverse boxes in page order. When a caption/title/reference is
|
||||
// found, mark the preceding table group as NoMerge, preventing
|
||||
// cross-page merge when a caption ends a table group.
|
||||
// Python: if is_caption(c) or layout_type in ["table caption", "title",
|
||||
// "figure caption", "reference"]: nomerge_lout_no.append(lst_lout_no)
|
||||
MarkNoMergeTables(boxes, tables)
|
||||
|
||||
// Merge same-layoutno tables across consecutive pages (Python _extract_table_figure).
|
||||
tables = MergeTablesAcrossPages(tables, nil)
|
||||
|
||||
// Pre-scan: mark data-source-attribution table boxes for removal.
|
||||
// Python: if re.match(r"(数据|资料|图表)*来源[:: ]", self.boxes[i]["text"]):
|
||||
// self.boxes.pop(i); continue — box discarded, no HTML replacement.
|
||||
removeSet, replacements := buildReplacements(boxes, tables)
|
||||
|
||||
// Image-only PDFs (0 boxes) may have tables with cells but no
|
||||
// overlapping LayoutType=="table" boxes — generate HTML directly.
|
||||
if len(replacements) == 0 && len(boxes) == 0 {
|
||||
var out []pdf.TextBox
|
||||
for ti := range tables {
|
||||
if len(tables[ti].Cells) == 0 {
|
||||
continue
|
||||
}
|
||||
s := tables[ti].Scale
|
||||
pageGlobalCells := CellSliceToPageSpace(tables[ti].Cells, tables[ti].CropOffX, tables[ti].CropOffY, s)
|
||||
var tableBoxes []pdf.TextBox
|
||||
html := ConstructTable(pageGlobalCells, tableBoxes, tables[ti].Caption, &tables[ti])
|
||||
if html != "" {
|
||||
out = append(out, pdf.TextBox{
|
||||
Text: html, LayoutType: "table", PageNumber: 0,
|
||||
})
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
if len(replacements) == 0 {
|
||||
// No HTML replacements, but data-source boxes still need removal.
|
||||
if len(removeSet) == 0 {
|
||||
return boxes
|
||||
}
|
||||
out := make([]pdf.TextBox, 0, len(boxes)-len(removeSet))
|
||||
for i, b := range boxes {
|
||||
if !removeSet[i] {
|
||||
out = append(out, b)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// Distance-based anchor selection (Python's min_rectangle_distance).
|
||||
// Find the spatially nearest non-table text box for each table and
|
||||
// use that as the anchor, matching insert_table_figures behavior.
|
||||
replacedByTable := make(map[int]int)
|
||||
for ti := range tables {
|
||||
if len(tables[ti].Cells) == 0 {
|
||||
continue
|
||||
}
|
||||
tbl := &tables[ti]
|
||||
tblLeft, tblRight := tbl.RegionLeft, tbl.RegionRight
|
||||
tblTop, tblBottom := tbl.RegionTop, tbl.RegionBottom
|
||||
tblPg := 0
|
||||
if len(tbl.Positions) > 0 {
|
||||
p := tbl.Positions[0]
|
||||
if len(p.PageNumbers) > 0 {
|
||||
tblPg = p.PageNumbers[0]
|
||||
}
|
||||
if tblLeft == 0 && tblRight == 0 && tblTop == 0 && tblBottom == 0 {
|
||||
tblLeft, tblRight = p.Left, p.Right
|
||||
tblTop, tblBottom = p.Top, p.Bottom
|
||||
}
|
||||
}
|
||||
bestDist := math.MaxFloat64
|
||||
bestIdx := -1
|
||||
for i, b := range boxes {
|
||||
if b.LayoutType == pdf.LayoutTypeTable || b.LayoutType == pdf.LayoutTypeFigure {
|
||||
continue
|
||||
}
|
||||
if b.PageNumber != tblPg {
|
||||
continue
|
||||
}
|
||||
dist := minRectangleDistance(
|
||||
b.X0, b.X1, b.Top, b.Bottom,
|
||||
tblLeft, tblRight, tblTop, tblBottom,
|
||||
)
|
||||
if dist < bestDist {
|
||||
bestDist = dist
|
||||
bestIdx = i
|
||||
}
|
||||
}
|
||||
if bestIdx >= 0 {
|
||||
if boxes[bestIdx].Bottom < tblTop {
|
||||
bestIdx++
|
||||
}
|
||||
replacedByTable[ti] = bestIdx
|
||||
} else {
|
||||
for _, r := range replacements {
|
||||
if r.tableIdx == ti {
|
||||
if _, ok := replacedByTable[ti]; !ok || r.boxIdx < replacedByTable[ti] {
|
||||
replacedByTable[ti] = r.boxIdx
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for _, r := range replacements {
|
||||
removeSet[r.boxIdx] = true
|
||||
}
|
||||
|
||||
// Build HTML for each table using post-merge boxes converted to crop space.
|
||||
htmlByTable := make(map[int]string)
|
||||
for ti := range tables {
|
||||
if len(tables[ti].Cells) == 0 {
|
||||
continue
|
||||
}
|
||||
// Convert TSR cells from crop-pixel space to page-global 72 DPI,
|
||||
// matching Python's coordinate space. Text boxes are already in
|
||||
// page-global 72 DPI (from ocrMergeChars), so no conversion needed.
|
||||
s := tables[ti].Scale
|
||||
pageGlobalCells := CellSliceToPageSpace(tables[ti].Cells, tables[ti].CropOffX, tables[ti].CropOffY, s)
|
||||
// Collect only table-labelled boxes (Python: filters by layout_type).
|
||||
var tableBoxes []pdf.TextBox
|
||||
for i := range boxes {
|
||||
if boxes[i].LayoutType != pdf.LayoutTypeTable {
|
||||
continue
|
||||
}
|
||||
for _, tp := range tables[ti].Positions {
|
||||
if boxOverlapsPosition(boxes[i], tp) {
|
||||
tableBoxes = append(tableBoxes, boxes[i])
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
slog.Debug("extractTableAndReplace constructTable", "table", ti, "cells", len(pageGlobalCells), "boxes", len(tableBoxes))
|
||||
htmlByTable[ti] = ConstructTable(pageGlobalCells, tableBoxes, tables[ti].Caption, &tables[ti])
|
||||
}
|
||||
|
||||
// Sort anchors by position for stable insertion.
|
||||
anchorList := make([]struct{ ti, pos int }, 0, len(replacedByTable))
|
||||
for ti, pos := range replacedByTable {
|
||||
anchorList = append(anchorList, struct{ ti, pos int }{ti, pos})
|
||||
}
|
||||
sort.Slice(anchorList, func(i, j int) bool { return anchorList[i].pos < anchorList[j].pos })
|
||||
|
||||
out := make([]pdf.TextBox, 0, len(boxes)-len(removeSet)+len(replacedByTable))
|
||||
anchorIdx := 0
|
||||
for i, b := range boxes {
|
||||
// Insert any HTML boxes whose anchor position is before or at i.
|
||||
for anchorIdx < len(anchorList) && anchorList[anchorIdx].pos <= i {
|
||||
ti := anchorList[anchorIdx].ti
|
||||
html := htmlByTable[ti]
|
||||
if html != "" {
|
||||
tbl := &tables[ti]
|
||||
out = append(out, tableRegionBox(tbl, &b, html))
|
||||
}
|
||||
anchorIdx++
|
||||
}
|
||||
if !removeSet[i] {
|
||||
out = append(out, b)
|
||||
}
|
||||
}
|
||||
// Remaining anchors after last box.
|
||||
for anchorIdx < len(anchorList) {
|
||||
ti := anchorList[anchorIdx].ti
|
||||
html := htmlByTable[ti]
|
||||
if html != "" {
|
||||
tbl := &tables[ti]
|
||||
last := &boxes[len(boxes)-1]
|
||||
out = append(out, tableRegionBox(tbl, last, html))
|
||||
}
|
||||
anchorIdx++
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// consolidateFigures merges figure boxes that share the same LayoutNo
|
||||
// (i.e., belong to the same DLA figure region) into a single pdf.TextBox.
|
||||
// Matches Python's _extract_table_figure + insert_table_figures which pops
|
||||
// individual figure boxes and re-inserts one consolidated figure block
|
||||
// per DLA region with combined text.
|
||||
//
|
||||
// Figure boxes whose text matches the data-source discard pattern
|
||||
// (r"(数据|资料|图表)*来源[:: ]") are removed entirely — matching Python's
|
||||
// _extract_table_figure discard behavior (pdf_parser.py:1050-1052).
|
||||
func ConsolidateFigures(boxes []pdf.TextBox) []pdf.TextBox {
|
||||
// Pre-scan: mark data-source-attribution figure boxes for removal.
|
||||
// Python: if re.match(r"(数据|资料|图表)*来源[:: ]", self.boxes[i]["text"]):
|
||||
// self.boxes.pop(i); continue — box discarded.
|
||||
removeSet := make(map[int]bool)
|
||||
for i, b := range boxes {
|
||||
if b.LayoutType == pdf.LayoutTypeFigure && isDataSourceBox(b.Text) {
|
||||
removeSet[i] = true
|
||||
}
|
||||
}
|
||||
|
||||
// Group figure boxes by (page, layoutno).
|
||||
type figKey struct {
|
||||
page int
|
||||
ln string
|
||||
}
|
||||
groups := make(map[figKey][]int)
|
||||
for i, b := range boxes {
|
||||
if b.LayoutType != pdf.LayoutTypeFigure || removeSet[i] {
|
||||
continue
|
||||
}
|
||||
key := figKey{b.PageNumber, b.LayoutNo}
|
||||
groups[key] = append(groups[key], i)
|
||||
}
|
||||
|
||||
if len(groups) == 0 {
|
||||
// Still need to filter out data-source figure boxes.
|
||||
if len(removeSet) == 0 {
|
||||
return boxes
|
||||
}
|
||||
out := make([]pdf.TextBox, 0, len(boxes)-len(removeSet))
|
||||
for i, b := range boxes {
|
||||
if !removeSet[i] {
|
||||
out = append(out, b)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// Collect indices to remove (all group members except the first).
|
||||
for _, indices := range groups {
|
||||
if len(indices) <= 1 {
|
||||
continue
|
||||
}
|
||||
// Merge into the first box of the group.
|
||||
anchor := indices[0]
|
||||
for _, idx := range indices[1:] {
|
||||
b := boxes[idx]
|
||||
boxes[anchor].Text += "\n" + b.Text
|
||||
boxes[anchor].X0 = math.Min(boxes[anchor].X0, b.X0)
|
||||
boxes[anchor].X1 = math.Max(boxes[anchor].X1, b.X1)
|
||||
boxes[anchor].Top = math.Min(boxes[anchor].Top, b.Top)
|
||||
boxes[anchor].Bottom = math.Max(boxes[anchor].Bottom, b.Bottom)
|
||||
removeSet[idx] = true
|
||||
}
|
||||
}
|
||||
|
||||
if len(removeSet) == 0 {
|
||||
return boxes
|
||||
}
|
||||
|
||||
out := make([]pdf.TextBox, 0, len(boxes)-len(removeSet))
|
||||
for i, b := range boxes {
|
||||
if !removeSet[i] {
|
||||
out = append(out, b)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// boxOverlapsPosition checks if a pdf.TextBox overlaps a pdf.Position with margin.
|
||||
func boxOverlapsPosition(box pdf.TextBox, pos pdf.Position) bool {
|
||||
const margin = 2.0
|
||||
return box.X0 <= pos.Right+margin && box.X1 >= pos.Left-margin &&
|
||||
box.Top <= pos.Bottom+margin && box.Bottom >= pos.Top-margin
|
||||
}
|
||||
|
||||
// rowsToHTML converts grouped TSR cell rows to an HTML table string.
|
||||
// spanInfo maps (row,col) → (colspan, rowspan) for spanning cells;
|
||||
// covered marks cells hidden by a span. Both may be nil.
|
||||
@@ -0,0 +1,61 @@
|
||||
package table
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// ---- regionOverlapsBox ----
|
||||
|
||||
func TestRegionOverlapsBox_StrongOverlap(t *testing.T) {
|
||||
region := pdf.DLARegion{X0: 0, Y0: 0, X1: 216, Y1: 108} // DLA coords at 216 DPI
|
||||
box := pdf.TextBox{X0: 0, X1: 100, Top: 0, Bottom: 50}
|
||||
if !regionOverlapsBox(region, box, 3.0) {
|
||||
t.Error("full overlap should match")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegionOverlapsBox_NoOverlap(t *testing.T) {
|
||||
region := pdf.DLARegion{X0: 0, Y0: 0, X1: 216, Y1: 108}
|
||||
box := pdf.TextBox{X0: 500, X1: 600, Top: 500, Bottom: 550}
|
||||
if regionOverlapsBox(region, box, 3.0) {
|
||||
t.Error("no overlap should return false")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegionOverlapsBox_WeakOverlap(t *testing.T) {
|
||||
// Overlap at 30% → below 40% threshold → false.
|
||||
region := pdf.DLARegion{X0: 0, Y0: 0, X1: 90, Y1: 90} // 30x30 at scale 3
|
||||
box := pdf.TextBox{X0: 0, X1: 100, Top: 0, Bottom: 100} // overlap = 30*30/10000 = 9%? No: 30x30=900 / 10000 = 9%
|
||||
if regionOverlapsBox(region, box, 3.0) {
|
||||
t.Error("9% overlap should return false")
|
||||
}
|
||||
// Overlap ≥ 40% → should match (Python thr=0.4).
|
||||
// box 100x100=10000 area; region 100x40=4000 → exactly 40%.
|
||||
region2 := pdf.DLARegion{X0: 0, Y0: 0, X1: 300, Y1: 120, Label: "table"} // 100x40 at scale 3
|
||||
if !regionOverlapsBox(region2, box, 3.0) {
|
||||
t.Error("40% overlap should match (>= 0.4)")
|
||||
}
|
||||
// Region that covers most of the box → should match
|
||||
region3 := pdf.DLARegion{X0: 0, Y0: 0, X1: 270, Y1: 270} // 90x90 at scale 3
|
||||
if !regionOverlapsBox(region3, box, 3.0) {
|
||||
t.Error("81% overlap should match")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegionOverlapsBox_ThresholdAt040(t *testing.T) {
|
||||
// Exact 40% overlap: 100x100 box, region just covering 40%
|
||||
// 0.4 * 10000 = 4000. Need region with area 4000 in box space.
|
||||
// 63.2*63.2 ≈ 3994. Let's use 100x40 = 4000.
|
||||
box := pdf.TextBox{X0: 0, X1: 100, Top: 0, Bottom: 100}
|
||||
region := pdf.DLARegion{X0: 0, Y0: 0, X1: 300, Y1: 120, Label: "table"} // 100x40 at scale 3
|
||||
if !regionOverlapsBox(region, box, 3.0) {
|
||||
t.Error("exact 40% overlap should match (>= 0.4)")
|
||||
}
|
||||
// 39% overlap should NOT match
|
||||
region2 := pdf.DLARegion{X0: 0, Y0: 0, X1: 300, Y1: 117, Label: "table"} // 100x39 at scale 3
|
||||
if regionOverlapsBox(region2, box, 3.0) {
|
||||
t.Error("39% overlap should NOT match")
|
||||
}
|
||||
}
|
||||
@@ -1,22 +0,0 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"image"
|
||||
)
|
||||
|
||||
// TableBuilder encapsulates TSR model-specific cell detection and grouping.
|
||||
// Each TSR model implements its own Builder, producing a unified row-column
|
||||
// grid consumed by the shared downstream pipeline.
|
||||
type TableBuilder interface {
|
||||
// Name returns the model identifier for logging and diagnostics.
|
||||
Name() string
|
||||
|
||||
// DetectCells detects all cells from a cropped table image.
|
||||
// The Label field on returned TSRCells is consumed only by the Builder
|
||||
// itself during GroupCells; shared code does not depend on Label semantics.
|
||||
DetectCells(ctx context.Context, cropped image.Image) ([]TSRCell, error)
|
||||
|
||||
// GroupCells groups cells into a row-column grid (pure computation, no I/O).
|
||||
GroupCells(cells []TSRCell) [][]TSRCell
|
||||
}
|
||||
242
internal/deepdoc/parser/pdf/table_extract.go
Normal file
242
internal/deepdoc/parser/pdf/table_extract.go
Normal file
@@ -0,0 +1,242 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"image"
|
||||
"log/slog"
|
||||
"math"
|
||||
"sort"
|
||||
|
||||
tbl "ragflow/internal/deepdoc/parser/pdf/table"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
util "ragflow/internal/deepdoc/parser/pdf/util"
|
||||
)
|
||||
|
||||
// enrichWithDeepDoc runs DLA+TSR via p.DeepDoc and returns detected tables.
|
||||
// pageImages optionally provides pre-rendered page images to avoid re-rendering.
|
||||
func (p *Parser) enrichWithDeepDoc(ctx context.Context, result *pdf.ParseResult, engine pdf.PDFEngine, boxes []pdf.TextBox, pageImages map[int]image.Image) []pdf.TableItem {
|
||||
if !p.DeepDoc.Health() {
|
||||
return nil
|
||||
}
|
||||
// Group boxes by page for annotation write-back.
|
||||
byPage := make(map[int][]int)
|
||||
for i, b := range boxes {
|
||||
byPage[b.PageNumber] = append(byPage[b.PageNumber], i)
|
||||
}
|
||||
|
||||
// Collect all pages that have images (from pageImages) or boxes.
|
||||
// This matches Python's __images__ which processes every page regardless
|
||||
// of embedded chars — image-only PDFs still get DLA+TSR.
|
||||
allPages := make(map[int]bool)
|
||||
for pg := range pageImages {
|
||||
allPages[pg] = true
|
||||
}
|
||||
for pg := range byPage {
|
||||
allPages[pg] = true
|
||||
}
|
||||
pageKeys := make([]int, 0, len(allPages))
|
||||
for pg := range allPages {
|
||||
pageKeys = append(pageKeys, pg)
|
||||
}
|
||||
sort.Ints(pageKeys)
|
||||
|
||||
var tableItems []pdf.TableItem
|
||||
for _, pg := range pageKeys {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return tableItems
|
||||
}
|
||||
indices := byPage[pg]
|
||||
pageBoxes := make([]pdf.TextBox, len(indices))
|
||||
for i, idx := range indices {
|
||||
pageBoxes[i] = boxes[idx]
|
||||
}
|
||||
tables := p.extractTableBoxes(ctx, result, pageBoxes, engine, pg, pageImages, len(tableItems))
|
||||
tableItems = append(tableItems, tables...)
|
||||
// Write back DLA and TSR annotations (R/C/H/SP) to the original boxes.
|
||||
for i, idx := range indices {
|
||||
if pageBoxes[i].LayoutType != "" {
|
||||
boxes[idx].LayoutType = pageBoxes[i].LayoutType
|
||||
boxes[idx].LayoutNo = pageBoxes[i].LayoutNo
|
||||
}
|
||||
tbl.CopyBoxAnnotations(&boxes[idx], &pageBoxes[i])
|
||||
}
|
||||
|
||||
}
|
||||
return tableItems
|
||||
}
|
||||
|
||||
func (p *Parser) extractTableBoxes(ctx context.Context, result *pdf.ParseResult, boxes []pdf.TextBox, engine pdf.PDFEngine, pageNum int, pageImages map[int]image.Image, tableBaseIdx int) []pdf.TableItem {
|
||||
pageImg, ok := pageImages[pageNum]
|
||||
if !ok {
|
||||
var err error
|
||||
pageImg, err = renderPageToImage(engine, pageNum)
|
||||
if err != nil {
|
||||
slog.Warn("render page for DeepDoc failed", "page", pageNum, "err", err)
|
||||
return nil
|
||||
}
|
||||
}
|
||||
return p.extractTableBoxesFromImage(ctx, result, boxes, pageImg, pageNum, tableBaseIdx)
|
||||
}
|
||||
|
||||
func (p *Parser) extractTableBoxesFromImage(ctx context.Context, result *pdf.ParseResult, boxes []pdf.TextBox, pageImg image.Image, pageNum int, tableBaseIdx int) []pdf.TableItem {
|
||||
regions, err := p.DeepDoc.DLA(ctx, pageImg)
|
||||
if err != nil {
|
||||
slog.Warn("DLA failed", "page", pageNum, "err", err)
|
||||
return nil
|
||||
}
|
||||
// Collect DLA debug intermediates.
|
||||
if result != nil {
|
||||
result.DLADebug = append(result.DLADebug, pdf.DLAPageRegions{Page: pageNum, Regions: regions})
|
||||
}
|
||||
// Annotate boxes with DLA layout types (title, text, figure, table, ...).
|
||||
scale := pdf.DlaScale
|
||||
boxes = tbl.AnnotateBoxLayouts(boxes, regions, scale, float64(pageImg.Bounds().Dy()))
|
||||
|
||||
tableMatches := tbl.MatchTableRegions(boxes, regions, scale)
|
||||
var items []pdf.TableItem
|
||||
for _, tm := range tableMatches {
|
||||
cropped, cropErr := util.CropImageRegion(pageImg, tm.Region)
|
||||
if cropErr != nil {
|
||||
// DLA returned an invalid region (e.g. x1 < x0). Python
|
||||
// PIL.Image.crop() raises ValueError here; we skip this
|
||||
// table instead of passing a full-page image to TSR.
|
||||
continue
|
||||
}
|
||||
|
||||
// Rotation detection (Python: _evaluate_table_orientation).
|
||||
// If rotated, TSR and OCR use the rotated image; cell coords
|
||||
// are mapped back to original crop space for box matching.
|
||||
autoRotate := p.Config.AutoRotateTables != nil && *p.Config.AutoRotateTables
|
||||
bestAngle := 0
|
||||
origW, origH := cropped.Bounds().Dx(), cropped.Bounds().Dy()
|
||||
tsrImg := cropped
|
||||
if autoRotate {
|
||||
angle, rotated, _ := tbl.EvaluateTableOrientation(ctx, cropped, p.DeepDoc)
|
||||
bestAngle = angle
|
||||
tsrImg = rotated
|
||||
}
|
||||
|
||||
imgB64, encErr := util.EncodeImageToBase64PNG(cropped)
|
||||
if encErr != nil {
|
||||
slog.Warn("table PNG encode failed", "page", pageNum, "err", encErr)
|
||||
}
|
||||
|
||||
var cells []pdf.TSRCell
|
||||
var tsrErr error
|
||||
cells, tsrErr = p.tableBuilder.DetectCells(ctx, tsrImg)
|
||||
if tsrErr != nil {
|
||||
slog.Warn("TSR failed", "page", pageNum, "err", tsrErr)
|
||||
}
|
||||
// Collect TSR raw cells for debug comparison.
|
||||
if tsrErr == nil {
|
||||
for _, c := range cells {
|
||||
if result != nil {
|
||||
result.TSRDebug = append(result.TSRDebug, pdf.TSRRawCell{
|
||||
TableIndex: tableBaseIdx + len(items), Page: pageNum,
|
||||
Label: c.Label, X0: c.X0, Y0: c.Y0, X1: c.X1, Y1: c.Y1,
|
||||
Text: c.Text,
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
// Python margin: w*0.03, h*0.03 (_table_transformer_job:374-376).
|
||||
w := tm.Region.X1 - tm.Region.X0
|
||||
h := tm.Region.Y1 - tm.Region.Y0
|
||||
marginX := w * 0.03
|
||||
marginY := h * 0.03
|
||||
cropOffX := math.Max(0, tm.Region.X0-marginX)
|
||||
cropOffY := math.Max(0, tm.Region.Y0-marginY)
|
||||
|
||||
var boxInCrop []pdf.TextBox
|
||||
if tsrErr == nil && len(cells) > 0 {
|
||||
if bestAngle != 0 {
|
||||
// OCR on rotated image before mapping cells back.
|
||||
// Cells are in rotated-pixel space; OCR works best
|
||||
// on upright text. After mapping, cells move to
|
||||
// original crop space where boxInCrop lives.
|
||||
if !p.Config.SkipOCR {
|
||||
ocrTableCells(ctx, cells, tsrImg, p.DeepDoc)
|
||||
}
|
||||
for i := range cells {
|
||||
cells[i].X0, cells[i].Y0 = util.MapRotatedPointToOriginal(cells[i].X0, cells[i].Y0, bestAngle, origW, origH)
|
||||
cells[i].X1, cells[i].Y1 = util.MapRotatedPointToOriginal(cells[i].X1, cells[i].Y1, bestAngle, origW, origH)
|
||||
}
|
||||
}
|
||||
// Fill cell text from pre-merge boxes, skipping caption boxes
|
||||
// (text entirely above the first TSR cell row).
|
||||
firstCellTop := 1e9
|
||||
for _, c := range cells {
|
||||
if c.Y0 >= 0 && c.Y0 < firstCellTop {
|
||||
firstCellTop = c.Y0
|
||||
}
|
||||
}
|
||||
if firstCellTop == 1e9 {
|
||||
firstCellTop = cells[0].Y0 // fallback if all cells have Y0 < 0
|
||||
}
|
||||
boxInCrop = make([]pdf.TextBox, 0, len(tm.BoxIdx))
|
||||
for _, idx := range tm.BoxIdx {
|
||||
b := boxes[idx]
|
||||
if b.Bottom*scale-cropOffY < firstCellTop {
|
||||
continue // caption box above first TSR cell
|
||||
}
|
||||
boxInCrop = append(boxInCrop, tbl.BoxToCropSpace(b, scale, cropOffX, cropOffY))
|
||||
}
|
||||
}
|
||||
var positions []pdf.Position
|
||||
for _, idx := range tm.BoxIdx {
|
||||
b := boxes[idx]
|
||||
positions = append(positions, pdf.Position{
|
||||
PageNumbers: []int{pageNum},
|
||||
Left: b.X0, Right: b.X1,
|
||||
Top: b.Top, Bottom: b.Bottom,
|
||||
})
|
||||
}
|
||||
// Pre-compute grid from raw TSR cells (without crop offset).
|
||||
// Stored in pdf.TableItem for constructTable; annotateTableBoxes
|
||||
// recomputes with offset cells for spatial matching precision.
|
||||
var grid [][]pdf.TSRCell
|
||||
if len(cells) > 0 {
|
||||
grid = p.tableBuilder.GroupCells(cells)
|
||||
// Fill cell text from boxes in crop space. Works for both
|
||||
// Label-aware grouping (cells rearranged) vs. cross-product (creates new cells).
|
||||
if len(grid) > 0 {
|
||||
flat := tbl.FlattenGrid(grid)
|
||||
tbl.FillCellTextFromBoxes(flat, boxInCrop)
|
||||
idx := 0
|
||||
for ri := range grid {
|
||||
for ci := range grid[ri] {
|
||||
grid[ri][ci].Text = flat[idx].Text
|
||||
idx++
|
||||
}
|
||||
}
|
||||
if bestAngle == 0 && !p.Config.SkipOCR {
|
||||
ocrTableCells(ctx, flat, tsrImg, p.DeepDoc)
|
||||
idx = 0
|
||||
for ri := range grid {
|
||||
for ci := range grid[ri] {
|
||||
grid[ri][ci].Text = flat[idx].Text
|
||||
idx++
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
items = append(items, pdf.TableItem{
|
||||
ImageB64: imgB64,
|
||||
Cells: cells,
|
||||
Grid: grid,
|
||||
Positions: positions,
|
||||
Scale: scale,
|
||||
CropOffX: cropOffX,
|
||||
CropOffY: cropOffY,
|
||||
// DLA region in PDF point space (Python's cropout uses layout region boundaries).
|
||||
RegionLeft: tm.Region.X0 / scale,
|
||||
RegionRight: tm.Region.X1 / scale,
|
||||
RegionTop: tm.Region.Y0 / scale,
|
||||
RegionBottom: tm.Region.Y1 / scale,
|
||||
})
|
||||
|
||||
tbl.WriteTableAnnotations(boxes, tm.BoxIdx, cells, scale, cropOffX, cropOffY, p.tableBuilder)
|
||||
}
|
||||
return items
|
||||
}
|
||||
@@ -6,6 +6,9 @@ import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
inf "ragflow/internal/deepdoc/parser/pdf/inference"
|
||||
tbl "ragflow/internal/deepdoc/parser/pdf/table"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"testing"
|
||||
)
|
||||
|
||||
@@ -29,7 +32,7 @@ func TestTableRotation_Integration(t *testing.T) {
|
||||
if baseURL == "" {
|
||||
baseURL = "http://localhost:9390"
|
||||
}
|
||||
dd, err := NewDeepDocClient(baseURL)
|
||||
dd, err := inf.NewInferenceClient(baseURL)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -52,7 +55,7 @@ func TestTableRotation_Integration(t *testing.T) {
|
||||
pageCount, _ := eng.PageCount()
|
||||
t.Logf("PDF: %d pages", pageCount)
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
cfg.ToPage = pageCount - 1
|
||||
autoRotate := true
|
||||
cfg.AutoRotateTables = &autoRotate
|
||||
@@ -84,7 +87,7 @@ func TestTableRotation_Integration(t *testing.T) {
|
||||
}
|
||||
|
||||
// Evaluate rotation
|
||||
angle, _, scores := evaluateTableOrientation(context.Background(), cropped, dd)
|
||||
angle, _, scores := tbl.EvaluateTableOrientation(context.Background(), cropped, dd)
|
||||
t.Logf(" Page %d Table %d: %dx%d, bestAngle=%d°, scores: 0=%.3f 90=%.3f 180=%.3f 270=%.3f",
|
||||
pg, tableCount, cropped.Bounds().Dx(), cropped.Bounds().Dy(),
|
||||
angle,
|
||||
@@ -127,7 +130,7 @@ func TestTableRotation_Stability(t *testing.T) {
|
||||
if baseURL == "" {
|
||||
baseURL = "http://localhost:9390"
|
||||
}
|
||||
dd, err := NewDeepDocClient(baseURL)
|
||||
dd, err := inf.NewInferenceClient(baseURL)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -178,7 +181,7 @@ func TestTableRotation_Stability(t *testing.T) {
|
||||
if cropped == nil {
|
||||
continue
|
||||
}
|
||||
angle, _, _ := evaluateTableOrientation(context.Background(), cropped, dd)
|
||||
angle, _, _ := tbl.EvaluateTableOrientation(context.Background(), cropped, dd)
|
||||
if angle != 0 {
|
||||
rotated++
|
||||
t.Logf(" %s: rotated table detected (angle=%d°)", e.Name(), angle)
|
||||
|
||||
@@ -5,20 +5,22 @@ import (
|
||||
"image"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// TestTableSection_TextFromTSR verifies that table Sections carry
|
||||
// TSR-structured text (from TableItem.Rows) rather than raw char text.
|
||||
// TSR-structured text (from pdf.TableItem.Rows) rather than raw char text.
|
||||
// Python _parse_loaded_window_into_bboxes runs _extract_table_figure
|
||||
// which pops table boxes and replaces them with consolidated table
|
||||
// entries. Go backfills Section.Text from TableItem.Rows after
|
||||
// entries. Go backfills pdf.Section.Text from pdf.TableItem.Rows after
|
||||
// linkTableSections.
|
||||
func TestTableSection_TextFromTSR(t *testing.T) {
|
||||
eng := &mockEngine{
|
||||
pageCount: 1,
|
||||
renderW: 900, // 300pt at 3x = 900px (216 DPI)
|
||||
renderH: 600,
|
||||
chars: map[int][]TextChar{0: {
|
||||
chars: map[int][]pdf.TextChar{0: {
|
||||
// PDF space (72 DPI): well inside DLA region
|
||||
{X0: 50, X1: 70, Top: 40, Bottom: 55, Text: "姓"},
|
||||
{X0: 80, X1: 100, Top: 40, Bottom: 55, Text: "名"},
|
||||
@@ -28,19 +30,19 @@ func TestTableSection_TextFromTSR(t *testing.T) {
|
||||
Healthy: true,
|
||||
// DLA table region in pixel space (216 DPI).
|
||||
// PDF space: x0=100/3≈33, y0=80/3≈27, x1=500/3≈167, y1=300/3≈100.
|
||||
DLARegions: []DLARegion{
|
||||
DLARegions: []pdf.DLARegion{
|
||||
{X0: 100, Y0: 80, X1: 500, Y1: 300, Label: "table", Confidence: 0.9},
|
||||
},
|
||||
// TSR returns structured 2x2 cells with text.
|
||||
// Pixel space (relative to cropped region).
|
||||
TSRCells: []TSRCell{
|
||||
TSRCells: []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 200, Y1: 100, Text: "姓名", Label: "table column header"},
|
||||
{X0: 200, Y0: 0, X1: 460, Y1: 100, Text: "年龄", Label: "table column header"},
|
||||
{X0: 0, Y0: 100, X1: 200, Y1: 220, Text: "张三", Label: "table row"},
|
||||
{X0: 200, Y0: 100, X1: 460, Y1: 220, Text: "25", Label: "table row"},
|
||||
},
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
@@ -49,15 +51,15 @@ func TestTableSection_TextFromTSR(t *testing.T) {
|
||||
|
||||
// ── Assert 1: Tables exist (Cells are filled by constructTable later) ──
|
||||
if len(result.Tables) == 0 {
|
||||
t.Fatal("expected at least 1 TableItem")
|
||||
t.Fatal("expected at least 1 pdf.TableItem")
|
||||
}
|
||||
tbl := result.Tables[0]
|
||||
if len(tbl.Cells) == 0 {
|
||||
t.Fatal("expected TSR cells in TableItem")
|
||||
t.Fatal("expected TSR cells in pdf.TableItem")
|
||||
}
|
||||
|
||||
// ── Assert 2: A table section exists with HTML output ──
|
||||
var tableSections []Section
|
||||
var tableSections []pdf.Section
|
||||
for _, s := range result.Sections {
|
||||
if s.LayoutType == "table" {
|
||||
tableSections = append(tableSections, s)
|
||||
@@ -68,15 +70,14 @@ func TestTableSection_TextFromTSR(t *testing.T) {
|
||||
}
|
||||
ts := tableSections[0]
|
||||
|
||||
// ── Assert 3: Section.Text is HTML table from constructTable ──
|
||||
// ── Assert 3: pdf.Section.Text is HTML table from constructTable ──
|
||||
if !strings.HasPrefix(ts.Text, "<table>") {
|
||||
t.Errorf("table Section.Text = %q, want HTML <table>", ts.Text)
|
||||
}
|
||||
// TSR cells have pre-filled text ("姓名", "年龄", "张三", "25") —
|
||||
// fillCellTextFromBoxes preserves it since cells already have text.
|
||||
if !strings.Contains(ts.Text, "姓名") || !strings.Contains(ts.Text, "年龄") {
|
||||
t.Errorf("table HTML should contain cell text, got %q", ts.Text)
|
||||
t.Errorf("table pdf.Section.Text = %q, want HTML <table>", ts.Text)
|
||||
}
|
||||
// OSS pipeline: TSR cell text is not preserved in the grid (OSS
|
||||
// GroupCells creates new cells from row×column cross product).
|
||||
// Cell text comes from fillCellTextFromBoxes matching PDF chars,
|
||||
// not from pre-filled TSR cell text (EE feature).
|
||||
}
|
||||
|
||||
// TestEnrichWithDeepDoc_ImageOnlyPage verifies that enrichWithDeepDoc
|
||||
@@ -85,21 +86,21 @@ func TestTableSection_TextFromTSR(t *testing.T) {
|
||||
func TestEnrichWithDeepDoc_ImageOnlyPage(t *testing.T) {
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []DLARegion{
|
||||
DLARegions: []pdf.DLARegion{
|
||||
{X0: 54, Y0: 100, X1: 846, Y1: 500, Label: "table", Confidence: 0.95},
|
||||
},
|
||||
TSRCells: []TSRCell{
|
||||
TSRCells: []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 200, Y1: 100, Text: "A", Label: "table row"},
|
||||
},
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
|
||||
// 0 text boxes, but page 0 has a rendered image.
|
||||
boxes := []TextBox{}
|
||||
boxes := []pdf.TextBox{}
|
||||
dummyImg := image.NewRGBA(image.Rect(0, 0, 900, 600))
|
||||
pageImages := map[int]image.Image{0: dummyImg}
|
||||
|
||||
tables := p.enrichWithDeepDoc(context.Background(), nil, boxes, pageImages)
|
||||
tables := p.enrichWithDeepDoc(context.Background(), nil, nil, boxes, pageImages)
|
||||
if len(tables) == 0 {
|
||||
t.Fatal("enrichWithDeepDoc: expected at least 1 table from DLA on page with image but no boxes, got 0")
|
||||
}
|
||||
@@ -108,55 +109,14 @@ func TestEnrichWithDeepDoc_ImageOnlyPage(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestMergeCaptions_Unit verifies mergeCaptions directly without full pipeline.
|
||||
func TestMergeCaptions_Unit(t *testing.T) {
|
||||
sections := []Section{
|
||||
{Text: "F", LayoutType: "figure", Positions: []Position{{PageNumbers: []int{0, 0}, Left: 40, Right: 60, Top: 30, Bottom: 45}}},
|
||||
{Text: "C", LayoutType: "figure caption", Positions: []Position{{PageNumbers: []int{0, 0}, Left: 40, Right: 60, Top: 80, Bottom: 95}}},
|
||||
}
|
||||
figures := CollectFigures(sections)
|
||||
|
||||
result := mergeCaptions(sections, figures)
|
||||
|
||||
// Caption removed.
|
||||
if len(result) != 1 {
|
||||
t.Fatalf("expected 1 section after merge, got %d", len(result))
|
||||
}
|
||||
// Figure text includes caption.
|
||||
if !strings.Contains(result[0].Text, "C") {
|
||||
t.Errorf("expected figure Text to contain caption 'C', got %q", result[0].Text)
|
||||
}
|
||||
if result[0].LayoutType != "figure" {
|
||||
t.Errorf("expected figure LayoutType, got %q", result[0].LayoutType)
|
||||
}
|
||||
}
|
||||
|
||||
// TestMergeCaptions_TableCaption verifies table caption merging directly.
|
||||
func TestMergeCaptions_TableCaption(t *testing.T) {
|
||||
sections := []Section{
|
||||
{Text: "T", LayoutType: "table", Positions: []Position{{PageNumbers: []int{0, 0}, Left: 40, Right: 60, Top: 30, Bottom: 45}}},
|
||||
{Text: "C", LayoutType: "table caption", Positions: []Position{{PageNumbers: []int{0, 0}, Left: 40, Right: 60, Top: 80, Bottom: 95}}},
|
||||
}
|
||||
figures := CollectFigures(sections)
|
||||
|
||||
result := mergeCaptions(sections, figures)
|
||||
|
||||
if len(result) != 1 {
|
||||
t.Fatalf("expected 1 section after merge, got %d", len(result))
|
||||
}
|
||||
if !strings.Contains(result[0].Text, "C") {
|
||||
t.Errorf("expected table Text to contain caption 'C', got %q", result[0].Text)
|
||||
}
|
||||
}
|
||||
|
||||
// TestFigureCaption_MergedIntoFigure verifies that "figure caption" text
|
||||
// is merged into the nearest "figure" Section and the caption Section is
|
||||
// is merged into the nearest "figure" pdf.Section and the caption pdf.Section is
|
||||
// removed. Matches Python _extract_table_figure caption matching.
|
||||
func TestFigureCaption_MergedIntoFigure(t *testing.T) {
|
||||
eng := &mockEngine{
|
||||
pageCount: 1,
|
||||
renderW: 1800, renderH: 2400,
|
||||
chars: map[int][]TextChar{0: {
|
||||
chars: map[int][]pdf.TextChar{0: {
|
||||
// Figure text — overlaps DLA figure region (pixel Y=80-300 → PDF 27-100).
|
||||
{X0: 40, X1: 60, Top: 30, Bottom: 45, Text: "F"},
|
||||
// Caption text — overlaps DLA figure caption region (pixel Y=310-340 → PDF 103-113).
|
||||
@@ -165,28 +125,28 @@ func TestFigureCaption_MergedIntoFigure(t *testing.T) {
|
||||
}
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []DLARegion{
|
||||
DLARegions: []pdf.DLARegion{
|
||||
{X0: 100, Y0: 80, X1: 500, Y1: 300, Label: "figure", Confidence: 0.9},
|
||||
// Caption is below the figure.
|
||||
{X0: 100, Y0: 310, X1: 500, Y1: 340, Label: "figure caption", Confidence: 0.9},
|
||||
},
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
|
||||
// Assert 1: figure caption Section removed.
|
||||
// Assert 1: figure caption pdf.Section removed.
|
||||
for _, s := range result.Sections {
|
||||
if s.LayoutType == "figure caption" {
|
||||
t.Errorf("figure caption Section should be removed after mergeCaptions, got %q", s.Text)
|
||||
t.Errorf("figure caption pdf.Section should be removed after mergeCaptions, got %q", s.Text)
|
||||
}
|
||||
}
|
||||
|
||||
// Assert 2: figure Section exists and has caption text appended.
|
||||
var fig *Section
|
||||
// Assert 2: figure pdf.Section exists and has caption text appended.
|
||||
var fig *pdf.Section
|
||||
for i := range result.Sections {
|
||||
if result.Sections[i].LayoutType == "figure" {
|
||||
fig = &result.Sections[i]
|
||||
@@ -194,25 +154,25 @@ func TestFigureCaption_MergedIntoFigure(t *testing.T) {
|
||||
}
|
||||
}
|
||||
if fig == nil {
|
||||
t.Fatal("expected a figure Section")
|
||||
t.Fatal("expected a figure pdf.Section")
|
||||
}
|
||||
if !strings.Contains(fig.Text, "C") {
|
||||
t.Errorf("figure Text should contain caption text 'C', got %q", fig.Text)
|
||||
}
|
||||
|
||||
// Assert 3: figure is in result.Figures.
|
||||
if len(result.Figures) == 0 {
|
||||
t.Error("expected at least 1 entry in result.Figures")
|
||||
// Assert 3: figure is in result.Figures().
|
||||
if len(result.Figures()) == 0 {
|
||||
t.Error("expected at least 1 entry in result.Figures()")
|
||||
}
|
||||
}
|
||||
|
||||
// TestTableCaption_MergedIntoTable verifies that "table caption" text
|
||||
// is merged into the nearest table Section and the caption is removed.
|
||||
// is merged into the nearest table pdf.Section and the caption is removed.
|
||||
func TestTableCaption_MergedIntoTable(t *testing.T) {
|
||||
eng := &mockEngine{
|
||||
pageCount: 1,
|
||||
renderW: 1800, renderH: 2400,
|
||||
chars: map[int][]TextChar{0: {
|
||||
chars: map[int][]pdf.TextChar{0: {
|
||||
// Table text — overlaps DLA table region (pixel Y=80-300 → PDF 27-100).
|
||||
{X0: 40, X1: 60, Top: 30, Bottom: 45, Text: "T"},
|
||||
// Caption text — overlaps DLA table caption region (pixel Y=310-340 → PDF 103-113).
|
||||
@@ -221,29 +181,29 @@ func TestTableCaption_MergedIntoTable(t *testing.T) {
|
||||
}
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []DLARegion{
|
||||
DLARegions: []pdf.DLARegion{
|
||||
{X0: 100, Y0: 80, X1: 500, Y1: 300, Label: "table", Confidence: 0.9},
|
||||
{X0: 100, Y0: 310, X1: 500, Y1: 340, Label: "table caption", Confidence: 0.9},
|
||||
},
|
||||
TSRCells: []TSRCell{
|
||||
TSRCells: []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 200, Y1: 100, Text: "A", Label: "table row"},
|
||||
{X0: 200, Y0: 0, X1: 460, Y1: 100, Text: "B", Label: "table row"},
|
||||
},
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
|
||||
// Assert: table caption Section removed, text merged into table Section.
|
||||
// Assert: table caption pdf.Section removed, text merged into table pdf.Section.
|
||||
for _, s := range result.Sections {
|
||||
if s.LayoutType == "table caption" {
|
||||
t.Errorf("table caption Section should be removed, got %q", s.Text)
|
||||
t.Errorf("table caption pdf.Section should be removed, got %q", s.Text)
|
||||
}
|
||||
}
|
||||
var tbl *Section
|
||||
var tbl *pdf.Section
|
||||
for i := range result.Sections {
|
||||
if result.Sections[i].LayoutType == "table" {
|
||||
tbl = &result.Sections[i]
|
||||
@@ -251,7 +211,7 @@ func TestTableCaption_MergedIntoTable(t *testing.T) {
|
||||
}
|
||||
}
|
||||
if tbl == nil {
|
||||
t.Fatal("expected a table Section")
|
||||
t.Fatal("expected a table pdf.Section")
|
||||
}
|
||||
if !strings.Contains(tbl.Text, "C") {
|
||||
t.Errorf("table Text should contain caption text 'C', got %q", tbl.Text)
|
||||
@@ -267,7 +227,7 @@ func TestTextSectionsInsideTableRegion_Suppressed(t *testing.T) {
|
||||
eng := &mockEngine{
|
||||
pageCount: 1,
|
||||
renderW: 1800, renderH: 2400,
|
||||
chars: map[int][]TextChar{0: {
|
||||
chars: map[int][]pdf.TextChar{0: {
|
||||
// Box A: inside DLA table region, labeled as "text" by DLA.
|
||||
{X0: 50, X1: 100, Top: 40, Bottom: 55, Text: "碎片文字"},
|
||||
// Box B: inside DLA table region, same situation.
|
||||
@@ -278,23 +238,23 @@ func TestTextSectionsInsideTableRegion_Suppressed(t *testing.T) {
|
||||
// Real DLA often splits large table regions this way.
|
||||
mock := &MockDocAnalyzer{
|
||||
Healthy: true,
|
||||
DLARegions: []DLARegion{
|
||||
DLARegions: []pdf.DLARegion{
|
||||
{X0: 100, Y0: 80, X1: 500, Y1: 300, Label: "table", Confidence: 0.9},
|
||||
{X0: 120, Y0: 100, X1: 180, Y1: 140, Label: "text", Confidence: 0.8},
|
||||
},
|
||||
TSRCells: []TSRCell{
|
||||
TSRCells: []pdf.TSRCell{
|
||||
{X0: 0, Y0: 0, X1: 200, Y1: 100, Text: "姓名", Label: "table row"},
|
||||
{X0: 200, Y0: 0, X1: 460, Y1: 100, Text: "年龄", Label: "table row"},
|
||||
},
|
||||
}
|
||||
p := NewParser(DefaultParserConfig(), mock)
|
||||
p := NewParser(pdf.DefaultParserConfig(), mock)
|
||||
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
}
|
||||
|
||||
// Assert 1: table Section exists with structured text.
|
||||
// Assert 1: table pdf.Section exists with structured text.
|
||||
var hasTable bool
|
||||
for _, s := range result.Sections {
|
||||
if s.LayoutType == "table" && s.Text != "" {
|
||||
@@ -303,7 +263,7 @@ func TestTextSectionsInsideTableRegion_Suppressed(t *testing.T) {
|
||||
}
|
||||
}
|
||||
if !hasTable {
|
||||
t.Fatal("expected a table Section with structured text")
|
||||
t.Fatal("expected a table pdf.Section with structured text")
|
||||
}
|
||||
|
||||
// Assert 2: NO "text" fragment sections remain — they were inside
|
||||
@@ -327,7 +287,7 @@ func TestTextSectionsInsideTableRegion_Suppressed(t *testing.T) {
|
||||
// TestEmptyDoc_NoCrash verifies Parse handles edge cases gracefully.
|
||||
func TestEmptyDoc_NoCrash(t *testing.T) {
|
||||
eng := &mockEngine{pageCount: 0}
|
||||
p := NewParser(DefaultParserConfig(), &MockDocAnalyzer{Healthy: true, Model: ModelSaas})
|
||||
p := NewParser(pdf.DefaultParserConfig(), &MockDocAnalyzer{Healthy: true})
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
@@ -340,7 +300,7 @@ func TestEmptyDoc_NoCrash(t *testing.T) {
|
||||
// TestNilChars_handled verifies zero-chars pages don't crash.
|
||||
func TestNilChars_Handled(t *testing.T) {
|
||||
eng := &mockEngine{pageCount: 1, renderW: 200, renderH: 200}
|
||||
p := NewParser(DefaultParserConfig(), &MockDocAnalyzer{Healthy: true, Model: ModelSaas})
|
||||
p := NewParser(pdf.DefaultParserConfig(), &MockDocAnalyzer{Healthy: true})
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
if err != nil {
|
||||
t.Fatalf("Parse: %v", err)
|
||||
@@ -349,68 +309,3 @@ func TestNilChars_Handled(t *testing.T) {
|
||||
t.Logf("nil chars + DeepDoc: sections=%d (may trigger OCR path)", len(result.Sections))
|
||||
}
|
||||
}
|
||||
|
||||
// TestMergeCaptions_EuclideanDistance verifies that caption matching uses
|
||||
// squared Euclidean distance (center-to-center), not Y-only distance.
|
||||
// Two captions at different X positions — the one closer by Euclidean
|
||||
// distance wins, even if its Y distance is slightly larger.
|
||||
func TestMergeCaptions_EuclideanDistance(t *testing.T) {
|
||||
sections := []Section{
|
||||
{Text: "F", LayoutType: "figure", Positions: []Position{
|
||||
{PageNumbers: []int{0, 0}, Left: 0, Right: 100, Top: 0, Bottom: 50},
|
||||
}},
|
||||
// Caption A: directly below figure (dx=0, dy=20) → Euclidean = 20²
|
||||
{Text: "close", LayoutType: "figure caption", Positions: []Position{
|
||||
{PageNumbers: []int{0, 0}, Left: 0, Right: 100, Top: 70, Bottom: 80},
|
||||
}},
|
||||
}
|
||||
figures := CollectFigures(sections)
|
||||
result := mergeCaptions(sections, figures)
|
||||
// Caption merged into figure — verified by figure Text containing caption.
|
||||
if len(result) != 1 {
|
||||
t.Fatalf("expected 1 section after merge, got %d", len(result))
|
||||
}
|
||||
if !strings.Contains(result[0].Text, "close") {
|
||||
t.Errorf("figure Text should contain caption 'close', got %q", result[0].Text)
|
||||
}
|
||||
}
|
||||
|
||||
// mockEngine is a minimal PDFEngine stub for unit tests.
|
||||
type mockEngine struct {
|
||||
chars map[int][]TextChar
|
||||
pageCount int
|
||||
renderW int
|
||||
renderH int
|
||||
}
|
||||
|
||||
func (m *mockEngine) ExtractChars(pg int) ([]TextChar, error) {
|
||||
return m.chars[pg], nil
|
||||
}
|
||||
func (m *mockEngine) RenderPage(pg int, dpi float64) ([]byte, error) {
|
||||
w, h := m.renderW, m.renderH
|
||||
if w <= 0 {
|
||||
w = 595
|
||||
}
|
||||
if h <= 0 {
|
||||
h = 842
|
||||
}
|
||||
return nil, nil
|
||||
}
|
||||
func (m *mockEngine) RenderPageImage(pg int, dpi float64) (image.Image, error) {
|
||||
w, h := m.renderW, m.renderH
|
||||
if w <= 0 {
|
||||
w = 100
|
||||
}
|
||||
if h <= 0 {
|
||||
h = 100
|
||||
}
|
||||
return image.NewRGBA(image.Rect(0, 0, w, h)), nil
|
||||
}
|
||||
func (m *mockEngine) PageCount() (int, error) {
|
||||
if m.pageCount <= 0 {
|
||||
return 1, nil
|
||||
}
|
||||
return m.pageCount, nil
|
||||
}
|
||||
func (m *mockEngine) RawData() []byte { return nil }
|
||||
func (m *mockEngine) Close() error { return nil }
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
50
internal/deepdoc/parser/pdf/test_helpers_cgo_test.go
Normal file
50
internal/deepdoc/parser/pdf/test_helpers_cgo_test.go
Normal file
@@ -0,0 +1,50 @@
|
||||
//go:build cgo
|
||||
|
||||
package parser
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
inf "ragflow/internal/deepdoc/parser/pdf/inference"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// ── Shared CGO test helpers ──────────────────────────────────────────────────
|
||||
// These helpers were previously duplicated across multiple test files with
|
||||
// different build tags (integration, manual). Consolidating them into one file
|
||||
// with the //go:build cgo tag makes them available to all cgo-tagged tests.
|
||||
|
||||
// mustConnectInferenceClient returns a InferenceClient pointed at the OSS service;
|
||||
// skips the test if the service reports a non-OSS model type.
|
||||
func mustConnectInferenceClient(t *testing.T) *inf.InferenceClient {
|
||||
t.Helper()
|
||||
url := os.Getenv("OSSDEEPDOC_URL")
|
||||
if url == "" {
|
||||
url = "http://localhost:9390"
|
||||
}
|
||||
client, err := inf.NewInferenceClient(url)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !client.Health() {
|
||||
t.Fatalf("OssDeepDoc not available at %s", url)
|
||||
}
|
||||
return client
|
||||
}
|
||||
|
||||
// mustOpenEngine opens a PDF from testdata/pdfs/ and returns a pdf.PDFEngine.
|
||||
func mustOpenEngine(t *testing.T, name string) pdf.PDFEngine {
|
||||
t.Helper()
|
||||
pdfPath := filepath.Join("testdata", "pdfs", name)
|
||||
data, err := os.ReadFile(pdfPath)
|
||||
if err != nil {
|
||||
t.Fatalf("read fixture %s: %v", name, err)
|
||||
}
|
||||
eng, err := NewEngine(data)
|
||||
if err != nil {
|
||||
t.Fatalf("open engine %s: %v", name, err)
|
||||
}
|
||||
return eng
|
||||
}
|
||||
66
internal/deepdoc/parser/pdf/test_helpers_test.go
Normal file
66
internal/deepdoc/parser/pdf/test_helpers_test.go
Normal file
@@ -0,0 +1,66 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"image"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// ── mockEngine: minimal pdf.PDFEngine stub for unit tests ─────────────
|
||||
|
||||
type mockEngine struct {
|
||||
chars map[int][]pdf.TextChar
|
||||
pageCount int
|
||||
renderW int
|
||||
renderH int
|
||||
}
|
||||
|
||||
func (m *mockEngine) ExtractChars(pg int) ([]pdf.TextChar, error) {
|
||||
return m.chars[pg], nil
|
||||
}
|
||||
func (m *mockEngine) RenderPage(pg int, dpi float64) ([]byte, error) {
|
||||
w, h := m.renderW, m.renderH
|
||||
if w <= 0 {
|
||||
w = 595
|
||||
}
|
||||
if h <= 0 {
|
||||
h = 842
|
||||
}
|
||||
return nil, nil
|
||||
}
|
||||
func (m *mockEngine) RenderPageImage(pg int, dpi float64) (image.Image, error) {
|
||||
w, h := m.renderW, m.renderH
|
||||
if w <= 0 {
|
||||
w = 100
|
||||
}
|
||||
if h <= 0 {
|
||||
h = 100
|
||||
}
|
||||
return image.NewRGBA(image.Rect(0, 0, w, h)), nil
|
||||
}
|
||||
func (m *mockEngine) PageCount() (int, error) {
|
||||
if m.pageCount <= 0 {
|
||||
return 1, nil
|
||||
}
|
||||
return m.pageCount, nil
|
||||
}
|
||||
func (m *mockEngine) RawData() []byte { return nil }
|
||||
func (m *mockEngine) Close() error { return nil }
|
||||
func (m *mockEngine) Outlines() ([]pdf.Outline, error) { return nil, nil }
|
||||
|
||||
// ── testPageImg: small test image for ocrMergeChars tests ─────────────
|
||||
// 90×120 px at 216 DPI → 30×40 pt in PDF space after /3.0 scaling.
|
||||
|
||||
func testPageImg() image.Image {
|
||||
return image.NewRGBA(image.Rect(0, 0, 90, 120))
|
||||
}
|
||||
|
||||
// ── cellTexts: extract text strings from TSRCells ─────────────────────
|
||||
|
||||
func cellTexts(cells []pdf.TSRCell) []string {
|
||||
t := make([]string, len(cells))
|
||||
for i, c := range cells {
|
||||
t[i] = c.Text
|
||||
}
|
||||
return t
|
||||
}
|
||||
@@ -6,6 +6,7 @@ import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
@@ -64,8 +65,8 @@ func TestDumpTextOutput(t *testing.T) {
|
||||
continue
|
||||
}
|
||||
|
||||
cfg := DefaultParserConfig()
|
||||
p := NewParser(cfg, &MockDocAnalyzer{Healthy: true, Model: ModelSaas})
|
||||
cfg := pdf.DefaultParserConfig()
|
||||
p := NewParser(cfg, &MockDocAnalyzer{Healthy: true})
|
||||
result, err := p.Parse(context.Background(), eng)
|
||||
eng.Close()
|
||||
if err != nil {
|
||||
|
||||
@@ -1,18 +1,20 @@
|
||||
package parser
|
||||
package tool
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"image"
|
||||
"os"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// PythonCharEngine implements PDFEngine by loading chars from a
|
||||
// PythonCharEngine implements pdf.PDFEngine by loading chars from a
|
||||
// charspy/{pdf}.json file exported by dump_py_results.py.
|
||||
// It is used for pipeline parity testing — same input chars as Python,
|
||||
// so any difference in pipeline output is a Go pipeline logic bug.
|
||||
type PythonCharEngine struct {
|
||||
chars map[int][]TextChar // pageNum → chars
|
||||
chars map[int][]pdf.TextChar // pageNum → chars
|
||||
pages int
|
||||
}
|
||||
|
||||
@@ -37,11 +39,11 @@ func LoadPythonChars(jsonPath string) (*PythonCharEngine, error) {
|
||||
return nil, fmt.Errorf("parse charspy json: %w", err)
|
||||
}
|
||||
|
||||
chars := make(map[int][]TextChar, len(wrapper.Pages))
|
||||
chars := make(map[int][]pdf.TextChar, len(wrapper.Pages))
|
||||
for pg, pageChars := range wrapper.Pages {
|
||||
result := make([]TextChar, len(pageChars))
|
||||
result := make([]pdf.TextChar, len(pageChars))
|
||||
for i, c := range pageChars {
|
||||
result[i] = TextChar{
|
||||
result[i] = pdf.TextChar{
|
||||
Text: c.Text,
|
||||
X0: c.X0,
|
||||
X1: c.X1,
|
||||
@@ -58,7 +60,7 @@ func LoadPythonChars(jsonPath string) (*PythonCharEngine, error) {
|
||||
}
|
||||
|
||||
// ExtractChars returns all characters for the given page (0-indexed).
|
||||
func (e *PythonCharEngine) ExtractChars(pageNum int) ([]TextChar, error) {
|
||||
func (e *PythonCharEngine) ExtractChars(pageNum int) ([]pdf.TextChar, error) {
|
||||
if pageNum < 0 || pageNum >= e.pages {
|
||||
return nil, fmt.Errorf("page %d out of range [0, %d)", pageNum, e.pages)
|
||||
}
|
||||
@@ -84,6 +86,8 @@ func (e *PythonCharEngine) PageCount() (int, error) {
|
||||
// for pipeline parity tests and does not hold PDF bytes.
|
||||
func (e *PythonCharEngine) RawData() []byte { return nil }
|
||||
|
||||
func (e *PythonCharEngine) Outlines() ([]pdf.Outline, error) { return nil, nil }
|
||||
|
||||
// Close is a no-op.
|
||||
func (e *PythonCharEngine) Close() error {
|
||||
return nil
|
||||
@@ -1,4 +1,4 @@
|
||||
package tools
|
||||
package tool
|
||||
|
||||
import (
|
||||
"encoding/csv"
|
||||
@@ -556,8 +556,10 @@ type tsrRawCell struct {
|
||||
TableIndex int `json:"table_index"`
|
||||
Page int `json:"page"`
|
||||
Label string `json:"label"`
|
||||
X0, Y0 float64 `json:"x0" y0:"y0"`
|
||||
X1, Y1 float64 `json:"x1" y1:"y1"`
|
||||
X0 float64 `json:"x0"`
|
||||
Y0 float64 `json:"y0"`
|
||||
X1 float64 `json:"x1"`
|
||||
Y1 float64 `json:"y1"`
|
||||
Text string `json:"text"`
|
||||
}
|
||||
|
||||
63
internal/deepdoc/parser/pdf/tool/compare_test.go
Normal file
63
internal/deepdoc/parser/pdf/tool/compare_test.go
Normal file
@@ -0,0 +1,63 @@
|
||||
//go:build manual
|
||||
|
||||
package tool
|
||||
|
||||
import (
|
||||
"log/slog"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// TestBatchCompareWithPython compares Go output against Python reference
|
||||
// across 4 dimensions (text, tables, DLA, TSR raw). It is read-only —
|
||||
// no generation, no CGO/DeepDoc dependency. Use BATCH_SKIP_OCR=1 to
|
||||
// compare the noocr variant; PY_OCR_SUFFIX to override the Python variant.
|
||||
func TestBatchCompareWithPython(t *testing.T) {
|
||||
level := slog.LevelInfo
|
||||
if os.Getenv("BATCH_LOG_LEVEL") == "debug" {
|
||||
level = slog.LevelDebug
|
||||
}
|
||||
if os.Getenv("BATCH_LOG_LEVEL") == "warn" {
|
||||
level = slog.LevelWarn
|
||||
}
|
||||
slog.SetDefault(slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: level})))
|
||||
|
||||
goVariant := "ocr"
|
||||
if os.Getenv("BATCH_SKIP_OCR") == "1" {
|
||||
goVariant = "noocr"
|
||||
}
|
||||
pyVariant := os.Getenv("PY_OCR_SUFFIX")
|
||||
if pyVariant == "" {
|
||||
pyVariant = goVariant
|
||||
}
|
||||
goTextDir := filepath.Join("testdata", "output", "go", goVariant, "text")
|
||||
pyTextDir := filepath.Join("testdata", "output", "py", pyVariant, "text")
|
||||
|
||||
// Read Go text files' #@meta (no aggregate JSON dependency).
|
||||
goResults, err := ReadGoTextMeta(goTextDir)
|
||||
if err != nil || len(goResults) == 0 {
|
||||
t.Fatalf("No Go text files in %s: %v", goTextDir, err)
|
||||
}
|
||||
|
||||
// Read Python text files' #@meta
|
||||
pyResults, err := ReadPythonTextMeta(pyTextDir)
|
||||
if err != nil || len(pyResults) == 0 {
|
||||
t.Fatalf("No Python text files in %s: %v", pyTextDir, err)
|
||||
}
|
||||
|
||||
t.Logf("Comparing %d Go × %d Python", len(goResults), len(pyResults))
|
||||
CompareWithPython(t, goResults, pyResults, goTextDir, pyTextDir)
|
||||
|
||||
// Compare tables.
|
||||
goTablesDir := filepath.Join("testdata", "output", "go", goVariant, "tables")
|
||||
pyTablesDir2 := filepath.Join("testdata", "output", "py", pyVariant, "tables")
|
||||
CompareTablesWithPython(t, goTablesDir, pyTablesDir2)
|
||||
// Compare DLA + TSR raw intermediates.
|
||||
goDLADir := filepath.Join("testdata", "output", "go", goVariant, "dla")
|
||||
pyDLADir := filepath.Join("testdata", "output", "py", pyVariant, "dla")
|
||||
CompareDLAWithPython(t, goDLADir, pyDLADir)
|
||||
goTSRRawDir := filepath.Join("testdata", "output", "go", goVariant, "tsr_raw")
|
||||
pyTSRRawDir := filepath.Join("testdata", "output", "py", pyVariant, "tsr_raw")
|
||||
CompareTSRRawWithPython(t, goTSRRawDir, pyTSRRawDir)
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
package tools
|
||||
package tool
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
@@ -1,4 +1,4 @@
|
||||
package tools
|
||||
package tool
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
@@ -1,4 +1,4 @@
|
||||
package tools
|
||||
package tool
|
||||
|
||||
import (
|
||||
"sort"
|
||||
90
internal/deepdoc/parser/pdf/tool/similarity_test.go
Normal file
90
internal/deepdoc/parser/pdf/tool/similarity_test.go
Normal file
@@ -0,0 +1,90 @@
|
||||
package tool
|
||||
|
||||
import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestStripMeta(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
input string
|
||||
want string
|
||||
}{
|
||||
{"no meta", "hello world", "hello world"},
|
||||
{"with meta", "hello\n#@meta\n{\"key\":\"val\"}", "hello"},
|
||||
{"multiple meta lines", "line1\nline2\n#@meta\n{}", "line1\nline2"},
|
||||
{"empty", "", ""},
|
||||
{"only meta no newline", "#@meta\n{}", "#@meta\n{}"},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
if got := StripMeta(tt.input); got != tt.want {
|
||||
t.Errorf("StripMeta = %q, want %q", got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestCharSimilarity(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
a, b string
|
||||
want float64
|
||||
}{
|
||||
{"identical", "hello", "hello", 100.0},
|
||||
{"completely different", "abc", "xyz", 0.0},
|
||||
{"partial overlap", "abc", "bcd", 66.66}, // returns percentage
|
||||
{"empty both", "", "", 100.0},
|
||||
{"empty a", "", "abc", 0.0},
|
||||
{"empty b", "abc", "", 0.0},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := CharSimilarity(tt.a, tt.b)
|
||||
if got < tt.want-1 || got > tt.want+1 {
|
||||
t.Errorf("CharSimilarity(%q, %q) = %v, want ~%v", tt.a, tt.b, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestLcsSimilarity(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
a, b string
|
||||
want float64
|
||||
}{
|
||||
{"identical", "hello", "hello", 100.0},
|
||||
{"completely different", "abc", "xyz", 0.0},
|
||||
{"partial", "abc", "abcd", 75.0}, // LCS "abc" len=3, max len=4 = 75%
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := LcsSimilarity(tt.a, tt.b)
|
||||
if got < tt.want-1 || got > tt.want+1 {
|
||||
t.Errorf("LcsSimilarity(%q, %q) = %v, want ~%v", tt.a, tt.b, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestSectionAlignedScore(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
goText string
|
||||
pyText string
|
||||
want float64
|
||||
}{
|
||||
{"identical", "hello world", "hello world", 100.0},
|
||||
{"different", "hello", "world", 20.0},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := SectionAlignedScore(tt.goText, tt.pyText)
|
||||
if got < tt.want-1 || got > tt.want+1 {
|
||||
t.Errorf("SectionAlignedScore(%q, %q) = %v, want ~%v",
|
||||
tt.goText, tt.pyText, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
package tools
|
||||
package tool
|
||||
|
||||
// BatchResult stores per-PDF pipeline stage output.
|
||||
type BatchResult struct {
|
||||
@@ -41,9 +41,12 @@ type TableItem struct {
|
||||
|
||||
// TSRCell mirrors parser.TSRCell for serialization.
|
||||
type TSRCell struct {
|
||||
X0, Y0, X1, Y1 float64 `json:"x0,y0,x1,y1"`
|
||||
Text string `json:"text"`
|
||||
Label string `json:"label"`
|
||||
X0 float64 `json:"x0"`
|
||||
Y0 float64 `json:"y0"`
|
||||
X1 float64 `json:"x1"`
|
||||
Y1 float64 `json:"y1"`
|
||||
Text string `json:"text"`
|
||||
Label string `json:"label"`
|
||||
}
|
||||
|
||||
// Position stores a bounding box.
|
||||
320
internal/deepdoc/parser/pdf/type/types.go
Normal file
320
internal/deepdoc/parser/pdf/type/types.go
Normal file
@@ -0,0 +1,320 @@
|
||||
// Package pdftypes provides shared types, interfaces, and constants for the
|
||||
// PDF parser pipeline. It has zero dependencies on sibling packages so that
|
||||
// sub-packages (tables, geometry, etc.) can import it without circular imports.
|
||||
package pdftype
|
||||
|
||||
import (
|
||||
"context"
|
||||
"image"
|
||||
"unicode"
|
||||
)
|
||||
|
||||
// ── Pipeline types ────────────────────────────────────────────────────────
|
||||
|
||||
// PipelineMetrics records diagnostic counts at each pipeline stage.
|
||||
type PipelineMetrics struct {
|
||||
BoxesInitial int
|
||||
BoxesTextMerge int
|
||||
BoxesVertMerge int
|
||||
BoxesFinal int
|
||||
TablesCount int
|
||||
}
|
||||
|
||||
// ParseResult encapsulates all outputs from a single Parse() call.
|
||||
type ParseResult struct {
|
||||
Sections []Section
|
||||
Tables []TableItem
|
||||
PageImages map[int]image.Image
|
||||
Metrics PipelineMetrics
|
||||
Outlines []Outline // PDF outlines/bookmarks extracted from the document
|
||||
|
||||
DLADebug []DLAPageRegions
|
||||
TSRDebug []TSRRawCell
|
||||
}
|
||||
|
||||
// Figures returns all sections with LayoutType "figure".
|
||||
// Computed on demand from Sections — no stored field.
|
||||
func (r *ParseResult) Figures() []Section {
|
||||
return CollectFigures(r.Sections)
|
||||
}
|
||||
|
||||
// DLAPageRegions holds DLA layout regions for one page.
|
||||
type DLAPageRegions struct {
|
||||
Page int
|
||||
Regions []DLARegion
|
||||
}
|
||||
|
||||
// TSRRawCell holds a raw TSR cell before row/column grouping.
|
||||
type TSRRawCell struct {
|
||||
TableIndex int `json:"table_index"`
|
||||
Page int `json:"page"`
|
||||
Label string `json:"label"`
|
||||
X0 float64 `json:"x0"`
|
||||
Y0 float64 `json:"y0"`
|
||||
X1 float64 `json:"x1"`
|
||||
Y1 float64 `json:"y1"`
|
||||
Text string `json:"text"`
|
||||
}
|
||||
|
||||
// ── Character and text box types ──────────────────────────────────────────
|
||||
|
||||
// TextChar represents a single character extracted from a PDF page.
|
||||
type TextChar struct {
|
||||
X0, X1 float64
|
||||
Top, Bottom float64
|
||||
Text string
|
||||
FontName string
|
||||
FontSize float64
|
||||
PageNumber int
|
||||
LayoutType string
|
||||
LayoutNo string
|
||||
ColID int
|
||||
R int
|
||||
}
|
||||
|
||||
func (c TextChar) Bounds() (float64, float64, float64, float64) {
|
||||
return c.X0, c.Top, c.X1, c.Bottom
|
||||
}
|
||||
|
||||
// TextBox represents a rectangular region of text on a PDF page.
|
||||
type TextBox struct {
|
||||
X0, X1 float64
|
||||
Top, Bottom float64
|
||||
Text string
|
||||
PageNumber int
|
||||
LayoutType string
|
||||
LayoutNo string
|
||||
ColID int
|
||||
R int
|
||||
// Post-TSR table annotation fields (Python: R/H/C/SP tags)
|
||||
RTop, RBott float64
|
||||
HTop, HBott float64
|
||||
HLeft, HRight float64
|
||||
H int
|
||||
C int
|
||||
CLeft, CRight float64
|
||||
SP int
|
||||
}
|
||||
|
||||
func (b TextBox) Bounds() (float64, float64, float64, float64) {
|
||||
return b.X0, b.Top, b.X1, b.Bottom
|
||||
}
|
||||
|
||||
// ── Position and section types ────────────────────────────────────────────
|
||||
|
||||
// Position represents a parsed position tag from @@...## format.
|
||||
type Position struct {
|
||||
PageNumbers []int
|
||||
Left float64
|
||||
Right float64
|
||||
Top float64
|
||||
Bottom float64
|
||||
}
|
||||
|
||||
// Section represents a text segment with its spatial position on a PDF page.
|
||||
type Section struct {
|
||||
Text string
|
||||
PositionTag string
|
||||
LayoutType string
|
||||
DocTypeKwd string // "text"/"table"/"image" — assigned during post-processing
|
||||
Positions []Position
|
||||
TableItem *TableItem
|
||||
Image string // base64-encoded cropped page image
|
||||
}
|
||||
|
||||
// SectionsByPage returns a slice of sections on the given page.
|
||||
func SectionsByPage(sections []Section, page int) []Section {
|
||||
var out []Section
|
||||
for _, s := range sections {
|
||||
for _, p := range s.Positions {
|
||||
for _, pn := range p.PageNumbers {
|
||||
if pn == page {
|
||||
out = append(out, s)
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// CollectFigures returns all sections with LayoutType "figure".
|
||||
func CollectFigures(sections []Section) []Section {
|
||||
if sections == nil {
|
||||
return nil
|
||||
}
|
||||
figures := make([]Section, 0)
|
||||
for _, s := range sections {
|
||||
if s.LayoutType == LayoutTypeFigure {
|
||||
figures = append(figures, s)
|
||||
}
|
||||
}
|
||||
return figures
|
||||
}
|
||||
|
||||
// ── Table types ───────────────────────────────────────────────────────────
|
||||
|
||||
// TableItem represents a detected table or figure region.
|
||||
type TableItem struct {
|
||||
ImageB64 string
|
||||
Rows [][]string
|
||||
Cells []TSRCell
|
||||
Positions []Position
|
||||
Scale float64
|
||||
CropOffX float64
|
||||
CropOffY float64
|
||||
Caption string
|
||||
|
||||
RegionLeft, RegionRight, RegionTop, RegionBottom float64
|
||||
NoMerge bool
|
||||
Grid [][]TSRCell
|
||||
}
|
||||
|
||||
// TSRCell represents one table cell from TSR.
|
||||
type TSRCell struct {
|
||||
X0, Y0, X1, Y1 float64
|
||||
Text string
|
||||
Label string
|
||||
}
|
||||
|
||||
func (c TSRCell) Bounds() (float64, float64, float64, float64) {
|
||||
return c.X0, c.Y0, c.X1, c.Y1
|
||||
}
|
||||
|
||||
// ── DeepDoc vision types ─────────────────────────────────────────────────
|
||||
|
||||
// DLARegion represents one detected layout region.
|
||||
type DLARegion struct {
|
||||
X0, Y0, X1, Y1 float64
|
||||
Label string
|
||||
Confidence float64
|
||||
}
|
||||
|
||||
func (r DLARegion) Bounds() (float64, float64, float64, float64) {
|
||||
return r.X0, r.Y0, r.X1, r.Y1
|
||||
}
|
||||
|
||||
// OCRBox represents a detected text region from DeepDoc OCR detection.
|
||||
type OCRBox struct {
|
||||
X0, Y0, X1, Y1, X2, Y2, X3, Y3 float64
|
||||
}
|
||||
|
||||
// OCRText represents recognized text with confidence from DeepDoc OCR rec.
|
||||
type OCRText struct {
|
||||
Text string
|
||||
Confidence float64
|
||||
}
|
||||
|
||||
// ── Parser configuration ──────────────────────────────────────────────────
|
||||
|
||||
// ParserConfig holds parser configuration.
|
||||
type ParserConfig struct {
|
||||
Zoom float64
|
||||
FromPage int
|
||||
ToPage int
|
||||
TableContextSize int
|
||||
ImageContextSize int
|
||||
AutoRotateTables *bool
|
||||
SeparateTablesFigs bool
|
||||
SortByTop bool
|
||||
BatchSize int
|
||||
SkipOCR bool
|
||||
MaxOCRConcurrency int
|
||||
TableBuilder TableBuilder
|
||||
}
|
||||
|
||||
// DefaultParserConfig returns a ParserConfig with sensible defaults.
|
||||
func DefaultParserConfig() ParserConfig {
|
||||
return ParserConfig{
|
||||
Zoom: 3,
|
||||
FromPage: 0,
|
||||
ToPage: -1,
|
||||
BatchSize: 50,
|
||||
TableContextSize: 0,
|
||||
ImageContextSize: 0,
|
||||
SeparateTablesFigs: false,
|
||||
}
|
||||
}
|
||||
|
||||
// DlaDPI is the DPI used for rendering page images for DeepDoc DLA/OCR.
|
||||
const DlaDPI = 216
|
||||
|
||||
// DlaScale is the scale factor from PDF points (72 DPI) to DLA image space.
|
||||
const DlaScale = DlaDPI / 72.0
|
||||
|
||||
// ── Layout type constants ─────────────────────────────────────────────────
|
||||
|
||||
const (
|
||||
LayoutTypeText = "text"
|
||||
LayoutTypeTable = "table"
|
||||
LayoutTypeFigure = "figure"
|
||||
LayoutTypeEquation = "equation"
|
||||
LayoutTypeTitle = "title"
|
||||
LayoutTypeReference = "reference"
|
||||
LayoutTypeFooter = "footer"
|
||||
LayoutTypeHeader = "header"
|
||||
|
||||
DLALabelFigureCaption = "figure caption"
|
||||
DLALabelTableCaption = "table caption"
|
||||
)
|
||||
|
||||
// ── Interfaces ────────────────────────────────────────────────────────────
|
||||
|
||||
// DocAnalyzer abstracts DeepDoc vision operations.
|
||||
type DocAnalyzer interface {
|
||||
DLA(ctx context.Context, pageImage image.Image) ([]DLARegion, error)
|
||||
TSR(ctx context.Context, cropped image.Image) ([]TSRCell, error)
|
||||
OCRDetect(ctx context.Context, cropped image.Image) ([]OCRBox, error)
|
||||
OCRRecognize(ctx context.Context, cropped image.Image) ([]OCRText, error)
|
||||
OCRRecognizeBatch(ctx context.Context, cropped []image.Image) ([][]OCRText, []error)
|
||||
Health() bool
|
||||
}
|
||||
|
||||
// ── Outline ────────────────────────────────────────────────────────────
|
||||
|
||||
// Outline represents one entry in a PDF's document outline (table of contents).
|
||||
// Python: extract_pdf_outlines() in deepdoc/parser/utils.py
|
||||
type Outline struct {
|
||||
Title string
|
||||
Level int
|
||||
PageNumber int // 1-indexed, matching Python
|
||||
}
|
||||
|
||||
// PDFEngine abstracts page extraction capabilities.
|
||||
type PDFEngine interface {
|
||||
ExtractChars(pageNum int) ([]TextChar, error)
|
||||
RenderPage(pageNum int, dpi float64) ([]byte, error)
|
||||
RenderPageImage(pageNum int, dpi float64) (image.Image, error)
|
||||
RawData() []byte
|
||||
PageCount() (int, error)
|
||||
Outlines() ([]Outline, error)
|
||||
Close() error
|
||||
}
|
||||
|
||||
// Tokenizer provides text tokenization matching rag_tokenizer.
|
||||
type Tokenizer interface {
|
||||
Tag(token string) string
|
||||
}
|
||||
|
||||
// SampleFunc samples up to n characters from a page's chars.
|
||||
type SampleFunc func(chars []TextChar, n int) string
|
||||
|
||||
// TableBuilder encapsulates TSR model-specific cell detection and grouping.
|
||||
type TableBuilder interface {
|
||||
Name() string
|
||||
DetectCells(ctx context.Context, cropped image.Image) ([]TSRCell, error)
|
||||
GroupCells(cells []TSRCell) [][]TSRCell
|
||||
}
|
||||
|
||||
// Rectangular is any 2D axis-aligned rectangle that can report its bounds.
|
||||
type Rectangular interface {
|
||||
Bounds() (x0, y0, x1, y1 float64)
|
||||
}
|
||||
|
||||
// IsCJK reports whether r is a CJK character.
|
||||
func IsCJK(r rune) bool {
|
||||
return unicode.Is(unicode.Han, r) ||
|
||||
unicode.Is(unicode.Hiragana, r) ||
|
||||
unicode.Is(unicode.Katakana, r) ||
|
||||
unicode.Is(unicode.Hangul, r)
|
||||
}
|
||||
86
internal/deepdoc/parser/pdf/type/types_test.go
Normal file
86
internal/deepdoc/parser/pdf/type/types_test.go
Normal file
@@ -0,0 +1,86 @@
|
||||
package pdftype
|
||||
|
||||
import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestIsCJK(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
r rune
|
||||
want bool
|
||||
}{
|
||||
{"chinese", '中', true},
|
||||
{"chinese2", '国', true},
|
||||
{"hiragana", 'あ', true},
|
||||
{"katakana", 'ア', true},
|
||||
{"hangul", '한', true},
|
||||
{"latin", 'A', false},
|
||||
{"digit", '1', false},
|
||||
{"space", ' ', false},
|
||||
{"punctuation", '.', false},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
if got := IsCJK(tt.r); got != tt.want {
|
||||
t.Errorf("IsCJK(%q) = %v, want %v", tt.r, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectFigures(t *testing.T) {
|
||||
t.Run("mixed layout types", func(t *testing.T) {
|
||||
sections := []Section{
|
||||
{LayoutType: "figure", Text: "fig1", Image: "img1"},
|
||||
{LayoutType: "text", Text: "text1"},
|
||||
{LayoutType: "table", Text: "tbl1"},
|
||||
{LayoutType: "figure", Text: "fig2", Image: "img2"},
|
||||
{LayoutType: "title", Text: "title1"},
|
||||
}
|
||||
figures := CollectFigures(sections)
|
||||
if len(figures) != 2 {
|
||||
t.Fatalf("expected 2 figures, got %d", len(figures))
|
||||
}
|
||||
if figures[0].Text != "fig1" || figures[0].Image != "img1" {
|
||||
t.Errorf("first figure: got (%s, %s)", figures[0].Text, figures[0].Image)
|
||||
}
|
||||
if figures[1].Text != "fig2" || figures[1].Image != "img2" {
|
||||
t.Errorf("second figure: got (%s, %s)", figures[1].Text, figures[1].Image)
|
||||
}
|
||||
})
|
||||
t.Run("no figures", func(t *testing.T) {
|
||||
figures := CollectFigures([]Section{
|
||||
{LayoutType: "text"}, {LayoutType: "table"}, {LayoutType: "title"},
|
||||
})
|
||||
if len(figures) != 0 {
|
||||
t.Fatalf("expected 0, got %d", len(figures))
|
||||
}
|
||||
})
|
||||
t.Run("nil input", func(t *testing.T) {
|
||||
if figures := CollectFigures(nil); figures != nil {
|
||||
t.Fatalf("expected nil, got %d elements", len(figures))
|
||||
}
|
||||
})
|
||||
t.Run("empty input", func(t *testing.T) {
|
||||
figures := CollectFigures([]Section{})
|
||||
if figures == nil || len(figures) != 0 {
|
||||
t.Fatal("expected empty slice for empty input")
|
||||
}
|
||||
})
|
||||
t.Run("case sensitive", func(t *testing.T) {
|
||||
figures := CollectFigures([]Section{
|
||||
{LayoutType: "Figure"}, {LayoutType: "FIGURE"}, {LayoutType: "figure", Text: "fig3"},
|
||||
})
|
||||
if len(figures) != 1 || figures[0].Text != "fig3" {
|
||||
t.Fatalf("only lowercase 'figure' should match, got %d", len(figures))
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// textBox implements Rectangular for testing.
|
||||
type textBox struct{ x0, y0, x1, y1 float64 }
|
||||
|
||||
func (b textBox) Bounds() (float64, float64, float64, float64) {
|
||||
return b.x0, b.y0, b.x1, b.y1
|
||||
}
|
||||
@@ -1,320 +0,0 @@
|
||||
// Package pdfparser provides Go equivalents of RAGFlow's deepdoc/parser/pdf_parser.py
|
||||
// layout analysis and text extraction logic.
|
||||
//
|
||||
// Each exported function documents its corresponding Python original with
|
||||
// file:line references to pdf_parser.py.
|
||||
package parser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"image"
|
||||
)
|
||||
|
||||
// PipelineMetrics records diagnostic counts at each pipeline stage.
|
||||
// Used for Go-vs-Python parity comparison and logging.
|
||||
type PipelineMetrics struct {
|
||||
BoxesInitial int
|
||||
BoxesTextMerge int
|
||||
BoxesVertMerge int
|
||||
BoxesFinal int
|
||||
TablesCount int
|
||||
}
|
||||
|
||||
// ParseResult encapsulates all outputs from a single Parse() call.
|
||||
// Parser itself is stateless and safe to reuse across documents.
|
||||
type ParseResult struct {
|
||||
Sections []Section
|
||||
Tables []TableItem
|
||||
PageImages map[int]image.Image
|
||||
Figures []Section
|
||||
Metrics PipelineMetrics
|
||||
|
||||
// Debug intermediates for DLA/TSR comparison with Python.
|
||||
// Populated only during fresh Parse, not from cached results.
|
||||
DLADebug []DLAPageRegions
|
||||
TSRDebug []TSRRawCell
|
||||
}
|
||||
|
||||
// DLAPageRegions holds DLA layout regions for one page.
|
||||
type DLAPageRegions struct {
|
||||
Page int
|
||||
Regions []DLARegion
|
||||
}
|
||||
|
||||
// TSRRawCell holds a raw TSR cell before row/column grouping.
|
||||
type TSRRawCell struct {
|
||||
TableIndex int `json:"table_index"`
|
||||
Page int `json:"page"`
|
||||
Label string `json:"label"`
|
||||
X0 float64 `json:"x0"`
|
||||
Y0 float64 `json:"y0"`
|
||||
X1 float64 `json:"x1"`
|
||||
Y1 float64 `json:"y1"`
|
||||
Text string `json:"text"`
|
||||
}
|
||||
|
||||
// TextChar represents a single character extracted from a PDF page.
|
||||
// Corresponds to pdfplumber page.chars dict elements in pdf_parser.py.
|
||||
//
|
||||
// Python equivalent:
|
||||
//
|
||||
// c = {"x0": 100.5, "x1": 108.2, "top": 200.0, "bottom": 212.0,
|
||||
// "text": "A", "fontname": "ABCDE+SimSun", "page_number": 3}
|
||||
//
|
||||
// Example:
|
||||
//
|
||||
// c := TextChar{X0: 100.5, X1: 108.2, Top: 200.0, Bottom: 212.0,
|
||||
// Text: "A", FontName: "ABCDE+SimSun", PageNumber: 3}
|
||||
type TextChar struct {
|
||||
X0, X1 float64 // horizontal bounds in PDF points
|
||||
Top, Bottom float64 // vertical bounds in PDF points
|
||||
Text string // single character (or small text run)
|
||||
FontName string // e.g. "ABCDE+SimSun"
|
||||
FontSize float64
|
||||
PageNumber int
|
||||
LayoutType string // "text", "table", "figure", "equation"
|
||||
LayoutNo string // layout identifier
|
||||
ColID int // column ID assigned by _assign_column
|
||||
R int // rotation/orientation marker
|
||||
}
|
||||
|
||||
func (c TextChar) Bounds() (float64, float64, float64, float64) {
|
||||
return c.X0, c.Top, c.X1, c.Bottom
|
||||
}
|
||||
|
||||
// TextBox represents a rectangular region of text on a PDF page,
|
||||
// typically a line or paragraph fragment. Created by layout analysis
|
||||
// (e.g. _assign_column, _text_merge).
|
||||
//
|
||||
// Python equivalent:
|
||||
//
|
||||
// b = {"x0": 50.0, "x1": 550.0, "top": 100.0, "bottom": 112.0,
|
||||
// "text": "第三章 财务分析", "page_number": 3, "layout_type": "text"}
|
||||
type TextBox struct {
|
||||
X0, X1 float64
|
||||
Top, Bottom float64
|
||||
Text string
|
||||
PageNumber int
|
||||
LayoutType string // "text", "table", "figure", "equation"
|
||||
LayoutNo string
|
||||
ColID int
|
||||
R int
|
||||
// Post-TSR table annotation fields (Python: R/H/C/SP tags)
|
||||
RTop, RBott float64 // row top/bottom
|
||||
HTop, HBott float64 // header top/bottom
|
||||
HLeft, HRight float64 // header left/right
|
||||
H int // header index
|
||||
C int // column index
|
||||
CLeft, CRight float64 // column left/right
|
||||
SP int // spanning cell index
|
||||
}
|
||||
|
||||
func (b TextBox) Bounds() (float64, float64, float64, float64) {
|
||||
return b.X0, b.Top, b.X1, b.Bottom
|
||||
}
|
||||
|
||||
// Position represents a parsed position tag from @@...## format.
|
||||
//
|
||||
// Python: pdf_parser.py:1872 extract_positions()
|
||||
//
|
||||
// Format: @@{page_range}\t{left}\t{right}\t{top}\t{bottom}##
|
||||
// Example: "@@0-1\t50.0\t300.0\t200.0\t400.0##"
|
||||
type Position struct {
|
||||
PageNumbers []int // e.g. [0, 1] for cross-page content
|
||||
Left float64
|
||||
Right float64
|
||||
Top float64
|
||||
Bottom float64
|
||||
}
|
||||
|
||||
// Section represents a text segment with its spatial position on a PDF page.
|
||||
// This is the primary output of layout analysis, consumed by NLP merge/split.
|
||||
//
|
||||
// Python equivalent: sections elements in naive.py::chunk()
|
||||
//
|
||||
// [(text_with_tags, position_tag_string), ...]
|
||||
type Section struct {
|
||||
Text string // text content
|
||||
PositionTag string // "@@page-left-right-top-bottom##" format
|
||||
LayoutType string // "text", "table", "title", "figure", ...
|
||||
Positions []Position // parsed from PositionTag
|
||||
TableItem *TableItem // non-nil when this section is a table
|
||||
Image string // base64-encoded PNG of the cropped region (Python: b["image"])
|
||||
}
|
||||
|
||||
// CollectFigures returns all sections with LayoutType "figure".
|
||||
// Returns nil if the input is nil, empty slice if no figures found.
|
||||
func CollectFigures(sections []Section) []Section {
|
||||
if sections == nil {
|
||||
return nil
|
||||
}
|
||||
figures := make([]Section, 0)
|
||||
for _, s := range sections {
|
||||
if s.LayoutType == LayoutTypeFigure {
|
||||
figures = append(figures, s)
|
||||
}
|
||||
}
|
||||
return figures
|
||||
}
|
||||
|
||||
// TableItem represents a detected table or figure region.
|
||||
//
|
||||
// Python equivalent: tables elements in naive.py::chunk()
|
||||
//
|
||||
// [((img, rows), positions), ...]
|
||||
type TableItem struct {
|
||||
ImageB64 string // base64-encoded PNG of the table/figure region
|
||||
Rows [][]string // DEPRECATED: replaced by Cells; kept for batch output compat
|
||||
Cells []TSRCell // raw TSR cells in crop pixel space
|
||||
Positions []Position // spatial positions (PDF points, pre-merge)
|
||||
Scale float64 // zoom factor for coordinate conversion
|
||||
CropOffX float64 // crop origin X in pixel space
|
||||
CropOffY float64 // crop origin Y in pixel space
|
||||
Caption string // caption text merged from adjacent caption box
|
||||
|
||||
// DLA table region boundaries in PDF point space (72 DPI).
|
||||
// Matches Python's cropout using DLA layout region boundaries
|
||||
// instead of text box anchor coordinates.
|
||||
RegionLeft, RegionRight, RegionTop, RegionBottom float64
|
||||
|
||||
// NoMerge prevents cross-page merging for this table. Python's
|
||||
// _extract_table_figure adds table keys to nomerge_lout_no when
|
||||
// the next box is a caption/title/reference, indicating the table
|
||||
// group ended and should not merge with its continuation.
|
||||
NoMerge bool
|
||||
|
||||
// Grid is the row-column grid produced by TableBuilder.GroupCells.
|
||||
// Consumed by constructTable Path 1 and annotateTableBoxes.
|
||||
// Nil for tables without TSR cells (fallback paths use boxes instead).
|
||||
Grid [][]TSRCell
|
||||
}
|
||||
|
||||
// ParserConfig holds parser configuration.
|
||||
//
|
||||
// Python equivalent: kwargs merged with parser_config in task_executor.py
|
||||
type ParserConfig struct {
|
||||
Zoom float64 // zoom factor for page rendering, default 3
|
||||
FromPage int // 0-based start page
|
||||
ToPage int // 0-based end page (-1 = all)
|
||||
TableContextSize int // tokens of surrounding context for tables
|
||||
ImageContextSize int // tokens of surrounding context for images
|
||||
AutoRotateTables *bool // enable auto table rotation detection
|
||||
SeparateTablesFigs bool // separate tables and figures
|
||||
SortByTop bool // true = Top-based sort (parity tests); false = Bottom (production)
|
||||
ChunkSize int // pages per chunk (0 = default 50, matching Python batch_size)
|
||||
SkipOCR bool // true = DLA+TSR only, no image OCR (matching Python SKIP_OCR=1)
|
||||
MaxOCRConcurrency int // max concurrent OCR pages (0 = sequential); matches Python PARALLEL_DEVICES
|
||||
TableBuilder TableBuilder // TSR model adapter; injected by caller via NewTableBuilderFor
|
||||
}
|
||||
|
||||
// DefaultParserConfig returns a ParserConfig with sensible defaults.
|
||||
func DefaultParserConfig() ParserConfig {
|
||||
return ParserConfig{
|
||||
Zoom: 3,
|
||||
FromPage: 0,
|
||||
ToPage: -1,
|
||||
ChunkSize: 50,
|
||||
TableContextSize: 0,
|
||||
ImageContextSize: 0,
|
||||
SeparateTablesFigs: false,
|
||||
}
|
||||
}
|
||||
|
||||
// DetectGarbled returns true if a page's text is likely garbled due to
|
||||
// font encoding issues, indicating OCR is needed.
|
||||
//
|
||||
// This is a convenience wrapper around IsGarbledByFontEncoding.
|
||||
//
|
||||
// Python: pdf_parser.py:264 _is_garbled_by_font_encoding()
|
||||
func DetectGarbled(chars []TextChar) bool {
|
||||
return IsGarbledByFontEncoding(chars, 20)
|
||||
}
|
||||
|
||||
// HasColor checks if a character has visible color (not invisible white-on-white).
|
||||
//
|
||||
// Python: pdf_parser.py:190 _has_color()
|
||||
//
|
||||
// All extracted chars are assumed visible since the PDF engine handles
|
||||
// rendering internally.
|
||||
func HasColor(c TextChar) bool {
|
||||
return true
|
||||
}
|
||||
|
||||
// ── DeepDoc interfaces (shared between cgo and non-cgo builds) ──────────
|
||||
|
||||
// ModelType identifies the DeepDoc TSR model flavour.
|
||||
type ModelType string
|
||||
|
||||
const (
|
||||
ModelSaas ModelType = "saas" // cpu DeepDoc — cell-level TSR output
|
||||
ModelOSS ModelType = "oss" // oss DeepDoc — column/row line TSR output
|
||||
)
|
||||
|
||||
// Layout type constants — used for LayoutType field comparisons across
|
||||
// the pipeline. Values match DLA label taxonomy.
|
||||
const (
|
||||
LayoutTypeText = "text"
|
||||
LayoutTypeTable = "table"
|
||||
LayoutTypeFigure = "figure"
|
||||
LayoutTypeEquation = "equation"
|
||||
LayoutTypeTitle = "title"
|
||||
LayoutTypeReference = "reference"
|
||||
LayoutTypeFooter = "footer"
|
||||
LayoutTypeHeader = "header"
|
||||
|
||||
// Compound DLA labels (used in priority-ordered annotation matching).
|
||||
DLALabelFigureCaption = "figure caption"
|
||||
DLALabelTableCaption = "table caption"
|
||||
)
|
||||
|
||||
// DocAnalyzer abstracts DeepDoc vision operations so the Parser can
|
||||
// work with either a live service or a test mock.
|
||||
// I/O methods accept a context for cancellation and deadline propagation.
|
||||
type DocAnalyzer interface {
|
||||
DLA(ctx context.Context, pageImage image.Image) ([]DLARegion, error)
|
||||
TSR(ctx context.Context, cropped image.Image) ([]TSRCell, error)
|
||||
OCRDetect(ctx context.Context, cropped image.Image) ([]OCRBox, error)
|
||||
OCRRecognize(ctx context.Context, cropped image.Image) ([]OCRText, error)
|
||||
OCRRecognizeBatch(ctx context.Context, cropped []image.Image) ([][]OCRText, []error)
|
||||
Health() bool
|
||||
ModelType() ModelType
|
||||
}
|
||||
|
||||
// OCRBox represents a detected text region from DeepDoc OCR detection.
|
||||
// DeepDoc /predict/ocr?operator=det returns:
|
||||
//
|
||||
// {"output": [[[[[x0,y0],[x1,y1],[x2,y2],[x3,y3]], ...]]]}
|
||||
type OCRBox struct {
|
||||
X0, Y0, X1, Y1, X2, Y2, X3, Y3 float64
|
||||
}
|
||||
|
||||
// OCRText represents recognized text with confidence from DeepDoc OCR rec.
|
||||
// DeepDoc /predict/ocr?operator=rec returns:
|
||||
//
|
||||
// {"output": [[[["text", confidence], ...]]]}
|
||||
type OCRText struct {
|
||||
Text string
|
||||
Confidence float64
|
||||
}
|
||||
|
||||
// DLARegion represents one detected layout region.
|
||||
type DLARegion struct {
|
||||
X0, Y0, X1, Y1 float64
|
||||
Label string
|
||||
Confidence float64
|
||||
}
|
||||
|
||||
func (r DLARegion) Bounds() (float64, float64, float64, float64) {
|
||||
return r.X0, r.Y0, r.X1, r.Y1
|
||||
}
|
||||
|
||||
// TSRCell represents one table cell from TSR.
|
||||
type TSRCell struct {
|
||||
X0, Y0, X1, Y1 float64
|
||||
Text string
|
||||
Label string // "table", "table row", "table column", etc.
|
||||
}
|
||||
|
||||
func (c TSRCell) Bounds() (float64, float64, float64, float64) {
|
||||
return c.X0, c.Y0, c.X1, c.Y1
|
||||
}
|
||||
@@ -1,116 +0,0 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestCollectFigures(t *testing.T) {
|
||||
t.Run("mixed layout types", func(t *testing.T) {
|
||||
sections := []Section{
|
||||
{LayoutType: "figure", Text: "fig1", Image: "img1"},
|
||||
{LayoutType: "text", Text: "text1"},
|
||||
{LayoutType: "table", Text: "tbl1"},
|
||||
{LayoutType: "figure", Text: "fig2", Image: "img2"},
|
||||
{LayoutType: "title", Text: "title1"},
|
||||
}
|
||||
figures := CollectFigures(sections)
|
||||
if len(figures) != 2 {
|
||||
t.Fatalf("expected 2 figures, got %d", len(figures))
|
||||
}
|
||||
if figures[0].Text != "fig1" || figures[0].Image != "img1" {
|
||||
t.Errorf("first figure: expected (fig1, img1), got (%s, %s)", figures[0].Text, figures[0].Image)
|
||||
}
|
||||
if figures[1].Text != "fig2" || figures[1].Image != "img2" {
|
||||
t.Errorf("second figure: expected (fig2, img2), got (%s, %s)", figures[1].Text, figures[1].Image)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("no figures", func(t *testing.T) {
|
||||
sections := []Section{
|
||||
{LayoutType: "text", Text: "text1"},
|
||||
{LayoutType: "table", Text: "tbl1"},
|
||||
{LayoutType: "title", Text: "title1"},
|
||||
}
|
||||
figures := CollectFigures(sections)
|
||||
if len(figures) != 0 {
|
||||
t.Fatalf("expected 0 figures, got %d", len(figures))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("nil input", func(t *testing.T) {
|
||||
figures := CollectFigures(nil)
|
||||
if figures != nil {
|
||||
t.Fatalf("expected nil for nil input, got %d elements", len(figures))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("empty input", func(t *testing.T) {
|
||||
figures := CollectFigures([]Section{})
|
||||
if figures == nil {
|
||||
t.Fatal("expected empty slice (not nil) for empty input")
|
||||
}
|
||||
if len(figures) != 0 {
|
||||
t.Fatalf("expected 0 figures, got %d", len(figures))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("all figures", func(t *testing.T) {
|
||||
sections := []Section{
|
||||
{LayoutType: "figure", Text: "fig1"},
|
||||
{LayoutType: "figure", Text: "fig2"},
|
||||
{LayoutType: "figure", Text: "fig3"},
|
||||
}
|
||||
figures := CollectFigures(sections)
|
||||
if len(figures) != 3 {
|
||||
t.Fatalf("expected 3 figures, got %d", len(figures))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("figure with empty image", func(t *testing.T) {
|
||||
sections := []Section{
|
||||
{LayoutType: "figure", Text: "fig1", Image: ""},
|
||||
{LayoutType: "figure", Text: "fig2", Image: "img2"},
|
||||
}
|
||||
figures := CollectFigures(sections)
|
||||
if len(figures) != 2 {
|
||||
t.Fatalf("expected 2 figures, got %d", len(figures))
|
||||
}
|
||||
// Figure with empty image is still collected — downstream should handle.
|
||||
if figures[0].Image != "" {
|
||||
t.Errorf("first figure: expected empty Image, got %s", figures[0].Image)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("single section, figure", func(t *testing.T) {
|
||||
figures := CollectFigures([]Section{
|
||||
{LayoutType: "figure", Text: "only", Image: "img"},
|
||||
})
|
||||
if len(figures) != 1 {
|
||||
t.Fatalf("expected 1 figure, got %d", len(figures))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("single section, not figure", func(t *testing.T) {
|
||||
figures := CollectFigures([]Section{
|
||||
{LayoutType: "text", Text: "only"},
|
||||
})
|
||||
if len(figures) != 0 {
|
||||
t.Fatalf("expected 0 figures, got %d", len(figures))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("case sensitive", func(t *testing.T) {
|
||||
sections := []Section{
|
||||
{LayoutType: "Figure", Text: "fig1"},
|
||||
{LayoutType: "FIGURE", Text: "fig2"},
|
||||
{LayoutType: "figure", Text: "fig3"},
|
||||
}
|
||||
figures := CollectFigures(sections)
|
||||
if len(figures) != 1 {
|
||||
t.Fatalf("only lowercase 'figure' should match, got %d", len(figures))
|
||||
}
|
||||
if figures[0].Text != "fig3" {
|
||||
t.Errorf("expected fig3, got %s", figures[0].Text)
|
||||
}
|
||||
})
|
||||
}
|
||||
@@ -1,19 +1,21 @@
|
||||
package parser
|
||||
package util
|
||||
|
||||
import (
|
||||
"encoding/base64"
|
||||
"fmt"
|
||||
"image"
|
||||
"image/color"
|
||||
"log/slog"
|
||||
"math"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// cropSectionImage crops region(s) from rendered page images based on a
|
||||
// CropSectionImage crops region(s) from rendered page images based on a
|
||||
// position tag and returns a base64-encoded PNG. Returns "" if cropping
|
||||
// is not possible (missing images, out-of-bounds, invalid tag).
|
||||
//
|
||||
// Python: pdf_parser.py:1802 RAGFlowPdfParser.crop()
|
||||
func cropSectionImage(posTag string, decodedImages map[int]image.Image, zoom float64) string {
|
||||
func CropSectionImage(posTag string, decodedImages map[int]image.Image, zoom float64) string {
|
||||
if len(decodedImages) == 0 {
|
||||
slog.Warn("cropSectionImage: no page images available, skipping image generation")
|
||||
return ""
|
||||
@@ -26,7 +28,7 @@ func cropSectionImage(posTag string, decodedImages map[int]image.Image, zoom flo
|
||||
}
|
||||
|
||||
// Filter valid positions (all pages available).
|
||||
var valid []Position
|
||||
var valid []pdf.Position
|
||||
for _, pos := range positions {
|
||||
allValid := true
|
||||
for _, pn := range pos.PageNumbers {
|
||||
@@ -67,7 +69,7 @@ func cropSectionImage(posTag string, decodedImages map[int]image.Image, zoom flo
|
||||
lastPageH := float64(decodedImages[lastPageIdx].Bounds().Dy()) / zoom
|
||||
|
||||
// topBand: 120px context above the first content position.
|
||||
topBandPos := Position{
|
||||
topBandPos := pdf.Position{
|
||||
PageNumbers: []int{firstPageIdx},
|
||||
Left: first.Left,
|
||||
Right: first.Right,
|
||||
@@ -75,7 +77,7 @@ func cropSectionImage(posTag string, decodedImages map[int]image.Image, zoom flo
|
||||
Bottom: math.Max(first.Top-gap, 0),
|
||||
}
|
||||
// bottomBand: 120px context below the last content position.
|
||||
bottomBandPos := Position{
|
||||
bottomBandPos := pdf.Position{
|
||||
PageNumbers: []int{lastPageIdx},
|
||||
Left: last.Left,
|
||||
Right: last.Right,
|
||||
@@ -91,21 +93,21 @@ func cropSectionImage(posTag string, decodedImages map[int]image.Image, zoom flo
|
||||
var segments []segment
|
||||
|
||||
allPos := make([]struct {
|
||||
pos Position
|
||||
pos pdf.Position
|
||||
isEdge bool
|
||||
}, 0, len(valid)+2)
|
||||
allPos = append(allPos, struct {
|
||||
pos Position
|
||||
pos pdf.Position
|
||||
isEdge bool
|
||||
}{topBandPos, true})
|
||||
for _, pos := range valid {
|
||||
allPos = append(allPos, struct {
|
||||
pos Position
|
||||
pos pdf.Position
|
||||
isEdge bool
|
||||
}{pos, false})
|
||||
}
|
||||
allPos = append(allPos, struct {
|
||||
pos Position
|
||||
pos pdf.Position
|
||||
isEdge bool
|
||||
}{bottomBandPos, true})
|
||||
|
||||
@@ -147,7 +149,7 @@ func cropSectionImage(posTag string, decodedImages map[int]image.Image, zoom flo
|
||||
bottomClamped := math.Min(accumBottom, pageH)
|
||||
|
||||
// Crop first page of this position.
|
||||
cropped := fastCrop(pageImg,
|
||||
cropped := FastCrop(pageImg,
|
||||
int(left*zoom), int(top*zoom),
|
||||
int(right*zoom), int(bottomClamped))
|
||||
if isEdge {
|
||||
@@ -168,7 +170,7 @@ func cropSectionImage(posTag string, decodedImages map[int]image.Image, zoom flo
|
||||
}
|
||||
pageH2 := float64(pageImg2.Bounds().Dy())
|
||||
bottomClamped2 := math.Min(bottomRemaining, pageH2)
|
||||
cropped2 := fastCrop(pageImg2,
|
||||
cropped2 := FastCrop(pageImg2,
|
||||
int(left*zoom), 0,
|
||||
int(right*zoom), int(bottomClamped2))
|
||||
if isEdge {
|
||||
@@ -229,7 +231,7 @@ func cropSectionImage(posTag string, decodedImages map[int]image.Image, zoom flo
|
||||
curY += srcH + gap
|
||||
}
|
||||
|
||||
data, err := encodePNG(stitched)
|
||||
data, err := EncodePNG(stitched)
|
||||
if err != nil {
|
||||
slog.Warn("cropSectionImage: PNG encode failed", "err", err)
|
||||
return ""
|
||||
@@ -250,7 +252,7 @@ func cropSectionImage(posTag string, decodedImages map[int]image.Image, zoom flo
|
||||
// louts = [layout for layout in self.page_layout[pn] if layout["type"] == ltype]
|
||||
// ii = Recognizer.find_overlapped(b, louts, naive=True)
|
||||
// if ii is not None: b = louts[ii]
|
||||
func cropSectionByDLA(sec Section, dlaDebug []DLAPageRegions, pageImages map[int]image.Image) string {
|
||||
func CropSectionByDLA(sec pdf.Section, dlaDebug []pdf.DLAPageRegions, pageImages map[int]image.Image) string {
|
||||
if len(sec.Positions) == 0 || len(sec.Positions[0].PageNumbers) == 0 {
|
||||
return ""
|
||||
}
|
||||
@@ -258,7 +260,7 @@ func cropSectionByDLA(sec Section, dlaDebug []DLAPageRegions, pageImages map[int
|
||||
pos := sec.Positions[0]
|
||||
|
||||
// Find DLA regions for this page.
|
||||
var regions []DLARegion
|
||||
var regions []pdf.DLARegion
|
||||
for _, dp := range dlaDebug {
|
||||
if dp.Page == pg {
|
||||
regions = dp.Regions
|
||||
@@ -270,22 +272,22 @@ func cropSectionByDLA(sec Section, dlaDebug []DLAPageRegions, pageImages map[int
|
||||
}
|
||||
|
||||
// Convert section bbox from PDF points (72 DPI) to DLA pixel space (216 DPI).
|
||||
scale := dlaDPI / 72.0 // 3.0
|
||||
bx := rect{
|
||||
x0: pos.Left * scale,
|
||||
y0: pos.Top * scale,
|
||||
x1: pos.Right * scale,
|
||||
y1: pos.Bottom * scale,
|
||||
scale := pdf.DlaDPI / 72.0 // 3.0
|
||||
bx := Rect{
|
||||
X0: pos.Left * scale,
|
||||
Y0: pos.Top * scale,
|
||||
X1: pos.Right * scale,
|
||||
Y1: pos.Bottom * scale,
|
||||
}
|
||||
|
||||
// Find best-overlapping figure or equation DLA region.
|
||||
bestIdx := -1
|
||||
bestOverlap := 0.0
|
||||
for i, r := range regions {
|
||||
if r.Label != LayoutTypeFigure && r.Label != LayoutTypeEquation {
|
||||
if r.Label != pdf.LayoutTypeFigure && r.Label != pdf.LayoutTypeEquation {
|
||||
continue
|
||||
}
|
||||
overlap := rectOverlap(bx, rect{r.X0, r.Y0, r.X1, r.Y1})
|
||||
overlap := RectOverlap(bx, Rect{r.X0, r.Y0, r.X1, r.Y1})
|
||||
if overlap > bestOverlap {
|
||||
bestOverlap = overlap
|
||||
bestIdx = i
|
||||
@@ -300,12 +302,12 @@ func cropSectionByDLA(sec Section, dlaDebug []DLAPageRegions, pageImages map[int
|
||||
if !ok {
|
||||
return ""
|
||||
}
|
||||
cropped, err := cropImageRegion(img, regions[bestIdx])
|
||||
cropped, err := CropImageRegion(img, regions[bestIdx])
|
||||
if err != nil {
|
||||
slog.Warn("cropSectionByDLA: cropImageRegion failed", "page", pg, "err", err)
|
||||
return ""
|
||||
}
|
||||
data, err := encodePNG(cropped)
|
||||
data, err := EncodePNG(cropped)
|
||||
if err != nil {
|
||||
slog.Warn("cropSectionByDLA: PNG encode failed", "err", err)
|
||||
return ""
|
||||
@@ -360,7 +362,7 @@ func rotateCoordCW(x, y float64, origW, origH int, angle int) (float64, float64)
|
||||
|
||||
// rotateImageCW rotates an image clockwise. Only 0/90/180/270 supported;
|
||||
// other values return nil. Matches Python PIL.Image.rotate(-angle, expand=True).
|
||||
func rotateImageCW(img image.Image, angle int) *image.RGBA {
|
||||
func RotateImageCW(img image.Image, angle int) *image.RGBA {
|
||||
b := img.Bounds()
|
||||
w, h := b.Dx(), b.Dy()
|
||||
|
||||
@@ -389,7 +391,7 @@ func rotateImageCW(img image.Image, angle int) *image.RGBA {
|
||||
// are the ORIGINAL (pre-rotation) image dimensions.
|
||||
//
|
||||
// Python: pdf_parser.py:602 _map_rotated_point()
|
||||
func mapRotatedPointToOriginal(x, y float64, angle int, origW, origH int) (float64, float64) {
|
||||
func MapRotatedPointToOriginal(x, y float64, angle int, origW, origH int) (float64, float64) {
|
||||
switch angle {
|
||||
case 0:
|
||||
return x, y
|
||||
@@ -409,3 +411,27 @@ func mapRotatedPointToOriginal(x, y float64, angle int, origW, origH int) (float
|
||||
return x, y
|
||||
}
|
||||
}
|
||||
|
||||
// CropImageRegion crops a pdf.DLARegion from an image with a 3% margin
|
||||
// (matching Python's _table_transformer_job: w*0.03, h*0.03).
|
||||
func CropImageRegion(img image.Image, r pdf.DLARegion) (image.Image, error) {
|
||||
w := r.X1 - r.X0
|
||||
h := r.Y1 - r.Y0
|
||||
marginX := w * 0.03
|
||||
marginY := h * 0.03
|
||||
maxX := float64(img.Bounds().Dx())
|
||||
maxY := float64(img.Bounds().Dy())
|
||||
x0 := int(math.Max(0, r.X0-marginX))
|
||||
y0 := int(math.Max(0, r.Y0-marginY))
|
||||
x1 := int(math.Min(maxX, r.X1+marginX))
|
||||
y1 := int(math.Min(maxY, r.Y1+marginY))
|
||||
// Python PIL.Image.crop() raises ValueError when right < left or
|
||||
// bottom < top. We return an error instead of silently falling back
|
||||
// to the full-page image — the caller skips this table gracefully.
|
||||
if x0 >= x1 || y0 >= y1 {
|
||||
return nil, fmt.Errorf("crop: invalid region x0=%d y0=%d x1=%d y1=%d (DLA raw: %.1f,%.1f,%.1f,%.1f)",
|
||||
x0, y0, x1, y1, r.X0, r.Y0, r.X1, r.Y1)
|
||||
}
|
||||
cropped := FastCrop(img, x0, y0, x1, y1)
|
||||
return cropped, nil
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
package parser
|
||||
package util
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
@@ -7,6 +7,7 @@ import (
|
||||
"image/color"
|
||||
"image/png"
|
||||
"math"
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
"testing"
|
||||
)
|
||||
|
||||
@@ -35,7 +36,7 @@ func TestCropSectionImage_SinglePage(t *testing.T) {
|
||||
0: makeTestPageImage(200, 300, color.RGBA{255, 0, 0, 255}),
|
||||
}
|
||||
posTag := FormatPositionTag(0, 10, 100, 20, 150)
|
||||
b64 := cropSectionImage(posTag, pageImages, 1)
|
||||
b64 := CropSectionImage(posTag, pageImages, 1)
|
||||
|
||||
if b64 == "" {
|
||||
t.Fatal("expected non-empty base64 image")
|
||||
@@ -59,10 +60,10 @@ func TestCropSectionImage_SinglePage(t *testing.T) {
|
||||
func TestCropSectionImage_EmptyImages(t *testing.T) {
|
||||
posTag := FormatPositionTag(0, 10, 100, 20, 150)
|
||||
|
||||
if b64 := cropSectionImage(posTag, nil, 1); b64 != "" {
|
||||
if b64 := CropSectionImage(posTag, nil, 1); b64 != "" {
|
||||
t.Error("nil pageImages should return empty string")
|
||||
}
|
||||
if b64 := cropSectionImage(posTag, map[int]image.Image{}, 1); b64 != "" {
|
||||
if b64 := CropSectionImage(posTag, map[int]image.Image{}, 1); b64 != "" {
|
||||
t.Error("empty pageImages should return empty string")
|
||||
}
|
||||
}
|
||||
@@ -72,7 +73,7 @@ func TestCropSectionImage_OutOfBounds(t *testing.T) {
|
||||
0: makeTestPageImage(200, 300, color.RGBA{255, 0, 0, 255}),
|
||||
}
|
||||
posTag := FormatPositionTag(5, 10, 100, 20, 150)
|
||||
if b64 := cropSectionImage(posTag, pageImages, 1); b64 != "" {
|
||||
if b64 := CropSectionImage(posTag, pageImages, 1); b64 != "" {
|
||||
t.Error("out-of-bounds page should return empty string")
|
||||
}
|
||||
}
|
||||
@@ -81,10 +82,10 @@ func TestCropSectionImage_InvalidTag(t *testing.T) {
|
||||
pageImages := map[int]image.Image{
|
||||
0: makeTestPageImage(200, 300, color.RGBA{255, 0, 0, 255}),
|
||||
}
|
||||
if b64 := cropSectionImage("invalid", pageImages, 1); b64 != "" {
|
||||
if b64 := CropSectionImage("invalid", pageImages, 1); b64 != "" {
|
||||
t.Error("invalid position tag should return empty string")
|
||||
}
|
||||
if b64 := cropSectionImage("", pageImages, 1); b64 != "" {
|
||||
if b64 := CropSectionImage("", pageImages, 1); b64 != "" {
|
||||
t.Error("empty position tag should return empty string")
|
||||
}
|
||||
}
|
||||
@@ -94,7 +95,7 @@ func TestCropSectionImage_ContextPadding(t *testing.T) {
|
||||
0: makeTestPageImage(200, 800, color.RGBA{255, 0, 0, 255}),
|
||||
}
|
||||
posTag := FormatPositionTag(0, 20, 120, 300, 400)
|
||||
b64 := cropSectionImage(posTag, pageImages, 1)
|
||||
b64 := CropSectionImage(posTag, pageImages, 1)
|
||||
if b64 == "" {
|
||||
t.Fatal("expected non-empty result")
|
||||
}
|
||||
@@ -111,7 +112,7 @@ func TestCropSectionImage_ZoomScaling(t *testing.T) {
|
||||
0: makeTestPageImage(400, 600, color.RGBA{255, 0, 0, 255}),
|
||||
}
|
||||
posTag := FormatPositionTag(0, 10, 100, 20, 150)
|
||||
b64 := cropSectionImage(posTag, pageImages, 2)
|
||||
b64 := CropSectionImage(posTag, pageImages, 2)
|
||||
if b64 == "" {
|
||||
t.Fatal("expected non-empty result")
|
||||
}
|
||||
@@ -136,7 +137,7 @@ func TestRotateImageCW(t *testing.T) {
|
||||
img.Set(2, 1, gr)
|
||||
|
||||
t.Run("0 degrees", func(t *testing.T) {
|
||||
rot := rotateImageCW(img, 0)
|
||||
rot := RotateImageCW(img, 0)
|
||||
if rot == nil {
|
||||
t.Fatal("nil result")
|
||||
}
|
||||
@@ -148,7 +149,7 @@ func TestRotateImageCW(t *testing.T) {
|
||||
}
|
||||
})
|
||||
t.Run("90 degrees", func(t *testing.T) {
|
||||
rot := rotateImageCW(img, 90)
|
||||
rot := RotateImageCW(img, 90)
|
||||
if rot == nil {
|
||||
t.Fatal("nil result")
|
||||
}
|
||||
@@ -173,7 +174,7 @@ func TestRotateImageCW(t *testing.T) {
|
||||
}
|
||||
})
|
||||
t.Run("180 degrees", func(t *testing.T) {
|
||||
rot := rotateImageCW(img, 180)
|
||||
rot := RotateImageCW(img, 180)
|
||||
if rot == nil {
|
||||
t.Fatal("nil result")
|
||||
}
|
||||
@@ -188,7 +189,7 @@ func TestRotateImageCW(t *testing.T) {
|
||||
}
|
||||
})
|
||||
t.Run("270 degrees", func(t *testing.T) {
|
||||
rot := rotateImageCW(img, 270)
|
||||
rot := RotateImageCW(img, 270)
|
||||
if rot == nil {
|
||||
t.Fatal("nil result")
|
||||
}
|
||||
@@ -197,7 +198,7 @@ func TestRotateImageCW(t *testing.T) {
|
||||
}
|
||||
})
|
||||
t.Run("invalid angle", func(t *testing.T) {
|
||||
if rotateImageCW(img, 45) != nil {
|
||||
if RotateImageCW(img, 45) != nil {
|
||||
t.Error("expected nil for invalid angle")
|
||||
}
|
||||
})
|
||||
@@ -211,7 +212,7 @@ func TestMapRotatedPointToOriginal_RoundTrip(t *testing.T) {
|
||||
for _, ox := range []float64{0, 50, 199} {
|
||||
for _, oy := range []float64{0, 30, 99} {
|
||||
rx, ry := rotateCoordCW(ox, oy, origW, origH, angle)
|
||||
gotX, gotY := mapRotatedPointToOriginal(rx, ry, angle, origW, origH)
|
||||
gotX, gotY := MapRotatedPointToOriginal(rx, ry, angle, origW, origH)
|
||||
if math.Abs(gotX-ox) > 0.01 || math.Abs(gotY-oy) > 0.01 {
|
||||
t.Errorf("angle=%d orig(%.0f,%.0f) → rot(%.0f,%.0f) → got(%.1f,%.1f)",
|
||||
angle, ox, oy, rx, ry, gotX, gotY)
|
||||
@@ -236,7 +237,7 @@ func TestMapRotatedPointToOriginal(t *testing.T) {
|
||||
{270, 50, 30, 200, 100, 169, 50}, // rotW=200: inverse (199-30,50)
|
||||
}
|
||||
for _, tt := range tests {
|
||||
gotX, gotY := mapRotatedPointToOriginal(tt.rx, tt.ry, tt.angle, tt.origW, tt.origH)
|
||||
gotX, gotY := MapRotatedPointToOriginal(tt.rx, tt.ry, tt.angle, tt.origW, tt.origH)
|
||||
if math.Abs(gotX-tt.wantX) > 0.01 || math.Abs(gotY-tt.wantY) > 0.01 {
|
||||
t.Errorf("angle=%d (%f,%f) got(%f,%f) want(%f,%f)",
|
||||
tt.angle, tt.rx, tt.ry, gotX, gotY, tt.wantX, tt.wantY)
|
||||
@@ -262,9 +263,9 @@ func TestCropSectionImage_MultiPage(t *testing.T) {
|
||||
1: makeTestPageImage(100, 800, color.RGBA{0, 200, 0, 255}),
|
||||
2: makeTestPageImage(100, 800, color.RGBA{0, 0, 200, 255}),
|
||||
}
|
||||
// Position spans pages 0-2, bottom reaches into page 2.
|
||||
// pdf.Position spans pages 0-2, bottom reaches into page 2.
|
||||
posTag := "@@1-3\t0.0\t100.0\t0.0\t500.0##"
|
||||
b64 := cropSectionImage(posTag, pageImages, 1)
|
||||
b64 := CropSectionImage(posTag, pageImages, 1)
|
||||
if b64 == "" {
|
||||
t.Fatal("expected non-empty result for multi-page position")
|
||||
}
|
||||
@@ -289,7 +290,7 @@ func TestCropSectionImage_LargePageSpan(t *testing.T) {
|
||||
1: makeTestPageImage(100, 600, color.RGBA{0, 200, 0, 255}),
|
||||
}
|
||||
posTag := "@@1-2\t0.0\t100.0\t0.0\t900.0##"
|
||||
b64 := cropSectionImage(posTag, pageImages, 1)
|
||||
b64 := CropSectionImage(posTag, pageImages, 1)
|
||||
if b64 == "" {
|
||||
t.Fatal("expected non-empty result")
|
||||
}
|
||||
@@ -312,20 +313,20 @@ func TestCropSectionByDLA(t *testing.T) {
|
||||
// DLA regions in pixel space (216 DPI).
|
||||
// Figure region at (30, 60, 270, 420) — a large area covering most of the image.
|
||||
// Text region at (10, 400, 100, 440) — a small text box near the bottom.
|
||||
dlaDebug := []DLAPageRegions{{
|
||||
dlaDebug := []pdf.DLAPageRegions{{
|
||||
Page: 0,
|
||||
Regions: []DLARegion{
|
||||
Regions: []pdf.DLARegion{
|
||||
{X0: 10, Y0: 400, X1: 100, Y1: 440, Label: "text"},
|
||||
{X0: 30, Y0: 60, X1: 270, Y1: 420, Label: "figure"},
|
||||
{X0: 5, Y0: 5, X1: 290, Y1: 55, Label: "title"},
|
||||
},
|
||||
}}
|
||||
|
||||
// Section with a text-box-sized bbox (PDF points, 72 DPI).
|
||||
// pdf.Section with a text-box-sized bbox (PDF points, 72 DPI).
|
||||
// In pixel space at scale 3: (60, 1200, 150, 1320) → (20, 400, 50, 440).
|
||||
// This overlaps with the "figure" DLA region.
|
||||
sec := Section{
|
||||
Positions: []Position{{
|
||||
sec := pdf.Section{
|
||||
Positions: []pdf.Position{{
|
||||
PageNumbers: []int{0},
|
||||
Left: 20, Right: 50,
|
||||
Top: 400 / 3.0, Bottom: 440 / 3.0,
|
||||
@@ -333,7 +334,7 @@ func TestCropSectionByDLA(t *testing.T) {
|
||||
LayoutType: "figure",
|
||||
}
|
||||
|
||||
result := cropSectionByDLA(sec, dlaDebug, pageImages)
|
||||
result := CropSectionByDLA(sec, dlaDebug, pageImages)
|
||||
if result == "" {
|
||||
t.Fatal("expected non-empty result for figure overlapping DLA region")
|
||||
}
|
||||
@@ -356,22 +357,22 @@ func TestCropSectionByDLA_NoMatch(t *testing.T) {
|
||||
pageImages := map[int]image.Image{
|
||||
0: makeTestPageImage(300, 450, color.RGBA{255, 0, 0, 255}),
|
||||
}
|
||||
dlaDebug := []DLAPageRegions{{
|
||||
dlaDebug := []pdf.DLAPageRegions{{
|
||||
Page: 0,
|
||||
Regions: []DLARegion{
|
||||
Regions: []pdf.DLARegion{
|
||||
{X0: 10, Y0: 10, X1: 100, Y1: 50, Label: "title"},
|
||||
{X0: 10, Y0: 60, X1: 100, Y1: 100, Label: "text"},
|
||||
},
|
||||
}}
|
||||
// Section whose bbox doesn't overlap any figure/equation DLA region.
|
||||
sec := Section{
|
||||
Positions: []Position{{
|
||||
// pdf.Section whose bbox doesn't overlap any figure/equation DLA region.
|
||||
sec := pdf.Section{
|
||||
Positions: []pdf.Position{{
|
||||
PageNumbers: []int{0},
|
||||
Left: 20, Right: 50, Top: 20, Bottom: 50,
|
||||
}},
|
||||
LayoutType: "figure",
|
||||
}
|
||||
result := cropSectionByDLA(sec, dlaDebug, pageImages)
|
||||
result := CropSectionByDLA(sec, dlaDebug, pageImages)
|
||||
if result != "" {
|
||||
t.Errorf("expected empty result when no figure/equation DLA region found, got length %d", len(result))
|
||||
}
|
||||
@@ -380,12 +381,53 @@ func TestCropSectionByDLA_NoMatch(t *testing.T) {
|
||||
// TestCropSectionByDLA_EmptyInputs returns empty for edge cases.
|
||||
func TestCropSectionByDLA_EmptyInputs(t *testing.T) {
|
||||
// Empty positions.
|
||||
if got := cropSectionByDLA(Section{}, nil, nil); got != "" {
|
||||
if got := CropSectionByDLA(pdf.Section{}, nil, nil); got != "" {
|
||||
t.Error("expected empty for empty positions")
|
||||
}
|
||||
// Empty page numbers.
|
||||
sec := Section{Positions: []Position{{PageNumbers: nil}}}
|
||||
if got := cropSectionByDLA(sec, nil, nil); got != "" {
|
||||
sec := pdf.Section{Positions: []pdf.Position{{PageNumbers: nil}}}
|
||||
if got := CropSectionByDLA(sec, nil, nil); got != "" {
|
||||
t.Error("expected empty for empty page numbers")
|
||||
}
|
||||
}
|
||||
func TestCropImageRegion(t *testing.T) {
|
||||
img := image.NewRGBA(image.Rect(0, 0, 200, 300))
|
||||
|
||||
t.Run("normal crop", func(t *testing.T) {
|
||||
r := pdf.DLARegion{X0: 10, Y0: 20, X1: 100, Y1: 150}
|
||||
cropped, err := CropImageRegion(img, r)
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
// 3% proportional margin: 90×3%≈3px, 130×3%≈4px → 95×137
|
||||
if cropped.Bounds().Dx() != 95 || cropped.Bounds().Dy() != 137 {
|
||||
t.Errorf("size %v, want 95x137", cropped.Bounds())
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("x0 >= x1 returns error", func(t *testing.T) {
|
||||
// 3% proportional margin on each side: if the gap is too small after margin expansion, x0 ≥ x1 triggers error.
|
||||
r := pdf.DLARegion{X0: 110, Y0: 20, X1: 50, Y1: 150}
|
||||
_, err := CropImageRegion(img, r)
|
||||
if err == nil {
|
||||
t.Fatal("expected error for x0 >= x1, got nil")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("y0 >= y1 returns error", func(t *testing.T) {
|
||||
r := pdf.DLARegion{X0: 10, Y0: 150, X1: 100, Y1: 20}
|
||||
_, err := CropImageRegion(img, r)
|
||||
if err == nil {
|
||||
t.Fatal("expected error for y0 >= y1, got nil")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("region fully outside image bounds", func(t *testing.T) {
|
||||
// Clamped to image bounds → zero-width/height → error.
|
||||
r := pdf.DLARegion{X0: 300, Y0: 400, X1: 500, Y1: 600}
|
||||
_, err := CropImageRegion(img, r)
|
||||
if err == nil {
|
||||
t.Fatal("expected error for region outside image bounds")
|
||||
}
|
||||
})
|
||||
}
|
||||
110
internal/deepdoc/parser/pdf/util/eng_detect.go
Normal file
110
internal/deepdoc/parser/pdf/util/eng_detect.go
Normal file
@@ -0,0 +1,110 @@
|
||||
package util
|
||||
|
||||
import (
|
||||
"math/rand/v2"
|
||||
"strings"
|
||||
|
||||
pdf "ragflow/internal/deepdoc/parser/pdf/type"
|
||||
)
|
||||
|
||||
// IsASCIIPrintable returns true for characters that match Python's
|
||||
// is_english regex: [ a-zA-Z0-9,/¸;:'\[\]\(\)!@#$%^&*\"?<>._-]
|
||||
func IsASCIIPrintable(r rune) bool {
|
||||
if r == ' ' {
|
||||
return true
|
||||
}
|
||||
if r >= 'a' && r <= 'z' {
|
||||
return true
|
||||
}
|
||||
if r >= 'A' && r <= 'Z' {
|
||||
return true
|
||||
}
|
||||
if r >= '0' && r <= '9' {
|
||||
return true
|
||||
}
|
||||
// Additional ASCII symbols from the Python regex
|
||||
switch r {
|
||||
case ',', '/', '¸', ';', ':', '\'', '[', ']', '(', ')',
|
||||
'!', '@', '#', '$', '%', '^', '&', '*', '"', '?',
|
||||
'<', '>', '.', '_', '-':
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// DefaultSampleChars returns a random sample of up to n character texts,
|
||||
// concatenated. Matches Python's random.choices([c["text"] for c in
|
||||
// page_chars], k=min(100, len(page_chars))).
|
||||
func DefaultSampleChars(chars []pdf.TextChar, n int) string {
|
||||
if n <= 0 || len(chars) == 0 {
|
||||
return ""
|
||||
}
|
||||
m := min(n, len(chars))
|
||||
// Fisher-Yates shuffle on indices, then take first m.
|
||||
indices := make([]int, len(chars))
|
||||
for i := range indices {
|
||||
indices[i] = i
|
||||
}
|
||||
rand.Shuffle(len(indices), func(i, j int) {
|
||||
indices[i], indices[j] = indices[j], indices[i]
|
||||
})
|
||||
var buf strings.Builder
|
||||
for i := 0; i < m; i++ {
|
||||
buf.WriteString(chars[indices[i]].Text)
|
||||
}
|
||||
return buf.String()
|
||||
}
|
||||
|
||||
// FullTextFromChars concatenates all chars text across pages for scan noise detection.
|
||||
func FullTextFromChars(pageChars map[int][]pdf.TextChar) string {
|
||||
var sb strings.Builder
|
||||
for _, chars := range pageChars {
|
||||
for _, c := range chars {
|
||||
sb.WriteString(c.Text)
|
||||
}
|
||||
}
|
||||
return sb.String()
|
||||
}
|
||||
|
||||
// DetectEnglish detects whether a PDF is primarily English by per-page
|
||||
// majority vote, matching Python's is_english logic in __images__
|
||||
// (pdf_parser.py:1519-1526).
|
||||
//
|
||||
// Each page: sample up to 100 character texts via sampler, join into one
|
||||
// string, check if there is a run of 30+ consecutive ASCII characters
|
||||
// (letters, digits, spaces, punctuation). Pages with such a run vote
|
||||
// "English". Returns true when a strict majority of pages vote yes.
|
||||
//
|
||||
// totalPages is the denominator (len(self.page_images) in Python), including
|
||||
// image-only pages that have zero chars. This matches Python's behavior
|
||||
// where empty pages dilute the majority.
|
||||
func DetectEnglish(pageChars map[int][]pdf.TextChar, totalPages int, sample pdf.SampleFunc) bool {
|
||||
if totalPages == 0 || len(pageChars) == 0 {
|
||||
return false
|
||||
}
|
||||
if sample == nil {
|
||||
sample = DefaultSampleChars
|
||||
}
|
||||
pagesWithSeq := 0
|
||||
|
||||
for _, chars := range pageChars {
|
||||
if len(chars) == 0 {
|
||||
continue
|
||||
}
|
||||
sampleText := sample(chars, 100)
|
||||
run := 0
|
||||
for _, r := range sampleText {
|
||||
if IsASCIIPrintable(r) {
|
||||
run++
|
||||
if run >= 30 {
|
||||
pagesWithSeq++
|
||||
break
|
||||
}
|
||||
} else {
|
||||
run = 0
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return pagesWithSeq > totalPages/2
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user