Files
ragflow/internal/ingestion/component/chunker/qa_test.go
Jack 9b0719fa94 fix: Go ingestion migration batch 5 (Parser 1.1/1.7/2.11, Chunker 1.7/1.8/2.6/2.7, Tokenizer 6x fixes) (#17419)
## Summary

Continuation of the Python→Go ingestion pipeline migration (File →
Parser → Chunker → Extractor → Tokenizer). Fixes cover Parser, Chunker,
and Tokenizer gaps identified. Fix page number (0-indexed and 1-index
mixed before fix; use 1-indexed after fix) and chunk order issues.

### Parser
- **Slides TCADP (1.7):** `pptx_tcadp.go` + TCADP branch in
`pptx_parser.go`/`ppt_parser.go` — PowerPoint files now support
`parse_method="tcadp"` via the TCADP cloud service, matching the
spreadsheet-family TCADP pattern. PPT containers pass `"PPT"` as
fileType (not hardcoded `"PPTX"`).
- **Audio default output_format (2.11):** `defaultSetups()` audio
default changed from `"text"` to `"json"`, aligning with Python
`parser.py:232` and `AllowedOutputFormat["audio"]={"json"}`.
- **PDF VLM enhancement (1.1):** `maybeDispatchPDFVisionEnhancement` in
`pdf_vision_dispatch.go` enriches image/table items with IMAGE2TEXT
model descriptions after PDF parsing, mirroring Python
`enhance_media_sections_with_vision`. Semaphore fix: acquire before
goroutine start to prevent unbounded goroutine creation.
- **json family (2.3):** reclassified as Keep Go — `json_parser.go` is a
functional enhancement, not a parity gap.
- **page number:** changed from "mixed use of 1-indexed & 0-indexed" to
"1-indexed"

### Chunker
- **BULLET_PATTERN fallback (1.7):** 4th-level fallback in
`resolveTitleLevels` (`title.go`) detects bullet/numbered-list patterns
(Chinese legal, numbering, English) when outline + regex levels produce
only bodyLevel. Guarded by `allBodyLevel` to never override existing
structure.
- **Tag/One chunker fields (1.8):** `tag.go` sets `TopInt` from source
row index; `one.go` preserves `Positions`/`PDFPositions` from source
items. TSV multi-line RowNum fix: tracks `contentStart` for correct row
attribution.
- **Overlapped_percent normalization (2.6):**
`NormalizeOverlappedPercent` in `schema/chunker.go` mirrors Python
`common/float_utils.py:50-58` — accepts `[0,1)` fraction or `[0,90]`
percent, normalizes to canonical `[0,90]`.
- **Paragraph splitting (2.7):** aligned to Python flow `naive_merge` —
`CRLF` normalization, `splitKeepingDelimiter` preserves sentence
delimiters, single-section merge with token-budget-governed chunking.
- **chunk order:** sort by reading order

### Tokenizer
- **Phantom chunk filtering (Omission 2):** `isPhantomChunk` + filter
loop in `chunksFromTokenizerUpstream` skips zero-value ChunkDocs.
- **Batch size env var (Omission 3):** `embeddingBatchSize()` reads
`TOKENIZER_EMBEDDING_BATCH_SIZE`, defaults to 16.
- **Summary empty check (Diff 5):** `TrimSpace(s) != ""` → `s != ""`,
matching Python truthy check.
- **chunk_order_int all paths (Diff 8):** set unconditionally before
full_text/embedding branching.
- **Timeout default (Diff 10):** `600s` → `60s`, matching Python
`@timeout(60)`.
- **Small maxTokens truncation (Diff 14):** `truncateForEmbedding`
returns `""` when `maxTokens <= 10`, matching Python.

### Code review fixes
- Semaphore acquire moved before goroutine in `pdf_vision_dispatch.go`
(concurrency control)
- Context propagation in `pptx_tcadp.go` (cancellation support)
- Test resolver leak fix in `media_dispatch_test.go` (defer restore)
- Migration history comments removed per AGENTS.md

## Test plan
```
bash build.sh --test ./internal/parser/parser/... ./internal/ingestion/component/...
```

## Notes
- Migration diff tracking: `docs/migration_python_go_diff.md`
- Remaining gaps: Extractor component only (21 items)
2026-07-28 11:12:52 +08:00

290 lines
8.1 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
//
// Copyright 2026 The InfiniFlow Authors. All Rights Reserved.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//
package chunker
import (
"context"
"strings"
"testing"
"ragflow/internal/agent/runtime"
)
func TestQAChunker_Registered(t *testing.T) {
factory, _, _, ok := runtime.DefaultRegistry.Lookup("QAChunker")
if !ok {
t.Fatal("QAChunker not found in registry")
}
comp, err := factory("QAChunker", nil)
if err != nil {
t.Fatalf("factory failed: %v", err)
}
if comp == nil {
t.Fatal("component is nil")
}
}
func TestQAChunker_DelimiterTab(t *testing.T) {
comp, err := NewQAChunker(map[string]any{"lang": "english"})
if err != nil {
t.Fatal(err)
}
inputs := map[string]any{
"name": "test.txt",
"output_format": "text",
"text": "What is Go?\tGo is a programming language.",
}
out, err := comp.Invoke(context.Background(), nil, inputs)
if err != nil {
t.Fatalf("Invoke failed: %v", err)
}
chunks, _ := out["chunks"].([]map[string]any)
if len(chunks) != 1 {
t.Fatalf("expected 1 chunk, got %d", len(chunks))
}
chunk := chunks[0]
cww, _ := chunk["content_with_weight"].(string)
if cww != "Question: What is Go?\tAnswer: Go is a programming language." {
t.Fatalf("unexpected content: %q", cww)
}
}
func TestQAChunker_DelimiterComma(t *testing.T) {
comp, err := NewQAChunker(map[string]any{"lang": "english"})
if err != nil {
t.Fatal(err)
}
inputs := map[string]any{
"name": "test.csv",
"output_format": "text",
"text": "What is Rust?,Rust is a systems language.",
}
out, err := comp.Invoke(context.Background(), nil, inputs)
if err != nil {
t.Fatalf("Invoke failed: %v", err)
}
chunks, _ := out["chunks"].([]map[string]any)
if len(chunks) != 1 {
t.Fatalf("expected 1 chunk, got %d", len(chunks))
}
chunk := chunks[0]
cww, _ := chunk["content_with_weight"].(string)
if cww != "Question: What is Rust?\tAnswer: Rust is a systems language." {
t.Fatalf("unexpected content: %q", cww)
}
}
func TestQAChunker_Markdown(t *testing.T) {
comp, err := NewQAChunker(nil)
if err != nil {
t.Fatal(err)
}
inputs := map[string]any{
"name": "test.md",
"output_format": "markdown",
"markdown": "# What is Go?\nGo is a programming language.\n\n# What is Rust?\nRust is a systems language.",
}
out, err := comp.Invoke(context.Background(), nil, inputs)
if err != nil {
t.Fatalf("Invoke failed: %v", err)
}
chunks, _ := out["chunks"].([]map[string]any)
if len(chunks) != 2 {
t.Fatalf("expected 2 chunks, got %d", len(chunks))
}
}
func TestQAChunker_HTMLTable(t *testing.T) {
comp, err := NewQAChunker(nil)
if err != nil {
t.Fatal(err)
}
inputs := map[string]any{
"name": "test.xlsx",
"output_format": "html",
"html": "<table><tr><td>Q1</td><td>A1</td></tr><tr><td>Q2</td><td>A2</td></tr></table>",
}
out, err := comp.Invoke(context.Background(), nil, inputs)
if err != nil {
t.Fatalf("Invoke failed: %v", err)
}
chunks, _ := out["chunks"].([]map[string]any)
if len(chunks) != 2 {
t.Fatalf("expected 2 chunks, got %d", len(chunks))
}
}
func TestQAChunker_RmQAPrefix(t *testing.T) {
comp, err := NewQAChunker(map[string]any{"lang": "english"})
if err != nil {
t.Fatal(err)
}
inputs := map[string]any{
"name": "test.txt",
"output_format": "text",
"text": "Question: What is Go?\tAnswer: Go is a language.",
}
out, err := comp.Invoke(context.Background(), nil, inputs)
if err != nil {
t.Fatalf("Invoke failed: %v", err)
}
chunks, _ := out["chunks"].([]map[string]any)
cww, _ := chunks[0]["content_with_weight"].(string)
if cww != "Question: What is Go?\tAnswer: Go is a language." {
t.Fatalf("prefix not stripped: %q", cww)
}
}
func TestQAChunker_Empty(t *testing.T) {
comp, err := NewQAChunker(nil)
if err != nil {
t.Fatal(err)
}
inputs := map[string]any{
"name": "empty.txt",
"output_format": "text",
"text": "",
}
out, err := comp.Invoke(context.Background(), nil, inputs)
if err != nil {
t.Fatalf("Invoke failed: %v", err)
}
chunks, _ := out["chunks"].([]map[string]any)
if len(chunks) != 0 {
t.Fatalf("expected 0 chunks, got %d", len(chunks))
}
}
func TestQAChunker_CaseInsensitivePrefix(t *testing.T) {
comp, err := NewQAChunker(map[string]any{"lang": "english"})
if err != nil {
t.Fatal(err)
}
inputs := map[string]any{
"name": "test.txt",
"output_format": "text",
"text": "QUESTION: Hello\tANSWER: World",
}
out, err := comp.Invoke(context.Background(), nil, inputs)
if err != nil {
t.Fatalf("Invoke failed: %v", err)
}
chunks, _ := out["chunks"].([]map[string]any)
if len(chunks) != 1 {
t.Fatalf("expected 1 chunk, got %d", len(chunks))
}
cww, _ := chunks[0]["content_with_weight"].(string)
if cww != "Question: Hello\tAnswer: World" {
t.Fatalf("case-insensitive prefix not stripped: %q", cww)
}
}
func TestQAChunker_PrefixSpaceSeparatorStrips(t *testing.T) {
comp, err := NewQAChunker(map[string]any{"lang": "english"})
if err != nil {
t.Fatal(err)
}
inputs := map[string]any{
"name": "test.txt",
"output_format": "text",
"text": "A language model is useful\tQ How does it work",
}
out, err := comp.Invoke(context.Background(), nil, inputs)
if err != nil {
t.Fatalf("Invoke failed: %v", err)
}
chunks, _ := out["chunks"].([]map[string]any)
if len(chunks) != 1 {
t.Fatalf("expected 1 chunk, got %d", len(chunks))
}
cww, _ := chunks[0]["content_with_weight"].(string)
// Python qa.py:241 uses `[\t: ]+`, so a space is a valid separator:
// a leading "A"/"Q" followed by a space is stripped. .
if cww != "Question: language model is useful\tAnswer: How does it work" {
t.Fatalf("space-separator prefix not stripped: %q", cww)
}
}
func TestQAChunker_HeadingNoTrailingSpace(t *testing.T) {
comp, err := NewQAChunker(nil)
if err != nil {
t.Fatal(err)
}
inputs := map[string]any{
"name": "test.md",
"output_format": "markdown",
"markdown": "#Hello\nWorld\n",
}
out, err := comp.Invoke(context.Background(), nil, inputs)
if err != nil {
t.Fatalf("Invoke failed: %v", err)
}
chunks, _ := out["chunks"].([]map[string]any)
if len(chunks) != 1 {
t.Fatalf("expected 1 chunk, got %d", len(chunks))
}
}
func TestQAChunker_ChineseLang(t *testing.T) {
comp, err := NewQAChunker(map[string]any{"lang": "Chinese"})
if err != nil {
t.Fatal(err)
}
inputs := map[string]any{
"name": "test.txt",
"output_format": "text",
"text": "什么是Go\tGo是一种编程语言。",
}
out, err := comp.Invoke(context.Background(), nil, inputs)
if err != nil {
t.Fatalf("Invoke failed: %v", err)
}
chunks, _ := out["chunks"].([]map[string]any)
if len(chunks) != 1 {
t.Fatalf("expected 1 chunk, got %d", len(chunks))
}
cww, _ := chunks[0]["content_with_weight"].(string)
if want := "问题什么是Go\t回答Go是一种编程语言。"; cww != want {
t.Fatalf("unexpected content: %q, want %q", cww, want)
}
}
func TestQAChunker_MarkdownRendersHTML(t *testing.T) {
comp, err := NewQAChunker(nil)
if err != nil {
t.Fatal(err)
}
inputs := map[string]any{
"name": "test.md",
"output_format": "markdown",
"markdown": "# Title\nThis is **bold** text.\n",
}
out, err := comp.Invoke(context.Background(), nil, inputs)
if err != nil {
t.Fatalf("Invoke failed: %v", err)
}
chunks, _ := out["chunks"].([]map[string]any)
if len(chunks) != 1 {
t.Fatalf("expected 1 chunk, got %d", len(chunks))
}
cww, _ := chunks[0]["content_with_weight"].(string)
if !strings.Contains(cww, "<strong>bold</strong>") &&
!strings.Contains(cww, "<b>bold</b>") {
t.Fatalf("markdown not rendered to HTML: %q", cww)
}
}