mirror of
https://github.com/infiniflow/ragflow.git
synced 2026-07-30 20:49:21 +08:00
## Summary
Continuation of the Python→Go ingestion pipeline migration (File →
Parser → Chunker → Extractor → Tokenizer). Fixes cover Parser, Chunker,
and Tokenizer gaps identified. Fix page number (0-indexed and 1-index
mixed before fix; use 1-indexed after fix) and chunk order issues.
### Parser
- **Slides TCADP (1.7):** `pptx_tcadp.go` + TCADP branch in
`pptx_parser.go`/`ppt_parser.go` — PowerPoint files now support
`parse_method="tcadp"` via the TCADP cloud service, matching the
spreadsheet-family TCADP pattern. PPT containers pass `"PPT"` as
fileType (not hardcoded `"PPTX"`).
- **Audio default output_format (2.11):** `defaultSetups()` audio
default changed from `"text"` to `"json"`, aligning with Python
`parser.py:232` and `AllowedOutputFormat["audio"]={"json"}`.
- **PDF VLM enhancement (1.1):** `maybeDispatchPDFVisionEnhancement` in
`pdf_vision_dispatch.go` enriches image/table items with IMAGE2TEXT
model descriptions after PDF parsing, mirroring Python
`enhance_media_sections_with_vision`. Semaphore fix: acquire before
goroutine start to prevent unbounded goroutine creation.
- **json family (2.3):** reclassified as Keep Go — `json_parser.go` is a
functional enhancement, not a parity gap.
- **page number:** changed from "mixed use of 1-indexed & 0-indexed" to
"1-indexed"
### Chunker
- **BULLET_PATTERN fallback (1.7):** 4th-level fallback in
`resolveTitleLevels` (`title.go`) detects bullet/numbered-list patterns
(Chinese legal, numbering, English) when outline + regex levels produce
only bodyLevel. Guarded by `allBodyLevel` to never override existing
structure.
- **Tag/One chunker fields (1.8):** `tag.go` sets `TopInt` from source
row index; `one.go` preserves `Positions`/`PDFPositions` from source
items. TSV multi-line RowNum fix: tracks `contentStart` for correct row
attribution.
- **Overlapped_percent normalization (2.6):**
`NormalizeOverlappedPercent` in `schema/chunker.go` mirrors Python
`common/float_utils.py:50-58` — accepts `[0,1)` fraction or `[0,90]`
percent, normalizes to canonical `[0,90]`.
- **Paragraph splitting (2.7):** aligned to Python flow `naive_merge` —
`CRLF` normalization, `splitKeepingDelimiter` preserves sentence
delimiters, single-section merge with token-budget-governed chunking.
- **chunk order:** sort by reading order
### Tokenizer
- **Phantom chunk filtering (Omission 2):** `isPhantomChunk` + filter
loop in `chunksFromTokenizerUpstream` skips zero-value ChunkDocs.
- **Batch size env var (Omission 3):** `embeddingBatchSize()` reads
`TOKENIZER_EMBEDDING_BATCH_SIZE`, defaults to 16.
- **Summary empty check (Diff 5):** `TrimSpace(s) != ""` → `s != ""`,
matching Python truthy check.
- **chunk_order_int all paths (Diff 8):** set unconditionally before
full_text/embedding branching.
- **Timeout default (Diff 10):** `600s` → `60s`, matching Python
`@timeout(60)`.
- **Small maxTokens truncation (Diff 14):** `truncateForEmbedding`
returns `""` when `maxTokens <= 10`, matching Python.
### Code review fixes
- Semaphore acquire moved before goroutine in `pdf_vision_dispatch.go`
(concurrency control)
- Context propagation in `pptx_tcadp.go` (cancellation support)
- Test resolver leak fix in `media_dispatch_test.go` (defer restore)
- Migration history comments removed per AGENTS.md
## Test plan
```
bash build.sh --test ./internal/parser/parser/... ./internal/ingestion/component/...
```
## Notes
- Migration diff tracking: `docs/migration_python_go_diff.md`
- Remaining gaps: Extractor component only (21 items)
96 lines
3.2 KiB
Go
96 lines
3.2 KiB
Go
package parser
|
|
|
|
import (
|
|
"context"
|
|
"encoding/base64"
|
|
"encoding/json"
|
|
"fmt"
|
|
"io"
|
|
"net/http"
|
|
"os"
|
|
"strings"
|
|
|
|
models "ragflow/internal/entity/models"
|
|
)
|
|
|
|
// parsePresentationWithTCADP sends binary presentation (PPTX/PPT) data
|
|
// to the TCADP cloud reconstruction service and returns the structured
|
|
// parse result. Mirrors the spreadsheet-family parseSpreadsheetWithTCADP
|
|
// in xls_tcadp.go
|
|
func parsePresentationWithTCADP(ctx context.Context, filename string, data []byte, fileType string,
|
|
tcadpAPIServer, tcadpAPIKey, tableResultType, markdownImageResponseType string,
|
|
outputFormat string,
|
|
) ParseResult {
|
|
if len(data) == 0 {
|
|
return emptyPDFResult(filename)
|
|
}
|
|
baseURL := strings.TrimSpace(tcadpAPIServer)
|
|
if baseURL == "" {
|
|
baseURL = strings.TrimSpace(os.Getenv("TCADP_APISERVER"))
|
|
}
|
|
if baseURL == "" {
|
|
return ParseResult{Err: fmt.Errorf("parser: TCADP requires tcadp_apiserver or TCADP_APISERVER")}
|
|
}
|
|
apiKey := strings.TrimSpace(tcadpAPIKey)
|
|
if apiKey == "" {
|
|
apiKey = strings.TrimSpace(os.Getenv("TCADP_API_KEY"))
|
|
}
|
|
requestBody := map[string]any{
|
|
"file_type": fileType,
|
|
"file_base64": base64.StdEncoding.EncodeToString(data),
|
|
"file_start_page_number": 1,
|
|
"file_end_page_number": 1000,
|
|
"config": map[string]any{
|
|
"TableResultType": tableResultType,
|
|
"MarkdownImageResponseType": markdownImageResponseType,
|
|
},
|
|
}
|
|
resp, err := models.PostJSONRequest(ctx, models.NewDriverHTTPClient(),
|
|
strings.TrimRight(baseURL, "/")+"/reconstruct_document", bearer(apiKey), requestBody)
|
|
if err != nil {
|
|
return ParseResult{Err: fmt.Errorf("parser: TCADP submit: %w", err)}
|
|
}
|
|
defer resp.Body.Close()
|
|
raw, err := io.ReadAll(resp.Body)
|
|
if err != nil {
|
|
return ParseResult{Err: fmt.Errorf("parser: TCADP read submit: %w", err)}
|
|
}
|
|
if resp.StatusCode >= 300 {
|
|
return ParseResult{Err: fmt.Errorf("parser: TCADP HTTP %d: %s", resp.StatusCode, string(raw))}
|
|
}
|
|
var payload struct {
|
|
DocumentRecognizeResultURL string `json:"DocumentRecognizeResultUrl"`
|
|
}
|
|
if err := json.Unmarshal(raw, &payload); err != nil {
|
|
return ParseResult{Err: fmt.Errorf("parser: TCADP decode submit: %w", err)}
|
|
}
|
|
if payload.DocumentRecognizeResultURL == "" {
|
|
return ParseResult{Err: fmt.Errorf("parser: TCADP returned no DocumentRecognizeResultUrl")}
|
|
}
|
|
downloadReq, err := http.NewRequestWithContext(ctx, http.MethodGet,
|
|
payload.DocumentRecognizeResultURL, nil)
|
|
if err != nil {
|
|
return ParseResult{Err: fmt.Errorf("parser: TCADP download request: %w", err)}
|
|
}
|
|
if auth := bearer(apiKey); auth != "" {
|
|
downloadReq.Header.Set("Authorization", auth)
|
|
}
|
|
downloadResp, err := models.NewDriverHTTPClient().Do(downloadReq)
|
|
if err != nil {
|
|
return ParseResult{Err: fmt.Errorf("parser: TCADP download: %w", err)}
|
|
}
|
|
defer downloadResp.Body.Close()
|
|
zipBytes, err := io.ReadAll(downloadResp.Body)
|
|
if err != nil {
|
|
return ParseResult{Err: fmt.Errorf("parser: TCADP read zip: %w", err)}
|
|
}
|
|
if downloadResp.StatusCode >= 300 {
|
|
return ParseResult{Err: fmt.Errorf("parser: TCADP download HTTP %d: %s", downloadResp.StatusCode, string(zipBytes))}
|
|
}
|
|
items, pageCount, err := tcadpItemsFromZip(zipBytes)
|
|
if err != nil {
|
|
return ParseResult{Err: err}
|
|
}
|
|
return pdfItemsToResult(filename, items, outputFormat, pageCount)
|
|
}
|