mirror of
https://github.com/infiniflow/ragflow.git
synced 2026-07-25 09:53:29 +08:00
Feat(ingestion): align image to MinIO upload, unify chunk-id computation and add PPT parsing support (#17111)
## Summary Align the Go ingestion pipeline with Python's `image` → `img_id` persistence semantics, and unify the chunk-id computation across all paths. ### Changes **1. Image upload at chunker stage ** - Add `ImageUploader` type and `DefaultImageUploader` in `internal/ingestion/component/image_uploader.go` — the write-side counterpart to `FetchBinary`, storing raw image bytes at `(bucket=kbID, key=chunkID)`, no re-encoding. - Add `uploadOneImage` — pure upload primitive (bytes in, `img_id` out), does not touch chunk maps. - Add `uploadChunkImages` / `uploadChunkImage` — caller-side helper: decodes `image` from a chunk, uploads bytes, writes `ck["img_id"]`, `delete(ck,"image")` , bounded by a process-wide semaphore (default 10, env `MAX_CONCURRENT_MINIO`). - Wire via `imageUploadDecorator` in `register.go`: every chunker runs the upload pass at invocation time, writing `ck["id"]` before upload and dropping image bytes right after — peak memory = single chunk image lifetime. **2. Unify chunk-id computation** - Consolidate three separate id-computation paths (`component.ChunkID`, `task.ChunkID`, inline `FormatUint` in API) into one: `common.ChunkID(docID, text string)`, using `%016x` + `xxhash.Sum64String(text+docID)` (matching Python `hexdigest()`). - The chunker decorator writes `ck["id"]` via `common.ChunkID`; the persist stage (`ProcessChunksForPipeline`) falls back to the same function (`if !exists id`). - The API AddChunk path now also calls `common.ChunkID` instead of the divergent `FormatUint(xxhash.Sum64(...))` — fixing a pre-existing inconsistency. - Delete `internal/ingestion/component/chunk_id.go` and `internal/ingestion/task/chunk_builder.go` (both were pure forwarding shells). **3. Preserve `img_id` (never deleted)** - `img_id` is a persistent index field (Infinity, OB) and the only consumer-side reference for image retrieval; it is NEVER removed from the chunk map. Only `image` (raw data URL) is dropped after upload. **4. PPT parser support** Previously PPT parsing failed. Add support to parse. ### Key design decisions | Decision | Choice | |----------|--------| | Upload timing | Chunker stage (not persist), so image bytes are dropped immediately — bounds peak memory to one chunk image | | Upload concurrency | Process-wide semaphore, default 10 (matches Python `minio_limiter`), env `MAX_CONCURRENT_MINIO` | | Image encoding | Store as-is, no JPEG re-encoding (unlike Python) | | `img_id` format | `"<kb_id>-<chunk_id>"` — matches Python task_executor path | | id function | Single `common.ChunkID(docID, text)`, concatenation `text+docID` inside hash (matching Python) | | `removeInternalChunkFields` | Retains `delete(ck,"image")` as defensive fallback for non-chunker paths | ### Files touched | File | Change | |------|--------| | `internal/common/format.go` | Add `ChunkID(docID, text)` | | `internal/common/format_test.go` | Add ChunkID golden-value test | | `internal/ingestion/component/image_uploader.go` | Add `ImageUploader` type + `DefaultImageUploader` | | `internal/ingestion/component/chunker/image_upload.go` | Add `uploadOneImage`, `uploadChunkImages`, `uploadChunkImage`, `decodeChunkImage`, semaphore | | `internal/ingestion/component/chunker/image_upload_test.go` | Tests: upload/drop, skip, no-image, concurrency, missing-id error | | `internal/ingestion/component/chunker/register.go` | Add `imageUploadDecorator` (writes `ck["id"]`, runs upload) | | `internal/ingestion/task/chunk_process.go` | Use `common.ChunkID` for persist fallback | | `internal/service/chunk/chunk.go` | Use `common.ChunkID` instead of `FormatUint` | | `internal/ingestion/component/chunk_id.go` | **Deleted** (moved to `common`) | | `internal/ingestion/task/chunk_builder.go` | **Deleted** (shell, no callers left) | | `internal/ingestion/task/chunk_builder_test.go` | **Deleted** (test migrated to `common/format_test.go`) | ### Verification ``` bash build.sh --test ./internal/service/chunk/... ./internal/common/... ./internal/ingestion/component/... ./internal/ingestion/task/... → ok service/chunk / common / component / chunker / schema / task ```
This commit is contained in:
@@ -29,13 +29,10 @@ func (p *PPTParser) String() string {
|
||||
}
|
||||
|
||||
// ParseWithResult delegates to PPTXParser's structured output
|
||||
// for the legacy PPT format. The two file families differ only
|
||||
// in the binary container; the python parser.py:slides branch
|
||||
// treats them uniformly.
|
||||
// for the legacy PPT format using the "ppt" container format
|
||||
// hint (OLE binary). The two file families differ only in the
|
||||
// binary container; the python parser.py:slides branch treats
|
||||
// them uniformly.
|
||||
func (p *PPTParser) ParseWithResult(filename string, data []byte) ParseResult {
|
||||
res := NewPPTXParser().ParseWithResult(filename, data)
|
||||
if res.File != nil {
|
||||
res.File["format"] = "ppt"
|
||||
}
|
||||
return res
|
||||
return (&PPTXParser{format: "ppt"}).ParseWithResult(filename, data)
|
||||
}
|
||||
|
||||
76
internal/parser/parser/ppt_parser_cgo_test.go
Normal file
76
internal/parser/parser/ppt_parser_cgo_test.go
Normal file
@@ -0,0 +1,76 @@
|
||||
//go:build cgo
|
||||
|
||||
package parser
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
officeOxide "github.com/yfedoseev/office_oxide/go"
|
||||
)
|
||||
|
||||
// TestPPTXParser_FormatField verifies the format field wiring:
|
||||
// NewPPTXParser() defaults to "pptx", and an explicit "ppt" can be set.
|
||||
func TestPPTXParser_FormatField(t *testing.T) {
|
||||
p := NewPPTXParser()
|
||||
if p.format != "pptx" {
|
||||
t.Errorf("NewPPTXParser().format = %q, want %q", p.format, "pptx")
|
||||
}
|
||||
p2 := &PPTXParser{format: "ppt"}
|
||||
if p2.format != "ppt" {
|
||||
t.Errorf("explicit PPTXParser{format: \"ppt\"}.format = %q, want %q", p2.format, "ppt")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPPTXParser_ParseWithResult_CGO verifies that PPTXParser can
|
||||
// parse a programmatically generated PPTX document into per-slide
|
||||
// JSON items. Uses office_oxide's own PptxWriter to produce the
|
||||
// test data so no external file is needed.
|
||||
func TestPPTXParser_ParseWithResult_CGO(t *testing.T) {
|
||||
p := NewPPTXParser()
|
||||
data := buildPPTX(t, "Hello World")
|
||||
res := p.ParseWithResult("test.pptx", data)
|
||||
if res.Err != nil {
|
||||
t.Fatalf("ParseWithResult: %v", res.Err)
|
||||
}
|
||||
if res.OutputFormat != "json" {
|
||||
t.Errorf("OutputFormat = %q, want %q", res.OutputFormat, "json")
|
||||
}
|
||||
if got := res.File["format"]; got != "pptx" {
|
||||
t.Errorf("File[format] = %v, want %q", got, "pptx")
|
||||
}
|
||||
if len(res.JSON) == 0 {
|
||||
t.Fatal("JSON items is empty; expected at least one slide")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPPTParser_ParseWithResult_CGO verifies that PPTParser
|
||||
// delegates correctly to PPTXParser{format:"ppt"} and produces
|
||||
// output with File["format"] = "ppt".
|
||||
func TestPPTParser_ParseWithResult_CGO(t *testing.T) {
|
||||
p := NewPPTParser()
|
||||
// Use PPTX content — office_oxide may reject it with format="ppt"
|
||||
// hint (expects OLE binary). When it does, skip gracefully; when
|
||||
// it succeeds, verify the metadata contract.
|
||||
data := buildPPTX(t, "Hello")
|
||||
res := p.ParseWithResult("test.ppt", data)
|
||||
if res.Err != nil {
|
||||
t.Skip("PPTParser with PPTX data (expected maybe to fail):", res.Err)
|
||||
}
|
||||
if got := res.File["format"]; got != "ppt" {
|
||||
t.Errorf("File[format] = %v, want %q", got, "ppt")
|
||||
}
|
||||
}
|
||||
|
||||
// buildPPTX creates a minimal valid PPTX document with one slide
|
||||
// containing the given text, using office_oxide's PptxWriter.
|
||||
func buildPPTX(t *testing.T, text string) []byte {
|
||||
t.Helper()
|
||||
w := officeOxide.NewPptxWriter()
|
||||
slide := w.AddSlide()
|
||||
w.SetSlideTitle(slide, text)
|
||||
data, err := w.ToBytes()
|
||||
if err != nil {
|
||||
t.Fatalf("PptxWriter.ToBytes: %v", err)
|
||||
}
|
||||
return data
|
||||
}
|
||||
@@ -25,10 +25,17 @@ import (
|
||||
officeOxide "github.com/yfedoseev/office_oxide/go"
|
||||
)
|
||||
|
||||
type PPTXParser struct{}
|
||||
// PPTXParser parses both .pptx (OOXML) and .ppt (OLE binary)
|
||||
// files via the office_oxide backend. The format field controls
|
||||
// the container format passed to OpenFromBytes — "pptx" for
|
||||
// ZIP-based OOXML presentations and "ppt" for the legacy binary
|
||||
// OLE format.
|
||||
type PPTXParser struct {
|
||||
format string
|
||||
}
|
||||
|
||||
func NewPPTXParser() *PPTXParser {
|
||||
return &PPTXParser{}
|
||||
return &PPTXParser{format: "pptx"}
|
||||
}
|
||||
|
||||
func (p *PPTXParser) String() string {
|
||||
@@ -39,9 +46,9 @@ func (p *PPTXParser) String() string {
|
||||
// plain text. Mirrors the python parser.py:slides branch which
|
||||
// forces output_format="json" for the slide family.
|
||||
func (p *PPTXParser) ParseWithResult(filename string, data []byte) ParseResult {
|
||||
doc, err := officeOxide.OpenFromBytes(data, "pptx")
|
||||
doc, err := officeOxide.OpenFromBytes(data, p.format)
|
||||
if err != nil {
|
||||
return ParseResult{Err: fmt.Errorf("pptx open: %w", err)}
|
||||
return ParseResult{Err: fmt.Errorf("presentation open: %w", err)}
|
||||
}
|
||||
defer doc.Close()
|
||||
|
||||
@@ -51,7 +58,7 @@ func (p *PPTXParser) ParseWithResult(filename string, data []byte) ParseResult {
|
||||
}
|
||||
|
||||
// Split on form-feed (the python TxtParser convention used by
|
||||
// ragflow's slide parser) — each block becomes a JSON item.
|
||||
// RAGFlow's slide parser) — each block becomes a JSON item.
|
||||
var items []map[string]any
|
||||
for i, raw := range strings.Split(text, "\f") {
|
||||
trimmed := strings.TrimSpace(raw)
|
||||
@@ -70,7 +77,7 @@ func (p *PPTXParser) ParseWithResult(filename string, data []byte) ParseResult {
|
||||
|
||||
return ParseResult{
|
||||
OutputFormat: "json",
|
||||
File: map[string]any{"name": filename, "format": "pptx"},
|
||||
File: map[string]any{"name": filename, "format": p.format},
|
||||
JSON: items,
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user