mirror of
https://github.com/infiniflow/ragflow.git
synced 2026-07-24 01:16:43 +08:00
feat(ingestion): mirror Go pipeline progress into the document table;
harden resume guards
- pipeline: bind the owning document via WithDocumentID; after each
TrackProgress event aggregate ingestion_task_log progress and mirror
progress/run/progress_msg back into the document table, so GET
/api/v1/datasets/{dataset_id}/documents reflects live Go pipeline
progress without a bespoke endpoint.
- canvas: extend the S3 resume guard to reject legacy no-op nodes (e.g.
ExitLoop) so component_total equals the count of progress-reporting
components and the aggregate percent can reach 100%.
- runtime/canvas: route progress through TrackProgress; add interrupt
test coverage (r3_interrupt_test.go).
- dao/entity: add IngestionTask.DocumentID column and AggregateProgress
support used by the mirror; IngestionTaskLog keeps a Checkpoint column
alongside the progress fields.
feat(deepdoc): cache DocAnalyzer inference results in Redis (1h TTL)
- Redis-backed DocAnalyzerCache decorator over inference.Client; cache
key = "ddoc:cache:<method>:" + sha256 of the JPEG-encoded image bytes
(deterministic).
- TTL = 1h; hits skip the inner HTTP call and return cached JSON; inner
errors are not cached.
refactor(deepdoc): align figure cropping with Python cropout + bounded
page caches
- CropSectionByDLA mirrors Python cropout: best-overlap DLA
figure/equation region, fallback to section bbox per page, vertical
concat on gray background.
- sliding-window page-image cache bounds peak memory to the recent
window instead of the whole PDF.
- rename DLADebug -> DLARegions across parser/chunker/tests.
refactor(parser): drop lib_type selector; align NewXxxParser with
NewPDFParser
- remove config["lib_type"] lookup and the libType param/field/switch
from all nine constructors; surface the CGO-required error at
ParseWithResult time instead of construction time; drop resolveLibType,
its test, and the four lib_type constants.
feat(utility): add a reusable workerpool for bounded concurrent
execution
- internal/utility/workerpool.go (+ tests).
refactor: translate Chinese prose comments to English in non-harness Go
files.
chore: upgrade github.com/cloudwego/eino from v0.9.9 to v0.9.12.
77 lines
2.0 KiB
Go
77 lines
2.0 KiB
Go
//go:build cgo
|
|
|
|
//
|
|
// Copyright 2026 The InfiniFlow Authors. All Rights Reserved.
|
|
//
|
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
// you may not use this file except in compliance with the License.
|
|
// You may obtain a copy of the License at
|
|
//
|
|
// http://www.apache.org/licenses/LICENSE-2.0
|
|
//
|
|
// Unless required by applicable law or agreed to in writing, software
|
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
// See the License for the specific language governing permissions and
|
|
// limitations under the License.
|
|
//
|
|
|
|
package parser
|
|
|
|
import (
|
|
"fmt"
|
|
"strings"
|
|
|
|
officeOxide "github.com/yfedoseev/office_oxide/go"
|
|
)
|
|
|
|
type PPTXParser struct{}
|
|
|
|
func NewPPTXParser() *PPTXParser {
|
|
return &PPTXParser{}
|
|
}
|
|
|
|
func (p *PPTXParser) String() string {
|
|
return "PPTXParser"
|
|
}
|
|
|
|
// ParseWithResult emits one JSON item per slide with the slide's
|
|
// plain text. Mirrors the python parser.py:slides branch which
|
|
// forces output_format="json" for the slide family.
|
|
func (p *PPTXParser) ParseWithResult(filename string, data []byte) ParseResult {
|
|
doc, err := officeOxide.OpenFromBytes(data, "pptx")
|
|
if err != nil {
|
|
return ParseResult{Err: fmt.Errorf("pptx open: %w", err)}
|
|
}
|
|
defer doc.Close()
|
|
|
|
text, err := doc.PlainText()
|
|
if err != nil {
|
|
return ParseResult{Err: fmt.Errorf("pptx plain-text: %w", err)}
|
|
}
|
|
|
|
// Split on form-feed (the python TxtParser convention used by
|
|
// ragflow's slide parser) — each block becomes a JSON item.
|
|
var items []map[string]any
|
|
for i, raw := range strings.Split(text, "\f") {
|
|
trimmed := strings.TrimSpace(raw)
|
|
if trimmed == "" {
|
|
continue
|
|
}
|
|
items = append(items, map[string]any{
|
|
"text": trimmed,
|
|
"doc_type_kwd": "text",
|
|
"slide_number": i + 1,
|
|
})
|
|
}
|
|
if items == nil {
|
|
items = []map[string]any{{"text": strings.TrimSpace(text), "doc_type_kwd": "text"}}
|
|
}
|
|
|
|
return ParseResult{
|
|
OutputFormat: "json",
|
|
File: map[string]any{"name": filename, "format": "pptx"},
|
|
JSON: items,
|
|
}
|
|
}
|