mirror of
https://github.com/infiniflow/ragflow.git
synced 2026-07-30 20:49:21 +08:00
## Summary Closes two Chunker migration gaps documented in `docs/migration_python_go_diff.md` (diffs **2.6** and **1.7**), improving parity with the Python ingestion pipeline. This is the code portion of commit `261e1fd0b` on branch `fix/batch-3-4`; the migration document itself is tracked separately (untracked in this commit). ### Diff 2.6 — `overlapped_percent` missing Python normalization Python's `normalize_overlapped_percent` (`common/float_utils.py:50-58`) accepts a `[0,1)` fraction (the flow-canvas UI validates `[0,1)`), multiplies it by 100, `int()`-truncates, and clamps to `[0,90]`. Go previously only accepted a raw `[0,90]` percentage and **rejected** out-of-range input, so a Python config passing `0.1` (meaning 10%) silently produced ~0% overlap. Added `normalizeOverlappedPercent` (`internal/ingestion/component/chunker/common.go`) mirroring the Python helper: - parses numbers and numeric strings (mirrors Python `float()`; bad / `NaN` / `Inf` → `0`), - `0 < v < 1 → v *= 100`, - `int()` truncation, - clamps to `[0,90]`. Wired into `tokenChunkerParam.Update` (`token.go`); the merge math (`token.go:705`, `(100-x)/100`) already matched Python. `TokenChunkerParam.Validate` (`schema/chunker.go`) now only guards direct struct construction. Tests: `TestNormalizeOverlappedPercent`, extended `TestTokenChunker_NewAcceptsPythonOverlappedRange` (adds fraction/clamp inputs), new `TestTokenChunker_NormalizesOverlappedPercent`. The existing reject-test cases for `<0` / `>90` were removed because they are now normalized/clamped (Python parity). ### Diff 1.7 — missing `BULLET_PATTERN` title-level fallback `resolveTitleLevels` (`title.go`) now applies a 4th-level fallback: when outline + regex + layout all yield body level, `bulletsCategory` selects the best-matching bullet-pattern group (Chinese legal / numbering / Chinese numbering / English legal — mirroring `rag/nlp/__init__.py:258-320`) and assigns structural levels. Guarded by `allBodyLevel` so it never overrides an existing outline/regex level. Tests: `TestResolveTitleLevels_BulletFallback` (4 subtests). ### Incidental test adjustments included in the commit - `token_batch1_test.go`: overlap input changed `0.3` → `30.0` to reflect the post-normalization 30% semantics. - `real_consumer_test.go`: updated `LoadFromIngestionTask(task)` → `LoadFromIngestionTask(ctx, task)` for the new context-first signature. ## Verification `bash build.sh --test ./internal/ingestion/component/...` — chunker + schema suites pass, no regression (CGO build). ## Migration doc reference `docs/migration_python_go_diff.md` §Chunker 1.7 and 2.6 are marked **Fixed** for these changes.
237 lines
7.5 KiB
Go
237 lines
7.5 KiB
Go
//
|
|
// Copyright 2026 The InfiniFlow Authors. All Rights Reserved.
|
|
//
|
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
// you may not use this file except in compliance with the License.
|
|
// You may obtain a copy of the License at
|
|
//
|
|
// http://www.apache.org/licenses/LICENSE-2.0
|
|
//
|
|
// Unless required by applicable law or agreed to in writing, software
|
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
// See the License for the specific language governing permissions and
|
|
// limitations under the License.
|
|
//
|
|
|
|
package chunker
|
|
|
|
import (
|
|
"fmt"
|
|
"regexp"
|
|
"sort"
|
|
"strings"
|
|
|
|
"ragflow/internal/agent/runtime"
|
|
"ragflow/internal/ingestion/component/schema"
|
|
"ragflow/internal/tokenizer"
|
|
)
|
|
|
|
// newChunkerByName dispatches the DSL name to a typed constructor.
|
|
// Centralised here so each chunker file only needs an init() that
|
|
// declares its registered name (see register.go). The returned
|
|
// runtime.Component interface is satisfied directly by each
|
|
// constructor (NewTokenChunker etc.) — no intermediate assertion
|
|
// is needed.
|
|
func newChunkerByName(name string, params map[string]any) (runtime.Component, error) {
|
|
switch name {
|
|
case ComponentNameTokenChunker:
|
|
return NewTokenChunker(params)
|
|
case ComponentNameTitleChunker:
|
|
return NewTitleChunker(params)
|
|
case ComponentNameGroupTitleChunker:
|
|
return NewGroupTitleChunker(params)
|
|
case ComponentNameHierarchyTitleChunker:
|
|
return NewHierarchyTitleChunker(params)
|
|
case ComponentNameQAChunker:
|
|
return NewQAChunker(params)
|
|
case ComponentNameOneChunker:
|
|
return NewOneChunker(params)
|
|
case ComponentNameTagChunker:
|
|
return NewTagChunker(params)
|
|
case ComponentNameTableChunker:
|
|
return NewTableChunker(params)
|
|
case ComponentNamePresentationChunker:
|
|
return NewPresentationChunker(params)
|
|
default:
|
|
return nil, fmt.Errorf("chunker: unknown component %q", name)
|
|
}
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// numeric / list conversion helpers (shared across chunker variants)
|
|
// ---------------------------------------------------------------------------
|
|
|
|
func stringListFromAny(in []any) []string {
|
|
out := make([]string, 0, len(in))
|
|
for _, x := range in {
|
|
if s, ok := x.(string); ok && s != "" {
|
|
out = append(out, s)
|
|
}
|
|
}
|
|
return out
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// regex / split helpers
|
|
// ---------------------------------------------------------------------------
|
|
|
|
// compileDelimPattern joins all delimiter entries into a single
|
|
// alternation. Entries wrapped in backticks are treated as regex
|
|
// literals and regex-escaped; plain strings are simply regex-escaped.
|
|
// Longer patterns win (matches python `sorted(set, key=len, reverse=True)`).
|
|
// Mirrors Python _compile_delimiter_pattern: only backtick-wrapped delimiters
|
|
// produce an active pattern. Plain delimiters are not compiled — they are only
|
|
// used by naive_merge / mergeByTokenSize for sentence-level splitting when no
|
|
// active pattern exists.
|
|
func compileDelimPattern(delims []string) *regexp.Regexp {
|
|
var custom []string
|
|
for _, d := range delims {
|
|
if d == "" {
|
|
continue
|
|
}
|
|
if strings.HasPrefix(d, "`") && strings.HasSuffix(d, "`") && len(d) >= 2 {
|
|
custom = append(custom, regexp.QuoteMeta(d[1:len(d)-1]))
|
|
}
|
|
}
|
|
if len(custom) == 0 {
|
|
return nil
|
|
}
|
|
sort.SliceStable(custom, func(i, j int) bool { return len(custom[i]) > len(custom[j]) })
|
|
return regexp.MustCompile(strings.Join(custom, "|"))
|
|
}
|
|
|
|
// splitKeepingDelim mirrors Python token_chunker._split_text_by_pattern
|
|
// (token_chunker.py:79-94): re.split with a captured delimiter group yields
|
|
// [text, delim, text, delim, ...]; each delimiter is glued to the END of the
|
|
// preceding text segment, so it never surfaces as a standalone chunk. A
|
|
// delimiter with no preceding text (a leading delimiter or one adjacent to
|
|
// another) is dropped together with the empty segment, matching Python's
|
|
// `if not chunk: continue`.
|
|
func splitKeepingDelim(text string, pattern *regexp.Regexp) []string {
|
|
if pattern == nil {
|
|
return []string{text}
|
|
}
|
|
idxs := pattern.FindAllStringIndex(text, -1)
|
|
if len(idxs) == 0 {
|
|
return []string{text}
|
|
}
|
|
var out []string
|
|
cursor := 0
|
|
for _, idx := range idxs {
|
|
start, end := idx[0], idx[1]
|
|
if start == cursor {
|
|
cursor = end
|
|
continue
|
|
}
|
|
out = append(out, text[cursor:end])
|
|
cursor = end
|
|
}
|
|
if cursor < len(text) {
|
|
out = append(out, text[cursor:])
|
|
}
|
|
return out
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// chunk-doc helpers
|
|
// ---------------------------------------------------------------------------
|
|
|
|
// itemText returns the text payload from a JSON-style chunk item,
|
|
// preferring "text", then "content_with_weight".
|
|
func itemText(it schema.ChunkDoc) (string, bool) {
|
|
if it.Text != "" {
|
|
return it.Text, true
|
|
}
|
|
if it.ContentWithWeight != "" {
|
|
return it.ContentWithWeight, true
|
|
}
|
|
return "", false
|
|
}
|
|
|
|
// itemDocType mirrors _build_json_chunks's type derivation.
|
|
func itemDocType(it schema.ChunkDoc) string {
|
|
switch strings.ToLower(strings.TrimSpace(it.DocType)) {
|
|
case "table":
|
|
return "table"
|
|
case "image":
|
|
return "image"
|
|
}
|
|
return "text"
|
|
}
|
|
|
|
// itemTextOrFallback returns the item's preferred text, or "".
|
|
func itemTextOrFallback(it schema.ChunkDoc) string {
|
|
if t, ok := itemText(it); ok {
|
|
return t
|
|
}
|
|
return ""
|
|
}
|
|
|
|
// tokenizeStr is the shared NumTokensFromString wrapper used by
|
|
// Table/Image context attachment. Lives here so we can centrally
|
|
// swizzle the count strategy in one place if needed.
|
|
func tokenizeStr(s string) int { return tokenizer.NumTokensFromString(s) }
|
|
|
|
// toString normalises a chunk-map field to a string. Empty strings
|
|
// for missing fields.
|
|
func toString(v any) string {
|
|
if v == nil {
|
|
return ""
|
|
}
|
|
if s, ok := v.(string); ok {
|
|
return s
|
|
}
|
|
return ""
|
|
}
|
|
|
|
// emptyOutputs returns the canonical no-chunks payload.
|
|
func emptyOutputs() map[string]any {
|
|
return map[string]any{
|
|
"output_format": "chunks",
|
|
"chunks": []map[string]any{},
|
|
}
|
|
}
|
|
|
|
func emptyChunkDocs() []schema.ChunkDoc { return []schema.ChunkDoc{} }
|
|
|
|
// chunkOutputs builds the canonical chunker output (output_format="chunks" +
|
|
// chunks). The Go runtime passes only this explicit output to the next node,
|
|
// so the run-level metadata that downstream components still need (e.g.
|
|
// `name` for Tokenizer title embedding, or tenant_id/kb_id for embedding
|
|
// model resolution) is NOT re-emitted here — it lives in the workflow-wide
|
|
// CanvasState.Globals bag (seeded at pipeline start, published by the File
|
|
// component) and read directly from ctx. See runtime.CanvasState.Globals.
|
|
func chunkOutputs(chunks []schema.ChunkDoc) map[string]any {
|
|
return map[string]any{
|
|
"output_format": "chunks",
|
|
"chunks": schema.ChunkDocsToMaps(chunks),
|
|
}
|
|
}
|
|
|
|
// withName returns a shallow copy of inputs with name set, so a component can
|
|
// guarantee `name` is present on the map it forwards to a decode step without
|
|
// mutating the caller's snapshot.
|
|
func withName(inputs map[string]any, name string) map[string]any {
|
|
cp := make(map[string]any, len(inputs)+1)
|
|
for k, v := range inputs {
|
|
cp[k] = v
|
|
}
|
|
cp["name"] = name
|
|
return cp
|
|
}
|
|
|
|
// cloneInputs returns a shallow copy of m with room for one extra key. Used to
|
|
// inject the Globals-resolved `name` into the decode input without mutating
|
|
// the caller's input snapshot.
|
|
func cloneInputs(m map[string]any) map[string]any {
|
|
if m == nil {
|
|
return map[string]any{}
|
|
}
|
|
cp := make(map[string]any, len(m)+1)
|
|
for k, v := range m {
|
|
cp[k] = v
|
|
}
|
|
return cp
|
|
}
|