2026-04-03 19:26:45 +08:00
|
|
|
|
#
|
|
|
|
|
|
# Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
|
|
|
|
|
|
#
|
|
|
|
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
|
|
|
|
# you may not use this file except in compliance with the License.
|
|
|
|
|
|
# You may obtain a copy of the License at
|
|
|
|
|
|
#
|
|
|
|
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
|
|
|
|
#
|
|
|
|
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
|
|
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
|
|
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
|
|
|
|
# See the License for the specific language governing permissions and
|
|
|
|
|
|
# limitations under the License.
|
|
|
|
|
|
import random
|
|
|
|
|
|
import re
|
|
|
|
|
|
from copy import deepcopy
|
|
|
|
|
|
|
|
|
|
|
|
from common.float_utils import normalize_overlapped_percent
|
|
|
|
|
|
from common.token_utils import num_tokens_from_string
|
|
|
|
|
|
from rag.flow.base import ProcessBase, ProcessParamBase
|
|
|
|
|
|
from rag.flow.chunker.schema import TokenChunkerFromUpstream
|
|
|
|
|
|
from rag.flow.parser.pdf_chunk_metadata import (
|
|
|
|
|
|
PDF_POSITIONS_KEY,
|
|
|
|
|
|
extract_pdf_positions,
|
|
|
|
|
|
finalize_pdf_chunk,
|
|
|
|
|
|
restore_pdf_text_previews,
|
|
|
|
|
|
)
|
|
|
|
|
|
from rag.nlp import naive_merge
|
|
|
|
|
|
|
2026-08-07 21:55:07 +08:00
|
|
|
|
# _TAG_RE matches parser-emitted coordinate tags of the form
|
|
|
|
|
|
# ``@@<page>\t<left>\t<right>\t<top>\t<bottom>##``. Mirrors Go's
|
|
|
|
|
|
# posTagRemove (internal/ingestion/component/chunker/group.go) so the two
|
|
|
|
|
|
# languages strip tags identically.
|
|
|
|
|
|
_TAG_RE = re.compile(r"@@[\t0-9.-]+?##")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def remove_tag(text):
|
|
|
|
|
|
"""Strip ``@@...##`` coordinate tags from text.
|
|
|
|
|
|
|
|
|
|
|
|
Used both when measuring the overlap prefix (so the cut lands on the
|
|
|
|
|
|
tag-free visible text, matching Go's computeOverlapPrefix) and on the
|
|
|
|
|
|
final chunk text (so coordinate tags never reach embedding/index).
|
|
|
|
|
|
"""
|
|
|
|
|
|
return _TAG_RE.sub("", text or "")
|
|
|
|
|
|
|
2026-04-03 19:26:45 +08:00
|
|
|
|
|
|
|
|
|
|
class TokenChunkerParam(ProcessParamBase):
|
|
|
|
|
|
def __init__(self):
|
|
|
|
|
|
super().__init__()
|
2026-08-07 16:11:42 +08:00
|
|
|
|
self.delimiter_mode = "delimiter"
|
2026-04-03 19:26:45 +08:00
|
|
|
|
self.chunk_token_size = 512
|
|
|
|
|
|
self.delimiters = ["\n"]
|
|
|
|
|
|
self.overlapped_percent = 0
|
|
|
|
|
|
self.children_delimiters = []
|
|
|
|
|
|
self.table_context_size = 0
|
|
|
|
|
|
self.image_context_size = 0
|
|
|
|
|
|
|
|
|
|
|
|
def check(self):
|
2026-08-07 16:11:42 +08:00
|
|
|
|
# Backward-compat: "token_size" was removed but is behaviorally identical
|
|
|
|
|
|
# to "delimiter" at runtime (both route through the same code path), so
|
|
|
|
|
|
# accept and coerce it instead of rejecting legacy configs / pre-fix
|
|
|
|
|
|
# frontends. Only genuinely unknown values are rejected.
|
|
|
|
|
|
if self.delimiter_mode == "token_size":
|
|
|
|
|
|
self.delimiter_mode = "delimiter"
|
|
|
|
|
|
self.check_valid_value(self.delimiter_mode, "Delimiter mode abnormal.", ["delimiter", "one"])
|
2026-04-03 19:26:45 +08:00
|
|
|
|
if self.delimiters is None:
|
|
|
|
|
|
self.delimiters = []
|
|
|
|
|
|
elif isinstance(self.delimiters, str):
|
|
|
|
|
|
self.delimiters = [self.delimiters]
|
|
|
|
|
|
else:
|
|
|
|
|
|
self.delimiters = [d for d in self.delimiters if isinstance(d, str)]
|
|
|
|
|
|
self.delimiters = [d for d in self.delimiters if d]
|
|
|
|
|
|
|
|
|
|
|
|
if self.children_delimiters is None:
|
|
|
|
|
|
self.children_delimiters = []
|
|
|
|
|
|
elif isinstance(self.children_delimiters, str):
|
|
|
|
|
|
self.children_delimiters = [self.children_delimiters]
|
|
|
|
|
|
else:
|
|
|
|
|
|
self.children_delimiters = [d for d in self.children_delimiters if isinstance(d, str)]
|
|
|
|
|
|
self.children_delimiters = [d for d in self.children_delimiters if d]
|
|
|
|
|
|
|
|
|
|
|
|
self.check_positive_integer(self.chunk_token_size, "Chunk token size.")
|
|
|
|
|
|
self.check_decimal_float(self.overlapped_percent, "Overlapped percentage: [0, 1)")
|
|
|
|
|
|
self.check_nonnegative_number(self.table_context_size, "Table context size.")
|
|
|
|
|
|
self.check_nonnegative_number(self.image_context_size, "Image context size.")
|
|
|
|
|
|
|
|
|
|
|
|
def get_input_form(self) -> dict[str, dict]:
|
|
|
|
|
|
return {}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _compile_delimiter_pattern(delimiters):
|
|
|
|
|
|
# Build the primary delimiter regex from active delimiters wrapped by backticks.
|
|
|
|
|
|
raw_delimiters = "".join(delimiter for delimiter in (delimiters or []) if delimiter)
|
|
|
|
|
|
custom_delimiters = [m.group(1) for m in re.finditer(r"`([^`]+)`", raw_delimiters)]
|
|
|
|
|
|
if not custom_delimiters:
|
|
|
|
|
|
return ""
|
|
|
|
|
|
return "|".join(re.escape(text) for text in sorted(set(custom_delimiters), key=len, reverse=True))
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _split_text_by_pattern(text, pattern):
|
2026-08-03 17:47:08 +08:00
|
|
|
|
# Split text by the compiled delimiter pattern and discard delimiters.
|
Fix: delimiter is chunk boundary, drop token_size atom-split (OVER_CAP default) (#17808)
## Summary
Fixes a regression introduced by #17203 (strict-cap atom-split) and a
secondary delimiter-handling bug from #17723.
**Root cause:**
- #17203 added `_split_oversized_unit` / `_compute_chunk_update`, which
split oversize units into ≤ token_size pieces. This collapsed
`token_size=1` into 1-token chunks and set the cap at 512, mismatching
the model-layer truncation boundary (embedding ~8191 / rerank
500/4096/8192/2048). Atom-split is unnecessary: oversize units stay
whole and the model layer truncates.
- #17723's delimiter handling dropped consecutive delimiters (`A####B`
-> `A##B`), glued JSON items with `"".join`, ignored
`children_delimiters`, and stripped whitespace delimiters.
## Changes
- New pure helper `merge_paragraphs(paragraphs, token_size, strategy)`
with a `MergeStrategy` enum (`UNDER_CAP` / `OVER_CAP`); **default
`OVER_CAP`**. `UNDER_CAP` is a strict cap (never overflows
`token_size`); `OVER_CAP` greedily accumulates adjacent paragraphs while
the projected total stays within `token_size`, merging one
boundary-overflow paragraph before closing. Oversize paragraphs stand
alone.
- `naive_merge` / `naive_merge_with_images` /
`RAGFlowTxtParser.parser_txt` now use `merge_paragraphs`; atom-split
removed. `naive_merge` / `naive_merge_with_images` always split a
section on the delimiter whenever one is present (even when the section
already fits `token_size`), so delimiter text never leaks into a chunk.
Only the empty-delimiter (size-only) mode skips splitting.
- `token_chunker`: delimiter text is dropped (not stripped); JSON flush
joins buffered items with `"\n"`; `children_delimiters` and
`PDF_POSITIONS_KEY` are preserved on the delimiter path. PDF positions
are now attributed **per segment** — each split chunk carries only the
positions of the item(s) that contributed to it — fixing a leak where
page-N coordinates were attached to page-M chunks and all segments
shared one preview image.
- `test_txt_parser.py` rewritten to assert the new contract (not the old
strict cap); `naive_merge` and delimiter-case-sensitive matrices
updated.
## Contract (refs #17799)
- user specified delimiter = chunk boundary; user specified delimiter
text never enters a chunk.
- `token_size` = soft target + merge strategy; no atom-split.
- Default strategy = `OVER_CAP`; migration can switch to `UNDER_CAP`
(strict cap).
- `OVER_CAP` has no hard cap; the model layer truncates oversize units.
`UNDER_CAP` enforces a strict cap.
## Notes
- Closes the wrong-object revert in #17774 (revert #17723 would
re-introduce delimiter-in-chunk and the strict cap).
- Go-side alignment (`internal/ingestion/component/chunker/token.go`) is
a follow-up PR.
---------
Co-authored-by: CodeBuddy <noreply@tencent.com>
2026-08-05 11:50:07 +08:00
|
|
|
|
# No atom-split is performed; empty segments between consecutive delimiters
|
|
|
|
|
|
# are dropped but whitespace-only segments are preserved (the delimiter is
|
|
|
|
|
|
# the boundary, not stripped away).
|
2026-04-03 19:26:45 +08:00
|
|
|
|
if not pattern:
|
|
|
|
|
|
return [text or ""]
|
|
|
|
|
|
|
|
|
|
|
|
split_texts = re.split(r"(%s)" % pattern, text or "", flags=re.DOTALL)
|
|
|
|
|
|
chunks = []
|
|
|
|
|
|
for i in range(0, len(split_texts), 2):
|
|
|
|
|
|
chunk = split_texts[i]
|
Fix: delimiter is chunk boundary, drop token_size atom-split (OVER_CAP default) (#17808)
## Summary
Fixes a regression introduced by #17203 (strict-cap atom-split) and a
secondary delimiter-handling bug from #17723.
**Root cause:**
- #17203 added `_split_oversized_unit` / `_compute_chunk_update`, which
split oversize units into ≤ token_size pieces. This collapsed
`token_size=1` into 1-token chunks and set the cap at 512, mismatching
the model-layer truncation boundary (embedding ~8191 / rerank
500/4096/8192/2048). Atom-split is unnecessary: oversize units stay
whole and the model layer truncates.
- #17723's delimiter handling dropped consecutive delimiters (`A####B`
-> `A##B`), glued JSON items with `"".join`, ignored
`children_delimiters`, and stripped whitespace delimiters.
## Changes
- New pure helper `merge_paragraphs(paragraphs, token_size, strategy)`
with a `MergeStrategy` enum (`UNDER_CAP` / `OVER_CAP`); **default
`OVER_CAP`**. `UNDER_CAP` is a strict cap (never overflows
`token_size`); `OVER_CAP` greedily accumulates adjacent paragraphs while
the projected total stays within `token_size`, merging one
boundary-overflow paragraph before closing. Oversize paragraphs stand
alone.
- `naive_merge` / `naive_merge_with_images` /
`RAGFlowTxtParser.parser_txt` now use `merge_paragraphs`; atom-split
removed. `naive_merge` / `naive_merge_with_images` always split a
section on the delimiter whenever one is present (even when the section
already fits `token_size`), so delimiter text never leaks into a chunk.
Only the empty-delimiter (size-only) mode skips splitting.
- `token_chunker`: delimiter text is dropped (not stripped); JSON flush
joins buffered items with `"\n"`; `children_delimiters` and
`PDF_POSITIONS_KEY` are preserved on the delimiter path. PDF positions
are now attributed **per segment** — each split chunk carries only the
positions of the item(s) that contributed to it — fixing a leak where
page-N coordinates were attached to page-M chunks and all segments
shared one preview image.
- `test_txt_parser.py` rewritten to assert the new contract (not the old
strict cap); `naive_merge` and delimiter-case-sensitive matrices
updated.
## Contract (refs #17799)
- user specified delimiter = chunk boundary; user specified delimiter
text never enters a chunk.
- `token_size` = soft target + merge strategy; no atom-split.
- Default strategy = `OVER_CAP`; migration can switch to `UNDER_CAP`
(strict cap).
- `OVER_CAP` has no hard cap; the model layer truncates oversize units.
`UNDER_CAP` enforces a strict cap.
## Notes
- Closes the wrong-object revert in #17774 (revert #17723 would
re-introduce delimiter-in-chunk and the strict cap).
- Go-side alignment (`internal/ingestion/component/chunker/token.go`) is
a follow-up PR.
---------
Co-authored-by: CodeBuddy <noreply@tencent.com>
2026-08-05 11:50:07 +08:00
|
|
|
|
if chunk:
|
2026-04-03 19:26:45 +08:00
|
|
|
|
chunks.append(chunk)
|
|
|
|
|
|
return chunks
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _build_json_chunks(json_result, delimiter_pattern):
|
|
|
|
|
|
# Convert upstream JSON items into internal working chunks.
|
|
|
|
|
|
chunks = []
|
|
|
|
|
|
for item in json_result:
|
|
|
|
|
|
doc_type = str(item.get("doc_type_kwd") or "").strip().lower()
|
|
|
|
|
|
if doc_type == "table":
|
|
|
|
|
|
ck_type = "table"
|
|
|
|
|
|
elif doc_type == "image":
|
|
|
|
|
|
ck_type = "image"
|
|
|
|
|
|
else:
|
|
|
|
|
|
ck_type = "text"
|
|
|
|
|
|
|
|
|
|
|
|
text = item.get("text")
|
|
|
|
|
|
if not isinstance(text, str):
|
|
|
|
|
|
text = item.get("content_with_weight")
|
|
|
|
|
|
if not isinstance(text, str):
|
|
|
|
|
|
text = ""
|
|
|
|
|
|
|
|
|
|
|
|
# Keep PDF coordinates as an internal preview field until the final
|
|
|
|
|
|
# output is assembled. This avoids leaking two public coordinate
|
|
|
|
|
|
# formats downstream.
|
|
|
|
|
|
preview_positions = extract_pdf_positions(item)
|
|
|
|
|
|
img_id = item.get("img_id")
|
|
|
|
|
|
|
|
|
|
|
|
if ck_type == "text":
|
|
|
|
|
|
text_segments = _split_text_by_pattern(text, delimiter_pattern) if delimiter_pattern else [text]
|
|
|
|
|
|
for segment in text_segments:
|
|
|
|
|
|
if not segment or not segment.strip():
|
|
|
|
|
|
continue
|
|
|
|
|
|
chunks.append(
|
|
|
|
|
|
{
|
|
|
|
|
|
"text": segment,
|
|
|
|
|
|
"doc_type_kwd": "text",
|
|
|
|
|
|
"ck_type": "text",
|
|
|
|
|
|
PDF_POSITIONS_KEY: deepcopy(preview_positions),
|
|
|
|
|
|
"tk_nums": num_tokens_from_string(segment),
|
|
|
|
|
|
}
|
|
|
|
|
|
)
|
|
|
|
|
|
continue
|
|
|
|
|
|
|
|
|
|
|
|
chunks.append(
|
|
|
|
|
|
{
|
|
|
|
|
|
"text": text or "",
|
|
|
|
|
|
"doc_type_kwd": ck_type,
|
|
|
|
|
|
"ck_type": ck_type,
|
|
|
|
|
|
"img_id": img_id,
|
|
|
|
|
|
PDF_POSITIONS_KEY: deepcopy(preview_positions),
|
|
|
|
|
|
"tk_nums": num_tokens_from_string(text or ""),
|
|
|
|
|
|
"context_above": "",
|
|
|
|
|
|
"context_below": "",
|
|
|
|
|
|
}
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
return chunks
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _take_sentences(text, need_tokens, from_end=False):
|
|
|
|
|
|
# Take text from one side until the target token budget is reached.
|
|
|
|
|
|
split_pat = r"([。!??;!\n]|\. )"
|
|
|
|
|
|
texts = re.split(split_pat, text or "", flags=re.DOTALL)
|
|
|
|
|
|
sentences = []
|
|
|
|
|
|
for i in range(0, len(texts), 2):
|
|
|
|
|
|
sentences.append(texts[i] + (texts[i + 1] if i + 1 < len(texts) else ""))
|
|
|
|
|
|
iterator = reversed(sentences) if from_end else sentences
|
|
|
|
|
|
collected = ""
|
|
|
|
|
|
for sentence in iterator:
|
|
|
|
|
|
collected = sentence + collected if from_end else collected + sentence
|
|
|
|
|
|
if num_tokens_from_string(collected) >= need_tokens:
|
|
|
|
|
|
break
|
|
|
|
|
|
return collected
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _attach_context_to_media_chunks(chunks, table_context_size, image_context_size):
|
|
|
|
|
|
# Add surrounding text to table/image chunks when context windows are enabled.
|
|
|
|
|
|
for i, chunk in enumerate(chunks):
|
|
|
|
|
|
if chunk["ck_type"] not in {"table", "image"}:
|
|
|
|
|
|
continue
|
|
|
|
|
|
|
|
|
|
|
|
context_size = image_context_size if chunk["ck_type"] == "image" else table_context_size
|
|
|
|
|
|
if context_size <= 0:
|
|
|
|
|
|
continue
|
|
|
|
|
|
|
|
|
|
|
|
remain_above = context_size
|
|
|
|
|
|
remain_below = context_size
|
|
|
|
|
|
parts_above = []
|
|
|
|
|
|
parts_below = []
|
|
|
|
|
|
|
|
|
|
|
|
prev = i - 1
|
|
|
|
|
|
while prev >= 0 and remain_above > 0:
|
|
|
|
|
|
prev_chunk = chunks[prev]
|
|
|
|
|
|
if prev_chunk["ck_type"] == "text":
|
|
|
|
|
|
if prev_chunk["tk_nums"] >= remain_above:
|
|
|
|
|
|
parts_above.insert(0, _take_sentences(prev_chunk["text"], remain_above, from_end=True))
|
|
|
|
|
|
remain_above = 0
|
|
|
|
|
|
break
|
|
|
|
|
|
parts_above.insert(0, prev_chunk["text"])
|
|
|
|
|
|
remain_above -= prev_chunk["tk_nums"]
|
|
|
|
|
|
prev -= 1
|
|
|
|
|
|
|
|
|
|
|
|
after = i + 1
|
|
|
|
|
|
while after < len(chunks) and remain_below > 0:
|
|
|
|
|
|
after_chunk = chunks[after]
|
|
|
|
|
|
if after_chunk["ck_type"] == "text":
|
|
|
|
|
|
if after_chunk["tk_nums"] >= remain_below:
|
|
|
|
|
|
parts_below.append(_take_sentences(after_chunk["text"], remain_below))
|
|
|
|
|
|
remain_below = 0
|
|
|
|
|
|
break
|
|
|
|
|
|
parts_below.append(after_chunk["text"])
|
|
|
|
|
|
remain_below -= after_chunk["tk_nums"]
|
|
|
|
|
|
after += 1
|
|
|
|
|
|
|
|
|
|
|
|
chunk["context_above"] = "".join(parts_above)
|
|
|
|
|
|
chunk["context_below"] = "".join(parts_below)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _merge_text_chunks_by_token_size(chunks, chunk_token_size, overlapped_percent):
|
|
|
|
|
|
# Merge adjacent text chunks when delimiter-based splitting is not active.
|
|
|
|
|
|
merged = []
|
|
|
|
|
|
prev_text_idx = -1
|
|
|
|
|
|
threshold = chunk_token_size * (100 - overlapped_percent) / 100.0
|
|
|
|
|
|
|
|
|
|
|
|
for chunk in chunks:
|
|
|
|
|
|
if chunk["ck_type"] != "text":
|
|
|
|
|
|
merged.append(deepcopy(chunk))
|
|
|
|
|
|
prev_text_idx = -1
|
|
|
|
|
|
continue
|
|
|
|
|
|
|
|
|
|
|
|
current = deepcopy(chunk)
|
|
|
|
|
|
should_start_new = prev_text_idx < 0 or merged[prev_text_idx]["tk_nums"] > threshold
|
2026-08-07 21:55:07 +08:00
|
|
|
|
# #17799: an over-budget unit stands alone — never merged into the
|
|
|
|
|
|
# previous chunk. This matches Python naive_merge and the Go
|
|
|
|
|
|
# TokenChunker (all three paths stand the over-budget unit alone), so
|
|
|
|
|
|
# the Python JSON path, Python text path, and Go TokenChunker share one
|
|
|
|
|
|
# contract.
|
|
|
|
|
|
if current["tk_nums"] > chunk_token_size:
|
|
|
|
|
|
should_start_new = True
|
2026-04-03 19:26:45 +08:00
|
|
|
|
if should_start_new:
|
|
|
|
|
|
if prev_text_idx >= 0 and overlapped_percent > 0 and merged[prev_text_idx]["text"]:
|
2026-08-07 21:55:07 +08:00
|
|
|
|
# Mirror Go computeOverlapPrefix: measure the overlap cut on the
|
|
|
|
|
|
# tag-free *visible* text, never on the raw text that still
|
|
|
|
|
|
# carries @@...## coordinate tags. This keeps the overlap prefix
|
|
|
|
|
|
# aligned with Go and prevents a partial tag from leaking into
|
|
|
|
|
|
# the next chunk when the cut would land inside a tag.
|
|
|
|
|
|
visible = remove_tag(merged[prev_text_idx]["text"])
|
|
|
|
|
|
overlap_start = int(len(visible) * (100 - overlapped_percent) / 100.0)
|
|
|
|
|
|
if 0 <= overlap_start < len(visible):
|
|
|
|
|
|
overlap_text = visible[overlap_start:]
|
|
|
|
|
|
else:
|
|
|
|
|
|
overlap_text = ""
|
|
|
|
|
|
current["text"] = overlap_text + current["text"]
|
2026-04-03 19:26:45 +08:00
|
|
|
|
current["tk_nums"] = num_tokens_from_string(current["text"])
|
|
|
|
|
|
merged.append(current)
|
|
|
|
|
|
prev_text_idx = len(merged) - 1
|
|
|
|
|
|
continue
|
|
|
|
|
|
|
|
|
|
|
|
if merged[prev_text_idx]["text"] and current["text"]:
|
|
|
|
|
|
merged[prev_text_idx]["text"] += "\n" + current["text"]
|
|
|
|
|
|
else:
|
|
|
|
|
|
merged[prev_text_idx]["text"] += current["text"]
|
|
|
|
|
|
merged[prev_text_idx][PDF_POSITIONS_KEY].extend(current.get(PDF_POSITIONS_KEY) or [])
|
|
|
|
|
|
merged[prev_text_idx]["tk_nums"] += current["tk_nums"]
|
|
|
|
|
|
|
|
|
|
|
|
return merged
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _finalize_json_chunks(chunks):
|
|
|
|
|
|
# Convert internal chunks into the final token chunker output format.
|
|
|
|
|
|
docs = []
|
|
|
|
|
|
for chunk in chunks:
|
2026-08-07 21:55:07 +08:00
|
|
|
|
# Strip parser coordinate tags from the final text so they never
|
|
|
|
|
|
# reach embedding/index (coordinates already live in the structured
|
|
|
|
|
|
# PDF_POSITIONS_KEY field). Mirrors Go's removeTag at the chunker
|
|
|
|
|
|
# output boundary (token.go:544).
|
|
|
|
|
|
text = remove_tag((chunk.get("context_above") or "") + (chunk.get("text") or "") + (chunk.get("context_below") or ""))
|
2026-04-03 19:26:45 +08:00
|
|
|
|
if not text.strip():
|
|
|
|
|
|
continue
|
|
|
|
|
|
|
|
|
|
|
|
# The internal preview coordinates are converted exactly once into the
|
|
|
|
|
|
# indexed fields consumed downstream.
|
|
|
|
|
|
doc = {
|
|
|
|
|
|
"text": text,
|
|
|
|
|
|
"doc_type_kwd": chunk.get("doc_type_kwd", "text"),
|
|
|
|
|
|
}
|
|
|
|
|
|
if chunk.get(PDF_POSITIONS_KEY):
|
|
|
|
|
|
doc[PDF_POSITIONS_KEY] = deepcopy(chunk[PDF_POSITIONS_KEY])
|
|
|
|
|
|
if chunk.get("mom"):
|
|
|
|
|
|
doc["mom"] = chunk["mom"]
|
|
|
|
|
|
if chunk.get("img_id"):
|
|
|
|
|
|
doc["img_id"] = chunk["img_id"]
|
|
|
|
|
|
docs.append(finalize_pdf_chunk(doc))
|
|
|
|
|
|
|
|
|
|
|
|
return docs
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _split_chunk_docs_by_children(chunks, pattern):
|
|
|
|
|
|
# Apply the secondary children_delimiters split to text chunks only.
|
|
|
|
|
|
if not pattern:
|
|
|
|
|
|
return chunks
|
|
|
|
|
|
|
|
|
|
|
|
docs = []
|
|
|
|
|
|
for chunk in chunks:
|
|
|
|
|
|
if chunk.get("doc_type_kwd", "text") != "text":
|
|
|
|
|
|
docs.append(chunk)
|
|
|
|
|
|
continue
|
|
|
|
|
|
|
|
|
|
|
|
split_texts = _split_text_by_pattern(chunk.get("text", ""), pattern)
|
|
|
|
|
|
|
|
|
|
|
|
mom = chunk.get("text", "")
|
|
|
|
|
|
for text in split_texts:
|
|
|
|
|
|
if not text.strip():
|
|
|
|
|
|
continue
|
|
|
|
|
|
child = deepcopy(chunk)
|
|
|
|
|
|
child["mom"] = mom
|
|
|
|
|
|
child["text"] = text
|
|
|
|
|
|
docs.append(child)
|
|
|
|
|
|
|
|
|
|
|
|
return docs
|
|
|
|
|
|
|
2026-07-03 12:53:39 +08:00
|
|
|
|
|
2026-04-03 19:26:45 +08:00
|
|
|
|
class TokenChunker(ProcessBase):
|
|
|
|
|
|
component_name = "TokenChunker"
|
|
|
|
|
|
|
|
|
|
|
|
async def _invoke(self, **kwargs):
|
|
|
|
|
|
try:
|
|
|
|
|
|
from_upstream = TokenChunkerFromUpstream.model_validate(kwargs)
|
|
|
|
|
|
except Exception as e:
|
|
|
|
|
|
self.set_output("_ERROR", f"Input error: {str(e)}")
|
|
|
|
|
|
return
|
|
|
|
|
|
|
|
|
|
|
|
# Build the primary delimiter regex. If no active custom delimiter exists,
|
|
|
|
|
|
# the token chunker falls back to token-size based merging.
|
|
|
|
|
|
delimiter_pattern = _compile_delimiter_pattern(self._param.delimiters)
|
|
|
|
|
|
custom_pattern = "|".join(re.escape(t) for t in sorted(set(self._param.children_delimiters), key=len, reverse=True))
|
|
|
|
|
|
|
|
|
|
|
|
self.set_output("output_format", "chunks")
|
|
|
|
|
|
self.callback(random.randint(1, 5) / 100.0, "Start to split into chunks.")
|
|
|
|
|
|
overlapped_percent = normalize_overlapped_percent(self._param.overlapped_percent)
|
|
|
|
|
|
if from_upstream.output_format in ["markdown", "text", "html"]:
|
|
|
|
|
|
payload = getattr(from_upstream, f"{from_upstream.output_format}_result") or ""
|
2026-04-10 13:11:22 +08:00
|
|
|
|
if self._param.delimiter_mode == "one":
|
2026-08-07 21:55:07 +08:00
|
|
|
|
# Strip parser coordinate tags so they never reach
|
|
|
|
|
|
# embedding/index (consistent with the JSON merge path).
|
|
|
|
|
|
self.set_output("chunks", [{"text": remove_tag(payload)}] if payload.strip() else [])
|
2026-04-10 13:11:22 +08:00
|
|
|
|
self.callback(1, "Done.")
|
|
|
|
|
|
return
|
2026-08-07 16:11:42 +08:00
|
|
|
|
if delimiter_pattern:
|
2026-08-03 17:47:08 +08:00
|
|
|
|
cks = _split_text_by_pattern(payload, delimiter_pattern)
|
|
|
|
|
|
else:
|
|
|
|
|
|
cks = naive_merge(
|
2026-07-03 12:53:39 +08:00
|
|
|
|
payload,
|
|
|
|
|
|
self._param.chunk_token_size,
|
fix(chunker): honor configured delimiters as soft boundaries in TokenChunker (#17721)
## Summary
`TokenChunker._invoke` discarded the user's configured `delimiters`
whenever no backtick-wrapped delimiter was present: it passed a
hardcoded `""` to `naive_merge`, so the configured delimiters (including
the default `["\n"]`) were never forwarded. `naive_merge` then ignored
the newline sentence boundary and cut chunks **mid-sentence** once the
token budget was exceeded.
## Root cause
`token_chunker.py:326` called `naive_merge(payload, chunk_token_size,
"", overlapped_percent)`. `_compile_delimiter_pattern` intentionally
returns `""` for bare (non-backtick) delimiters — that return value is a
*path selector* (empty → token-budget merge; non-empty → hard
`_split_text_by_pattern` split). The bug was not in that selector but in
the `else` branch, which threw away `self._param.delimiters` instead of
forwarding it.
## Fix
Forward the configured delimiters as a soft boundary:
```python
else naive_merge(
payload,
self._param.chunk_token_size,
"".join(self._param.delimiters),
overlapped_percent,
)
```
`naive_merge` already parses the string via the canonical
`parse_delimiter_field`, so bare and backtick-wrapped delimiters are
honored as soft boundaries while the token budget is still respected.
The path-selection role of `_compile_delimiter_pattern` is untouched:
backtick-wrapped delimiters still select the hard
`_split_text_by_pattern` path; bare delimiters still take the
token-budget merge path. No regression for any previously-working
(wrapped-delimiter) configuration.
## Test
Adds `rag/flow/tests/test_token_chunker_delimiter.py`:
- `test_token_chunker_token_size_mode_does_not_split_sentences` — fails
on the old code (3 sentences cut mid-stream), passes after the fix.
- `test_naive_merge_empty_delimiter_ignores_newline_break` — root-cause
companion asserting `naive_merge("")` cuts while `naive_merge("\n")`
preserves boundaries.
Both tests skip when the tokenizer is unavailable (dead-tokenizer
guard).
## Note
This branch contains only this one-line fix on top of `main`; it is
intentionally independent of the unrelated tokenizer/offline-BPE work.
Co-authored-by: CodeBuddy <noreply@codebuddy.ai>
2026-08-03 15:31:40 +08:00
|
|
|
|
"".join(self._param.delimiters),
|
2026-07-03 12:53:39 +08:00
|
|
|
|
overlapped_percent,
|
|
|
|
|
|
)
|
2026-04-03 19:26:45 +08:00
|
|
|
|
if custom_pattern:
|
|
|
|
|
|
docs = []
|
|
|
|
|
|
for c in cks:
|
|
|
|
|
|
if not c.strip():
|
|
|
|
|
|
continue
|
|
|
|
|
|
for text in _split_text_by_pattern(c, custom_pattern):
|
|
|
|
|
|
if not text.strip():
|
|
|
|
|
|
continue
|
|
|
|
|
|
docs.append({"text": text, "mom": c})
|
|
|
|
|
|
self.set_output("chunks", docs)
|
|
|
|
|
|
else:
|
|
|
|
|
|
self.set_output("chunks", [{"text": c.strip()} for c in cks if c.strip()])
|
|
|
|
|
|
|
|
|
|
|
|
self.callback(1, "Done.")
|
|
|
|
|
|
return
|
|
|
|
|
|
|
|
|
|
|
|
# json
|
Fix: TokenChunker discards TitleChunker chunks when output_format is 'chunks' (#16825)
Fixes #16812
### Problem
In the `rag/flow` ingestion pipeline, when `TitleChunker` feeds
`TokenChunker`, the chapter-aware chunks are silently discarded and the
parser's raw flat json is re-chunked instead.
`TitleChunker` emits `output_format="chunks"` and writes its
chapter-aware output to the `chunks` field
(`rag/flow/chunker/title_chunker/common.py`,
`set_output("output_format", "chunks")`). But `TokenChunker._invoke`
only handles `output_format` in `["markdown", "text", "html"]`, then
falls through to the `# json` path which reads
`from_upstream.json_result`. There is no branch for `"chunks"`, so
`from_upstream.chunks` is never read.
Downstream effects reported in #16812: PageIndex/TOC extraction receives
flat line-level text instead of structured chapter blocks
(incorrect/duplicate/missing chapters), and retrieval quality degrades
because chunks are no longer aligned to document structure.
### Fix
Select the source list based on `output_format`, mirroring the exact
pattern already used in `title_chunker/common.py`:
```python
json_result = (from_upstream.chunks if from_upstream.output_format == "chunks" else from_upstream.json_result) or []
```
`chunks` items share the same dict shape as `json_result` items (both
consumed via `.get("text")`, `.get("doc_type_kwd")`, etc.), so they flow
through the existing token-sizing path unchanged. One-line change, no
behavior change for the `json`/`markdown`/`text`/`html` paths.
### Test
Adds `rag/flow/tests/test_token_chunker.py`, an isolated unit test that
runs the real `TokenChunker._invoke` (heavy deps stubbed; real pydantic
schema used when available) and asserts that with
`output_format="chunks"` the upstream `chunks` are consumed rather than
the raw parser `json`.
Verified RED -> GREEN: the test fails against the current code (reads
the raw json) and passes with the fix.
Signed-off-by: Yash Raj Pandey <yashpn62@gmail.com>
2026-07-13 10:04:56 -04:00
|
|
|
|
json_result = (from_upstream.chunks if from_upstream.output_format == "chunks" else from_upstream.json_result) or []
|
2026-04-10 13:11:22 +08:00
|
|
|
|
if self._param.delimiter_mode == "one":
|
|
|
|
|
|
sections = []
|
|
|
|
|
|
for item in json_result:
|
|
|
|
|
|
text = item.get("text")
|
|
|
|
|
|
if not isinstance(text, str):
|
|
|
|
|
|
text = item.get("content_with_weight")
|
|
|
|
|
|
if isinstance(text, str) and text.strip():
|
2026-08-07 21:55:07 +08:00
|
|
|
|
# Strip parser coordinate tags so they never reach
|
|
|
|
|
|
# embedding/index (consistent with the JSON merge path).
|
|
|
|
|
|
sections.append(remove_tag(text))
|
2026-04-10 13:11:22 +08:00
|
|
|
|
merged_text = "\n".join(sections)
|
|
|
|
|
|
self.set_output("chunks", [{"text": merged_text}] if merged_text.strip() else [])
|
|
|
|
|
|
self.callback(1, "Done.")
|
|
|
|
|
|
return
|
2026-08-03 17:47:08 +08:00
|
|
|
|
|
2026-08-07 16:11:42 +08:00
|
|
|
|
# Both branches start from per-item chunks (no pre-split by the
|
|
|
|
|
|
# delimiter pattern). The delimiter branch splits the buffered text
|
|
|
|
|
|
# stream while preserving per-segment PDF positions; the no-delimiter
|
|
|
|
|
|
# branch merges adjacent text items to chunk_token_size (the removed
|
|
|
|
|
|
# "token_size" behaviour, and a parity match with the Go JSON path).
|
|
|
|
|
|
text_chunks = _build_json_chunks(json_result, "")
|
|
|
|
|
|
|
|
|
|
|
|
if delimiter_pattern:
|
2026-08-03 17:47:08 +08:00
|
|
|
|
chunks = []
|
|
|
|
|
|
text_buffer = []
|
Fix: delimiter is chunk boundary, drop token_size atom-split (OVER_CAP default) (#17808)
## Summary
Fixes a regression introduced by #17203 (strict-cap atom-split) and a
secondary delimiter-handling bug from #17723.
**Root cause:**
- #17203 added `_split_oversized_unit` / `_compute_chunk_update`, which
split oversize units into ≤ token_size pieces. This collapsed
`token_size=1` into 1-token chunks and set the cap at 512, mismatching
the model-layer truncation boundary (embedding ~8191 / rerank
500/4096/8192/2048). Atom-split is unnecessary: oversize units stay
whole and the model layer truncates.
- #17723's delimiter handling dropped consecutive delimiters (`A####B`
-> `A##B`), glued JSON items with `"".join`, ignored
`children_delimiters`, and stripped whitespace delimiters.
## Changes
- New pure helper `merge_paragraphs(paragraphs, token_size, strategy)`
with a `MergeStrategy` enum (`UNDER_CAP` / `OVER_CAP`); **default
`OVER_CAP`**. `UNDER_CAP` is a strict cap (never overflows
`token_size`); `OVER_CAP` greedily accumulates adjacent paragraphs while
the projected total stays within `token_size`, merging one
boundary-overflow paragraph before closing. Oversize paragraphs stand
alone.
- `naive_merge` / `naive_merge_with_images` /
`RAGFlowTxtParser.parser_txt` now use `merge_paragraphs`; atom-split
removed. `naive_merge` / `naive_merge_with_images` always split a
section on the delimiter whenever one is present (even when the section
already fits `token_size`), so delimiter text never leaks into a chunk.
Only the empty-delimiter (size-only) mode skips splitting.
- `token_chunker`: delimiter text is dropped (not stripped); JSON flush
joins buffered items with `"\n"`; `children_delimiters` and
`PDF_POSITIONS_KEY` are preserved on the delimiter path. PDF positions
are now attributed **per segment** — each split chunk carries only the
positions of the item(s) that contributed to it — fixing a leak where
page-N coordinates were attached to page-M chunks and all segments
shared one preview image.
- `test_txt_parser.py` rewritten to assert the new contract (not the old
strict cap); `naive_merge` and delimiter-case-sensitive matrices
updated.
## Contract (refs #17799)
- user specified delimiter = chunk boundary; user specified delimiter
text never enters a chunk.
- `token_size` = soft target + merge strategy; no atom-split.
- Default strategy = `OVER_CAP`; migration can switch to `UNDER_CAP`
(strict cap).
- `OVER_CAP` has no hard cap; the model layer truncates oversize units.
`UNDER_CAP` enforces a strict cap.
## Notes
- Closes the wrong-object revert in #17774 (revert #17723 would
re-introduce delimiter-in-chunk and the strict cap).
- Go-side alignment (`internal/ingestion/component/chunker/token.go`) is
a follow-up PR.
---------
Co-authored-by: CodeBuddy <noreply@tencent.com>
2026-08-05 11:50:07 +08:00
|
|
|
|
text_buffer_pos = []
|
2026-08-03 17:47:08 +08:00
|
|
|
|
|
|
|
|
|
|
def flush_text_buffer():
|
|
|
|
|
|
if not text_buffer:
|
|
|
|
|
|
return
|
Fix: delimiter is chunk boundary, drop token_size atom-split (OVER_CAP default) (#17808)
## Summary
Fixes a regression introduced by #17203 (strict-cap atom-split) and a
secondary delimiter-handling bug from #17723.
**Root cause:**
- #17203 added `_split_oversized_unit` / `_compute_chunk_update`, which
split oversize units into ≤ token_size pieces. This collapsed
`token_size=1` into 1-token chunks and set the cap at 512, mismatching
the model-layer truncation boundary (embedding ~8191 / rerank
500/4096/8192/2048). Atom-split is unnecessary: oversize units stay
whole and the model layer truncates.
- #17723's delimiter handling dropped consecutive delimiters (`A####B`
-> `A##B`), glued JSON items with `"".join`, ignored
`children_delimiters`, and stripped whitespace delimiters.
## Changes
- New pure helper `merge_paragraphs(paragraphs, token_size, strategy)`
with a `MergeStrategy` enum (`UNDER_CAP` / `OVER_CAP`); **default
`OVER_CAP`**. `UNDER_CAP` is a strict cap (never overflows
`token_size`); `OVER_CAP` greedily accumulates adjacent paragraphs while
the projected total stays within `token_size`, merging one
boundary-overflow paragraph before closing. Oversize paragraphs stand
alone.
- `naive_merge` / `naive_merge_with_images` /
`RAGFlowTxtParser.parser_txt` now use `merge_paragraphs`; atom-split
removed. `naive_merge` / `naive_merge_with_images` always split a
section on the delimiter whenever one is present (even when the section
already fits `token_size`), so delimiter text never leaks into a chunk.
Only the empty-delimiter (size-only) mode skips splitting.
- `token_chunker`: delimiter text is dropped (not stripped); JSON flush
joins buffered items with `"\n"`; `children_delimiters` and
`PDF_POSITIONS_KEY` are preserved on the delimiter path. PDF positions
are now attributed **per segment** — each split chunk carries only the
positions of the item(s) that contributed to it — fixing a leak where
page-N coordinates were attached to page-M chunks and all segments
shared one preview image.
- `test_txt_parser.py` rewritten to assert the new contract (not the old
strict cap); `naive_merge` and delimiter-case-sensitive matrices
updated.
## Contract (refs #17799)
- user specified delimiter = chunk boundary; user specified delimiter
text never enters a chunk.
- `token_size` = soft target + merge strategy; no atom-split.
- Default strategy = `OVER_CAP`; migration can switch to `UNDER_CAP`
(strict cap).
- `OVER_CAP` has no hard cap; the model layer truncates oversize units.
`UNDER_CAP` enforces a strict cap.
## Notes
- Closes the wrong-object revert in #17774 (revert #17723 would
re-introduce delimiter-in-chunk and the strict cap).
- Go-side alignment (`internal/ingestion/component/chunker/token.go`) is
a follow-up PR.
---------
Co-authored-by: CodeBuddy <noreply@tencent.com>
2026-08-05 11:50:07 +08:00
|
|
|
|
# Join buffered text items with "\n" so adjacent item text is not
|
|
|
|
|
|
# glued together (e.g. "hello" + "world" must not become "helloworld").
|
|
|
|
|
|
# The delimiter is then applied to the combined text; a segment may
|
|
|
|
|
|
# span across item boundaries (the "\n" glue is not itself a
|
|
|
|
|
|
# delimiter), so each segment carries only the PDF positions of the
|
2026-08-07 16:11:42 +08:00
|
|
|
|
# item(s) that contributed to it -- never the union of every item
|
|
|
|
|
|
# (which previously leaked page-N coordinates into page-M chunks and
|
|
|
|
|
|
# made all segments share one preview image).
|
Fix: delimiter is chunk boundary, drop token_size atom-split (OVER_CAP default) (#17808)
## Summary
Fixes a regression introduced by #17203 (strict-cap atom-split) and a
secondary delimiter-handling bug from #17723.
**Root cause:**
- #17203 added `_split_oversized_unit` / `_compute_chunk_update`, which
split oversize units into ≤ token_size pieces. This collapsed
`token_size=1` into 1-token chunks and set the cap at 512, mismatching
the model-layer truncation boundary (embedding ~8191 / rerank
500/4096/8192/2048). Atom-split is unnecessary: oversize units stay
whole and the model layer truncates.
- #17723's delimiter handling dropped consecutive delimiters (`A####B`
-> `A##B`), glued JSON items with `"".join`, ignored
`children_delimiters`, and stripped whitespace delimiters.
## Changes
- New pure helper `merge_paragraphs(paragraphs, token_size, strategy)`
with a `MergeStrategy` enum (`UNDER_CAP` / `OVER_CAP`); **default
`OVER_CAP`**. `UNDER_CAP` is a strict cap (never overflows
`token_size`); `OVER_CAP` greedily accumulates adjacent paragraphs while
the projected total stays within `token_size`, merging one
boundary-overflow paragraph before closing. Oversize paragraphs stand
alone.
- `naive_merge` / `naive_merge_with_images` /
`RAGFlowTxtParser.parser_txt` now use `merge_paragraphs`; atom-split
removed. `naive_merge` / `naive_merge_with_images` always split a
section on the delimiter whenever one is present (even when the section
already fits `token_size`), so delimiter text never leaks into a chunk.
Only the empty-delimiter (size-only) mode skips splitting.
- `token_chunker`: delimiter text is dropped (not stripped); JSON flush
joins buffered items with `"\n"`; `children_delimiters` and
`PDF_POSITIONS_KEY` are preserved on the delimiter path. PDF positions
are now attributed **per segment** — each split chunk carries only the
positions of the item(s) that contributed to it — fixing a leak where
page-N coordinates were attached to page-M chunks and all segments
shared one preview image.
- `test_txt_parser.py` rewritten to assert the new contract (not the old
strict cap); `naive_merge` and delimiter-case-sensitive matrices
updated.
## Contract (refs #17799)
- user specified delimiter = chunk boundary; user specified delimiter
text never enters a chunk.
- `token_size` = soft target + merge strategy; no atom-split.
- Default strategy = `OVER_CAP`; migration can switch to `UNDER_CAP`
(strict cap).
- `OVER_CAP` has no hard cap; the model layer truncates oversize units.
`UNDER_CAP` enforces a strict cap.
## Notes
- Closes the wrong-object revert in #17774 (revert #17723 would
re-introduce delimiter-in-chunk and the strict cap).
- Go-side alignment (`internal/ingestion/component/chunker/token.go`) is
a follow-up PR.
---------
Co-authored-by: CodeBuddy <noreply@tencent.com>
2026-08-05 11:50:07 +08:00
|
|
|
|
parts = []
|
|
|
|
|
|
item_ranges = [] # (start, end) of each buffered item in combined_text
|
|
|
|
|
|
offset = 0
|
|
|
|
|
|
for text in text_buffer:
|
|
|
|
|
|
start = offset
|
|
|
|
|
|
parts.append(text)
|
|
|
|
|
|
offset += len(text)
|
|
|
|
|
|
item_ranges.append((start, offset))
|
|
|
|
|
|
parts.append("\n")
|
|
|
|
|
|
offset += 1
|
|
|
|
|
|
combined_text = "".join(parts[:-1]) # drop the trailing glue
|
|
|
|
|
|
|
2026-08-07 16:11:42 +08:00
|
|
|
|
raw = re.split(r"(%s)" % delimiter_pattern, combined_text, flags=re.DOTALL)
|
|
|
|
|
|
segments = [] # (text, start, end) within combined_text
|
|
|
|
|
|
pos = 0
|
|
|
|
|
|
for i in range(0, len(raw), 2):
|
|
|
|
|
|
seg = raw[i]
|
|
|
|
|
|
seg_start = pos
|
|
|
|
|
|
seg_end = pos + len(seg)
|
|
|
|
|
|
if seg:
|
|
|
|
|
|
segments.append((seg, seg_start, seg_end))
|
|
|
|
|
|
pos = seg_end
|
|
|
|
|
|
if i + 1 < len(raw):
|
|
|
|
|
|
pos += len(raw[i + 1])
|
Fix: delimiter is chunk boundary, drop token_size atom-split (OVER_CAP default) (#17808)
## Summary
Fixes a regression introduced by #17203 (strict-cap atom-split) and a
secondary delimiter-handling bug from #17723.
**Root cause:**
- #17203 added `_split_oversized_unit` / `_compute_chunk_update`, which
split oversize units into ≤ token_size pieces. This collapsed
`token_size=1` into 1-token chunks and set the cap at 512, mismatching
the model-layer truncation boundary (embedding ~8191 / rerank
500/4096/8192/2048). Atom-split is unnecessary: oversize units stay
whole and the model layer truncates.
- #17723's delimiter handling dropped consecutive delimiters (`A####B`
-> `A##B`), glued JSON items with `"".join`, ignored
`children_delimiters`, and stripped whitespace delimiters.
## Changes
- New pure helper `merge_paragraphs(paragraphs, token_size, strategy)`
with a `MergeStrategy` enum (`UNDER_CAP` / `OVER_CAP`); **default
`OVER_CAP`**. `UNDER_CAP` is a strict cap (never overflows
`token_size`); `OVER_CAP` greedily accumulates adjacent paragraphs while
the projected total stays within `token_size`, merging one
boundary-overflow paragraph before closing. Oversize paragraphs stand
alone.
- `naive_merge` / `naive_merge_with_images` /
`RAGFlowTxtParser.parser_txt` now use `merge_paragraphs`; atom-split
removed. `naive_merge` / `naive_merge_with_images` always split a
section on the delimiter whenever one is present (even when the section
already fits `token_size`), so delimiter text never leaks into a chunk.
Only the empty-delimiter (size-only) mode skips splitting.
- `token_chunker`: delimiter text is dropped (not stripped); JSON flush
joins buffered items with `"\n"`; `children_delimiters` and
`PDF_POSITIONS_KEY` are preserved on the delimiter path. PDF positions
are now attributed **per segment** — each split chunk carries only the
positions of the item(s) that contributed to it — fixing a leak where
page-N coordinates were attached to page-M chunks and all segments
shared one preview image.
- `test_txt_parser.py` rewritten to assert the new contract (not the old
strict cap); `naive_merge` and delimiter-case-sensitive matrices
updated.
## Contract (refs #17799)
- user specified delimiter = chunk boundary; user specified delimiter
text never enters a chunk.
- `token_size` = soft target + merge strategy; no atom-split.
- Default strategy = `OVER_CAP`; migration can switch to `UNDER_CAP`
(strict cap).
- `OVER_CAP` has no hard cap; the model layer truncates oversize units.
`UNDER_CAP` enforces a strict cap.
## Notes
- Closes the wrong-object revert in #17774 (revert #17723 would
re-introduce delimiter-in-chunk and the strict cap).
- Go-side alignment (`internal/ingestion/component/chunker/token.go`) is
a follow-up PR.
---------
Co-authored-by: CodeBuddy <noreply@tencent.com>
2026-08-05 11:50:07 +08:00
|
|
|
|
|
|
|
|
|
|
for text, seg_start, seg_end in segments:
|
|
|
|
|
|
if not text.strip():
|
|
|
|
|
|
continue
|
|
|
|
|
|
seg_pos = []
|
2026-08-07 16:11:42 +08:00
|
|
|
|
for (istart, iend), item_pos in zip(item_ranges, text_buffer_pos, strict=True):
|
Fix: delimiter is chunk boundary, drop token_size atom-split (OVER_CAP default) (#17808)
## Summary
Fixes a regression introduced by #17203 (strict-cap atom-split) and a
secondary delimiter-handling bug from #17723.
**Root cause:**
- #17203 added `_split_oversized_unit` / `_compute_chunk_update`, which
split oversize units into ≤ token_size pieces. This collapsed
`token_size=1` into 1-token chunks and set the cap at 512, mismatching
the model-layer truncation boundary (embedding ~8191 / rerank
500/4096/8192/2048). Atom-split is unnecessary: oversize units stay
whole and the model layer truncates.
- #17723's delimiter handling dropped consecutive delimiters (`A####B`
-> `A##B`), glued JSON items with `"".join`, ignored
`children_delimiters`, and stripped whitespace delimiters.
## Changes
- New pure helper `merge_paragraphs(paragraphs, token_size, strategy)`
with a `MergeStrategy` enum (`UNDER_CAP` / `OVER_CAP`); **default
`OVER_CAP`**. `UNDER_CAP` is a strict cap (never overflows
`token_size`); `OVER_CAP` greedily accumulates adjacent paragraphs while
the projected total stays within `token_size`, merging one
boundary-overflow paragraph before closing. Oversize paragraphs stand
alone.
- `naive_merge` / `naive_merge_with_images` /
`RAGFlowTxtParser.parser_txt` now use `merge_paragraphs`; atom-split
removed. `naive_merge` / `naive_merge_with_images` always split a
section on the delimiter whenever one is present (even when the section
already fits `token_size`), so delimiter text never leaks into a chunk.
Only the empty-delimiter (size-only) mode skips splitting.
- `token_chunker`: delimiter text is dropped (not stripped); JSON flush
joins buffered items with `"\n"`; `children_delimiters` and
`PDF_POSITIONS_KEY` are preserved on the delimiter path. PDF positions
are now attributed **per segment** — each split chunk carries only the
positions of the item(s) that contributed to it — fixing a leak where
page-N coordinates were attached to page-M chunks and all segments
shared one preview image.
- `test_txt_parser.py` rewritten to assert the new contract (not the old
strict cap); `naive_merge` and delimiter-case-sensitive matrices
updated.
## Contract (refs #17799)
- user specified delimiter = chunk boundary; user specified delimiter
text never enters a chunk.
- `token_size` = soft target + merge strategy; no atom-split.
- Default strategy = `OVER_CAP`; migration can switch to `UNDER_CAP`
(strict cap).
- `OVER_CAP` has no hard cap; the model layer truncates oversize units.
`UNDER_CAP` enforces a strict cap.
## Notes
- Closes the wrong-object revert in #17774 (revert #17723 would
re-introduce delimiter-in-chunk and the strict cap).
- Go-side alignment (`internal/ingestion/component/chunker/token.go`) is
a follow-up PR.
---------
Co-authored-by: CodeBuddy <noreply@tencent.com>
2026-08-05 11:50:07 +08:00
|
|
|
|
# A segment overlaps an item when their character ranges
|
|
|
|
|
|
# intersect; collect that item's coordinates.
|
|
|
|
|
|
if seg_start < iend and istart < seg_end:
|
|
|
|
|
|
seg_pos.extend(item_pos or [])
|
|
|
|
|
|
chunks.append(
|
|
|
|
|
|
{
|
|
|
|
|
|
"text": text,
|
|
|
|
|
|
"doc_type_kwd": "text",
|
|
|
|
|
|
"ck_type": "text",
|
|
|
|
|
|
PDF_POSITIONS_KEY: deepcopy(seg_pos),
|
|
|
|
|
|
"tk_nums": num_tokens_from_string(text),
|
|
|
|
|
|
}
|
|
|
|
|
|
)
|
2026-08-03 17:47:08 +08:00
|
|
|
|
text_buffer.clear()
|
Fix: delimiter is chunk boundary, drop token_size atom-split (OVER_CAP default) (#17808)
## Summary
Fixes a regression introduced by #17203 (strict-cap atom-split) and a
secondary delimiter-handling bug from #17723.
**Root cause:**
- #17203 added `_split_oversized_unit` / `_compute_chunk_update`, which
split oversize units into ≤ token_size pieces. This collapsed
`token_size=1` into 1-token chunks and set the cap at 512, mismatching
the model-layer truncation boundary (embedding ~8191 / rerank
500/4096/8192/2048). Atom-split is unnecessary: oversize units stay
whole and the model layer truncates.
- #17723's delimiter handling dropped consecutive delimiters (`A####B`
-> `A##B`), glued JSON items with `"".join`, ignored
`children_delimiters`, and stripped whitespace delimiters.
## Changes
- New pure helper `merge_paragraphs(paragraphs, token_size, strategy)`
with a `MergeStrategy` enum (`UNDER_CAP` / `OVER_CAP`); **default
`OVER_CAP`**. `UNDER_CAP` is a strict cap (never overflows
`token_size`); `OVER_CAP` greedily accumulates adjacent paragraphs while
the projected total stays within `token_size`, merging one
boundary-overflow paragraph before closing. Oversize paragraphs stand
alone.
- `naive_merge` / `naive_merge_with_images` /
`RAGFlowTxtParser.parser_txt` now use `merge_paragraphs`; atom-split
removed. `naive_merge` / `naive_merge_with_images` always split a
section on the delimiter whenever one is present (even when the section
already fits `token_size`), so delimiter text never leaks into a chunk.
Only the empty-delimiter (size-only) mode skips splitting.
- `token_chunker`: delimiter text is dropped (not stripped); JSON flush
joins buffered items with `"\n"`; `children_delimiters` and
`PDF_POSITIONS_KEY` are preserved on the delimiter path. PDF positions
are now attributed **per segment** — each split chunk carries only the
positions of the item(s) that contributed to it — fixing a leak where
page-N coordinates were attached to page-M chunks and all segments
shared one preview image.
- `test_txt_parser.py` rewritten to assert the new contract (not the old
strict cap); `naive_merge` and delimiter-case-sensitive matrices
updated.
## Contract (refs #17799)
- user specified delimiter = chunk boundary; user specified delimiter
text never enters a chunk.
- `token_size` = soft target + merge strategy; no atom-split.
- Default strategy = `OVER_CAP`; migration can switch to `UNDER_CAP`
(strict cap).
- `OVER_CAP` has no hard cap; the model layer truncates oversize units.
`UNDER_CAP` enforces a strict cap.
## Notes
- Closes the wrong-object revert in #17774 (revert #17723 would
re-introduce delimiter-in-chunk and the strict cap).
- Go-side alignment (`internal/ingestion/component/chunker/token.go`) is
a follow-up PR.
---------
Co-authored-by: CodeBuddy <noreply@tencent.com>
2026-08-05 11:50:07 +08:00
|
|
|
|
text_buffer_pos.clear()
|
2026-08-03 17:47:08 +08:00
|
|
|
|
|
|
|
|
|
|
for chunk in text_chunks:
|
|
|
|
|
|
if chunk["ck_type"] == "text":
|
|
|
|
|
|
text_buffer.append(chunk["text"])
|
Fix: delimiter is chunk boundary, drop token_size atom-split (OVER_CAP default) (#17808)
## Summary
Fixes a regression introduced by #17203 (strict-cap atom-split) and a
secondary delimiter-handling bug from #17723.
**Root cause:**
- #17203 added `_split_oversized_unit` / `_compute_chunk_update`, which
split oversize units into ≤ token_size pieces. This collapsed
`token_size=1` into 1-token chunks and set the cap at 512, mismatching
the model-layer truncation boundary (embedding ~8191 / rerank
500/4096/8192/2048). Atom-split is unnecessary: oversize units stay
whole and the model layer truncates.
- #17723's delimiter handling dropped consecutive delimiters (`A####B`
-> `A##B`), glued JSON items with `"".join`, ignored
`children_delimiters`, and stripped whitespace delimiters.
## Changes
- New pure helper `merge_paragraphs(paragraphs, token_size, strategy)`
with a `MergeStrategy` enum (`UNDER_CAP` / `OVER_CAP`); **default
`OVER_CAP`**. `UNDER_CAP` is a strict cap (never overflows
`token_size`); `OVER_CAP` greedily accumulates adjacent paragraphs while
the projected total stays within `token_size`, merging one
boundary-overflow paragraph before closing. Oversize paragraphs stand
alone.
- `naive_merge` / `naive_merge_with_images` /
`RAGFlowTxtParser.parser_txt` now use `merge_paragraphs`; atom-split
removed. `naive_merge` / `naive_merge_with_images` always split a
section on the delimiter whenever one is present (even when the section
already fits `token_size`), so delimiter text never leaks into a chunk.
Only the empty-delimiter (size-only) mode skips splitting.
- `token_chunker`: delimiter text is dropped (not stripped); JSON flush
joins buffered items with `"\n"`; `children_delimiters` and
`PDF_POSITIONS_KEY` are preserved on the delimiter path. PDF positions
are now attributed **per segment** — each split chunk carries only the
positions of the item(s) that contributed to it — fixing a leak where
page-N coordinates were attached to page-M chunks and all segments
shared one preview image.
- `test_txt_parser.py` rewritten to assert the new contract (not the old
strict cap); `naive_merge` and delimiter-case-sensitive matrices
updated.
## Contract (refs #17799)
- user specified delimiter = chunk boundary; user specified delimiter
text never enters a chunk.
- `token_size` = soft target + merge strategy; no atom-split.
- Default strategy = `OVER_CAP`; migration can switch to `UNDER_CAP`
(strict cap).
- `OVER_CAP` has no hard cap; the model layer truncates oversize units.
`UNDER_CAP` enforces a strict cap.
## Notes
- Closes the wrong-object revert in #17774 (revert #17723 would
re-introduce delimiter-in-chunk and the strict cap).
- Go-side alignment (`internal/ingestion/component/chunker/token.go`) is
a follow-up PR.
---------
Co-authored-by: CodeBuddy <noreply@tencent.com>
2026-08-05 11:50:07 +08:00
|
|
|
|
text_buffer_pos.append(chunk.get(PDF_POSITIONS_KEY))
|
2026-08-03 17:47:08 +08:00
|
|
|
|
else:
|
|
|
|
|
|
flush_text_buffer()
|
|
|
|
|
|
chunks.append(chunk)
|
|
|
|
|
|
flush_text_buffer()
|
Fix: delimiter is chunk boundary, drop token_size atom-split (OVER_CAP default) (#17808)
## Summary
Fixes a regression introduced by #17203 (strict-cap atom-split) and a
secondary delimiter-handling bug from #17723.
**Root cause:**
- #17203 added `_split_oversized_unit` / `_compute_chunk_update`, which
split oversize units into ≤ token_size pieces. This collapsed
`token_size=1` into 1-token chunks and set the cap at 512, mismatching
the model-layer truncation boundary (embedding ~8191 / rerank
500/4096/8192/2048). Atom-split is unnecessary: oversize units stay
whole and the model layer truncates.
- #17723's delimiter handling dropped consecutive delimiters (`A####B`
-> `A##B`), glued JSON items with `"".join`, ignored
`children_delimiters`, and stripped whitespace delimiters.
## Changes
- New pure helper `merge_paragraphs(paragraphs, token_size, strategy)`
with a `MergeStrategy` enum (`UNDER_CAP` / `OVER_CAP`); **default
`OVER_CAP`**. `UNDER_CAP` is a strict cap (never overflows
`token_size`); `OVER_CAP` greedily accumulates adjacent paragraphs while
the projected total stays within `token_size`, merging one
boundary-overflow paragraph before closing. Oversize paragraphs stand
alone.
- `naive_merge` / `naive_merge_with_images` /
`RAGFlowTxtParser.parser_txt` now use `merge_paragraphs`; atom-split
removed. `naive_merge` / `naive_merge_with_images` always split a
section on the delimiter whenever one is present (even when the section
already fits `token_size`), so delimiter text never leaks into a chunk.
Only the empty-delimiter (size-only) mode skips splitting.
- `token_chunker`: delimiter text is dropped (not stripped); JSON flush
joins buffered items with `"\n"`; `children_delimiters` and
`PDF_POSITIONS_KEY` are preserved on the delimiter path. PDF positions
are now attributed **per segment** — each split chunk carries only the
positions of the item(s) that contributed to it — fixing a leak where
page-N coordinates were attached to page-M chunks and all segments
shared one preview image.
- `test_txt_parser.py` rewritten to assert the new contract (not the old
strict cap); `naive_merge` and delimiter-case-sensitive matrices
updated.
## Contract (refs #17799)
- user specified delimiter = chunk boundary; user specified delimiter
text never enters a chunk.
- `token_size` = soft target + merge strategy; no atom-split.
- Default strategy = `OVER_CAP`; migration can switch to `UNDER_CAP`
(strict cap).
- `OVER_CAP` has no hard cap; the model layer truncates oversize units.
`UNDER_CAP` enforces a strict cap.
## Notes
- Closes the wrong-object revert in #17774 (revert #17723 would
re-introduce delimiter-in-chunk and the strict cap).
- Go-side alignment (`internal/ingestion/component/chunker/token.go`) is
a follow-up PR.
---------
Co-authored-by: CodeBuddy <noreply@tencent.com>
2026-08-05 11:50:07 +08:00
|
|
|
|
# Apply children_delimiters (secondary split) before finalizing.
|
|
|
|
|
|
if custom_pattern:
|
|
|
|
|
|
chunks = _split_chunk_docs_by_children(chunks, custom_pattern)
|
2026-08-03 17:47:08 +08:00
|
|
|
|
_attach_context_to_media_chunks(chunks, self._param.table_context_size, self._param.image_context_size)
|
2026-08-07 16:11:42 +08:00
|
|
|
|
else:
|
|
|
|
|
|
# No active delimiter: merge adjacent text items to chunk_token_size.
|
|
|
|
|
|
# This runs on the per-item chunks (NOT a single concatenated chunk),
|
|
|
|
|
|
# so the token cap is actually enforced -- matching the previous
|
|
|
|
|
|
# "token_size" mode and the Go JSON path. Media chunks break the merge.
|
|
|
|
|
|
# Media context is attached on the per-item chunks before merging, as
|
|
|
|
|
|
# the removed "token_size" branch did, to preserve context windows.
|
|
|
|
|
|
_attach_context_to_media_chunks(text_chunks, self._param.table_context_size, self._param.image_context_size)
|
|
|
|
|
|
chunks = _merge_text_chunks_by_token_size(text_chunks, self._param.chunk_token_size, overlapped_percent)
|
|
|
|
|
|
if custom_pattern:
|
|
|
|
|
|
chunks = _split_chunk_docs_by_children(chunks, custom_pattern)
|
2026-04-03 19:26:45 +08:00
|
|
|
|
|
|
|
|
|
|
await restore_pdf_text_previews(chunks, from_upstream, self._canvas)
|
2026-08-07 16:11:42 +08:00
|
|
|
|
self.set_output("chunks", _finalize_json_chunks(chunks))
|
2026-04-03 19:26:45 +08:00
|
|
|
|
self.callback(1, "Done.")
|
2026-08-07 16:11:42 +08:00
|
|
|
|
return
|