refactor(nlp): collapse 6 delimiter-parser implementations into one canonical helper (#17383) (#17387)

## Summary

Six sites used to read the same `parser_config.delimiter` field with
divergent grammars:

- `rag.nlp.get_delimiters` (PDF/DOCX/HTML/EPUB/JSON/CSV/XLSX/email/book)
- `rag.nlp.naive_merge` (custom-delimiter branch)
- `rag.nlp.naive_merge_with_images`
- `rag.nlp._build_cks`
- `deepdoc.parser.txt_parser.parser_txt` (.txt, code)
-
`deepdoc.parser.markdown_parser.MarkdownElementExtractor.get_delimiters`

The six implementations disagreed on bare-vs-wrapped chars, dedupe, sort
order, CRLF normalization, and `re.I` (#17384). The shipped default ``
`\n!?;。;!?` `` was a no-op for `.md` because the markdown path only
matched backtick-wrapped tokens.

## Changes

- **new:** `rag/nlp/delim.py` with `parse_delimiter_field` and
`compile_delimiter_pattern`. Single source of truth. CRLF normalization
at the top; longest-first stable sort; insertion-ordered dedupe; no
`re.I`.
- **refactor:** all six call sites delegate to the helper.
  - `rag/nlp/__init__.py::get_delimiters` becomes a thin shim.
- `deepdoc/parser/txt_parser.py::parser_txt` drops the
`[encode/decode/unicode_escape]` round-trip.
- `deepdoc/parser/markdown_parser.py::get_delimiters` honors bare chars
(fixes [1]).
- **tests:** `test/unit_test/rag/test_delim.py` (85 tests) — helper,
acceptance table, frontend parity, static guard against re-inlining.
- **tests:** `test/unit_test/rag/test_delimiter_case_sensitive.py` (from
#17386) updated to retarget the static check at the new helper +
AST-based broader scan.

## Acceptance criteria

- All six sites produce the same regex pattern for the same input.
- Shipped default keeps working for `.txt` / `.pdf` / `.docx`.
- Shipped default for `.md` now splits (was a silent no-op).
- Tooltip example `` `\n##;` `` produces three effective delimiters
regardless of file type.
- Bare whitespace inputs split on every occurrence.
- Backtick-wrapped whitespace splits only on the exact N-char sequence.
- CRLF-line-ending documents split identically to LF-line-ending
documents.
- 123 tests pass (85 new + 38 existing).

## Rebase protocol

As #17385 and #17386 evolve, this branch will be rebased on top. The
only overlap between this PR's diff and the other two is
`test_delimiter_case_sensitive.py`, where #17383 modifies the static
check to point at the new helper location.

---------

Co-authored-by: kiloconnect[bot] <240665456+kiloconnect[bot]@users.noreply.github.com>
This commit is contained in:
S
2026-08-02 14:37:14 +05:30
committed by GitHub
parent 01d667296d
commit d4ceeee4ed
14 changed files with 1551 additions and 321 deletions

View File

@@ -19,11 +19,11 @@ package chunker
import (
"fmt"
"regexp"
"sort"
"strings"
"ragflow/internal/agent/runtime"
"ragflow/internal/ingestion/component/schema"
"ragflow/internal/parser/chunk"
"ragflow/internal/tokenizer"
)
@@ -76,87 +76,14 @@ func stringListFromAny(in []any) []string {
// regex / split helpers
// ---------------------------------------------------------------------------
// backtickDelimRE extracts backtick-wrapped delimiter tokens.
// Case-sensitive on purpose (mirrors Python after #17384 / PR #17386):
// never add (?i) here — delimiter matching must preserve letter casing.
var backtickDelimRE = regexp.MustCompile("`([^`]+)`")
// escapeAndSort QuoteMeta-escapes tokens and sorts longest-first.
// Empty tokens are dropped. When dedup is true, exact duplicate tokens
// are collapsed (first occurrence wins). Matching stays case-sensitive.
func escapeAndSort(tokens []string, dedup bool) []string {
out := make([]string, 0, len(tokens))
var seen map[string]struct{}
if dedup {
seen = make(map[string]struct{}, len(tokens))
}
for _, tok := range tokens {
if tok == "" {
continue
}
if dedup {
if _, ok := seen[tok]; ok {
continue
}
seen[tok] = struct{}{}
}
out = append(out, regexp.QuoteMeta(tok))
}
sort.SliceStable(out, func(i, j int) bool { return len(out[i]) > len(out[j]) })
return out
}
// getDelimiters ports rag.nlp.get_delimiters (rag/nlp/__init__.py).
//
// It walks a single delimiter string, pulling out backtick-wrapped tokens
// and every bare character outside those spans, then returns a
// length-sorted, QuoteMeta-escaped alternation suitable for regexp.Split.
// Matching is case-sensitive: "a" does not match "A", and "`end`" does not
// match "End" / "END".
func getDelimiters(delimiters string) string {
var dels []string
s := 0
for _, m := range backtickDelimRE.FindAllStringSubmatchIndex(delimiters, -1) {
// m = [fullStart, fullEnd, g1Start, g1End]
f, t := m[0], m[1]
dels = append(dels, delimiters[m[2]:m[3]])
for _, r := range delimiters[s:f] {
dels = append(dels, string(r))
}
s = t
}
if s < len(delimiters) {
for _, r := range delimiters[s:] {
dels = append(dels, string(r))
}
}
// Python get_delimiters does not dedup; preserve that behavior.
return strings.Join(escapeAndSort(dels, false), "|")
}
// compileDelimPattern builds an alternation from backtick-wrapped tokens
// across delimiter entries (mirrors Python _compile_delimiter_pattern).
// Each entry is scanned independently so token boundaries cannot span
// adjacent slice elements. Extraction is case-sensitive — see backtickDelimRE.
//
// Plain (non-backtick) delimiters are not compiled here; callers that
// need bare-char splitting use getDelimiters (naive_merge path).
// compileDelimPattern compiles a TokenChunker-style []string delimiter list.
// Only backtick-wrapped entries produce an active pattern (Python
// token_chunker / rag/nlp/delim list helper). Plain entries are ignored here
// and used by mergeByTokenSize for sentence-level splitting when no active
// pattern exists. Canonical single-string parser_config.delimiter parsing
// lives in ragflow/internal/parser/chunk (ParseDelimiterField).
func compileDelimPattern(delims []string) *regexp.Regexp {
var tokens []string
for _, d := range delims {
if d == "" {
continue
}
for _, m := range backtickDelimRE.FindAllStringSubmatch(d, -1) {
tokens = append(tokens, m[1])
}
}
// Python _compile_delimiter_pattern dedups via set(...).
custom := escapeAndSort(tokens, true)
if len(custom) == 0 {
return nil
}
return regexp.MustCompile(strings.Join(custom, "|"))
return chunk.CompileDelimiterListPattern(delims)
}
// splitKeepingDelim mirrors Python token_chunker._split_text_by_pattern

View File

@@ -32,8 +32,18 @@ import (
"regexp"
"strings"
"testing"
"ragflow/internal/parser/chunk"
)
func getDelimiters(delimiters string) string {
p := chunk.CompileDelimiterPattern(chunk.ParseDelimiterField(delimiters))
if p == nil {
return ""
}
return p.String()
}
func TestGetDelimiters_BareCharAReturnsLiteralPattern(t *testing.T) {
// Bare-char delimiter "a" must produce the pattern "a", not "a|A".
if got, want := getDelimiters("a"), "a"; got != want {
@@ -219,15 +229,20 @@ func TestTokenChunker_BacktickASplitsOnlyAtLowercase(t *testing.T) {
}
}
func TestBacktickDelimRE_HasNoIgnoreCaseFlag(t *testing.T) {
// Structural guard: the shared backtick extractor must stay case-sensitive.
src := backtickDelimRE.String()
if strings.Contains(src, "(?i)") || strings.HasPrefix(src, "(?i)") {
t.Fatalf("backtickDelimRE must not use (?i); got %q", src)
func TestBacktickDelimiterIsCaseSensitive(t *testing.T) {
// Extraction and compiled pattern must both preserve letter casing.
parsed := chunk.ParseDelimiterField("`End`")
if len(parsed) != 1 || parsed[0] != "End" {
t.Fatalf("ParseDelimiterField(`End`) = %#v, want [End]", parsed)
}
// Sanity: the pattern still extracts backtick contents.
m := backtickDelimRE.FindStringSubmatch("`End`")
if len(m) < 2 || m[1] != "End" {
t.Fatalf("backtickDelimRE(`End`) = %#v, want group1=End", m)
p := chunk.CompileDelimiterPattern(parsed)
if p == nil {
t.Fatal("CompileDelimiterPattern returned nil")
}
if !p.MatchString("End") || p.MatchString("end") || p.MatchString("END") {
t.Fatalf("pattern %q is not case-sensitive", p.String())
}
if strings.Contains(p.String(), "(?i)") {
t.Fatalf("compiled pattern must not carry (?i); got %q", p.String())
}
}

View File

@@ -23,10 +23,11 @@
// mirroring Python's normalize_overlapped_percent), table_context_size ≥ 0,
// image_context_size ≥ 0. enum/range checks live in param.Check.
//
// - DELIMITER PARSING mirrors python `_compile_delimiter_pattern`:
// entries wrapped in backticks (e.g. "`\\n\\n`") are treated as
// regex split points; plain strings are regex-escaped and joined
// into the same alternation. Empty entries are filtered.
// - DELIMITER PARSING for the TokenChunker list API mirrors Python
// token_chunker: only entries wrapped in backticks (e.g. "`\\n\\n`")
// produce an active split pattern. Plain list entries are not
// compiled into the pattern. Single-string parser_config.delimiter
// parsing lives in ragflow/internal/parser/chunk (ParseDelimiterField).
//
// - CHILDREN DELIMITERS (the secondary split) is implemented via the
// shared splitKeepingDelim helper; emitted chunks carry the parent
@@ -63,12 +64,13 @@ import (
"strings"
"sync"
"gorm.io/gorm"
"ragflow/internal/agent/runtime"
deepdoctype "ragflow/internal/deepdoc/parser/type"
"ragflow/internal/ingestion/component/globals"
"ragflow/internal/ingestion/component/schema"
"gorm.io/gorm"
"ragflow/internal/parser/chunk"
)
const ComponentNameTokenChunker = "TokenChunker"
@@ -1075,14 +1077,9 @@ func hasActiveDelimiter(p *regexp.Regexp) bool {
// hasCustomDelim reports whether any delimiter uses backtick syntax
// (`pattern`). Python's naive_merge skips token-size merging when
// custom delimiters are present (naive_merge:1194-1213).
// custom delimiters are present. Delegates to the canonical helper.
func hasCustomDelim(delims []string) bool {
for _, d := range delims {
if strings.HasPrefix(d, "`") && strings.HasSuffix(d, "`") && len(d) >= 2 {
return true
}
}
return false
return chunk.HasCustomDelimiterList(delims)
}
// applyChildrenDelim mirrors token_chunker.py:325-334.