diff --git a/internal/ingestion/component/media_dispatch.go b/internal/ingestion/component/media_dispatch.go
index 5285220a82..e7a14f6721 100644
--- a/internal/ingestion/component/media_dispatch.go
+++ b/internal/ingestion/component/media_dispatch.go
@@ -212,7 +212,10 @@ func maybeDispatchImage(
dataURI := "data:" + mimeType + ";base64," + imageB64
prompt := "Describe this image in detail."
- if v, ok := setup["prompt"].(string); ok && v != "" {
+ // image family's contract key is system_prompt (parser.go:295),
+ // mirroring Python parser.py:1119. Do NOT read setup["prompt"]
+ // here — that key is for the video family, not image.
+ if v, ok := setup["system_prompt"].(string); ok && v != "" {
prompt = v
}
messages := []modelModule.Message{{
diff --git a/internal/ingestion/component/media_dispatch_test.go b/internal/ingestion/component/media_dispatch_test.go
new file mode 100644
index 0000000000..0b12589fad
--- /dev/null
+++ b/internal/ingestion/component/media_dispatch_test.go
@@ -0,0 +1,118 @@
+//
+// Copyright 2026 The InfiniFlow Authors. All Rights Reserved.
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+// http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+package component
+
+import (
+ "context"
+ "sync"
+ "testing"
+
+ "ragflow/internal/common"
+ "ragflow/internal/entity"
+ modelModule "ragflow/internal/entity/models"
+ "ragflow/internal/utility"
+)
+
+// imagePromptCaptureDriver embeds ModelDriver so it satisfies the interface
+// without listing every method; only ChatWithMessages is overridden to record
+// the messages the image branch selected.
+type imagePromptCaptureDriver struct {
+ modelModule.ModelDriver
+ mu sync.Mutex
+ captured []modelModule.Message
+}
+
+func (d *imagePromptCaptureDriver) ChatWithMessages(ctx context.Context, modelName string, messages []modelModule.Message, apiConfig *modelModule.APIConfig, chatModelConfig *modelModule.ChatConfig, usage *common.ModelUsage) (*modelModule.ChatResponse, error) {
+ d.mu.Lock()
+ d.captured = append(d.captured, messages...)
+ d.mu.Unlock()
+ ans := "captured"
+ return &modelModule.ChatResponse{Answer: &ans}, nil
+}
+
+// firstUserText extracts the text of the first "text" content part from the
+// first captured user message. It scans parts by the "type" discriminator so
+// the test stays valid regardless of part ordering (image_url may precede text).
+func firstUserText(msgs []modelModule.Message) (string, bool) {
+ if len(msgs) == 0 {
+ return "", false
+ }
+ parts, ok := msgs[0].Content.([]interface{})
+ if !ok || len(parts) == 0 {
+ return "", false
+ }
+ for _, raw := range parts {
+ part, ok := raw.(map[string]any)
+ if !ok {
+ continue
+ }
+ if part["type"] != "text" {
+ continue
+ }
+ if txt, ok := part["text"].(string); ok {
+ return txt, true
+ }
+ }
+ return "", false
+}
+
+// TestMaybeDispatchImage_UsesSystemPrompt pins that the image branch reads the
+// `system_prompt` setup key (Python/DSL contract), not `prompt`. Before the fix
+// the branch read setup["prompt"] which is empty for image, so the VLM received
+// the hardcoded default instead of the user-configured value.
+func TestMaybeDispatchImage_UsesSystemPrompt(t *testing.T) {
+ origResolver := resolveTenantModelByType
+ defer func() { resolveTenantModelByType = origResolver }()
+
+ drv := &imagePromptCaptureDriver{}
+ resolveTenantModelByType = func(tenantID string, modelType entity.ModelType) (modelModule.ModelDriver, string, *modelModule.APIConfig, int, error) {
+ return drv, "img-model", &modelModule.APIConfig{}, 0, nil
+ }
+
+ setups := defaultSetups()
+ // image family's contract key is system_prompt (parser.go:295).
+ // Also set a legacy `prompt` sentinel to assert system_prompt wins
+ // when both keys are present (regression guard against re-reading
+ // setup["prompt"], which is the video-family key, not image).
+ setups["image"]["prompt"] = "legacy prompt"
+ setups["image"]["system_prompt"] = "自定义视觉提示"
+
+ res, dispatched, err := maybeDispatchImage(
+ context.Background(),
+ utility.FileTypeVISUAL,
+ "test.png",
+ []byte("not-a-real-image"),
+ map[string]any{"tenant_id": "t1"},
+ setups,
+ )
+ if err != nil {
+ t.Fatalf("maybeDispatchImage: %v", err)
+ }
+ if !dispatched {
+ t.Fatalf("expected dispatched=true for VISUAL file")
+ }
+ if res.Text == "" {
+ t.Fatalf("expected non-empty combined text")
+ }
+
+ got, ok := firstUserText(drv.captured)
+ if !ok {
+ t.Fatalf("no user text captured in VLM messages: %#v", drv.captured)
+ }
+ if got != "自定义视觉提示" {
+ t.Fatalf("VLM user text = %q, want %q (image branch must read system_prompt)", got, "自定义视觉提示")
+ }
+}
diff --git a/internal/ingestion/task/pipeline_real_integration_test.go b/internal/ingestion/task/pipeline_real_integration_test.go
index e7907818e8..5b8178aad3 100644
--- a/internal/ingestion/task/pipeline_real_integration_test.go
+++ b/internal/ingestion/task/pipeline_real_integration_test.go
@@ -204,7 +204,8 @@ func TestPipelineExecutor_Run_RealPDF_WritesAndReadsBackFromElasticsearch(t *tes
pdfPath := filepath.Join(taskRepoRoot(t), "internal", "deepdoc", "parser", "pdf", "testdata", "pdfs", "01_english_simple.pdf")
pdfBytes, err := os.ReadFile(pdfPath)
if err != nil {
- t.Fatalf("read pdf fixture: %v", err)
+ t.Skipf("read pdf fixture %s: %v", pdfPath, err)
+ return
}
suffix := fmt.Sprintf("%d", time.Now().UnixNano())
diff --git a/internal/parser/parser/docx_ir.go b/internal/parser/parser/docx_ir.go
new file mode 100644
index 0000000000..36fae673a8
--- /dev/null
+++ b/internal/parser/parser/docx_ir.go
@@ -0,0 +1,444 @@
+//
+// Copyright 2026 The InfiniFlow Authors. All Rights Reserved.
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+// http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+//
+
+// Package parser: this file holds the pure-Go office_oxide IR data
+// model and JSON helpers for DOCX. It is intentionally cgo-free so
+// that postprocessing (docx_postprocess.go) and unit tests run
+// without the office_oxide native library. The cgo boundary lives
+// entirely in docx_parser.go (officeOxide.OpenFromBytes).
+
+package parser
+
+import (
+ "encoding/base64"
+ "encoding/json"
+ "html"
+ "strings"
+)
+
+// DOCXFigure represents one embedded image plus its surrounding text
+// context, mirroring the chunk-level shape that Python's
+// naive_merge_docx produces for vision_figure_parser_docx_wrapper_naive.
+type DOCXFigure struct {
+ Image string `json:"image"` // base64-encoded image bytes
+ ContextAbove string `json:"context_above"` // text before the image block
+ ContextBelow string `json:"context_below"` // text after the image block
+ Marker string `json:"marker"` // substring to locate image position in markdown
+}
+
+// --- office_oxide IR types (local copy, independent of deepdoc) ---
+
+type docxIRDocument struct {
+ Sections []docxIRSection `json:"sections"`
+}
+
+type docxIRSection struct {
+ Title string `json:"title"`
+ Elements []docxIRElement `json:"elements"`
+}
+
+type docxIRElement struct {
+ Type string `json:"type"` // "paragraph", "heading", "table", "image", "list", "text_box", ...
+ Level int `json:"level"` // heading level (1-6) or list nesting level
+ Style string `json:"style"` // Word style name (e.g. "Normal", "Heading 1")
+ Content json.RawMessage `json:"content"` // rich text runs or block-level content; decoded per type
+ Data []byte `json:"data"` // raw image bytes (for "image" type)
+ Rows []docxIRRow `json:"rows"` // table rows
+ Ordered bool `json:"ordered"` // true=numbered list, false=bullet list (for "list" type)
+ Items []docxIRListItem `json:"items"` // list items (for "list" type)
+}
+
+// contentRuns decodes Content as flat text runs (paragraph/heading type).
+func (e docxIRElement) contentRuns() []docxIRRun {
+ var runs []docxIRRun
+ if len(e.Content) > 0 {
+ _ = json.Unmarshal(e.Content, &runs)
+ }
+ return runs
+}
+
+// contentBlocks decodes Content as block-level elements (text_box type).
+func (e docxIRElement) contentBlocks() []docxIRElement {
+ var blocks []docxIRElement
+ if len(e.Content) > 0 {
+ _ = json.Unmarshal(e.Content, &blocks)
+ }
+ return blocks
+}
+
+// docxIRListItem represents one item in an ordered/unordered list.
+type docxIRListItem struct {
+ Content []docxIRElement `json:"content"` // block-level content (typically a single Paragraph)
+ Nested *docxIRList `json:"nested,omitempty"` // optional nested sub-list; null/absent when none
+}
+
+// docxIRList mirrors office_oxide's ir::List. Only Items is needed for
+// text extraction; the remaining List fields (ordered, start_number,
+// style, level) are not consumed here, so they are not modeled.
+type docxIRList struct {
+ Items []docxIRListItem `json:"items"`
+}
+
+type docxIRRun struct {
+ Type string `json:"type"` // "text", "image"
+ Text string `json:"text"`
+ Content []docxIRElement `json:"content"` // nested elements (used in table cells)
+}
+
+type docxIRRow struct {
+ Cells []docxIRCell `json:"cells"`
+}
+
+type docxIRCell struct {
+ Content []docxIRElement `json:"content"` // nested paragraphs inside table cell
+}
+
+func joinDOCXIRRuns(runs []docxIRRun) string {
+ var b strings.Builder
+ for _, r := range runs {
+ if r.Type == "text" {
+ b.WriteString(r.Text)
+ }
+ }
+ return b.String()
+}
+
+// extractTextFromListItem extracts the plain text content from a list item.
+// Each list item contains block-level elements (typically a Paragraph),
+// whose text runs are concatenated. Nested sub-lists (multi-level
+// bullets/numbered items) are decoded and recursed so their text is not
+// silently dropped. Mirrors office_oxide ir::ListItem { content, nested }.
+func extractTextFromListItem(item docxIRListItem) string {
+ var parts []string
+ for _, el := range item.Content {
+ if el.Type == "paragraph" || el.Type == "heading" {
+ t := joinDOCXIRRuns(el.contentRuns())
+ if t != "" {
+ parts = append(parts, t)
+ }
+ }
+ }
+ if item.Nested != nil {
+ for _, sub := range item.Nested.Items {
+ if t := extractTextFromListItem(sub); t != "" {
+ parts = append(parts, t)
+ }
+ }
+ }
+ if len(parts) == 0 {
+ return ""
+ }
+ return strings.TrimSpace(strings.Join(parts, "\n"))
+}
+
+// extractTextFromBlockElements extracts text from a slice of block-level
+// elements (paragraphs/headings), used by text_box and other compound
+// element types.
+func extractTextFromBlockElements(blocks []docxIRElement) string {
+ var parts []string
+ for _, el := range blocks {
+ if el.Type == "paragraph" || el.Type == "heading" {
+ t := joinDOCXIRRuns(el.contentRuns())
+ if t != "" {
+ parts = append(parts, t)
+ }
+ }
+ }
+ if len(parts) == 0 {
+ return ""
+ }
+ return strings.TrimSpace(strings.Join(parts, "\n"))
+}
+
+// joinCellText concatenates all paragraph texts inside a table cell,
+// joined by newlines.
+func joinCellText(cell docxIRCell) string {
+ var parts []string
+ for _, el := range cell.Content {
+ if text := joinDOCXIRRuns(el.contentRuns()); text != "" {
+ parts = append(parts, text)
+ }
+ }
+ return strings.Join(parts, "\n")
+}
+
+// docxIRTableToHTML converts a table IR element to an HTML table string.
+func docxIRTableToHTML(el docxIRElement) string {
+ var sb strings.Builder
+ sb.WriteString("
")
+ for _, row := range el.Rows {
+ sb.WriteString("")
+ for _, cell := range row.Cells {
+ sb.WriteString("")
+ sb.WriteString(html.EscapeString(joinCellText(cell)))
+ sb.WriteString(" ")
+ }
+ sb.WriteString(" ")
+ }
+ sb.WriteString("
")
+ return sb.String()
+}
+
+// docxElementText returns the plain-text rendering of any supported
+// IR element type. Used by extractDOCXFiguresFromIR so that tables,
+// lists, and text boxes contribute to image surrounding context
+// instead of becoming empty flatBlocks (which would drop adjacent
+// VLM context). Returns "" for image and unknown types.
+func docxElementText(el docxIRElement) string {
+ switch el.Type {
+ case "paragraph", "heading":
+ return joinDOCXIRRuns(el.contentRuns())
+ case "table":
+ var lines []string
+ for _, row := range el.Rows {
+ for _, cell := range row.Cells {
+ if t := joinCellText(cell); t != "" {
+ lines = append(lines, t)
+ }
+ }
+ }
+ return strings.Join(lines, "\n")
+ case "list":
+ var lines []string
+ for _, item := range el.Items {
+ if t := extractTextFromListItem(item); t != "" {
+ lines = append(lines, t)
+ }
+ }
+ return strings.Join(lines, "\n")
+ case "text_box":
+ return extractTextFromBlockElements(el.contentBlocks())
+ default:
+ return ""
+ }
+}
+
+// buildDOCXJSONSections converts an office_oxide IR JSON string into a
+// slice of structured items compatible with the chunker's JSON input
+// contract. Each item carries at least text and doc_type_kwd.
+func buildDOCXJSONSections(irJSON string) []map[string]any {
+ var ir docxIRDocument
+ if err := json.Unmarshal([]byte(irJSON), &ir); err != nil {
+ return nil
+ }
+ var sections []map[string]any
+ for _, sec := range ir.Sections {
+ for _, el := range sec.Elements {
+ switch el.Type {
+ case "paragraph", "heading":
+ text := joinDOCXIRRuns(el.contentRuns())
+ if strings.TrimSpace(text) == "" {
+ continue
+ }
+ item := map[string]any{
+ "text": text,
+ "image": nil,
+ "doc_type_kwd": "text",
+ }
+ if el.Type == "heading" {
+ item["ck_type"] = "heading"
+ }
+ sections = append(sections, item)
+
+ case "image":
+ b64 := base64.StdEncoding.EncodeToString(el.Data)
+ sections = append(sections, map[string]any{
+ "text": "",
+ "image": b64,
+ "doc_type_kwd": "image",
+ })
+
+ case "table":
+ html := docxIRTableToHTML(el)
+ if html == "" {
+ continue
+ }
+ sections = append(sections, map[string]any{
+ "text": html,
+ "image": nil,
+ "doc_type_kwd": "table",
+ })
+
+ case "list":
+ for _, item := range el.Items {
+ text := extractTextFromListItem(item)
+ if text == "" {
+ continue
+ }
+ sections = append(sections, map[string]any{
+ "text": text,
+ "image": nil,
+ "doc_type_kwd": "text",
+ })
+ }
+
+ case "text_box":
+ text := extractTextFromBlockElements(el.contentBlocks())
+ if text == "" {
+ continue
+ }
+ sections = append(sections, map[string]any{
+ "text": text,
+ "image": nil,
+ "doc_type_kwd": "text",
+ })
+ }
+ }
+ }
+ return sections
+}
+
+// --- figure extraction (used by the cgo parser path) ---
+
+// extractDOCXFiguresFromIR parses the office_oxide IR JSON and
+// returns every embedded image block together with the plain text
+// immediately surrounding it. The context matches what Python's
+// naive_merge_docx attaches as context_above / context_below on
+// each chunk that carries an image.
+//
+// Reuses the IR already obtained from the doc handle in
+// ParseWithResult so the binary is not opened twice.
+func extractDOCXFiguresFromIR(irJSON string) []DOCXFigure {
+ var ir docxIRDocument
+ if err := json.Unmarshal([]byte(irJSON), &ir); err != nil {
+ return nil
+ }
+
+ var flat []flatBlock
+ for _, sec := range ir.Sections {
+ for _, el := range sec.Elements {
+ if el.Type == "image" {
+ b64 := base64.StdEncoding.EncodeToString(el.Data)
+ flat = append(flat, flatBlock{image: b64})
+ continue
+ }
+ text := docxElementText(el)
+ flat = append(flat, flatBlock{text: text})
+ }
+ }
+
+ var figures []DOCXFigure
+ for i, block := range flat {
+ if block.image == "" {
+ continue
+ }
+ fig := DOCXFigure{Image: block.image}
+
+ // Collect text above (backward scan up to docxContextWindow
+ // chars, or until another image is hit).
+ above := collectDOCXPrevText(flat, i, 512)
+ fig.ContextAbove = strings.TrimSpace(above)
+
+ // Collect text below (forward scan up to docxContextWindow
+ // chars, or until another image is hit).
+ below := collectDOCXNextText(flat, i, 512)
+ fig.ContextBelow = strings.TrimSpace(below)
+
+ // Marker: text of the immediately preceding flat block,
+ // used by the vision dispatcher to locate the image position
+ // in the rendered markdown for inline insertion.
+ for j := i - 1; j >= 0; j-- {
+ if flat[j].text != "" {
+ fig.Marker = flat[j].text
+ break
+ }
+ }
+
+ figures = append(figures, fig)
+ }
+ return figures
+}
+
+// --- internal types ---
+
+// flatBlock is a flattened IR element used internally to collect
+// text / image context around embedded figures.
+type flatBlock struct {
+ text string
+ image string // base64-encoded image data (empty for non-image)
+}
+
+const docxContextWindow = 512
+
+func collectDOCXPrevText(flat []flatBlock, idx, maxLen int) string {
+ var parts []string
+ remaining := maxLen
+ for i := idx - 1; i >= 0 && remaining > 0; i-- {
+ if flat[i].image != "" {
+ break // stop at previous image
+ }
+ if flat[i].text == "" {
+ continue
+ }
+ r := []rune(flat[i].text)
+ if len(r) > remaining {
+ r = r[len(r)-remaining:]
+ }
+ parts = append([]string{string(r)}, parts...)
+ remaining -= len(r)
+ }
+ // parts are in document order (farthest first, closest last);
+ // keep the tail = text nearest the image when truncating for the
+ // newline separators that Join inserts (they are not counted by
+ // `remaining`, so the joined length can otherwise exceed maxLen).
+ joined := strings.Join(parts, "\n")
+ if r := []rune(joined); len(r) > maxLen {
+ joined = string(r[len(r)-maxLen:])
+ }
+ return joined
+}
+
+func collectDOCXNextText(flat []flatBlock, idx, maxLen int) string {
+ var parts []string
+ remaining := maxLen
+ for i := idx + 1; i < len(flat) && remaining > 0; i++ {
+ if flat[i].image != "" {
+ break // stop at next image
+ }
+ if flat[i].text == "" {
+ continue
+ }
+ r := []rune(flat[i].text)
+ if len(r) > remaining {
+ r = r[:remaining]
+ }
+ parts = append(parts, string(r))
+ remaining -= len(r)
+ }
+ // parts are in document order (closest first, farthest last);
+ // keep the head = text nearest the image when truncating for the
+ // newline separators that Join inserts.
+ joined := strings.Join(parts, "\n")
+ if r := []rune(joined); len(r) > maxLen {
+ joined = string(r[:maxLen])
+ }
+ return joined
+}
+
+// buildFiguresMap converts the internal DOCXFigure slice to the
+// map form attached to fileMeta["figures"].
+func buildFiguresMap(figures []DOCXFigure) []map[string]any {
+ figs := make([]map[string]any, 0, len(figures))
+ for _, f := range figures {
+ figs = append(figs, map[string]any{
+ "image": f.Image,
+ "context_above": f.ContextAbove,
+ "context_below": f.ContextBelow,
+ "marker": f.Marker,
+ })
+ }
+ return figs
+}
diff --git a/internal/parser/parser/docx_ir_test.go b/internal/parser/parser/docx_ir_test.go
new file mode 100644
index 0000000000..155fff3b45
--- /dev/null
+++ b/internal/parser/parser/docx_ir_test.go
@@ -0,0 +1,253 @@
+package parser
+
+import (
+ "encoding/json"
+ "reflect"
+ "testing"
+)
+
+// TestBuildDOCXJSONSections_FromJSON feeds an office_oxide IR JSON
+// string directly into buildDOCXJSONSections and asserts the
+// paragraph / heading / image / table / list branches. This pins the
+// pure-Go IR → sections transform without requiring the office_oxide
+// native library, so it runs under CGO_ENABLED=0.
+func TestBuildDOCXJSONSections_FromJSON(t *testing.T) {
+ // "data":"aGVsbG8=" is base64 for "hello"; buildDOCXJSONSections
+ // re-encodes el.Data, so the item's "image" must equal "aGVsbG8=".
+ irJSON := `{"sections":[{"title":"","elements":[
+ {"type":"heading","level":1,"content":[{"type":"text","text":"Title"}]},
+ {"type":"paragraph","content":[{"type":"text","text":"Hello"}]},
+ {"type":"image","data":"aGVsbG8="},
+ {"type":"table","rows":[{"cells":[{"content":[{"type":"paragraph","content":[{"type":"text","text":"cell"}]}]}]}]},
+ {"type":"list","items":[{"content":[{"type":"paragraph","content":[{"type":"text","text":"item1"}]}]}]}
+ ]}]}`
+
+ got := buildDOCXJSONSections(irJSON)
+ want := []map[string]any{
+ {"text": "Title", "image": nil, "doc_type_kwd": "text", "ck_type": "heading"},
+ {"text": "Hello", "image": nil, "doc_type_kwd": "text"},
+ {"text": "", "image": "aGVsbG8=", "doc_type_kwd": "image"},
+ {"text": "", "image": nil, "doc_type_kwd": "table"},
+ {"text": "item1", "image": nil, "doc_type_kwd": "text"},
+ }
+ if !reflect.DeepEqual(got, want) {
+ t.Fatalf("buildDOCXJSONSections mismatch:\n got: %+v\nwant: %+v", got, want)
+ }
+}
+
+// TestBuildDOCXJSONSections_EmptyTableSkipped verifies that a table
+// with no rows renders as "" and is dropped by the
+// "if html == \"\" { continue }" guard.
+func TestBuildDOCXJSONSections_EmptyTableSkipped(t *testing.T) {
+ irJSON := `{"sections":[{"elements":[
+ {"type":"paragraph","content":[{"type":"text","text":"keep"}]},
+ {"type":"table","rows":[]}
+ ]}]}`
+ got := buildDOCXJSONSections(irJSON)
+ if len(got) != 1 || got[0]["text"] != "keep" {
+ t.Fatalf("expected only the paragraph to survive, got %+v", got)
+ }
+}
+
+// TestJoinDOCXIRRuns pins that only text-type runs are concatenated;
+// non-text runs (e.g. nested image runs) are skipped.
+func TestJoinDOCXIRRuns(t *testing.T) {
+ runs := []docxIRRun{
+ {Type: "text", Text: "Hello"},
+ {Type: "image", Text: "ignored"},
+ {Type: "text", Text: " World"},
+ }
+ if got := joinDOCXIRRuns(runs); got != "Hello World" {
+ t.Fatalf("joinDOCXIRRuns = %q, want %q", got, "Hello World")
+ }
+ if got := joinDOCXIRRuns(nil); got != "" {
+ t.Fatalf("joinDOCXIRRuns(nil) = %q, want empty", got)
+ }
+}
+
+// TestExtractDOCXFiguresFromIR verifies the image-figure context
+// extraction: an image block carries the immediately surrounding text
+// as ContextAbove / ContextBelow / Marker. Pure-Go, runs under !cgo.
+func TestExtractDOCXFiguresFromIR(t *testing.T) {
+ irJSON := `{"sections":[{"elements":[
+ {"type":"paragraph","content":[{"type":"text","text":"before"}]},
+ {"type":"image","data":"aGVsbG8="},
+ {"type":"paragraph","content":[{"type":"text","text":"after"}]}
+ ]}]}`
+
+ figs := extractDOCXFiguresFromIR(irJSON)
+ if len(figs) != 1 {
+ t.Fatalf("expected 1 figure, got %d", len(figs))
+ }
+ fig := figs[0]
+ want := DOCXFigure{
+ Image: "aGVsbG8=",
+ ContextAbove: "before",
+ ContextBelow: "after",
+ Marker: "before",
+ }
+ if !reflect.DeepEqual(fig, want) {
+ t.Fatalf("figure mismatch:\n got: %+v\nwant: %+v", fig, want)
+ }
+}
+
+// TestExtractDOCXFiguresFromIR_NoImage returns nil when the IR has no
+// image blocks.
+func TestExtractDOCXFiguresFromIR_NoImage(t *testing.T) {
+ irJSON := `{"sections":[{"elements":[
+ {"type":"paragraph","content":[{"type":"text","text":"only text"}]}
+ ]}]}`
+ if figs := extractDOCXFiguresFromIR(irJSON); figs != nil {
+ t.Fatalf("expected nil, got %+v", figs)
+ }
+}
+
+// TestExtractDOCXFiguresFromIR_BadJSON returns nil on unparseable IR,
+// mirroring the json.Unmarshal error guard.
+func TestExtractDOCXFiguresFromIR_BadJSON(t *testing.T) {
+ if figs := extractDOCXFiguresFromIR("{not json"); figs != nil {
+ t.Fatalf("expected nil for bad JSON, got %+v", figs)
+ }
+}
+
+// TestExtractDOCXFiguresFromIR_NonParagraphContext verifies that
+// tables, lists, and text boxes adjacent to an image contribute their
+// text to ContextAbove / ContextBelow. Previously the flatten loop
+// called joinDOCXIRRuns(el.contentRuns()) for every non-image element,
+// which returned "" for table/list/text_box (their Content is not text
+// runs), silently dropping the surrounding VLM context.
+func TestExtractDOCXFiguresFromIR_NonParagraphContext(t *testing.T) {
+ irJSON := `{"sections":[{"elements":[
+ {"type":"table","rows":[{"cells":[{"content":[{"type":"paragraph","content":[{"type":"text","text":"table cell"}]}]}]}]},
+ {"type":"image","data":"aGVsbG8="},
+ {"type":"list","items":[{"content":[{"type":"paragraph","content":[{"type":"text","text":"list item"}]}]}]},
+ {"type":"text_box","content":[{"type":"paragraph","content":[{"type":"text","text":"box text"}]}]}
+ ]}]}`
+ figs := extractDOCXFiguresFromIR(irJSON)
+ if len(figs) != 1 {
+ t.Fatalf("expected 1 figure, got %d", len(figs))
+ }
+ fig := figs[0]
+ if fig.ContextAbove != "table cell" {
+ t.Errorf("ContextAbove = %q, want %q", fig.ContextAbove, "table cell")
+ }
+ // Below the image: list item then text_box, joined by newline.
+ if fig.ContextBelow != "list item\nbox text" {
+ t.Errorf("ContextBelow = %q, want %q", fig.ContextBelow, "list item\nbox text")
+ }
+}
+
+// TestCollectDOCXText_BoundedByMaxLen pins that the joined context
+// never exceeds maxLen runes, even though newline separators are
+// inserted between blocks (they are not counted by the per-block
+// `remaining` decrement). Previously 512 one-rune blocks produced
+// 1023 chars (512 runes + 511 separators).
+func TestCollectDOCXText_BoundedByMaxLen(t *testing.T) {
+ // 600 one-rune text blocks before and after an image block.
+ const maxLen = 512
+ var flat []flatBlock
+ for i := 0; i < 600; i++ {
+ flat = append(flat, flatBlock{text: "x"})
+ }
+ imgIdx := len(flat)
+ flat = append(flat, flatBlock{image: "img"})
+ for i := 0; i < 600; i++ {
+ flat = append(flat, flatBlock{text: "x"})
+ }
+
+ prev := collectDOCXPrevText(flat, imgIdx, maxLen)
+ next := collectDOCXNextText(flat, imgIdx, maxLen)
+ if r := []rune(prev); len(r) > maxLen {
+ t.Errorf("prev context = %d runes, want <= %d", len(r), maxLen)
+ }
+ if r := []rune(next); len(r) > maxLen {
+ t.Errorf("next context = %d runes, want <= %d", len(r), maxLen)
+ }
+ // Both must still carry content (the closest blocks survive).
+ if prev == "" {
+ t.Error("prev context empty; expected some text")
+ }
+ if next == "" {
+ t.Error("next context empty; expected some text")
+ }
+}
+
+// rawJSON marshals v and returns it as json.RawMessage, for populating
+// docxIRElement.Content in tests. Defined here (tagless) so it is
+// available to both cgo and !cgo test files in package parser.
+func rawJSON(v any) json.RawMessage {
+ data, _ := json.Marshal(v)
+ return json.RawMessage(data)
+}
+
+// para is a test helper returning a paragraph element whose single text
+// run carries the given string.
+func para(text string) docxIRElement {
+ return docxIRElement{
+ Type: "paragraph",
+ Content: rawJSON([]docxIRRun{{Type: "text", Text: text}}),
+ }
+}
+
+// TestExtractTextFromListItem_Nested verifies that nested sub-list text
+// is decoded and recursed, not silently dropped. The IR shape mirrors
+// office_oxide ir::ListItem { content, nested: Option } where
+// List { items: [ListItem, ...] } is recursive.
+func TestExtractTextFromListItem_Nested(t *testing.T) {
+ // Top-level item "Top" carries a nested sub-list with two children
+ // "Child A" and "Child B"; Child B itself has a deeper sub-list
+ // "Grandchild" to exercise multi-level recursion.
+ item := docxIRListItem{
+ Content: []docxIRElement{para("Top")},
+ Nested: &docxIRList{Items: []docxIRListItem{
+ {Content: []docxIRElement{para("Child A")}},
+ {
+ Content: []docxIRElement{para("Child B")},
+ Nested: &docxIRList{Items: []docxIRListItem{
+ {Content: []docxIRElement{para("Grandchild")}},
+ }},
+ },
+ }},
+ }
+ got := extractTextFromListItem(item)
+ want := "Top\nChild A\nChild B\nGrandchild"
+ if got != want {
+ t.Errorf("extractTextFromListItem(nested) = %q, want %q", got, want)
+ }
+}
+
+// TestExtractTextFromListItem_NestedNull pins that "nested": null (the
+// serde serialization of Option::None) does not trip recursion and the
+// item's own text still comes through.
+func TestExtractTextFromListItem_NestedNull(t *testing.T) {
+ irJSON := `{"sections":[{"elements":[
+ {"type":"list","items":[
+ {"content":[{"type":"paragraph","content":[{"type":"text","text":"only item"}]}],"nested":null}
+ ]}
+ ]}]}`
+ sections := buildDOCXJSONSections(irJSON)
+ if len(sections) != 1 || sections[0]["text"] != "only item" {
+ t.Fatalf("expected single item \"only item\", got %+v", sections)
+ }
+}
+
+// TestBuildDOCXJSONSections_NestedList verifies end-to-end that a
+// multi-level list survives into the section output with all levels.
+func TestBuildDOCXJSONSections_NestedList(t *testing.T) {
+ irJSON := `{"sections":[{"elements":[
+ {"type":"list","items":[
+ {"content":[{"type":"paragraph","content":[{"type":"text","text":"L1"}]}],
+ "nested":{"items":[
+ {"content":[{"type":"paragraph","content":[{"type":"text","text":"L2a"}]}]},
+ {"content":[{"type":"paragraph","content":[{"type":"text","text":"L2b"}]}]}
+ ]}}
+ ]}
+ ]}]}`
+ sections := buildDOCXJSONSections(irJSON)
+ if len(sections) != 1 {
+ t.Fatalf("expected 1 section, got %d: %+v", len(sections), sections)
+ }
+ if got := sections[0]["text"]; got != "L1\nL2a\nL2b" {
+ t.Errorf("nested list text = %q, want %q", got, "L1\nL2a\nL2b")
+ }
+}
diff --git a/internal/parser/parser/docx_parser.go b/internal/parser/parser/docx_parser.go
index 7e3619ebea..6e4177ced5 100644
--- a/internal/parser/parser/docx_parser.go
+++ b/internal/parser/parser/docx_parser.go
@@ -20,28 +20,22 @@ package parser
import (
"context"
- "encoding/base64"
- "encoding/json"
"fmt"
- "html"
"strings"
officeOxide "github.com/yfedoseev/office_oxide/go"
)
-// DOCXFigure represents one embedded image plus its surrounding text
-// context, mirroring the chunk-level shape that Python's
-// naive_merge_docx produces for vision_figure_parser_docx_wrapper_naive.
-type DOCXFigure struct {
- Image string `json:"image"` // base64-encoded image bytes
- ContextAbove string `json:"context_above"` // text before the image block
- ContextBelow string `json:"context_below"` // text after the image block
- Marker string `json:"marker"` // substring to locate image position in markdown
-}
-
+// DOCXParser is the cgo-backed DOCX parser. It is the only DOCX
+// entrypoint that depends on office_oxide; the IR data model and
+// postprocessing live in cgo-free files (docx_ir.go, docx_postprocess.go)
+// so they compile and test without native libraries. The !cgo build
+// provides a stub DOCXParser in office_parsers_no_cgo.go.
type DOCXParser struct {
- libType string
- outputFormat string // from DSL config; "json" or "markdown"
+ libType string
+ outputFormat string // from DSL config; "json" or "markdown"
+ RemoveTOC bool
+ RemoveHeaderFooter bool
}
func NewDOCXParser() *DOCXParser {
@@ -58,6 +52,12 @@ func (p *DOCXParser) ConfigureFromSetup(setup map[string]any) {
if v, ok := setup["output_format"].(string); ok && v != "" {
p.outputFormat = v
}
+ if v, ok := setup["remove_toc"].(bool); ok {
+ p.RemoveTOC = v
+ }
+ if v, ok := setup["remove_header_footer"].(bool); ok {
+ p.RemoveHeaderFooter = v
+ }
}
// ParseWithResult produces structured JSON items (when
@@ -96,6 +96,20 @@ func (p *DOCXParser) ParseWithResult(ctx context.Context, filename string, data
}
var sections []map[string]any
sections = buildDOCXJSONSections(irJSON)
+ // remove_header_footer: drop sections whose normalized text
+ // matches a docx header/footer entry (mirrors Python
+ // parser.py:889-891 extract_docx_header_footer_texts +
+ // remove_header_footer_docx_sections).
+ if p.RemoveHeaderFooter {
+ hfTexts := extractDOCXHeaderFooterTexts(data)
+ sections = removeDOCXHeaderFooterSections(sections, hfTexts)
+ }
+ // remove_toc: filter TOC entries using heading outlines
+ // (mirrors Python parser.py:892-893 remove_toc_word).
+ if p.RemoveTOC {
+ outlines := extractDOCXOutlines(irJSON)
+ sections = removeTOCWord(sections, outlines, false)
+ }
if len(sections) == 0 {
sections = []map[string]any{{"text": "", "doc_type_kwd": "text"}}
}
@@ -111,6 +125,38 @@ func (p *DOCXParser) ParseWithResult(ctx context.Context, filename string, data
if err != nil {
return ParseResult{Err: fmt.Errorf("docx to-markdown: %w", err)}
}
+ // remove_header_footer on markdown: filter lines by exact match
+ // (mirrors Python parser.py:923-926 split lines → filter → rejoin).
+ if p.RemoveHeaderFooter {
+ hfTexts := extractDOCXHeaderFooterTexts(data)
+ lines := strings.Split(md, "\n")
+ lineItems := make([]map[string]any, 0, len(lines))
+ for _, ln := range lines {
+ lineItems = append(lineItems, map[string]any{"text": ln})
+ }
+ lineItems = removeDOCXHeaderFooterSections(lineItems, hfTexts)
+ rebuilt := make([]string, 0, len(lineItems))
+ for _, item := range lineItems {
+ rebuilt = append(rebuilt, itemText(item))
+ }
+ md = strings.Join(rebuilt, "\n")
+ }
+ // remove_toc on markdown: split lines, filter, rejoin
+ // (mirrors Python parser.py:927-928 remove_toc_word on markdown).
+ if p.RemoveTOC && irErr == nil {
+ outlines := extractDOCXOutlines(irJSON)
+ lines := strings.Split(md, "\n")
+ lineItems := make([]map[string]any, 0, len(lines))
+ for _, ln := range lines {
+ lineItems = append(lineItems, map[string]any{"text": ln})
+ }
+ filtered := removeTOCWord(lineItems, outlines, false)
+ rebuilt := make([]string, 0, len(filtered))
+ for _, item := range filtered {
+ rebuilt = append(rebuilt, itemText(item))
+ }
+ md = strings.Join(rebuilt, "\n")
+ }
return ParseResult{
OutputFormat: "markdown",
File: fileMeta,
@@ -118,343 +164,6 @@ func (p *DOCXParser) ParseWithResult(ctx context.Context, filename string, data
}
}
-// extractDOCXFiguresFromIR parses the office_oxide IR JSON and
-// returns every embedded image block together with the plain text
-// immediately surrounding it. The context matches what Python's
-// naive_merge_docx attaches as context_above / context_below on
-// each chunk that carries an image.
-//
-// Reuses the IR already obtained from the doc handle in
-// ParseWithResult so the binary is not opened twice.
-func extractDOCXFiguresFromIR(irJSON string) []DOCXFigure {
- var ir docxIRDocument
- if err := json.Unmarshal([]byte(irJSON), &ir); err != nil {
- return nil
- }
-
- var flat []flatBlock
- for _, sec := range ir.Sections {
- for _, el := range sec.Elements {
- if el.Type == "image" {
- b64 := base64.StdEncoding.EncodeToString(el.Data)
- flat = append(flat, flatBlock{image: b64})
- continue
- }
- text := joinDOCXIRRuns(el.contentRuns())
- flat = append(flat, flatBlock{text: text})
- }
- }
-
- var figures []DOCXFigure
- for i, block := range flat {
- if block.image == "" {
- continue
- }
- fig := DOCXFigure{Image: block.image}
-
- // Collect text above (backward scan up to docxContextWindow
- // chars, or until another image is hit).
- above := collectDOCXPrevText(flat, i, 512)
- fig.ContextAbove = strings.TrimSpace(above)
-
- // Collect text below (forward scan up to docxContextWindow
- // chars, or until another image is hit).
- below := collectDOCXNextText(flat, i, 512)
- fig.ContextBelow = strings.TrimSpace(below)
-
- // Marker: text of the immediately preceding flat block,
- // used by the vision dispatcher to locate the image position
- // in the rendered markdown for inline insertion.
- for j := i - 1; j >= 0; j-- {
- if flat[j].text != "" {
- fig.Marker = flat[j].text
- break
- }
- }
-
- figures = append(figures, fig)
- }
- return figures
-}
-
-// --- internal types ---
-
-// flatBlock is a flattened IR element used internally to collect
-// text / image context around embedded figures.
-type flatBlock struct {
- text string
- image string // base64-encoded image data (empty for non-image)
-}
-
-const docxContextWindow = 512
-
-func collectDOCXPrevText(flat []flatBlock, idx, maxLen int) string {
- var parts []string
- remaining := maxLen
- for i := idx - 1; i >= 0 && remaining > 0; i-- {
- if flat[i].image != "" {
- break // stop at previous image
- }
- if flat[i].text == "" {
- continue
- }
- r := []rune(flat[i].text)
- if len(r) > remaining {
- r = r[len(r)-remaining:]
- }
- parts = append([]string{string(r)}, parts...)
- remaining -= len(r)
- }
- return strings.Join(parts, "\n")
-}
-
-func collectDOCXNextText(flat []flatBlock, idx, maxLen int) string {
- var parts []string
- remaining := maxLen
- for i := idx + 1; i < len(flat) && remaining > 0; i++ {
- if flat[i].image != "" {
- break // stop at next image
- }
- if flat[i].text == "" {
- continue
- }
- r := []rune(flat[i].text)
- if len(r) > remaining {
- r = r[:remaining]
- }
- parts = append(parts, string(r))
- remaining -= len(r)
- }
- return strings.Join(parts, "\n")
-}
-
-func joinDOCXIRRuns(runs []docxIRRun) string {
- var b strings.Builder
- for _, r := range runs {
- if r.Type == "text" {
- b.WriteString(r.Text)
- }
- }
- return b.String()
-}
-
-// extractTextFromListItem extracts the plain text content from a list item.
-// Each list item contains block-level elements (typically a Paragraph),
-// whose text runs are concatenated.
-func extractTextFromListItem(item docxIRListItem) string {
- var parts []string
- for _, el := range item.Content {
- if el.Type == "paragraph" || el.Type == "heading" {
- t := joinDOCXIRRuns(el.contentRuns())
- if t != "" {
- parts = append(parts, t)
- }
- }
- }
- if len(parts) == 0 {
- return ""
- }
- return strings.TrimSpace(strings.Join(parts, "\n"))
-}
-
-// extractTextFromBlockElements extracts text from a slice of block-level
-// elements (paragraphs/headings), used by text_box and other compound
-// element types.
-func extractTextFromBlockElements(blocks []docxIRElement) string {
- var parts []string
- for _, el := range blocks {
- if el.Type == "paragraph" || el.Type == "heading" {
- t := joinDOCXIRRuns(el.contentRuns())
- if t != "" {
- parts = append(parts, t)
- }
- }
- }
- if len(parts) == 0 {
- return ""
- }
- return strings.TrimSpace(strings.Join(parts, "\n"))
-}
-
-// buildFiguresMap converts the internal DOCXFigure slice to the
-// map form attached to fileMeta["figures"].
-func buildFiguresMap(figures []DOCXFigure) []map[string]any {
- figs := make([]map[string]any, 0, len(figures))
- for _, f := range figures {
- figs = append(figs, map[string]any{
- "image": f.Image,
- "context_above": f.ContextAbove,
- "context_below": f.ContextBelow,
- "marker": f.Marker,
- })
- }
- return figs
-}
-
-// joinCellText concatenates all paragraph texts inside a table cell,
-// joined by newlines.
-func joinCellText(cell docxIRCell) string {
- var parts []string
- for _, el := range cell.Content {
- if text := joinDOCXIRRuns(el.contentRuns()); text != "" {
- parts = append(parts, text)
- }
- }
- return strings.Join(parts, "\n")
-}
-
-// docxIRTableToHTML converts a table IR element to an HTML table string.
-func docxIRTableToHTML(el docxIRElement) string {
- var sb strings.Builder
- sb.WriteString("")
- for _, row := range el.Rows {
- sb.WriteString("")
- for _, cell := range row.Cells {
- sb.WriteString("")
- sb.WriteString(html.EscapeString(joinCellText(cell)))
- sb.WriteString(" ")
- }
- sb.WriteString(" ")
- }
- sb.WriteString("
")
- return sb.String()
-}
-
-// buildDOCXJSONSections converts an office_oxide IR JSON string into a
-// slice of structured items compatible with the chunker's JSON input
-// contract. Each item carries at least text and doc_type_kwd.
-func buildDOCXJSONSections(irJSON string) []map[string]any {
- var ir docxIRDocument
- if err := json.Unmarshal([]byte(irJSON), &ir); err != nil {
- return nil
- }
- var sections []map[string]any
- for _, sec := range ir.Sections {
- for _, el := range sec.Elements {
- switch el.Type {
- case "paragraph", "heading":
- text := joinDOCXIRRuns(el.contentRuns())
- if strings.TrimSpace(text) == "" {
- continue
- }
- item := map[string]any{
- "text": text,
- "image": nil,
- "doc_type_kwd": "text",
- }
- if el.Type == "heading" {
- item["ck_type"] = "heading"
- }
- sections = append(sections, item)
-
- case "image":
- b64 := base64.StdEncoding.EncodeToString(el.Data)
- sections = append(sections, map[string]any{
- "text": "",
- "image": b64,
- "doc_type_kwd": "image",
- })
-
- case "table":
- html := docxIRTableToHTML(el)
- if html == "" {
- continue
- }
- sections = append(sections, map[string]any{
- "text": html,
- "image": nil,
- "doc_type_kwd": "table",
- })
-
- case "list":
- for _, item := range el.Items {
- text := extractTextFromListItem(item)
- if text == "" {
- continue
- }
- sections = append(sections, map[string]any{
- "text": text,
- "image": nil,
- "doc_type_kwd": "text",
- })
- }
-
- case "text_box":
- text := extractTextFromBlockElements(el.contentBlocks())
- if text == "" {
- continue
- }
- sections = append(sections, map[string]any{
- "text": text,
- "image": nil,
- "doc_type_kwd": "text",
- })
- }
- }
- }
- return sections
-}
-
-// --- office_oxide IR types (local copy, independent of deepdoc) ---
-
-type docxIRDocument struct {
- Sections []docxIRSection `json:"sections"`
-}
-
-type docxIRSection struct {
- Title string `json:"title"`
- Elements []docxIRElement `json:"elements"`
-}
-
-type docxIRElement struct {
- Type string `json:"type"` // "paragraph", "heading", "table", "image", "list", "text_box", ...
- Level int `json:"level"` // heading level (1-6) or list nesting level
- Style string `json:"style"` // Word style name (e.g. "Normal", "Heading 1")
- Content json.RawMessage `json:"content"` // rich text runs or block-level content; decoded per type
- Data []byte `json:"data"` // raw image bytes (for "image" type)
- Rows []docxIRRow `json:"rows"` // table rows
- Ordered bool `json:"ordered"` // true=numbered list, false=bullet list (for "list" type)
- Items []docxIRListItem `json:"items"` // list items (for "list" type)
-}
-
-// contentRuns decodes Content as flat text runs (paragraph/heading type).
-func (e docxIRElement) contentRuns() []docxIRRun {
- var runs []docxIRRun
- if len(e.Content) > 0 {
- _ = json.Unmarshal(e.Content, &runs)
- }
- return runs
-}
-
-// contentBlocks decodes Content as block-level elements (text_box type).
-func (e docxIRElement) contentBlocks() []docxIRElement {
- var blocks []docxIRElement
- if len(e.Content) > 0 {
- _ = json.Unmarshal(e.Content, &blocks)
- }
- return blocks
-}
-
-// docxIRListItem represents one item in an ordered/unordered list.
-type docxIRListItem struct {
- Content []docxIRElement `json:"content"` // block-level content (typically a single Paragraph)
- Nested json.RawMessage `json:"nested,omitempty"` // nested sub-list (stored as raw JSON for now)
-}
-
-type docxIRRun struct {
- Type string `json:"type"` // "text", "image"
- Text string `json:"text"`
- Content []docxIRElement `json:"content"` // nested elements (used in table cells)
-}
-
-type docxIRRow struct {
- Cells []docxIRCell `json:"cells"`
-}
-
-type docxIRCell struct {
- Content []docxIRElement `json:"content"` // nested paragraphs inside table cell
-}
-
func (p *DOCXParser) String() string {
return "DOCXParser"
}
diff --git a/internal/parser/parser/docx_parser_test.go b/internal/parser/parser/docx_parser_test.go
index c03be77a4c..2d82ebec92 100644
--- a/internal/parser/parser/docx_parser_test.go
+++ b/internal/parser/parser/docx_parser_test.go
@@ -25,11 +25,6 @@ import (
"testing"
)
-func rawJSON(v any) json.RawMessage {
- data, _ := json.Marshal(v)
- return json.RawMessage(data)
-}
-
func TestBuildDOCXJSONSections_Paragraphs(t *testing.T) {
ir := `{"sections":[{"title":"","elements":[
{"type":"paragraph","content":[{"type":"text","text":"Hello world"}],"style":"Normal"}
diff --git a/internal/parser/parser/docx_postprocess.go b/internal/parser/parser/docx_postprocess.go
new file mode 100644
index 0000000000..811d097b69
--- /dev/null
+++ b/internal/parser/parser/docx_postprocess.go
@@ -0,0 +1,214 @@
+package parser
+
+import (
+ "archive/zip"
+ "bytes"
+ "encoding/json"
+ "encoding/xml"
+ "io"
+ "regexp"
+ "strings"
+)
+
+// docxOutline is a heading outline entry extracted from the
+// office_oxide IR. Mirrors Python (title, level, None) tuples from
+// rag/flow/parser/utils.py:77-89 extract_word_outlines.
+type docxOutline struct {
+ Title string `json:"title"`
+ Level int `json:"level"`
+}
+
+// dotPageNumberPattern matches ".... 123" / "·····5" style TOC
+// leader lines. Mirrors Python utils.py:285 regex
+// (\.{2,}|…{2,}|·{2,}|[ ]{2,})\s*\d+\s*$.
+var dotPageNumberPattern = regexp.MustCompile(`(\.{2,}|…{2,}|·{2,}|[ ]{2,})\s*\d+\s*$`)
+
+// extractDOCXOutlines parses the office_oxide IR JSON and returns
+// heading outlines (title + 0-based level). Mirrors Python
+// rag/flow/parser/utils.py:77-89 extract_word_outlines.
+func extractDOCXOutlines(irJSON string) []docxOutline {
+ var ir docxIRDocument
+ if err := json.Unmarshal([]byte(irJSON), &ir); err != nil {
+ return nil
+ }
+ var outlines []docxOutline
+ for _, sec := range ir.Sections {
+ for _, el := range sec.Elements {
+ if el.Type != "heading" {
+ continue
+ }
+ title := strings.TrimSpace(joinDOCXIRRuns(el.contentRuns()))
+ if title == "" {
+ continue
+ }
+ outlines = append(outlines, docxOutline{
+ Title: title,
+ Level: el.Level - 1, // 0-based, matching Python
+ })
+ }
+ }
+ return outlines
+}
+
+// removeTOCWord filters TOC entries from docx items using heading
+// outlines. Mirrors Python rag/flow/parser/utils.py:115-144
+// remove_toc_word.
+//
+// Algorithm:
+// 1. No outlines → delegate to removeContentsTable (text heuristic).
+// 2. With outlines → locate the "目录/Contents" heading, drop it,
+// then drop following entries that match an outline-title prefix
+// or the "dots + page number" regex.
+// 3. Finally run removeContentsTable as a fallback sweep.
+func removeTOCWord(items []map[string]any, outlines []docxOutline, eng bool) []map[string]any {
+ if len(outlines) == 0 {
+ return removeContentsTable(items, eng)
+ }
+
+ outlineTitles := make([]string, 0, len(outlines))
+ for _, o := range outlines {
+ t := strings.ToLower(strings.TrimSpace(o.Title))
+ if t != "" {
+ outlineTitles = append(outlineTitles, t)
+ }
+ }
+
+ if len(outlineTitles) > 0 {
+ i := 0
+ for i < len(items) {
+ text := strings.ToLower(strings.TrimSpace(itemText(items[i])))
+ normalized := whitespacePattern.ReplaceAllString(strings.SplitN(text, "@@", 2)[0], "")
+ if !tocHeadingPattern.MatchString(normalized) {
+ i++
+ continue
+ }
+ // Drop the TOC heading.
+ items = append(items[:i], items[i+1:]...)
+ // Drop following entries matching outline-title prefix
+ // or dot+page-number pattern.
+ for i < len(items) {
+ raw := itemText(items[i])
+ normalized := strings.ToLower(strings.TrimSpace(strings.SplitN(raw, "@@", 2)[0]))
+ if normalized == "" {
+ items = append(items[:i], items[i+1:]...)
+ continue
+ }
+ matched := false
+ for _, title := range outlineTitles {
+ if strings.HasPrefix(normalized, title) || strings.HasPrefix(title, normalized) {
+ matched = true
+ break
+ }
+ }
+ if matched || dotPageNumberPattern.MatchString(raw) {
+ items = append(items[:i], items[i+1:]...)
+ continue
+ }
+ break
+ }
+ break
+ }
+ }
+
+ return removeContentsTable(items, eng)
+}
+
+// whitespaceCollapsePattern collapses runs of whitespace (spaces,
+// tabs, newlines) into a single space — mirrors Python
+// re.sub(r"\s+", " ", text).strip() used in
+// extract_docx_header_footer_texts (utils.py:210,229).
+var whitespaceCollapsePattern = regexp.MustCompile(`\s+`)
+
+// extractDOCXHeaderFooterTexts opens the docx ZIP, reads every
+// word/header*.xml and word/footer*.xml part, and returns the set
+// of normalized text strings found inside elements. Mirrors
+// Python rag/flow/parser/utils.py:38-53
+// extract_docx_header_footer_texts (python-docx sections[].header/
+// footer paragraphs + tables).
+func extractDOCXHeaderFooterTexts(data []byte) map[string]bool {
+ texts := make(map[string]bool)
+ zr, err := zip.NewReader(bytes.NewReader(data), int64(len(data)))
+ if err != nil {
+ return texts
+ }
+ for _, f := range zr.File {
+ name := f.Name
+ if !strings.HasPrefix(name, "word/") {
+ continue
+ }
+ base := name[len("word/"):]
+ if !strings.HasPrefix(base, "header") && !strings.HasPrefix(base, "footer") {
+ continue
+ }
+ if !strings.HasSuffix(base, ".xml") {
+ continue
+ }
+ rc, err := f.Open()
+ if err != nil {
+ continue
+ }
+ text := extractWtTextFromXML(rc)
+ rc.Close()
+ normalized := strings.TrimSpace(whitespaceCollapsePattern.ReplaceAllString(text, " "))
+ if normalized != "" {
+ texts[normalized] = true
+ }
+ }
+ return texts
+}
+
+// extractWtTextFromXML decodes the XML stream and concatenates the
+// content of every element (the WordprocessingML text node).
+func extractWtTextFromXML(r io.Reader) string {
+ dec := xml.NewDecoder(r)
+ var sb strings.Builder
+ for {
+ tok, err := dec.Token()
+ if err != nil {
+ break
+ }
+ start, ok := tok.(xml.StartElement)
+ if !ok {
+ continue
+ }
+ if start.Name.Local != "t" {
+ continue
+ }
+ // The next token inside ... is CharData.
+ for {
+ inner, err := dec.Token()
+ if err != nil {
+ return sb.String()
+ }
+ switch it := inner.(type) {
+ case xml.CharData:
+ sb.Write(it)
+ case xml.EndElement:
+ if it.Name.Local == "t" {
+ goto nextToken
+ }
+ }
+ }
+ nextToken:
+ }
+ return sb.String()
+}
+
+// removeDOCXHeaderFooterSections drops items whose normalized text
+// exactly matches a header/footer text string. Mirrors Python
+// rag/flow/parser/utils.py:56-67 remove_header_footer_docx_sections.
+func removeDOCXHeaderFooterSections(items []map[string]any, hfTexts map[string]bool) []map[string]any {
+ if len(hfTexts) == 0 {
+ return items
+ }
+ filtered := items[:0]
+ for _, item := range items {
+ text := itemText(item)
+ normalized := strings.TrimSpace(whitespaceCollapsePattern.ReplaceAllString(text, " "))
+ if normalized != "" && hfTexts[normalized] {
+ continue
+ }
+ filtered = append(filtered, item)
+ }
+ return filtered
+}
diff --git a/internal/parser/parser/docx_postprocess_test.go b/internal/parser/parser/docx_postprocess_test.go
new file mode 100644
index 0000000000..0afa44f0f5
--- /dev/null
+++ b/internal/parser/parser/docx_postprocess_test.go
@@ -0,0 +1,188 @@
+package parser
+
+import (
+ "archive/zip"
+ "bytes"
+ "encoding/json"
+ "reflect"
+ "testing"
+)
+
+// helper: build a minimal docx ZIP in-memory with given header/footer XML.
+func buildDocxZIP(t *testing.T, headerText, footerText string) []byte {
+ t.Helper()
+ buf := new(bytes.Buffer)
+ zw := zip.NewWriter(buf)
+ // Minimal required parts for a valid docx: [Content_Types].xml + word/document.xml.
+ mustWrite := func(name, body string) {
+ w, err := zw.Create(name)
+ if err != nil {
+ t.Fatal(err)
+ }
+ _, _ = w.Write([]byte(body))
+ }
+ mustWrite("[Content_Types].xml", ` `)
+ mustWrite("word/document.xml", `Body text `)
+ if headerText != "" {
+ mustWrite("word/header1.xml", ``+headerText+` `)
+ }
+ if footerText != "" {
+ mustWrite("word/footer1.xml", ``+footerText+` `)
+ }
+ if err := zw.Close(); err != nil {
+ t.Fatal(err)
+ }
+ return buf.Bytes()
+}
+
+func TestExtractDOCXHeaderFooterTexts(t *testing.T) {
+ data := buildDocxZIP(t, "Company Confidential", "Page 1")
+ texts := extractDOCXHeaderFooterTexts(data)
+ if !texts["Company Confidential"] {
+ t.Error("missing header text 'Company Confidential'")
+ }
+ if !texts["Page 1"] {
+ t.Error("missing footer text 'Page 1'")
+ }
+}
+
+func TestExtractDOCXHeaderFooterTexts_NoHeaders(t *testing.T) {
+ data := buildDocxZIP(t, "", "")
+ texts := extractDOCXHeaderFooterTexts(data)
+ if len(texts) != 0 {
+ t.Errorf("expected empty map, got %v", texts)
+ }
+}
+
+func TestExtractDOCXHeaderFooterTexts_Normalization(t *testing.T) {
+ // Whitespace runs should be collapsed and trimmed.
+ data := buildDocxZIP(t, " Multiple Spaces ", "")
+ texts := extractDOCXHeaderFooterTexts(data)
+ if !texts["Multiple Spaces"] {
+ t.Errorf("expected normalized 'Multiple Spaces', got %v", texts)
+ }
+}
+
+// TestRemoveDOCXHeaderFooterSections filters JSON items whose
+// normalized text exactly matches a header/footer text.
+func TestRemoveDOCXHeaderFooterSections(t *testing.T) {
+ items := []map[string]any{
+ {"text": "Company Confidential"},
+ {"text": "Real content"},
+ {"text": "Page 1"},
+ {"text": "More content"},
+ }
+ hfTexts := map[string]bool{"Company Confidential": true, "Page 1": true}
+ got := removeDOCXHeaderFooterSections(items, hfTexts)
+ want := []map[string]any{
+ {"text": "Real content"},
+ {"text": "More content"},
+ }
+ if !reflect.DeepEqual(got, want) {
+ t.Errorf("got %+v, want %+v", got, want)
+ }
+}
+
+// TestRemoveDOCXHeaderFooterSections_EmptyHF keeps all items when no
+// header/footer texts are provided.
+func TestRemoveDOCXHeaderFooterSections_EmptyHF(t *testing.T) {
+ items := []map[string]any{{"text": "A"}, {"text": "B"}}
+ got := removeDOCXHeaderFooterSections(items, nil)
+ if !reflect.DeepEqual(got, items) {
+ t.Errorf("expected unchanged, got %+v", got)
+ }
+}
+
+// TestExtractDOCXOutlines verifies heading elements are extracted
+// from the office_oxide IR as (title, level) outlines.
+func TestExtractDOCXOutlines(t *testing.T) {
+ irJSON := `{"sections":[{"title":"","elements":[
+ {"type":"heading","level":1,"content":[{"type":"text","text":"第一章 概述"}]},
+ {"type":"paragraph","content":[{"type":"text","text":"正文"}]},
+ {"type":"heading","level":2,"content":[{"type":"text","text":"1.1 背景"}]}
+ ]}]}`
+ outlines := extractDOCXOutlines(irJSON)
+ want := []docxOutline{{Title: "第一章 概述", Level: 0}, {Title: "1.1 背景", Level: 1}}
+ if !reflect.DeepEqual(outlines, want) {
+ t.Errorf("got %+v, want %+v", outlines, want)
+ }
+}
+
+// TestRemoveTOCWord_NoOutlines delegates to removeContentsTable when
+// no outlines are available (mirrors Python utils.py:263-264).
+func TestRemoveTOCWord_NoOutlines(t *testing.T) {
+ items := []map[string]any{
+ {"text": "前言"},
+ {"text": "目录"},
+ {"text": "第一章 概述"},
+ {"text": "正文"},
+ }
+ got := removeTOCWord(items, nil, false)
+ want := []map[string]any{
+ {"text": "前言"},
+ {"text": "正文"},
+ }
+ if !reflect.DeepEqual(got, want) {
+ t.Errorf("got %+v, want %+v", got, want)
+ }
+}
+
+// TestRemoveTOCWord_WithOutlines deletes the TOC heading and
+// following entries that match outline-title prefixes or the
+// "dots + page number" regex (mirrors Python utils.py:115-144).
+func TestRemoveTOCWord_WithOutlines(t *testing.T) {
+ items := []map[string]any{
+ {"text": "前言"},
+ {"text": "目录"},
+ {"text": "第一章 概述"},
+ {"text": "第一章 概述 .......... 1"},
+ {"text": "第二章 方法 .......... 5"},
+ {"text": "正文开始"},
+ }
+ outlines := []docxOutline{
+ {Title: "第一章 概述", Level: 0},
+ {Title: "第二章 方法", Level: 0},
+ }
+ got := removeTOCWord(items, outlines, false)
+ // "目录" heading + "第一章 概述" (outline prefix match) +
+ // "第一章 概述 .......... 1" (prefix match) +
+ // "第二章 方法 .......... 5" (prefix match + dot+page regex) removed.
+ // "前言" and "正文开始" survive.
+ want := []map[string]any{
+ {"text": "前言"},
+ {"text": "正文开始"},
+ }
+ if !reflect.DeepEqual(got, want) {
+ t.Errorf("got %+v, want %+v", got, want)
+ }
+}
+
+// TestRemoveTOCWord_NoTOCHeading keeps items unchanged when no TOC
+// heading is found (only removeContentsTable fallback runs).
+func TestRemoveTOCWord_NoTOCHeading(t *testing.T) {
+ items := []map[string]any{
+ {"text": "第一章 概述"},
+ {"text": "正文"},
+ }
+ outlines := []docxOutline{{Title: "第一章 概述", Level: 0}}
+ got := removeTOCWord(items, outlines, false)
+ // No "目录/Contents" heading → outline-prefix loop skipped;
+ // removeContentsTable fallback finds no TOC heading either.
+ want := []map[string]any{
+ {"text": "第一章 概述"},
+ {"text": "正文"},
+ }
+ if !reflect.DeepEqual(got, want) {
+ t.Errorf("got %+v, want %+v", got, want)
+ }
+}
+
+func TestDocxOutline_JSONRoundtrip(t *testing.T) {
+ o := docxOutline{Title: "Heading 1", Level: 2}
+ b, _ := json.Marshal(o)
+ var got docxOutline
+ _ = json.Unmarshal(b, &got)
+ if !reflect.DeepEqual(got, o) {
+ t.Errorf("roundtrip got %+v, want %+v", got, o)
+ }
+}
diff --git a/internal/parser/parser/html_parser.go b/internal/parser/parser/html_parser.go
index 315195838e..a6c831bdc2 100644
--- a/internal/parser/parser/html_parser.go
+++ b/internal/parser/parser/html_parser.go
@@ -25,7 +25,10 @@ import (
"golang.org/x/net/html"
)
-type HTMLParser struct{}
+type HTMLParser struct {
+ RemoveHeaderFooter bool
+ RemoveTOC bool
+}
func NewHTMLParser() *HTMLParser {
return &HTMLParser{}
@@ -35,6 +38,21 @@ func (p *HTMLParser) String() string {
return "HTMLParser"
}
+// ConfigureFromSetup reads the HTML family setup map. Mirrors the
+// Python parser.py HTML setup keys: remove_header_footer (pre-parse
+// tag strip) and remove_toc (post-parse text heuristic).
+func (p *HTMLParser) ConfigureFromSetup(setup map[string]any) {
+ if p == nil || setup == nil {
+ return
+ }
+ if v, ok := setup["remove_header_footer"].(bool); ok {
+ p.RemoveHeaderFooter = v
+ }
+ if v, ok := setup["remove_toc"].(bool); ok {
+ p.RemoveTOC = v
+ }
+}
+
// ParseWithResult emits one item per block-level HTML element
// (headings, paragraphs, lists, pre blocks). The walker is a
// pure-Go replacement for the previous `fmt.Printf` debug output:
@@ -49,12 +67,27 @@ func (p *HTMLParser) String() string {
// separate ck_type — the python HtmlParser collapses inline
// formatting into the parent block's text.
func (p *HTMLParser) ParseWithResult(ctx context.Context, filename string, data []byte) ParseResult {
+ // remove_header_footer: pre-parse strip of / tags
+ // and ARIA role=banner/contentinfo elements (mirrors Python
+ // parser.py:1083-1084 remove_header_footer_html_blob).
+ if p.RemoveHeaderFooter {
+ cleaned, err := stripHTMLHeaderFooter(data)
+ if err != nil {
+ return ParseResult{Err: fmt.Errorf("html remove_header_footer: %w", err)}
+ }
+ data = cleaned
+ }
doc, err := html.Parse(bytes.NewReader(data))
if err != nil {
return ParseResult{Err: fmt.Errorf("html parse: %w", err)}
}
var items []map[string]any
walkHTMLBlocks(doc, &items)
+ // remove_toc: post-parse text heuristic (mirrors Python
+ // parser.py:1087-1088 remove_toc → remove_contents_table).
+ if p.RemoveTOC {
+ items = removeContentsTable(items, false)
+ }
if items == nil {
items = []map[string]any{{"text": "", "doc_type_kwd": "text"}}
}
diff --git a/internal/parser/parser/html_parser_test.go b/internal/parser/parser/html_parser_test.go
new file mode 100644
index 0000000000..6110678611
--- /dev/null
+++ b/internal/parser/parser/html_parser_test.go
@@ -0,0 +1,138 @@
+package parser
+
+import (
+ "context"
+ "strings"
+ "testing"
+)
+
+func TestHTMLParser_ConfigureFromSetup(t *testing.T) {
+ p := NewHTMLParser()
+ p.ConfigureFromSetup(map[string]any{
+ "remove_header_footer": true,
+ "remove_toc": true,
+ })
+ if !p.RemoveHeaderFooter {
+ t.Error("RemoveHeaderFooter = false, want true")
+ }
+ if !p.RemoveTOC {
+ t.Error("RemoveTOC = false, want true")
+ }
+}
+
+func TestHTMLParser_ConfigureFromSetup_NilSafe(t *testing.T) {
+ p := NewHTMLParser()
+ p.ConfigureFromSetup(nil) // should not panic
+ if p.RemoveHeaderFooter || p.RemoveTOC {
+ t.Error("flags should stay false after nil setup")
+ }
+}
+
+// TestHTMLParser_RemoveHeaderFooter verifies that /
+// tags and role="banner"/role="contentinfo" elements are stripped
+// before parsing (mirrors Python parser.py:1083-1084 pre-parse).
+func TestHTMLParser_RemoveHeaderFooter(t *testing.T) {
+
+ p := NewHTMLParser()
+ p.ConfigureFromSetup(map[string]any{"remove_header_footer": true})
+
+ html := `
+
+
+Main content here
+Ad banner
+
+Legal info
+
+`
+
+ res := p.ParseWithResult(context.Background(), "test.html", []byte(html))
+ if res.Err != nil {
+ t.Fatalf("ParseWithResult: %v", res.Err)
+ }
+ for _, item := range res.JSON {
+ text, _ := item["text"].(string)
+ for _, banned := range []string{"Site Navigation", "Copyright 2024", "Ad banner", "Legal info"} {
+ if strings.Contains(text, banned) {
+ t.Errorf("header/footer text %q leaked into output item: %q", banned, text)
+ }
+ }
+ }
+ // Main content must survive.
+ found := false
+ for _, item := range res.JSON {
+ if text, _ := item["text"].(string); strings.Contains(text, "Main content here") {
+ found = true
+ }
+ }
+ if !found {
+ t.Error("main content was stripped along with header/footer")
+ }
+}
+
+// TestHTMLParser_RemoveHeaderFooter_DisabledByDefault verifies that
+// without the flag, / content is preserved.
+func TestHTMLParser_RemoveHeaderFooter_DisabledByDefault(t *testing.T) {
+
+ p := NewHTMLParser()
+
+ html := `Body
`
+ res := p.ParseWithResult(context.Background(), "test.html", []byte(html))
+ if res.Err != nil {
+ t.Fatalf("ParseWithResult: %v", res.Err)
+ }
+ found := false
+ for _, item := range res.JSON {
+ if text, _ := item["text"].(string); strings.Contains(text, "Nav") {
+ found = true
+ }
+ }
+ if !found {
+ t.Error("header text should be preserved when flag is off")
+ }
+}
+
+// TestHTMLParser_RemoveTOC verifies that TOC sections are filtered
+// after parsing (mirrors Python parser.py:1087-1088 post-parse).
+func TestHTMLParser_RemoveTOC(t *testing.T) {
+
+ p := NewHTMLParser()
+ p.ConfigureFromSetup(map[string]any{"remove_toc": true})
+
+ html := `
+前言
+目录
+第一章 概述
+第二章 方法
+正文开始
+`
+
+ res := p.ParseWithResult(context.Background(), "test.html", []byte(html))
+ if res.Err != nil {
+ t.Fatalf("ParseWithResult: %v", res.Err)
+ }
+ // "目录" heading and "第一章 概述" (first entry after heading)
+ // should be removed by removeContentsTable.
+ for _, item := range res.JSON {
+ text, _ := item["text"].(string)
+ if text == "目录" {
+ t.Errorf("TOC heading %q was not removed", text)
+ }
+ if strings.HasPrefix(text, "第一章") {
+ t.Errorf("first TOC entry %q was not removed", text)
+ }
+ }
+ // "前言" and "正文开始" should survive.
+ wantTexts := map[string]bool{"前言": false, "正文开始": false}
+ for _, item := range res.JSON {
+ text, _ := item["text"].(string)
+ if _, ok := wantTexts[text]; ok {
+ wantTexts[text] = true
+ }
+ }
+ for text, found := range wantTexts {
+ if !found {
+ t.Errorf("expected %q to survive, but it was removed", text)
+ }
+ }
+}
diff --git a/internal/parser/parser/html_postprocess.go b/internal/parser/parser/html_postprocess.go
new file mode 100644
index 0000000000..cdeab1d01b
--- /dev/null
+++ b/internal/parser/parser/html_postprocess.go
@@ -0,0 +1,51 @@
+package parser
+
+import (
+ "bytes"
+
+ "golang.org/x/net/html"
+)
+
+// stripHTMLHeaderFooter removes / elements and nodes
+// with role="banner"/role="contentinfo" from the HTML byte stream,
+// then re-serializes. Mirrors Python rag/flow/parser/utils.py:70-74
+// remove_header_footer_html_blob (BeautifulSoup decompose).
+func stripHTMLHeaderFooter(data []byte) ([]byte, error) {
+ doc, err := html.Parse(bytes.NewReader(data))
+ if err != nil {
+ return nil, err
+ }
+ removeHeaderFooterNodes(doc)
+ var buf bytes.Buffer
+ if err := html.Render(&buf, doc); err != nil {
+ return nil, err
+ }
+ return buf.Bytes(), nil
+}
+
+// removeHeaderFooterNodes walks the parsed tree and detaches every
+// / element and every element whose role attribute
+// is "banner" or "contentinfo".
+func removeHeaderFooterNodes(n *html.Node) {
+ var next *html.Node
+ for child := n.FirstChild; child != nil; child = next {
+ next = child.NextSibling
+ if child.Type == html.ElementNode && isHeaderFooterNode(child) {
+ n.RemoveChild(child)
+ continue
+ }
+ removeHeaderFooterNodes(child)
+ }
+}
+
+func isHeaderFooterNode(n *html.Node) bool {
+ if n.Data == "header" || n.Data == "footer" {
+ return true
+ }
+ for _, attr := range n.Attr {
+ if attr.Key == "role" && (attr.Val == "banner" || attr.Val == "contentinfo") {
+ return true
+ }
+ }
+ return false
+}
diff --git a/internal/parser/parser/markdown_parser.go b/internal/parser/parser/markdown_parser.go
index 55ce6b5b51..1457f5de45 100644
--- a/internal/parser/parser/markdown_parser.go
+++ b/internal/parser/parser/markdown_parser.go
@@ -49,10 +49,11 @@ const GoMarkdown = "go_markdown"
var ssrfAllowLoopback bool
type MarkdownParser struct {
- libType string
- ParseMethod string
- OutputFormat string
- VLM map[string]any
+ libType string
+ ParseMethod string
+ OutputFormat string
+ VLM map[string]any
+ FlattenMediaToText bool
}
func NewMarkdownParser(libType string) (*MarkdownParser, error) {
@@ -79,6 +80,9 @@ func (p *MarkdownParser) ConfigureFromSetup(setup map[string]any) {
if v, ok := setup["vlm"].(map[string]any); ok {
p.VLM = v
}
+ if v, ok := setup["flatten_media_to_text"].(bool); ok {
+ p.FlattenMediaToText = v
+ }
}
// ParseWithResult implements ParseResultProducer (plan §6.5) and
@@ -94,7 +98,7 @@ func (p *MarkdownParser) ParseWithResult(ctx context.Context, filename string, d
rawText := string(data)
var items []map[string]any
- walkMarkdownBlocksWithImages(doc, rawText, &items)
+ walkMarkdownBlocksWithImages(doc, rawText, &items, p.FlattenMediaToText)
if items == nil {
items = []map[string]any{{"text": "", "doc_type_kwd": "text"}}
}
@@ -123,8 +127,10 @@ func markdownNew() *parser.Parser {
// emitted with their text. When a block contains a markdown image
// reference (), the image data is resolved via
// resolveMarkdownImage and the item carries `doc_type_kwd: "image"`
-// together with the base64-encoded image payload.
-func walkMarkdownBlocksWithImages(doc ast.Node, rawText string, out *[]map[string]any) {
+// together with the base64-encoded image payload. When flatten is
+// true, all items are forced to doc_type_kwd="text" (mirrors Python
+// parser.py:1034 flatten_media_to_text).
+func walkMarkdownBlocksWithImages(doc ast.Node, rawText string, out *[]map[string]any, flatten bool) {
for _, child := range doc.GetChildren() {
var ckType string
var docTypeKwd string
@@ -166,9 +172,13 @@ func walkMarkdownBlocksWithImages(doc ast.Node, rawText string, out *[]map[strin
// Detect markdown image references in the raw source text
// that corresponds to this block. When found, resolve the
// image data so downstream vision enhancement can describe it.
+ // When flatten is true, keep doc_type_kwd="text" (Python
+ // parser.py:1034: flatten_media_to_text overrides image).
if imgData, imgFound := resolveMarkdownImage(txt, rawText); imgFound && imgData != "" {
- item["doc_type_kwd"] = "image"
item["image"] = imgData
+ if !flatten {
+ item["doc_type_kwd"] = "image"
+ }
}
*out = append(*out, item)
diff --git a/internal/parser/parser/markdown_parser_test.go b/internal/parser/parser/markdown_parser_test.go
index b0c18c71ec..8247710ddd 100644
--- a/internal/parser/parser/markdown_parser_test.go
+++ b/internal/parser/parser/markdown_parser_test.go
@@ -118,6 +118,44 @@ func TestMarkdownParser_ConfigureFromSetup_NilSafe(t *testing.T) {
}
}
+// TestMarkdownParser_FlattenMediaToText verifies that when
+// flatten_media_to_text is true, image items are emitted with
+// doc_type_kwd="text" (mirroring Python parser.py:1034). When false,
+// image items keep doc_type_kwd="image".
+func TestMarkdownParser_FlattenMediaToText(t *testing.T) {
+ ctx := t.Context()
+ pixelB64 := base64.StdEncoding.EncodeToString([]byte("fake-png-data"))
+ md := "Some text with an image\n\n"
+
+ cases := []struct {
+ name string
+ flatten bool
+ wantDocType string
+ }{
+ {"flatten=false keeps image", false, "image"},
+ {"flatten=true forces text", true, "text"},
+ }
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ p, _ := NewMarkdownParser(GoMarkdown)
+ p.ConfigureFromSetup(map[string]any{"flatten_media_to_text": tc.flatten})
+ if p.FlattenMediaToText != tc.flatten {
+ t.Fatalf("FlattenMediaToText field = %v, want %v", p.FlattenMediaToText, tc.flatten)
+ }
+ res := p.ParseWithResult(ctx, "test.md", []byte(md))
+ if res.Err != nil {
+ t.Fatalf("ParseWithResult: %v", res.Err)
+ }
+ for _, item := range res.JSON {
+ if kd, _ := item["doc_type_kwd"].(string); kd != tc.wantDocType {
+ t.Errorf("doc_type_kwd = %q, want %q (item text=%q)",
+ kd, tc.wantDocType, item["text"])
+ }
+ }
+ })
+ }
+}
+
func TestResolveMarkdownImage_DataURI(t *testing.T) {
b64 := base64.StdEncoding.EncodeToString([]byte("fakeimage"))
md := ""
diff --git a/internal/parser/parser/text_toc.go b/internal/parser/parser/text_toc.go
new file mode 100644
index 0000000000..bfdedd8db3
--- /dev/null
+++ b/internal/parser/parser/text_toc.go
@@ -0,0 +1,107 @@
+package parser
+
+import (
+ "regexp"
+ "strings"
+)
+
+// tocHeadingPattern matches TOC heading text. Mirrors Python
+// rag/nlp/__init__.py:945 — case-insensitive match against
+// contents/目录/目次/table of contents/致谢/acknowledge after
+// stripping all spaces (ASCII + fullwidth).
+var tocHeadingPattern = regexp.MustCompile(`(?i)^(contents|目录|目次|table of contents|致谢|acknowledge)$`)
+
+// whitespacePattern matches ASCII spaces and fullwidth space (U+3000)
+// for normalization before heading match.
+var whitespacePattern = regexp.MustCompile("[ \t\u3000]+")
+
+// itemText extracts the text field from a parser item map. Falls
+// back to empty string when the field is missing or not a string.
+func itemText(item map[string]any) string {
+ if s, ok := item["text"].(string); ok {
+ return strings.TrimSpace(s)
+ }
+ return ""
+}
+
+// removeContentsTable drops the table-of-contents entries from a list
+// of parser items. Mirrors Python rag/nlp/__init__.py:937-965
+// (remove_contents_table).
+//
+// Algorithm:
+// 1. Scan for a TOC heading (contents/目录/目次/...).
+// 2. Drop the heading.
+// 3. Take the next entry's prefix (first 3 chars for CJK, first 2
+// words for English) as the TOC entry pattern.
+// 4. Drop that entry, then drop up to 128 following entries that
+// start with the prefix.
+// 5. Non-matching entries are kept.
+func removeContentsTable(items []map[string]any, eng bool) []map[string]any {
+ i := 0
+ for i < len(items) {
+ text := itemText(items[i])
+ // Strip @@ suffix and whitespace before matching.
+ normalized := whitespacePattern.ReplaceAllString(strings.SplitN(text, "@@", 2)[0], "")
+ if !tocHeadingPattern.MatchString(normalized) {
+ i++
+ continue
+ }
+ // Drop the TOC heading.
+ items = append(items[:i], items[i+1:]...)
+ if i >= len(items) {
+ break
+ }
+ // Determine the prefix from the entry right after the heading.
+ prefix := tocPrefix(itemText(items[i]), eng)
+ for prefix == "" {
+ items = append(items[:i], items[i+1:]...)
+ if i >= len(items) {
+ break
+ }
+ prefix = tocPrefix(itemText(items[i]), eng)
+ }
+ if i >= len(items) || prefix == "" {
+ break
+ }
+ // Drop the first TOC entry (the one that supplied the prefix).
+ items = append(items[:i], items[i+1:]...)
+ if i >= len(items) {
+ break
+ }
+ // Drop up to 128 following entries that start with prefix.
+ limit := i + 128
+ if limit > len(items) {
+ limit = len(items)
+ }
+ for j := i; j < limit; j++ {
+ if !strings.HasPrefix(itemText(items[j]), prefix) {
+ continue
+ }
+ // Drop entries [i, j).
+ items = append(items[:i], items[j:]...)
+ break
+ }
+ }
+ return items
+}
+
+// tocPrefix returns the TOC-entry prefix: first 3 chars for CJK,
+// first 2 whitespace-separated words for English. Mirrors Python
+// remove_contents_table line 951.
+func tocPrefix(text string, eng bool) string {
+ if eng {
+ words := strings.Fields(text)
+ if len(words) >= 2 {
+ return words[0] + " " + words[1]
+ }
+ if len(words) == 1 {
+ return words[0]
+ }
+ return ""
+ }
+ runes := []rune(text)
+ if len(runes) < 3 {
+ return string(runes)
+ }
+ return string(runes[:3])
+}
diff --git a/internal/parser/parser/text_toc_test.go b/internal/parser/parser/text_toc_test.go
new file mode 100644
index 0000000000..a4c3a66a63
--- /dev/null
+++ b/internal/parser/parser/text_toc_test.go
@@ -0,0 +1,108 @@
+package parser
+
+import (
+ "reflect"
+ "testing"
+)
+
+// TestRemoveContentsTable mirrors the Python remove_contents_table
+// (rag/nlp/__init__.py:937-965). It locates a TOC heading
+// (contents/目录/目次/...), drops it, then drops the following
+// entries that share a common prefix with the first TOC entry.
+func TestRemoveContentsTable(t *testing.T) {
+ cases := []struct {
+ name string
+ items []map[string]any
+ eng bool
+ want []map[string]any
+ }{
+ {
+ name: "no TOC heading → unchanged",
+ items: []map[string]any{{"text": "Intro"}, {"text": "Body"}},
+ eng: false,
+ want: []map[string]any{{"text": "Intro"}, {"text": "Body"}},
+ },
+ {
+ name: "Chinese 目录 heading + first TOC entry removed",
+ items: []map[string]any{
+ {"text": "前言"},
+ {"text": "目录"},
+ {"text": "第一章 概述"},
+ {"text": "第二章 方法"},
+ {"text": "正文开始"},
+ },
+ eng: false,
+ // Python remove_contents_table drops the heading and the
+ // first entry after it (prefix "第一章"). Subsequent
+ // entries with a different prefix ("第二章") are kept
+ // because re.match("第一章", "第二章 方法") fails.
+ want: []map[string]any{
+ {"text": "前言"},
+ {"text": "第二章 方法"},
+ {"text": "正文开始"},
+ },
+ },
+ {
+ name: "English Contents heading + first TOC entry removed",
+ items: []map[string]any{
+ {"text": "Intro"},
+ {"text": "Contents"},
+ {"text": "Chapter 1 Overview"},
+ {"text": "Chapter 2 Method"},
+ {"text": "Body starts"},
+ },
+ eng: true,
+ want: []map[string]any{
+ {"text": "Intro"},
+ {"text": "Chapter 2 Method"},
+ {"text": "Body starts"},
+ },
+ },
+ {
+ name: "目录 heading with no following entry → heading dropped only",
+ items: []map[string]any{
+ {"text": "前言"},
+ {"text": "目录"},
+ },
+ eng: false,
+ want: []map[string]any{
+ {"text": "前言"},
+ },
+ },
+ {
+ name: "prefix-matching subsequent entries collapse non-matching gap",
+ items: []map[string]any{
+ {"text": "前言"},
+ {"text": "目录"},
+ {"text": "第一节 概述"},
+ {"text": "杂项"},
+ {"text": "第一节 背景"},
+ {"text": "正文"},
+ },
+ eng: false,
+ // prefix="第一节"; after dropping heading + first entry,
+ // scan finds "第一节 背景" at j=2 (gap "杂项" at j=1);
+ // Python deletes [i, j) = the gap, keeps the match.
+ want: []map[string]any{
+ {"text": "前言"},
+ {"text": "第一节 背景"},
+ {"text": "正文"},
+ },
+ },
+ {
+ name: "empty input → unchanged",
+ items: nil,
+ eng: false,
+ want: nil,
+ },
+ }
+
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ got := removeContentsTable(tc.items, tc.eng)
+ if !reflect.DeepEqual(got, tc.want) {
+ t.Errorf("got %v, want %v", got, tc.want)
+ }
+ })
+ }
+}