mirror of
https://github.com/infiniflow/ragflow.git
synced 2026-07-25 18:03:29 +08:00
fix: honor parser params and image VLM system_prompt in Go ingestion (#17334)
## Summary Fix the Go ingestion pipeline so that several parser setup switches and the image VLM prompt are actually honored end-to-end (previously the DSL fields existed but the Go code never read them). - **DOCX** (`docx_parser.go`, `docx_postprocess.go`): read `remove_toc` and `remove_header_footer`; apply to both JSON and markdown output paths (outline-based TOC removal with a text-heuristic fallback, plus header/footer section filtering). - **HTML** (`html_parser.go`, `html_postprocess.go`, `text_toc.go`): read `remove_header_footer` (pre-parse strip of `<header>`/`<footer>` and ARIA `banner`/`contentinfo`) and `remove_toc` (post-parse `remove_contents_table` heuristic). - **Markdown** (`markdown_parser.go`): read `flatten_media_to_text` and force media blocks to text when enabled. - **Image VLM** (`media_dispatch.go`): read `system_prompt` instead of `prompt` so the user-configured image VLM prompt is no longer silently dropped (`prompt` remains the video family key). All flags are wired through `ConfigureFromSetup`, which the dispatch layer already invokes for every family, so the behavior is live rather than dead code. ## Test plan - New unit tests: `docx_postprocess_test.go`, `html_parser_test.go`, `text_toc_test.go`, `markdown_parser_test.go`, `media_dispatch_test.go`. - `bash build.sh --test ./internal/parser/parser/... ./internal/ingestion/component/...` ## Notes - The `File` component is excluded from this migration scope. - Relates to the Python→Go parity diff (Parser 1.8–1.11, 1.15).
This commit is contained in:
@@ -212,7 +212,10 @@ func maybeDispatchImage(
|
|||||||
dataURI := "data:" + mimeType + ";base64," + imageB64
|
dataURI := "data:" + mimeType + ";base64," + imageB64
|
||||||
|
|
||||||
prompt := "Describe this image in detail."
|
prompt := "Describe this image in detail."
|
||||||
if v, ok := setup["prompt"].(string); ok && v != "" {
|
// image family's contract key is system_prompt (parser.go:295),
|
||||||
|
// mirroring Python parser.py:1119. Do NOT read setup["prompt"]
|
||||||
|
// here — that key is for the video family, not image.
|
||||||
|
if v, ok := setup["system_prompt"].(string); ok && v != "" {
|
||||||
prompt = v
|
prompt = v
|
||||||
}
|
}
|
||||||
messages := []modelModule.Message{{
|
messages := []modelModule.Message{{
|
||||||
|
|||||||
118
internal/ingestion/component/media_dispatch_test.go
Normal file
118
internal/ingestion/component/media_dispatch_test.go
Normal file
@@ -0,0 +1,118 @@
|
|||||||
|
//
|
||||||
|
// Copyright 2026 The InfiniFlow Authors. All Rights Reserved.
|
||||||
|
//
|
||||||
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
// you may not use this file except in compliance with the License.
|
||||||
|
// You may obtain a copy of the License at
|
||||||
|
//
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, software
|
||||||
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
// See the License for the specific language governing permissions and
|
||||||
|
// limitations under the License.
|
||||||
|
|
||||||
|
package component
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"sync"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"ragflow/internal/common"
|
||||||
|
"ragflow/internal/entity"
|
||||||
|
modelModule "ragflow/internal/entity/models"
|
||||||
|
"ragflow/internal/utility"
|
||||||
|
)
|
||||||
|
|
||||||
|
// imagePromptCaptureDriver embeds ModelDriver so it satisfies the interface
|
||||||
|
// without listing every method; only ChatWithMessages is overridden to record
|
||||||
|
// the messages the image branch selected.
|
||||||
|
type imagePromptCaptureDriver struct {
|
||||||
|
modelModule.ModelDriver
|
||||||
|
mu sync.Mutex
|
||||||
|
captured []modelModule.Message
|
||||||
|
}
|
||||||
|
|
||||||
|
func (d *imagePromptCaptureDriver) ChatWithMessages(ctx context.Context, modelName string, messages []modelModule.Message, apiConfig *modelModule.APIConfig, chatModelConfig *modelModule.ChatConfig, usage *common.ModelUsage) (*modelModule.ChatResponse, error) {
|
||||||
|
d.mu.Lock()
|
||||||
|
d.captured = append(d.captured, messages...)
|
||||||
|
d.mu.Unlock()
|
||||||
|
ans := "captured"
|
||||||
|
return &modelModule.ChatResponse{Answer: &ans}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// firstUserText extracts the text of the first "text" content part from the
|
||||||
|
// first captured user message. It scans parts by the "type" discriminator so
|
||||||
|
// the test stays valid regardless of part ordering (image_url may precede text).
|
||||||
|
func firstUserText(msgs []modelModule.Message) (string, bool) {
|
||||||
|
if len(msgs) == 0 {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
parts, ok := msgs[0].Content.([]interface{})
|
||||||
|
if !ok || len(parts) == 0 {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
for _, raw := range parts {
|
||||||
|
part, ok := raw.(map[string]any)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if part["type"] != "text" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if txt, ok := part["text"].(string); ok {
|
||||||
|
return txt, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestMaybeDispatchImage_UsesSystemPrompt pins that the image branch reads the
|
||||||
|
// `system_prompt` setup key (Python/DSL contract), not `prompt`. Before the fix
|
||||||
|
// the branch read setup["prompt"] which is empty for image, so the VLM received
|
||||||
|
// the hardcoded default instead of the user-configured value.
|
||||||
|
func TestMaybeDispatchImage_UsesSystemPrompt(t *testing.T) {
|
||||||
|
origResolver := resolveTenantModelByType
|
||||||
|
defer func() { resolveTenantModelByType = origResolver }()
|
||||||
|
|
||||||
|
drv := &imagePromptCaptureDriver{}
|
||||||
|
resolveTenantModelByType = func(tenantID string, modelType entity.ModelType) (modelModule.ModelDriver, string, *modelModule.APIConfig, int, error) {
|
||||||
|
return drv, "img-model", &modelModule.APIConfig{}, 0, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
setups := defaultSetups()
|
||||||
|
// image family's contract key is system_prompt (parser.go:295).
|
||||||
|
// Also set a legacy `prompt` sentinel to assert system_prompt wins
|
||||||
|
// when both keys are present (regression guard against re-reading
|
||||||
|
// setup["prompt"], which is the video-family key, not image).
|
||||||
|
setups["image"]["prompt"] = "legacy prompt"
|
||||||
|
setups["image"]["system_prompt"] = "自定义视觉提示"
|
||||||
|
|
||||||
|
res, dispatched, err := maybeDispatchImage(
|
||||||
|
context.Background(),
|
||||||
|
utility.FileTypeVISUAL,
|
||||||
|
"test.png",
|
||||||
|
[]byte("not-a-real-image"),
|
||||||
|
map[string]any{"tenant_id": "t1"},
|
||||||
|
setups,
|
||||||
|
)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("maybeDispatchImage: %v", err)
|
||||||
|
}
|
||||||
|
if !dispatched {
|
||||||
|
t.Fatalf("expected dispatched=true for VISUAL file")
|
||||||
|
}
|
||||||
|
if res.Text == "" {
|
||||||
|
t.Fatalf("expected non-empty combined text")
|
||||||
|
}
|
||||||
|
|
||||||
|
got, ok := firstUserText(drv.captured)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("no user text captured in VLM messages: %#v", drv.captured)
|
||||||
|
}
|
||||||
|
if got != "自定义视觉提示" {
|
||||||
|
t.Fatalf("VLM user text = %q, want %q (image branch must read system_prompt)", got, "自定义视觉提示")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -204,7 +204,8 @@ func TestPipelineExecutor_Run_RealPDF_WritesAndReadsBackFromElasticsearch(t *tes
|
|||||||
pdfPath := filepath.Join(taskRepoRoot(t), "internal", "deepdoc", "parser", "pdf", "testdata", "pdfs", "01_english_simple.pdf")
|
pdfPath := filepath.Join(taskRepoRoot(t), "internal", "deepdoc", "parser", "pdf", "testdata", "pdfs", "01_english_simple.pdf")
|
||||||
pdfBytes, err := os.ReadFile(pdfPath)
|
pdfBytes, err := os.ReadFile(pdfPath)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("read pdf fixture: %v", err)
|
t.Skipf("read pdf fixture %s: %v", pdfPath, err)
|
||||||
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
suffix := fmt.Sprintf("%d", time.Now().UnixNano())
|
suffix := fmt.Sprintf("%d", time.Now().UnixNano())
|
||||||
|
|||||||
444
internal/parser/parser/docx_ir.go
Normal file
444
internal/parser/parser/docx_ir.go
Normal file
@@ -0,0 +1,444 @@
|
|||||||
|
//
|
||||||
|
// Copyright 2026 The InfiniFlow Authors. All Rights Reserved.
|
||||||
|
//
|
||||||
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
// you may not use this file except in compliance with the License.
|
||||||
|
// You may obtain a copy of the License at
|
||||||
|
//
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, software
|
||||||
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
// See the License for the specific language governing permissions and
|
||||||
|
// limitations under the License.
|
||||||
|
//
|
||||||
|
|
||||||
|
// Package parser: this file holds the pure-Go office_oxide IR data
|
||||||
|
// model and JSON helpers for DOCX. It is intentionally cgo-free so
|
||||||
|
// that postprocessing (docx_postprocess.go) and unit tests run
|
||||||
|
// without the office_oxide native library. The cgo boundary lives
|
||||||
|
// entirely in docx_parser.go (officeOxide.OpenFromBytes).
|
||||||
|
|
||||||
|
package parser
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/base64"
|
||||||
|
"encoding/json"
|
||||||
|
"html"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// DOCXFigure represents one embedded image plus its surrounding text
|
||||||
|
// context, mirroring the chunk-level shape that Python's
|
||||||
|
// naive_merge_docx produces for vision_figure_parser_docx_wrapper_naive.
|
||||||
|
type DOCXFigure struct {
|
||||||
|
Image string `json:"image"` // base64-encoded image bytes
|
||||||
|
ContextAbove string `json:"context_above"` // text before the image block
|
||||||
|
ContextBelow string `json:"context_below"` // text after the image block
|
||||||
|
Marker string `json:"marker"` // substring to locate image position in markdown
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- office_oxide IR types (local copy, independent of deepdoc) ---
|
||||||
|
|
||||||
|
type docxIRDocument struct {
|
||||||
|
Sections []docxIRSection `json:"sections"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type docxIRSection struct {
|
||||||
|
Title string `json:"title"`
|
||||||
|
Elements []docxIRElement `json:"elements"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type docxIRElement struct {
|
||||||
|
Type string `json:"type"` // "paragraph", "heading", "table", "image", "list", "text_box", ...
|
||||||
|
Level int `json:"level"` // heading level (1-6) or list nesting level
|
||||||
|
Style string `json:"style"` // Word style name (e.g. "Normal", "Heading 1")
|
||||||
|
Content json.RawMessage `json:"content"` // rich text runs or block-level content; decoded per type
|
||||||
|
Data []byte `json:"data"` // raw image bytes (for "image" type)
|
||||||
|
Rows []docxIRRow `json:"rows"` // table rows
|
||||||
|
Ordered bool `json:"ordered"` // true=numbered list, false=bullet list (for "list" type)
|
||||||
|
Items []docxIRListItem `json:"items"` // list items (for "list" type)
|
||||||
|
}
|
||||||
|
|
||||||
|
// contentRuns decodes Content as flat text runs (paragraph/heading type).
|
||||||
|
func (e docxIRElement) contentRuns() []docxIRRun {
|
||||||
|
var runs []docxIRRun
|
||||||
|
if len(e.Content) > 0 {
|
||||||
|
_ = json.Unmarshal(e.Content, &runs)
|
||||||
|
}
|
||||||
|
return runs
|
||||||
|
}
|
||||||
|
|
||||||
|
// contentBlocks decodes Content as block-level elements (text_box type).
|
||||||
|
func (e docxIRElement) contentBlocks() []docxIRElement {
|
||||||
|
var blocks []docxIRElement
|
||||||
|
if len(e.Content) > 0 {
|
||||||
|
_ = json.Unmarshal(e.Content, &blocks)
|
||||||
|
}
|
||||||
|
return blocks
|
||||||
|
}
|
||||||
|
|
||||||
|
// docxIRListItem represents one item in an ordered/unordered list.
|
||||||
|
type docxIRListItem struct {
|
||||||
|
Content []docxIRElement `json:"content"` // block-level content (typically a single Paragraph)
|
||||||
|
Nested *docxIRList `json:"nested,omitempty"` // optional nested sub-list; null/absent when none
|
||||||
|
}
|
||||||
|
|
||||||
|
// docxIRList mirrors office_oxide's ir::List. Only Items is needed for
|
||||||
|
// text extraction; the remaining List fields (ordered, start_number,
|
||||||
|
// style, level) are not consumed here, so they are not modeled.
|
||||||
|
type docxIRList struct {
|
||||||
|
Items []docxIRListItem `json:"items"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type docxIRRun struct {
|
||||||
|
Type string `json:"type"` // "text", "image"
|
||||||
|
Text string `json:"text"`
|
||||||
|
Content []docxIRElement `json:"content"` // nested elements (used in table cells)
|
||||||
|
}
|
||||||
|
|
||||||
|
type docxIRRow struct {
|
||||||
|
Cells []docxIRCell `json:"cells"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type docxIRCell struct {
|
||||||
|
Content []docxIRElement `json:"content"` // nested paragraphs inside table cell
|
||||||
|
}
|
||||||
|
|
||||||
|
func joinDOCXIRRuns(runs []docxIRRun) string {
|
||||||
|
var b strings.Builder
|
||||||
|
for _, r := range runs {
|
||||||
|
if r.Type == "text" {
|
||||||
|
b.WriteString(r.Text)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// extractTextFromListItem extracts the plain text content from a list item.
|
||||||
|
// Each list item contains block-level elements (typically a Paragraph),
|
||||||
|
// whose text runs are concatenated. Nested sub-lists (multi-level
|
||||||
|
// bullets/numbered items) are decoded and recursed so their text is not
|
||||||
|
// silently dropped. Mirrors office_oxide ir::ListItem { content, nested }.
|
||||||
|
func extractTextFromListItem(item docxIRListItem) string {
|
||||||
|
var parts []string
|
||||||
|
for _, el := range item.Content {
|
||||||
|
if el.Type == "paragraph" || el.Type == "heading" {
|
||||||
|
t := joinDOCXIRRuns(el.contentRuns())
|
||||||
|
if t != "" {
|
||||||
|
parts = append(parts, t)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if item.Nested != nil {
|
||||||
|
for _, sub := range item.Nested.Items {
|
||||||
|
if t := extractTextFromListItem(sub); t != "" {
|
||||||
|
parts = append(parts, t)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(parts) == 0 {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
return strings.TrimSpace(strings.Join(parts, "\n"))
|
||||||
|
}
|
||||||
|
|
||||||
|
// extractTextFromBlockElements extracts text from a slice of block-level
|
||||||
|
// elements (paragraphs/headings), used by text_box and other compound
|
||||||
|
// element types.
|
||||||
|
func extractTextFromBlockElements(blocks []docxIRElement) string {
|
||||||
|
var parts []string
|
||||||
|
for _, el := range blocks {
|
||||||
|
if el.Type == "paragraph" || el.Type == "heading" {
|
||||||
|
t := joinDOCXIRRuns(el.contentRuns())
|
||||||
|
if t != "" {
|
||||||
|
parts = append(parts, t)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(parts) == 0 {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
return strings.TrimSpace(strings.Join(parts, "\n"))
|
||||||
|
}
|
||||||
|
|
||||||
|
// joinCellText concatenates all paragraph texts inside a table cell,
|
||||||
|
// joined by newlines.
|
||||||
|
func joinCellText(cell docxIRCell) string {
|
||||||
|
var parts []string
|
||||||
|
for _, el := range cell.Content {
|
||||||
|
if text := joinDOCXIRRuns(el.contentRuns()); text != "" {
|
||||||
|
parts = append(parts, text)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return strings.Join(parts, "\n")
|
||||||
|
}
|
||||||
|
|
||||||
|
// docxIRTableToHTML converts a table IR element to an HTML table string.
|
||||||
|
func docxIRTableToHTML(el docxIRElement) string {
|
||||||
|
var sb strings.Builder
|
||||||
|
sb.WriteString("<table>")
|
||||||
|
for _, row := range el.Rows {
|
||||||
|
sb.WriteString("<tr>")
|
||||||
|
for _, cell := range row.Cells {
|
||||||
|
sb.WriteString("<td>")
|
||||||
|
sb.WriteString(html.EscapeString(joinCellText(cell)))
|
||||||
|
sb.WriteString("</td>")
|
||||||
|
}
|
||||||
|
sb.WriteString("</tr>")
|
||||||
|
}
|
||||||
|
sb.WriteString("</table>")
|
||||||
|
return sb.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// docxElementText returns the plain-text rendering of any supported
|
||||||
|
// IR element type. Used by extractDOCXFiguresFromIR so that tables,
|
||||||
|
// lists, and text boxes contribute to image surrounding context
|
||||||
|
// instead of becoming empty flatBlocks (which would drop adjacent
|
||||||
|
// VLM context). Returns "" for image and unknown types.
|
||||||
|
func docxElementText(el docxIRElement) string {
|
||||||
|
switch el.Type {
|
||||||
|
case "paragraph", "heading":
|
||||||
|
return joinDOCXIRRuns(el.contentRuns())
|
||||||
|
case "table":
|
||||||
|
var lines []string
|
||||||
|
for _, row := range el.Rows {
|
||||||
|
for _, cell := range row.Cells {
|
||||||
|
if t := joinCellText(cell); t != "" {
|
||||||
|
lines = append(lines, t)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return strings.Join(lines, "\n")
|
||||||
|
case "list":
|
||||||
|
var lines []string
|
||||||
|
for _, item := range el.Items {
|
||||||
|
if t := extractTextFromListItem(item); t != "" {
|
||||||
|
lines = append(lines, t)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return strings.Join(lines, "\n")
|
||||||
|
case "text_box":
|
||||||
|
return extractTextFromBlockElements(el.contentBlocks())
|
||||||
|
default:
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// buildDOCXJSONSections converts an office_oxide IR JSON string into a
|
||||||
|
// slice of structured items compatible with the chunker's JSON input
|
||||||
|
// contract. Each item carries at least text and doc_type_kwd.
|
||||||
|
func buildDOCXJSONSections(irJSON string) []map[string]any {
|
||||||
|
var ir docxIRDocument
|
||||||
|
if err := json.Unmarshal([]byte(irJSON), &ir); err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
var sections []map[string]any
|
||||||
|
for _, sec := range ir.Sections {
|
||||||
|
for _, el := range sec.Elements {
|
||||||
|
switch el.Type {
|
||||||
|
case "paragraph", "heading":
|
||||||
|
text := joinDOCXIRRuns(el.contentRuns())
|
||||||
|
if strings.TrimSpace(text) == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
item := map[string]any{
|
||||||
|
"text": text,
|
||||||
|
"image": nil,
|
||||||
|
"doc_type_kwd": "text",
|
||||||
|
}
|
||||||
|
if el.Type == "heading" {
|
||||||
|
item["ck_type"] = "heading"
|
||||||
|
}
|
||||||
|
sections = append(sections, item)
|
||||||
|
|
||||||
|
case "image":
|
||||||
|
b64 := base64.StdEncoding.EncodeToString(el.Data)
|
||||||
|
sections = append(sections, map[string]any{
|
||||||
|
"text": "",
|
||||||
|
"image": b64,
|
||||||
|
"doc_type_kwd": "image",
|
||||||
|
})
|
||||||
|
|
||||||
|
case "table":
|
||||||
|
html := docxIRTableToHTML(el)
|
||||||
|
if html == "<table></table>" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
sections = append(sections, map[string]any{
|
||||||
|
"text": html,
|
||||||
|
"image": nil,
|
||||||
|
"doc_type_kwd": "table",
|
||||||
|
})
|
||||||
|
|
||||||
|
case "list":
|
||||||
|
for _, item := range el.Items {
|
||||||
|
text := extractTextFromListItem(item)
|
||||||
|
if text == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
sections = append(sections, map[string]any{
|
||||||
|
"text": text,
|
||||||
|
"image": nil,
|
||||||
|
"doc_type_kwd": "text",
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
case "text_box":
|
||||||
|
text := extractTextFromBlockElements(el.contentBlocks())
|
||||||
|
if text == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
sections = append(sections, map[string]any{
|
||||||
|
"text": text,
|
||||||
|
"image": nil,
|
||||||
|
"doc_type_kwd": "text",
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return sections
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- figure extraction (used by the cgo parser path) ---
|
||||||
|
|
||||||
|
// extractDOCXFiguresFromIR parses the office_oxide IR JSON and
|
||||||
|
// returns every embedded image block together with the plain text
|
||||||
|
// immediately surrounding it. The context matches what Python's
|
||||||
|
// naive_merge_docx attaches as context_above / context_below on
|
||||||
|
// each chunk that carries an image.
|
||||||
|
//
|
||||||
|
// Reuses the IR already obtained from the doc handle in
|
||||||
|
// ParseWithResult so the binary is not opened twice.
|
||||||
|
func extractDOCXFiguresFromIR(irJSON string) []DOCXFigure {
|
||||||
|
var ir docxIRDocument
|
||||||
|
if err := json.Unmarshal([]byte(irJSON), &ir); err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
var flat []flatBlock
|
||||||
|
for _, sec := range ir.Sections {
|
||||||
|
for _, el := range sec.Elements {
|
||||||
|
if el.Type == "image" {
|
||||||
|
b64 := base64.StdEncoding.EncodeToString(el.Data)
|
||||||
|
flat = append(flat, flatBlock{image: b64})
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
text := docxElementText(el)
|
||||||
|
flat = append(flat, flatBlock{text: text})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
var figures []DOCXFigure
|
||||||
|
for i, block := range flat {
|
||||||
|
if block.image == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
fig := DOCXFigure{Image: block.image}
|
||||||
|
|
||||||
|
// Collect text above (backward scan up to docxContextWindow
|
||||||
|
// chars, or until another image is hit).
|
||||||
|
above := collectDOCXPrevText(flat, i, 512)
|
||||||
|
fig.ContextAbove = strings.TrimSpace(above)
|
||||||
|
|
||||||
|
// Collect text below (forward scan up to docxContextWindow
|
||||||
|
// chars, or until another image is hit).
|
||||||
|
below := collectDOCXNextText(flat, i, 512)
|
||||||
|
fig.ContextBelow = strings.TrimSpace(below)
|
||||||
|
|
||||||
|
// Marker: text of the immediately preceding flat block,
|
||||||
|
// used by the vision dispatcher to locate the image position
|
||||||
|
// in the rendered markdown for inline insertion.
|
||||||
|
for j := i - 1; j >= 0; j-- {
|
||||||
|
if flat[j].text != "" {
|
||||||
|
fig.Marker = flat[j].text
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
figures = append(figures, fig)
|
||||||
|
}
|
||||||
|
return figures
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- internal types ---
|
||||||
|
|
||||||
|
// flatBlock is a flattened IR element used internally to collect
|
||||||
|
// text / image context around embedded figures.
|
||||||
|
type flatBlock struct {
|
||||||
|
text string
|
||||||
|
image string // base64-encoded image data (empty for non-image)
|
||||||
|
}
|
||||||
|
|
||||||
|
const docxContextWindow = 512
|
||||||
|
|
||||||
|
func collectDOCXPrevText(flat []flatBlock, idx, maxLen int) string {
|
||||||
|
var parts []string
|
||||||
|
remaining := maxLen
|
||||||
|
for i := idx - 1; i >= 0 && remaining > 0; i-- {
|
||||||
|
if flat[i].image != "" {
|
||||||
|
break // stop at previous image
|
||||||
|
}
|
||||||
|
if flat[i].text == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
r := []rune(flat[i].text)
|
||||||
|
if len(r) > remaining {
|
||||||
|
r = r[len(r)-remaining:]
|
||||||
|
}
|
||||||
|
parts = append([]string{string(r)}, parts...)
|
||||||
|
remaining -= len(r)
|
||||||
|
}
|
||||||
|
// parts are in document order (farthest first, closest last);
|
||||||
|
// keep the tail = text nearest the image when truncating for the
|
||||||
|
// newline separators that Join inserts (they are not counted by
|
||||||
|
// `remaining`, so the joined length can otherwise exceed maxLen).
|
||||||
|
joined := strings.Join(parts, "\n")
|
||||||
|
if r := []rune(joined); len(r) > maxLen {
|
||||||
|
joined = string(r[len(r)-maxLen:])
|
||||||
|
}
|
||||||
|
return joined
|
||||||
|
}
|
||||||
|
|
||||||
|
func collectDOCXNextText(flat []flatBlock, idx, maxLen int) string {
|
||||||
|
var parts []string
|
||||||
|
remaining := maxLen
|
||||||
|
for i := idx + 1; i < len(flat) && remaining > 0; i++ {
|
||||||
|
if flat[i].image != "" {
|
||||||
|
break // stop at next image
|
||||||
|
}
|
||||||
|
if flat[i].text == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
r := []rune(flat[i].text)
|
||||||
|
if len(r) > remaining {
|
||||||
|
r = r[:remaining]
|
||||||
|
}
|
||||||
|
parts = append(parts, string(r))
|
||||||
|
remaining -= len(r)
|
||||||
|
}
|
||||||
|
// parts are in document order (closest first, farthest last);
|
||||||
|
// keep the head = text nearest the image when truncating for the
|
||||||
|
// newline separators that Join inserts.
|
||||||
|
joined := strings.Join(parts, "\n")
|
||||||
|
if r := []rune(joined); len(r) > maxLen {
|
||||||
|
joined = string(r[:maxLen])
|
||||||
|
}
|
||||||
|
return joined
|
||||||
|
}
|
||||||
|
|
||||||
|
// buildFiguresMap converts the internal DOCXFigure slice to the
|
||||||
|
// map form attached to fileMeta["figures"].
|
||||||
|
func buildFiguresMap(figures []DOCXFigure) []map[string]any {
|
||||||
|
figs := make([]map[string]any, 0, len(figures))
|
||||||
|
for _, f := range figures {
|
||||||
|
figs = append(figs, map[string]any{
|
||||||
|
"image": f.Image,
|
||||||
|
"context_above": f.ContextAbove,
|
||||||
|
"context_below": f.ContextBelow,
|
||||||
|
"marker": f.Marker,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
return figs
|
||||||
|
}
|
||||||
253
internal/parser/parser/docx_ir_test.go
Normal file
253
internal/parser/parser/docx_ir_test.go
Normal file
@@ -0,0 +1,253 @@
|
|||||||
|
package parser
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/json"
|
||||||
|
"reflect"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestBuildDOCXJSONSections_FromJSON feeds an office_oxide IR JSON
|
||||||
|
// string directly into buildDOCXJSONSections and asserts the
|
||||||
|
// paragraph / heading / image / table / list branches. This pins the
|
||||||
|
// pure-Go IR → sections transform without requiring the office_oxide
|
||||||
|
// native library, so it runs under CGO_ENABLED=0.
|
||||||
|
func TestBuildDOCXJSONSections_FromJSON(t *testing.T) {
|
||||||
|
// "data":"aGVsbG8=" is base64 for "hello"; buildDOCXJSONSections
|
||||||
|
// re-encodes el.Data, so the item's "image" must equal "aGVsbG8=".
|
||||||
|
irJSON := `{"sections":[{"title":"","elements":[
|
||||||
|
{"type":"heading","level":1,"content":[{"type":"text","text":"Title"}]},
|
||||||
|
{"type":"paragraph","content":[{"type":"text","text":"Hello"}]},
|
||||||
|
{"type":"image","data":"aGVsbG8="},
|
||||||
|
{"type":"table","rows":[{"cells":[{"content":[{"type":"paragraph","content":[{"type":"text","text":"cell"}]}]}]}]},
|
||||||
|
{"type":"list","items":[{"content":[{"type":"paragraph","content":[{"type":"text","text":"item1"}]}]}]}
|
||||||
|
]}]}`
|
||||||
|
|
||||||
|
got := buildDOCXJSONSections(irJSON)
|
||||||
|
want := []map[string]any{
|
||||||
|
{"text": "Title", "image": nil, "doc_type_kwd": "text", "ck_type": "heading"},
|
||||||
|
{"text": "Hello", "image": nil, "doc_type_kwd": "text"},
|
||||||
|
{"text": "", "image": "aGVsbG8=", "doc_type_kwd": "image"},
|
||||||
|
{"text": "<table><tr><td>cell</td></tr></table>", "image": nil, "doc_type_kwd": "table"},
|
||||||
|
{"text": "item1", "image": nil, "doc_type_kwd": "text"},
|
||||||
|
}
|
||||||
|
if !reflect.DeepEqual(got, want) {
|
||||||
|
t.Fatalf("buildDOCXJSONSections mismatch:\n got: %+v\nwant: %+v", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestBuildDOCXJSONSections_EmptyTableSkipped verifies that a table
|
||||||
|
// with no rows renders as "<table></table>" and is dropped by the
|
||||||
|
// "if html == \"<table></table>\" { continue }" guard.
|
||||||
|
func TestBuildDOCXJSONSections_EmptyTableSkipped(t *testing.T) {
|
||||||
|
irJSON := `{"sections":[{"elements":[
|
||||||
|
{"type":"paragraph","content":[{"type":"text","text":"keep"}]},
|
||||||
|
{"type":"table","rows":[]}
|
||||||
|
]}]}`
|
||||||
|
got := buildDOCXJSONSections(irJSON)
|
||||||
|
if len(got) != 1 || got[0]["text"] != "keep" {
|
||||||
|
t.Fatalf("expected only the paragraph to survive, got %+v", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestJoinDOCXIRRuns pins that only text-type runs are concatenated;
|
||||||
|
// non-text runs (e.g. nested image runs) are skipped.
|
||||||
|
func TestJoinDOCXIRRuns(t *testing.T) {
|
||||||
|
runs := []docxIRRun{
|
||||||
|
{Type: "text", Text: "Hello"},
|
||||||
|
{Type: "image", Text: "ignored"},
|
||||||
|
{Type: "text", Text: " World"},
|
||||||
|
}
|
||||||
|
if got := joinDOCXIRRuns(runs); got != "Hello World" {
|
||||||
|
t.Fatalf("joinDOCXIRRuns = %q, want %q", got, "Hello World")
|
||||||
|
}
|
||||||
|
if got := joinDOCXIRRuns(nil); got != "" {
|
||||||
|
t.Fatalf("joinDOCXIRRuns(nil) = %q, want empty", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestExtractDOCXFiguresFromIR verifies the image-figure context
|
||||||
|
// extraction: an image block carries the immediately surrounding text
|
||||||
|
// as ContextAbove / ContextBelow / Marker. Pure-Go, runs under !cgo.
|
||||||
|
func TestExtractDOCXFiguresFromIR(t *testing.T) {
|
||||||
|
irJSON := `{"sections":[{"elements":[
|
||||||
|
{"type":"paragraph","content":[{"type":"text","text":"before"}]},
|
||||||
|
{"type":"image","data":"aGVsbG8="},
|
||||||
|
{"type":"paragraph","content":[{"type":"text","text":"after"}]}
|
||||||
|
]}]}`
|
||||||
|
|
||||||
|
figs := extractDOCXFiguresFromIR(irJSON)
|
||||||
|
if len(figs) != 1 {
|
||||||
|
t.Fatalf("expected 1 figure, got %d", len(figs))
|
||||||
|
}
|
||||||
|
fig := figs[0]
|
||||||
|
want := DOCXFigure{
|
||||||
|
Image: "aGVsbG8=",
|
||||||
|
ContextAbove: "before",
|
||||||
|
ContextBelow: "after",
|
||||||
|
Marker: "before",
|
||||||
|
}
|
||||||
|
if !reflect.DeepEqual(fig, want) {
|
||||||
|
t.Fatalf("figure mismatch:\n got: %+v\nwant: %+v", fig, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestExtractDOCXFiguresFromIR_NoImage returns nil when the IR has no
|
||||||
|
// image blocks.
|
||||||
|
func TestExtractDOCXFiguresFromIR_NoImage(t *testing.T) {
|
||||||
|
irJSON := `{"sections":[{"elements":[
|
||||||
|
{"type":"paragraph","content":[{"type":"text","text":"only text"}]}
|
||||||
|
]}]}`
|
||||||
|
if figs := extractDOCXFiguresFromIR(irJSON); figs != nil {
|
||||||
|
t.Fatalf("expected nil, got %+v", figs)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestExtractDOCXFiguresFromIR_BadJSON returns nil on unparseable IR,
|
||||||
|
// mirroring the json.Unmarshal error guard.
|
||||||
|
func TestExtractDOCXFiguresFromIR_BadJSON(t *testing.T) {
|
||||||
|
if figs := extractDOCXFiguresFromIR("{not json"); figs != nil {
|
||||||
|
t.Fatalf("expected nil for bad JSON, got %+v", figs)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestExtractDOCXFiguresFromIR_NonParagraphContext verifies that
|
||||||
|
// tables, lists, and text boxes adjacent to an image contribute their
|
||||||
|
// text to ContextAbove / ContextBelow. Previously the flatten loop
|
||||||
|
// called joinDOCXIRRuns(el.contentRuns()) for every non-image element,
|
||||||
|
// which returned "" for table/list/text_box (their Content is not text
|
||||||
|
// runs), silently dropping the surrounding VLM context.
|
||||||
|
func TestExtractDOCXFiguresFromIR_NonParagraphContext(t *testing.T) {
|
||||||
|
irJSON := `{"sections":[{"elements":[
|
||||||
|
{"type":"table","rows":[{"cells":[{"content":[{"type":"paragraph","content":[{"type":"text","text":"table cell"}]}]}]}]},
|
||||||
|
{"type":"image","data":"aGVsbG8="},
|
||||||
|
{"type":"list","items":[{"content":[{"type":"paragraph","content":[{"type":"text","text":"list item"}]}]}]},
|
||||||
|
{"type":"text_box","content":[{"type":"paragraph","content":[{"type":"text","text":"box text"}]}]}
|
||||||
|
]}]}`
|
||||||
|
figs := extractDOCXFiguresFromIR(irJSON)
|
||||||
|
if len(figs) != 1 {
|
||||||
|
t.Fatalf("expected 1 figure, got %d", len(figs))
|
||||||
|
}
|
||||||
|
fig := figs[0]
|
||||||
|
if fig.ContextAbove != "table cell" {
|
||||||
|
t.Errorf("ContextAbove = %q, want %q", fig.ContextAbove, "table cell")
|
||||||
|
}
|
||||||
|
// Below the image: list item then text_box, joined by newline.
|
||||||
|
if fig.ContextBelow != "list item\nbox text" {
|
||||||
|
t.Errorf("ContextBelow = %q, want %q", fig.ContextBelow, "list item\nbox text")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestCollectDOCXText_BoundedByMaxLen pins that the joined context
|
||||||
|
// never exceeds maxLen runes, even though newline separators are
|
||||||
|
// inserted between blocks (they are not counted by the per-block
|
||||||
|
// `remaining` decrement). Previously 512 one-rune blocks produced
|
||||||
|
// 1023 chars (512 runes + 511 separators).
|
||||||
|
func TestCollectDOCXText_BoundedByMaxLen(t *testing.T) {
|
||||||
|
// 600 one-rune text blocks before and after an image block.
|
||||||
|
const maxLen = 512
|
||||||
|
var flat []flatBlock
|
||||||
|
for i := 0; i < 600; i++ {
|
||||||
|
flat = append(flat, flatBlock{text: "x"})
|
||||||
|
}
|
||||||
|
imgIdx := len(flat)
|
||||||
|
flat = append(flat, flatBlock{image: "img"})
|
||||||
|
for i := 0; i < 600; i++ {
|
||||||
|
flat = append(flat, flatBlock{text: "x"})
|
||||||
|
}
|
||||||
|
|
||||||
|
prev := collectDOCXPrevText(flat, imgIdx, maxLen)
|
||||||
|
next := collectDOCXNextText(flat, imgIdx, maxLen)
|
||||||
|
if r := []rune(prev); len(r) > maxLen {
|
||||||
|
t.Errorf("prev context = %d runes, want <= %d", len(r), maxLen)
|
||||||
|
}
|
||||||
|
if r := []rune(next); len(r) > maxLen {
|
||||||
|
t.Errorf("next context = %d runes, want <= %d", len(r), maxLen)
|
||||||
|
}
|
||||||
|
// Both must still carry content (the closest blocks survive).
|
||||||
|
if prev == "" {
|
||||||
|
t.Error("prev context empty; expected some text")
|
||||||
|
}
|
||||||
|
if next == "" {
|
||||||
|
t.Error("next context empty; expected some text")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// rawJSON marshals v and returns it as json.RawMessage, for populating
|
||||||
|
// docxIRElement.Content in tests. Defined here (tagless) so it is
|
||||||
|
// available to both cgo and !cgo test files in package parser.
|
||||||
|
func rawJSON(v any) json.RawMessage {
|
||||||
|
data, _ := json.Marshal(v)
|
||||||
|
return json.RawMessage(data)
|
||||||
|
}
|
||||||
|
|
||||||
|
// para is a test helper returning a paragraph element whose single text
|
||||||
|
// run carries the given string.
|
||||||
|
func para(text string) docxIRElement {
|
||||||
|
return docxIRElement{
|
||||||
|
Type: "paragraph",
|
||||||
|
Content: rawJSON([]docxIRRun{{Type: "text", Text: text}}),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestExtractTextFromListItem_Nested verifies that nested sub-list text
|
||||||
|
// is decoded and recursed, not silently dropped. The IR shape mirrors
|
||||||
|
// office_oxide ir::ListItem { content, nested: Option<List> } where
|
||||||
|
// List { items: [ListItem, ...] } is recursive.
|
||||||
|
func TestExtractTextFromListItem_Nested(t *testing.T) {
|
||||||
|
// Top-level item "Top" carries a nested sub-list with two children
|
||||||
|
// "Child A" and "Child B"; Child B itself has a deeper sub-list
|
||||||
|
// "Grandchild" to exercise multi-level recursion.
|
||||||
|
item := docxIRListItem{
|
||||||
|
Content: []docxIRElement{para("Top")},
|
||||||
|
Nested: &docxIRList{Items: []docxIRListItem{
|
||||||
|
{Content: []docxIRElement{para("Child A")}},
|
||||||
|
{
|
||||||
|
Content: []docxIRElement{para("Child B")},
|
||||||
|
Nested: &docxIRList{Items: []docxIRListItem{
|
||||||
|
{Content: []docxIRElement{para("Grandchild")}},
|
||||||
|
}},
|
||||||
|
},
|
||||||
|
}},
|
||||||
|
}
|
||||||
|
got := extractTextFromListItem(item)
|
||||||
|
want := "Top\nChild A\nChild B\nGrandchild"
|
||||||
|
if got != want {
|
||||||
|
t.Errorf("extractTextFromListItem(nested) = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestExtractTextFromListItem_NestedNull pins that "nested": null (the
|
||||||
|
// serde serialization of Option::None) does not trip recursion and the
|
||||||
|
// item's own text still comes through.
|
||||||
|
func TestExtractTextFromListItem_NestedNull(t *testing.T) {
|
||||||
|
irJSON := `{"sections":[{"elements":[
|
||||||
|
{"type":"list","items":[
|
||||||
|
{"content":[{"type":"paragraph","content":[{"type":"text","text":"only item"}]}],"nested":null}
|
||||||
|
]}
|
||||||
|
]}]}`
|
||||||
|
sections := buildDOCXJSONSections(irJSON)
|
||||||
|
if len(sections) != 1 || sections[0]["text"] != "only item" {
|
||||||
|
t.Fatalf("expected single item \"only item\", got %+v", sections)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestBuildDOCXJSONSections_NestedList verifies end-to-end that a
|
||||||
|
// multi-level list survives into the section output with all levels.
|
||||||
|
func TestBuildDOCXJSONSections_NestedList(t *testing.T) {
|
||||||
|
irJSON := `{"sections":[{"elements":[
|
||||||
|
{"type":"list","items":[
|
||||||
|
{"content":[{"type":"paragraph","content":[{"type":"text","text":"L1"}]}],
|
||||||
|
"nested":{"items":[
|
||||||
|
{"content":[{"type":"paragraph","content":[{"type":"text","text":"L2a"}]}]},
|
||||||
|
{"content":[{"type":"paragraph","content":[{"type":"text","text":"L2b"}]}]}
|
||||||
|
]}}
|
||||||
|
]}
|
||||||
|
]}]}`
|
||||||
|
sections := buildDOCXJSONSections(irJSON)
|
||||||
|
if len(sections) != 1 {
|
||||||
|
t.Fatalf("expected 1 section, got %d: %+v", len(sections), sections)
|
||||||
|
}
|
||||||
|
if got := sections[0]["text"]; got != "L1\nL2a\nL2b" {
|
||||||
|
t.Errorf("nested list text = %q, want %q", got, "L1\nL2a\nL2b")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -20,28 +20,22 @@ package parser
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
"encoding/base64"
|
|
||||||
"encoding/json"
|
|
||||||
"fmt"
|
"fmt"
|
||||||
"html"
|
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
officeOxide "github.com/yfedoseev/office_oxide/go"
|
officeOxide "github.com/yfedoseev/office_oxide/go"
|
||||||
)
|
)
|
||||||
|
|
||||||
// DOCXFigure represents one embedded image plus its surrounding text
|
// DOCXParser is the cgo-backed DOCX parser. It is the only DOCX
|
||||||
// context, mirroring the chunk-level shape that Python's
|
// entrypoint that depends on office_oxide; the IR data model and
|
||||||
// naive_merge_docx produces for vision_figure_parser_docx_wrapper_naive.
|
// postprocessing live in cgo-free files (docx_ir.go, docx_postprocess.go)
|
||||||
type DOCXFigure struct {
|
// so they compile and test without native libraries. The !cgo build
|
||||||
Image string `json:"image"` // base64-encoded image bytes
|
// provides a stub DOCXParser in office_parsers_no_cgo.go.
|
||||||
ContextAbove string `json:"context_above"` // text before the image block
|
|
||||||
ContextBelow string `json:"context_below"` // text after the image block
|
|
||||||
Marker string `json:"marker"` // substring to locate image position in markdown
|
|
||||||
}
|
|
||||||
|
|
||||||
type DOCXParser struct {
|
type DOCXParser struct {
|
||||||
libType string
|
libType string
|
||||||
outputFormat string // from DSL config; "json" or "markdown"
|
outputFormat string // from DSL config; "json" or "markdown"
|
||||||
|
RemoveTOC bool
|
||||||
|
RemoveHeaderFooter bool
|
||||||
}
|
}
|
||||||
|
|
||||||
func NewDOCXParser() *DOCXParser {
|
func NewDOCXParser() *DOCXParser {
|
||||||
@@ -58,6 +52,12 @@ func (p *DOCXParser) ConfigureFromSetup(setup map[string]any) {
|
|||||||
if v, ok := setup["output_format"].(string); ok && v != "" {
|
if v, ok := setup["output_format"].(string); ok && v != "" {
|
||||||
p.outputFormat = v
|
p.outputFormat = v
|
||||||
}
|
}
|
||||||
|
if v, ok := setup["remove_toc"].(bool); ok {
|
||||||
|
p.RemoveTOC = v
|
||||||
|
}
|
||||||
|
if v, ok := setup["remove_header_footer"].(bool); ok {
|
||||||
|
p.RemoveHeaderFooter = v
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ParseWithResult produces structured JSON items (when
|
// ParseWithResult produces structured JSON items (when
|
||||||
@@ -96,6 +96,20 @@ func (p *DOCXParser) ParseWithResult(ctx context.Context, filename string, data
|
|||||||
}
|
}
|
||||||
var sections []map[string]any
|
var sections []map[string]any
|
||||||
sections = buildDOCXJSONSections(irJSON)
|
sections = buildDOCXJSONSections(irJSON)
|
||||||
|
// remove_header_footer: drop sections whose normalized text
|
||||||
|
// matches a docx header/footer entry (mirrors Python
|
||||||
|
// parser.py:889-891 extract_docx_header_footer_texts +
|
||||||
|
// remove_header_footer_docx_sections).
|
||||||
|
if p.RemoveHeaderFooter {
|
||||||
|
hfTexts := extractDOCXHeaderFooterTexts(data)
|
||||||
|
sections = removeDOCXHeaderFooterSections(sections, hfTexts)
|
||||||
|
}
|
||||||
|
// remove_toc: filter TOC entries using heading outlines
|
||||||
|
// (mirrors Python parser.py:892-893 remove_toc_word).
|
||||||
|
if p.RemoveTOC {
|
||||||
|
outlines := extractDOCXOutlines(irJSON)
|
||||||
|
sections = removeTOCWord(sections, outlines, false)
|
||||||
|
}
|
||||||
if len(sections) == 0 {
|
if len(sections) == 0 {
|
||||||
sections = []map[string]any{{"text": "", "doc_type_kwd": "text"}}
|
sections = []map[string]any{{"text": "", "doc_type_kwd": "text"}}
|
||||||
}
|
}
|
||||||
@@ -111,6 +125,38 @@ func (p *DOCXParser) ParseWithResult(ctx context.Context, filename string, data
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return ParseResult{Err: fmt.Errorf("docx to-markdown: %w", err)}
|
return ParseResult{Err: fmt.Errorf("docx to-markdown: %w", err)}
|
||||||
}
|
}
|
||||||
|
// remove_header_footer on markdown: filter lines by exact match
|
||||||
|
// (mirrors Python parser.py:923-926 split lines → filter → rejoin).
|
||||||
|
if p.RemoveHeaderFooter {
|
||||||
|
hfTexts := extractDOCXHeaderFooterTexts(data)
|
||||||
|
lines := strings.Split(md, "\n")
|
||||||
|
lineItems := make([]map[string]any, 0, len(lines))
|
||||||
|
for _, ln := range lines {
|
||||||
|
lineItems = append(lineItems, map[string]any{"text": ln})
|
||||||
|
}
|
||||||
|
lineItems = removeDOCXHeaderFooterSections(lineItems, hfTexts)
|
||||||
|
rebuilt := make([]string, 0, len(lineItems))
|
||||||
|
for _, item := range lineItems {
|
||||||
|
rebuilt = append(rebuilt, itemText(item))
|
||||||
|
}
|
||||||
|
md = strings.Join(rebuilt, "\n")
|
||||||
|
}
|
||||||
|
// remove_toc on markdown: split lines, filter, rejoin
|
||||||
|
// (mirrors Python parser.py:927-928 remove_toc_word on markdown).
|
||||||
|
if p.RemoveTOC && irErr == nil {
|
||||||
|
outlines := extractDOCXOutlines(irJSON)
|
||||||
|
lines := strings.Split(md, "\n")
|
||||||
|
lineItems := make([]map[string]any, 0, len(lines))
|
||||||
|
for _, ln := range lines {
|
||||||
|
lineItems = append(lineItems, map[string]any{"text": ln})
|
||||||
|
}
|
||||||
|
filtered := removeTOCWord(lineItems, outlines, false)
|
||||||
|
rebuilt := make([]string, 0, len(filtered))
|
||||||
|
for _, item := range filtered {
|
||||||
|
rebuilt = append(rebuilt, itemText(item))
|
||||||
|
}
|
||||||
|
md = strings.Join(rebuilt, "\n")
|
||||||
|
}
|
||||||
return ParseResult{
|
return ParseResult{
|
||||||
OutputFormat: "markdown",
|
OutputFormat: "markdown",
|
||||||
File: fileMeta,
|
File: fileMeta,
|
||||||
@@ -118,343 +164,6 @@ func (p *DOCXParser) ParseWithResult(ctx context.Context, filename string, data
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// extractDOCXFiguresFromIR parses the office_oxide IR JSON and
|
|
||||||
// returns every embedded image block together with the plain text
|
|
||||||
// immediately surrounding it. The context matches what Python's
|
|
||||||
// naive_merge_docx attaches as context_above / context_below on
|
|
||||||
// each chunk that carries an image.
|
|
||||||
//
|
|
||||||
// Reuses the IR already obtained from the doc handle in
|
|
||||||
// ParseWithResult so the binary is not opened twice.
|
|
||||||
func extractDOCXFiguresFromIR(irJSON string) []DOCXFigure {
|
|
||||||
var ir docxIRDocument
|
|
||||||
if err := json.Unmarshal([]byte(irJSON), &ir); err != nil {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
var flat []flatBlock
|
|
||||||
for _, sec := range ir.Sections {
|
|
||||||
for _, el := range sec.Elements {
|
|
||||||
if el.Type == "image" {
|
|
||||||
b64 := base64.StdEncoding.EncodeToString(el.Data)
|
|
||||||
flat = append(flat, flatBlock{image: b64})
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
text := joinDOCXIRRuns(el.contentRuns())
|
|
||||||
flat = append(flat, flatBlock{text: text})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
var figures []DOCXFigure
|
|
||||||
for i, block := range flat {
|
|
||||||
if block.image == "" {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
fig := DOCXFigure{Image: block.image}
|
|
||||||
|
|
||||||
// Collect text above (backward scan up to docxContextWindow
|
|
||||||
// chars, or until another image is hit).
|
|
||||||
above := collectDOCXPrevText(flat, i, 512)
|
|
||||||
fig.ContextAbove = strings.TrimSpace(above)
|
|
||||||
|
|
||||||
// Collect text below (forward scan up to docxContextWindow
|
|
||||||
// chars, or until another image is hit).
|
|
||||||
below := collectDOCXNextText(flat, i, 512)
|
|
||||||
fig.ContextBelow = strings.TrimSpace(below)
|
|
||||||
|
|
||||||
// Marker: text of the immediately preceding flat block,
|
|
||||||
// used by the vision dispatcher to locate the image position
|
|
||||||
// in the rendered markdown for inline insertion.
|
|
||||||
for j := i - 1; j >= 0; j-- {
|
|
||||||
if flat[j].text != "" {
|
|
||||||
fig.Marker = flat[j].text
|
|
||||||
break
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
figures = append(figures, fig)
|
|
||||||
}
|
|
||||||
return figures
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- internal types ---
|
|
||||||
|
|
||||||
// flatBlock is a flattened IR element used internally to collect
|
|
||||||
// text / image context around embedded figures.
|
|
||||||
type flatBlock struct {
|
|
||||||
text string
|
|
||||||
image string // base64-encoded image data (empty for non-image)
|
|
||||||
}
|
|
||||||
|
|
||||||
const docxContextWindow = 512
|
|
||||||
|
|
||||||
func collectDOCXPrevText(flat []flatBlock, idx, maxLen int) string {
|
|
||||||
var parts []string
|
|
||||||
remaining := maxLen
|
|
||||||
for i := idx - 1; i >= 0 && remaining > 0; i-- {
|
|
||||||
if flat[i].image != "" {
|
|
||||||
break // stop at previous image
|
|
||||||
}
|
|
||||||
if flat[i].text == "" {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
r := []rune(flat[i].text)
|
|
||||||
if len(r) > remaining {
|
|
||||||
r = r[len(r)-remaining:]
|
|
||||||
}
|
|
||||||
parts = append([]string{string(r)}, parts...)
|
|
||||||
remaining -= len(r)
|
|
||||||
}
|
|
||||||
return strings.Join(parts, "\n")
|
|
||||||
}
|
|
||||||
|
|
||||||
func collectDOCXNextText(flat []flatBlock, idx, maxLen int) string {
|
|
||||||
var parts []string
|
|
||||||
remaining := maxLen
|
|
||||||
for i := idx + 1; i < len(flat) && remaining > 0; i++ {
|
|
||||||
if flat[i].image != "" {
|
|
||||||
break // stop at next image
|
|
||||||
}
|
|
||||||
if flat[i].text == "" {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
r := []rune(flat[i].text)
|
|
||||||
if len(r) > remaining {
|
|
||||||
r = r[:remaining]
|
|
||||||
}
|
|
||||||
parts = append(parts, string(r))
|
|
||||||
remaining -= len(r)
|
|
||||||
}
|
|
||||||
return strings.Join(parts, "\n")
|
|
||||||
}
|
|
||||||
|
|
||||||
func joinDOCXIRRuns(runs []docxIRRun) string {
|
|
||||||
var b strings.Builder
|
|
||||||
for _, r := range runs {
|
|
||||||
if r.Type == "text" {
|
|
||||||
b.WriteString(r.Text)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return b.String()
|
|
||||||
}
|
|
||||||
|
|
||||||
// extractTextFromListItem extracts the plain text content from a list item.
|
|
||||||
// Each list item contains block-level elements (typically a Paragraph),
|
|
||||||
// whose text runs are concatenated.
|
|
||||||
func extractTextFromListItem(item docxIRListItem) string {
|
|
||||||
var parts []string
|
|
||||||
for _, el := range item.Content {
|
|
||||||
if el.Type == "paragraph" || el.Type == "heading" {
|
|
||||||
t := joinDOCXIRRuns(el.contentRuns())
|
|
||||||
if t != "" {
|
|
||||||
parts = append(parts, t)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if len(parts) == 0 {
|
|
||||||
return ""
|
|
||||||
}
|
|
||||||
return strings.TrimSpace(strings.Join(parts, "\n"))
|
|
||||||
}
|
|
||||||
|
|
||||||
// extractTextFromBlockElements extracts text from a slice of block-level
|
|
||||||
// elements (paragraphs/headings), used by text_box and other compound
|
|
||||||
// element types.
|
|
||||||
func extractTextFromBlockElements(blocks []docxIRElement) string {
|
|
||||||
var parts []string
|
|
||||||
for _, el := range blocks {
|
|
||||||
if el.Type == "paragraph" || el.Type == "heading" {
|
|
||||||
t := joinDOCXIRRuns(el.contentRuns())
|
|
||||||
if t != "" {
|
|
||||||
parts = append(parts, t)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if len(parts) == 0 {
|
|
||||||
return ""
|
|
||||||
}
|
|
||||||
return strings.TrimSpace(strings.Join(parts, "\n"))
|
|
||||||
}
|
|
||||||
|
|
||||||
// buildFiguresMap converts the internal DOCXFigure slice to the
|
|
||||||
// map form attached to fileMeta["figures"].
|
|
||||||
func buildFiguresMap(figures []DOCXFigure) []map[string]any {
|
|
||||||
figs := make([]map[string]any, 0, len(figures))
|
|
||||||
for _, f := range figures {
|
|
||||||
figs = append(figs, map[string]any{
|
|
||||||
"image": f.Image,
|
|
||||||
"context_above": f.ContextAbove,
|
|
||||||
"context_below": f.ContextBelow,
|
|
||||||
"marker": f.Marker,
|
|
||||||
})
|
|
||||||
}
|
|
||||||
return figs
|
|
||||||
}
|
|
||||||
|
|
||||||
// joinCellText concatenates all paragraph texts inside a table cell,
|
|
||||||
// joined by newlines.
|
|
||||||
func joinCellText(cell docxIRCell) string {
|
|
||||||
var parts []string
|
|
||||||
for _, el := range cell.Content {
|
|
||||||
if text := joinDOCXIRRuns(el.contentRuns()); text != "" {
|
|
||||||
parts = append(parts, text)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return strings.Join(parts, "\n")
|
|
||||||
}
|
|
||||||
|
|
||||||
// docxIRTableToHTML converts a table IR element to an HTML table string.
|
|
||||||
func docxIRTableToHTML(el docxIRElement) string {
|
|
||||||
var sb strings.Builder
|
|
||||||
sb.WriteString("<table>")
|
|
||||||
for _, row := range el.Rows {
|
|
||||||
sb.WriteString("<tr>")
|
|
||||||
for _, cell := range row.Cells {
|
|
||||||
sb.WriteString("<td>")
|
|
||||||
sb.WriteString(html.EscapeString(joinCellText(cell)))
|
|
||||||
sb.WriteString("</td>")
|
|
||||||
}
|
|
||||||
sb.WriteString("</tr>")
|
|
||||||
}
|
|
||||||
sb.WriteString("</table>")
|
|
||||||
return sb.String()
|
|
||||||
}
|
|
||||||
|
|
||||||
// buildDOCXJSONSections converts an office_oxide IR JSON string into a
|
|
||||||
// slice of structured items compatible with the chunker's JSON input
|
|
||||||
// contract. Each item carries at least text and doc_type_kwd.
|
|
||||||
func buildDOCXJSONSections(irJSON string) []map[string]any {
|
|
||||||
var ir docxIRDocument
|
|
||||||
if err := json.Unmarshal([]byte(irJSON), &ir); err != nil {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
var sections []map[string]any
|
|
||||||
for _, sec := range ir.Sections {
|
|
||||||
for _, el := range sec.Elements {
|
|
||||||
switch el.Type {
|
|
||||||
case "paragraph", "heading":
|
|
||||||
text := joinDOCXIRRuns(el.contentRuns())
|
|
||||||
if strings.TrimSpace(text) == "" {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
item := map[string]any{
|
|
||||||
"text": text,
|
|
||||||
"image": nil,
|
|
||||||
"doc_type_kwd": "text",
|
|
||||||
}
|
|
||||||
if el.Type == "heading" {
|
|
||||||
item["ck_type"] = "heading"
|
|
||||||
}
|
|
||||||
sections = append(sections, item)
|
|
||||||
|
|
||||||
case "image":
|
|
||||||
b64 := base64.StdEncoding.EncodeToString(el.Data)
|
|
||||||
sections = append(sections, map[string]any{
|
|
||||||
"text": "",
|
|
||||||
"image": b64,
|
|
||||||
"doc_type_kwd": "image",
|
|
||||||
})
|
|
||||||
|
|
||||||
case "table":
|
|
||||||
html := docxIRTableToHTML(el)
|
|
||||||
if html == "<table></table>" {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
sections = append(sections, map[string]any{
|
|
||||||
"text": html,
|
|
||||||
"image": nil,
|
|
||||||
"doc_type_kwd": "table",
|
|
||||||
})
|
|
||||||
|
|
||||||
case "list":
|
|
||||||
for _, item := range el.Items {
|
|
||||||
text := extractTextFromListItem(item)
|
|
||||||
if text == "" {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
sections = append(sections, map[string]any{
|
|
||||||
"text": text,
|
|
||||||
"image": nil,
|
|
||||||
"doc_type_kwd": "text",
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|
||||||
case "text_box":
|
|
||||||
text := extractTextFromBlockElements(el.contentBlocks())
|
|
||||||
if text == "" {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
sections = append(sections, map[string]any{
|
|
||||||
"text": text,
|
|
||||||
"image": nil,
|
|
||||||
"doc_type_kwd": "text",
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return sections
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- office_oxide IR types (local copy, independent of deepdoc) ---
|
|
||||||
|
|
||||||
type docxIRDocument struct {
|
|
||||||
Sections []docxIRSection `json:"sections"`
|
|
||||||
}
|
|
||||||
|
|
||||||
type docxIRSection struct {
|
|
||||||
Title string `json:"title"`
|
|
||||||
Elements []docxIRElement `json:"elements"`
|
|
||||||
}
|
|
||||||
|
|
||||||
type docxIRElement struct {
|
|
||||||
Type string `json:"type"` // "paragraph", "heading", "table", "image", "list", "text_box", ...
|
|
||||||
Level int `json:"level"` // heading level (1-6) or list nesting level
|
|
||||||
Style string `json:"style"` // Word style name (e.g. "Normal", "Heading 1")
|
|
||||||
Content json.RawMessage `json:"content"` // rich text runs or block-level content; decoded per type
|
|
||||||
Data []byte `json:"data"` // raw image bytes (for "image" type)
|
|
||||||
Rows []docxIRRow `json:"rows"` // table rows
|
|
||||||
Ordered bool `json:"ordered"` // true=numbered list, false=bullet list (for "list" type)
|
|
||||||
Items []docxIRListItem `json:"items"` // list items (for "list" type)
|
|
||||||
}
|
|
||||||
|
|
||||||
// contentRuns decodes Content as flat text runs (paragraph/heading type).
|
|
||||||
func (e docxIRElement) contentRuns() []docxIRRun {
|
|
||||||
var runs []docxIRRun
|
|
||||||
if len(e.Content) > 0 {
|
|
||||||
_ = json.Unmarshal(e.Content, &runs)
|
|
||||||
}
|
|
||||||
return runs
|
|
||||||
}
|
|
||||||
|
|
||||||
// contentBlocks decodes Content as block-level elements (text_box type).
|
|
||||||
func (e docxIRElement) contentBlocks() []docxIRElement {
|
|
||||||
var blocks []docxIRElement
|
|
||||||
if len(e.Content) > 0 {
|
|
||||||
_ = json.Unmarshal(e.Content, &blocks)
|
|
||||||
}
|
|
||||||
return blocks
|
|
||||||
}
|
|
||||||
|
|
||||||
// docxIRListItem represents one item in an ordered/unordered list.
|
|
||||||
type docxIRListItem struct {
|
|
||||||
Content []docxIRElement `json:"content"` // block-level content (typically a single Paragraph)
|
|
||||||
Nested json.RawMessage `json:"nested,omitempty"` // nested sub-list (stored as raw JSON for now)
|
|
||||||
}
|
|
||||||
|
|
||||||
type docxIRRun struct {
|
|
||||||
Type string `json:"type"` // "text", "image"
|
|
||||||
Text string `json:"text"`
|
|
||||||
Content []docxIRElement `json:"content"` // nested elements (used in table cells)
|
|
||||||
}
|
|
||||||
|
|
||||||
type docxIRRow struct {
|
|
||||||
Cells []docxIRCell `json:"cells"`
|
|
||||||
}
|
|
||||||
|
|
||||||
type docxIRCell struct {
|
|
||||||
Content []docxIRElement `json:"content"` // nested paragraphs inside table cell
|
|
||||||
}
|
|
||||||
|
|
||||||
func (p *DOCXParser) String() string {
|
func (p *DOCXParser) String() string {
|
||||||
return "DOCXParser"
|
return "DOCXParser"
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -25,11 +25,6 @@ import (
|
|||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
|
|
||||||
func rawJSON(v any) json.RawMessage {
|
|
||||||
data, _ := json.Marshal(v)
|
|
||||||
return json.RawMessage(data)
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestBuildDOCXJSONSections_Paragraphs(t *testing.T) {
|
func TestBuildDOCXJSONSections_Paragraphs(t *testing.T) {
|
||||||
ir := `{"sections":[{"title":"","elements":[
|
ir := `{"sections":[{"title":"","elements":[
|
||||||
{"type":"paragraph","content":[{"type":"text","text":"Hello world"}],"style":"Normal"}
|
{"type":"paragraph","content":[{"type":"text","text":"Hello world"}],"style":"Normal"}
|
||||||
|
|||||||
214
internal/parser/parser/docx_postprocess.go
Normal file
214
internal/parser/parser/docx_postprocess.go
Normal file
@@ -0,0 +1,214 @@
|
|||||||
|
package parser
|
||||||
|
|
||||||
|
import (
|
||||||
|
"archive/zip"
|
||||||
|
"bytes"
|
||||||
|
"encoding/json"
|
||||||
|
"encoding/xml"
|
||||||
|
"io"
|
||||||
|
"regexp"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// docxOutline is a heading outline entry extracted from the
|
||||||
|
// office_oxide IR. Mirrors Python (title, level, None) tuples from
|
||||||
|
// rag/flow/parser/utils.py:77-89 extract_word_outlines.
|
||||||
|
type docxOutline struct {
|
||||||
|
Title string `json:"title"`
|
||||||
|
Level int `json:"level"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// dotPageNumberPattern matches ".... 123" / "·····5" style TOC
|
||||||
|
// leader lines. Mirrors Python utils.py:285 regex
|
||||||
|
// (\.{2,}|…{2,}|·{2,}|[ ]{2,})\s*\d+\s*$.
|
||||||
|
var dotPageNumberPattern = regexp.MustCompile(`(\.{2,}|…{2,}|·{2,}|[ ]{2,})\s*\d+\s*$`)
|
||||||
|
|
||||||
|
// extractDOCXOutlines parses the office_oxide IR JSON and returns
|
||||||
|
// heading outlines (title + 0-based level). Mirrors Python
|
||||||
|
// rag/flow/parser/utils.py:77-89 extract_word_outlines.
|
||||||
|
func extractDOCXOutlines(irJSON string) []docxOutline {
|
||||||
|
var ir docxIRDocument
|
||||||
|
if err := json.Unmarshal([]byte(irJSON), &ir); err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
var outlines []docxOutline
|
||||||
|
for _, sec := range ir.Sections {
|
||||||
|
for _, el := range sec.Elements {
|
||||||
|
if el.Type != "heading" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
title := strings.TrimSpace(joinDOCXIRRuns(el.contentRuns()))
|
||||||
|
if title == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
outlines = append(outlines, docxOutline{
|
||||||
|
Title: title,
|
||||||
|
Level: el.Level - 1, // 0-based, matching Python
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return outlines
|
||||||
|
}
|
||||||
|
|
||||||
|
// removeTOCWord filters TOC entries from docx items using heading
|
||||||
|
// outlines. Mirrors Python rag/flow/parser/utils.py:115-144
|
||||||
|
// remove_toc_word.
|
||||||
|
//
|
||||||
|
// Algorithm:
|
||||||
|
// 1. No outlines → delegate to removeContentsTable (text heuristic).
|
||||||
|
// 2. With outlines → locate the "目录/Contents" heading, drop it,
|
||||||
|
// then drop following entries that match an outline-title prefix
|
||||||
|
// or the "dots + page number" regex.
|
||||||
|
// 3. Finally run removeContentsTable as a fallback sweep.
|
||||||
|
func removeTOCWord(items []map[string]any, outlines []docxOutline, eng bool) []map[string]any {
|
||||||
|
if len(outlines) == 0 {
|
||||||
|
return removeContentsTable(items, eng)
|
||||||
|
}
|
||||||
|
|
||||||
|
outlineTitles := make([]string, 0, len(outlines))
|
||||||
|
for _, o := range outlines {
|
||||||
|
t := strings.ToLower(strings.TrimSpace(o.Title))
|
||||||
|
if t != "" {
|
||||||
|
outlineTitles = append(outlineTitles, t)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if len(outlineTitles) > 0 {
|
||||||
|
i := 0
|
||||||
|
for i < len(items) {
|
||||||
|
text := strings.ToLower(strings.TrimSpace(itemText(items[i])))
|
||||||
|
normalized := whitespacePattern.ReplaceAllString(strings.SplitN(text, "@@", 2)[0], "")
|
||||||
|
if !tocHeadingPattern.MatchString(normalized) {
|
||||||
|
i++
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Drop the TOC heading.
|
||||||
|
items = append(items[:i], items[i+1:]...)
|
||||||
|
// Drop following entries matching outline-title prefix
|
||||||
|
// or dot+page-number pattern.
|
||||||
|
for i < len(items) {
|
||||||
|
raw := itemText(items[i])
|
||||||
|
normalized := strings.ToLower(strings.TrimSpace(strings.SplitN(raw, "@@", 2)[0]))
|
||||||
|
if normalized == "" {
|
||||||
|
items = append(items[:i], items[i+1:]...)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
matched := false
|
||||||
|
for _, title := range outlineTitles {
|
||||||
|
if strings.HasPrefix(normalized, title) || strings.HasPrefix(title, normalized) {
|
||||||
|
matched = true
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if matched || dotPageNumberPattern.MatchString(raw) {
|
||||||
|
items = append(items[:i], items[i+1:]...)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
break
|
||||||
|
}
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return removeContentsTable(items, eng)
|
||||||
|
}
|
||||||
|
|
||||||
|
// whitespaceCollapsePattern collapses runs of whitespace (spaces,
|
||||||
|
// tabs, newlines) into a single space — mirrors Python
|
||||||
|
// re.sub(r"\s+", " ", text).strip() used in
|
||||||
|
// extract_docx_header_footer_texts (utils.py:210,229).
|
||||||
|
var whitespaceCollapsePattern = regexp.MustCompile(`\s+`)
|
||||||
|
|
||||||
|
// extractDOCXHeaderFooterTexts opens the docx ZIP, reads every
|
||||||
|
// word/header*.xml and word/footer*.xml part, and returns the set
|
||||||
|
// of normalized text strings found inside <w:t> elements. Mirrors
|
||||||
|
// Python rag/flow/parser/utils.py:38-53
|
||||||
|
// extract_docx_header_footer_texts (python-docx sections[].header/
|
||||||
|
// footer paragraphs + tables).
|
||||||
|
func extractDOCXHeaderFooterTexts(data []byte) map[string]bool {
|
||||||
|
texts := make(map[string]bool)
|
||||||
|
zr, err := zip.NewReader(bytes.NewReader(data), int64(len(data)))
|
||||||
|
if err != nil {
|
||||||
|
return texts
|
||||||
|
}
|
||||||
|
for _, f := range zr.File {
|
||||||
|
name := f.Name
|
||||||
|
if !strings.HasPrefix(name, "word/") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
base := name[len("word/"):]
|
||||||
|
if !strings.HasPrefix(base, "header") && !strings.HasPrefix(base, "footer") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !strings.HasSuffix(base, ".xml") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
rc, err := f.Open()
|
||||||
|
if err != nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
text := extractWtTextFromXML(rc)
|
||||||
|
rc.Close()
|
||||||
|
normalized := strings.TrimSpace(whitespaceCollapsePattern.ReplaceAllString(text, " "))
|
||||||
|
if normalized != "" {
|
||||||
|
texts[normalized] = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return texts
|
||||||
|
}
|
||||||
|
|
||||||
|
// extractWtTextFromXML decodes the XML stream and concatenates the
|
||||||
|
// content of every <w:t> element (the WordprocessingML text node).
|
||||||
|
func extractWtTextFromXML(r io.Reader) string {
|
||||||
|
dec := xml.NewDecoder(r)
|
||||||
|
var sb strings.Builder
|
||||||
|
for {
|
||||||
|
tok, err := dec.Token()
|
||||||
|
if err != nil {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
start, ok := tok.(xml.StartElement)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if start.Name.Local != "t" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// The next token inside <w:t>...</w:t> is CharData.
|
||||||
|
for {
|
||||||
|
inner, err := dec.Token()
|
||||||
|
if err != nil {
|
||||||
|
return sb.String()
|
||||||
|
}
|
||||||
|
switch it := inner.(type) {
|
||||||
|
case xml.CharData:
|
||||||
|
sb.Write(it)
|
||||||
|
case xml.EndElement:
|
||||||
|
if it.Name.Local == "t" {
|
||||||
|
goto nextToken
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
nextToken:
|
||||||
|
}
|
||||||
|
return sb.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// removeDOCXHeaderFooterSections drops items whose normalized text
|
||||||
|
// exactly matches a header/footer text string. Mirrors Python
|
||||||
|
// rag/flow/parser/utils.py:56-67 remove_header_footer_docx_sections.
|
||||||
|
func removeDOCXHeaderFooterSections(items []map[string]any, hfTexts map[string]bool) []map[string]any {
|
||||||
|
if len(hfTexts) == 0 {
|
||||||
|
return items
|
||||||
|
}
|
||||||
|
filtered := items[:0]
|
||||||
|
for _, item := range items {
|
||||||
|
text := itemText(item)
|
||||||
|
normalized := strings.TrimSpace(whitespaceCollapsePattern.ReplaceAllString(text, " "))
|
||||||
|
if normalized != "" && hfTexts[normalized] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
filtered = append(filtered, item)
|
||||||
|
}
|
||||||
|
return filtered
|
||||||
|
}
|
||||||
188
internal/parser/parser/docx_postprocess_test.go
Normal file
188
internal/parser/parser/docx_postprocess_test.go
Normal file
@@ -0,0 +1,188 @@
|
|||||||
|
package parser
|
||||||
|
|
||||||
|
import (
|
||||||
|
"archive/zip"
|
||||||
|
"bytes"
|
||||||
|
"encoding/json"
|
||||||
|
"reflect"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// helper: build a minimal docx ZIP in-memory with given header/footer XML.
|
||||||
|
func buildDocxZIP(t *testing.T, headerText, footerText string) []byte {
|
||||||
|
t.Helper()
|
||||||
|
buf := new(bytes.Buffer)
|
||||||
|
zw := zip.NewWriter(buf)
|
||||||
|
// Minimal required parts for a valid docx: [Content_Types].xml + word/document.xml.
|
||||||
|
mustWrite := func(name, body string) {
|
||||||
|
w, err := zw.Create(name)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
_, _ = w.Write([]byte(body))
|
||||||
|
}
|
||||||
|
mustWrite("[Content_Types].xml", `<?xml version="1.0"?><Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types"><Default Extension="xml" ContentType="application/xml"/><Default Extension="rels" ContentType="application/vnd.openxmlformats-package.relationships+xml"/></Types>`)
|
||||||
|
mustWrite("word/document.xml", `<?xml version="1.0"?><w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main"><w:body><w:p><w:r><w:t>Body text</w:t></w:r></w:p></w:body></w:document>`)
|
||||||
|
if headerText != "" {
|
||||||
|
mustWrite("word/header1.xml", `<?xml version="1.0"?><w:hdr xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main"><w:p><w:r><w:t>`+headerText+`</w:t></w:r></w:p></w:hdr>`)
|
||||||
|
}
|
||||||
|
if footerText != "" {
|
||||||
|
mustWrite("word/footer1.xml", `<?xml version="1.0"?><w:ftr xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main"><w:p><w:r><w:t>`+footerText+`</w:t></w:r></w:p></w:ftr>`)
|
||||||
|
}
|
||||||
|
if err := zw.Close(); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return buf.Bytes()
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestExtractDOCXHeaderFooterTexts(t *testing.T) {
|
||||||
|
data := buildDocxZIP(t, "Company Confidential", "Page 1")
|
||||||
|
texts := extractDOCXHeaderFooterTexts(data)
|
||||||
|
if !texts["Company Confidential"] {
|
||||||
|
t.Error("missing header text 'Company Confidential'")
|
||||||
|
}
|
||||||
|
if !texts["Page 1"] {
|
||||||
|
t.Error("missing footer text 'Page 1'")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestExtractDOCXHeaderFooterTexts_NoHeaders(t *testing.T) {
|
||||||
|
data := buildDocxZIP(t, "", "")
|
||||||
|
texts := extractDOCXHeaderFooterTexts(data)
|
||||||
|
if len(texts) != 0 {
|
||||||
|
t.Errorf("expected empty map, got %v", texts)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestExtractDOCXHeaderFooterTexts_Normalization(t *testing.T) {
|
||||||
|
// Whitespace runs should be collapsed and trimmed.
|
||||||
|
data := buildDocxZIP(t, " Multiple Spaces ", "")
|
||||||
|
texts := extractDOCXHeaderFooterTexts(data)
|
||||||
|
if !texts["Multiple Spaces"] {
|
||||||
|
t.Errorf("expected normalized 'Multiple Spaces', got %v", texts)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRemoveDOCXHeaderFooterSections filters JSON items whose
|
||||||
|
// normalized text exactly matches a header/footer text.
|
||||||
|
func TestRemoveDOCXHeaderFooterSections(t *testing.T) {
|
||||||
|
items := []map[string]any{
|
||||||
|
{"text": "Company Confidential"},
|
||||||
|
{"text": "Real content"},
|
||||||
|
{"text": "Page 1"},
|
||||||
|
{"text": "More content"},
|
||||||
|
}
|
||||||
|
hfTexts := map[string]bool{"Company Confidential": true, "Page 1": true}
|
||||||
|
got := removeDOCXHeaderFooterSections(items, hfTexts)
|
||||||
|
want := []map[string]any{
|
||||||
|
{"text": "Real content"},
|
||||||
|
{"text": "More content"},
|
||||||
|
}
|
||||||
|
if !reflect.DeepEqual(got, want) {
|
||||||
|
t.Errorf("got %+v, want %+v", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRemoveDOCXHeaderFooterSections_EmptyHF keeps all items when no
|
||||||
|
// header/footer texts are provided.
|
||||||
|
func TestRemoveDOCXHeaderFooterSections_EmptyHF(t *testing.T) {
|
||||||
|
items := []map[string]any{{"text": "A"}, {"text": "B"}}
|
||||||
|
got := removeDOCXHeaderFooterSections(items, nil)
|
||||||
|
if !reflect.DeepEqual(got, items) {
|
||||||
|
t.Errorf("expected unchanged, got %+v", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestExtractDOCXOutlines verifies heading elements are extracted
|
||||||
|
// from the office_oxide IR as (title, level) outlines.
|
||||||
|
func TestExtractDOCXOutlines(t *testing.T) {
|
||||||
|
irJSON := `{"sections":[{"title":"","elements":[
|
||||||
|
{"type":"heading","level":1,"content":[{"type":"text","text":"第一章 概述"}]},
|
||||||
|
{"type":"paragraph","content":[{"type":"text","text":"正文"}]},
|
||||||
|
{"type":"heading","level":2,"content":[{"type":"text","text":"1.1 背景"}]}
|
||||||
|
]}]}`
|
||||||
|
outlines := extractDOCXOutlines(irJSON)
|
||||||
|
want := []docxOutline{{Title: "第一章 概述", Level: 0}, {Title: "1.1 背景", Level: 1}}
|
||||||
|
if !reflect.DeepEqual(outlines, want) {
|
||||||
|
t.Errorf("got %+v, want %+v", outlines, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRemoveTOCWord_NoOutlines delegates to removeContentsTable when
|
||||||
|
// no outlines are available (mirrors Python utils.py:263-264).
|
||||||
|
func TestRemoveTOCWord_NoOutlines(t *testing.T) {
|
||||||
|
items := []map[string]any{
|
||||||
|
{"text": "前言"},
|
||||||
|
{"text": "目录"},
|
||||||
|
{"text": "第一章 概述"},
|
||||||
|
{"text": "正文"},
|
||||||
|
}
|
||||||
|
got := removeTOCWord(items, nil, false)
|
||||||
|
want := []map[string]any{
|
||||||
|
{"text": "前言"},
|
||||||
|
{"text": "正文"},
|
||||||
|
}
|
||||||
|
if !reflect.DeepEqual(got, want) {
|
||||||
|
t.Errorf("got %+v, want %+v", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRemoveTOCWord_WithOutlines deletes the TOC heading and
|
||||||
|
// following entries that match outline-title prefixes or the
|
||||||
|
// "dots + page number" regex (mirrors Python utils.py:115-144).
|
||||||
|
func TestRemoveTOCWord_WithOutlines(t *testing.T) {
|
||||||
|
items := []map[string]any{
|
||||||
|
{"text": "前言"},
|
||||||
|
{"text": "目录"},
|
||||||
|
{"text": "第一章 概述"},
|
||||||
|
{"text": "第一章 概述 .......... 1"},
|
||||||
|
{"text": "第二章 方法 .......... 5"},
|
||||||
|
{"text": "正文开始"},
|
||||||
|
}
|
||||||
|
outlines := []docxOutline{
|
||||||
|
{Title: "第一章 概述", Level: 0},
|
||||||
|
{Title: "第二章 方法", Level: 0},
|
||||||
|
}
|
||||||
|
got := removeTOCWord(items, outlines, false)
|
||||||
|
// "目录" heading + "第一章 概述" (outline prefix match) +
|
||||||
|
// "第一章 概述 .......... 1" (prefix match) +
|
||||||
|
// "第二章 方法 .......... 5" (prefix match + dot+page regex) removed.
|
||||||
|
// "前言" and "正文开始" survive.
|
||||||
|
want := []map[string]any{
|
||||||
|
{"text": "前言"},
|
||||||
|
{"text": "正文开始"},
|
||||||
|
}
|
||||||
|
if !reflect.DeepEqual(got, want) {
|
||||||
|
t.Errorf("got %+v, want %+v", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRemoveTOCWord_NoTOCHeading keeps items unchanged when no TOC
|
||||||
|
// heading is found (only removeContentsTable fallback runs).
|
||||||
|
func TestRemoveTOCWord_NoTOCHeading(t *testing.T) {
|
||||||
|
items := []map[string]any{
|
||||||
|
{"text": "第一章 概述"},
|
||||||
|
{"text": "正文"},
|
||||||
|
}
|
||||||
|
outlines := []docxOutline{{Title: "第一章 概述", Level: 0}}
|
||||||
|
got := removeTOCWord(items, outlines, false)
|
||||||
|
// No "目录/Contents" heading → outline-prefix loop skipped;
|
||||||
|
// removeContentsTable fallback finds no TOC heading either.
|
||||||
|
want := []map[string]any{
|
||||||
|
{"text": "第一章 概述"},
|
||||||
|
{"text": "正文"},
|
||||||
|
}
|
||||||
|
if !reflect.DeepEqual(got, want) {
|
||||||
|
t.Errorf("got %+v, want %+v", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDocxOutline_JSONRoundtrip(t *testing.T) {
|
||||||
|
o := docxOutline{Title: "Heading 1", Level: 2}
|
||||||
|
b, _ := json.Marshal(o)
|
||||||
|
var got docxOutline
|
||||||
|
_ = json.Unmarshal(b, &got)
|
||||||
|
if !reflect.DeepEqual(got, o) {
|
||||||
|
t.Errorf("roundtrip got %+v, want %+v", got, o)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -25,7 +25,10 @@ import (
|
|||||||
"golang.org/x/net/html"
|
"golang.org/x/net/html"
|
||||||
)
|
)
|
||||||
|
|
||||||
type HTMLParser struct{}
|
type HTMLParser struct {
|
||||||
|
RemoveHeaderFooter bool
|
||||||
|
RemoveTOC bool
|
||||||
|
}
|
||||||
|
|
||||||
func NewHTMLParser() *HTMLParser {
|
func NewHTMLParser() *HTMLParser {
|
||||||
return &HTMLParser{}
|
return &HTMLParser{}
|
||||||
@@ -35,6 +38,21 @@ func (p *HTMLParser) String() string {
|
|||||||
return "HTMLParser"
|
return "HTMLParser"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ConfigureFromSetup reads the HTML family setup map. Mirrors the
|
||||||
|
// Python parser.py HTML setup keys: remove_header_footer (pre-parse
|
||||||
|
// tag strip) and remove_toc (post-parse text heuristic).
|
||||||
|
func (p *HTMLParser) ConfigureFromSetup(setup map[string]any) {
|
||||||
|
if p == nil || setup == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if v, ok := setup["remove_header_footer"].(bool); ok {
|
||||||
|
p.RemoveHeaderFooter = v
|
||||||
|
}
|
||||||
|
if v, ok := setup["remove_toc"].(bool); ok {
|
||||||
|
p.RemoveTOC = v
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ParseWithResult emits one item per block-level HTML element
|
// ParseWithResult emits one item per block-level HTML element
|
||||||
// (headings, paragraphs, lists, pre blocks). The walker is a
|
// (headings, paragraphs, lists, pre blocks). The walker is a
|
||||||
// pure-Go replacement for the previous `fmt.Printf` debug output:
|
// pure-Go replacement for the previous `fmt.Printf` debug output:
|
||||||
@@ -49,12 +67,27 @@ func (p *HTMLParser) String() string {
|
|||||||
// separate ck_type — the python HtmlParser collapses inline
|
// separate ck_type — the python HtmlParser collapses inline
|
||||||
// formatting into the parent block's text.
|
// formatting into the parent block's text.
|
||||||
func (p *HTMLParser) ParseWithResult(ctx context.Context, filename string, data []byte) ParseResult {
|
func (p *HTMLParser) ParseWithResult(ctx context.Context, filename string, data []byte) ParseResult {
|
||||||
|
// remove_header_footer: pre-parse strip of <header>/<footer> tags
|
||||||
|
// and ARIA role=banner/contentinfo elements (mirrors Python
|
||||||
|
// parser.py:1083-1084 remove_header_footer_html_blob).
|
||||||
|
if p.RemoveHeaderFooter {
|
||||||
|
cleaned, err := stripHTMLHeaderFooter(data)
|
||||||
|
if err != nil {
|
||||||
|
return ParseResult{Err: fmt.Errorf("html remove_header_footer: %w", err)}
|
||||||
|
}
|
||||||
|
data = cleaned
|
||||||
|
}
|
||||||
doc, err := html.Parse(bytes.NewReader(data))
|
doc, err := html.Parse(bytes.NewReader(data))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return ParseResult{Err: fmt.Errorf("html parse: %w", err)}
|
return ParseResult{Err: fmt.Errorf("html parse: %w", err)}
|
||||||
}
|
}
|
||||||
var items []map[string]any
|
var items []map[string]any
|
||||||
walkHTMLBlocks(doc, &items)
|
walkHTMLBlocks(doc, &items)
|
||||||
|
// remove_toc: post-parse text heuristic (mirrors Python
|
||||||
|
// parser.py:1087-1088 remove_toc → remove_contents_table).
|
||||||
|
if p.RemoveTOC {
|
||||||
|
items = removeContentsTable(items, false)
|
||||||
|
}
|
||||||
if items == nil {
|
if items == nil {
|
||||||
items = []map[string]any{{"text": "", "doc_type_kwd": "text"}}
|
items = []map[string]any{{"text": "", "doc_type_kwd": "text"}}
|
||||||
}
|
}
|
||||||
|
|||||||
138
internal/parser/parser/html_parser_test.go
Normal file
138
internal/parser/parser/html_parser_test.go
Normal file
@@ -0,0 +1,138 @@
|
|||||||
|
package parser
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestHTMLParser_ConfigureFromSetup(t *testing.T) {
|
||||||
|
p := NewHTMLParser()
|
||||||
|
p.ConfigureFromSetup(map[string]any{
|
||||||
|
"remove_header_footer": true,
|
||||||
|
"remove_toc": true,
|
||||||
|
})
|
||||||
|
if !p.RemoveHeaderFooter {
|
||||||
|
t.Error("RemoveHeaderFooter = false, want true")
|
||||||
|
}
|
||||||
|
if !p.RemoveTOC {
|
||||||
|
t.Error("RemoveTOC = false, want true")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestHTMLParser_ConfigureFromSetup_NilSafe(t *testing.T) {
|
||||||
|
p := NewHTMLParser()
|
||||||
|
p.ConfigureFromSetup(nil) // should not panic
|
||||||
|
if p.RemoveHeaderFooter || p.RemoveTOC {
|
||||||
|
t.Error("flags should stay false after nil setup")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestHTMLParser_RemoveHeaderFooter verifies that <header>/<footer>
|
||||||
|
// tags and role="banner"/role="contentinfo" elements are stripped
|
||||||
|
// before parsing (mirrors Python parser.py:1083-1084 pre-parse).
|
||||||
|
func TestHTMLParser_RemoveHeaderFooter(t *testing.T) {
|
||||||
|
|
||||||
|
p := NewHTMLParser()
|
||||||
|
p.ConfigureFromSetup(map[string]any{"remove_header_footer": true})
|
||||||
|
|
||||||
|
html := `<html><body>
|
||||||
|
<header>Site Navigation</header>
|
||||||
|
<main>
|
||||||
|
<p>Main content here</p>
|
||||||
|
<div role="banner">Ad banner</div>
|
||||||
|
<footer>Copyright 2024</footer>
|
||||||
|
<div role="contentinfo">Legal info</div>
|
||||||
|
</main>
|
||||||
|
</body></html>`
|
||||||
|
|
||||||
|
res := p.ParseWithResult(context.Background(), "test.html", []byte(html))
|
||||||
|
if res.Err != nil {
|
||||||
|
t.Fatalf("ParseWithResult: %v", res.Err)
|
||||||
|
}
|
||||||
|
for _, item := range res.JSON {
|
||||||
|
text, _ := item["text"].(string)
|
||||||
|
for _, banned := range []string{"Site Navigation", "Copyright 2024", "Ad banner", "Legal info"} {
|
||||||
|
if strings.Contains(text, banned) {
|
||||||
|
t.Errorf("header/footer text %q leaked into output item: %q", banned, text)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Main content must survive.
|
||||||
|
found := false
|
||||||
|
for _, item := range res.JSON {
|
||||||
|
if text, _ := item["text"].(string); strings.Contains(text, "Main content here") {
|
||||||
|
found = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !found {
|
||||||
|
t.Error("main content was stripped along with header/footer")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestHTMLParser_RemoveHeaderFooter_DisabledByDefault verifies that
|
||||||
|
// without the flag, <header>/<footer> content is preserved.
|
||||||
|
func TestHTMLParser_RemoveHeaderFooter_DisabledByDefault(t *testing.T) {
|
||||||
|
|
||||||
|
p := NewHTMLParser()
|
||||||
|
|
||||||
|
html := `<html><body><header>Nav</header><p>Body</p></body></html>`
|
||||||
|
res := p.ParseWithResult(context.Background(), "test.html", []byte(html))
|
||||||
|
if res.Err != nil {
|
||||||
|
t.Fatalf("ParseWithResult: %v", res.Err)
|
||||||
|
}
|
||||||
|
found := false
|
||||||
|
for _, item := range res.JSON {
|
||||||
|
if text, _ := item["text"].(string); strings.Contains(text, "Nav") {
|
||||||
|
found = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !found {
|
||||||
|
t.Error("header text should be preserved when flag is off")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestHTMLParser_RemoveTOC verifies that TOC sections are filtered
|
||||||
|
// after parsing (mirrors Python parser.py:1087-1088 post-parse).
|
||||||
|
func TestHTMLParser_RemoveTOC(t *testing.T) {
|
||||||
|
|
||||||
|
p := NewHTMLParser()
|
||||||
|
p.ConfigureFromSetup(map[string]any{"remove_toc": true})
|
||||||
|
|
||||||
|
html := `<html><body>
|
||||||
|
<h1>前言</h1>
|
||||||
|
<h1>目录</h1>
|
||||||
|
<p>第一章 概述</p>
|
||||||
|
<p>第二章 方法</p>
|
||||||
|
<p>正文开始</p>
|
||||||
|
</body></html>`
|
||||||
|
|
||||||
|
res := p.ParseWithResult(context.Background(), "test.html", []byte(html))
|
||||||
|
if res.Err != nil {
|
||||||
|
t.Fatalf("ParseWithResult: %v", res.Err)
|
||||||
|
}
|
||||||
|
// "目录" heading and "第一章 概述" (first entry after heading)
|
||||||
|
// should be removed by removeContentsTable.
|
||||||
|
for _, item := range res.JSON {
|
||||||
|
text, _ := item["text"].(string)
|
||||||
|
if text == "目录" {
|
||||||
|
t.Errorf("TOC heading %q was not removed", text)
|
||||||
|
}
|
||||||
|
if strings.HasPrefix(text, "第一章") {
|
||||||
|
t.Errorf("first TOC entry %q was not removed", text)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// "前言" and "正文开始" should survive.
|
||||||
|
wantTexts := map[string]bool{"前言": false, "正文开始": false}
|
||||||
|
for _, item := range res.JSON {
|
||||||
|
text, _ := item["text"].(string)
|
||||||
|
if _, ok := wantTexts[text]; ok {
|
||||||
|
wantTexts[text] = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for text, found := range wantTexts {
|
||||||
|
if !found {
|
||||||
|
t.Errorf("expected %q to survive, but it was removed", text)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
51
internal/parser/parser/html_postprocess.go
Normal file
51
internal/parser/parser/html_postprocess.go
Normal file
@@ -0,0 +1,51 @@
|
|||||||
|
package parser
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
|
||||||
|
"golang.org/x/net/html"
|
||||||
|
)
|
||||||
|
|
||||||
|
// stripHTMLHeaderFooter removes <header>/<footer> elements and nodes
|
||||||
|
// with role="banner"/role="contentinfo" from the HTML byte stream,
|
||||||
|
// then re-serializes. Mirrors Python rag/flow/parser/utils.py:70-74
|
||||||
|
// remove_header_footer_html_blob (BeautifulSoup decompose).
|
||||||
|
func stripHTMLHeaderFooter(data []byte) ([]byte, error) {
|
||||||
|
doc, err := html.Parse(bytes.NewReader(data))
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
removeHeaderFooterNodes(doc)
|
||||||
|
var buf bytes.Buffer
|
||||||
|
if err := html.Render(&buf, doc); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return buf.Bytes(), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// removeHeaderFooterNodes walks the parsed tree and detaches every
|
||||||
|
// <header>/<footer> element and every element whose role attribute
|
||||||
|
// is "banner" or "contentinfo".
|
||||||
|
func removeHeaderFooterNodes(n *html.Node) {
|
||||||
|
var next *html.Node
|
||||||
|
for child := n.FirstChild; child != nil; child = next {
|
||||||
|
next = child.NextSibling
|
||||||
|
if child.Type == html.ElementNode && isHeaderFooterNode(child) {
|
||||||
|
n.RemoveChild(child)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
removeHeaderFooterNodes(child)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func isHeaderFooterNode(n *html.Node) bool {
|
||||||
|
if n.Data == "header" || n.Data == "footer" {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
for _, attr := range n.Attr {
|
||||||
|
if attr.Key == "role" && (attr.Val == "banner" || attr.Val == "contentinfo") {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
@@ -49,10 +49,11 @@ const GoMarkdown = "go_markdown"
|
|||||||
var ssrfAllowLoopback bool
|
var ssrfAllowLoopback bool
|
||||||
|
|
||||||
type MarkdownParser struct {
|
type MarkdownParser struct {
|
||||||
libType string
|
libType string
|
||||||
ParseMethod string
|
ParseMethod string
|
||||||
OutputFormat string
|
OutputFormat string
|
||||||
VLM map[string]any
|
VLM map[string]any
|
||||||
|
FlattenMediaToText bool
|
||||||
}
|
}
|
||||||
|
|
||||||
func NewMarkdownParser(libType string) (*MarkdownParser, error) {
|
func NewMarkdownParser(libType string) (*MarkdownParser, error) {
|
||||||
@@ -79,6 +80,9 @@ func (p *MarkdownParser) ConfigureFromSetup(setup map[string]any) {
|
|||||||
if v, ok := setup["vlm"].(map[string]any); ok {
|
if v, ok := setup["vlm"].(map[string]any); ok {
|
||||||
p.VLM = v
|
p.VLM = v
|
||||||
}
|
}
|
||||||
|
if v, ok := setup["flatten_media_to_text"].(bool); ok {
|
||||||
|
p.FlattenMediaToText = v
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ParseWithResult implements ParseResultProducer (plan §6.5) and
|
// ParseWithResult implements ParseResultProducer (plan §6.5) and
|
||||||
@@ -94,7 +98,7 @@ func (p *MarkdownParser) ParseWithResult(ctx context.Context, filename string, d
|
|||||||
rawText := string(data)
|
rawText := string(data)
|
||||||
|
|
||||||
var items []map[string]any
|
var items []map[string]any
|
||||||
walkMarkdownBlocksWithImages(doc, rawText, &items)
|
walkMarkdownBlocksWithImages(doc, rawText, &items, p.FlattenMediaToText)
|
||||||
if items == nil {
|
if items == nil {
|
||||||
items = []map[string]any{{"text": "", "doc_type_kwd": "text"}}
|
items = []map[string]any{{"text": "", "doc_type_kwd": "text"}}
|
||||||
}
|
}
|
||||||
@@ -123,8 +127,10 @@ func markdownNew() *parser.Parser {
|
|||||||
// emitted with their text. When a block contains a markdown image
|
// emitted with their text. When a block contains a markdown image
|
||||||
// reference (), the image data is resolved via
|
// reference (), the image data is resolved via
|
||||||
// resolveMarkdownImage and the item carries `doc_type_kwd: "image"`
|
// resolveMarkdownImage and the item carries `doc_type_kwd: "image"`
|
||||||
// together with the base64-encoded image payload.
|
// together with the base64-encoded image payload. When flatten is
|
||||||
func walkMarkdownBlocksWithImages(doc ast.Node, rawText string, out *[]map[string]any) {
|
// true, all items are forced to doc_type_kwd="text" (mirrors Python
|
||||||
|
// parser.py:1034 flatten_media_to_text).
|
||||||
|
func walkMarkdownBlocksWithImages(doc ast.Node, rawText string, out *[]map[string]any, flatten bool) {
|
||||||
for _, child := range doc.GetChildren() {
|
for _, child := range doc.GetChildren() {
|
||||||
var ckType string
|
var ckType string
|
||||||
var docTypeKwd string
|
var docTypeKwd string
|
||||||
@@ -166,9 +172,13 @@ func walkMarkdownBlocksWithImages(doc ast.Node, rawText string, out *[]map[strin
|
|||||||
// Detect markdown image references in the raw source text
|
// Detect markdown image references in the raw source text
|
||||||
// that corresponds to this block. When found, resolve the
|
// that corresponds to this block. When found, resolve the
|
||||||
// image data so downstream vision enhancement can describe it.
|
// image data so downstream vision enhancement can describe it.
|
||||||
|
// When flatten is true, keep doc_type_kwd="text" (Python
|
||||||
|
// parser.py:1034: flatten_media_to_text overrides image).
|
||||||
if imgData, imgFound := resolveMarkdownImage(txt, rawText); imgFound && imgData != "" {
|
if imgData, imgFound := resolveMarkdownImage(txt, rawText); imgFound && imgData != "" {
|
||||||
item["doc_type_kwd"] = "image"
|
|
||||||
item["image"] = imgData
|
item["image"] = imgData
|
||||||
|
if !flatten {
|
||||||
|
item["doc_type_kwd"] = "image"
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
*out = append(*out, item)
|
*out = append(*out, item)
|
||||||
|
|||||||
@@ -118,6 +118,44 @@ func TestMarkdownParser_ConfigureFromSetup_NilSafe(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestMarkdownParser_FlattenMediaToText verifies that when
|
||||||
|
// flatten_media_to_text is true, image items are emitted with
|
||||||
|
// doc_type_kwd="text" (mirroring Python parser.py:1034). When false,
|
||||||
|
// image items keep doc_type_kwd="image".
|
||||||
|
func TestMarkdownParser_FlattenMediaToText(t *testing.T) {
|
||||||
|
ctx := t.Context()
|
||||||
|
pixelB64 := base64.StdEncoding.EncodeToString([]byte("fake-png-data"))
|
||||||
|
md := "Some text with an image\n\n"
|
||||||
|
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
flatten bool
|
||||||
|
wantDocType string
|
||||||
|
}{
|
||||||
|
{"flatten=false keeps image", false, "image"},
|
||||||
|
{"flatten=true forces text", true, "text"},
|
||||||
|
}
|
||||||
|
for _, tc := range cases {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
p, _ := NewMarkdownParser(GoMarkdown)
|
||||||
|
p.ConfigureFromSetup(map[string]any{"flatten_media_to_text": tc.flatten})
|
||||||
|
if p.FlattenMediaToText != tc.flatten {
|
||||||
|
t.Fatalf("FlattenMediaToText field = %v, want %v", p.FlattenMediaToText, tc.flatten)
|
||||||
|
}
|
||||||
|
res := p.ParseWithResult(ctx, "test.md", []byte(md))
|
||||||
|
if res.Err != nil {
|
||||||
|
t.Fatalf("ParseWithResult: %v", res.Err)
|
||||||
|
}
|
||||||
|
for _, item := range res.JSON {
|
||||||
|
if kd, _ := item["doc_type_kwd"].(string); kd != tc.wantDocType {
|
||||||
|
t.Errorf("doc_type_kwd = %q, want %q (item text=%q)",
|
||||||
|
kd, tc.wantDocType, item["text"])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestResolveMarkdownImage_DataURI(t *testing.T) {
|
func TestResolveMarkdownImage_DataURI(t *testing.T) {
|
||||||
b64 := base64.StdEncoding.EncodeToString([]byte("fakeimage"))
|
b64 := base64.StdEncoding.EncodeToString([]byte("fakeimage"))
|
||||||
md := ""
|
md := ""
|
||||||
|
|||||||
107
internal/parser/parser/text_toc.go
Normal file
107
internal/parser/parser/text_toc.go
Normal file
@@ -0,0 +1,107 @@
|
|||||||
|
package parser
|
||||||
|
|
||||||
|
import (
|
||||||
|
"regexp"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// tocHeadingPattern matches TOC heading text. Mirrors Python
|
||||||
|
// rag/nlp/__init__.py:945 — case-insensitive match against
|
||||||
|
// contents/目录/目次/table of contents/致谢/acknowledge after
|
||||||
|
// stripping all spaces (ASCII + fullwidth).
|
||||||
|
var tocHeadingPattern = regexp.MustCompile(`(?i)^(contents|目录|目次|table of contents|致谢|acknowledge)$`)
|
||||||
|
|
||||||
|
// whitespacePattern matches ASCII spaces and fullwidth space (U+3000)
|
||||||
|
// for normalization before heading match.
|
||||||
|
var whitespacePattern = regexp.MustCompile("[ \t\u3000]+")
|
||||||
|
|
||||||
|
// itemText extracts the text field from a parser item map. Falls
|
||||||
|
// back to empty string when the field is missing or not a string.
|
||||||
|
func itemText(item map[string]any) string {
|
||||||
|
if s, ok := item["text"].(string); ok {
|
||||||
|
return strings.TrimSpace(s)
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// removeContentsTable drops the table-of-contents entries from a list
|
||||||
|
// of parser items. Mirrors Python rag/nlp/__init__.py:937-965
|
||||||
|
// (remove_contents_table).
|
||||||
|
//
|
||||||
|
// Algorithm:
|
||||||
|
// 1. Scan for a TOC heading (contents/目录/目次/...).
|
||||||
|
// 2. Drop the heading.
|
||||||
|
// 3. Take the next entry's prefix (first 3 chars for CJK, first 2
|
||||||
|
// words for English) as the TOC entry pattern.
|
||||||
|
// 4. Drop that entry, then drop up to 128 following entries that
|
||||||
|
// start with the prefix.
|
||||||
|
// 5. Non-matching entries are kept.
|
||||||
|
func removeContentsTable(items []map[string]any, eng bool) []map[string]any {
|
||||||
|
i := 0
|
||||||
|
for i < len(items) {
|
||||||
|
text := itemText(items[i])
|
||||||
|
// Strip @@ suffix and whitespace before matching.
|
||||||
|
normalized := whitespacePattern.ReplaceAllString(strings.SplitN(text, "@@", 2)[0], "")
|
||||||
|
if !tocHeadingPattern.MatchString(normalized) {
|
||||||
|
i++
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Drop the TOC heading.
|
||||||
|
items = append(items[:i], items[i+1:]...)
|
||||||
|
if i >= len(items) {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
// Determine the prefix from the entry right after the heading.
|
||||||
|
prefix := tocPrefix(itemText(items[i]), eng)
|
||||||
|
for prefix == "" {
|
||||||
|
items = append(items[:i], items[i+1:]...)
|
||||||
|
if i >= len(items) {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
prefix = tocPrefix(itemText(items[i]), eng)
|
||||||
|
}
|
||||||
|
if i >= len(items) || prefix == "" {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
// Drop the first TOC entry (the one that supplied the prefix).
|
||||||
|
items = append(items[:i], items[i+1:]...)
|
||||||
|
if i >= len(items) {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
// Drop up to 128 following entries that start with prefix.
|
||||||
|
limit := i + 128
|
||||||
|
if limit > len(items) {
|
||||||
|
limit = len(items)
|
||||||
|
}
|
||||||
|
for j := i; j < limit; j++ {
|
||||||
|
if !strings.HasPrefix(itemText(items[j]), prefix) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Drop entries [i, j).
|
||||||
|
items = append(items[:i], items[j:]...)
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return items
|
||||||
|
}
|
||||||
|
|
||||||
|
// tocPrefix returns the TOC-entry prefix: first 3 chars for CJK,
|
||||||
|
// first 2 whitespace-separated words for English. Mirrors Python
|
||||||
|
// remove_contents_table line 951.
|
||||||
|
func tocPrefix(text string, eng bool) string {
|
||||||
|
if eng {
|
||||||
|
words := strings.Fields(text)
|
||||||
|
if len(words) >= 2 {
|
||||||
|
return words[0] + " " + words[1]
|
||||||
|
}
|
||||||
|
if len(words) == 1 {
|
||||||
|
return words[0]
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
runes := []rune(text)
|
||||||
|
if len(runes) < 3 {
|
||||||
|
return string(runes)
|
||||||
|
}
|
||||||
|
return string(runes[:3])
|
||||||
|
}
|
||||||
108
internal/parser/parser/text_toc_test.go
Normal file
108
internal/parser/parser/text_toc_test.go
Normal file
@@ -0,0 +1,108 @@
|
|||||||
|
package parser
|
||||||
|
|
||||||
|
import (
|
||||||
|
"reflect"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestRemoveContentsTable mirrors the Python remove_contents_table
|
||||||
|
// (rag/nlp/__init__.py:937-965). It locates a TOC heading
|
||||||
|
// (contents/目录/目次/...), drops it, then drops the following
|
||||||
|
// entries that share a common prefix with the first TOC entry.
|
||||||
|
func TestRemoveContentsTable(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
items []map[string]any
|
||||||
|
eng bool
|
||||||
|
want []map[string]any
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "no TOC heading → unchanged",
|
||||||
|
items: []map[string]any{{"text": "Intro"}, {"text": "Body"}},
|
||||||
|
eng: false,
|
||||||
|
want: []map[string]any{{"text": "Intro"}, {"text": "Body"}},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "Chinese 目录 heading + first TOC entry removed",
|
||||||
|
items: []map[string]any{
|
||||||
|
{"text": "前言"},
|
||||||
|
{"text": "目录"},
|
||||||
|
{"text": "第一章 概述"},
|
||||||
|
{"text": "第二章 方法"},
|
||||||
|
{"text": "正文开始"},
|
||||||
|
},
|
||||||
|
eng: false,
|
||||||
|
// Python remove_contents_table drops the heading and the
|
||||||
|
// first entry after it (prefix "第一章"). Subsequent
|
||||||
|
// entries with a different prefix ("第二章") are kept
|
||||||
|
// because re.match("第一章", "第二章 方法") fails.
|
||||||
|
want: []map[string]any{
|
||||||
|
{"text": "前言"},
|
||||||
|
{"text": "第二章 方法"},
|
||||||
|
{"text": "正文开始"},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "English Contents heading + first TOC entry removed",
|
||||||
|
items: []map[string]any{
|
||||||
|
{"text": "Intro"},
|
||||||
|
{"text": "Contents"},
|
||||||
|
{"text": "Chapter 1 Overview"},
|
||||||
|
{"text": "Chapter 2 Method"},
|
||||||
|
{"text": "Body starts"},
|
||||||
|
},
|
||||||
|
eng: true,
|
||||||
|
want: []map[string]any{
|
||||||
|
{"text": "Intro"},
|
||||||
|
{"text": "Chapter 2 Method"},
|
||||||
|
{"text": "Body starts"},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "目录 heading with no following entry → heading dropped only",
|
||||||
|
items: []map[string]any{
|
||||||
|
{"text": "前言"},
|
||||||
|
{"text": "目录"},
|
||||||
|
},
|
||||||
|
eng: false,
|
||||||
|
want: []map[string]any{
|
||||||
|
{"text": "前言"},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "prefix-matching subsequent entries collapse non-matching gap",
|
||||||
|
items: []map[string]any{
|
||||||
|
{"text": "前言"},
|
||||||
|
{"text": "目录"},
|
||||||
|
{"text": "第一节 概述"},
|
||||||
|
{"text": "杂项"},
|
||||||
|
{"text": "第一节 背景"},
|
||||||
|
{"text": "正文"},
|
||||||
|
},
|
||||||
|
eng: false,
|
||||||
|
// prefix="第一节"; after dropping heading + first entry,
|
||||||
|
// scan finds "第一节 背景" at j=2 (gap "杂项" at j=1);
|
||||||
|
// Python deletes [i, j) = the gap, keeps the match.
|
||||||
|
want: []map[string]any{
|
||||||
|
{"text": "前言"},
|
||||||
|
{"text": "第一节 背景"},
|
||||||
|
{"text": "正文"},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "empty input → unchanged",
|
||||||
|
items: nil,
|
||||||
|
eng: false,
|
||||||
|
want: nil,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tc := range cases {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
got := removeContentsTable(tc.items, tc.eng)
|
||||||
|
if !reflect.DeepEqual(got, tc.want) {
|
||||||
|
t.Errorf("got %v, want %v", got, tc.want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user