Files
ragflow/internal/parser/parser/html_postprocess.go

52 lines
1.3 KiB
Go
Raw Normal View History

fix: honor parser params and image VLM system_prompt in Go ingestion (#17334) ## Summary Fix the Go ingestion pipeline so that several parser setup switches and the image VLM prompt are actually honored end-to-end (previously the DSL fields existed but the Go code never read them). - **DOCX** (`docx_parser.go`, `docx_postprocess.go`): read `remove_toc` and `remove_header_footer`; apply to both JSON and markdown output paths (outline-based TOC removal with a text-heuristic fallback, plus header/footer section filtering). - **HTML** (`html_parser.go`, `html_postprocess.go`, `text_toc.go`): read `remove_header_footer` (pre-parse strip of `<header>`/`<footer>` and ARIA `banner`/`contentinfo`) and `remove_toc` (post-parse `remove_contents_table` heuristic). - **Markdown** (`markdown_parser.go`): read `flatten_media_to_text` and force media blocks to text when enabled. - **Image VLM** (`media_dispatch.go`): read `system_prompt` instead of `prompt` so the user-configured image VLM prompt is no longer silently dropped (`prompt` remains the video family key). All flags are wired through `ConfigureFromSetup`, which the dispatch layer already invokes for every family, so the behavior is live rather than dead code. ## Test plan - New unit tests: `docx_postprocess_test.go`, `html_parser_test.go`, `text_toc_test.go`, `markdown_parser_test.go`, `media_dispatch_test.go`. - `bash build.sh --test ./internal/parser/parser/... ./internal/ingestion/component/...` ## Notes - The `File` component is excluded from this migration scope. - Relates to the Python→Go parity diff (Parser 1.8–1.11, 1.15).
2026-07-24 14:42:26 +08:00
package parser
import (
"bytes"
"golang.org/x/net/html"
)
// stripHTMLHeaderFooter removes <header>/<footer> elements and nodes
// with role="banner"/role="contentinfo" from the HTML byte stream,
// then re-serializes. Mirrors Python rag/flow/parser/utils.py:70-74
// remove_header_footer_html_blob (BeautifulSoup decompose).
func stripHTMLHeaderFooter(data []byte) ([]byte, error) {
doc, err := html.Parse(bytes.NewReader(data))
if err != nil {
return nil, err
}
removeHeaderFooterNodes(doc)
var buf bytes.Buffer
if err := html.Render(&buf, doc); err != nil {
return nil, err
}
return buf.Bytes(), nil
}
// removeHeaderFooterNodes walks the parsed tree and detaches every
// <header>/<footer> element and every element whose role attribute
// is "banner" or "contentinfo".
func removeHeaderFooterNodes(n *html.Node) {
var next *html.Node
for child := n.FirstChild; child != nil; child = next {
next = child.NextSibling
if child.Type == html.ElementNode && isHeaderFooterNode(child) {
n.RemoveChild(child)
continue
}
removeHeaderFooterNodes(child)
}
}
func isHeaderFooterNode(n *html.Node) bool {
if n.Data == "header" || n.Data == "footer" {
return true
}
for _, attr := range n.Attr {
if attr.Key == "role" && (attr.Val == "banner" || attr.Val == "contentinfo") {
return true
}
}
return false
}