mirror of
https://github.com/infiniflow/ragflow.git
synced 2026-08-02 13:57:30 +08:00
fix: honor parser params and image VLM system_prompt in Go ingestion (#17334)
## Summary Fix the Go ingestion pipeline so that several parser setup switches and the image VLM prompt are actually honored end-to-end (previously the DSL fields existed but the Go code never read them). - **DOCX** (`docx_parser.go`, `docx_postprocess.go`): read `remove_toc` and `remove_header_footer`; apply to both JSON and markdown output paths (outline-based TOC removal with a text-heuristic fallback, plus header/footer section filtering). - **HTML** (`html_parser.go`, `html_postprocess.go`, `text_toc.go`): read `remove_header_footer` (pre-parse strip of `<header>`/`<footer>` and ARIA `banner`/`contentinfo`) and `remove_toc` (post-parse `remove_contents_table` heuristic). - **Markdown** (`markdown_parser.go`): read `flatten_media_to_text` and force media blocks to text when enabled. - **Image VLM** (`media_dispatch.go`): read `system_prompt` instead of `prompt` so the user-configured image VLM prompt is no longer silently dropped (`prompt` remains the video family key). All flags are wired through `ConfigureFromSetup`, which the dispatch layer already invokes for every family, so the behavior is live rather than dead code. ## Test plan - New unit tests: `docx_postprocess_test.go`, `html_parser_test.go`, `text_toc_test.go`, `markdown_parser_test.go`, `media_dispatch_test.go`. - `bash build.sh --test ./internal/parser/parser/... ./internal/ingestion/component/...` ## Notes - The `File` component is excluded from this migration scope. - Relates to the Python→Go parity diff (Parser 1.8–1.11, 1.15).
This commit is contained in:
51
internal/parser/parser/html_postprocess.go
Normal file
51
internal/parser/parser/html_postprocess.go
Normal file
@@ -0,0 +1,51 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
|
||||
"golang.org/x/net/html"
|
||||
)
|
||||
|
||||
// stripHTMLHeaderFooter removes <header>/<footer> elements and nodes
|
||||
// with role="banner"/role="contentinfo" from the HTML byte stream,
|
||||
// then re-serializes. Mirrors Python rag/flow/parser/utils.py:70-74
|
||||
// remove_header_footer_html_blob (BeautifulSoup decompose).
|
||||
func stripHTMLHeaderFooter(data []byte) ([]byte, error) {
|
||||
doc, err := html.Parse(bytes.NewReader(data))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
removeHeaderFooterNodes(doc)
|
||||
var buf bytes.Buffer
|
||||
if err := html.Render(&buf, doc); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return buf.Bytes(), nil
|
||||
}
|
||||
|
||||
// removeHeaderFooterNodes walks the parsed tree and detaches every
|
||||
// <header>/<footer> element and every element whose role attribute
|
||||
// is "banner" or "contentinfo".
|
||||
func removeHeaderFooterNodes(n *html.Node) {
|
||||
var next *html.Node
|
||||
for child := n.FirstChild; child != nil; child = next {
|
||||
next = child.NextSibling
|
||||
if child.Type == html.ElementNode && isHeaderFooterNode(child) {
|
||||
n.RemoveChild(child)
|
||||
continue
|
||||
}
|
||||
removeHeaderFooterNodes(child)
|
||||
}
|
||||
}
|
||||
|
||||
func isHeaderFooterNode(n *html.Node) bool {
|
||||
if n.Data == "header" || n.Data == "footer" {
|
||||
return true
|
||||
}
|
||||
for _, attr := range n.Attr {
|
||||
if attr.Key == "role" && (attr.Val == "banner" || attr.Val == "contentinfo") {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
Reference in New Issue
Block a user