mirror of
https://github.com/infiniflow/ragflow.git
synced 2026-07-20 14:41:05 +08:00
### Problem Parsing a Q&A `.csv` can splice unrelated text into the wrong answers (reported in #16791). ### Root cause The `.csv` branch of `rag/app/qa.py`'s `chunk()` builds records with `csv.reader(lines, delimiter=delimiter)` (default `quotechar='"'`), but then indexes `lines[i]` by the reader's *record* index in `answer += "\n" + lines[i]`. When a line's field opens with a `"`, `csv.reader` treats it as an unclosed quoted field and merges several physical lines into one record. From there the record index permanently desyncs from the physical line numbers, so `lines[i]` returns the wrong line and unrelated Q&A content gets appended to the wrong answer. ### Reproduction (stdlib only) ```python import csv lines = 'Q1,A1\n"quoted answer start\ncontinues here,extra\nQ2,A2\n'.split("\n") list(csv.reader(lines, delimiter=",")) # record 1 swallows 3 physical lines: ['quoted answer startcontinues here,extraQ2,A2'] # -> the reader index no longer matches lines[i] list(csv.reader(lines, delimiter=",", quoting=csv.QUOTE_NONE)) # one physical line per record; index stays aligned ``` ### Fix Pass `quoting=csv.QUOTE_NONE` so one physical line maps to one record, keeping the reader index aligned with `lines[i]` (the surrounding code already relies on that 1:1 mapping). Fixes #16791. --------- Signed-off-by: Yash Raj Pandey <yashpn62@gmail.com> Co-authored-by: Yingfeng <yingfeng.zhang@gmail.com>
94 lines
2.7 KiB
Python
94 lines
2.7 KiB
Python
#
|
|
# Copyright 2026 The InfiniFlow Authors. All Rights Reserved.
|
|
#
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
# you may not use this file except in compliance with the License.
|
|
# You may obtain a copy of the License at
|
|
#
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
#
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
# See the License for the specific language governing permissions and
|
|
# limitations under the License.
|
|
#
|
|
|
|
from __future__ import annotations
|
|
|
|
import warnings
|
|
|
|
with warnings.catch_warnings():
|
|
warnings.filterwarnings("ignore", message=".*pkg_resources is deprecated.*", category=UserWarning)
|
|
import pkg_resources # noqa: F401 - stabilize xgboost import during collection
|
|
|
|
import pytest
|
|
|
|
from rag.app import qa
|
|
|
|
|
|
def _noop_callback(*_args, **_kwargs):
|
|
pass
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def _stub_rag_tokenizer(monkeypatch):
|
|
def fake_tokenize(text):
|
|
return str(text)
|
|
|
|
monkeypatch.setattr("rag.nlp.rag_tokenizer.tokenize", fake_tokenize)
|
|
monkeypatch.setattr("rag.nlp.rag_tokenizer.fine_grained_tokenize", fake_tokenize)
|
|
|
|
|
|
@pytest.mark.p2
|
|
def test_csv_final_pair_uses_last_line_number():
|
|
chunks = qa.chunk(
|
|
"qa.csv",
|
|
binary=b"Question 1,Answer 1\nQuestion 2,Answer 2",
|
|
lang="English",
|
|
callback=_noop_callback,
|
|
)
|
|
|
|
assert len(chunks) == 2
|
|
assert chunks[0]["top_int"] == [1]
|
|
assert chunks[1]["top_int"] == [2]
|
|
|
|
|
|
@pytest.mark.p2
|
|
def test_csv_quoted_comma_stays_in_one_field():
|
|
chunks = qa.chunk(
|
|
"qa.csv",
|
|
binary=b'"Question, one",Answer 1\nQuestion 2,Answer 2',
|
|
lang="English",
|
|
callback=_noop_callback,
|
|
)
|
|
|
|
assert len(chunks) == 2
|
|
assert chunks[0]["content_with_weight"] == "Question: Question, one\tAnswer: 1"
|
|
|
|
|
|
@pytest.mark.p2
|
|
def test_csv_quoted_field_preserves_embedded_newline():
|
|
chunks = qa.chunk(
|
|
"qa.csv",
|
|
binary=b'"first line\nsecond line",some answer\n',
|
|
lang="English",
|
|
callback=_noop_callback,
|
|
)
|
|
|
|
assert len(chunks) == 1
|
|
assert chunks[0]["content_with_weight"] == "Question: first line\nsecond line\tAnswer: some answer"
|
|
|
|
|
|
@pytest.mark.p2
|
|
def test_csv_multiline_quote_uses_physical_continuation_line():
|
|
chunks = qa.chunk(
|
|
"qa.csv",
|
|
binary=b'first,one\n"second\nquestion",two\ntwo continued\nthird,three',
|
|
lang="English",
|
|
callback=_noop_callback,
|
|
)
|
|
|
|
assert len(chunks) == 3
|
|
assert chunks[1]["content_with_weight"] == "Question: second\nquestion\tAnswer: two\ntwo continued"
|