2024-08-15 09:17:36 +08:00
#
# Copyright 2024 The InfiniFlow Authors. All Rights Reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
2026-01-13 09:41:35 +08:00
import asyncio
2024-09-03 19:49:14 +08:00
import binascii
2025-03-24 13:18:47 +08:00
import logging
2024-08-15 09:17:36 +08:00
import re
2025-03-24 13:18:47 +08:00
import time
2024-08-15 09:17:36 +08:00
from copy import deepcopy
2025-05-19 19:34:05 +08:00
from datetime import datetime
2025-03-24 13:18:47 +08:00
from functools import partial
2024-09-09 12:08:50 +08:00
from timeit import default_timer as timer
2025-03-24 13:18:47 +08:00
from langfuse import Langfuse
2025-08-06 10:33:52 +08:00
from peewee import fn
2025-11-28 19:25:32 +08:00
from api . db . services . file_service import FileService
2025-11-05 08:01:39 +08:00
from common . constants import LLMType , ParserType , StatusEnum
2025-03-24 13:18:47 +08:00
from api . db . db_models import DB , Dialog
2024-08-15 09:17:36 +08:00
from api . db . services . common_service import CommonService
2026-01-28 13:29:34 +08:00
from api . db . services . doc_metadata_service import DocMetadataService
2024-08-15 09:17:36 +08:00
from api . db . services . knowledgebase_service import KnowledgebaseService
2025-03-24 13:18:47 +08:00
from api . db . services . langfuse_service import TenantLangfuseService
2025-08-13 16:41:01 +08:00
from api . db . services . llm_service import LLMBundle
2025-12-12 17:12:38 +08:00
from common . metadata_utils import apply_meta_data_filter
2025-08-13 16:41:01 +08:00
from api . db . services . tenant_llm_service import TenantLLMService
2026-03-05 17:27:17 +08:00
from api . db . joint_services . tenant_model_service import get_model_config_by_id , get_model_config_by_type_and_name , get_tenant_default_model_by_type
2025-10-28 19:09:14 +08:00
from common . time_utils import current_timestamp , datetime_format
Feature rtl support (#13118)
### What problem does this PR solve?
This PR adds comprehensive **Right-to-Left (RTL) language support**,
primarily targeting Arabic and other RTL scripts (Hebrew, Persian, Urdu,
etc.).
Previously, RTL content had multiple rendering issues:
- Incorrect sentence splitting for Arabic punctuation in citation logic
- Misaligned text in chat messages and markdown components
- Improper positioning of blockquotes and “think” sections
- Incorrect table alignment
- Citation placement ambiguity in RTL prompts
- UI layout inconsistencies when mixing LTR and RTL text
This PR introduces backend and frontend improvements to properly detect,
render, and style RTL content while preserving existing LTR behavior.
#### Backend
- Updated sentence boundary regex in `rag/nlp/search.py` to include
Arabic punctuation:
- `،` (comma)
- `؛` (semicolon)
- `؟` (question mark)
- `۔` (Arabic full stop)
- Ensures citation insertion works correctly in RTL sentences.
- Updated citation prompt instructions to clarify citation placement
rules for RTL languages.
#### Frontend
- Introduced a new utility: `text-direction.ts`
- Detects text direction based on Unicode ranges.
- Supports Arabic, Hebrew, Syriac, Thaana, and related scripts.
- Provides `getDirAttribute()` for automatic `dir` assignment.
- Applied dynamic `dir` attributes across:
- Markdown rendering
- Chat messages
- Search results
- Tables
- Hover cards and reference popovers
- Added proper RTL styling in LESS:
- Text alignment adjustments
- Blockquote border flipping
- Section indentation correction
- Table direction switching
- Use of `<bdi>` for figure labels to prevent bidirectional conflicts
#### DevOps / Environment
- Added Windows backend launch script with retry handling.
- Updated dependency metadata.
- Adjusted development-only React debugging behavior.
---
### Type of change
- [x] Bug Fix (non-breaking change which fixes RTL rendering and
citation issues)
- [x] New Feature (non-breaking change which adds RTL detection and
dynamic direction handling)
---------
Co-authored-by: 6ba3i <isbaaoui09@gmail.com>
Co-authored-by: Ahmad Intisar <ahmadintisar@Ahmads-MacBook-M4-Pro.local>
Co-authored-by: Ahmad Intisar <168020872+ahmadintisar@users.noreply.github.com>
Co-authored-by: Liu An <asiro@qq.com>
2026-03-02 08:03:44 +03:00
from common . text_utils import normalize_arabic_digits
2026-01-29 14:23:26 +08:00
from rag . graphrag . general . mind_map_extractor import MindMapExtractor
2026-01-13 09:41:35 +08:00
from rag . advanced_rag import DeepResearcher
2025-02-26 15:40:52 +08:00
from rag . app . tag import label_question
2024-08-15 09:17:36 +08:00
from rag . nlp . search import index_name
2025-09-23 10:19:25 +08:00
from rag . prompts . generator import chunks_format , citation_prompt , cross_languages , full_question , kb_prompt , keyword_extraction , message_fit_in , \
2025-12-12 17:12:38 +08:00
PROMPT_JINJA_ENV , ASK_SUMMARY
2025-11-03 08:50:05 +08:00
from common . token_utils import num_tokens_from_string
2025-02-26 10:21:04 +08:00
from rag . utils . tavily_conn import Tavily
2025-10-28 09:46:32 +08:00
from common . string_utils import remove_redundant_spaces
2025-11-06 09:36:38 +08:00
from common import settings
2024-08-15 09:17:36 +08:00
class DialogService ( CommonService ) :
model = Dialog
2025-04-15 10:20:33 +08:00
@classmethod
def save ( cls , * * kwargs ) :
""" Save a new record to database.
This method creates a new record in the database with the provided field values,
forcing an insert operation rather than an update.
Args:
**kwargs: Record field values as keyword arguments.
Returns:
Model instance: The created record object.
"""
sample_obj = cls . model ( * * kwargs ) . save ( force_insert = True )
return sample_obj
@classmethod
def update_many_by_id ( cls , data_list ) :
""" Update multiple records by their IDs.
This method updates multiple records in the database, identified by their IDs.
It automatically updates the update_time and update_date fields for each record.
Args:
data_list (list): List of dictionaries containing record data to update.
Each dictionary must include an ' id ' field.
"""
with DB . atomic ( ) :
for data in data_list :
data [ " update_time " ] = current_timestamp ( )
data [ " update_date " ] = datetime_format ( datetime . now ( ) )
cls . model . update ( data ) . where ( cls . model . id == data [ " id " ] ) . execute ( )
2024-10-12 13:48:43 +08:00
@classmethod
@DB.connection_context ( )
2025-03-24 13:18:47 +08:00
def get_list ( cls , tenant_id , page_number , items_per_page , orderby , desc , id , name ) :
2024-10-12 13:48:43 +08:00
chats = cls . model . select ( )
if id :
chats = chats . where ( cls . model . id == id )
if name :
chats = chats . where ( cls . model . name == name )
2025-03-24 13:18:47 +08:00
chats = chats . where ( ( cls . model . tenant_id == tenant_id ) & ( cls . model . status == StatusEnum . VALID . value ) )
2024-10-12 13:48:43 +08:00
if desc :
chats = chats . order_by ( cls . model . getter_by ( orderby ) . desc ( ) )
else :
chats = chats . order_by ( cls . model . getter_by ( orderby ) . asc ( ) )
chats = chats . paginate ( page_number , items_per_page )
return list ( chats . dicts ( ) )
2025-08-06 10:33:52 +08:00
@classmethod
@DB.connection_context ( )
def get_by_tenant_ids ( cls , joined_tenant_ids , user_id , page_number , items_per_page , orderby , desc , keywords , parser_id = None ) :
from api . db . db_models import User
fields = [
cls . model . id ,
cls . model . tenant_id ,
cls . model . name ,
cls . model . description ,
cls . model . language ,
cls . model . llm_id ,
cls . model . llm_setting ,
cls . model . prompt_type ,
cls . model . prompt_config ,
cls . model . similarity_threshold ,
cls . model . vector_similarity_weight ,
cls . model . top_n ,
cls . model . top_k ,
cls . model . do_refer ,
cls . model . rerank_id ,
cls . model . kb_ids ,
2025-08-13 10:26:26 +08:00
cls . model . icon ,
2025-08-06 10:33:52 +08:00
cls . model . status ,
User . nickname ,
User . avatar . alias ( " tenant_avatar " ) ,
cls . model . update_time ,
cls . model . create_time ,
]
if keywords :
dialogs = (
cls . model . select ( * fields )
. join ( User , on = ( cls . model . tenant_id == User . id ) )
. where (
( cls . model . tenant_id . in_ ( joined_tenant_ids ) | ( cls . model . tenant_id == user_id ) ) & ( cls . model . status == StatusEnum . VALID . value ) ,
( fn . LOWER ( cls . model . name ) . contains ( keywords . lower ( ) ) ) ,
)
)
else :
dialogs = (
cls . model . select ( * fields )
. join ( User , on = ( cls . model . tenant_id == User . id ) )
. where (
( cls . model . tenant_id . in_ ( joined_tenant_ids ) | ( cls . model . tenant_id == user_id ) ) & ( cls . model . status == StatusEnum . VALID . value ) ,
)
)
if parser_id :
dialogs = dialogs . where ( cls . model . parser_id == parser_id )
if desc :
dialogs = dialogs . order_by ( cls . model . getter_by ( orderby ) . desc ( ) )
else :
dialogs = dialogs . order_by ( cls . model . getter_by ( orderby ) . asc ( ) )
count = dialogs . count ( )
if page_number and items_per_page :
dialogs = dialogs . paginate ( page_number , items_per_page )
return list ( dialogs . dicts ( ) ) , count
2025-09-29 10:16:13 +08:00
@classmethod
@DB.connection_context ( )
def get_all_dialogs_by_tenant_id ( cls , tenant_id ) :
fields = [ cls . model . id ]
dialogs = cls . model . select ( * fields ) . where ( cls . model . tenant_id == tenant_id )
dialogs . order_by ( cls . model . create_time . asc ( ) )
offset , limit = 0 , 100
res = [ ]
while True :
d_batch = dialogs . offset ( offset ) . limit ( limit )
_temp = list ( d_batch . dicts ( ) )
if not _temp :
break
res . extend ( _temp )
offset + = limit
return res
2025-08-06 10:33:52 +08:00
2026-03-05 17:27:17 +08:00
@classmethod
@DB.connection_context ( )
def get_null_tenant_llm_id_row ( cls ) :
fields = [
cls . model . id ,
cls . model . tenant_id ,
cls . model . llm_id
]
objs = cls . model . select ( * fields ) . where ( cls . model . tenant_llm_id . is_null ( ) )
return list ( objs )
@classmethod
@DB.connection_context ( )
def get_null_tenant_rerank_id_row ( cls ) :
fields = [
cls . model . id ,
cls . model . tenant_id ,
cls . model . rerank_id
]
objs = cls . model . select ( * fields ) . where ( cls . model . tenant_rerank_id . is_null ( ) )
return list ( objs )
2025-12-08 09:43:03 +08:00
async def async_chat_solo ( dialog , messages , stream = True ) :
2026-02-11 09:47:33 +08:00
llm_type = TenantLLMService . llm_id2llm_type ( dialog . llm_id )
2025-12-01 16:54:57 +08:00
attachments = " "
2026-02-11 09:47:33 +08:00
image_attachments = [ ]
image_files = [ ]
2025-12-01 16:54:57 +08:00
if " files " in messages [ - 1 ] :
2026-02-11 09:47:33 +08:00
if llm_type == " chat " :
text_attachments , image_attachments = split_file_attachments ( messages [ - 1 ] [ " files " ] )
else :
text_attachments , image_files = split_file_attachments ( messages [ - 1 ] [ " files " ] , raw = True )
attachments = " \n \n " . join ( text_attachments )
2026-03-05 17:27:17 +08:00
model_config = get_model_config_by_id ( dialog . tenant_llm_id )
chat_mdl = LLMBundle ( dialog . tenant_id , model_config )
factory = model_config . get ( " llm_factory " , " " ) if model_config else " "
2025-02-21 12:24:02 +08:00
prompt_config = dialog . prompt_config
tts_mdl = None
if prompt_config . get ( " tts " ) :
2026-03-05 17:27:17 +08:00
default_tts_model = get_tenant_default_model_by_type ( dialog . tenant_id , LLMType . TTS )
tts_mdl = LLMBundle ( dialog . tenant_id , default_tts_model )
2025-03-24 13:18:47 +08:00
msg = [ { " role " : m [ " role " ] , " content " : re . sub ( r " ## \ d+ \ $ \ $ " , " " , m [ " content " ] ) } for m in messages if m [ " role " ] != " system " ]
2025-12-01 16:54:57 +08:00
if attachments and msg :
msg [ - 1 ] [ " content " ] + = attachments
2026-02-11 09:47:33 +08:00
if llm_type == " chat " and image_attachments :
convert_last_user_msg_to_multimodal ( msg , image_attachments , factory )
2025-02-21 12:24:02 +08:00
if stream :
2026-02-11 09:47:33 +08:00
if llm_type == " chat " :
stream_iter = chat_mdl . async_chat_streamly_delta ( prompt_config . get ( " system " , " " ) , msg , dialog . llm_setting )
else :
stream_iter = chat_mdl . async_chat_streamly_delta ( prompt_config . get ( " system " , " " ) , msg , dialog . llm_setting , images = image_files )
2026-01-08 13:34:16 +08:00
async for kind , value , state in _stream_with_think_delta ( stream_iter ) :
if kind == " marker " :
flags = { " start_to_think " : True } if value == " <think> " else { " end_to_think " : True }
yield { " answer " : " " , " reference " : { } , " audio_binary " : None , " prompt " : " " , " created_at " : time . time ( ) , " final " : False , * * flags }
2025-02-21 12:24:02 +08:00
continue
2026-01-08 13:34:16 +08:00
yield { " answer " : value , " reference " : { } , " audio_binary " : tts ( tts_mdl , value ) , " prompt " : " " , " created_at " : time . time ( ) , " final " : False }
2025-02-21 12:24:02 +08:00
else :
2026-02-11 09:47:33 +08:00
if llm_type == " chat " :
answer = await chat_mdl . async_chat ( prompt_config . get ( " system " , " " ) , msg , dialog . llm_setting )
else :
answer = await chat_mdl . async_chat ( prompt_config . get ( " system " , " " ) , msg , dialog . llm_setting , images = image_files )
2025-02-21 12:24:02 +08:00
user_content = msg [ - 1 ] . get ( " content " , " [content not available] " )
logging . debug ( " User: {} |Assistant: {} " . format ( user_content , answer ) )
yield { " answer " : answer , " reference " : { } , " audio_binary " : tts ( tts_mdl , answer ) , " prompt " : " " , " created_at " : time . time ( ) }
2025-06-05 13:00:43 +08:00
def get_models ( dialog ) :
embd_mdl , chat_mdl , rerank_mdl , tts_mdl = None , None , None , None
kbs = KnowledgebaseService . get_by_ids ( dialog . kb_ids )
embedding_list = list ( set ( [ kb . embd_id for kb in kbs ] ) )
if len ( embedding_list ) > 1 :
raise Exception ( " **ERROR**: Knowledge bases use different embedding models. " )
if embedding_list :
2026-03-05 17:27:17 +08:00
embd_model_config = get_model_config_by_type_and_name ( dialog . tenant_id , LLMType . EMBEDDING , embedding_list [ 0 ] )
embd_mdl = LLMBundle ( dialog . tenant_id , embd_model_config )
2025-06-05 13:00:43 +08:00
if not embd_mdl :
raise LookupError ( " Embedding model( %s ) not found " % embedding_list [ 0 ] )
2026-03-05 17:27:17 +08:00
if dialog . tenant_llm_id :
chat_model_config = get_model_config_by_id ( dialog . tenant_llm_id )
elif dialog . llm_id :
chat_model_config = get_model_config_by_type_and_name ( dialog . tenant_id , LLMType . CHAT , dialog . llm_id )
2025-06-05 13:00:43 +08:00
else :
2026-03-05 17:27:17 +08:00
chat_model_config = get_tenant_default_model_by_type ( dialog . tenant_id , LLMType . CHAT )
chat_mdl = LLMBundle ( dialog . tenant_id , chat_model_config )
2025-06-05 13:00:43 +08:00
if dialog . rerank_id :
2026-03-05 17:27:17 +08:00
rerank_model_config = get_model_config_by_type_and_name ( dialog . tenant_id , LLMType . RERANK , dialog . rerank_id )
rerank_mdl = LLMBundle ( dialog . tenant_id , rerank_model_config )
2025-06-05 13:00:43 +08:00
if dialog . prompt_config . get ( " tts " ) :
2026-03-05 17:27:17 +08:00
default_tts_model_config = get_tenant_default_model_by_type ( dialog . tenant_id , LLMType . TTS )
tts_mdl = LLMBundle ( dialog . tenant_id , default_tts_model_config )
2025-06-05 13:00:43 +08:00
return kbs , embd_mdl , rerank_mdl , chat_mdl , tts_mdl
2026-02-11 09:47:33 +08:00
def split_file_attachments ( files : list [ dict ] | None , raw : bool = False ) - > tuple [ list [ str ] , list [ str ] | list [ dict ] ] :
if not files :
return [ ] , [ ]
text_attachments = [ ]
if raw :
file_contents , image_files = FileService . get_files ( files , raw = True )
for content in file_contents :
if not isinstance ( content , str ) :
content = str ( content )
text_attachments . append ( content )
return text_attachments , image_files
image_attachments = [ ]
for content in FileService . get_files ( files , raw = False ) :
if not isinstance ( content , str ) :
content = str ( content )
if content . strip ( ) . startswith ( " data: " ) :
image_attachments . append ( content . strip ( ) )
continue
text_attachments . append ( content )
return text_attachments , image_attachments
_DATA_URI_RE = re . compile ( r " ^data:(?P<mime>[^;]+);base64,(?P<b64>[A-Za-z0-9+/= \ s]+)$ " )
def _parse_data_uri_or_b64 ( s : str , default_mime : str = " image/png " ) - > tuple [ str , str ] :
s = ( s or " " ) . strip ( )
match = _DATA_URI_RE . match ( s )
if match :
mime = match . group ( " mime " ) . strip ( )
b64 = match . group ( " b64 " ) . strip ( )
return mime , b64
return default_mime , s
def _normalize_text_from_content ( content ) - > str :
if content is None :
return " "
if isinstance ( content , str ) :
return content
if isinstance ( content , list ) :
texts = [ ]
for blk in content :
if isinstance ( blk , dict ) :
if blk . get ( " type " ) in { " text " , " input_text " } :
txt = blk . get ( " text " )
if txt :
texts . append ( str ( txt ) )
elif " text " in blk and isinstance ( blk . get ( " text " ) , ( str , int , float ) ) :
texts . append ( str ( blk [ " text " ] ) )
return " \n " . join ( texts ) . strip ( )
return str ( content )
def convert_last_user_msg_to_multimodal ( msg : list [ dict ] , image_data_uris : list [ str ] , factory : str ) - > None :
if not msg or not image_data_uris :
return
factory_norm = ( factory or " " ) . strip ( ) . lower ( )
for idx in range ( len ( msg ) - 1 , - 1 , - 1 ) :
if msg [ idx ] . get ( " role " ) != " user " :
continue
original_content = msg [ idx ] . get ( " content " , " " )
text = _normalize_text_from_content ( original_content )
if factory_norm == " gemini " :
parts = [ ]
if text :
parts . append ( { " text " : text } )
for image in image_data_uris :
mime , b64 = _parse_data_uri_or_b64 ( str ( image ) , default_mime = " image/png " )
parts . append ( { " inline_data " : { " mime_type " : mime , " data " : b64 } } )
msg [ idx ] [ " content " ] = parts
return
if factory_norm == " anthropic " :
blocks = [ ]
if text :
blocks . append ( { " type " : " text " , " text " : text } )
for image in image_data_uris :
mime , b64 = _parse_data_uri_or_b64 ( str ( image ) , default_mime = " image/png " )
blocks . append (
{
" type " : " image " ,
" source " : { " type " : " base64 " , " media_type " : mime , " data " : b64 } ,
}
)
msg [ idx ] [ " content " ] = blocks
return
multimodal_content = [ ]
if isinstance ( original_content , list ) :
multimodal_content = deepcopy ( original_content )
else :
text_content = " " if original_content is None else str ( original_content )
if text_content :
multimodal_content . append ( { " type " : " text " , " text " : text_content } )
for data_uri in image_data_uris :
image_url = data_uri
if not isinstance ( image_url , str ) :
image_url = str ( image_url )
if not image_url . startswith ( " data: " ) :
image_url = f " data:image/png;base64, { image_url } "
multimodal_content . append ( { " type " : " image_url " , " image_url " : { " url " : image_url } } )
msg [ idx ] [ " content " ] = multimodal_content
return
2025-05-29 10:03:51 +08:00
BAD_CITATION_PATTERNS = [
re . compile ( r " \ ( \ s*ID \ s*[: ]* \ s*( \ d+) \ s* \ ) " ) , # (ID: 12)
re . compile ( r " \ [ \ s*ID \ s*[: ]* \ s*( \ d+) \ s* \ ] " ) , # [ID: 12]
re . compile ( r " 【 \ s*ID \ s*[: ]* \ s*( \ d+) \ s*】 " ) , # 【ID: 12】
re . compile ( r " ref \ s*( \ d+) " , flags = re . IGNORECASE ) , # ref12、REF 12
]
Feature rtl support (#13118)
### What problem does this PR solve?
This PR adds comprehensive **Right-to-Left (RTL) language support**,
primarily targeting Arabic and other RTL scripts (Hebrew, Persian, Urdu,
etc.).
Previously, RTL content had multiple rendering issues:
- Incorrect sentence splitting for Arabic punctuation in citation logic
- Misaligned text in chat messages and markdown components
- Improper positioning of blockquotes and “think” sections
- Incorrect table alignment
- Citation placement ambiguity in RTL prompts
- UI layout inconsistencies when mixing LTR and RTL text
This PR introduces backend and frontend improvements to properly detect,
render, and style RTL content while preserving existing LTR behavior.
#### Backend
- Updated sentence boundary regex in `rag/nlp/search.py` to include
Arabic punctuation:
- `،` (comma)
- `؛` (semicolon)
- `؟` (question mark)
- `۔` (Arabic full stop)
- Ensures citation insertion works correctly in RTL sentences.
- Updated citation prompt instructions to clarify citation placement
rules for RTL languages.
#### Frontend
- Introduced a new utility: `text-direction.ts`
- Detects text direction based on Unicode ranges.
- Supports Arabic, Hebrew, Syriac, Thaana, and related scripts.
- Provides `getDirAttribute()` for automatic `dir` assignment.
- Applied dynamic `dir` attributes across:
- Markdown rendering
- Chat messages
- Search results
- Tables
- Hover cards and reference popovers
- Added proper RTL styling in LESS:
- Text alignment adjustments
- Blockquote border flipping
- Section indentation correction
- Table direction switching
- Use of `<bdi>` for figure labels to prevent bidirectional conflicts
#### DevOps / Environment
- Added Windows backend launch script with retry handling.
- Updated dependency metadata.
- Adjusted development-only React debugging behavior.
---
### Type of change
- [x] Bug Fix (non-breaking change which fixes RTL rendering and
citation issues)
- [x] New Feature (non-breaking change which adds RTL detection and
dynamic direction handling)
---------
Co-authored-by: 6ba3i <isbaaoui09@gmail.com>
Co-authored-by: Ahmad Intisar <ahmadintisar@Ahmads-MacBook-M4-Pro.local>
Co-authored-by: Ahmad Intisar <168020872+ahmadintisar@users.noreply.github.com>
Co-authored-by: Liu An <asiro@qq.com>
2026-03-02 08:03:44 +03:00
CITATION_MARKER_PATTERN = re . compile ( r " \ [(?:ID:)?([0-9 \ u0660- \ u0669 \ u06F0- \ u06F9]+) \ ] " )
2025-05-29 10:03:51 +08:00
2025-06-18 16:45:42 +08:00
2025-06-05 13:00:43 +08:00
def repair_bad_citation_formats ( answer : str , kbinfos : dict , idx : set ) :
max_index = len ( kbinfos [ " chunks " ] )
Feature rtl support (#13118)
### What problem does this PR solve?
This PR adds comprehensive **Right-to-Left (RTL) language support**,
primarily targeting Arabic and other RTL scripts (Hebrew, Persian, Urdu,
etc.).
Previously, RTL content had multiple rendering issues:
- Incorrect sentence splitting for Arabic punctuation in citation logic
- Misaligned text in chat messages and markdown components
- Improper positioning of blockquotes and “think” sections
- Incorrect table alignment
- Citation placement ambiguity in RTL prompts
- UI layout inconsistencies when mixing LTR and RTL text
This PR introduces backend and frontend improvements to properly detect,
render, and style RTL content while preserving existing LTR behavior.
#### Backend
- Updated sentence boundary regex in `rag/nlp/search.py` to include
Arabic punctuation:
- `،` (comma)
- `؛` (semicolon)
- `؟` (question mark)
- `۔` (Arabic full stop)
- Ensures citation insertion works correctly in RTL sentences.
- Updated citation prompt instructions to clarify citation placement
rules for RTL languages.
#### Frontend
- Introduced a new utility: `text-direction.ts`
- Detects text direction based on Unicode ranges.
- Supports Arabic, Hebrew, Syriac, Thaana, and related scripts.
- Provides `getDirAttribute()` for automatic `dir` assignment.
- Applied dynamic `dir` attributes across:
- Markdown rendering
- Chat messages
- Search results
- Tables
- Hover cards and reference popovers
- Added proper RTL styling in LESS:
- Text alignment adjustments
- Blockquote border flipping
- Section indentation correction
- Table direction switching
- Use of `<bdi>` for figure labels to prevent bidirectional conflicts
#### DevOps / Environment
- Added Windows backend launch script with retry handling.
- Updated dependency metadata.
- Adjusted development-only React debugging behavior.
---
### Type of change
- [x] Bug Fix (non-breaking change which fixes RTL rendering and
citation issues)
- [x] New Feature (non-breaking change which adds RTL detection and
dynamic direction handling)
---------
Co-authored-by: 6ba3i <isbaaoui09@gmail.com>
Co-authored-by: Ahmad Intisar <ahmadintisar@Ahmads-MacBook-M4-Pro.local>
Co-authored-by: Ahmad Intisar <168020872+ahmadintisar@users.noreply.github.com>
Co-authored-by: Liu An <asiro@qq.com>
2026-03-02 08:03:44 +03:00
normalized_answer = normalize_arabic_digits ( answer ) or " "
2025-06-05 13:00:43 +08:00
def safe_add ( i ) :
if 0 < = i < max_index :
idx . add ( i )
return True
return False
Feature rtl support (#13118)
### What problem does this PR solve?
This PR adds comprehensive **Right-to-Left (RTL) language support**,
primarily targeting Arabic and other RTL scripts (Hebrew, Persian, Urdu,
etc.).
Previously, RTL content had multiple rendering issues:
- Incorrect sentence splitting for Arabic punctuation in citation logic
- Misaligned text in chat messages and markdown components
- Improper positioning of blockquotes and “think” sections
- Incorrect table alignment
- Citation placement ambiguity in RTL prompts
- UI layout inconsistencies when mixing LTR and RTL text
This PR introduces backend and frontend improvements to properly detect,
render, and style RTL content while preserving existing LTR behavior.
#### Backend
- Updated sentence boundary regex in `rag/nlp/search.py` to include
Arabic punctuation:
- `،` (comma)
- `؛` (semicolon)
- `؟` (question mark)
- `۔` (Arabic full stop)
- Ensures citation insertion works correctly in RTL sentences.
- Updated citation prompt instructions to clarify citation placement
rules for RTL languages.
#### Frontend
- Introduced a new utility: `text-direction.ts`
- Detects text direction based on Unicode ranges.
- Supports Arabic, Hebrew, Syriac, Thaana, and related scripts.
- Provides `getDirAttribute()` for automatic `dir` assignment.
- Applied dynamic `dir` attributes across:
- Markdown rendering
- Chat messages
- Search results
- Tables
- Hover cards and reference popovers
- Added proper RTL styling in LESS:
- Text alignment adjustments
- Blockquote border flipping
- Section indentation correction
- Table direction switching
- Use of `<bdi>` for figure labels to prevent bidirectional conflicts
#### DevOps / Environment
- Added Windows backend launch script with retry handling.
- Updated dependency metadata.
- Adjusted development-only React debugging behavior.
---
### Type of change
- [x] Bug Fix (non-breaking change which fixes RTL rendering and
citation issues)
- [x] New Feature (non-breaking change which adds RTL detection and
dynamic direction handling)
---------
Co-authored-by: 6ba3i <isbaaoui09@gmail.com>
Co-authored-by: Ahmad Intisar <ahmadintisar@Ahmads-MacBook-M4-Pro.local>
Co-authored-by: Ahmad Intisar <168020872+ahmadintisar@users.noreply.github.com>
Co-authored-by: Liu An <asiro@qq.com>
2026-03-02 08:03:44 +03:00
def find_and_replace ( pattern , group_index = 1 , repl = lambda digits : f " ID: { digits } " ) :
2025-06-05 13:00:43 +08:00
nonlocal answer
Feature rtl support (#13118)
### What problem does this PR solve?
This PR adds comprehensive **Right-to-Left (RTL) language support**,
primarily targeting Arabic and other RTL scripts (Hebrew, Persian, Urdu,
etc.).
Previously, RTL content had multiple rendering issues:
- Incorrect sentence splitting for Arabic punctuation in citation logic
- Misaligned text in chat messages and markdown components
- Improper positioning of blockquotes and “think” sections
- Incorrect table alignment
- Citation placement ambiguity in RTL prompts
- UI layout inconsistencies when mixing LTR and RTL text
This PR introduces backend and frontend improvements to properly detect,
render, and style RTL content while preserving existing LTR behavior.
#### Backend
- Updated sentence boundary regex in `rag/nlp/search.py` to include
Arabic punctuation:
- `،` (comma)
- `؛` (semicolon)
- `؟` (question mark)
- `۔` (Arabic full stop)
- Ensures citation insertion works correctly in RTL sentences.
- Updated citation prompt instructions to clarify citation placement
rules for RTL languages.
#### Frontend
- Introduced a new utility: `text-direction.ts`
- Detects text direction based on Unicode ranges.
- Supports Arabic, Hebrew, Syriac, Thaana, and related scripts.
- Provides `getDirAttribute()` for automatic `dir` assignment.
- Applied dynamic `dir` attributes across:
- Markdown rendering
- Chat messages
- Search results
- Tables
- Hover cards and reference popovers
- Added proper RTL styling in LESS:
- Text alignment adjustments
- Blockquote border flipping
- Section indentation correction
- Table direction switching
- Use of `<bdi>` for figure labels to prevent bidirectional conflicts
#### DevOps / Environment
- Added Windows backend launch script with retry handling.
- Updated dependency metadata.
- Adjusted development-only React debugging behavior.
---
### Type of change
- [x] Bug Fix (non-breaking change which fixes RTL rendering and
citation issues)
- [x] New Feature (non-breaking change which adds RTL detection and
dynamic direction handling)
---------
Co-authored-by: 6ba3i <isbaaoui09@gmail.com>
Co-authored-by: Ahmad Intisar <ahmadintisar@Ahmads-MacBook-M4-Pro.local>
Co-authored-by: Ahmad Intisar <168020872+ahmadintisar@users.noreply.github.com>
Co-authored-by: Liu An <asiro@qq.com>
2026-03-02 08:03:44 +03:00
nonlocal normalized_answer
2025-06-05 13:00:43 +08:00
Feature rtl support (#13118)
### What problem does this PR solve?
This PR adds comprehensive **Right-to-Left (RTL) language support**,
primarily targeting Arabic and other RTL scripts (Hebrew, Persian, Urdu,
etc.).
Previously, RTL content had multiple rendering issues:
- Incorrect sentence splitting for Arabic punctuation in citation logic
- Misaligned text in chat messages and markdown components
- Improper positioning of blockquotes and “think” sections
- Incorrect table alignment
- Citation placement ambiguity in RTL prompts
- UI layout inconsistencies when mixing LTR and RTL text
This PR introduces backend and frontend improvements to properly detect,
render, and style RTL content while preserving existing LTR behavior.
#### Backend
- Updated sentence boundary regex in `rag/nlp/search.py` to include
Arabic punctuation:
- `،` (comma)
- `؛` (semicolon)
- `؟` (question mark)
- `۔` (Arabic full stop)
- Ensures citation insertion works correctly in RTL sentences.
- Updated citation prompt instructions to clarify citation placement
rules for RTL languages.
#### Frontend
- Introduced a new utility: `text-direction.ts`
- Detects text direction based on Unicode ranges.
- Supports Arabic, Hebrew, Syriac, Thaana, and related scripts.
- Provides `getDirAttribute()` for automatic `dir` assignment.
- Applied dynamic `dir` attributes across:
- Markdown rendering
- Chat messages
- Search results
- Tables
- Hover cards and reference popovers
- Added proper RTL styling in LESS:
- Text alignment adjustments
- Blockquote border flipping
- Section indentation correction
- Table direction switching
- Use of `<bdi>` for figure labels to prevent bidirectional conflicts
#### DevOps / Environment
- Added Windows backend launch script with retry handling.
- Updated dependency metadata.
- Adjusted development-only React debugging behavior.
---
### Type of change
- [x] Bug Fix (non-breaking change which fixes RTL rendering and
citation issues)
- [x] New Feature (non-breaking change which adds RTL detection and
dynamic direction handling)
---------
Co-authored-by: 6ba3i <isbaaoui09@gmail.com>
Co-authored-by: Ahmad Intisar <ahmadintisar@Ahmads-MacBook-M4-Pro.local>
Co-authored-by: Ahmad Intisar <168020872+ahmadintisar@users.noreply.github.com>
Co-authored-by: Liu An <asiro@qq.com>
2026-03-02 08:03:44 +03:00
matches = list ( pattern . finditer ( normalized_answer ) )
if not matches :
return
parts = [ ]
last_idx = 0
for match in matches :
parts . append ( answer [ last_idx : match . start ( ) ] )
2025-06-05 13:00:43 +08:00
try :
i = int ( match . group ( group_index ) )
except Exception :
Feature rtl support (#13118)
### What problem does this PR solve?
This PR adds comprehensive **Right-to-Left (RTL) language support**,
primarily targeting Arabic and other RTL scripts (Hebrew, Persian, Urdu,
etc.).
Previously, RTL content had multiple rendering issues:
- Incorrect sentence splitting for Arabic punctuation in citation logic
- Misaligned text in chat messages and markdown components
- Improper positioning of blockquotes and “think” sections
- Incorrect table alignment
- Citation placement ambiguity in RTL prompts
- UI layout inconsistencies when mixing LTR and RTL text
This PR introduces backend and frontend improvements to properly detect,
render, and style RTL content while preserving existing LTR behavior.
#### Backend
- Updated sentence boundary regex in `rag/nlp/search.py` to include
Arabic punctuation:
- `،` (comma)
- `؛` (semicolon)
- `؟` (question mark)
- `۔` (Arabic full stop)
- Ensures citation insertion works correctly in RTL sentences.
- Updated citation prompt instructions to clarify citation placement
rules for RTL languages.
#### Frontend
- Introduced a new utility: `text-direction.ts`
- Detects text direction based on Unicode ranges.
- Supports Arabic, Hebrew, Syriac, Thaana, and related scripts.
- Provides `getDirAttribute()` for automatic `dir` assignment.
- Applied dynamic `dir` attributes across:
- Markdown rendering
- Chat messages
- Search results
- Tables
- Hover cards and reference popovers
- Added proper RTL styling in LESS:
- Text alignment adjustments
- Blockquote border flipping
- Section indentation correction
- Table direction switching
- Use of `<bdi>` for figure labels to prevent bidirectional conflicts
#### DevOps / Environment
- Added Windows backend launch script with retry handling.
- Updated dependency metadata.
- Adjusted development-only React debugging behavior.
---
### Type of change
- [x] Bug Fix (non-breaking change which fixes RTL rendering and
citation issues)
- [x] New Feature (non-breaking change which adds RTL detection and
dynamic direction handling)
---------
Co-authored-by: 6ba3i <isbaaoui09@gmail.com>
Co-authored-by: Ahmad Intisar <ahmadintisar@Ahmads-MacBook-M4-Pro.local>
Co-authored-by: Ahmad Intisar <168020872+ahmadintisar@users.noreply.github.com>
Co-authored-by: Liu An <asiro@qq.com>
2026-03-02 08:03:44 +03:00
parts . append ( answer [ match . start ( ) : match . end ( ) ] )
last_idx = match . end ( )
continue
if safe_add ( i ) :
digit_start , digit_end = match . span ( group_index )
digits_original = answer [ digit_start : digit_end ]
parts . append ( f " [ { repl ( digits_original ) } ] " )
else :
parts . append ( answer [ match . start ( ) : match . end ( ) ] )
last_idx = match . end ( )
2025-06-05 13:00:43 +08:00
Feature rtl support (#13118)
### What problem does this PR solve?
This PR adds comprehensive **Right-to-Left (RTL) language support**,
primarily targeting Arabic and other RTL scripts (Hebrew, Persian, Urdu,
etc.).
Previously, RTL content had multiple rendering issues:
- Incorrect sentence splitting for Arabic punctuation in citation logic
- Misaligned text in chat messages and markdown components
- Improper positioning of blockquotes and “think” sections
- Incorrect table alignment
- Citation placement ambiguity in RTL prompts
- UI layout inconsistencies when mixing LTR and RTL text
This PR introduces backend and frontend improvements to properly detect,
render, and style RTL content while preserving existing LTR behavior.
#### Backend
- Updated sentence boundary regex in `rag/nlp/search.py` to include
Arabic punctuation:
- `،` (comma)
- `؛` (semicolon)
- `؟` (question mark)
- `۔` (Arabic full stop)
- Ensures citation insertion works correctly in RTL sentences.
- Updated citation prompt instructions to clarify citation placement
rules for RTL languages.
#### Frontend
- Introduced a new utility: `text-direction.ts`
- Detects text direction based on Unicode ranges.
- Supports Arabic, Hebrew, Syriac, Thaana, and related scripts.
- Provides `getDirAttribute()` for automatic `dir` assignment.
- Applied dynamic `dir` attributes across:
- Markdown rendering
- Chat messages
- Search results
- Tables
- Hover cards and reference popovers
- Added proper RTL styling in LESS:
- Text alignment adjustments
- Blockquote border flipping
- Section indentation correction
- Table direction switching
- Use of `<bdi>` for figure labels to prevent bidirectional conflicts
#### DevOps / Environment
- Added Windows backend launch script with retry handling.
- Updated dependency metadata.
- Adjusted development-only React debugging behavior.
---
### Type of change
- [x] Bug Fix (non-breaking change which fixes RTL rendering and
citation issues)
- [x] New Feature (non-breaking change which adds RTL detection and
dynamic direction handling)
---------
Co-authored-by: 6ba3i <isbaaoui09@gmail.com>
Co-authored-by: Ahmad Intisar <ahmadintisar@Ahmads-MacBook-M4-Pro.local>
Co-authored-by: Ahmad Intisar <168020872+ahmadintisar@users.noreply.github.com>
Co-authored-by: Liu An <asiro@qq.com>
2026-03-02 08:03:44 +03:00
parts . append ( answer [ last_idx : ] )
answer = " " . join ( parts )
normalized_answer = normalize_arabic_digits ( answer ) or " "
2025-06-05 13:00:43 +08:00
for pattern in BAD_CITATION_PATTERNS :
find_and_replace ( pattern )
return answer , idx
2025-05-29 10:03:51 +08:00
2025-12-08 09:43:03 +08:00
async def async_chat ( dialog , messages , stream = True , * * kwargs ) :
2026-01-19 19:35:14 +08:00
logging . debug ( " Begin async_chat " )
2024-08-15 09:17:36 +08:00
assert messages [ - 1 ] [ " role " ] == " user " , " The last content of this conversation is not from user. "
2025-06-05 13:00:43 +08:00
if not dialog . kb_ids and not dialog . prompt_config . get ( " tavily_api_key " ) :
2025-12-08 09:43:03 +08:00
async for ans in async_chat_solo ( dialog , messages , stream ) :
2025-02-21 12:24:02 +08:00
yield ans
2025-12-08 09:43:03 +08:00
return
2024-12-19 18:13:33 +08:00
chat_start_ts = timer ( )
2026-02-11 09:47:33 +08:00
llm_type = TenantLLMService . llm_id2llm_type ( dialog . llm_id )
if llm_type == " image2text " :
2025-02-18 13:42:22 +08:00
llm_model_config = TenantLLMService . get_model_config ( dialog . tenant_id , LLMType . IMAGE2TEXT , dialog . llm_id )
2024-08-15 09:17:36 +08:00
else :
2025-02-18 13:42:22 +08:00
llm_model_config = TenantLLMService . get_model_config ( dialog . tenant_id , LLMType . CHAT , dialog . llm_id )
2026-02-11 09:47:33 +08:00
factory = llm_model_config . get ( " llm_factory " , " " ) if llm_model_config else " "
2025-02-18 13:42:22 +08:00
max_tokens = llm_model_config . get ( " max_tokens " , 8192 )
2024-12-19 18:13:33 +08:00
check_llm_ts = timer ( )
2025-03-24 13:18:47 +08:00
langfuse_tracer = None
2025-08-04 14:45:43 +08:00
trace_context = { }
2025-03-24 13:18:47 +08:00
langfuse_keys = TenantLangfuseService . filter_by_tenant ( tenant_id = dialog . tenant_id )
if langfuse_keys :
langfuse = Langfuse ( public_key = langfuse_keys . public_key , secret_key = langfuse_keys . secret_key , host = langfuse_keys . host )
2026-01-14 19:23:15 -08:00
try :
if langfuse . auth_check ( ) :
langfuse_tracer = langfuse
trace_id = langfuse_tracer . create_trace_id ( )
trace_context = { " trace_id " : trace_id }
except Exception :
# Skip langfuse tracing if connection fails
pass
2025-03-24 13:18:47 +08:00
check_langfuse_tracer_ts = timer ( )
2025-06-05 13:00:43 +08:00
kbs , embd_mdl , rerank_mdl , chat_mdl , tts_mdl = get_models ( dialog )
toolcall_session , tools = kwargs . get ( " toolcall_session " ) , kwargs . get ( " tools " )
if toolcall_session and tools :
chat_mdl . bind_tools ( toolcall_session , tools )
bind_models_ts = timer ( )
2024-12-19 18:13:33 +08:00
2025-11-06 09:36:38 +08:00
retriever = settings . retriever
2024-08-15 09:17:36 +08:00
questions = [ m [ " content " ] for m in messages if m [ " role " ] == " user " ] [ - 3 : ]
2025-08-12 14:12:56 +08:00
attachments = kwargs [ " doc_ids " ] . split ( " , " ) if " doc_ids " in kwargs else [ ]
2025-11-28 19:25:32 +08:00
attachments_ = " "
2026-02-11 09:47:33 +08:00
image_attachments = [ ]
image_files = [ ]
2024-08-15 09:17:36 +08:00
if " doc_ids " in messages [ - 1 ] :
attachments = messages [ - 1 ] [ " doc_ids " ]
2025-11-28 19:25:32 +08:00
if " files " in messages [ - 1 ] :
2026-02-11 09:47:33 +08:00
if llm_type == " chat " :
text_attachments , image_attachments = split_file_attachments ( messages [ - 1 ] [ " files " ] )
else :
text_attachments , image_files = split_file_attachments ( messages [ - 1 ] [ " files " ] , raw = True )
attachments_ = " \n \n " . join ( text_attachments )
2025-08-12 14:12:56 +08:00
2024-08-15 09:17:36 +08:00
prompt_config = dialog . prompt_config
field_map = KnowledgebaseService . get_field_map ( dialog . kb_ids )
2026-01-19 19:35:14 +08:00
logging . debug ( f " field_map retrieved: { field_map } " )
2024-08-15 09:17:36 +08:00
# try to use sql if field mapping is good to go
if field_map :
2024-11-14 17:13:48 +08:00
logging . debug ( " Use SQL to retrieval: {} " . format ( questions [ - 1 ] ) )
2025-12-11 17:38:17 +08:00
ans = await use_sql ( questions [ - 1 ] , field_map , dialog . tenant_id , chat_mdl , prompt_config . get ( " quote " , True ) , dialog . kb_ids )
2026-01-19 19:35:14 +08:00
# For aggregate queries (COUNT, SUM, etc.), chunks may be empty but answer is still valid
if ans and ( ans . get ( " reference " , { } ) . get ( " chunks " ) or ans . get ( " answer " ) ) :
2024-08-15 09:17:36 +08:00
yield ans
2025-12-08 09:43:03 +08:00
return
2026-01-19 19:35:14 +08:00
else :
logging . debug ( " SQL failed or returned no results, falling back to vector search " )
param_keys = [ p [ " key " ] for p in prompt_config . get ( " parameters " , [ ] ) ]
logging . debug ( f " attachments= { attachments } , param_keys= { param_keys } , embd_mdl= { embd_mdl } " )
2024-08-15 09:17:36 +08:00
for p in prompt_config [ " parameters " ] :
if p [ " key " ] == " knowledge " :
continue
if p [ " key " ] not in kwargs and not p [ " optional " ] :
raise KeyError ( " Miss parameter: " + p [ " key " ] )
if p [ " key " ] not in kwargs :
2025-03-24 13:18:47 +08:00
prompt_config [ " system " ] = prompt_config [ " system " ] . replace ( " { %s } " % p [ " key " ] , " " )
2024-08-15 09:17:36 +08:00
2024-09-20 17:25:55 +08:00
if len ( questions ) > 1 and prompt_config . get ( " refine_multiturn " ) :
2025-12-11 17:38:17 +08:00
questions = [ await full_question ( dialog . tenant_id , dialog . llm_id , messages ) ]
2024-09-20 17:25:55 +08:00
else :
questions = questions [ - 1 : ]
2024-12-19 18:13:33 +08:00
2025-05-09 15:32:02 +08:00
if prompt_config . get ( " cross_languages " ) :
2025-12-11 17:38:17 +08:00
questions = [ await cross_languages ( dialog . tenant_id , dialog . llm_id , questions [ 0 ] , prompt_config [ " cross_languages " ] ) ]
2025-05-09 15:32:02 +08:00
2025-08-12 14:12:56 +08:00
if dialog . meta_data_filter :
2026-01-28 13:29:34 +08:00
metas = DocMetadataService . get_flatted_meta_by_kbs ( dialog . kb_ids )
2025-12-12 17:12:38 +08:00
attachments = await apply_meta_data_filter (
dialog . meta_data_filter ,
metas ,
questions [ - 1 ] ,
chat_mdl ,
attachments ,
)
2025-08-12 14:12:56 +08:00
2025-06-05 13:00:43 +08:00
if prompt_config . get ( " keyword " , False ) :
2025-12-11 17:38:17 +08:00
questions [ - 1 ] + = await keyword_extraction ( chat_mdl , questions [ - 1 ] )
2024-09-20 17:25:55 +08:00
2025-06-05 13:00:43 +08:00
refine_question_ts = timer ( )
2024-08-15 09:17:36 +08:00
2025-02-20 17:41:01 +08:00
thought = " "
kbinfos = { " total " : 0 , " chunks " : [ ] , " doc_aggs " : [ ] }
2025-08-13 12:43:31 +08:00
knowledges = [ ]
2024-12-19 18:13:33 +08:00
2026-01-19 19:35:14 +08:00
if attachments is not None and " knowledge " in param_keys :
logging . debug ( " Proceeding with retrieval " )
2024-10-29 13:19:01 +08:00
tenant_ids = list ( set ( [ kb . tenant_id for kb in kbs ] ) )
2025-02-20 17:41:01 +08:00
knowledges = [ ]
2026-01-22 15:34:08 +08:00
if prompt_config . get ( " reasoning " , False ) or kwargs . get ( " reasoning " ) :
2025-03-24 13:18:47 +08:00
reasoner = DeepResearcher (
chat_mdl ,
prompt_config ,
2025-08-15 17:44:58 +08:00
partial (
retriever . retrieval ,
embd_mdl = embd_mdl ,
tenant_ids = tenant_ids ,
kb_ids = dialog . kb_ids ,
page = 1 ,
page_size = dialog . top_n ,
similarity_threshold = 0.2 ,
vector_similarity_weight = 0.3 ,
doc_ids = attachments ,
) ,
2025-03-24 13:18:47 +08:00
)
2026-01-13 09:41:35 +08:00
queue = asyncio . Queue ( )
async def callback ( msg : str ) :
nonlocal queue
await queue . put ( msg + " <br/> " )
await callback ( " <START_DEEP_RESEARCH> " )
task = asyncio . create_task ( reasoner . research ( kbinfos , questions [ - 1 ] , questions [ - 1 ] , callback = callback ) )
while True :
msg = await queue . get ( )
if msg . find ( " <START_DEEP_RESEARCH> " ) == 0 :
yield { " answer " : " " , " reference " : { } , " audio_binary " : None , " final " : False , " start_to_think " : True }
elif msg . find ( " <END_DEEP_RESEARCH> " ) == 0 :
yield { " answer " : " " , " reference " : { } , " audio_binary " : None , " final " : False , " end_to_think " : True }
break
else :
yield { " answer " : msg , " reference " : { } , " audio_binary " : None , " final " : False }
await task
2026-01-15 12:28:49 +08:00
2025-02-20 17:41:01 +08:00
else :
2025-06-05 13:00:43 +08:00
if embd_mdl :
2026-01-15 12:28:49 +08:00
kbinfos = await retriever . retrieval (
2025-06-05 13:00:43 +08:00
" " . join ( questions ) ,
embd_mdl ,
tenant_ids ,
dialog . kb_ids ,
1 ,
dialog . top_n ,
dialog . similarity_threshold ,
dialog . vector_similarity_weight ,
doc_ids = attachments ,
top = dialog . top_k ,
2025-12-23 15:57:27 +08:00
aggs = True ,
2025-06-05 13:00:43 +08:00
rerank_mdl = rerank_mdl ,
rank_feature = label_question ( " " . join ( questions ) , kbs ) ,
)
2025-10-10 17:07:55 +08:00
if prompt_config . get ( " toc_enhance " ) :
2026-01-07 15:35:30 +08:00
cks = await retriever . retrieval_by_toc ( " " . join ( questions ) , kbinfos [ " chunks " ] , tenant_ids , chat_mdl , dialog . top_n )
2025-10-10 17:07:55 +08:00
if cks :
kbinfos [ " chunks " ] = cks
2025-12-01 14:03:09 +08:00
kbinfos [ " chunks " ] = retriever . retrieval_by_children ( kbinfos [ " chunks " ] , tenant_ids )
2025-02-26 10:21:04 +08:00
if prompt_config . get ( " tavily_api_key " ) :
tav = Tavily ( prompt_config [ " tavily_api_key " ] )
tav_res = tav . retrieve_chunks ( " " . join ( questions ) )
kbinfos [ " chunks " ] . extend ( tav_res [ " chunks " ] )
kbinfos [ " doc_aggs " ] . extend ( tav_res [ " doc_aggs " ] )
2025-02-20 17:41:01 +08:00
if prompt_config . get ( " use_kg " ) :
2026-03-05 17:27:17 +08:00
default_chat_model = get_tenant_default_model_by_type ( dialog . tenant_id , LLMType . CHAT )
2025-12-31 14:40:27 +08:00
ck = await settings . kg_retriever . retrieval ( " " . join ( questions ) , tenant_ids , dialog . kb_ids , embd_mdl ,
2026-03-05 17:27:17 +08:00
LLMBundle ( dialog . tenant_id , default_chat_model ) )
2025-02-20 17:41:01 +08:00
if ck [ " content_with_weight " ] :
kbinfos [ " chunks " ] . insert ( 0 , ck )
2026-01-13 09:41:35 +08:00
knowledges = kb_prompt ( kbinfos , max_tokens )
2025-03-24 13:18:47 +08:00
logging . debug ( " {} -> {} " . format ( " " . join ( questions ) , " \n -> " . join ( knowledges ) ) )
2024-08-15 09:17:36 +08:00
2025-02-20 17:41:01 +08:00
retrieval_ts = timer ( )
2024-08-15 09:17:36 +08:00
if not knowledges and prompt_config . get ( " empty_response " ) :
2024-09-03 19:49:14 +08:00
empty_res = prompt_config [ " empty_response " ]
2025-09-23 10:19:25 +08:00
yield { " answer " : empty_res , " reference " : kbinfos , " prompt " : " \n \n ### Query: \n %s " % " " . join ( questions ) ,
2026-01-08 13:34:16 +08:00
" audio_binary " : tts ( tts_mdl , empty_res ) , " final " : True }
2025-12-08 09:43:03 +08:00
return
2024-08-15 09:17:36 +08:00
2025-01-13 14:35:24 +08:00
kwargs [ " knowledge " ] = " \n ------ \n " + " \n \n ------ \n \n " . join ( knowledges )
2024-08-15 09:17:36 +08:00
gen_conf = dialog . llm_setting
2025-11-28 19:25:32 +08:00
msg = [ { " role " : " system " , " content " : prompt_config [ " system " ] . format ( * * kwargs ) + attachments_ } ]
2025-03-11 19:56:21 +08:00
prompt4citation = " "
if knowledges and ( prompt_config . get ( " quote " , True ) and kwargs . get ( " quote " , True ) ) :
prompt4citation = citation_prompt ( )
2025-03-24 13:18:47 +08:00
msg . extend ( [ { " role " : m [ " role " ] , " content " : re . sub ( r " ## \ d+ \ $ \ $ " , " " , m [ " content " ] ) } for m in messages if m [ " role " ] != " system " ] )
2025-03-11 19:56:21 +08:00
used_token_count , msg = message_fit_in ( msg , int ( max_tokens * 0.95 ) )
2026-02-11 09:47:33 +08:00
if llm_type == " chat " and image_attachments :
convert_last_user_msg_to_multimodal ( msg , image_attachments , factory )
2024-08-15 09:17:36 +08:00
assert len ( msg ) > = 2 , f " message_fit_in has bug: { msg } "
2024-08-26 16:14:15 +08:00
prompt = msg [ 0 ] [ " content " ]
2024-08-15 09:17:36 +08:00
if " max_tokens " in gen_conf :
2025-03-24 13:18:47 +08:00
gen_conf [ " max_tokens " ] = min ( gen_conf [ " max_tokens " ] , max_tokens - used_token_count )
2024-08-15 09:17:36 +08:00
def decorate_answer ( answer ) :
2025-06-05 13:00:43 +08:00
nonlocal embd_mdl , prompt_config , knowledges , kwargs , kbinfos , prompt , retrieval_ts , questions , langfuse_tracer
2024-12-19 18:13:33 +08:00
2024-08-15 09:17:36 +08:00
refs = [ ]
2025-02-20 17:41:01 +08:00
ans = answer . split ( " </think> " )
think = " "
if len ( ans ) == 2 :
think = ans [ 0 ] + " </think> "
answer = ans [ 1 ]
2025-04-15 09:33:53 +08:00
2024-08-15 09:17:36 +08:00
if knowledges and ( prompt_config . get ( " quote " , True ) and kwargs . get ( " quote " , True ) ) :
2025-04-15 09:33:53 +08:00
idx = set ( [ ] )
Feature rtl support (#13118)
### What problem does this PR solve?
This PR adds comprehensive **Right-to-Left (RTL) language support**,
primarily targeting Arabic and other RTL scripts (Hebrew, Persian, Urdu,
etc.).
Previously, RTL content had multiple rendering issues:
- Incorrect sentence splitting for Arabic punctuation in citation logic
- Misaligned text in chat messages and markdown components
- Improper positioning of blockquotes and “think” sections
- Incorrect table alignment
- Citation placement ambiguity in RTL prompts
- UI layout inconsistencies when mixing LTR and RTL text
This PR introduces backend and frontend improvements to properly detect,
render, and style RTL content while preserving existing LTR behavior.
#### Backend
- Updated sentence boundary regex in `rag/nlp/search.py` to include
Arabic punctuation:
- `،` (comma)
- `؛` (semicolon)
- `؟` (question mark)
- `۔` (Arabic full stop)
- Ensures citation insertion works correctly in RTL sentences.
- Updated citation prompt instructions to clarify citation placement
rules for RTL languages.
#### Frontend
- Introduced a new utility: `text-direction.ts`
- Detects text direction based on Unicode ranges.
- Supports Arabic, Hebrew, Syriac, Thaana, and related scripts.
- Provides `getDirAttribute()` for automatic `dir` assignment.
- Applied dynamic `dir` attributes across:
- Markdown rendering
- Chat messages
- Search results
- Tables
- Hover cards and reference popovers
- Added proper RTL styling in LESS:
- Text alignment adjustments
- Blockquote border flipping
- Section indentation correction
- Table direction switching
- Use of `<bdi>` for figure labels to prevent bidirectional conflicts
#### DevOps / Environment
- Added Windows backend launch script with retry handling.
- Updated dependency metadata.
- Adjusted development-only React debugging behavior.
---
### Type of change
- [x] Bug Fix (non-breaking change which fixes RTL rendering and
citation issues)
- [x] New Feature (non-breaking change which adds RTL detection and
dynamic direction handling)
---------
Co-authored-by: 6ba3i <isbaaoui09@gmail.com>
Co-authored-by: Ahmad Intisar <ahmadintisar@Ahmads-MacBook-M4-Pro.local>
Co-authored-by: Ahmad Intisar <168020872+ahmadintisar@users.noreply.github.com>
Co-authored-by: Liu An <asiro@qq.com>
2026-03-02 08:03:44 +03:00
normalized_answer = normalize_arabic_digits ( answer ) or " "
if embd_mdl and not CITATION_MARKER_PATTERN . search ( normalized_answer ) :
2025-03-24 13:18:47 +08:00
answer , idx = retriever . insert_citations (
answer ,
[ ck [ " content_ltks " ] for ck in kbinfos [ " chunks " ] ] ,
[ ck [ " vector " ] for ck in kbinfos [ " chunks " ] ] ,
embd_mdl ,
tkweight = 1 - dialog . vector_similarity_weight ,
vtweight = dialog . vector_similarity_weight ,
)
2025-03-11 19:56:21 +08:00
else :
Feature rtl support (#13118)
### What problem does this PR solve?
This PR adds comprehensive **Right-to-Left (RTL) language support**,
primarily targeting Arabic and other RTL scripts (Hebrew, Persian, Urdu,
etc.).
Previously, RTL content had multiple rendering issues:
- Incorrect sentence splitting for Arabic punctuation in citation logic
- Misaligned text in chat messages and markdown components
- Improper positioning of blockquotes and “think” sections
- Incorrect table alignment
- Citation placement ambiguity in RTL prompts
- UI layout inconsistencies when mixing LTR and RTL text
This PR introduces backend and frontend improvements to properly detect,
render, and style RTL content while preserving existing LTR behavior.
#### Backend
- Updated sentence boundary regex in `rag/nlp/search.py` to include
Arabic punctuation:
- `،` (comma)
- `؛` (semicolon)
- `؟` (question mark)
- `۔` (Arabic full stop)
- Ensures citation insertion works correctly in RTL sentences.
- Updated citation prompt instructions to clarify citation placement
rules for RTL languages.
#### Frontend
- Introduced a new utility: `text-direction.ts`
- Detects text direction based on Unicode ranges.
- Supports Arabic, Hebrew, Syriac, Thaana, and related scripts.
- Provides `getDirAttribute()` for automatic `dir` assignment.
- Applied dynamic `dir` attributes across:
- Markdown rendering
- Chat messages
- Search results
- Tables
- Hover cards and reference popovers
- Added proper RTL styling in LESS:
- Text alignment adjustments
- Blockquote border flipping
- Section indentation correction
- Table direction switching
- Use of `<bdi>` for figure labels to prevent bidirectional conflicts
#### DevOps / Environment
- Added Windows backend launch script with retry handling.
- Updated dependency metadata.
- Adjusted development-only React debugging behavior.
---
### Type of change
- [x] Bug Fix (non-breaking change which fixes RTL rendering and
citation issues)
- [x] New Feature (non-breaking change which adds RTL detection and
dynamic direction handling)
---------
Co-authored-by: 6ba3i <isbaaoui09@gmail.com>
Co-authored-by: Ahmad Intisar <ahmadintisar@Ahmads-MacBook-M4-Pro.local>
Co-authored-by: Ahmad Intisar <168020872+ahmadintisar@users.noreply.github.com>
Co-authored-by: Liu An <asiro@qq.com>
2026-03-02 08:03:44 +03:00
for match in CITATION_MARKER_PATTERN . finditer ( normalized_answer ) :
2025-04-15 09:33:53 +08:00
i = int ( match . group ( 1 ) )
if i < len ( kbinfos [ " chunks " ] ) :
idx . add ( i )
2025-05-19 19:34:05 +08:00
answer , idx = repair_bad_citation_formats ( answer , kbinfos , idx )
2025-03-11 19:56:21 +08:00
2024-08-15 09:17:36 +08:00
idx = set ( [ kbinfos [ " chunks " ] [ int ( i ) ] [ " doc_id " ] for i in idx ] )
2025-03-24 13:18:47 +08:00
recall_docs = [ d for d in kbinfos [ " doc_aggs " ] if d [ " doc_id " ] in idx ]
2024-12-08 14:21:12 +08:00
if not recall_docs :
recall_docs = kbinfos [ " doc_aggs " ]
2024-08-15 09:17:36 +08:00
kbinfos [ " doc_aggs " ] = recall_docs
refs = deepcopy ( kbinfos )
for c in refs [ " chunks " ] :
if c . get ( " vector " ) :
del c [ " vector " ]
if answer . lower ( ) . find ( " invalid key " ) > = 0 or answer . lower ( ) . find ( " invalid api " ) > = 0 :
2024-12-07 11:04:36 +08:00
answer + = " Please set LLM API-Key in ' User Setting -> Model providers -> API-Key ' "
2024-12-19 18:13:33 +08:00
finish_chat_ts = timer ( )
total_time_cost = ( finish_chat_ts - chat_start_ts ) * 1000
check_llm_time_cost = ( check_llm_ts - chat_start_ts ) * 1000
2025-03-24 13:18:47 +08:00
check_langfuse_tracer_cost = ( check_langfuse_tracer_ts - check_llm_ts ) * 1000
2025-06-05 13:00:43 +08:00
bind_embedding_time_cost = ( bind_models_ts - check_langfuse_tracer_ts ) * 1000
refine_question_time_cost = ( refine_question_ts - bind_models_ts ) * 1000
retrieval_time_cost = ( retrieval_ts - refine_question_ts ) * 1000
2024-12-19 18:13:33 +08:00
generate_result_time_cost = ( finish_chat_ts - retrieval_ts ) * 1000
2025-03-24 13:18:47 +08:00
tk_num = num_tokens_from_string ( think + answer )
2025-03-03 13:12:38 +08:00
prompt + = " \n \n ### Query: \n %s " % " " . join ( questions )
2025-03-14 13:37:31 +08:00
prompt = (
2025-03-24 13:18:47 +08:00
f " { prompt } \n \n "
" ## Time elapsed: \n "
f " - Total: { total_time_cost : .1f } ms \n "
f " - Check LLM: { check_llm_time_cost : .1f } ms \n "
f " - Check Langfuse tracer: { check_langfuse_tracer_cost : .1f } ms \n "
2025-06-05 13:00:43 +08:00
f " - Bind models: { bind_embedding_time_cost : .1f } ms \n "
f " - Query refinement(LLM): { refine_question_time_cost : .1f } ms \n "
2025-03-24 13:18:47 +08:00
f " - Retrieval: { retrieval_time_cost : .1f } ms \n "
f " - Generate answer: { generate_result_time_cost : .1f } ms \n \n "
" ## Token usage: \n "
f " - Generated tokens(approximately): { tk_num } \n "
f " - Token speed: { int ( tk_num / ( generate_result_time_cost / 1000.0 ) ) } /s "
2025-03-14 13:37:31 +08:00
)
2025-03-24 13:18:47 +08:00
2025-03-24 15:14:36 +08:00
# Add a condition check to call the end method only if langfuse_tracer exists
2025-04-08 16:09:03 +08:00
if langfuse_tracer and " langfuse_generation " in locals ( ) :
2025-08-04 14:45:43 +08:00
langfuse_output = " \n " + re . sub ( r " ^.*?(### Query:.*) " , r " \ 1 " , prompt , flags = re . DOTALL )
langfuse_output = { " time_elapsed: " : re . sub ( r " \ n " , " \n " , langfuse_output ) , " created_at " : time . time ( ) }
langfuse_generation . update ( output = langfuse_output )
langfuse_generation . end ( )
2025-03-24 13:18:47 +08:00
return { " answer " : think + answer , " reference " : refs , " prompt " : re . sub ( r " \ n " , " \n " , prompt ) , " created_at " : time . time ( ) }
if langfuse_tracer :
2025-08-04 14:45:43 +08:00
langfuse_generation = langfuse_tracer . start_generation (
2025-09-23 10:19:25 +08:00
trace_context = trace_context , name = " chat " , model = llm_model_config [ " llm_name " ] ,
input = { " prompt " : prompt , " prompt4citation " : prompt4citation , " messages " : msg }
2025-08-04 14:45:43 +08:00
)
2024-08-15 09:17:36 +08:00
if stream :
2026-02-11 09:47:33 +08:00
if llm_type == " chat " :
stream_iter = chat_mdl . async_chat_streamly_delta ( prompt + prompt4citation , msg [ 1 : ] , gen_conf )
else :
stream_iter = chat_mdl . async_chat_streamly_delta ( prompt + prompt4citation , msg [ 1 : ] , gen_conf , images = image_files )
2026-01-08 13:34:16 +08:00
last_state = None
async for kind , value , state in _stream_with_think_delta ( stream_iter ) :
last_state = state
if kind == " marker " :
flags = { " start_to_think " : True } if value == " <think> " else { " end_to_think " : True }
yield { " answer " : " " , " reference " : { } , " audio_binary " : None , " final " : False , * * flags }
2024-09-03 19:49:14 +08:00
continue
2026-01-08 13:34:16 +08:00
yield { " answer " : value , " reference " : { } , " audio_binary " : tts ( tts_mdl , value ) , " final " : False }
full_answer = last_state . full_text if last_state else " "
if full_answer :
final = decorate_answer ( thought + full_answer )
final [ " final " ] = True
final [ " audio_binary " ] = None
final [ " answer " ] = " "
yield final
2024-08-15 09:17:36 +08:00
else :
2026-02-11 09:47:33 +08:00
if llm_type == " chat " :
answer = await chat_mdl . async_chat ( prompt + prompt4citation , msg [ 1 : ] , gen_conf )
else :
answer = await chat_mdl . async_chat ( prompt + prompt4citation , msg [ 1 : ] , gen_conf , images = image_files )
2025-02-13 23:27:01 -03:00
user_content = msg [ - 1 ] . get ( " content " , " [content not available] " )
logging . debug ( " User: {} |Assistant: {} " . format ( user_content , answer ) )
2024-09-03 19:49:14 +08:00
res = decorate_answer ( answer )
res [ " audio_binary " ] = tts ( tts_mdl , answer )
yield res
2024-08-15 09:17:36 +08:00
2025-12-08 09:43:03 +08:00
return
2025-11-16 19:29:20 +08:00
2024-08-15 09:17:36 +08:00
2025-12-11 17:38:17 +08:00
async def use_sql ( question , field_map , tenant_id , chat_mdl , quota = True , kb_ids = None ) :
2026-01-19 19:35:14 +08:00
logging . debug ( f " use_sql: Question: { question } " )
# Determine which document engine we're using
2026-01-31 15:11:54 +08:00
if settings . DOC_ENGINE_INFINITY :
doc_engine = " infinity "
elif settings . DOC_ENGINE_OCEANBASE :
doc_engine = " oceanbase "
else :
doc_engine = " es "
2026-01-19 19:35:14 +08:00
# Construct the full table name
# For Elasticsearch: ragflow_{tenant_id} (kb_id is in WHERE clause)
# For Infinity: ragflow_{tenant_id}_{kb_id} (each KB has its own table)
base_table = index_name ( tenant_id )
if doc_engine == " infinity " and kb_ids and len ( kb_ids ) == 1 :
# Infinity: append kb_id to table name
table_name = f " { base_table } _ { kb_ids [ 0 ] } "
logging . debug ( f " use_sql: Using Infinity table name: { table_name } " )
else :
# Elasticsearch/OpenSearch: use base index name
table_name = base_table
logging . debug ( f " use_sql: Using ES/OS table name: { table_name } " )
2026-03-04 19:10:06 +08:00
expected_doc_name_column = " docnm " if doc_engine == " infinity " else " docnm_kwd "
def has_source_columns ( columns ) :
normalized_names = { str ( col . get ( " name " , " " ) ) . lower ( ) for col in columns }
return " doc_id " in normalized_names and bool ( { " docnm_kwd " , " docnm " } & normalized_names )
def is_aggregate_sql ( sql_text ) :
return bool ( re . search ( r " (count|sum|avg|max|min|distinct) \ s* \ ( " , ( sql_text or " " ) . lower ( ) ) )
def normalize_sql ( sql ) :
logging . debug ( f " use_sql: Raw SQL from LLM: { repr ( sql [ : 500 ] ) } " )
# Remove think blocks if present (format: </think>...)
sql = re . sub ( r " </think> \ n.*? \ n \ s* " , " " , sql , flags = re . DOTALL )
sql = re . sub ( r " 思考 \ n.*? \ n " , " " , sql , flags = re . DOTALL )
# Remove markdown code blocks (```sql ... ```)
sql = re . sub ( r " ```(?:sql)? \ s* " , " " , sql , flags = re . IGNORECASE )
sql = re . sub ( r " ``` \ s*$ " , " " , sql , flags = re . IGNORECASE )
# Remove trailing semicolon that ES SQL parser doesn't like
return sql . rstrip ( ) . rstrip ( ' ; ' ) . strip ( )
def add_kb_filter ( sql ) :
# Add kb_id filter for ES/OS only (Infinity already has it in table name)
if doc_engine == " infinity " or not kb_ids :
return sql
# Build kb_filter: single KB or multiple KBs with OR
if len ( kb_ids ) == 1 :
kb_filter = f " kb_id = ' { kb_ids [ 0 ] } ' "
else :
kb_filter = " ( " + " OR " . join ( [ f " kb_id = ' { kb_id } ' " for kb_id in kb_ids ] ) + " ) "
if " where " not in sql . lower ( ) :
o = sql . lower ( ) . split ( " order by " )
if len ( o ) > 1 :
sql = o [ 0 ] + f " WHERE { kb_filter } order by " + o [ 1 ]
else :
sql + = f " WHERE { kb_filter } "
elif " kb_id = " not in sql . lower ( ) and " kb_id= " not in sql . lower ( ) :
sql = re . sub ( r " \ bwhere \ b " , f " where { kb_filter } and " , sql , flags = re . IGNORECASE )
return sql
2026-02-09 14:56:10 +08:00
def is_row_count_question ( q : str ) - > bool :
q = ( q or " " ) . lower ( )
if not re . search ( r " \ bhow many rows \ b| \ bnumber of rows \ b| \ brow count \ b " , q ) :
return False
return bool ( re . search ( r " \ bdataset \ b| \ btable \ b| \ bspreadsheet \ b| \ bexcel \ b " , q ) )
2026-01-19 19:35:14 +08:00
# Generate engine-specific SQL prompts
if doc_engine == " infinity " :
# Build Infinity prompts with JSON extraction context
json_field_names = list ( field_map . keys ( ) )
2026-02-09 14:56:10 +08:00
row_count_override = (
f " SELECT COUNT(*) AS rows FROM { table_name } "
if is_row_count_question ( question )
else None
)
2026-01-19 19:35:14 +08:00
sys_prompt = """ You are a Database Administrator. Write SQL for a table with JSON ' chunk_data ' column.
JSON Extraction: json_extract_string(chunk_data, ' $.FieldName ' )
Numeric Cast: CAST(json_extract_string(chunk_data, ' $.FieldName ' ) AS INTEGER/FLOAT)
NULL Check: json_extract_isnull(chunk_data, ' $.FieldName ' ) == false
RULES:
1. Use EXACT field names (case-sensitive) from the list below
2. For SELECT: include doc_id, docnm, and json_extract_string() for requested fields
3. For COUNT: use COUNT(*) or COUNT(DISTINCT json_extract_string(...))
4. Add AS alias for extracted field names
5. DO NOT select ' content ' field
6. Only add NULL check (json_extract_isnull() == false) in WHERE clause when:
- Question asks to " show me " or " display " specific columns
- Question mentions " not null " or " excluding null "
- Add NULL check for count specific column
- DO NOT add NULL check for COUNT(*) queries (COUNT(*) counts all rows including nulls)
7. Output ONLY the SQL, no explanations """
user_prompt = """ Table: {}
Fields (EXACT case): {}
2024-08-15 09:17:36 +08:00
{}
2026-01-19 19:35:14 +08:00
Question: {}
Write SQL using json_extract_string() with exact field names. Include doc_id, docnm for data queries. Only SQL. """ . format (
table_name ,
" , " . join ( json_field_names ) ,
" \n " . join ( [ f " - { field } " for field in json_field_names ] ) ,
question
)
2026-01-31 15:11:54 +08:00
elif doc_engine == " oceanbase " :
# Build OceanBase prompts with JSON extraction context
json_field_names = list ( field_map . keys ( ) )
2026-02-09 14:56:10 +08:00
row_count_override = (
f " SELECT COUNT(*) AS rows FROM { table_name } "
if is_row_count_question ( question )
else None
)
2026-01-31 15:11:54 +08:00
sys_prompt = """ You are a Database Administrator. Write SQL for a table with JSON ' chunk_data ' column.
JSON Extraction: json_extract_string(chunk_data, ' $.FieldName ' )
Numeric Cast: CAST(json_extract_string(chunk_data, ' $.FieldName ' ) AS INTEGER/FLOAT)
NULL Check: json_extract_isnull(chunk_data, ' $.FieldName ' ) == false
RULES:
1. Use EXACT field names (case-sensitive) from the list below
2. For SELECT: include doc_id, docnm_kwd, and json_extract_string() for requested fields
3. For COUNT: use COUNT(*) or COUNT(DISTINCT json_extract_string(...))
4. Add AS alias for extracted field names
5. DO NOT select ' content ' field
6. Only add NULL check (json_extract_isnull() == false) in WHERE clause when:
- Question asks to " show me " or " display " specific columns
- Question mentions " not null " or " excluding null "
- Add NULL check for count specific column
- DO NOT add NULL check for COUNT(*) queries (COUNT(*) counts all rows including nulls)
7. Output ONLY the SQL, no explanations """
user_prompt = """ Table: {}
Fields (EXACT case): {}
{}
Question: {}
Write SQL using json_extract_string() with exact field names. Include doc_id, docnm_kwd for data queries. Only SQL. """ . format (
table_name ,
" , " . join ( json_field_names ) ,
" \n " . join ( [ f " - { field } " for field in json_field_names ] ) ,
question
)
2026-01-19 19:35:14 +08:00
else :
# Build ES/OS prompts with direct field access
2026-02-09 14:56:10 +08:00
row_count_override = None
2026-01-19 19:35:14 +08:00
sys_prompt = """ You are a Database Administrator. Write SQL queries.
RULES:
1. Use EXACT field names from the schema below (e.g., product_tks, not product)
2. Quote field names starting with digit: " 123_field "
3. Add IS NOT NULL in WHERE clause when:
- Question asks to " show me " or " display " specific columns
4. Include doc_id/docnm in non-aggregate statement
5. Output ONLY the SQL, no explanations """
user_prompt = """ Table: {}
Available fields:
2024-08-15 09:17:36 +08:00
{}
2026-01-19 19:35:14 +08:00
Question: {}
Write SQL using exact field names above. Include doc_id, docnm_kwd for data queries. Only SQL. """ . format (
table_name ,
" \n " . join ( [ f " - { k } ( { v } ) " for k , v in field_map . items ( ) ] ) ,
question
)
2024-08-15 09:17:36 +08:00
tried_times = 0
2026-03-04 19:10:06 +08:00
async def get_table ( custom_user_prompt = None ) :
2026-02-09 14:56:10 +08:00
nonlocal sys_prompt , user_prompt , question , tried_times , row_count_override
2026-03-04 19:10:06 +08:00
if row_count_override and custom_user_prompt is None :
2026-02-09 14:56:10 +08:00
sql = row_count_override
else :
2026-03-04 19:10:06 +08:00
prompt = custom_user_prompt if custom_user_prompt is not None else user_prompt
sql = await chat_mdl . async_chat ( sys_prompt , [ { " role " : " user " , " content " : prompt } ] , { " temperature " : 0.06 } )
sql = normalize_sql ( sql )
sql = add_kb_filter ( sql )
2025-09-04 16:51:13 +08:00
2024-11-14 17:13:48 +08:00
logging . debug ( f " { question } get SQL(refined): { sql } " )
2024-08-15 09:17:36 +08:00
tried_times + = 1
2026-01-19 19:35:14 +08:00
logging . debug ( f " use_sql: Executing SQL retrieval (attempt { tried_times } ) " )
tbl = settings . retriever . sql_retrieval ( sql , format = " json " )
if tbl is None :
logging . debug ( " use_sql: SQL retrieval returned None " )
return None , sql
logging . debug ( f " use_sql: SQL retrieval completed, got { len ( tbl . get ( ' rows ' , [ ] ) ) } rows " )
return tbl , sql
2024-08-15 09:17:36 +08:00
2026-03-04 19:10:06 +08:00
async def repair_table_for_missing_source_columns ( previous_sql ) :
if doc_engine in ( " infinity " , " oceanbase " ) :
json_field_names = list ( field_map . keys ( ) )
repair_prompt = """ Table name: {} ;
JSON fields available in ' chunk_data ' column (use exact names):
{}
Question: {}
Previous SQL:
{}
The previous SQL result is missing required source columns for citations.
Rewrite SQL to keep the same query intent and include doc_id and {} in the SELECT list.
For extracted JSON fields, use json_extract_string(chunk_data, ' $.field_name ' ).
Return ONLY SQL. """ . format (
table_name ,
" \n " . join ( [ f " - { field } " for field in json_field_names ] ) ,
question ,
previous_sql ,
expected_doc_name_column
)
else :
repair_prompt = """ Table name: {}
Available fields:
{}
Question: {}
Previous SQL:
{}
The previous SQL result is missing required source columns for citations.
Rewrite SQL to keep the same query intent and include doc_id and docnm_kwd in the SELECT list.
Return ONLY SQL. """ . format (
table_name ,
" \n " . join ( [ f " - { k } ( { v } ) " for k , v in field_map . items ( ) ] ) ,
question ,
previous_sql
)
return await get_table ( custom_user_prompt = repair_prompt )
2025-12-01 12:42:35 +08:00
try :
2025-12-11 17:38:17 +08:00
tbl , sql = await get_table ( )
2026-01-19 19:35:14 +08:00
logging . debug ( f " use_sql: Initial SQL execution SUCCESS. SQL: { sql } " )
logging . debug ( f " use_sql: Retrieved { len ( tbl . get ( ' rows ' , [ ] ) ) } rows, columns: { [ c [ ' name ' ] for c in tbl . get ( ' columns ' , [ ] ) ] } " )
2025-12-01 12:42:35 +08:00
except Exception as e :
2026-01-19 19:35:14 +08:00
logging . warning ( f " use_sql: Initial SQL execution FAILED with error: { e } " )
# Build retry prompt with error information
2026-01-31 15:11:54 +08:00
if doc_engine in ( " infinity " , " oceanbase " ) :
2026-01-19 19:35:14 +08:00
# Build Infinity error retry prompt
json_field_names = list ( field_map . keys ( ) )
user_prompt = """
Table name: {} ;
JSON fields available in ' chunk_data ' column (use these exact names in json_extract_string):
{}
Question: {}
Please write the SQL using json_extract_string(chunk_data, ' $.field_name ' ) with the field names from the list above. Only SQL, no explanations.
The SQL error you provided last time is as follows:
{}
Please correct the error and write SQL again using json_extract_string(chunk_data, ' $.field_name ' ) syntax with the correct field names. Only SQL, no explanations.
""" . format ( table_name , " \n " . join ( [ f " - { field } " for field in json_field_names ] ) , question , e )
else :
# Build ES/OS error retry prompt
user_prompt = """
2024-12-19 18:13:33 +08:00
Table name: {} ;
2026-01-19 19:35:14 +08:00
Table of database fields are as follows (use the field names directly in SQL):
2024-08-15 09:17:36 +08:00
{}
2025-04-15 10:20:33 +08:00
2024-12-19 18:13:33 +08:00
Question are as follows:
2024-08-15 09:17:36 +08:00
{}
2026-01-19 19:35:14 +08:00
Please write the SQL using the exact field names above, only SQL, without any other explanations or text.
2025-04-15 10:20:33 +08:00
2024-08-15 09:17:36 +08:00
2024-12-19 18:13:33 +08:00
The SQL error you provided last time is as follows:
2024-08-15 09:17:36 +08:00
{}
2026-01-19 19:35:14 +08:00
Please correct the error and write SQL again using the exact field names above, only SQL, without any other explanations or text.
""" . format ( table_name , " \n " . join ( [ f " { k } ( { v } ) " for k , v in field_map . items ( ) ] ) , question , e )
2025-12-01 12:42:35 +08:00
try :
2025-12-11 17:38:17 +08:00
tbl , sql = await get_table ( )
2026-01-19 19:35:14 +08:00
logging . debug ( f " use_sql: Retry SQL execution SUCCESS. SQL: { sql } " )
logging . debug ( f " use_sql: Retrieved { len ( tbl . get ( ' rows ' , [ ] ) ) } rows on retry " )
2025-12-01 12:42:35 +08:00
except Exception :
2026-01-19 19:35:14 +08:00
logging . error ( " use_sql: Retry SQL execution also FAILED, returning None " )
2025-12-01 12:42:35 +08:00
return
2024-08-15 09:17:36 +08:00
2025-12-01 12:42:35 +08:00
if len ( tbl [ " rows " ] ) == 0 :
2026-01-19 19:35:14 +08:00
logging . warning ( f " use_sql: No rows returned from SQL query, returning None. SQL: { sql } " )
2024-08-15 09:17:36 +08:00
return None
2026-03-04 19:10:06 +08:00
if not is_aggregate_sql ( sql ) and not has_source_columns ( tbl . get ( " columns " , [ ] ) ) :
logging . warning ( f " use_sql: Non-aggregate SQL missing required source columns; retrying once. SQL: { sql } " )
try :
repaired_tbl , repaired_sql = await repair_table_for_missing_source_columns ( sql )
if (
repaired_tbl
and len ( repaired_tbl . get ( " rows " , [ ] ) ) > 0
and has_source_columns ( repaired_tbl . get ( " columns " , [ ] ) )
) :
tbl , sql = repaired_tbl , repaired_sql
logging . info ( f " use_sql: Source-column SQL repair succeeded. SQL: { sql } " )
else :
logging . warning ( f " use_sql: Source-column SQL repair did not provide required columns. Repaired SQL: { repaired_sql } " )
except Exception as e :
logging . warning ( f " use_sql: Source-column SQL repair failed, returning best-effort answer. Error: { e } " )
2026-01-19 19:35:14 +08:00
logging . debug ( f " use_sql: Proceeding with { len ( tbl [ ' rows ' ] ) } rows to build answer " )
docid_idx = set ( [ ii for ii , c in enumerate ( tbl [ " columns " ] ) if c [ " name " ] . lower ( ) == " doc_id " ] )
doc_name_idx = set ( [ ii for ii , c in enumerate ( tbl [ " columns " ] ) if c [ " name " ] . lower ( ) in [ " docnm_kwd " , " docnm " ] ] )
logging . debug ( f " use_sql: All columns: { [ ( i , c [ ' name ' ] ) for i , c in enumerate ( tbl [ ' columns ' ] ) ] } " )
logging . debug ( f " use_sql: docid_idx= { docid_idx } , doc_name_idx= { doc_name_idx } " )
2025-03-24 13:18:47 +08:00
column_idx = [ ii for ii in range ( len ( tbl [ " columns " ] ) ) if ii not in ( docid_idx | doc_name_idx ) ]
2024-08-15 09:17:36 +08:00
2026-01-19 19:35:14 +08:00
logging . debug ( f " use_sql: column_idx= { column_idx } " )
logging . debug ( f " use_sql: field_map= { field_map } " )
# Helper function to map column names to display names
def map_column_name ( col_name ) :
if col_name . lower ( ) == " count(star) " :
return " COUNT(*) "
# First, try to extract AS alias from any expression (aggregate functions, json_extract_string, etc.)
# Pattern: anything AS alias_name
as_match = re . search ( r ' \ s+AS \ s+([^ \ s,)]+) ' , col_name , re . IGNORECASE )
if as_match :
alias = as_match . group ( 1 ) . strip ( ' " \' ' )
# Use the alias for display name lookup
if alias in field_map :
display = field_map [ alias ]
return re . sub ( r " (/.*|( [^( ) ]+) ) " , " " , display )
# If alias not in field_map, try to match case-insensitively
for field_key , display_value in field_map . items ( ) :
if field_key . lower ( ) == alias . lower ( ) :
return re . sub ( r " (/.*|( [^( ) ]+) ) " , " " , display_value )
# Return alias as-is if no mapping found
return alias
# Try direct mapping first (for simple column names)
if col_name in field_map :
display = field_map [ col_name ]
# Clean up any suffix patterns
return re . sub ( r " (/.*|( [^( ) ]+) ) " , " " , display )
# Try case-insensitive match for simple column names
col_lower = col_name . lower ( )
for field_key , display_value in field_map . items ( ) :
if field_key . lower ( ) == col_lower :
return re . sub ( r " (/.*|( [^( ) ]+) ) " , " " , display_value )
# For aggregate expressions or complex expressions without AS alias,
# try to replace field names with display names
result = col_name
for field_name , display_name in field_map . items ( ) :
# Replace field_name with display_name in the expression
result = result . replace ( field_name , display_name )
# Clean up any suffix patterns
result = re . sub ( r " (/.*|( [^( ) ]+) ) " , " " , result )
return result
2024-12-19 18:13:33 +08:00
# compose Markdown table
2025-03-24 13:18:47 +08:00
columns = (
2025-09-23 10:19:25 +08:00
" | " + " | " . join (
2026-01-19 19:35:14 +08:00
[ map_column_name ( tbl [ " columns " ] [ i ] [ " name " ] ) for i in column_idx ] ) + (
" |Source| " if docid_idx and doc_name_idx else " | " )
2025-03-24 13:18:47 +08:00
)
2024-08-15 09:17:36 +08:00
2025-03-24 13:18:47 +08:00
line = " | " + " | " . join ( [ " ------ " for _ in range ( len ( column_idx ) ) ] ) + ( " |------| " if docid_idx and docid_idx else " " )
2024-08-15 09:17:36 +08:00
2026-01-19 19:35:14 +08:00
# Build rows ensuring column names match values - create a dict for each row
# keyed by column name to handle any SQL column order
rows = [ ]
for row_idx , r in enumerate ( tbl [ " rows " ] ) :
row_dict = { tbl [ " columns " ] [ i ] [ " name " ] : r [ i ] for i in range ( len ( tbl [ " columns " ] ) ) if i < len ( r ) }
if row_idx == 0 :
logging . debug ( f " use_sql: First row data: { row_dict } " )
row_values = [ ]
for col_idx in column_idx :
col_name = tbl [ " columns " ] [ col_idx ] [ " name " ]
value = row_dict . get ( col_name , " " )
row_values . append ( remove_redundant_spaces ( str ( value ) ) . replace ( " None " , " " ) )
# Add Source column with citation marker if Source column exists
if docid_idx and doc_name_idx :
row_values . append ( f " ## { row_idx } $$ " )
row_str = " | " + " | " . join ( row_values ) + " | "
if re . sub ( r " [ |]+ " , " " , row_str ) :
rows . append ( row_str )
2024-08-15 09:17:36 +08:00
if quota :
2026-01-19 19:35:14 +08:00
rows = " \n " . join ( rows )
2024-08-15 09:17:36 +08:00
else :
2026-01-19 19:35:14 +08:00
rows = " \n " . join ( rows )
2024-08-15 09:17:36 +08:00
rows = re . sub ( r " T[0-9] {2} :[0-9] {2} :[0-9] {2} ( \ .[0-9]+Z)? \ | " , " | " , rows )
2024-12-19 18:13:33 +08:00
if not docid_idx or not doc_name_idx :
2026-01-19 19:35:14 +08:00
logging . warning ( f " use_sql: SQL missing required doc_id or docnm_kwd field. docid_idx= { docid_idx } , doc_name_idx= { doc_name_idx } . SQL: { sql } " )
# For aggregate queries (COUNT, SUM, AVG, MAX, MIN, DISTINCT), fetch doc_id, docnm_kwd separately
# to provide source chunks, but keep the original table format answer
2026-03-04 19:10:06 +08:00
if is_aggregate_sql ( sql ) :
2026-01-19 19:35:14 +08:00
# Keep original table format as answer
answer = " \n " . join ( [ columns , line , rows ] )
# Now fetch doc_id, docnm_kwd to provide source chunks
# Extract WHERE clause from the original SQL
where_match = re . search ( r " \ bwhere \ b(.+?)(?: \ bgroup by \ b| \ border by \ b| \ blimit \ b|$) " , sql , re . IGNORECASE )
if where_match :
where_clause = where_match . group ( 1 ) . strip ( )
# Build a query to get doc_id and docnm_kwd with the same WHERE clause
chunks_sql = f " select doc_id, docnm_kwd from { table_name } where { where_clause } "
# Add LIMIT to avoid fetching too many chunks
if " limit " not in chunks_sql . lower ( ) :
chunks_sql + = " limit 20 "
logging . debug ( f " use_sql: Fetching chunks with SQL: { chunks_sql } " )
try :
chunks_tbl = settings . retriever . sql_retrieval ( chunks_sql , format = " json " )
if chunks_tbl . get ( " rows " ) and len ( chunks_tbl [ " rows " ] ) > 0 :
# Build chunks reference - use case-insensitive matching
chunks_did_idx = next ( ( i for i , c in enumerate ( chunks_tbl [ " columns " ] ) if c [ " name " ] . lower ( ) == " doc_id " ) , None )
chunks_dn_idx = next ( ( i for i , c in enumerate ( chunks_tbl [ " columns " ] ) if c [ " name " ] . lower ( ) in [ " docnm_kwd " , " docnm " ] ) , None )
if chunks_did_idx is not None and chunks_dn_idx is not None :
chunks = [ { " doc_id " : r [ chunks_did_idx ] , " docnm_kwd " : r [ chunks_dn_idx ] } for r in chunks_tbl [ " rows " ] ]
# Build doc_aggs
doc_aggs = { }
for r in chunks_tbl [ " rows " ] :
doc_id = r [ chunks_did_idx ]
doc_name = r [ chunks_dn_idx ]
if doc_id not in doc_aggs :
doc_aggs [ doc_id ] = { " doc_name " : doc_name , " count " : 0 }
doc_aggs [ doc_id ] [ " count " ] + = 1
doc_aggs_list = [ { " doc_id " : did , " doc_name " : d [ " doc_name " ] , " count " : d [ " count " ] } for did , d in doc_aggs . items ( ) ]
logging . debug ( f " use_sql: Returning aggregate answer with { len ( chunks ) } chunks from { len ( doc_aggs ) } documents " )
return { " answer " : answer , " reference " : { " chunks " : chunks , " doc_aggs " : doc_aggs_list } , " prompt " : sys_prompt }
except Exception as e :
logging . warning ( f " use_sql: Failed to fetch chunks: { e } " )
# Fallback: return answer without chunks
return { " answer " : answer , " reference " : { " chunks " : [ ] , " doc_aggs " : [ ] } , " prompt " : sys_prompt }
# Fallback to table format for other cases
2025-03-24 13:18:47 +08:00
return { " answer " : " \n " . join ( [ columns , line , rows ] ) , " reference " : { " chunks " : [ ] , " doc_aggs " : [ ] } , " prompt " : sys_prompt }
2024-08-15 09:17:36 +08:00
docid_idx = list ( docid_idx ) [ 0 ]
2024-12-19 18:13:33 +08:00
doc_name_idx = list ( doc_name_idx ) [ 0 ]
2024-08-15 09:17:36 +08:00
doc_aggs = { }
for r in tbl [ " rows " ] :
if r [ docid_idx ] not in doc_aggs :
2024-12-19 18:13:33 +08:00
doc_aggs [ r [ docid_idx ] ] = { " doc_name " : r [ doc_name_idx ] , " count " : 0 }
2024-08-15 09:17:36 +08:00
doc_aggs [ r [ docid_idx ] ] [ " count " ] + = 1
2026-01-19 19:35:14 +08:00
result = {
2024-12-19 18:13:33 +08:00
" answer " : " \n " . join ( [ columns , line , rows ] ) ,
2025-03-24 13:18:47 +08:00
" reference " : {
" chunks " : [ { " doc_id " : r [ docid_idx ] , " docnm_kwd " : r [ doc_name_idx ] } for r in tbl [ " rows " ] ] ,
" doc_aggs " : [ { " doc_id " : did , " doc_name " : d [ " doc_name " ] , " count " : d [ " count " ] } for did , d in doc_aggs . items ( ) ] ,
} ,
" prompt " : sys_prompt ,
2024-08-15 09:17:36 +08:00
}
2026-01-19 19:35:14 +08:00
logging . debug ( f " use_sql: Returning answer with { len ( result [ ' reference ' ] [ ' chunks ' ] ) } chunks from { len ( doc_aggs ) } documents " )
return result
2024-08-15 09:17:36 +08:00
2025-12-03 12:03:59 +08:00
def clean_tts_text ( text : str ) - > str :
if not text :
return " "
text = text . encode ( " utf-8 " , " ignore " ) . decode ( " utf-8 " , " ignore " )
text = re . sub ( r " [ \ x00- \ x08 \ x0B- \ x0C \ x0E- \ x1F \ x7F] " , " " , text )
emoji_pattern = re . compile (
" [ \U0001F600 - \U0001F64F "
" \U0001F300 - \U0001F5FF "
" \U0001F680 - \U0001F6FF "
" \U0001F1E0 - \U0001F1FF "
" \U00002700 - \U000027BF "
" \U0001F900 - \U0001F9FF "
" \U0001FA70 - \U0001FAFF "
" \U0001FAD0 - \U0001FAFF ]+ " ,
flags = re . UNICODE
)
text = emoji_pattern . sub ( " " , text )
text = re . sub ( r " \ s+ " , " " , text ) . strip ( )
MAX_LEN = 500
if len ( text ) > MAX_LEN :
text = text [ : MAX_LEN ]
return text
2024-08-15 09:17:36 +08:00
2024-09-03 19:49:14 +08:00
def tts ( tts_mdl , text ) :
2024-12-08 14:21:12 +08:00
if not tts_mdl or not text :
2025-11-16 19:29:20 +08:00
return None
2025-12-03 12:03:59 +08:00
text = clean_tts_text ( text )
if not text :
return None
2024-09-03 19:49:14 +08:00
bin = b " "
2025-12-03 12:03:59 +08:00
try :
for chunk in tts_mdl . tts ( text ) :
bin + = chunk
except Exception as e :
logging . error ( f " TTS failed: { e } , text= { text !r} " )
return None
2024-09-09 12:08:50 +08:00
return binascii . hexlify ( bin ) . decode ( " utf-8 " )
2026-01-08 13:34:16 +08:00
class _ThinkStreamState :
def __init__ ( self ) - > None :
self . full_text = " "
self . last_idx = 0
self . endswith_think = False
self . last_full = " "
self . last_model_full = " "
self . in_think = False
self . buffer = " "
def _next_think_delta ( state : _ThinkStreamState ) - > str :
full_text = state . full_text
if full_text == state . last_full :
return " "
state . last_full = full_text
delta_ans = full_text [ state . last_idx : ]
if delta_ans . find ( " <think> " ) == 0 :
state . last_idx + = len ( " <think> " )
return " <think> "
if delta_ans . find ( " <think> " ) > 0 :
delta_text = full_text [ state . last_idx : state . last_idx + delta_ans . find ( " <think> " ) ]
state . last_idx + = delta_ans . find ( " <think> " )
return delta_text
if delta_ans . endswith ( " </think> " ) :
state . endswith_think = True
elif state . endswith_think :
state . endswith_think = False
return " </think> "
state . last_idx = len ( full_text )
if full_text . endswith ( " </think> " ) :
state . last_idx - = len ( " </think> " )
return re . sub ( r " (<think>|</think>) " , " " , delta_ans )
async def _stream_with_think_delta ( stream_iter , min_tokens : int = 16 ) :
state = _ThinkStreamState ( )
async for chunk in stream_iter :
if not chunk :
continue
if chunk . startswith ( state . last_model_full ) :
new_part = chunk [ len ( state . last_model_full ) : ]
state . last_model_full = chunk
else :
new_part = chunk
state . last_model_full + = chunk
if not new_part :
continue
state . full_text + = new_part
delta = _next_think_delta ( state )
if not delta :
continue
if delta in ( " <think> " , " </think> " ) :
if delta == " <think> " and state . in_think :
continue
if delta == " </think> " and not state . in_think :
continue
if state . buffer :
yield ( " text " , state . buffer , state )
state . buffer = " "
state . in_think = delta == " <think> "
yield ( " marker " , delta , state )
continue
state . buffer + = delta
if num_tokens_from_string ( state . buffer ) < min_tokens :
continue
yield ( " text " , state . buffer , state )
state . buffer = " "
if state . buffer :
yield ( " text " , state . buffer , state )
state . buffer = " "
if state . endswith_think :
yield ( " marker " , " </think> " , state )
2025-12-08 09:43:03 +08:00
async def async_ask ( question , kb_ids , tenant_id , chat_llm_name = None , search_config = { } ) :
2025-08-19 17:25:44 +08:00
doc_ids = search_config . get ( " doc_ids " , [ ] )
2025-08-19 09:33:33 +08:00
rerank_mdl = None
2025-08-19 17:25:44 +08:00
kb_ids = search_config . get ( " kb_ids " , kb_ids )
chat_llm_name = search_config . get ( " chat_id " , chat_llm_name )
rerank_id = search_config . get ( " rerank_id " , " " )
meta_data_filter = search_config . get ( " meta_data_filter " )
2025-08-19 09:33:33 +08:00
2024-09-09 12:08:50 +08:00
kbs = KnowledgebaseService . get_by_ids ( kb_ids )
2024-12-19 18:13:33 +08:00
embedding_list = list ( set ( [ kb . embd_id for kb in kbs ] ) )
2024-09-09 12:08:50 +08:00
2024-12-19 18:13:33 +08:00
is_knowledge_graph = all ( [ kb . parser_id == ParserType . KG for kb in kbs ] )
2025-11-06 09:36:38 +08:00
retriever = settings . retriever if not is_knowledge_graph else settings . kg_retriever
2026-03-05 17:27:17 +08:00
embd_model_config = get_model_config_by_type_and_name ( tenant_id , LLMType . EMBEDDING , embedding_list [ 0 ] )
embd_mdl = LLMBundle ( tenant_id , embd_model_config )
chat_model_config = get_model_config_by_type_and_name ( tenant_id , LLMType . CHAT , chat_llm_name )
chat_mdl = LLMBundle ( tenant_id , chat_model_config )
2025-08-19 09:33:33 +08:00
if rerank_id :
2026-03-05 17:27:17 +08:00
rerank_model_config = get_model_config_by_type_and_name ( tenant_id , LLMType . RERANK , rerank_id )
rerank_mdl = LLMBundle ( tenant_id , rerank_model_config )
2024-09-09 12:08:50 +08:00
max_tokens = chat_mdl . max_length
2024-12-10 17:03:24 +08:00
tenant_ids = list ( set ( [ kb . tenant_id for kb in kbs ] ) )
2025-08-19 10:27:24 +08:00
2025-08-19 17:25:44 +08:00
if meta_data_filter :
2026-01-28 13:29:34 +08:00
metas = DocMetadataService . get_flatted_meta_by_kbs ( kb_ids )
2025-12-12 17:12:38 +08:00
doc_ids = await apply_meta_data_filter ( meta_data_filter , metas , question , chat_mdl , doc_ids )
2025-08-19 17:25:44 +08:00
2026-01-15 12:28:49 +08:00
kbinfos = await retriever . retrieval (
2025-09-23 10:19:25 +08:00
question = question ,
2025-08-19 09:33:33 +08:00
embd_mdl = embd_mdl ,
tenant_ids = tenant_ids ,
kb_ids = kb_ids ,
page = 1 ,
page_size = 12 ,
2025-08-19 17:25:44 +08:00
similarity_threshold = search_config . get ( " similarity_threshold " , 0.1 ) ,
vector_similarity_weight = search_config . get ( " vector_similarity_weight " , 0.3 ) ,
top = search_config . get ( " top_k " , 1024 ) ,
2025-08-19 09:33:33 +08:00
doc_ids = doc_ids ,
2025-12-23 15:57:27 +08:00
aggs = True ,
2025-08-19 09:33:33 +08:00
rerank_mdl = rerank_mdl ,
rank_feature = label_question ( question , kbs )
)
2024-12-10 17:03:24 +08:00
knowledges = kb_prompt ( kbinfos , max_tokens )
2025-08-19 10:27:24 +08:00
sys_prompt = PROMPT_JINJA_ENV . from_string ( ASK_SUMMARY ) . render ( knowledge = " \n " . join ( knowledges ) )
2024-09-09 12:08:50 +08:00
msg = [ { " role " : " user " , " content " : question } ]
def decorate_answer ( answer ) :
2025-08-19 10:27:24 +08:00
nonlocal knowledges , kbinfos , sys_prompt
2025-09-23 10:19:25 +08:00
answer , idx = retriever . insert_citations ( answer , [ ck [ " content_ltks " ] for ck in kbinfos [ " chunks " ] ] , [ ck [ " vector " ] for ck in kbinfos [ " chunks " ] ] ,
embd_mdl , tkweight = 0.7 , vtweight = 0.3 )
2024-09-09 12:08:50 +08:00
idx = set ( [ kbinfos [ " chunks " ] [ int ( i ) ] [ " doc_id " ] for i in idx ] )
2025-03-24 13:18:47 +08:00
recall_docs = [ d for d in kbinfos [ " doc_aggs " ] if d [ " doc_id " ] in idx ]
2024-12-08 14:21:12 +08:00
if not recall_docs :
recall_docs = kbinfos [ " doc_aggs " ]
2024-09-09 12:08:50 +08:00
kbinfos [ " doc_aggs " ] = recall_docs
refs = deepcopy ( kbinfos )
for c in refs [ " chunks " ] :
if c . get ( " vector " ) :
del c [ " vector " ]
if answer . lower ( ) . find ( " invalid key " ) > = 0 or answer . lower ( ) . find ( " invalid api " ) > = 0 :
2024-12-10 17:03:24 +08:00
answer + = " Please set LLM API-Key in ' User Setting -> Model Providers -> API-Key ' "
2025-03-05 17:25:47 +08:00
refs [ " chunks " ] = chunks_format ( refs )
return { " answer " : answer , " reference " : refs }
2024-09-09 12:08:50 +08:00
2026-01-08 13:34:16 +08:00
stream_iter = chat_mdl . async_chat_streamly_delta ( sys_prompt , msg , { " temperature " : 0.1 } )
last_state = None
async for kind , value , state in _stream_with_think_delta ( stream_iter ) :
last_state = state
if kind == " marker " :
flags = { " start_to_think " : True } if value == " <think> " else { " end_to_think " : True }
yield { " answer " : " " , " reference " : { } , " final " : False , * * flags }
continue
yield { " answer " : value , " reference " : { } , " final " : False }
full_answer = last_state . full_text if last_state else " "
final = decorate_answer ( full_answer )
final [ " final " ] = True
final [ " answer " ] = " "
yield final
2025-08-19 17:25:44 +08:00
2025-12-11 09:47:44 +08:00
async def gen_mindmap ( question , kb_ids , tenant_id , search_config = { } ) :
2025-08-19 17:25:44 +08:00
meta_data_filter = search_config . get ( " meta_data_filter " , { } )
doc_ids = search_config . get ( " doc_ids " , [ ] )
rerank_id = search_config . get ( " rerank_id " , " " )
rerank_mdl = None
kbs = KnowledgebaseService . get_by_ids ( kb_ids )
2025-08-19 18:57:35 +08:00
if not kbs :
return { " error " : " No KB selected " }
2026-03-05 17:27:17 +08:00
tenant_embedding_list = list ( set ( [ kb . tenant_embd_id for kb in kbs ] ) )
2025-08-19 17:25:44 +08:00
tenant_ids = list ( set ( [ kb . tenant_id for kb in kbs ] ) )
2026-03-05 17:27:17 +08:00
if tenant_embedding_list [ 0 ] :
embd_model_config = get_model_config_by_id ( tenant_embedding_list [ 0 ] )
else :
embd_model_config = get_model_config_by_type_and_name ( tenant_id , LLMType . EMBEDDING , kbs [ 0 ] . embd_id )
embd_mdl = LLMBundle ( tenant_id , embd_model_config )
chat_id = search_config . get ( " chat_id " , " " )
if chat_id :
chat_model_config = get_model_config_by_type_and_name ( tenant_id , LLMType . CHAT , chat_id )
else :
chat_model_config = get_tenant_default_model_by_type ( tenant_id , LLMType . CHAT )
chat_mdl = LLMBundle ( tenant_id , chat_model_config )
2025-08-19 17:25:44 +08:00
if rerank_id :
2026-03-05 17:27:17 +08:00
rerank_model_config = get_model_config_by_type_and_name ( tenant_id , LLMType . RERANK , rerank_id )
rerank_mdl = LLMBundle ( tenant_id , rerank_model_config )
2025-08-19 17:25:44 +08:00
if meta_data_filter :
2026-01-28 13:29:34 +08:00
metas = DocMetadataService . get_flatted_meta_by_kbs ( kb_ids )
2025-12-12 17:12:38 +08:00
doc_ids = await apply_meta_data_filter ( meta_data_filter , metas , question , chat_mdl , doc_ids )
2025-08-19 17:25:44 +08:00
2026-01-15 12:28:49 +08:00
ranks = await settings . retriever . retrieval (
2025-08-19 17:25:44 +08:00
question = question ,
embd_mdl = embd_mdl ,
tenant_ids = tenant_ids ,
kb_ids = kb_ids ,
page = 1 ,
page_size = 12 ,
similarity_threshold = search_config . get ( " similarity_threshold " , 0.2 ) ,
vector_similarity_weight = search_config . get ( " vector_similarity_weight " , 0.3 ) ,
top = search_config . get ( " top_k " , 1024 ) ,
doc_ids = doc_ids ,
aggs = False ,
rerank_mdl = rerank_mdl ,
rank_feature = label_question ( question , kbs ) ,
)
mindmap = MindMapExtractor ( chat_mdl )
2025-12-11 09:47:44 +08:00
mind_map = await mindmap ( [ c [ " content_with_weight " ] for c in ranks [ " chunks " ] ] )
2025-09-04 16:51:13 +08:00
return mind_map . output