From 4753a53487f80b3db3654a82fe72b1742c670766 Mon Sep 17 00:00:00 2001 From: lacerate551 <128470311+lacerate551@users.noreply.github.com> Date: Sun, 21 Jun 2026 22:41:30 +0800 Subject: [PATCH] =?UTF-8?q?fix(rag):=20=E5=9B=BE=E7=89=87=E6=8F=8F?= =?UTF-8?q?=E8=BF=B0=E6=B3=A8=E5=85=A5=E5=AE=89=E5=85=A8=E7=BD=91=E4=B8=8E?= =?UTF-8?q?=20chart=5Fcontexts=20=E5=AE=BD=E6=9D=BE=E9=98=88=E5=80=BC?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 问题:图片/图表切片的 CrossEncoder 评分系统性偏低(0.002-0.08 vs 文本 0.3-0.9), 导致 chart_contexts 被 min_score 过滤,图片描述无法进入 LLM context。 典型案例:雷达图查询(ci=73)rerank score=0.039 < min_score=0.05, 虽然 select_images 正确选中了雷达图,但描述未出现在 LLM context 中。 修复: 1. chart_contexts 使用宽松阈值 min_score*0.5(与 table 保护逻辑一致) - 图片描述的 CE 分数天然偏低,不应与文本使用相同阈值 - 实际内容相关性由 select_images 独立评分保证 2. P0 安全网:在现有图片注入代码后检查 selected_images 完整性 - 若【相关图片信息】未生成,补注入所有选中图片描述 - 若已有但遗漏部分图片,追加遗漏的描述 3. images_selected 调试事件增加 chunk_id/actual_score/id/desc_preview 4. P0 完整性日志:DEV 模式下检查图片描述是否完整注入 context --- api/chat_routes.py | 53 ++++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 51 insertions(+), 2 deletions(-) diff --git a/api/chat_routes.py b/api/chat_routes.py index 3ee13c2..2a9e198 100644 --- a/api/chat_routes.py +++ b/api/chat_routes.py @@ -352,7 +352,11 @@ def _order_text_contexts_for_prompt(contexts: List[Dict], query: str, max_chunks and (c.get('meta', {}).get('source', ''), c.get('meta', {}).get('section', '') or c.get('meta', {}).get('section_path', '')) in passing_sections ) ] - chart_contexts = [c for c in chart_contexts if c.get('score', 0) >= min_score] + # 图片/图表切片:CrossEncoder 对图片描述评分系统性偏低(0.002-0.08 vs 文本 0.3-0.9), + # 使用更宽松的阈值(min_score * 0.5),避免被误过滤。 + # 实际内容相关性由 select_images 独立评分保证,此处仅做基础过滤。 + _chart_floor = min_score * 0.5 + chart_contexts = [c for c in chart_contexts if c.get('score', 0) >= _chart_floor] # 合并:文本切片优先,图表切片补充 # 限制图表切片数量,避免过多 @@ -2752,7 +2756,7 @@ def rag(): # 调试事件:图片选择详情 if IS_DEV: - yield f"data: {json.dumps({'type': 'images_selected', 'data': {'total_scored': len([c for c in contexts if c.get('meta',{}).get('chunk_type') in ('image','chart','table')]), 'selected_count': len(selected_images), 'images': [{'source': img.get('source',''), 'page': img.get('page',0), 'score': round(img.get('_image_boost', 1.0), 2)} for img in selected_images]}}, ensure_ascii=False)}\n\n" + yield f"data: {json.dumps({'type': 'images_selected', 'data': {'total_scored': len([c for c in contexts if c.get('meta',{}).get('chunk_type') in ('image','chart','table')]), 'selected_count': len(selected_images), 'images': [{'source': img.get('source',''), 'page': img.get('page',0), 'score': round(img.get('_image_boost', 1.0), 2), 'chunk_id': img.get('chunk_id',''), 'actual_score': round(img.get('score',0), 2), 'id': img.get('id',''), 'desc_preview': (img.get('full_description','') or img.get('description',''))[:60]} for img in selected_images]}}, ensure_ascii=False)}\n\n" # 4. 构建 prompt(Phase 6:LLM 图片感知) # Bug 1 修复:文本切片用于 top 5 名额竞争,图片描述不参与竞争 @@ -2854,8 +2858,53 @@ def rag(): # 添加指令让 LLM 介绍图片 context_text += "\n\n【回答要求】回答时请简要介绍每张图片的内容和用途。" + # P0 安全网:确保 select_images 选中的图片描述一定出现在 LLM context 中 + # 场景:chart_contexts 被 min_score 过滤、_build_context_with_budget 截断、 + # 或其他过滤逻辑意外排除时,selected_images 仍能提供描述 + if selected_images and '【相关图片信息】' not in context_text: + # 现有注入代码未执行(selected_images 为空时不会进入),补注入 + image_descriptions = [] + for i, img in enumerate(selected_images, 1): + full_desc = img.get('full_description', '') or img.get('description', '') + if full_desc and len(full_desc.strip()) >= 5: + img_source = img.get('source', '') + img_page = img.get('page', '') + source_info = f"(来源:{img_source} 第{img_page}页)" if img_source and img_page else "" + image_descriptions.append(f"【图片{i}】{full_desc}{source_info}") + if image_descriptions: + context_text += "\n\n【相关图片信息】\n" + "\n\n".join(image_descriptions) + context_text += "\n\n【回答要求】回答时请简要介绍每张图片的内容和用途。" + elif selected_images: + # 已有【相关图片信息】,检查是否有遗漏(被过滤但 select_images 选中的图片) + _existing_desc_ids = set() + for img in selected_images: + img_id = img.get('id', '') + if img_id and img_id in context_text: + _existing_desc_ids.add(img_id) + _missing = [img for img in selected_images if img.get('id', '') not in _existing_desc_ids] + if _missing: + _missing_descs = [] + _start_idx = len(selected_images) - len(_missing) + 1 + for i, img in enumerate(_missing, _start_idx): + full_desc = img.get('full_description', '') or img.get('description', '') + if full_desc and len(full_desc.strip()) >= 5: + img_source = img.get('source', '') + img_page = img.get('page', '') + source_info = f"(来源:{img_source} 第{img_page}页)" if img_source and img_page else "" + _missing_descs.append(f"【图片{i}】{full_desc}{source_info}") + if _missing_descs: + context_text += "\n\n" + "\n\n".join(_missing_descs) + enhanced_context = context_text + # P0 日志:图片描述注入完整性检查 + if selected_images and IS_DEV: + _img_ids_in_ctx = {img.get('id', '') for img in selected_images if img.get('id', '') and img.get('id', '') in enhanced_context} + _img_ids_all = {img.get('id', '') for img in selected_images if img.get('id', '')} + _missing_ids = _img_ids_all - _img_ids_in_ctx + if _missing_ids: + logger.warning(f"[P0] 图片描述未完整注入 context: missing={_missing_ids}, total={len(selected_images)}") + # 对比类查询:添加结构化对比指令 if intent and intent.intent == "comparison" and intent.sub_queries: comparison_instruction = (