feat: QYLAW 法律法规知识库
- 语义检索(FAISS + embedding)+ 精确查找(法规名+条号) - RAG 问答(SSE 流式,支持 thinking 折叠显示) - 法规浏览(原文阅读) - 历史记录(检索+对话持久化到 SQLite) - 设置页(系统提示词/模板/LLM 参数可配置) - 检索质量评估脚本 Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
@@ -0,0 +1,140 @@
|
||||
"""LLM 客户端 — 调用 114:7000 Qwen3.5-35B 服务(流式)"""
|
||||
import json
|
||||
import logging
|
||||
from typing import AsyncGenerator
|
||||
|
||||
import httpx
|
||||
|
||||
from app.config import LLM_URL, LLM_MODEL
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_client: httpx.AsyncClient | None = None
|
||||
|
||||
|
||||
def _get_client() -> httpx.AsyncClient:
|
||||
global _client
|
||||
if _client is None or _client.is_closed:
|
||||
_client = httpx.AsyncClient(timeout=120)
|
||||
return _client
|
||||
|
||||
|
||||
DEFAULT_SYSTEM_PROMPT = "你是法律助手,根据提供的法规条文回答问题,必须引用法规名和条号。"
|
||||
|
||||
|
||||
async def stream_chat(
|
||||
prompt: str,
|
||||
system: str | None = None,
|
||||
temperature: float = 0.3,
|
||||
max_tokens: int = 2048,
|
||||
thinking_enabled: bool = True,
|
||||
thinking_budget: int = 2048,
|
||||
) -> AsyncGenerator[tuple[str, str], None]:
|
||||
"""流式对话,区分 thinking 和 answer
|
||||
|
||||
模型输出格式:thinking 内容以 "Here's a thinking process:" 开头,
|
||||
以 </thinking> 结尾,之后是正式回答。
|
||||
|
||||
Args:
|
||||
prompt: 用户 prompt(已含 context)
|
||||
system: system prompt
|
||||
temperature: 温度
|
||||
max_tokens: 最大生成 token(含 thinking)
|
||||
thinking_enabled: 是否启用 thinking
|
||||
thinking_budget: thinking 预算(thinking 最大 token 数)
|
||||
|
||||
Yields:
|
||||
(type, content) 元组,type 为 "thinking" 或 "answer"
|
||||
|
||||
Raises:
|
||||
RuntimeError: LLM 服务不可用
|
||||
"""
|
||||
client = _get_client()
|
||||
sys_content = system if system else DEFAULT_SYSTEM_PROMPT
|
||||
|
||||
# thinking 关闭时,通过 system prompt 补充指示
|
||||
if not thinking_enabled:
|
||||
sys_content += "\n\n注意:不要输出思考过程,直接给出答案。"
|
||||
|
||||
# payload:使用 chat_template_kwargs.enable_thinking 控制 thinking 开关
|
||||
# 这是 Qwen3.5 vLLM 的原生参数,比 system prompt 更可靠
|
||||
payload = {
|
||||
"model": LLM_MODEL,
|
||||
"messages": [
|
||||
{"role": "system", "content": sys_content},
|
||||
{"role": "user", "content": prompt},
|
||||
],
|
||||
"temperature": temperature,
|
||||
"max_tokens": max_tokens,
|
||||
"stream": True,
|
||||
"chat_template_kwargs": {"enable_thinking": thinking_enabled},
|
||||
}
|
||||
|
||||
# 状态机:跟踪当前是否在 thinking 区域
|
||||
in_thinking = thinking_enabled # thinking 开启时,输出通常以 thinking 开头
|
||||
# 小缓冲:仅用于检测跨块的 thinking 结束标记
|
||||
tail_buffer = ""
|
||||
# Qwen3.5 的 thinking 结束标记: </think>
|
||||
THINKING_END_TAG = "</think>"
|
||||
|
||||
last_err = None
|
||||
for attempt in range(2): # 重试 1 次
|
||||
try:
|
||||
async with client.stream(
|
||||
"POST",
|
||||
f"{LLM_URL}/chat/completions",
|
||||
json=payload,
|
||||
timeout=120,
|
||||
) as resp:
|
||||
resp.raise_for_status()
|
||||
async for line in resp.aiter_lines():
|
||||
if not line or not line.startswith("data: "):
|
||||
continue
|
||||
data_str = line[6:]
|
||||
if data_str.strip() == "[DONE]":
|
||||
return
|
||||
try:
|
||||
chunk = json.loads(data_str)
|
||||
delta = chunk["choices"][0].get("delta", {})
|
||||
content = delta.get("content", "")
|
||||
if not content:
|
||||
continue
|
||||
|
||||
if not thinking_enabled:
|
||||
# thinking 关闭,全部作为 answer
|
||||
yield ("answer", content)
|
||||
else:
|
||||
# thinking 开启,需要解析 THINKING_END_TAG 标记
|
||||
if in_thinking:
|
||||
if THINKING_END_TAG in content:
|
||||
# 分割:前面是 thinking,后面是 answer
|
||||
parts = content.split(THINKING_END_TAG, 1)
|
||||
thinking_part = parts[0]
|
||||
answer_part = parts[1] if len(parts) > 1 else ""
|
||||
if thinking_part:
|
||||
yield ("thinking", thinking_part)
|
||||
if answer_part:
|
||||
yield ("answer", answer_part.lstrip())
|
||||
in_thinking = False
|
||||
tail_buffer = ""
|
||||
else:
|
||||
# 检查是否是 THINKING_END_TAG 的前缀(跨块情况)
|
||||
tail_buffer = (tail_buffer + content)[-12:]
|
||||
# 检查 tail_buffer 是否是 THINKING_END_TAG 的前缀
|
||||
is_prefix = any(
|
||||
tail_buffer.endswith(THINKING_END_TAG[:k])
|
||||
for k in range(1, len(THINKING_END_TAG))
|
||||
)
|
||||
if is_prefix:
|
||||
continue
|
||||
yield ("thinking", content)
|
||||
else:
|
||||
yield ("answer", content)
|
||||
except (json.JSONDecodeError, KeyError, IndexError):
|
||||
continue
|
||||
return
|
||||
except Exception as e:
|
||||
last_err = e
|
||||
logger.warning(f"LLM 流式调用第 {attempt+1} 次失败: {e}")
|
||||
|
||||
raise RuntimeError(f"LLM 服务调用失败: {last_err}")
|
||||
Reference in New Issue
Block a user