mirror of
https://github.com/zhenxun-org/zhenxun_bot.git
synced 2026-09-29 00:32:06 +08:00
* ✨ feat!(llm): 重构并升级大语言模型服务为全新 AI 智能体框架 - 【重构】将原 services/llm 重构并迁移至全新的 services/ai 架构,提供向下兼容垫片 - 【新增】引入 Agent、Team、Workflow 三大智能体与工作流编排范式 - 【新增】引入基于 RAG 的长期向量记忆与中期槽位记忆系统 - 【新增】引入基于 Docker 的安全代码执行沙箱环境 - 【新增】支持 MCP 协议,允许动态管理和调用 MCP 服务 - 【新增】引入输入输出安全合规护栏与自愈反思机制 - 【优化】重构并优化多厂商 API 适配器 (Gemini, OpenAI, DeepSeek, GLM 等) - 【优化】优化日志脱敏与 Token 预估机制 - 【移除】移除旧版 llm default 和 llm reset-key 命令,新增 llm mcp 管理命令 * 🔧 chore(deps): 更新项目依赖与配置 - 添加 mcp、jieba 和 aiodocker 依赖到配置文件及 requirements.txt - 在 pyright 配置中设置 reportMissingImports 为 none - 调整 .gitignore 中 resources 目录的忽略规则 * ♻️ refactor(tools): 重构工具终止机制并清理知识库日志输出 - 统一使用 `context.state["__end_run__"]` 替代 `EndRunResult` 控制任务结束 - 移除文件系统和向量知识库检索工具中 `ToolResult` 的 `.with_log` 调用 - 调整指令处理器(Directive)的返回值为 `tool_res.output` - 修复部分类型检查警告并优化联合类型判断语法 * ♻️ refactor(tools): 重构工具副作用指令与控制流熔断机制 - 引入 `DirectivePayload` 及 `ToolResult` 的子类以结构化表达工具副作用 - 移除通过 `context.state` 传递魔术变量的隐式控制流设计 - 重构 `DirectiveManager` 处理器接口,直接在处理器中修改 `AgentState` 并构建 `AgentRunResult` - 在 `StandardAgentExecutor` 中统一通过 `directive_manager` 调度工具返回的副作用指令 - 补全 `MessageBuilder` 中部分核心方法的文档注释 * 🐛 fix(sandbox): 修复 Docker 沙箱容器状态检测与会话清理逻辑 -【修复】修正 `is_alive` 中直接读取私有属性的问题,改用 `show()` 返回值 -【修复】解决 `execute_code` 中缓存的执行器与当前会话不一致的问题 -【优化】在清理工作区前增加容器存活检测,避免向已死容器发送请求 -【优化】创建容器时增加运行状态校验,若已停止则自动从缓存中移除并重建 -【优化】优化容器销毁和清理逻辑,静默处理容器不存在 (404) 的异常 * 📝 docs(core): 补充核心模块初始化方法的文档注释 * 🚨 auto fix by pre-commit hooks --------- Co-authored-by: webjoin111 <455457521@qq.com> Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
459 lines
20 KiB
Python
459 lines
20 KiB
Python
import copy
|
|
import re
|
|
from typing import Any
|
|
|
|
from nonebot.adapters import Message, MessageSegment
|
|
|
|
|
|
def _truncate_base64_string(value: str, threshold: int = 256) -> str:
|
|
"""如果字符串是超长的base64或data URI,则截断它。"""
|
|
if not isinstance(value, str):
|
|
return value
|
|
|
|
prefixes = ("base64://", "data:image", "data:video", "data:audio")
|
|
if value.startswith(prefixes) and len(value) > threshold:
|
|
prefix = next((p for p in prefixes if value.startswith(p)), "base64")
|
|
return f"[{prefix}_data_omitted_len={len(value)}]"
|
|
|
|
embedded_patterns = (
|
|
(re.compile(r"base64://[A-Za-z0-9+/=\s]{80,}"), "base64"),
|
|
(
|
|
re.compile(r"data:(?:image|video|audio)[^,]*,[A-Za-z0-9+/=\s]{80,}"),
|
|
"data_uri",
|
|
),
|
|
)
|
|
for pattern, tag in embedded_patterns:
|
|
if pattern.search(value):
|
|
value = pattern.sub(
|
|
lambda m: f"[{tag}_data_omitted_len={len(m.group(0))}]",
|
|
value,
|
|
)
|
|
|
|
return value
|
|
|
|
|
|
def _truncate_vector_list(vector: list, threshold: int = 10) -> list:
|
|
"""如果列表过长(通常是embedding向量),则截断它用于日志显示。"""
|
|
if isinstance(vector, list) and len(vector) > threshold:
|
|
return [*vector[:3], f"...({len(vector)} floats omitted)...", *vector[-3:]]
|
|
return vector
|
|
|
|
|
|
def _recursive_sanitize_any(obj: Any) -> Any:
|
|
"""递归清洗任何对象中的长字符串"""
|
|
if isinstance(obj, dict):
|
|
sanitized_dict = {}
|
|
for k, v in obj.items():
|
|
if (
|
|
k in ("data", "b64_json", "inlineData", "image_base64", "b64_data")
|
|
and isinstance(v, str)
|
|
and len(v) > 512
|
|
):
|
|
sanitized_dict[k] = f"[raw_base64_data_omitted_key={k}_len={len(v)}]"
|
|
else:
|
|
sanitized_dict[k] = _recursive_sanitize_any(v)
|
|
return sanitized_dict
|
|
elif isinstance(obj, list):
|
|
return [_recursive_sanitize_any(v) for v in obj]
|
|
elif isinstance(obj, str):
|
|
if len(obj) > 2048 and set(obj).issubset(
|
|
set(
|
|
"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/=\r\n\t"
|
|
)
|
|
):
|
|
return f"[heuristic_base64_omitted_len={len(obj)}]"
|
|
return _truncate_base64_string(obj)
|
|
return obj
|
|
|
|
|
|
def _sanitize_ui_html(html_string: str) -> str:
|
|
"""
|
|
专门用于净化UI渲染调试HTML的函数。
|
|
它会查找所有内联的base64数据(如字体、图片)并将其截断。
|
|
同时会折叠冗长的样式标签,避免主题 CSS 在日志中撑爆。
|
|
"""
|
|
if not isinstance(html_string, str):
|
|
return html_string
|
|
|
|
pattern = re.compile(r"(data:[^;]+;base64,)[A-Za-z0-9+/=\s]{100,}")
|
|
|
|
def replacer(match):
|
|
prefix = match.group(1)
|
|
original_len = len(match.group(0)) - len(prefix)
|
|
return f"{prefix}[...base64_omitted_len={original_len}...]"
|
|
|
|
html_string = pattern.sub(replacer, html_string)
|
|
|
|
pattern_style = re.compile(
|
|
r"(<style[^>]*>)(.*?)(</style>)", re.DOTALL | re.IGNORECASE
|
|
)
|
|
|
|
def replacer_style(match):
|
|
start_tag, content, end_tag = match.group(1), match.group(2), match.group(3)
|
|
keywords = [
|
|
"Base Styles - Assembled by Jinja2",
|
|
"Utility Classes",
|
|
"@layer reset, base, components, utilities;",
|
|
]
|
|
|
|
if len(content) > 600 or any(keyword in content for keyword in keywords):
|
|
excerpt = (
|
|
f"\n /* [theme.css.jinja content hidden for brevity - "
|
|
f"{len(content)} chars] */\n"
|
|
)
|
|
return f"{start_tag}{excerpt}{end_tag}"
|
|
|
|
return match.group(0)
|
|
|
|
return pattern_style.sub(replacer_style, html_string)
|
|
|
|
|
|
def _sanitize_nonebot_message(message: Message) -> Message:
|
|
"""净化nonebot.adapter.Message对象,用于日志记录。"""
|
|
sanitized_message = copy.deepcopy(message)
|
|
for seg in sanitized_message:
|
|
seg: MessageSegment
|
|
if seg.type in ("image", "record", "video"):
|
|
file_info = seg.data.get("file", "")
|
|
if isinstance(file_info, str):
|
|
seg.data["file"] = _truncate_base64_string(file_info)
|
|
return sanitized_message
|
|
|
|
|
|
def _sanitize_openai_response(response_json: dict) -> dict:
|
|
"""净化OpenAI兼容API的响应体。"""
|
|
from zhenxun.services.ai.config import (
|
|
DebugLogOptions,
|
|
get_llm_config,
|
|
)
|
|
|
|
debug_conf = get_llm_config().debug_log
|
|
if isinstance(debug_conf, bool):
|
|
debug_conf = DebugLogOptions(
|
|
show_tools=debug_conf, show_schema=debug_conf, show_safety=debug_conf
|
|
)
|
|
|
|
try:
|
|
sanitized_json = _recursive_sanitize_any(copy.deepcopy(response_json))
|
|
|
|
if "tools" in sanitized_json and not debug_conf.show_tools:
|
|
tools = sanitized_json["tools"]
|
|
if isinstance(tools, list):
|
|
tool_names = []
|
|
for t in tools:
|
|
if isinstance(t, dict):
|
|
name = None
|
|
if "function" in t and isinstance(t["function"], dict):
|
|
name = t["function"].get("name")
|
|
if not name and "name" in t:
|
|
name = t.get("name")
|
|
if not name and "type" in t:
|
|
name = t.get("type")
|
|
tool_names.append(name or "unknown")
|
|
sanitized_json["tools"] = (
|
|
f"<{len(tool_names)} tools hidden: {', '.join(tool_names)}>"
|
|
)
|
|
|
|
if not debug_conf.show_safety:
|
|
for safety_key in ("content_filters", "prompt_annotations"):
|
|
if safety_key in sanitized_json:
|
|
sanitized_json[safety_key] = "<Safety Ratings Hidden>"
|
|
|
|
if "choices" in sanitized_json and isinstance(sanitized_json["choices"], list):
|
|
for choice in sanitized_json["choices"]:
|
|
if not debug_conf.show_safety and "content_filter_results" in choice:
|
|
choice["content_filter_results"] = "<Safety Ratings Hidden>"
|
|
if "message" in choice and isinstance(choice["message"], dict):
|
|
message = choice["message"]
|
|
if "images" in message and isinstance(message["images"], list):
|
|
for i, image_info in enumerate(message["images"]):
|
|
if "image_url" in image_info and isinstance(
|
|
image_info["image_url"], dict
|
|
):
|
|
url = image_info["image_url"].get("url", "")
|
|
message["images"][i]["image_url"]["url"] = (
|
|
_truncate_base64_string(url)
|
|
)
|
|
if "reasoning_details" in message and isinstance(
|
|
message["reasoning_details"], list
|
|
):
|
|
for detail in message["reasoning_details"]:
|
|
if isinstance(detail, dict):
|
|
if "data" in detail and isinstance(detail["data"], str):
|
|
if len(detail["data"]) > 100:
|
|
detail["data"] = (
|
|
f"[encrypted_data_omitted_len={len(detail['data'])}]"
|
|
)
|
|
if "text" in detail and isinstance(detail["text"], str):
|
|
detail["text"] = _truncate_base64_string(
|
|
detail["text"], threshold=2000
|
|
)
|
|
if "data" in sanitized_json and isinstance(sanitized_json["data"], list):
|
|
for item in sanitized_json["data"]:
|
|
if "embedding" in item and isinstance(item["embedding"], list):
|
|
item["embedding"] = _truncate_vector_list(item["embedding"])
|
|
if "b64_json" in item and isinstance(item["b64_json"], str):
|
|
if len(item["b64_json"]) > 256:
|
|
item["b64_json"] = (
|
|
f"[base64_json_omitted_len={len(item['b64_json'])}]"
|
|
)
|
|
if "input" in sanitized_json and isinstance(sanitized_json["input"], list):
|
|
for item in sanitized_json["input"]:
|
|
if "content" in item and isinstance(item["content"], list):
|
|
for part in item["content"]:
|
|
if isinstance(part, dict) and part.get("type") == "input_image":
|
|
image_url = part.get("image_url")
|
|
if isinstance(image_url, str):
|
|
part["image_url"] = _truncate_base64_string(image_url)
|
|
if "output" in sanitized_json and isinstance(sanitized_json["output"], list):
|
|
for item in sanitized_json["output"]:
|
|
if isinstance(item, dict) and "encrypted_content" in item:
|
|
content_val = item["encrypted_content"]
|
|
if isinstance(content_val, str) and len(content_val) > 64:
|
|
item["encrypted_content"] = (
|
|
f"[encrypted_content_omitted_len={len(content_val)}]"
|
|
)
|
|
return sanitized_json
|
|
except Exception:
|
|
return response_json
|
|
|
|
|
|
def _sanitize_openai_request(body: dict) -> dict:
|
|
"""净化OpenAI兼容API的请求体,主要截断图片base64。"""
|
|
from zhenxun.services.ai.config import (
|
|
DebugLogOptions,
|
|
get_llm_config,
|
|
)
|
|
|
|
debug_conf = get_llm_config().debug_log
|
|
if isinstance(debug_conf, bool):
|
|
debug_conf = DebugLogOptions(
|
|
show_tools=debug_conf, show_schema=debug_conf, show_safety=debug_conf
|
|
)
|
|
|
|
try:
|
|
sanitized_json = _recursive_sanitize_any(copy.deepcopy(body))
|
|
if "tools" in sanitized_json and not debug_conf.show_tools:
|
|
tools = sanitized_json["tools"]
|
|
if isinstance(tools, list):
|
|
tool_names = []
|
|
for t in tools:
|
|
if isinstance(t, dict):
|
|
name = None
|
|
if "function" in t and isinstance(t["function"], dict):
|
|
name = t["function"].get("name")
|
|
if not name and "name" in t:
|
|
name = t.get("name")
|
|
if not name and "type" in t:
|
|
name = t.get("type")
|
|
tool_names.append(name or "unknown")
|
|
sanitized_json["tools"] = (
|
|
f"<{len(tool_names)} tools hidden: {', '.join(tool_names)}>"
|
|
)
|
|
|
|
if "response_format" in sanitized_json and not debug_conf.show_schema:
|
|
response_format = sanitized_json["response_format"]
|
|
if isinstance(response_format, dict):
|
|
if response_format.get("type") == "json_schema":
|
|
sanitized_json["response_format"] = {
|
|
"type": "json_schema",
|
|
"json_schema": "<JSON Schema Hidden>",
|
|
}
|
|
|
|
return sanitized_json
|
|
except Exception:
|
|
return body
|
|
|
|
|
|
def _sanitize_gemini_response(response_json: dict) -> dict:
|
|
"""净化Gemini API的响应体,处理文本和图片生成两种格式。"""
|
|
from zhenxun.services.ai.config import (
|
|
DebugLogOptions,
|
|
get_llm_config,
|
|
)
|
|
|
|
debug_conf = get_llm_config().debug_log
|
|
if isinstance(debug_conf, bool):
|
|
debug_conf = DebugLogOptions(
|
|
show_tools=debug_conf, show_schema=debug_conf, show_safety=debug_conf
|
|
)
|
|
|
|
try:
|
|
sanitized_json = _recursive_sanitize_any(copy.deepcopy(response_json))
|
|
|
|
if "thoughtSignature" in sanitized_json:
|
|
sig = sanitized_json["thoughtSignature"]
|
|
if isinstance(sig, str) and len(sig) > 64:
|
|
sanitized_json["thoughtSignature"] = (
|
|
f"[signature_omitted_len={len(sig)}]"
|
|
)
|
|
|
|
def _process_candidates(candidates_list: list):
|
|
"""辅助函数,用于处理任何 candidates 列表。"""
|
|
if not isinstance(candidates_list, list):
|
|
return
|
|
for candidate in candidates_list:
|
|
if "content" in candidate and isinstance(candidate["content"], dict):
|
|
content = candidate["content"]
|
|
if "parts" in content and isinstance(content["parts"], list):
|
|
for i, part in enumerate(content["parts"]):
|
|
if "inlineData" in part and isinstance(
|
|
part["inlineData"], dict
|
|
):
|
|
data = part["inlineData"].get("data", "")
|
|
if isinstance(data, str) and len(data) > 256:
|
|
content["parts"][i]["inlineData"]["data"] = (
|
|
f"[base64_data_omitted_len={len(data)}]"
|
|
)
|
|
if (
|
|
"thoughtSignature" in part
|
|
or "thought_signature" in part
|
|
):
|
|
sig_key = (
|
|
"thoughtSignature"
|
|
if "thoughtSignature" in part
|
|
else "thought_signature"
|
|
)
|
|
signature = part.get(sig_key, "")
|
|
if isinstance(signature, str) and len(signature) > 64:
|
|
content["parts"][i][sig_key] = (
|
|
f"[signature_omitted_len={len(signature)}]"
|
|
)
|
|
if not debug_conf.show_safety and isinstance(candidate, dict):
|
|
if "safetyRatings" in candidate:
|
|
candidate["safetyRatings"] = "<Safety Ratings Hidden>"
|
|
|
|
if "candidates" in sanitized_json:
|
|
_process_candidates(sanitized_json["candidates"])
|
|
|
|
if "image_generation" in sanitized_json and isinstance(
|
|
sanitized_json["image_generation"], dict
|
|
):
|
|
if "candidates" in sanitized_json["image_generation"]:
|
|
_process_candidates(sanitized_json["image_generation"]["candidates"])
|
|
|
|
if "embeddings" in sanitized_json and isinstance(
|
|
sanitized_json["embeddings"], list
|
|
):
|
|
for embedding in sanitized_json["embeddings"]:
|
|
if "values" in embedding and isinstance(embedding["values"], list):
|
|
embedding["values"] = _truncate_vector_list(embedding["values"])
|
|
|
|
if not debug_conf.show_safety and "promptFeedback" in sanitized_json:
|
|
prompt_feedback = sanitized_json.get("promptFeedback") or {}
|
|
if isinstance(prompt_feedback, dict) and "safetyRatings" in prompt_feedback:
|
|
prompt_feedback["safetyRatings"] = "<Safety Ratings Hidden>"
|
|
sanitized_json["promptFeedback"] = prompt_feedback
|
|
|
|
return sanitized_json
|
|
except Exception:
|
|
return response_json
|
|
|
|
|
|
def _sanitize_gemini_request(body: dict) -> dict:
|
|
"""净化Gemini API的请求体,进行结构转换和总结。"""
|
|
from zhenxun.services.ai.config import (
|
|
DebugLogOptions,
|
|
get_llm_config,
|
|
)
|
|
|
|
debug_conf = get_llm_config().debug_log
|
|
if isinstance(debug_conf, bool):
|
|
debug_conf = DebugLogOptions(
|
|
show_tools=debug_conf, show_schema=debug_conf, show_safety=debug_conf
|
|
)
|
|
|
|
try:
|
|
sanitized_body = _recursive_sanitize_any(copy.deepcopy(body))
|
|
if "tools" in sanitized_body and not debug_conf.show_tools:
|
|
tool_names = []
|
|
for tool_group in sanitized_body["tools"]:
|
|
if isinstance(tool_group, dict):
|
|
for key, value in tool_group.items():
|
|
if key == "functionDeclarations" and isinstance(value, list):
|
|
for func in value:
|
|
if isinstance(func, dict):
|
|
tool_names.append(func.get("name", "unknown"))
|
|
else:
|
|
tool_names.append(key)
|
|
sanitized_body["tools"] = (
|
|
f"<{len(tool_names)} tools hidden: {', '.join(tool_names)}>"
|
|
)
|
|
|
|
if not debug_conf.show_safety and "safetySettings" in sanitized_body:
|
|
sanitized_body["safetySettings"] = "<Safety Settings Hidden>"
|
|
|
|
if not debug_conf.show_schema and "generationConfig" in sanitized_body:
|
|
generation_config = sanitized_body["generationConfig"]
|
|
if (
|
|
isinstance(generation_config, dict)
|
|
and "responseJsonSchema" in generation_config
|
|
):
|
|
generation_config["responseJsonSchema"] = "<JSON Schema Hidden>"
|
|
|
|
if "contents" in sanitized_body and isinstance(
|
|
sanitized_body["contents"], list
|
|
):
|
|
for content_item in sanitized_body["contents"]:
|
|
if "parts" in content_item and isinstance(content_item["parts"], list):
|
|
new_parts = []
|
|
for part in content_item["parts"]:
|
|
if "inlineData" in part and isinstance(
|
|
part["inlineData"], dict
|
|
):
|
|
data = part["inlineData"].get("data")
|
|
if isinstance(data, str):
|
|
mime_type = part["inlineData"].get(
|
|
"mimeType", "unknown"
|
|
)
|
|
new_parts.append(
|
|
{"text": f"[多模态图片/文件: {mime_type}]"}
|
|
)
|
|
continue
|
|
new_parts.append(part)
|
|
|
|
if "thoughtSignature" in part:
|
|
sig = part["thoughtSignature"]
|
|
if isinstance(sig, str) and len(sig) > 64:
|
|
part["thoughtSignature"] = (
|
|
f"[signature_omitted_len={len(sig)}]"
|
|
)
|
|
|
|
content_item["parts"] = new_parts
|
|
return sanitized_body
|
|
except Exception:
|
|
return body
|
|
|
|
|
|
def sanitize_for_logging(data: Any, context: str | None = None) -> Any:
|
|
"""
|
|
统一的日志净化入口。
|
|
|
|
Args:
|
|
data: 需要净化的数据 (dict, Message, etc.).
|
|
context: 净化场景的上下文标识,例如 'gemini_request', 'openai_response'.
|
|
|
|
Returns:
|
|
净化后的数据。
|
|
"""
|
|
if context == "nonebot_message":
|
|
if isinstance(data, Message):
|
|
return _sanitize_nonebot_message(data)
|
|
elif context in ("openai_response", "openai_responses_response"):
|
|
if isinstance(data, dict):
|
|
return _sanitize_openai_response(data)
|
|
elif context == "gemini_response":
|
|
if isinstance(data, dict):
|
|
return _sanitize_gemini_response(data)
|
|
elif context == "gemini_request":
|
|
if isinstance(data, dict):
|
|
return _sanitize_gemini_request(data)
|
|
elif context in ("openai_request", "openai_responses_request"):
|
|
if isinstance(data, dict):
|
|
return _sanitize_openai_request(data)
|
|
elif context == "ui_html":
|
|
if isinstance(data, str):
|
|
return _sanitize_ui_html(data)
|
|
|
|
return _recursive_sanitize_any(data)
|