Files
zhenxun_bot/zhenxun/services/ai/flow/workflow/policies.py
T
922d092650 ♻️ refactor(core): 重构 AI 能力与定时任务调度系统 (#2148)
* ♻️ refactor(core): 重构 AI 能力与定时任务调度系统

- 【AI 能力与工具】重构 Capability 注册与管理机制,引入 CapabilityManager 统一管理
- 移除全局能力注册表,改用声明式装饰器 `@capability` 进行解耦注册
- 重构工具解析器链,使用统一的 BaseToolResolver 代替原有的多个特定解析器
- 增强工具查询过滤,支持通配符匹配、工具箱过滤和排除标签
- 【定时任务调度】重构定时任务管理器,引入 SchedulerRegistry 统一管理任务元数据
- 引入 JobConfig 聚合定时任务配置,支持用户维度的定时任务调度
- 重构执行分发器,支持并发限制、串行间隔和随机延迟打散
- 【运行上下文】引入 ScheduledDeps 以支持后台和定时任务环境下的依赖注入
- 优化 RunContext,支持从定时任务上下文快速构造,并提供 emit 辅助方法
- 【日志与监控】引入 AILoggerProxy,实现 AI 各模块的专属日志输出
- 将各模块的全局 logger 替换为对应的模块专属日志代理
- 【其他优化】修复 Pydantic V1 兼容层中 model_validator 的装饰器兼容性问题
- 在非交互式环境(如定时任务)中自动隐藏 HITL 交互工具以节省 Token

* ♻️ refactor(core): 优化内部导入路径并提升 Pydantic 兼容性

- 【重构】将 `services/ai` 模块内的绝对导入重构为相对导入,优化包结构
- 【重构】移除不必要的 `if TYPE_CHECKING` 保护,通过 `from __future__ import annotations` 直接导入类型
- 【清理】清理 `core/messages/types.py` 中未使用的 `AssistantContentUnion` 等联合类型定义
- 【优化】在 `utils/pydantic_compat.py` 中新增 `model_rebuild` 兼容函数,统一 Pydantic V1/V2 的模型重建逻辑
- 【优化】将部分函数内部的延迟导入提升至模块顶部,规范代码结构

* ♻️ refactor(imports): 优化导入路径为相对导入并清理冗余导入

- 【重构】将 AI 服务相关模块中的绝对导入路径修改为相对导入,提升模块内聚性与可移植性
- 【清理】移除多处函数内部或类方法中未使用的冗余导入,避免循环引用和资源浪费
- 【格式化】微调部分工具装饰器和返回语句的格式与尾随逗号

* 🚨 auto fix by pre-commit hooks

---------

Co-authored-by: webjoin111 <455457521@qq.com>
Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
2026-07-10 09:14:06 +08:00

186 lines
5.7 KiB
Python

import copy
from enum import Enum
from typing import Any
from pydantic import BaseModel, Field
from zhenxun.services.ai.llm.api import generate_structured
from zhenxun.services.ai.utils.logger import log_flow as logger
from .types import StepInput
class PolicyAction(str, Enum):
RETRY = "retry"
CONTINUE = "continue"
ABORT = "abort"
FALLBACK = "fallback"
class PolicyResult(BaseModel):
"""错误策略执行结果"""
action: PolicyAction
"""采取的具体恢复策略动作"""
delay: float = 0.0
"""执行延迟或重试前需要等待的缓冲秒数"""
new_input: StepInput | None = None
"""用于动态纠错自愈时替换传入的新参数结构"""
fallback_node: Any | None = None
"""策略裁定降级时所指定的备用工作流节点"""
healer_agent_name: str | None = None
"""执行了高级自愈的大模型或修复者名称"""
class BaseFailurePolicy:
"""错误处理策略抽象基类"""
async def handle_failure(
self, node: Any, exception: BaseException, step_input: StepInput, context: Any
) -> PolicyResult:
"""
处理节点执行失败的策略入口方法。
参数:
node: 发生异常的目标工作流节点。
exception: 捕获到的具体异常实例。
step_input: 节点执行时的原始输入数据。
context: 当前工作流运行上下文。
返回:
PolicyResult: 包含错误恢复动作、延迟时间以及备用参数等决策信息的策略结果对象。
""" # noqa: E501
raise NotImplementedError
class AbortPolicy(BaseFailurePolicy):
"""直接中断策略"""
async def handle_failure(
self, node: Any, exception: BaseException, step_input: StepInput, context: Any
) -> PolicyResult:
return PolicyResult(action=PolicyAction.ABORT)
class SkipPolicy(BaseFailurePolicy):
"""跳过并继续策略"""
async def handle_failure(
self, node: Any, exception: BaseException, step_input: StepInput, context: Any
) -> PolicyResult:
return PolicyResult(action=PolicyAction.CONTINUE)
class RetryPolicy(BaseFailurePolicy):
"""退避重试策略"""
def __init__(self, max_retries: int = 3, delay: float = 1.0):
"""
初始化退避重试策略。
参数:
max_retries: 最大允许重试的次数限制,默认 3。
delay: 每次重试前需要等待和睡眠的秒数,默认 1.0。
"""
self.max_retries = max_retries
self.delay = delay
async def handle_failure(
self, node: Any, exception: BaseException, step_input: StepInput, context: Any
) -> PolicyResult:
counts = context.state.setdefault("__retry_counts__", {})
key = f"{node.name}_{id(self)}"
counts[key] = counts.get(key, 0) + 1
if counts[key] <= self.max_retries:
return PolicyResult(action=PolicyAction.RETRY, delay=self.delay)
return PolicyResult(action=PolicyAction.ABORT)
class FallbackPolicy(BaseFailurePolicy):
"""降级路由策略"""
def __init__(self, fallback_node: Any):
"""
初始化降级路由策略。
参数:
fallback_node: 当主节点发生致命故障时,直接转入执行的备用降级节点。
"""
self.fallback_node = fallback_node
async def handle_failure(
self, node: Any, exception: BaseException, step_input: StepInput, context: Any
) -> PolicyResult:
return PolicyResult(
action=PolicyAction.FALLBACK, fallback_node=self.fallback_node
)
class SelfHealingPolicy(BaseFailurePolicy):
"""大模型高级自愈策略"""
def __init__(self, healer_model: str, max_retries: int = 2):
"""
初始化大模型高级自愈策略。
参数:
healer_model: 用于分析错误原因并智能修复入参的大模型名称。
max_retries: 最大尝试自愈修复的次数,默认 2。
"""
self.healer_model = healer_model
self.max_retries = max_retries
async def handle_failure(
self, node: Any, exception: BaseException, step_input: StepInput, context: Any
) -> PolicyResult:
counts = context.state.setdefault("__heal_counts__", {})
key = f"{node.name}_{id(self)}"
counts[key] = counts.get(key, 0) + 1
if counts[key] > self.max_retries:
logger.warning(f"节点 '{node.name}' 自愈次数达上限,宣告失败。")
return PolicyResult(action=PolicyAction.ABORT)
class HealedInput(BaseModel):
"""自愈后输入结构"""
fixed_input: str = Field(
description="""修复后的输入参数 必须是完全合法的数据结构"""
)
prompt = f"""# Self-Healing Task
请修复节点 `{node.name}` 的参数错误。
## Original Input
{step_input.input}
## Exception
{exception}
## Requirements
- 分析错误原因
- 将输入修复为可被程序正确解析的格式
- 只输出修复后的结果,不要输出额外解释
"""
try:
logger.info(f"🩹 触发 AI 自愈分析 (节点: {node.name})...")
res = await generate_structured(
prompt, response_model=HealedInput, model=self.healer_model
)
new_input = copy.copy(step_input)
new_input.input = res.fixed_input
return PolicyResult(
action=PolicyAction.RETRY,
new_input=new_input,
healer_agent_name=self.healer_model,
)
except Exception as e:
logger.error(f"自愈过程发生大模型调用异常: {e}")
return PolicyResult(action=PolicyAction.ABORT)