ForcePilot/backend/package/yuxi/channels/infrastructure/host_shutdown.py

566 lines
25 KiB
Python
Raw Normal View History

"""宿主关停编排器。
本模块是编排层的关停编排入口按反向顺序释放渠道网关
资源每步有超时超时强制进入下一步并告警
关停不得丢弃在途请求除非超时强制终止
关停顺序
1. 标记宿主为 draining停止接受新请求
2. 等待在途请求完成带超时 30s
3. 拆除管道
4. 停止传输引擎管理器入站Worker
5. 停止渠道插件按反向依赖拓扑
6. 关闭被驱动适配器
7. 释放领域核心资源
- §15.2 规范定义了"注销驱动适配器"步骤当前系统未接入驱动适配器
机制``driving_adapter_registrar`` / ``driving_adapter_unregistrar``
未实现故此处省略该步骤当驱动适配器实现就绪后需在步骤 2 3
之间补充注销逻辑并恢复与规范一致的 7 步序列
- Outbox 恢复扫描FR-22配对过期扫描FR-33与审计日志保留期
清理FR-34已迁移至 scheduler worker handler不再由 api 进程后台
任务执行因此关停时无需取消对应任务
"""
from __future__ import annotations
import asyncio
from collections.abc import Awaitable, Callable, Sequence
from typing import Any
from yuxi.channels.application.extension.config_source_registry import (
ConfigSourceRegistry,
)
from yuxi.channels.application.extension.event_bus import EventBus
from yuxi.channels.application.extension.event_subscription_registry import (
EventSubscriptionRegistry,
)
from yuxi.channels.application.extension.stage_slot_registry import StageSlotRegistry
from yuxi.channels.application.lifecycle.plugin_dependency_resolver import (
PluginDependencyResolver,
)
from yuxi.channels.application.lifecycle.plugin_lifecycle_manager import (
PluginLifecycleManager,
)
from yuxi.channels.application.pipeline.stage_slot_injector import StageSlotInjector
from yuxi.channels.application.transport import TransportManager
from yuxi.channels.contract.dtos.lifecycle import LifecycleResult
from yuxi.channels.contract.errors.server import OperationTimeoutError
from yuxi.channels.contract.plugin.lifecycle import LifecycleState
from yuxi.channels.contract.plugin.manifest import ChannelManifest
from yuxi.channels.contract.ports.driven.closeable_port import CloseablePort
from yuxi.channels.contract.ports.driven.logger_port import LoggerPort
from yuxi.channels.core.registry.plugin_registry import PluginRegistry
from yuxi.channels.infrastructure.host_bootstrap import HostBootstrap
__all__ = [
"HostShutdown",
]
class HostShutdown:
"""宿主关停编排器。
按反向顺序释放渠道网关资源
关停顺序
1. 标记宿主为 draining停止接受新请求
2. 等待在途请求完成带超时 30s
3. 拆除管道
4. 停止传输引擎管理器入站Worker
5. 停止渠道插件按反向依赖拓扑
6. 关闭被驱动适配器
7. 释放领域核心资源
关键约束
- 每个步骤 **必须** 有超时超时后强制进入下一步并告警
- 关停 **不得** 丢弃在途请求除非超时强制终止
- 插件停止 **必须** 按反向依赖拓扑顺序被依赖者后停止
- 关停过程中任一步异常不得中断后续步骤记录告警后继续
- §15.2 规范定义了"注销驱动适配器"步骤当前系统未接入驱动适配器
机制故省略该步骤详见模块级文档字符串
- Outbox 恢复扫描等任务已迁移至 scheduler worker关停时不再取消
"""
#: 在途请求等待超时(秒),超时强制进入下一步并告警
SHUTDOWN_TIMEOUT: float = 30.0
#: 标记 draining 超时(秒)
_MARK_DRAINING_TIMEOUT: float = 10.0
#: 拆除管道超时(秒)
_TEARDOWN_PIPELINE_TIMEOUT: float = 30.0
#: 停止传输引擎管理器超时(秒)
_STOP_TRANSPORT_MANAGER_TIMEOUT: float = 30.0
#: 单个插件停止超时(秒)
_STOP_PLUGIN_TIMEOUT: float = 30.0
#: 关闭被驱动适配器超时(秒)
_CLOSE_DRIVEN_TIMEOUT: float = 30.0
#: 释放核心资源超时(秒)
_RELEASE_CORE_TIMEOUT: float = 30.0
def __init__(
self,
plugin_lifecycle_manager: PluginLifecycleManager,
plugin_registry: PluginRegistry,
plugin_dependency_resolver: PluginDependencyResolver,
stage_slot_injector: StageSlotInjector,
stage_slot_registry: StageSlotRegistry,
event_subscription_registry: EventSubscriptionRegistry,
config_source_registry: ConfigSourceRegistry,
event_bus: EventBus,
transport_manager: TransportManager,
logger: LoggerPort,
host_bootstrap: HostBootstrap,
inflight_drain_waiter: Callable[[float], Awaitable[None]] | None = None,
app_driven_adapters: Sequence[Any] | None = None,
) -> None:
"""初始化宿主关停编排器。
Args:
plugin_lifecycle_manager: 插件生命周期管理器编排 stop/unload
plugin_registry: 插件注册表按状态列出待停止插件
plugin_dependency_resolver: 插件依赖解析器用于计算反向拓扑序
stage_slot_injector: 管道阶段插槽注入器
stage_slot_registry: 阶段槽位注册中心
event_subscription_registry: 事件订阅注册中心
config_source_registry: 配置源注册中心
event_bus: 事件总线
transport_manager: 传输引擎管理器在关停步骤4调用stop()停止入站Worker
logger: 日志端口记录关停过程
host_bootstrap: 宿主启动编排器实例在步骤 7 释放核心资源时
惰性读取其 ``pluginReloadTask`` / ``transportManagerTask``
属性并取消未完成的后台任务传入实例而非任务引用是为了在
``HostShutdown`` 创建于 ``bootstrap()`` 之前时仍能取到
``bootstrap()`` 执行后填充的任务引用
inflight_drain_waiter: 在途请求排空回调可选接受超时参数
在步骤 2 调用以等待在途请求完成 ``None`` 时步骤 2 直接
等待固定超时
app_driven_adapters: 应用级被驱动适配器列表可选INF-017
包含 ``RedisCacheAdapter`` / ``ChannelPersistenceAdapter`` /
``ARQQueueAdapter`` / ``RedisConfigAdapter`` 等在
``HostBootstrap`` 构造时创建跨请求共享的应用级适配器实例
在关停步骤 5 显式调用 ``close()`` 释放 Redis 连接DB 会话等
资源 ``None`` 或空时跳过应用级适配器关闭向后兼容
"""
self._plugin_lifecycle = plugin_lifecycle_manager
self._plugin_registry = plugin_registry
self._resolver = plugin_dependency_resolver
self._stage_slot_injector = stage_slot_injector
self._stage_slot_registry = stage_slot_registry
self._event_subscription_registry = event_subscription_registry
self._config_source_registry = config_source_registry
self._event_bus = event_bus
self._transport_manager = transport_manager
self._logger = logger
self._inflight_drain_waiter = inflight_drain_waiter
self._host_bootstrap = host_bootstrap
self._app_driven_adapters: tuple[Any, ...] = tuple(app_driven_adapters) if app_driven_adapters else ()
self._draining = False
async def shutdown(self) -> None:
"""执行关停编排。
每步有超时超时强制进入下一步并告警关停不得丢弃在途
请求除非超时强制终止插件停止按反向依赖拓扑顺序被依赖者
后停止任一步异常不中断后续步骤记录告警后继续
"""
await self._logger.info("宿主关停开始")
# 1. 标记宿主为 draining停止接受新请求
await self._markDraining()
# 2. 等待在途请求完成(带超时)
await self._waitForInflightRequests(self.SHUTDOWN_TIMEOUT)
# 3. 拆除管道
await self._teardownPipelines()
# 4. 停止传输引擎管理器入站Worker
await self._stopTransportManager()
# 5. 停止渠道插件(按反向依赖拓扑)
await self._stopPlugins()
# 6. 关闭被驱动适配器
await self._closeDrivenAdapters()
# 7. 释放领域核心资源
await self._releaseCoreResources()
await self._logger.info("宿主关停完成")
async def _markDraining(self) -> None:
"""标记宿主为 draining停止接受新请求。
超时强制进入下一步并告警
"""
try:
await asyncio.wait_for(
self._setDrainingState(),
timeout=self._MARK_DRAINING_TIMEOUT,
)
# 将内置 asyncio.TimeoutError 翻译为契约 OperationTimeoutError 记录结构化告警
# (硬约束:所有模块必须使用自定义异常类而非原生异常)
except TimeoutError:
await self._handleStepTimeout("标记 draining", self._MARK_DRAINING_TIMEOUT)
except Exception as e:
await self._logger.warn(
f"标记 draining 异常,强制进入下一步: {e}",
error_type=type(e).__name__,
error=str(e),
)
async def _setDrainingState(self) -> None:
"""设置 draining 状态。"""
self._draining = True
await self._logger.info("宿主已标记为 draining停止接受新请求")
async def _handleStepTimeout(self, step_name: str, timeout_seconds: float) -> None:
"""记录关停步骤超时告警(将内置 ``TimeoutError`` 翻译为契约 ``OperationTimeoutError``)。
满足硬约束所有模块必须使用自定义异常类而非原生异常
e.g., ``TimeoutError``, ``ValueError``防止原生异常穿透至核心层
shutdown 路径属降级处理将内置 ``asyncio.TimeoutError`` 翻译为
``OperationTimeoutError`` 后仅记录结构化告警不重新抛出避免中断
后续关停步骤 ``HostBootstrap`` 关键步骤"包装并终止"的语义不同
shutdown 要求"超时强制进入下一步"
Args:
step_name: 步骤名称用于告警标识
timeout_seconds: 超时秒数转换为 ``timeout_ms`` 写入告警字段
"""
timeout_err = OperationTimeoutError(
timeout_ms=int(timeout_seconds * 1000),
message=f"{step_name} 超时: timeout={timeout_seconds}s",
)
await self._logger.warn(
f"{step_name} 超时,强制进入下一步",
step=step_name,
timeout_ms=timeout_err.timeout_ms,
error_type=type(timeout_err).__name__,
error=timeout_err.message,
)
async def _waitForInflightRequests(self, timeout: float) -> None:
"""等待在途请求完成(带超时)。
若调用方提供了 ``inflight_drain_waiter`` 回调则调用之并传入超时
参数由回调内部实现超时 ``InflightRequestTracker.wait_drained``
使用 ``asyncio.wait_for`` 等待归零事件 ``None`` 时等待固定超时
后继续超时强制进入下一步并告警
Args:
timeout: 等待超时时间超时强制进入下一步并告警
"""
try:
if self._inflight_drain_waiter is not None:
await self._inflight_drain_waiter(timeout)
else:
await self._logger.info(
f"无在途请求排空回调,直接继续: timeout={timeout}s",
)
# 将内置 asyncio.TimeoutError 翻译为契约 OperationTimeoutError 记录结构化告警
# (硬约束:所有模块必须使用自定义异常类而非原生异常)
except TimeoutError:
await self._handleStepTimeout("等待在途请求", timeout)
except Exception as e:
await self._logger.warn(
f"等待在途请求异常,强制进入下一步: {e}",
error_type=type(e).__name__,
error=str(e),
)
async def _teardownPipelines(self) -> None:
"""拆除管道INF-016清理阶段槽位注册
管道为请求级组件在途请求完成后已无活跃管道实例本步骤记录管道
拆除状态并调用 ``StageSlotRegistry.clear`` 清理所有插件注册的
阶段槽位确保下一次 ``bootstrap()`` 不会复用上次启动残留的槽位
否则 ``_assemblePipelines`` 的锚点校验可能命中陈旧锚点超时
强制进入下一步并告警
"""
try:
await asyncio.wait_for(
self._teardownPipelinesInternal(),
timeout=self._TEARDOWN_PIPELINE_TIMEOUT,
)
# 将内置 asyncio.TimeoutError 翻译为契约 OperationTimeoutError 记录结构化告警
# (硬约束:所有模块必须使用自定义异常类而非原生异常)
except TimeoutError:
await self._handleStepTimeout("拆除管道", self._TEARDOWN_PIPELINE_TIMEOUT)
except Exception as e:
await self._logger.warn(
f"拆除管道异常,强制进入下一步: {e}",
error_type=type(e).__name__,
error=str(e),
)
async def _teardownPipelinesInternal(self) -> None:
"""记录管道拆除状态并清理阶段槽位注册INF-016"""
inbound_slots = self._stage_slot_registry.findByPipeline("inbound")
outbound_slots = self._stage_slot_registry.findByPipeline("outbound")
control_slots = self._stage_slot_registry.findByPipeline("control")
# 清理所有阶段槽位注册,避免下次 bootstrap 复用陈旧槽位
self._stage_slot_registry.clear()
await self._logger.info(
"管道已拆除(请求级,在途请求完成后无活跃实例,槽位注册已清理)",
inbound_slot_count=len(inbound_slots),
outbound_slot_count=len(outbound_slots),
control_slot_count=len(control_slots),
)
async def _stopTransportManager(self) -> None:
"""停止传输引擎管理器。
调用 ``TransportManager.stop()`` 优雅停止所有入站传输 Worker
Puller/Stream停止接收新消息并释放连接资源超时强制
进入下一步并告警
"""
try:
await asyncio.wait_for(
self._transport_manager.stop(),
timeout=self._STOP_TRANSPORT_MANAGER_TIMEOUT,
)
except TimeoutError:
await self._handleStepTimeout("停止传输引擎管理器", self._STOP_TRANSPORT_MANAGER_TIMEOUT)
except Exception as e:
await self._logger.warn(
f"停止传输引擎管理器异常,强制进入下一步: {e}",
error_type=type(e).__name__,
error=str(e),
)
async def _stopPlugins(self) -> None:
"""按反向依赖拓扑停止渠道插件。
列出所有 ``STARTED`` ``PAUSED`` 状态的插件按依赖解析器计算反向
拓扑序被依赖者后停止逐个调用 ``stop``单个插件停止超时或失败
记录告警并继续不阻塞后续插件停止
``PAUSED`` 状态插件同样需要调用 ``onStop`` 释放资源否则其资源
无法被清理
"""
started_plugins = self._plugin_registry.listPluginsByState(LifecycleState.STARTED)
paused_plugins = self._plugin_registry.listPluginsByState(LifecycleState.PAUSED)
plugins_to_stop = started_plugins + paused_plugins
if not plugins_to_stop:
await self._logger.info("无 STARTED/PAUSED 状态插件可停止")
return
# 计算反向拓扑序(被依赖者后停止)
manifests = [pm.manifest for pm in plugins_to_stop]
try:
sorted_manifests = self._resolver.resolve(manifests)
stop_order: list[ChannelManifest] = list(reversed(sorted_manifests))
except Exception as e:
# 解析失败常见原因:
# 1. 依赖项不在待停止列表中(如依赖插件已 FAILED不在
# STARTED/PAUSED 列表中)→ resolve 抛 ValidationError
# 2. 清单间存在循环依赖 → resolve 抛 RuleViolationError
# 回退为列表反转序:无法保证严格拓扑序,但保证所有插件都会被停止
await self._logger.warn(
f"插件依赖解析失败,回退为列表反转序(可能原因:依赖项不在待停止列表中或存在循环依赖),"
f"无法保证严格拓扑序但所有插件都会被停止: {e}",
error=str(e),
)
stop_order = list(reversed(manifests))
await self._logger.info(
f"开始停止 {len(stop_order)} 个插件(反向依赖拓扑序)",
)
for manifest in stop_order:
await self._stopSinglePlugin(manifest)
async def _stopSinglePlugin(self, manifest: ChannelManifest) -> None:
"""停止单个插件,超时或失败记录告警并继续。
Args:
manifest: 渠道清单
"""
try:
result: LifecycleResult = await asyncio.wait_for(
self._plugin_lifecycle.stop(manifest.id),
timeout=self._STOP_PLUGIN_TIMEOUT,
)
# 将内置 asyncio.TimeoutError 翻译为契约 OperationTimeoutError 记录结构化告警
# (硬约束:所有模块必须使用自定义异常类而非原生异常)
except TimeoutError:
await self._handleStepTimeout(
f"插件停止: plugin_id={manifest.id}",
self._STOP_PLUGIN_TIMEOUT,
)
return
except Exception as e:
await self._logger.warn(
f"插件停止异常,强制中止: plugin_id={manifest.id}, error={e}",
plugin_id=manifest.id,
error_type=type(e).__name__,
error=str(e),
)
return
if result.state == "failed":
await self._logger.warn(
f"插件停止失败: plugin_id={manifest.id}, error={result.error}",
plugin_id=manifest.id,
error=result.error or "",
)
else:
await self._logger.info(
f"插件已停止: plugin_id={manifest.id}",
plugin_id=manifest.id,
)
async def _closeDrivenAdapters(self) -> None:
"""关闭被驱动适配器。
遍历所有已加载插件注册的 ``DrivenAdapters``调用 ``close()``
释放数据库会话连接池等资源单个适配器关闭异常不中断其他
适配器关闭记录告警后继续超时强制进入下一步并告警
"""
try:
await asyncio.wait_for(
self._closeAllDrivenAdapters(),
timeout=self._CLOSE_DRIVEN_TIMEOUT,
)
# 将内置 asyncio.TimeoutError 翻译为契约 OperationTimeoutError 记录结构化告警
# (硬约束:所有模块必须使用自定义异常类而非原生异常)
except TimeoutError:
await self._handleStepTimeout("关闭被驱动适配器", self._CLOSE_DRIVEN_TIMEOUT)
except Exception as e:
await self._logger.warn(
f"关闭被驱动适配器异常,强制进入下一步: {e}",
error_type=type(e).__name__,
error=str(e),
)
async def _closeAllDrivenAdapters(self) -> None:
"""关闭所有被驱动适配器,按 §15.2 顺序先应用级再插件级INF-017
应用级被驱动适配器``cache_port`` / ``persistence_port`` /
``queue_port`` / ``config_port`` ``HostBootstrap`` 构造时创建
跨请求共享需在关停时显式关闭以释放 Redis 连接DB 会话等资源
插件被驱动适配器由 ``PluginRegistry`` 管理通过
``listPluginAdapters`` 遍历关闭
单个适配器关闭异常记录告警并继续确保后续适配器仍能关闭避免
资源泄漏INV-9 回退路径完整性
"""
# 1. 关闭应用级被驱动适配器cache / persistence / queue / config
for adapter in self._app_driven_adapters:
try:
if not isinstance(adapter, CloseablePort):
await self._logger.info(
"应用级被驱动适配器未实现 CloseablePort跳过",
adapter_type=type(adapter).__name__,
)
continue
await adapter.aclose()
except Exception as exc:
await self._logger.warn(
"shutdown: close app driven adapter failed, continue",
adapter_type=type(adapter).__name__,
error_type=type(exc).__name__,
error=str(exc),
)
continue
await self._logger.info(
"应用级被驱动适配器已关闭",
adapter_type=type(adapter).__name__,
)
# 2. 关闭插件被驱动适配器(保持原有逻辑)
adapters_list = self._plugin_registry.listPluginAdapters()
for channel_type, adapters in adapters_list:
try:
await adapters.close()
await self._logger.info(
f"被驱动适配器已关闭: channel_type={channel_type}",
channel_type=channel_type,
)
except Exception as e:
await self._logger.warn(
f"关闭被驱动适配器失败,继续关闭其他: channel_type={channel_type}, error={e}",
channel_type=channel_type,
error=str(e),
)
if not adapters_list:
await self._logger.info("无被驱动适配器需要关闭")
async def _releaseCoreResources(self) -> None:
"""释放领域核心资源。
按启动的反向顺序取消后台任务传输引擎管理器 / 失败插件重载
记录核心资源释放状态超时强制进入下一步并告警
"""
try:
await asyncio.wait_for(
self._releaseCoreResourcesInternal(),
timeout=self._RELEASE_CORE_TIMEOUT,
)
# 将内置 asyncio.TimeoutError 翻译为契约 OperationTimeoutError 记录结构化告警
# (硬约束:所有模块必须使用自定义异常类而非原生异常)
except TimeoutError:
await self._handleStepTimeout("释放核心资源", self._RELEASE_CORE_TIMEOUT)
except Exception as e:
await self._logger.warn(
f"释放核心资源异常,强制进入下一步: {e}",
error_type=type(e).__name__,
error=str(e),
)
async def _releaseCoreResourcesInternal(self) -> None:
"""释放核心资源内部实现。
惰性读取 ``host_bootstrap`` 上的后台任务引用``bootstrap()`` 执行
后才填充 **启动的反向顺序** 取消并等待未完成的任务以释放资源
后启动的先取消符合"关停 = 启动的反向"原则``HostShutdown``
可能在 ``bootstrap()`` 之前创建因此必须在关停时而非构造时读取
任务引用
启动顺序plugin_reload transport_manager
关停顺序transport_manager plugin_reload
"""
# 按启动的反向顺序取消后台任务(后启动的先取消)
await self._cancelBackgroundTask(
self._host_bootstrap.transportManagerTask,
"传输引擎管理器任务",
)
await self._cancelBackgroundTask(
self._host_bootstrap.pluginReloadTask,
"失败插件重载任务FR-36",
)
await self._logger.info("领域核心资源已释放")
async def _cancelBackgroundTask(
self,
task: asyncio.Task[None] | None,
name: str,
) -> None:
"""取消并等待单个后台任务完成。
若任务为 ``None````bootstrap()`` 未执行到对应阶段或已完成
直接跳过取消后等待任务真正终止``CancelledError`` 视为正常
取消结果其余异常记录告警后继续
Args:
task: 后台任务引用可能为 ``None``
name: 任务名称用于日志标识
"""
if task is None or task.done():
return
task.cancel()
try:
await task
except asyncio.CancelledError:
pass
except Exception as e:
await self._logger.warn(
f"{name}取消异常: {e}",
error=str(e),
)
await self._logger.info(f"{name}已取消")