# -*- coding: utf-8 -*- """一键部署编排引擎(Issue #60 / PRD 5.6「⑥ 部署底座」)。 PRD 5.6 验收:一套 Helm Chart 部署内核 + 模板,GPU/NPU 推理后端可插拔。 本模块把部署落为**可测试的编排引擎**——把部署拆为有序步骤序列 (pre_check → deploy → health_check → post_check),编排器按序执行(模拟, 不真调 k8s/helm),逐步产出步骤日志与状态,失败即触发回滚。 设计要点 -------- 1. **部署即编排序列**(``DeployStep``):每步声明 ``kind``(四类)+ ``name`` + 可执行动作(``action`` 回调,返回 ok/detail);步骤序列即 :class:`DeployPlan`。 - ``pre_check`` 部署前检查(集群可达 / 镜像存在 / values 合法 / 资源配额); - ``deploy`` 部署动作(helm install/upgrade,模拟); - ``health_check``健康检查(探活 /health,对齐 ``deploy/k8s/healthz``); - ``post_check`` 部署后验证(业务接口回归 / 监控接入)。 2. **编排器**(``DeployOrchestrator``):按序执行步骤,任何一步失败即中止并 标记需回滚(``DeployOutcome.needs_rollback``),产出 :class:`DeployReport` (逐步结果 + 总状态 + 可解释 reason)。 3. **健康检查契约**(``HealthCheckContract``):定义健康检查的判定口径 (超时/重试/目标可用率,对齐 ``deploy/k8s/healthz`` 的 99.8% 目标)。 4. **纯标准库**:action 为回调,默认实现模拟成功;真实环境注入真 action 即可。 5. **不真执行 k8s/helm**:编排只产日志与状态,便于离线测试与 CI 集成。 用法:: plan = DeployPlan.default_helm_release(release="iaop", namespace="iaop") report = DeployOrchestrator(plan).execute() if not report.outcome.succeeded: print("需回滚:", report.outcome.reason) """ from __future__ import annotations import time from dataclasses import dataclass, field from enum import Enum from typing import Callable, Dict, List, Optional, Sequence, Tuple #: 健康检查默认超时(秒,对齐 deploy/k8s/healthz 探针口径)。 DEFAULT_HEALTH_TIMEOUT_S = 30.0 #: 健康检查默认重试次数。 DEFAULT_HEALTH_RETRIES = 3 #: 可用率目标(对齐 deploy/k8s/healthz probe_availability 的 99.8%)。 DEFAULT_AVAILABILITY_TARGET = 0.998 class DeployError(ValueError): """部署计划声明/编排错误(步骤序列非法、kind 未知、回调异常等)。""" class DeployStepKind(str, Enum): """部署步骤类型(决定执行顺序与失败后果)。""" PRE_CHECK = "pre_check" # 部署前检查 DEPLOY = "deploy" # 部署动作 HEALTH_CHECK = "health_check" # 健康检查 POST_CHECK = "post_check" # 部署后验证 @property def label(self) -> str: return { DeployStepKind.PRE_CHECK: "部署前检查", DeployStepKind.DEPLOY: "部署动作", DeployStepKind.HEALTH_CHECK: "健康检查", DeployStepKind.POST_CHECK: "部署后验证", }[self] @property def is_checkpoint(self) -> bool: """该步失败是否触发回滚(deploy/health_check 失败必须回滚)。""" return self in (DeployStepKind.DEPLOY, DeployStepKind.HEALTH_CHECK) #: 步骤动作回调:返回 (ok, detail)。 StepAction = Callable[[Dict[str, object]], Tuple[bool, str]] def _default_action(ctx: Dict[str, object]) -> Tuple[bool, str]: """默认动作:模拟成功(真实环境注入真回调)。""" return True, "模拟成功(无 k8s/helm 环境)" @dataclass class DeployStep: """单个部署步骤(声明式)。 Attributes: kind: 步骤类型。 name: 步骤名(可读,如 "helm upgrade")。 action: 可执行动作(ctx → (ok, detail));默认模拟成功。 required: 是否必需(必需步骤失败即中止并标记回滚;非必需失败仅告警)。 detail: 步骤说明(引 PRD/SOP,可解释)。 """ kind: DeployStepKind name: str action: StepAction = field(default=_default_action, repr=False) required: bool = True detail: str = "" def __post_init__(self) -> None: if not self.name: raise DeployError("DeployStep.name 不能为空") if not isinstance(self.kind, DeployStepKind): raise DeployError(f"kind 必须是 DeployStepKind,实际 {type(self.kind)}") @dataclass class HealthCheckContract: """健康检查契约(对齐 deploy/k8s/healthz 的探针口径)。 Attributes: endpoint: 健康端点(如 http://iaop:8000/health)。 timeout_s: 单次探测超时。 retries: 重试次数(窗口内达到目标可用率即通过)。 availability_target: 可用率目标(默认 0.998 = 99.8%)。 """ endpoint: str = "http://iaop:8000/health" timeout_s: float = DEFAULT_HEALTH_TIMEOUT_S retries: int = DEFAULT_HEALTH_RETRIES availability_target: float = DEFAULT_AVAILABILITY_TARGET def __post_init__(self) -> None: if self.timeout_s <= 0: raise DeployError(f"timeout_s 必须 > 0,实际 {self.timeout_s}") if self.retries < 1: raise DeployError(f"retries 必须 ≥ 1,实际 {self.retries}") if not (0.0 < self.availability_target <= 1.0): raise DeployError( f"availability_target 须在 (0,1],实际 {self.availability_target}") @dataclass class DeployPlan: """部署计划:步骤序列 + 发布元信息。 步骤顺序约束:pre_check 必须在 deploy 前;health_check 必须在 deploy 后、 post_check 前。``__post_init__`` 校验该顺序,违例即拒绝(防编排漂移)。 Attributes: name: 计划名(如 "iaop-helm-upgrade")。 release: Helm release 名(如 "iaop")。 namespace: 命名空间。 chart: Chart 引用(如 "deploy/k8s/helm/iaop")。 version: 目标版本(appVersion,如 "v1.0.0")。 steps: 步骤序列(按执行顺序)。 health: 健康检查契约(health_check 步骤引用)。 """ name: str release: str = "iaop" namespace: str = "iaop" chart: str = "deploy/k8s/helm/iaop" version: str = "v1.0.0" steps: List[DeployStep] = field(default_factory=list) health: HealthCheckContract = field(default_factory=HealthCheckContract) def __post_init__(self) -> None: if not self.name: raise DeployError("DeployPlan.name 不能为空") self._validate_order() # ------------------------------------------------------------------ def _validate_order(self) -> None: """校验步骤顺序:pre_check < deploy < health_check < post_check。""" if not self.steps: return # 空计划在 execute() 时报错 order_rank = { DeployStepKind.PRE_CHECK: 0, DeployStepKind.DEPLOY: 1, DeployStepKind.HEALTH_CHECK: 2, DeployStepKind.POST_CHECK: 3, } prev_rank = -1 for s in self.steps: r = order_rank[s.kind] if r < prev_rank: raise DeployError( f"步骤顺序违例:{s.kind.value}({s.name}) 不能排在" f" rank {prev_rank} 之后(pre_check < deploy < " f"health_check < post_check)") prev_rank = r # ------------------------------------------------------------------ @classmethod def default_helm_release( cls, release: str = "iaop", namespace: str = "iaop", version: str = "v1.0.0", backend: str = "gpu", health: Optional[HealthCheckContract] = None, step_actions: Optional[Dict[DeployStepKind, StepAction]] = None, ) -> "DeployPlan": """默认 Helm 发布计划(pre_check → deploy → health_check → post_check)。 Args: release: release 名。 namespace: 命名空间。 version: 目标 appVersion。 backend: 推理后端(gpu|npu,对齐 PRD 5.6 配置点)。 health: 健康检查契约(默认 :class:`HealthCheckContract`)。 step_actions: 各类步骤的动作回调覆盖(默认模拟成功)。 """ actions = step_actions or {} plan = cls( name=f"{release}-helm-upgrade", release=release, namespace=namespace, version=version, health=health or HealthCheckContract(), steps=[ DeployStep( kind=DeployStepKind.PRE_CHECK, name="pre-check-cluster-values", action=actions.get(DeployStepKind.PRE_CHECK, _default_action), detail="部署前检查:集群可达 / 镜像存在 / values 合法 / 资源配额"), DeployStep( kind=DeployStepKind.DEPLOY, name=f"helm-upgrade-{backend}", action=actions.get(DeployStepKind.DEPLOY, _default_action), detail=f"helm upgrade --install {release}(backend={backend},PRD 5.6)"), DeployStep( kind=DeployStepKind.HEALTH_CHECK, name="health-check-probe", action=actions.get(DeployStepKind.HEALTH_CHECK, _default_action), detail="健康检查:探活 /health(目标可用率 99.8%,对齐 healthz)"), DeployStep( kind=DeployStepKind.POST_CHECK, name="post-check-smoke", action=actions.get(DeployStepKind.POST_CHECK, _default_action), required=False, # 烟测非阻断 detail="部署后验证:业务接口烟测 / 监控接入(非阻断)"), ], ) return plan # --------------------------------------------------------------------------- # 执行结果 # --------------------------------------------------------------------------- class DeployStatus(str, Enum): """单步执行状态。""" PENDING = "pending" SUCCESS = "success" FAILED = "failed" SKIPPED = "skipped" @dataclass class DeployStepResult: """单步执行结果。""" step: DeployStep status: DeployStatus = DeployStatus.PENDING detail: str = "" duration_ms: float = 0.0 index: int = 0 @dataclass class DeployOutcome: """部署总结果(可解释:是否成功 + 是否需回滚 + reason)。""" succeeded: bool needs_rollback: bool failed_step: Optional[str] = None reason: str = "" @dataclass class DeployReport: """部署编排报告(逐步结果 + 总状态 + 日志行)。""" plan_name: str results: List[DeployStepResult] = field(default_factory=list) outcome: Optional[DeployOutcome] = None log_lines: List[str] = field(default_factory=list) context: Dict[str, object] = field(default_factory=dict) @property def succeeded(self) -> bool: return self.outcome.succeeded if self.outcome else False def to_dict(self) -> dict: return { "plan_name": self.plan_name, "succeeded": self.succeeded, "needs_rollback": self.outcome.needs_rollback if self.outcome else False, "reason": self.outcome.reason if self.outcome else "", "steps": [ {"index": r.index, "kind": r.step.kind.value, "name": r.step.name, "status": r.status.value, "detail": r.detail, "duration_ms": round(r.duration_ms, 2)} for r in self.results ], "log": list(self.log_lines), } # --------------------------------------------------------------------------- # 编排器 # --------------------------------------------------------------------------- class DeployOrchestrator: """部署编排器:按序执行 DeployPlan 步骤,失败即中止并标记回滚。 Args: plan: 部署计划。 dry_run: 空跑模式(不执行 action,全部标记 SKIPPED;用于计划校验)。 logger: 日志回调(默认 print 到内部 log_lines 缓冲)。 """ def __init__( self, plan: DeployPlan, dry_run: bool = False, logger: Optional[Callable[[str], None]] = None, ) -> None: self.plan = plan self.dry_run = bool(dry_run) self._log: List[str] = [] self._logger = logger # ------------------------------------------------------------------ def _log_line(self, line: str) -> None: self._log.append(line) if self._logger is not None: self._logger(line) # ------------------------------------------------------------------ def execute(self, context: Optional[Dict[str, object]] = None) -> DeployReport: """执行部署计划,返回报告。""" report = DeployReport(plan_name=self.plan.name) ctx: Dict[str, object] = { "release": self.plan.release, "namespace": self.plan.namespace, "chart": self.plan.chart, "version": self.plan.version, "health": { "endpoint": self.plan.health.endpoint, "timeout_s": self.plan.health.timeout_s, "retries": self.plan.health.retries, "availability_target": self.plan.health.availability_target, }, } if context: ctx.update(context) report.context = dict(ctx) if not self.plan.steps: report.outcome = DeployOutcome( succeeded=False, needs_rollback=False, reason="部署计划无步骤(DeployPlan.steps 为空)") self._log_line("[FAIL] 部署计划无步骤,未执行") report.log_lines = list(self._log) return report self._log_line( f"[START] 部署 {self.plan.name}:release={self.plan.release} " f"namespace={self.plan.namespace} version={self.plan.version} " f"steps={len(self.plan.steps)}") failed_result: Optional[DeployStepResult] = None for idx, step in enumerate(self.plan.steps): result = DeployStepResult(step=step, index=idx) if self.dry_run: result.status = DeployStatus.SKIPPED result.detail = "dry-run 跳过" report.results.append(result) self._log_line( f"[SKIP] #{idx} {step.kind.label}/{step.name}(dry-run)") continue self._log_line(f"[RUN] #{idx} {step.kind.label}/{step.name}") started = time.monotonic() try: ok, detail = step.action(ctx) except Exception as exc: # noqa: BLE001 - 任意 action 异常视为失败 ok, detail = False, f"action 异常:{exc}" result.duration_ms = (time.monotonic() - started) * 1000.0 result.detail = detail result.status = DeployStatus.SUCCESS if ok else DeployStatus.FAILED report.results.append(result) tag = "OK" if ok else "FAIL" self._log_line( f"[{tag}] #{idx} {step.name}:{detail} " f"({result.duration_ms:.0f}ms)") if not ok: if step.required: failed_result = result break # 非必需步骤失败:告警但继续 self._log_line( f"[WARN] #{idx} {step.name} 非必需步骤失败,继续编排") report.log_lines = list(self._log) if self.dry_run: report.outcome = DeployOutcome( succeeded=True, needs_rollback=False, reason=f"dry-run:{len(self.plan.steps)} 步全部跳过(计划校验通过)") return report if failed_result is not None: needs_rollback = failed_result.step.kind.is_checkpoint report.outcome = DeployOutcome( succeeded=False, needs_rollback=needs_rollback, failed_step=failed_result.step.name, reason=( f"步骤 #{failed_result.index} {failed_result.step.kind.label}/" f"{failed_result.step.name} 失败:{failed_result.detail}" + (",需回滚到上一稳定版本" if needs_rollback else "(非回滚检查点,无需回滚)"))) else: report.outcome = DeployOutcome( succeeded=True, needs_rollback=False, reason=f"全部 {len(report.results)} 步执行成功,部署完成") return report