# -*- coding: utf-8 -*- """iAOP 推理服务入口(部署用最小 HTTP 服务)。 对齐 `deploy/k8s/helm/iaop` Chart 的部署契约: - 监听 :8000(`service.port`); - 启动时读取 `/etc/iaop/backends.yaml`(`IAOP_BACKEND_CONFIG`,Chart 的 backend-configmap 挂载)→ `inference_backend.factory.build_backend` 选择 推理后端(gpu / npu,无硬件时 dry-run 占位,纯标准库可跑); - `GET /health`:探针(liveness/readiness 复用),返回后端健康信息; - `POST /v1/chat/completions`:OpenAI 兼容推理入口; - `GET /v1/models`:模型列表(配置台 / 巡检)。 依赖仅 PyYAML(解析 backends.yaml),零第三方框架。 """ from __future__ import annotations import json import os import sys from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer try: import yaml except ImportError: # pragma: no cover yaml = None # 允许直接以脚本方式运行(python /app/serve.py;inference-backend 为同目录子包) sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) from inference_backend.factory import build_backend # noqa: E402 from inference_backend.base import InferRequest # noqa: E402 PORT = int(os.environ.get("IAOP_PORT", "8000")) CONFIG_PATH = os.environ.get( "IAOP_BACKEND_CONFIG", "/etc/iaop/backends.yaml") BACKEND = None def _load_config(path: str) -> dict: if yaml is None: raise RuntimeError("缺少 PyYAML 依赖,无法解析 backends.yaml") with open(path, encoding="utf-8") as f: return yaml.safe_load(f) or {} def _init_backend() -> None: global BACKEND cfg = _load_config(CONFIG_PATH) infer = cfg.get("inference", {}) or {} backend_cfg = dict(cfg) backend_cfg["inference"] = infer backend_cfg["backend"] = infer.get("backend") or cfg.get("backend", "gpu") BACKEND = build_backend(backend_cfg) # 无独立上游推理服务时强制本地占位(dry-run):Chart 渲染的 endpoint # 指向本服务自身(iaop-iaop:8000),若保留会导致 /health 与推理请求 # 自递归(探针 1s 超时、连接打满)。清空 endpoint 后所有网络调用快速 # 失败并进入 _do_infer / health 兜底,保证探针与推理契约稳定。 BACKEND.endpoint = "" # 启动时加载模型。无上游时 load_model 仅维护本地状态。 try: BACKEND.load_model(infer.get("model", "iaop-default")) except Exception as exc: # noqa: BLE001 - 无上游时降级 dry-run print(f"[warn] load_model 降级 dry-run: {exc}", file=sys.stderr) def _do_infer(prompt: str, max_tokens: int, temperature: float, model: str) -> dict: """调用后端推理;无上游(endpoint 不可达)时返回 dry-run 占位。""" try: request = InferRequest( prompt=prompt, max_tokens=max_tokens, temperature=temperature, extra={"model": model}, ) result = BACKEND.infer(request) return {"text": result.text, "meta": result.meta, "backend": result.backend, "latency_ms": result.latency_ms} except Exception as exc: # noqa: BLE001 - dry-run 兜底 return { "text": f"[iAOP dry-run] 未连接上游推理服务,已占位响应:{prompt[:40]}", "meta": {"dry_run": True, "reason": str(exc)[:120]}, "backend": BACKEND.backend_name, "latency_ms": 0.0, } class Handler(BaseHTTPRequestHandler): server_version = "iAOPInference/1.0" # -- 工具 ------------------------------------------------------------ def _json(self, code: int, payload: dict) -> None: body = json.dumps(payload, ensure_ascii=False).encode("utf-8") self.send_response(code) self.send_header("Content-Type", "application/json; charset=utf-8") self.send_header("Content-Length", str(len(body))) self.end_headers() self.wfile.write(body) def _read_body(self) -> dict: length = int(self.headers.get("Content-Length", "0") or "0") raw = self.rfile.read(length) if length else b"{}" try: return json.loads(raw.decode("utf-8") or "{}") except json.JSONDecodeError: return {} def log_message(self, fmt, *args): # noqa: D401 - 精简访问日志 sys.stderr.write("%s %s\n" % (self.address_string(), fmt % args)) # -- 路由 ------------------------------------------------------------ def do_GET(self): # noqa: N802 - http.server 命名 if self.path == "/health": try: h = BACKEND.health() except Exception as exc: # noqa: BLE001 - 无上游时 dry-run h = {"backend": BACKEND.backend_name, "status": "dry-run", "reason": str(exc)[:120]} self._json(200, {"status": "ok", **h}) return if self.path == "/v1/models": self._json(200, {"models": [BACKEND.model], "backend": BACKEND.backend_name}) return self._json(404, {"error": "not found", "path": self.path}) def do_POST(self): # noqa: N802 - http.server 命名 if self.path == "/v1/chat/completions": body = self._read_body() prompt = (body.get("prompt") or (body.get("messages") or [{}])[-1].get("content", "")) result = _do_infer( prompt=prompt or "", max_tokens=body.get("max_tokens", 1024), temperature=body.get("temperature", 0.1), model=body.get("model", ""), ) resp = { "id": "iaop-%08d" % (abs(hash(result["text"])) % 10**8), "object": "chat.completion", "choices": [{ "index": 0, "message": {"role": "assistant", "content": result["text"]}, "finish_reason": "stop", }], "model": body.get("model", BACKEND.model), "meta": result["meta"], } self._json(200, resp) return self._json(404, {"error": "not found", "path": self.path}) def main() -> int: try: _init_backend() except Exception as exc: # noqa: BLE001 - 启动失败即退出,让 K8s 重启 print(f"后端初始化失败: {exc}", file=sys.stderr) return 1 server = ThreadingHTTPServer(("0.0.0.0", PORT), Handler) print(f"iAOP inference listening on :{PORT} (config={CONFIG_PATH})", flush=True) server.serve_forever() return 0 if __name__ == "__main__": sys.exit(main())