2026-08-05 08:59:45 +08:00
|
|
|
|
# -*- coding: utf-8 -*-
|
|
|
|
|
|
"""iAOP 推理服务入口(部署用最小 HTTP 服务)。
|
|
|
|
|
|
|
|
|
|
|
|
对齐 `deploy/k8s/helm/iaop` Chart 的部署契约:
|
|
|
|
|
|
- 监听 :8000(`service.port`);
|
|
|
|
|
|
- 启动时读取 `/etc/iaop/backends.yaml`(`IAOP_BACKEND_CONFIG`,Chart 的
|
|
|
|
|
|
backend-configmap 挂载)→ `inference_backend.factory.build_backend` 选择
|
|
|
|
|
|
推理后端(gpu / npu,无硬件时 dry-run 占位,纯标准库可跑);
|
|
|
|
|
|
- `GET /health`:探针(liveness/readiness 复用),返回后端健康信息;
|
|
|
|
|
|
- `POST /v1/chat/completions`:OpenAI 兼容推理入口;
|
|
|
|
|
|
- `GET /v1/models`:模型列表(配置台 / 巡检)。
|
|
|
|
|
|
|
|
|
|
|
|
依赖仅 PyYAML(解析 backends.yaml),零第三方框架。
|
|
|
|
|
|
"""
|
|
|
|
|
|
from __future__ import annotations
|
|
|
|
|
|
|
|
|
|
|
|
import json
|
|
|
|
|
|
import os
|
|
|
|
|
|
import sys
|
|
|
|
|
|
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
|
|
|
|
|
|
|
|
|
|
try:
|
|
|
|
|
|
import yaml
|
|
|
|
|
|
except ImportError: # pragma: no cover
|
|
|
|
|
|
yaml = None
|
|
|
|
|
|
|
|
|
|
|
|
# 允许直接以脚本方式运行(python /app/serve.py;inference-backend 为同目录子包)
|
|
|
|
|
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
|
|
|
|
|
|
|
|
|
|
from inference_backend.factory import build_backend # noqa: E402
|
|
|
|
|
|
from inference_backend.base import InferRequest # noqa: E402
|
|
|
|
|
|
|
|
|
|
|
|
PORT = int(os.environ.get("IAOP_PORT", "8000"))
|
|
|
|
|
|
CONFIG_PATH = os.environ.get(
|
|
|
|
|
|
"IAOP_BACKEND_CONFIG", "/etc/iaop/backends.yaml")
|
|
|
|
|
|
|
|
|
|
|
|
BACKEND = None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _load_config(path: str) -> dict:
|
|
|
|
|
|
if yaml is None:
|
|
|
|
|
|
raise RuntimeError("缺少 PyYAML 依赖,无法解析 backends.yaml")
|
|
|
|
|
|
with open(path, encoding="utf-8") as f:
|
|
|
|
|
|
return yaml.safe_load(f) or {}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _init_backend() -> None:
|
|
|
|
|
|
global BACKEND
|
|
|
|
|
|
cfg = _load_config(CONFIG_PATH)
|
|
|
|
|
|
infer = cfg.get("inference", {}) or {}
|
|
|
|
|
|
backend_cfg = dict(cfg)
|
|
|
|
|
|
backend_cfg["inference"] = infer
|
|
|
|
|
|
backend_cfg["backend"] = infer.get("backend") or cfg.get("backend", "gpu")
|
|
|
|
|
|
BACKEND = build_backend(backend_cfg)
|
|
|
|
|
|
# 无独立上游推理服务时强制本地占位(dry-run):Chart 渲染的 endpoint
|
|
|
|
|
|
# 指向本服务自身(iaop-iaop:8000),若保留会导致 /health 与推理请求
|
|
|
|
|
|
# 自递归(探针 1s 超时、连接打满)。清空 endpoint 后所有网络调用快速
|
|
|
|
|
|
# 失败并进入 _do_infer / health 兜底,保证探针与推理契约稳定。
|
|
|
|
|
|
BACKEND.endpoint = ""
|
|
|
|
|
|
# 启动时加载模型。无上游时 load_model 仅维护本地状态。
|
|
|
|
|
|
try:
|
|
|
|
|
|
BACKEND.load_model(infer.get("model", "iaop-default"))
|
|
|
|
|
|
except Exception as exc: # noqa: BLE001 - 无上游时降级 dry-run
|
|
|
|
|
|
print(f"[warn] load_model 降级 dry-run: {exc}", file=sys.stderr)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _do_infer(prompt: str, max_tokens: int, temperature: float,
|
|
|
|
|
|
model: str) -> dict:
|
|
|
|
|
|
"""调用后端推理;无上游(endpoint 不可达)时返回 dry-run 占位。"""
|
|
|
|
|
|
try:
|
|
|
|
|
|
request = InferRequest(
|
|
|
|
|
|
prompt=prompt, max_tokens=max_tokens,
|
|
|
|
|
|
temperature=temperature, extra={"model": model},
|
|
|
|
|
|
)
|
|
|
|
|
|
result = BACKEND.infer(request)
|
|
|
|
|
|
return {"text": result.text, "meta": result.meta,
|
|
|
|
|
|
"backend": result.backend, "latency_ms": result.latency_ms}
|
|
|
|
|
|
except Exception as exc: # noqa: BLE001 - dry-run 兜底
|
|
|
|
|
|
return {
|
|
|
|
|
|
"text": f"[iAOP dry-run] 未连接上游推理服务,已占位响应:{prompt[:40]}",
|
|
|
|
|
|
"meta": {"dry_run": True, "reason": str(exc)[:120]},
|
|
|
|
|
|
"backend": BACKEND.backend_name, "latency_ms": 0.0,
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
class Handler(BaseHTTPRequestHandler):
|
|
|
|
|
|
server_version = "iAOPInference/1.0"
|
|
|
|
|
|
|
|
|
|
|
|
# -- 工具 ------------------------------------------------------------
|
|
|
|
|
|
def _json(self, code: int, payload: dict) -> None:
|
|
|
|
|
|
body = json.dumps(payload, ensure_ascii=False).encode("utf-8")
|
|
|
|
|
|
self.send_response(code)
|
|
|
|
|
|
self.send_header("Content-Type", "application/json; charset=utf-8")
|
|
|
|
|
|
self.send_header("Content-Length", str(len(body)))
|
|
|
|
|
|
self.end_headers()
|
|
|
|
|
|
self.wfile.write(body)
|
|
|
|
|
|
|
|
|
|
|
|
def _read_body(self) -> dict:
|
|
|
|
|
|
length = int(self.headers.get("Content-Length", "0") or "0")
|
|
|
|
|
|
raw = self.rfile.read(length) if length else b"{}"
|
|
|
|
|
|
try:
|
|
|
|
|
|
return json.loads(raw.decode("utf-8") or "{}")
|
|
|
|
|
|
except json.JSONDecodeError:
|
|
|
|
|
|
return {}
|
|
|
|
|
|
|
|
|
|
|
|
def log_message(self, fmt, *args): # noqa: D401 - 精简访问日志
|
|
|
|
|
|
sys.stderr.write("%s %s\n" % (self.address_string(), fmt % args))
|
|
|
|
|
|
|
|
|
|
|
|
# -- 路由 ------------------------------------------------------------
|
|
|
|
|
|
def do_GET(self): # noqa: N802 - http.server 命名
|
|
|
|
|
|
if self.path == "/health":
|
|
|
|
|
|
try:
|
|
|
|
|
|
h = BACKEND.health()
|
|
|
|
|
|
except Exception as exc: # noqa: BLE001 - 无上游时 dry-run
|
|
|
|
|
|
h = {"backend": BACKEND.backend_name,
|
|
|
|
|
|
"status": "dry-run", "reason": str(exc)[:120]}
|
|
|
|
|
|
self._json(200, {"status": "ok", **h})
|
|
|
|
|
|
return
|
|
|
|
|
|
if self.path == "/v1/models":
|
|
|
|
|
|
self._json(200, {"models": [BACKEND.model],
|
|
|
|
|
|
"backend": BACKEND.backend_name})
|
|
|
|
|
|
return
|
2026-08-05 09:12:33 +08:00
|
|
|
|
if self.path in ("/", "/index.html"):
|
|
|
|
|
|
self._send_index()
|
|
|
|
|
|
return
|
2026-08-05 08:59:45 +08:00
|
|
|
|
self._json(404, {"error": "not found", "path": self.path})
|
|
|
|
|
|
|
2026-08-05 09:12:33 +08:00
|
|
|
|
def _send_index(self) -> None:
|
|
|
|
|
|
"""根路径欢迎页:列出可用接口,便于浏览器直接访问。"""
|
|
|
|
|
|
try:
|
|
|
|
|
|
h = BACKEND.health()
|
|
|
|
|
|
health_s = json.dumps(h, ensure_ascii=False)
|
|
|
|
|
|
except Exception as exc: # noqa: BLE001
|
|
|
|
|
|
health_s = json.dumps({"status": "dry-run",
|
|
|
|
|
|
"reason": str(exc)[:80]},
|
|
|
|
|
|
ensure_ascii=False)
|
|
|
|
|
|
html = f"""<!DOCTYPE html>
|
|
|
|
|
|
<html lang="zh-CN">
|
|
|
|
|
|
<head>
|
|
|
|
|
|
<meta charset="utf-8">
|
|
|
|
|
|
<title>iAOP Inference Service</title>
|
|
|
|
|
|
<style>
|
|
|
|
|
|
body {{ font-family: "Microsoft YaHei", sans-serif; margin: 40px auto;
|
|
|
|
|
|
max-width: 720px; padding: 0 16px; color: #222; }}
|
|
|
|
|
|
h1 {{ color: #0b5; }}
|
|
|
|
|
|
code {{ background: #f4f4f4; padding: 2px 6px; border-radius: 4px; }}
|
|
|
|
|
|
li {{ margin: 8px 0; }}
|
|
|
|
|
|
.health {{ background: #eef; padding: 10px; border-radius: 6px;
|
|
|
|
|
|
word-break: break-all; }}
|
|
|
|
|
|
</style>
|
|
|
|
|
|
</head>
|
|
|
|
|
|
<body>
|
|
|
|
|
|
<h1>iAOP Inference Service</h1>
|
|
|
|
|
|
<p>iAOP-Core 推理服务({BACKEND.backend_name} 后端 · {BACKEND.model})。
|
|
|
|
|
|
当前为 dry-run 占位模式(未接入真实推理硬件/模型服务)。</p>
|
|
|
|
|
|
<h2>可用接口</h2>
|
|
|
|
|
|
<ul>
|
|
|
|
|
|
<li><a href="/health"><code>GET /health</code></a> — 健康巡检</li>
|
|
|
|
|
|
<li><a href="/v1/models"><code>GET /v1/models</code></a> — 模型信息</li>
|
|
|
|
|
|
<li><code>POST /v1/chat/completions</code> — OpenAI 兼容推理
|
|
|
|
|
|
<br>body: <code>{"{model, prompt, max_tokens}"}</code></li>
|
|
|
|
|
|
</ul>
|
|
|
|
|
|
<h2>当前健康状态</h2>
|
|
|
|
|
|
<div class="health">{health_s}</div>
|
|
|
|
|
|
</body>
|
|
|
|
|
|
</html>"""
|
|
|
|
|
|
body = html.encode("utf-8")
|
|
|
|
|
|
self.send_response(200)
|
|
|
|
|
|
self.send_header("Content-Type", "text/html; charset=utf-8")
|
|
|
|
|
|
self.send_header("Content-Length", str(len(body)))
|
|
|
|
|
|
self.end_headers()
|
|
|
|
|
|
self.wfile.write(body)
|
|
|
|
|
|
|
2026-08-05 08:59:45 +08:00
|
|
|
|
def do_POST(self): # noqa: N802 - http.server 命名
|
|
|
|
|
|
if self.path == "/v1/chat/completions":
|
|
|
|
|
|
body = self._read_body()
|
|
|
|
|
|
prompt = (body.get("prompt")
|
|
|
|
|
|
or (body.get("messages") or [{}])[-1].get("content", ""))
|
|
|
|
|
|
result = _do_infer(
|
|
|
|
|
|
prompt=prompt or "",
|
|
|
|
|
|
max_tokens=body.get("max_tokens", 1024),
|
|
|
|
|
|
temperature=body.get("temperature", 0.1),
|
|
|
|
|
|
model=body.get("model", ""),
|
|
|
|
|
|
)
|
|
|
|
|
|
resp = {
|
|
|
|
|
|
"id": "iaop-%08d" % (abs(hash(result["text"])) % 10**8),
|
|
|
|
|
|
"object": "chat.completion",
|
|
|
|
|
|
"choices": [{
|
|
|
|
|
|
"index": 0,
|
|
|
|
|
|
"message": {"role": "assistant", "content": result["text"]},
|
|
|
|
|
|
"finish_reason": "stop",
|
|
|
|
|
|
}],
|
|
|
|
|
|
"model": body.get("model", BACKEND.model),
|
|
|
|
|
|
"meta": result["meta"],
|
|
|
|
|
|
}
|
|
|
|
|
|
self._json(200, resp)
|
|
|
|
|
|
return
|
|
|
|
|
|
self._json(404, {"error": "not found", "path": self.path})
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def main() -> int:
|
|
|
|
|
|
try:
|
|
|
|
|
|
_init_backend()
|
|
|
|
|
|
except Exception as exc: # noqa: BLE001 - 启动失败即退出,让 K8s 重启
|
|
|
|
|
|
print(f"后端初始化失败: {exc}", file=sys.stderr)
|
|
|
|
|
|
return 1
|
|
|
|
|
|
server = ThreadingHTTPServer(("0.0.0.0", PORT), Handler)
|
|
|
|
|
|
print(f"iAOP inference listening on :{PORT} (config={CONFIG_PATH})",
|
|
|
|
|
|
flush=True)
|
|
|
|
|
|
server.serve_forever()
|
|
|
|
|
|
return 0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
if __name__ == "__main__":
|
|
|
|
|
|
sys.exit(main())
|