Files

216 lines
8.5 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
"""iAOP 推理服务入口(部署用最小 HTTP 服务)。
对齐 `deploy/k8s/helm/iaop` Chart 的部署契约:
- 监听 :8000(`service.port`);
- 启动时读取 `/etc/iaop/backends.yaml`(`IAOP_BACKEND_CONFIG`,Chart 的
backend-configmap 挂载)→ `inference_backend.factory.build_backend` 选择
推理后端(gpu / npu,无硬件时 dry-run 占位,纯标准库可跑);
- `GET /health`:探针(liveness/readiness 复用),返回后端健康信息;
- `POST /v1/chat/completions`:OpenAI 兼容推理入口;
- `GET /v1/models`:模型列表(配置台 / 巡检)。
依赖仅 PyYAML(解析 backends.yaml),零第三方框架。
"""
from __future__ import annotations
import json
import os
import sys
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
try:
import yaml
except ImportError: # pragma: no cover
yaml = None
# 允许直接以脚本方式运行(python /app/serve.py;inference-backend 为同目录子包)
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from inference_backend.factory import build_backend # noqa: E402
from inference_backend.base import InferRequest # noqa: E402
PORT = int(os.environ.get("IAOP_PORT", "8000"))
CONFIG_PATH = os.environ.get(
"IAOP_BACKEND_CONFIG", "/etc/iaop/backends.yaml")
BACKEND = None
def _load_config(path: str) -> dict:
if yaml is None:
raise RuntimeError("缺少 PyYAML 依赖,无法解析 backends.yaml")
with open(path, encoding="utf-8") as f:
return yaml.safe_load(f) or {}
def _init_backend() -> None:
global BACKEND
cfg = _load_config(CONFIG_PATH)
infer = cfg.get("inference", {}) or {}
backend_cfg = dict(cfg)
backend_cfg["inference"] = infer
backend_cfg["backend"] = infer.get("backend") or cfg.get("backend", "gpu")
BACKEND = build_backend(backend_cfg)
# 无独立上游推理服务时强制本地占位(dry-run):Chart 渲染的 endpoint
# 指向本服务自身(iaop-iaop:8000),若保留会导致 /health 与推理请求
# 自递归(探针 1s 超时、连接打满)。清空 endpoint 后所有网络调用快速
# 失败并进入 _do_infer / health 兜底,保证探针与推理契约稳定。
BACKEND.endpoint = ""
# 启动时加载模型。无上游时 load_model 仅维护本地状态。
try:
BACKEND.load_model(infer.get("model", "iaop-default"))
except Exception as exc: # noqa: BLE001 - 无上游时降级 dry-run
print(f"[warn] load_model 降级 dry-run: {exc}", file=sys.stderr)
def _do_infer(prompt: str, max_tokens: int, temperature: float,
model: str) -> dict:
"""调用后端推理;无上游(endpoint 不可达)时返回 dry-run 占位。"""
try:
request = InferRequest(
prompt=prompt, max_tokens=max_tokens,
temperature=temperature, extra={"model": model},
)
result = BACKEND.infer(request)
return {"text": result.text, "meta": result.meta,
"backend": result.backend, "latency_ms": result.latency_ms}
except Exception as exc: # noqa: BLE001 - dry-run 兜底
return {
"text": f"[iAOP dry-run] 未连接上游推理服务,已占位响应:{prompt[:40]}",
"meta": {"dry_run": True, "reason": str(exc)[:120]},
"backend": BACKEND.backend_name, "latency_ms": 0.0,
}
class Handler(BaseHTTPRequestHandler):
server_version = "iAOPInference/1.0"
# -- 工具 ------------------------------------------------------------
def _json(self, code: int, payload: dict) -> None:
body = json.dumps(payload, ensure_ascii=False).encode("utf-8")
self.send_response(code)
self.send_header("Content-Type", "application/json; charset=utf-8")
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def _read_body(self) -> dict:
length = int(self.headers.get("Content-Length", "0") or "0")
raw = self.rfile.read(length) if length else b"{}"
try:
return json.loads(raw.decode("utf-8") or "{}")
except json.JSONDecodeError:
return {}
def log_message(self, fmt, *args): # noqa: D401 - 精简访问日志
sys.stderr.write("%s %s\n" % (self.address_string(), fmt % args))
# -- 路由 ------------------------------------------------------------
def do_GET(self): # noqa: N802 - http.server 命名
if self.path == "/health":
try:
h = BACKEND.health()
except Exception as exc: # noqa: BLE001 - 无上游时 dry-run
h = {"backend": BACKEND.backend_name,
"status": "dry-run", "reason": str(exc)[:120]}
self._json(200, {"status": "ok", **h})
return
if self.path == "/v1/models":
self._json(200, {"models": [BACKEND.model],
"backend": BACKEND.backend_name})
return
if self.path in ("/", "/index.html"):
self._send_index()
return
self._json(404, {"error": "not found", "path": self.path})
def _send_index(self) -> None:
"""根路径欢迎页:列出可用接口,便于浏览器直接访问。"""
try:
h = BACKEND.health()
health_s = json.dumps(h, ensure_ascii=False)
except Exception as exc: # noqa: BLE001
health_s = json.dumps({"status": "dry-run",
"reason": str(exc)[:80]},
ensure_ascii=False)
html = f"""<!DOCTYPE html>
<html lang="zh-CN">
<head>
<meta charset="utf-8">
<title>iAOP Inference Service</title>
<style>
body {{ font-family: "Microsoft YaHei", sans-serif; margin: 40px auto;
max-width: 720px; padding: 0 16px; color: #222; }}
h1 {{ color: #0b5; }}
code {{ background: #f4f4f4; padding: 2px 6px; border-radius: 4px; }}
li {{ margin: 8px 0; }}
.health {{ background: #eef; padding: 10px; border-radius: 6px;
word-break: break-all; }}
</style>
</head>
<body>
<h1>iAOP Inference Service</h1>
<p>iAOP-Core 推理服务({BACKEND.backend_name} 后端 · {BACKEND.model})。
当前为 dry-run 占位模式(未接入真实推理硬件/模型服务)。</p>
<h2>可用接口</h2>
<ul>
<li><a href="/health"><code>GET /health</code></a> — 健康巡检</li>
<li><a href="/v1/models"><code>GET /v1/models</code></a> — 模型信息</li>
<li><code>POST /v1/chat/completions</code> — OpenAI 兼容推理
<br>body: <code>{"{model, prompt, max_tokens}"}</code></li>
</ul>
<h2>当前健康状态</h2>
<div class="health">{health_s}</div>
</body>
</html>"""
body = html.encode("utf-8")
self.send_response(200)
self.send_header("Content-Type", "text/html; charset=utf-8")
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def do_POST(self): # noqa: N802 - http.server 命名
if self.path == "/v1/chat/completions":
body = self._read_body()
prompt = (body.get("prompt")
or (body.get("messages") or [{}])[-1].get("content", ""))
result = _do_infer(
prompt=prompt or "",
max_tokens=body.get("max_tokens", 1024),
temperature=body.get("temperature", 0.1),
model=body.get("model", ""),
)
resp = {
"id": "iaop-%08d" % (abs(hash(result["text"])) % 10**8),
"object": "chat.completion",
"choices": [{
"index": 0,
"message": {"role": "assistant", "content": result["text"]},
"finish_reason": "stop",
}],
"model": body.get("model", BACKEND.model),
"meta": result["meta"],
}
self._json(200, resp)
return
self._json(404, {"error": "not found", "path": self.path})
def main() -> int:
try:
_init_backend()
except Exception as exc: # noqa: BLE001 - 启动失败即退出,让 K8s 重启
print(f"后端初始化失败: {exc}", file=sys.stderr)
return 1
server = ThreadingHTTPServer(("0.0.0.0", PORT), Handler)
print(f"iAOP inference listening on :{PORT} (config={CONFIG_PATH})",
flush=True)
server.serve_forever()
return 0
if __name__ == "__main__":
sys.exit(main())