目录
环境安装:
bash
pip install git+https://github.com/huggingface/diffusers
pip install accelerate
推理代码:
python
import torch
from diffusers import QwenImage21Pipeline
pipe = QwenImage21Pipeline.from_pretrained(
"/data/feature/0lbg/models/Qwen_Qwen-Image-2.1",
torch_dtype=torch.bfloat16
)
# 关键:用 offload 替代 .to("cuda")
pipe.enable_model_cpu_offload()
# 关键:VAE 解码阶段显存不够,必须开 tiling
pipe.vae.enable_tiling()
image = pipe(
prompt='A neon shop sign that reads "QWEN IMAGE 2.1", rainy night, reflections on wet pavement',
width=2048, height=2048,
num_inference_steps=40,
generator=torch.Generator("cuda").manual_seed(42),
).images[0]
image.save("t2i_example.png")
封装server:
python
import io
import base64
import time
import uuid
from contextlib import asynccontextmanager
import torch
from fastapi import FastAPI, HTTPException
from fastapi.responses import Response
from pydantic import BaseModel, Field
from diffusers import QwenImage21Pipeline
import uvicorn
# ---------- 全局 pipeline ----------
pipe = None
@asynccontextmanager
async def lifespan(app: FastAPI):
global pipe
print("Loading Qwen-Image-2.1 pipeline ...")
t0 = time.time()
pipe = QwenImage21Pipeline.from_pretrained(
"/data/feature/0lbg/models/Qwen_Qwen-Image-2.1",
torch_dtype=torch.bfloat16,
)
pipe.enable_model_cpu_offload()
pipe.vae.enable_tiling()
print(f"Pipeline loaded in {time.time() - t0:.1f}s")
yield
# 关闭时释放
del pipe
torch.cuda.empty_cache()
app = FastAPI(title="Qwen-Image-2.1 T2I Server", lifespan=lifespan)
# ---------- 请求体 ----------
class T2IRequest(BaseModel):
prompt: str = Field(..., description="正向提示词")
negative_prompt: str = Field("", description="负向提示词")
width: int = Field(2048, ge=256, le=2048)
height: int = Field(2048, ge=256, le=2048)
num_inference_steps: int = Field(40, ge=1, le=100)
true_cfg_scale: float = Field(4.0, ge=0.0, le=20.0)
seed: int | None = Field(None, description="不传则随机")
# ---------- 接口 ----------
@app.get("/health")
def health():
return {"status": "ok", "model_loaded": pipe is not None}
@app.post("/generate")
def generate(req: T2IRequest):
if pipe is None:
raise HTTPException(status_code=503, detail="Model not loaded yet")
# seed:不传则随机
if req.seed is None:
seed = torch.randint(0, 2**31 - 1, (1,)).item()
else:
seed = req.seed
generator = torch.Generator("cuda").manual_seed(seed)
try:
image = pipe(
prompt=req.prompt,
negative_prompt=req.negative_prompt or None,
width=req.width,
height=req.height,
num_inference_steps=req.num_inference_steps,
true_cfg_scale=req.true_cfg_scale,
generator=generator,
).images[0]
except torch.cuda.OutOfMemoryError as e:
torch.cuda.empty_cache()
raise HTTPException(status_code=507, detail=f"CUDA OOM: {e}")
# 返回 PNG 二进制
buf = io.BytesIO()
image.save(buf, format="PNG")
buf.seek(0)
return Response(
content=buf.getvalue(),
media_type="image/png",
headers={
"X-Seed": str(seed),
"Content-Disposition": f'attachment; filename="{uuid.uuid4().hex}.png"',
},
)
@app.post("/generate_b64")
def generate_b64(req: T2IRequest):
"""返回 base64,方便前端直接展示。"""
if pipe is None:
raise HTTPException(status_code=503, detail="Model not loaded yet")
seed = req.seed if req.seed is not None else torch.randint(0, 2**31 - 1, (1,)).item()
generator = torch.Generator("cuda").manual_seed(seed)
try:
image = pipe(
prompt=req.prompt,
negative_prompt=req.negative_prompt or None,
width=req.width,
height=req.height,
num_inference_steps=req.num_inference_steps,
true_cfg_scale=req.true_cfg_scale,
generator=generator,
).images[0]
except torch.cuda.OutOfMemoryError as e:
torch.cuda.empty_cache()
raise HTTPException(status_code=507, detail=f"CUDA OOM: {e}")
buf = io.BytesIO()
image.save(buf, format="PNG")
b64 = base64.b64encode(buf.getvalue()).decode("utf-8")
return {"seed": seed, "format": "png", "image_base64": b64}
if __name__ == "__main__":
uvicorn.run(app, host="0.0.0.0", port=7999)