本地部署 Qwen-Image 文生图:完整流程与避坑指南

目录

环境安装:

推理代码:

封装server:


环境安装:

bash 复制代码
pip install git+https://github.com/huggingface/diffusers

pip install accelerate

推理代码:

python 复制代码
import torch
from diffusers import QwenImage21Pipeline

pipe = QwenImage21Pipeline.from_pretrained(
    "/data/feature/0lbg/models/Qwen_Qwen-Image-2.1", 
    torch_dtype=torch.bfloat16
)

# 关键:用 offload 替代 .to("cuda")
pipe.enable_model_cpu_offload()

# 关键:VAE 解码阶段显存不够,必须开 tiling
pipe.vae.enable_tiling()

image = pipe(
    prompt='A neon shop sign that reads "QWEN IMAGE 2.1", rainy night, reflections on wet pavement',
    width=2048, height=2048,
    num_inference_steps=40,
    generator=torch.Generator("cuda").manual_seed(42),
).images[0]

image.save("t2i_example.png")

封装server:

python 复制代码
import io
import base64
import time
import uuid
from contextlib import asynccontextmanager

import torch
from fastapi import FastAPI, HTTPException
from fastapi.responses import Response
from pydantic import BaseModel, Field
from diffusers import QwenImage21Pipeline
import uvicorn

# ---------- 全局 pipeline ----------
pipe = None


@asynccontextmanager
async def lifespan(app: FastAPI):
    global pipe
    print("Loading Qwen-Image-2.1 pipeline ...")
    t0 = time.time()
    pipe = QwenImage21Pipeline.from_pretrained(
        "/data/feature/0lbg/models/Qwen_Qwen-Image-2.1",
        torch_dtype=torch.bfloat16,
    )
    pipe.enable_model_cpu_offload()
    pipe.vae.enable_tiling()
    print(f"Pipeline loaded in {time.time() - t0:.1f}s")
    yield
    # 关闭时释放
    del pipe
    torch.cuda.empty_cache()


app = FastAPI(title="Qwen-Image-2.1 T2I Server", lifespan=lifespan)


# ---------- 请求体 ----------
class T2IRequest(BaseModel):
    prompt: str = Field(..., description="正向提示词")
    negative_prompt: str = Field("", description="负向提示词")
    width: int = Field(2048, ge=256, le=2048)
    height: int = Field(2048, ge=256, le=2048)
    num_inference_steps: int = Field(40, ge=1, le=100)
    true_cfg_scale: float = Field(4.0, ge=0.0, le=20.0)
    seed: int | None = Field(None, description="不传则随机")


# ---------- 接口 ----------
@app.get("/health")
def health():
    return {"status": "ok", "model_loaded": pipe is not None}


@app.post("/generate")
def generate(req: T2IRequest):
    if pipe is None:
        raise HTTPException(status_code=503, detail="Model not loaded yet")

    # seed:不传则随机
    if req.seed is None:
        seed = torch.randint(0, 2**31 - 1, (1,)).item()
    else:
        seed = req.seed
    generator = torch.Generator("cuda").manual_seed(seed)

    try:
        image = pipe(
            prompt=req.prompt,
            negative_prompt=req.negative_prompt or None,
            width=req.width,
            height=req.height,
            num_inference_steps=req.num_inference_steps,
            true_cfg_scale=req.true_cfg_scale,
            generator=generator,
        ).images[0]
    except torch.cuda.OutOfMemoryError as e:
        torch.cuda.empty_cache()
        raise HTTPException(status_code=507, detail=f"CUDA OOM: {e}")

    # 返回 PNG 二进制
    buf = io.BytesIO()
    image.save(buf, format="PNG")
    buf.seek(0)

    return Response(
        content=buf.getvalue(),
        media_type="image/png",
        headers={
            "X-Seed": str(seed),
            "Content-Disposition": f'attachment; filename="{uuid.uuid4().hex}.png"',
        },
    )


@app.post("/generate_b64")
def generate_b64(req: T2IRequest):
    """返回 base64,方便前端直接展示。"""
    if pipe is None:
        raise HTTPException(status_code=503, detail="Model not loaded yet")

    seed = req.seed if req.seed is not None else torch.randint(0, 2**31 - 1, (1,)).item()
    generator = torch.Generator("cuda").manual_seed(seed)

    try:
        image = pipe(
            prompt=req.prompt,
            negative_prompt=req.negative_prompt or None,
            width=req.width,
            height=req.height,
            num_inference_steps=req.num_inference_steps,
            true_cfg_scale=req.true_cfg_scale,
            generator=generator,
        ).images[0]
    except torch.cuda.OutOfMemoryError as e:
        torch.cuda.empty_cache()
        raise HTTPException(status_code=507, detail=f"CUDA OOM: {e}")

    buf = io.BytesIO()
    image.save(buf, format="PNG")
    b64 = base64.b64encode(buf.getvalue()).decode("utf-8")
    return {"seed": seed, "format": "png", "image_base64": b64}

if __name__ == "__main__":
    uvicorn.run(app, host="0.0.0.0", port=7999)
相关推荐
梦想不只是梦与想2 小时前
LangGraph的运行时:编译、执行与持久化(三)
大模型·langgraph·checkpointer
leoZ2312 小时前
第 40 篇 AI 团队搭建与角色分工
人工智能·大模型·agent
Alice-YUE2 小时前
A2A 协议详解:Agent 间通信标准、四大核心机制与 MCP 互补
大模型·多智能体·ai agent·mcp·a2a协议
小范的技术工坊2 小时前
大模型同步、异步、流式输出
大模型·结构化
浅安的邂逅17 小时前
20929-OpenAI 一天踩三脚急刹:暂停前沿训练、叫停 Astra、披露越权访问澳政府网站
人工智能·大模型·ai编程·行业动态·ai日报
问天_观心21 小时前
大模型训练与推理优化(二)
人工智能·深度学习·学习·大模型·transformer
小范的技术工坊1 天前
大模型蒸馏
人工智能·算法·数据挖掘·大模型
liferecords1 天前
笔记本硬跑 744B 大模型:GitHub 上的『蜂鸟』把 SSD 当显存用
人工智能·开源·大模型·推理优化
爱喝雪碧的可乐1 天前
CSDN|爆火哑巴AI Jev模型深度实战|技术博客
人工智能·大模型·jev