多彩编程 多彩编程MZPH · CODE BLOG
ARTICLE DETAIL

文章详情

深耕前端与后端开发技术的一线实战笔记与踩坑复盘。

本地部署 Qwen-Image 文生图:完整流程与避坑指南

本地部署 Qwen-Image 文生图:完整流程与避坑指南 目录环境安装推理代码封装server环境安装pip install githttps://github.com/huggingface/diffusers pip install accelerate推理代码import torch from diffusers import QwenImage21Pipeline pipe QwenImage21Pipeline.from_pretrained( /data/feature/0lbg/models/Qwen_Qwen-Image-2.1, torch_dtypetorch.bfloat16 ) # 关键用 offload 替代 .to(cuda) pipe.enable_model_cpu_offload() # 关键VAE 解码阶段显存不够必须开 tiling pipe.vae.enable_tiling() image pipe( promptA neon shop sign that reads QWEN IMAGE 2.1, rainy night, reflections on wet pavement, width2048, height2048, num_inference_steps40, generatortorch.Generator(cuda).manual_seed(42), ).images[0] image.save(t2i_example.png)封装serverimport io import base64 import time import uuid from contextlib import asynccontextmanager import torch from fastapi import FastAPI, HTTPException from fastapi.responses import Response from pydantic import BaseModel, Field from diffusers import QwenImage21Pipeline import uvicorn # ---------- 全局 pipeline ---------- pipe None asynccontextmanager async def lifespan(app: FastAPI): global pipe print(Loading Qwen-Image-2.1 pipeline ...) t0 time.time() pipe QwenImage21Pipeline.from_pretrained( /data/feature/0lbg/models/Qwen_Qwen-Image-2.1, torch_dtypetorch.bfloat16, ) pipe.enable_model_cpu_offload() pipe.vae.enable_tiling() print(fPipeline loaded in {time.time() - t0:.1f}s) yield # 关闭时释放 del pipe torch.cuda.empty_cache() app FastAPI(titleQwen-Image-2.1 T2I Server, lifespanlifespan) # ---------- 请求体 ---------- class T2IRequest(BaseModel): prompt: str Field(..., description正向提示词) negative_prompt: str Field(, description负向提示词) width: int Field(2048, ge256, le2048) height: int Field(2048, ge256, le2048) num_inference_steps: int Field(40, ge1, le100) true_cfg_scale: float Field(4.0, ge0.0, le20.0) seed: int | None Field(None, description不传则随机) # ---------- 接口 ---------- app.get(/health) def health(): return {status: ok, model_loaded: pipe is not None} app.post(/generate) def generate(req: T2IRequest): if pipe is None: raise HTTPException(status_code503, detailModel not loaded yet) # seed不传则随机 if req.seed is None: seed torch.randint(0, 2**31 - 1, (1,)).item() else: seed req.seed generator torch.Generator(cuda).manual_seed(seed) try: image pipe( promptreq.prompt, negative_promptreq.negative_prompt or None, widthreq.width, heightreq.height, num_inference_stepsreq.num_inference_steps, true_cfg_scalereq.true_cfg_scale, generatorgenerator, ).images[0] except torch.cuda.OutOfMemoryError as e: torch.cuda.empty_cache() raise HTTPException(status_code507, detailfCUDA OOM: {e}) # 返回 PNG 二进制 buf io.BytesIO() image.save(buf, formatPNG) buf.seek(0) return Response( contentbuf.getvalue(), media_typeimage/png, headers{ X-Seed: str(seed), Content-Disposition: fattachment; filename{uuid.uuid4().hex}.png, }, ) app.post(/generate_b64) def generate_b64(req: T2IRequest): 返回 base64方便前端直接展示。 if pipe is None: raise HTTPException(status_code503, detailModel not loaded yet) seed req.seed if req.seed is not None else torch.randint(0, 2**31 - 1, (1,)).item() generator torch.Generator(cuda).manual_seed(seed) try: image pipe( promptreq.prompt, negative_promptreq.negative_prompt or None, widthreq.width, heightreq.height, num_inference_stepsreq.num_inference_steps, true_cfg_scalereq.true_cfg_scale, generatorgenerator, ).images[0] except torch.cuda.OutOfMemoryError as e: torch.cuda.empty_cache() raise HTTPException(status_code507, detailfCUDA OOM: {e}) buf io.BytesIO() image.save(buf, formatPNG) b64 base64.b64encode(buf.getvalue()).decode(utf-8) return {seed: seed, format: png, image_base64: b64} if __name__ __main__: uvicorn.run(app, host0.0.0.0, port7999)
返回列表