问题:接口2 与接口3/5 乱序调用时耗时抖动(最差 15~22s)。两个根因: 1. GPU 24G 常驻 21.4G,Flux-2(3.9G) 无法完全驻留显存,每次采样动态换页, 速度随空闲显存波动(2s~8s); 2. ComfyUI 单队列 FIFO,接口2 排在接口3/5 批量任务后面。 改动: - hairline/comfyui.py: run() 新增 front 参数,/prompt 带 "front": true 插队到队列最前; redraw.py 透传;service.py 接口2 三处调用(女重绘 + 男有/无遮罩)传 front=True, 接口3/5 仍走普通队列。 - add_hair.json / 0716add-hair-api.json: 节点61 CLIPLoader device default→cpu。 qwen CLIP(4G) 不再占显存(文本条件缓存常年命中),ComfyUI 显存 8.8G→4.5G, Flux-2 完全驻留,采样稳定 ~3-5s。代价:换 prompt 后首次请求 CPU 编码 ~11s(一次性)。 - 提示词全局统一为「填充遮罩区域的头发,皮肤加一点磨皮,再加一点美颜」: app.py 4处默认值、service.py _REDRAW_PROMPT、redraw.py _DEFAULT_PROMPT、 4个工作流节点60内置文案、测试页(test_interface2/3/7/12/12_final)、local_test。 任何两个不同 prompt 交替提交都会打爆 CLIP 编码缓存(--cache-classic 只存最近一次), 之前测试页旧文案与服务端不一致导致交替测试每次 +11s。 - app.py: 接口7 /api/v1/hair/grow-v2 下线(业务弃用;add_hair2.json 的 Klein-9b 会把常驻 Klein-4b 挤出显存)。保留 stub 返回 1007 明确报错,避免裸 404。 实测(1024 档):接口2女 8.5~10s、接口2男 ~5s、接口3 ~7-10s,交替混跑无尖刺。 Co-authored-by: Cursor <cursoragent@cursor.com>
78 lines
3.1 KiB
Python
78 lines
3.1 KiB
Python
#!/usr/bin/env python3
|
|
"""Benchmark ComfyUI hair-inpaint workflow across model / dtype / steps."""
|
|
import io, time, sys
|
|
import requests
|
|
import numpy as np
|
|
from PIL import Image, ImageFilter
|
|
import app as A
|
|
|
|
COMFY = "http://127.0.0.1:8188"
|
|
|
|
|
|
def prep_and_upload():
|
|
image = Image.open("用来重绘.jpg").convert("RGB")
|
|
mask_img = Image.open("用来重绘.png").convert("RGBA")
|
|
mask_data = np.max(np.array(mask_img), axis=2)
|
|
m = Image.fromarray(mask_data, mode="L")
|
|
if m.size != image.size:
|
|
m = m.resize(image.size, Image.LANCZOS)
|
|
m = m.filter(ImageFilter.GaussianBlur(radius=4))
|
|
alpha = Image.eval(m, lambda x: 255 - x)
|
|
r, g, b = image.split()
|
|
rgba = Image.merge("RGBA", (r, g, b, alpha))
|
|
buf = io.BytesIO(); rgba.save(buf, format="PNG"); buf.seek(0)
|
|
up = requests.post(f"{COMFY}/upload/image",
|
|
files={"image": ("hair_input.png", buf, "image/png")}).json()
|
|
return up["name"]
|
|
|
|
|
|
def run_once(fname, model, dtype, steps):
|
|
wf = A.build_workflow(fname, "填充遮罩区域的头发,皮肤加一点磨皮,再加一点美颜")
|
|
wf["16"]["inputs"]["unet_name"] = model
|
|
wf["16"]["inputs"]["weight_dtype"] = dtype
|
|
wf["1"]["inputs"]["steps"] = steps
|
|
r = requests.post(f"{COMFY}/prompt", json={"prompt": wf}).json()
|
|
if "prompt_id" not in r:
|
|
raise RuntimeError(f"submit failed: {str(r)[:300]}")
|
|
pid = r["prompt_id"]
|
|
deadline = time.time() + 180
|
|
while time.time() < deadline:
|
|
time.sleep(0.1)
|
|
h = requests.get(f"{COMFY}/history/{pid}").json()
|
|
if pid not in h:
|
|
continue
|
|
st = h[pid].get("status", {})
|
|
if st.get("status_str") == "error":
|
|
for m in st.get("messages", []):
|
|
if m[0] == "execution_error":
|
|
raise RuntimeError(str(m[1])[:300])
|
|
raise RuntimeError("execution error")
|
|
if "17" in h[pid].get("outputs", {}):
|
|
ts = {mm[0]: mm[1].get("timestamp") for mm in st["messages"]}
|
|
return (ts["execution_success"] - ts["execution_start"]) / 1000.0
|
|
raise TimeoutError("run exceeded 180s")
|
|
|
|
|
|
CONFIGS = [
|
|
("flux2.0/flux-2-klein-9b-fp8.safetensors", "fp8_e4m3fn", 6, "9B fp8 (当前)"),
|
|
("flux2.0/flux-2-klein-9b-fp8.safetensors", "fp8_e4m3fn_fast", 6, "9B fp8-fast"),
|
|
("flux2.0/flux-2-klein-9b-fp8.safetensors", "fp8_e4m3fn_fast", 4, "9B fp8-fast s4"),
|
|
("flux-2-klein-4b-fp8.safetensors", "fp8_e4m3fn", 6, "4B fp8"),
|
|
("flux-2-klein-4b-fp8.safetensors", "fp8_e4m3fn_fast", 6, "4B fp8-fast"),
|
|
("flux-2-klein-4b-fp8.safetensors", "fp8_e4m3fn_fast", 4, "4B fp8-fast s4"),
|
|
]
|
|
|
|
fname = prep_and_upload()
|
|
print("input uploaded:", fname)
|
|
print(f"{'配置':<22}{'warmup':>10}{'run1':>10}{'run2':>10}{'best':>10}")
|
|
for model, dtype, steps, label in CONFIGS:
|
|
times = []
|
|
for i in range(3): # 1 warmup + 2 measured
|
|
try:
|
|
t = run_once(fname, model, dtype, steps)
|
|
except Exception as e:
|
|
t = float('nan'); print("ERR", label, e)
|
|
times.append(t)
|
|
best = min(times[1:])
|
|
print(f"{label:<22}{times[0]:>9.2f}s{times[1]:>9.2f}s{times[2]:>9.2f}s{best:>9.2f}s")
|