Uploaded sanitized BC250/ROCm Repository.
This commit is contained in:
@@ -0,0 +1,229 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Fix VAE decode hang: kill stuck, check available flags, restart with --cpu-vae."""
|
||||
import paramiko, json, time, textwrap
|
||||
|
||||
ssh = paramiko.SSHClient()
|
||||
ssh.set_missing_host_key_policy(paramiko.AutoAddPolicy())
|
||||
ssh.connect('192.168.178.150', username='fabian', key_filename=r'C:\Users\fabia\.ssh\id_ed25519')
|
||||
|
||||
def run(cmd, timeout=60, desc=""):
|
||||
if desc:
|
||||
print(f"\n{'='*60}\n {desc}\n{'='*60}")
|
||||
_, stdout, stderr = ssh.exec_command(cmd, timeout=timeout)
|
||||
out = stdout.read().decode()
|
||||
err = stderr.read().decode()
|
||||
rc = stdout.channel.recv_exit_status()
|
||||
if out.strip():
|
||||
lines = out.strip().split('\n')
|
||||
if len(lines) > 50:
|
||||
print(f" ... ({len(lines)} lines, showing last 50)")
|
||||
print('\n'.join(lines[-50:]))
|
||||
else:
|
||||
print(out.strip())
|
||||
if err.strip():
|
||||
for l in err.strip().split('\n')[-5:]:
|
||||
print(f" STDERR: {l}")
|
||||
return rc, out, err
|
||||
|
||||
# Kill stuck
|
||||
run("pkill -f 'python3 main.py' 2>/dev/null; sleep 2; pkill -9 -f 'python3 main.py' 2>/dev/null; sleep 1; echo killed",
|
||||
desc="Kill stuck ComfyUI")
|
||||
|
||||
# Check available VAE flags
|
||||
run("bash -c 'source /home/fabian/comfyui-env/bin/activate && cd /home/fabian/ComfyUI && "
|
||||
"python3 main.py --help 2>&1 | grep -i -E \"vae|fp16|fp32|force|cpu|novram|lowvram\"'",
|
||||
desc="ComfyUI VAE/VRAM flags")
|
||||
|
||||
# Update startup script: add --cpu-vae to keep diffusion on GPU but VAE on CPU
|
||||
startup_script = textwrap.dedent("""\
|
||||
#!/bin/bash
|
||||
# BC-250 ComfyUI Launcher — GPU inference with CPU VAE decode
|
||||
# Diffusion sampling: GPU (~6s/step, 8 steps = 51s total)
|
||||
# VAE decode: CPU (GPU hangs on float32 VAE ops on Cyan Skillfish)
|
||||
# Text encoding: CPU (GGUF model, dequant on CPU)
|
||||
|
||||
# GPU identity
|
||||
export HSA_OVERRIDE_GFX_VERSION=10.1.0
|
||||
export HIP_VISIBLE_DEVICES=0
|
||||
export HSA_ENABLE_SDMA=0
|
||||
export HSA_TOOLS_LIB=""
|
||||
export HSA_TOOLS_REPORT_LOAD_FAILURE=0
|
||||
|
||||
# Use all 12 CPU cores
|
||||
export OMP_NUM_THREADS=12
|
||||
export MKL_NUM_THREADS=12
|
||||
export OPENBLAS_NUM_THREADS=12
|
||||
|
||||
# Activate venv
|
||||
source /home/fabian/comfyui-env/bin/activate
|
||||
cd /home/fabian/ComfyUI
|
||||
|
||||
# --novram: send one layer at a time to GPU (needed for 7.6GB shared VRAM)
|
||||
# --force-fp16: halve VRAM usage for diffusion model
|
||||
# --cpu-vae: decode VAE on CPU (GPU hangs on VAE float32 conv2d ops)
|
||||
# --disable-smart-memory: prevent memory heuristics from interfering
|
||||
exec python3 main.py \\
|
||||
--listen 0.0.0.0 --port 8188 \\
|
||||
--novram \\
|
||||
--force-fp16 \\
|
||||
--cpu-vae \\
|
||||
--disable-smart-memory
|
||||
""")
|
||||
|
||||
sftp = ssh.open_sftp()
|
||||
with sftp.open('/home/fabian/start_comfyui.sh', 'w') as f:
|
||||
f.write(startup_script)
|
||||
sftp.close()
|
||||
run("chmod +x /home/fabian/start_comfyui.sh")
|
||||
print("\n Updated: added --cpu-vae (GPU sampler + CPU VAE)")
|
||||
|
||||
# Launch
|
||||
run("bash -c 'nohup /home/fabian/start_comfyui.sh > /home/fabian/comfyui.log 2>&1 &'; sleep 1; echo launched",
|
||||
desc="Launch ComfyUI")
|
||||
|
||||
print("\n Waiting for server...")
|
||||
for i in range(40):
|
||||
time.sleep(3)
|
||||
rc, out, _ = run("bash -c 'curl -s -o /dev/null -w \"%{http_code}\" http://localhost:8188/ 2>/dev/null || echo 0'")
|
||||
if out.strip() == '200':
|
||||
print(f" Server ready! ({(i+1)*3}s)")
|
||||
break
|
||||
if i % 5 == 4:
|
||||
rc2, log, _ = run("tail -3 /home/fabian/comfyui.log 2>/dev/null")
|
||||
print(f" [{(i+1)*3}s] waiting... {log.strip().split(chr(10))[-1][:80]}")
|
||||
else:
|
||||
print(" Timeout!")
|
||||
run("tail -40 /home/fabian/comfyui.log", desc="Log")
|
||||
ssh.close()
|
||||
exit(1)
|
||||
|
||||
# Submit workflow
|
||||
workflow = {
|
||||
"prompt": {
|
||||
"1": {
|
||||
"class_type": "UnetLoaderGGUF",
|
||||
"inputs": {"unet_name": "z_image_turbo-Q5_K_S.gguf"}
|
||||
},
|
||||
"2": {
|
||||
"class_type": "CLIPLoaderGGUF",
|
||||
"inputs": {"clip_name": "Qwen3-4B.i1-Q5_K_S.gguf", "type": "qwen_image"}
|
||||
},
|
||||
"3": {
|
||||
"class_type": "VAELoader",
|
||||
"inputs": {"vae_name": "ae.safetensors"}
|
||||
},
|
||||
"4": {
|
||||
"class_type": "CLIPTextEncode",
|
||||
"inputs": {
|
||||
"text": "A red fox in a snowy forest, photorealistic",
|
||||
"clip": ["2", 0]
|
||||
}
|
||||
},
|
||||
"5": {
|
||||
"class_type": "CLIPTextEncode",
|
||||
"inputs": {"text": "", "clip": ["2", 0]}
|
||||
},
|
||||
"6": {
|
||||
"class_type": "EmptyLatentImage",
|
||||
"inputs": {"width": 512, "height": 512, "batch_size": 1}
|
||||
},
|
||||
"7": {
|
||||
"class_type": "KSampler",
|
||||
"inputs": {
|
||||
"model": ["1", 0],
|
||||
"seed": 42,
|
||||
"steps": 8,
|
||||
"cfg": 1.0,
|
||||
"sampler_name": "euler",
|
||||
"scheduler": "simple",
|
||||
"positive": ["4", 0],
|
||||
"negative": ["5", 0],
|
||||
"latent_image": ["6", 0],
|
||||
"denoise": 1.0
|
||||
}
|
||||
},
|
||||
"8": {
|
||||
"class_type": "VAEDecode",
|
||||
"inputs": {"samples": ["7", 0], "vae": ["3", 0]}
|
||||
},
|
||||
"9": {
|
||||
"class_type": "SaveImage",
|
||||
"inputs": {"images": ["8", 0], "filename_prefix": "ZImageTurbo_test"}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
sftp2 = ssh.open_sftp()
|
||||
with sftp2.open('/tmp/zimage_workflow.json', 'w') as f:
|
||||
f.write(json.dumps(workflow))
|
||||
sftp2.close()
|
||||
|
||||
rc, out, _ = run("bash -c 'curl -s -X POST http://localhost:8188/prompt "
|
||||
"-H \"Content-Type: application/json\" "
|
||||
"-d @/tmp/zimage_workflow.json'",
|
||||
desc="Submit workflow (GPU sampling + CPU VAE)")
|
||||
|
||||
try:
|
||||
resp = json.loads(out.strip())
|
||||
if 'error' in resp:
|
||||
print(f" ERROR: {resp['error']}")
|
||||
if 'node_errors' in resp:
|
||||
for nid, e in resp['node_errors'].items():
|
||||
print(f" Node {nid}: {e}")
|
||||
ssh.close()
|
||||
exit(1)
|
||||
print(f" Prompt ID: {resp.get('prompt_id')}")
|
||||
except:
|
||||
print(f" Response: {out.strip()[:500]}")
|
||||
|
||||
# Monitor
|
||||
print("\n Monitoring GPU generation + CPU VAE decode...")
|
||||
last_log = ""
|
||||
for i in range(120):
|
||||
time.sleep(15)
|
||||
|
||||
stats = run("bash -c 'PID=$(pgrep -f \"python3 main.py\" | head -1); "
|
||||
"if [ -n \"$PID\" ]; then "
|
||||
" CPU=$(ps -p $PID -o %cpu --no-headers); "
|
||||
" MEM=$(ps -p $PID -o rss --no-headers); "
|
||||
" GPU_TEMP=$(cat /sys/class/drm/card1/device/hwmon/hwmon*/temp1_input 2>/dev/null || echo 0); "
|
||||
" GPU_POWER=$(cat /sys/class/drm/card1/device/hwmon/hwmon*/power1_average 2>/dev/null || echo 0); "
|
||||
" echo \"CPU:${CPU}% RSS:$((MEM/1024))MB GPU_T:$((GPU_TEMP/1000))C GPU_P:$((GPU_POWER/1000000))W\"; "
|
||||
"else echo DEAD; fi'")[1].strip()
|
||||
|
||||
log = run("tail -10 /home/fabian/comfyui.log 2>/dev/null")[1].strip()
|
||||
|
||||
elapsed = (i+1)*15
|
||||
m, s = divmod(elapsed, 60)
|
||||
|
||||
print(f" [{m}m{s:02d}s] {stats}")
|
||||
|
||||
# Show last meaningful log line if changed
|
||||
if log != last_log:
|
||||
for line in reversed(log.split('\n')):
|
||||
l = line.strip()
|
||||
if l and not l.startswith('FETCH') and not l.startswith('[DEPRECATION'):
|
||||
print(f" LOG: {l[:120]}")
|
||||
break
|
||||
last_log = log
|
||||
|
||||
if 'DEAD' in stats:
|
||||
print("\n PROCESS DIED!")
|
||||
run("tail -60 /home/fabian/comfyui.log", desc="Death log")
|
||||
break
|
||||
|
||||
if 'Prompt executed in' in log:
|
||||
print(f"\n IMAGE GENERATED!")
|
||||
run("tail -30 /home/fabian/comfyui.log", desc="Success log")
|
||||
break
|
||||
|
||||
if 'Traceback' in log or 'CUDA out of memory' in log:
|
||||
print("\n ERROR!")
|
||||
run("tail -60 /home/fabian/comfyui.log", desc="Error log")
|
||||
break
|
||||
|
||||
# Output
|
||||
run("ls -lah /home/fabian/ComfyUI/output/", desc="Output files")
|
||||
|
||||
ssh.close()
|
||||
print("\nDone.")
|
||||
Reference in New Issue
Block a user