This repository has been archived on 2026-08-19. You can view files and clone it. You cannot open issues or pull requests or push a commit.
Files
ROCm-Research-Archive/ComfyUI Scripts/bc250_shared.py
T
2026-08-20 00:45:43 +02:00

280 lines
10 KiB
Python

"""Patch ComfyUI for BC-250 APU: Use SHARED VRAM mode + force fp16 VAE.
This is the correct mode for an APU where CPU and GPU share the same physical memory."""
import paramiko, time, json, textwrap
k = paramiko.Ed25519Key.from_private_key_file(r'C:\Users\fabia\.ssh\id_ed25519')
c = paramiko.SSHClient()
c.set_missing_host_key_policy(paramiko.AutoAddPolicy())
c.connect('192.168.178.150', username='fabian', pkey=k, timeout=15)
sftp = c.open_sftp()
def sh(cmd, timeout=60):
chan = c.get_transport().open_session()
chan.settimeout(timeout)
chan.exec_command('/bin/bash -l -c ' + "'" + cmd.replace("'", "'\\''") + "'")
out = b""
while True:
try:
chunk = chan.recv(65536)
if not chunk: break
out += chunk
except: break
chan.close()
return out.decode(errors='replace').strip()
# ======================================
# 1) Kill
# ======================================
print("1) Kill ComfyUI")
sh('pkill -9 -f "python3.*main.py" 2>/dev/null; sleep 2')
# ======================================
# 2) Backup + Patch model_management.py
# ======================================
print("2) Patch model_management.py: SHARED mode for APU")
with sftp.open('/home/fabian/ComfyUI/comfy/model_management.py', 'r') as f:
code = f.read().decode()
# Backup
with sftp.open('/home/fabian/ComfyUI/comfy/model_management.py.bak', 'w') as f:
f.write(code)
print(" Backup saved")
# PATCH 1: After MPS sets SHARED, also set SHARED for this AMD APU
# Current code (L458-462):
# if cpu_state != CPUState.GPU:
# vram_state = VRAMState.DISABLED
# if cpu_state == CPUState.MPS:
# vram_state = VRAMState.SHARED
#
# We add: if the GPU has shared memory (small dedicated VRAM), set SHARED
old_block = '''if cpu_state == CPUState.MPS:
vram_state = VRAMState.SHARED
logging.info(f"Set vram state to: {vram_state.name}")'''
new_block = '''if cpu_state == CPUState.MPS:
vram_state = VRAMState.SHARED
# BC-250 APU: shared memory between CPU and GPU. Dedicated VRAM is tiny (512MB)
# but the full system RAM is accessible to both. SHARED mode loads models
# directly on GPU (zero-copy for shared memory APUs).
if cpu_state == CPUState.GPU and vram_state not in (VRAMState.DISABLED, VRAMState.SHARED):
try:
import os
if os.environ.get("COMFYUI_SHARED_MEMORY") == "1":
vram_state = VRAMState.SHARED
logging.info("Forcing SHARED vram state (COMFYUI_SHARED_MEMORY=1)")
except:
pass
logging.info(f"Set vram state to: {vram_state.name}")'''
if old_block in code:
code = code.replace(old_block, new_block)
print(" PATCH 1 applied: COMFYUI_SHARED_MEMORY env var support")
else:
print(" PATCH 1: Could not find exact block, trying alternate...")
# Try line by line
lines = code.split('\n')
for i, line in enumerate(lines):
if 'cpu_state == CPUState.MPS' in line and 'SHARED' in lines[i+1] if i+1 < len(lines) else '':
# Insert after the MPS block
insert_idx = i + 2 # After "vram_state = VRAMState.SHARED"
patch_lines = [
'',
'# BC-250 APU shared memory support',
'if cpu_state == CPUState.GPU and vram_state not in (VRAMState.DISABLED, VRAMState.SHARED):',
' try:',
' import os',
' if os.environ.get("COMFYUI_SHARED_MEMORY") == "1":',
' vram_state = VRAMState.SHARED',
' logging.info("Forcing SHARED vram state (COMFYUI_SHARED_MEMORY=1)")',
' except:',
' pass',
]
for j, pl in enumerate(patch_lines):
lines.insert(insert_idx + j, pl)
code = '\n'.join(lines)
print(f" PATCH 1 applied (alternate) at line {insert_idx}")
break
# Write patched file
with sftp.open('/home/fabian/ComfyUI/comfy/model_management.py', 'w') as f:
f.write(code)
print(" File written")
# Verify patch
with sftp.open('/home/fabian/ComfyUI/comfy/model_management.py', 'r') as f:
verify = f.read().decode()
if 'COMFYUI_SHARED_MEMORY' in verify:
print(" Patch verified!")
else:
print(" ERROR: Patch not found in file!")
# ======================================
# 3) Write launcher with SHARED mode
# ======================================
print("3) Write launcher with COMFYUI_SHARED_MEMORY=1")
launcher = textwrap.dedent("""\
#!/bin/bash
# GPU
export HSA_OVERRIDE_GFX_VERSION=10.1.0
export HIP_VISIBLE_DEVICES=0
export HSA_ENABLE_SDMA=0
export HSA_TOOLS_LIB=""
export HSA_TOOLS_REPORT_LOAD_FAILURE=0
export PYTORCH_HIP_ALLOC_CONF=expandable_segments:False
# Threading
export OMP_NUM_THREADS=12
export MKL_NUM_THREADS=12
export OPENBLAS_NUM_THREADS=12
# MIOpen
export MIOPEN_FIND_MODE=3
# Shared memory APU mode: CPU and GPU share the same physical RAM
export COMFYUI_SHARED_MEMORY=1
cd ~/ComfyUI
source ~/comfyui-env/bin/activate
# --force-fp16: half precision (saves memory)
# --fp16-vae: VAE in fp16 (320MB instead of 640MB, fits in GPU memory)
# SHARED mode: models load directly on GPU, no offloading overhead
exec python3 main.py \\
--listen 0.0.0.0 --port 8188 \\
--force-fp16 \\
--fp16-vae
""")
with sftp.open('/tmp/run_comfyui.sh', 'w') as f:
f.write(launcher)
sh('chmod +x /tmp/run_comfyui.sh')
print(" Flags: --force-fp16 --fp16-vae + COMFYUI_SHARED_MEMORY=1")
# ======================================
# 4) Start
# ======================================
print("4) Start ComfyUI")
sh('rm -f /tmp/comfyui.log; rm -f ~/ComfyUI/output/ZImageTurbo_GPU*.png')
sh('nohup /tmp/run_comfyui.sh > /tmp/comfyui.log 2>&1 &')
time.sleep(3)
pid = sh('pgrep -f "python3.*main.py"')
print(f" PID: {pid}")
# ======================================
# 5) Wait ready
# ======================================
print("5) Wait HTTP", end='', flush=True)
for i in range(90):
code = sh('curl -s -o /dev/null -w "%{http_code}" http://127.0.0.1:8188/ 2>/dev/null', timeout=5)
if '200' in code:
print(f" OK ({i*2}s)")
break
if i % 10 == 0 and i > 0:
try:
with sftp.open('/tmp/comfyui.log', 'r') as f:
log = f.read().decode(errors='replace')
ls = [l.strip() for l in log.split('\n') if l.strip() and 'FETCH' not in l and 'DEPRECATION' not in l]
print(f"\n [{i*2}s] {ls[-1][:80] if ls else ''}", end='', flush=True)
except: pass
else:
print('.', end='', flush=True)
time.sleep(2)
# Verify SHARED mode
with sftp.open('/tmp/comfyui.log', 'r') as f:
log = f.read().decode(errors='replace')
for line in log.split('\n'):
s = line.strip()
if any(k in s for k in ['vram state', 'SHARED', 'Device:', 'Total VRAM', 'pytorch version']):
print(f" {s}")
# ======================================
# 6) Submit
# ======================================
print("6) Submit")
wf = {"prompt": {
"1": {"class_type": "UnetLoaderGGUF", "inputs": {"unet_name": "z_image_turbo-Q5_K_S.gguf"}},
"2": {"class_type": "CLIPLoaderGGUF", "inputs": {"clip_name": "Qwen3-4B.i1-Q5_K_S.gguf", "type": "qwen_image"}},
"3": {"class_type": "VAELoader", "inputs": {"vae_name": "ae.safetensors"}},
"4": {"class_type": "CLIPTextEncode", "inputs": {"text": "A red fox in a snowy forest, photorealistic", "clip": ["2", 0]}},
"5": {"class_type": "EmptyLatentImage", "inputs": {"width": 512, "height": 512, "batch_size": 1}},
"6": {"class_type": "KSampler", "inputs": {"model": ["1", 0], "positive": ["4", 0], "negative": ["4", 0],
"latent_image": ["5", 0], "seed": 999, "steps": 8, "cfg": 1.0,
"sampler_name": "euler", "scheduler": "simple", "denoise": 1.0}},
"7": {"class_type": "VAEDecode", "inputs": {"samples": ["6", 0], "vae": ["3", 0]}},
"8": {"class_type": "SaveImage", "inputs": {"images": ["7", 0], "filename_prefix": "ZImageTurbo_GPU"}}
}}
with sftp.open('/tmp/wf.json', 'w') as f:
f.write(json.dumps(wf))
resp = sh('curl -s -X POST http://127.0.0.1:8188/prompt -H "Content-Type: application/json" -d @/tmp/wf.json')
print(f" {resp[:150]}")
# ======================================
# 7) Monitor
# ======================================
print("7) Monitor (SHARED mode = everything on GPU)")
t0 = time.time()
for i in range(200):
el = int(time.time() - t0)
temp = sh('cat /sys/class/drm/card0/device/hwmon/hwmon*/temp1_input 2>/dev/null', timeout=5)
tc = int(temp)//1000 if temp.isdigit() else '?'
try:
with sftp.open('/tmp/comfyui.log', 'r') as f:
log = f.read().decode(errors='replace')
except: log = ''
samp = ''
last = ''
for line in log.split('\n'):
s = line.strip()
if '/8' in s and ('it/s' in s or 's/it' in s): samp = s
if s and 'FETCH' not in s and 'startup' not in s and 'DEPRECATION' not in s: last = s
print(f" [{el:>4}s] {tc}C | {(samp or last)[-90:]}")
imgs = sh('ls ~/ComfyUI/output/ZImageTurbo_GPU*.png 2>/dev/null', timeout=5)
if imgs:
et = ''
for line in log.split('\n'):
if 'Prompt executed' in line: et = line.strip()
print(f"\n *** DONE! ***")
print(f" File: {imgs}")
print(f" {et}")
print(f" Wall: {el}s")
for line in log.split('\n'):
s = line.strip()
if any(k in s for k in ['loaded', '/8', 'Prompt executed', 'VAE load', 'Requested']):
if 'FETCH' not in s:
print(f" {s}")
break
q = sh('curl -s http://127.0.0.1:8188/queue 2>/dev/null', timeout=5)
try:
qd = json.loads(q)
if not qd.get('queue_running') and not qd.get('queue_pending') and el > 30:
time.sleep(2)
imgs = sh('ls ~/ComfyUI/output/ZImageTurbo_GPU*.png 2>/dev/null', timeout=5)
if imgs:
print(f"\n *** DONE: {imgs} ***")
else:
print(f"\n Queue empty, no image. Log:")
for line in log.split('\n')[-20:]:
if line.strip() and 'FETCH' not in line: print(f" {line.strip()}")
break
except: pass
if sh('pgrep -f "python3.*main.py" >/dev/null && echo Y || echo N', timeout=5) == 'N':
print("\n CRASHED!")
for line in log.split('\n')[-30:]:
if line.strip(): print(f" {line.strip()}")
break
time.sleep(10)
sftp.close()
c.close()
print("\nDone.")