"""Patch ComfyUI for BC-250 APU: Use SHARED VRAM mode + force fp16 VAE. This is the correct mode for an APU where CPU and GPU share the same physical memory.""" import paramiko, time, json, textwrap k = paramiko.Ed25519Key.from_private_key_file(r'C:\Users\fabia\.ssh\id_ed25519') c = paramiko.SSHClient() c.set_missing_host_key_policy(paramiko.AutoAddPolicy()) c.connect('192.168.178.150', username='fabian', pkey=k, timeout=15) sftp = c.open_sftp() def sh(cmd, timeout=60): chan = c.get_transport().open_session() chan.settimeout(timeout) chan.exec_command('/bin/bash -l -c ' + "'" + cmd.replace("'", "'\\''") + "'") out = b"" while True: try: chunk = chan.recv(65536) if not chunk: break out += chunk except: break chan.close() return out.decode(errors='replace').strip() # ====================================== # 1) Kill # ====================================== print("1) Kill ComfyUI") sh('pkill -9 -f "python3.*main.py" 2>/dev/null; sleep 2') # ====================================== # 2) Backup + Patch model_management.py # ====================================== print("2) Patch model_management.py: SHARED mode for APU") with sftp.open('/home/fabian/ComfyUI/comfy/model_management.py', 'r') as f: code = f.read().decode() # Backup with sftp.open('/home/fabian/ComfyUI/comfy/model_management.py.bak', 'w') as f: f.write(code) print(" Backup saved") # PATCH 1: After MPS sets SHARED, also set SHARED for this AMD APU # Current code (L458-462): # if cpu_state != CPUState.GPU: # vram_state = VRAMState.DISABLED # if cpu_state == CPUState.MPS: # vram_state = VRAMState.SHARED # # We add: if the GPU has shared memory (small dedicated VRAM), set SHARED old_block = '''if cpu_state == CPUState.MPS: vram_state = VRAMState.SHARED logging.info(f"Set vram state to: {vram_state.name}")''' new_block = '''if cpu_state == CPUState.MPS: vram_state = VRAMState.SHARED # BC-250 APU: shared memory between CPU and GPU. Dedicated VRAM is tiny (512MB) # but the full system RAM is accessible to both. SHARED mode loads models # directly on GPU (zero-copy for shared memory APUs). if cpu_state == CPUState.GPU and vram_state not in (VRAMState.DISABLED, VRAMState.SHARED): try: import os if os.environ.get("COMFYUI_SHARED_MEMORY") == "1": vram_state = VRAMState.SHARED logging.info("Forcing SHARED vram state (COMFYUI_SHARED_MEMORY=1)") except: pass logging.info(f"Set vram state to: {vram_state.name}")''' if old_block in code: code = code.replace(old_block, new_block) print(" PATCH 1 applied: COMFYUI_SHARED_MEMORY env var support") else: print(" PATCH 1: Could not find exact block, trying alternate...") # Try line by line lines = code.split('\n') for i, line in enumerate(lines): if 'cpu_state == CPUState.MPS' in line and 'SHARED' in lines[i+1] if i+1 < len(lines) else '': # Insert after the MPS block insert_idx = i + 2 # After "vram_state = VRAMState.SHARED" patch_lines = [ '', '# BC-250 APU shared memory support', 'if cpu_state == CPUState.GPU and vram_state not in (VRAMState.DISABLED, VRAMState.SHARED):', ' try:', ' import os', ' if os.environ.get("COMFYUI_SHARED_MEMORY") == "1":', ' vram_state = VRAMState.SHARED', ' logging.info("Forcing SHARED vram state (COMFYUI_SHARED_MEMORY=1)")', ' except:', ' pass', ] for j, pl in enumerate(patch_lines): lines.insert(insert_idx + j, pl) code = '\n'.join(lines) print(f" PATCH 1 applied (alternate) at line {insert_idx}") break # Write patched file with sftp.open('/home/fabian/ComfyUI/comfy/model_management.py', 'w') as f: f.write(code) print(" File written") # Verify patch with sftp.open('/home/fabian/ComfyUI/comfy/model_management.py', 'r') as f: verify = f.read().decode() if 'COMFYUI_SHARED_MEMORY' in verify: print(" Patch verified!") else: print(" ERROR: Patch not found in file!") # ====================================== # 3) Write launcher with SHARED mode # ====================================== print("3) Write launcher with COMFYUI_SHARED_MEMORY=1") launcher = textwrap.dedent("""\ #!/bin/bash # GPU export HSA_OVERRIDE_GFX_VERSION=10.1.0 export HIP_VISIBLE_DEVICES=0 export HSA_ENABLE_SDMA=0 export HSA_TOOLS_LIB="" export HSA_TOOLS_REPORT_LOAD_FAILURE=0 export PYTORCH_HIP_ALLOC_CONF=expandable_segments:False # Threading export OMP_NUM_THREADS=12 export MKL_NUM_THREADS=12 export OPENBLAS_NUM_THREADS=12 # MIOpen export MIOPEN_FIND_MODE=3 # Shared memory APU mode: CPU and GPU share the same physical RAM export COMFYUI_SHARED_MEMORY=1 cd ~/ComfyUI source ~/comfyui-env/bin/activate # --force-fp16: half precision (saves memory) # --fp16-vae: VAE in fp16 (320MB instead of 640MB, fits in GPU memory) # SHARED mode: models load directly on GPU, no offloading overhead exec python3 main.py \\ --listen 0.0.0.0 --port 8188 \\ --force-fp16 \\ --fp16-vae """) with sftp.open('/tmp/run_comfyui.sh', 'w') as f: f.write(launcher) sh('chmod +x /tmp/run_comfyui.sh') print(" Flags: --force-fp16 --fp16-vae + COMFYUI_SHARED_MEMORY=1") # ====================================== # 4) Start # ====================================== print("4) Start ComfyUI") sh('rm -f /tmp/comfyui.log; rm -f ~/ComfyUI/output/ZImageTurbo_GPU*.png') sh('nohup /tmp/run_comfyui.sh > /tmp/comfyui.log 2>&1 &') time.sleep(3) pid = sh('pgrep -f "python3.*main.py"') print(f" PID: {pid}") # ====================================== # 5) Wait ready # ====================================== print("5) Wait HTTP", end='', flush=True) for i in range(90): code = sh('curl -s -o /dev/null -w "%{http_code}" http://127.0.0.1:8188/ 2>/dev/null', timeout=5) if '200' in code: print(f" OK ({i*2}s)") break if i % 10 == 0 and i > 0: try: with sftp.open('/tmp/comfyui.log', 'r') as f: log = f.read().decode(errors='replace') ls = [l.strip() for l in log.split('\n') if l.strip() and 'FETCH' not in l and 'DEPRECATION' not in l] print(f"\n [{i*2}s] {ls[-1][:80] if ls else ''}", end='', flush=True) except: pass else: print('.', end='', flush=True) time.sleep(2) # Verify SHARED mode with sftp.open('/tmp/comfyui.log', 'r') as f: log = f.read().decode(errors='replace') for line in log.split('\n'): s = line.strip() if any(k in s for k in ['vram state', 'SHARED', 'Device:', 'Total VRAM', 'pytorch version']): print(f" {s}") # ====================================== # 6) Submit # ====================================== print("6) Submit") wf = {"prompt": { "1": {"class_type": "UnetLoaderGGUF", "inputs": {"unet_name": "z_image_turbo-Q5_K_S.gguf"}}, "2": {"class_type": "CLIPLoaderGGUF", "inputs": {"clip_name": "Qwen3-4B.i1-Q5_K_S.gguf", "type": "qwen_image"}}, "3": {"class_type": "VAELoader", "inputs": {"vae_name": "ae.safetensors"}}, "4": {"class_type": "CLIPTextEncode", "inputs": {"text": "A red fox in a snowy forest, photorealistic", "clip": ["2", 0]}}, "5": {"class_type": "EmptyLatentImage", "inputs": {"width": 512, "height": 512, "batch_size": 1}}, "6": {"class_type": "KSampler", "inputs": {"model": ["1", 0], "positive": ["4", 0], "negative": ["4", 0], "latent_image": ["5", 0], "seed": 999, "steps": 8, "cfg": 1.0, "sampler_name": "euler", "scheduler": "simple", "denoise": 1.0}}, "7": {"class_type": "VAEDecode", "inputs": {"samples": ["6", 0], "vae": ["3", 0]}}, "8": {"class_type": "SaveImage", "inputs": {"images": ["7", 0], "filename_prefix": "ZImageTurbo_GPU"}} }} with sftp.open('/tmp/wf.json', 'w') as f: f.write(json.dumps(wf)) resp = sh('curl -s -X POST http://127.0.0.1:8188/prompt -H "Content-Type: application/json" -d @/tmp/wf.json') print(f" {resp[:150]}") # ====================================== # 7) Monitor # ====================================== print("7) Monitor (SHARED mode = everything on GPU)") t0 = time.time() for i in range(200): el = int(time.time() - t0) temp = sh('cat /sys/class/drm/card0/device/hwmon/hwmon*/temp1_input 2>/dev/null', timeout=5) tc = int(temp)//1000 if temp.isdigit() else '?' try: with sftp.open('/tmp/comfyui.log', 'r') as f: log = f.read().decode(errors='replace') except: log = '' samp = '' last = '' for line in log.split('\n'): s = line.strip() if '/8' in s and ('it/s' in s or 's/it' in s): samp = s if s and 'FETCH' not in s and 'startup' not in s and 'DEPRECATION' not in s: last = s print(f" [{el:>4}s] {tc}C | {(samp or last)[-90:]}") imgs = sh('ls ~/ComfyUI/output/ZImageTurbo_GPU*.png 2>/dev/null', timeout=5) if imgs: et = '' for line in log.split('\n'): if 'Prompt executed' in line: et = line.strip() print(f"\n *** DONE! ***") print(f" File: {imgs}") print(f" {et}") print(f" Wall: {el}s") for line in log.split('\n'): s = line.strip() if any(k in s for k in ['loaded', '/8', 'Prompt executed', 'VAE load', 'Requested']): if 'FETCH' not in s: print(f" {s}") break q = sh('curl -s http://127.0.0.1:8188/queue 2>/dev/null', timeout=5) try: qd = json.loads(q) if not qd.get('queue_running') and not qd.get('queue_pending') and el > 30: time.sleep(2) imgs = sh('ls ~/ComfyUI/output/ZImageTurbo_GPU*.png 2>/dev/null', timeout=5) if imgs: print(f"\n *** DONE: {imgs} ***") else: print(f"\n Queue empty, no image. Log:") for line in log.split('\n')[-20:]: if line.strip() and 'FETCH' not in line: print(f" {line.strip()}") break except: pass if sh('pgrep -f "python3.*main.py" >/dev/null && echo Y || echo N', timeout=5) == 'N': print("\n CRASHED!") for line in log.split('\n')[-30:]: if line.strip(): print(f" {line.strip()}") break time.sleep(10) sftp.close() c.close() print("\nDone.")