Uploaded sanitized BC250/ROCm Repository.
This commit is contained in:
@@ -0,0 +1,65 @@
|
||||
"""Fix v20: Add torch.cuda.synchronize() before GPU transfers to prevent kernel queue buildup"""
|
||||
import sys
|
||||
|
||||
path = '/home/fabian/ComfyUI/bc250_softmax_patch.py'
|
||||
with open(path, 'r') as f:
|
||||
content = f.read()
|
||||
|
||||
# Add sync before EACH .to(cuda) call in _bc250_cast_bias_weight
|
||||
# This prevents GPU kernel queue buildup that blocks subsequent transfers
|
||||
|
||||
# Fix: Add sync before bias.to(cuda)
|
||||
old_bias_to = ''' if bias.device != device or bias.dtype != bias_dtype:
|
||||
if is_first:
|
||||
logger.warning(f"[BC-250] CBW#{_fwd_count[0]}: bias.to({device})...")
|
||||
bias = bias.to(device=device, dtype=bias_dtype, non_blocking=False)'''
|
||||
|
||||
new_bias_to = ''' if bias.device != device or bias.dtype != bias_dtype:
|
||||
# gfx1010: sync GPU before transfer to prevent kernel queue deadlock
|
||||
if device is not None and hasattr(device, 'type') and device.type == 'cuda':
|
||||
torch.cuda.synchronize()
|
||||
if is_first:
|
||||
logger.warning(f"[BC-250] CBW#{_fwd_count[0]}: bias.to({device})...")
|
||||
bias = bias.to(device=device, dtype=bias_dtype, non_blocking=False)'''
|
||||
|
||||
if old_bias_to not in content:
|
||||
print("ERROR: bias.to block not found!")
|
||||
sys.exit(1)
|
||||
content = content.replace(old_bias_to, new_bias_to)
|
||||
print("OK: sync before bias.to(cuda)")
|
||||
|
||||
# Fix: Add sync before weight.to(cuda)
|
||||
old_weight_to = ''' if weight.device != device or weight.dtype != dtype:
|
||||
if is_first:
|
||||
logger.warning(f"[BC-250] CBW#{_fwd_count[0]}: weight.to({device}, {dtype})...")
|
||||
weight = weight.to(device=device, dtype=dtype, non_blocking=False)'''
|
||||
|
||||
new_weight_to = ''' if weight.device != device or weight.dtype != dtype:
|
||||
# gfx1010: sync GPU before transfer to prevent kernel queue deadlock
|
||||
if device is not None and hasattr(device, 'type') and device.type == 'cuda':
|
||||
torch.cuda.synchronize()
|
||||
if is_first:
|
||||
logger.warning(f"[BC-250] CBW#{_fwd_count[0]}: weight.to({device}, {dtype})...")
|
||||
weight = weight.to(device=device, dtype=dtype, non_blocking=False)'''
|
||||
|
||||
if old_weight_to not in content:
|
||||
print("ERROR: weight.to block not found!")
|
||||
sys.exit(1)
|
||||
content = content.replace(old_weight_to, new_weight_to)
|
||||
print("OK: sync before weight.to(cuda)")
|
||||
|
||||
# Update version
|
||||
content = content.replace('Comprehensive Monkey-Patch v19', 'Comprehensive Monkey-Patch v20')
|
||||
content = content.replace('v19 ready', 'v20 ready — sync before every GPU transfer')
|
||||
|
||||
# Add more progress logging (every 50 layers instead of 100)
|
||||
old_progress = ''' if _fwd_count[0] % 100 == 0:'''
|
||||
new_progress = ''' if _fwd_count[0] % 50 == 0:'''
|
||||
if old_progress in content:
|
||||
content = content.replace(old_progress, new_progress)
|
||||
print("OK: progress logging every 50 layers")
|
||||
|
||||
with open(path, 'w') as f:
|
||||
f.write(content)
|
||||
|
||||
print("\nALL v20 PATCHES APPLIED")
|
||||
Reference in New Issue
Block a user