Uploaded sanitized BC250/ROCm Repository.

This commit is contained in:
Fabian
2026-08-20 00:45:43 +02:00
parent 7d2184f1e8
commit d7d22e93b3
678 changed files with 65963 additions and 1 deletions
@@ -0,0 +1,46 @@
import os
os.environ["HSA_OVERRIDE_GFX_VERSION"] = "10.1.0"
os.environ["HSA_ENABLE_SDMA"] = "0"
os.environ["HIP_VISIBLE_DEVICES"] = "0"
os.environ["TORCHDYNAMO_DISABLE"] = "1"
import torch
import time
print("CUDA available:", torch.cuda.is_available())
print("Device:", torch.cuda.get_device_name(0))
# Test 1: Small CPU→GPU
x = torch.randn(100, 100)
print("CPU tensor created, starting small transfer...")
t0 = time.time()
y = x.to("cuda:0")
t1 = time.time()
print(f"Small transfer: {t1-t0:.3f}s, device={y.device}")
# Test 2: Medium CPU→GPU (f16)
x2 = torch.randn(2048, 2048, dtype=torch.float16)
mb = x2.numel() * 2 / 1e6
print(f"Medium tensor: {mb:.1f} MB, starting transfer...")
t0 = time.time()
y2 = x2.to("cuda:0")
t1 = time.time()
print(f"Medium transfer: {t1-t0:.3f}s")
# Test 3: Big CPU→GPU (f16)
x3 = torch.randn(4096, 4096, dtype=torch.float16)
mb3 = x3.numel() * 2 / 1e6
print(f"Big tensor: {mb3:.1f} MB, starting transfer...")
t0 = time.time()
y3 = x3.to("cuda:0")
t1 = time.time()
print(f"Big transfer: {t1-t0:.3f}s")
# Test 4: GPU compute
print("Testing GPU compute...")
t0 = time.time()
z = torch.matmul(y2, y2.T)
torch.cuda.synchronize()
t1 = time.time()
print(f"GPU matmul 2048x2048: {t1-t0:.3f}s")
print("ALL TESTS PASSED")