diff --git a/example/002-example/example_cudacode.py b/example/002-example/example_cudacode.py new file mode 100644 index 0000000..03da087 --- /dev/null +++ b/example/002-example/example_cudacode.py @@ -0,0 +1,48 @@ +import torch +from torch.utils.cpp_extension import load_inline + +# Swish激活函数的CUDA实现 (x * sigmoid(x)) +swish_source = """ +#include +#include + +__global__ void swish_kernel(const float* x, float* y, int size) { + int idx = blockIdx.x * blockDim.x + threadIdx.x; + if (idx < size) { + // 高效计算Swish: x * (1 / (1 + exp(-x))) + float val = x[idx]; + float sigmoid = 1.0f / (1.0f + expf(-val)); + y[idx] = val * sigmoid; + } +} + +torch::Tensor swish_cuda(torch::Tensor x) { + auto size = x.numel(); + auto y = torch::empty_like(x); + const int block_size = 256; + int num_blocks = (size + block_size - 1) / block_size; + swish_kernel<<>>(x.data_ptr(), y.data_ptr(), size); + return y; +} +""" + +swish_cpp_source = """ +torch::Tensor swish_cuda(torch::Tensor x); +""" + +# 编译内联CUDA代码 +swish = load_inline( + name="swish", + cpp_sources=swish_cpp_source, + cuda_sources=swish_source, + functions=["swish_cuda"], + verbose=True +) + +class ModelNew(torch.nn.Module): + def __init__(self): + super(ModelNew, self).__init__() + self.swish = swish # 包含自定义Swish算子的模块 + + def forward(self, x): + return self.swish.swish_cuda(x) \ No newline at end of file diff --git a/example/002-example/example_torchcode.py b/example/002-example/example_torchcode.py new file mode 100644 index 0000000..cac6a94 --- /dev/null +++ b/example/002-example/example_torchcode.py @@ -0,0 +1,20 @@ +import torch +import torch.nn as nn + +class Model(nn.Module): + def __init__(self): + super(Model, self).__init__() + + def forward(self, x: torch.Tensor) -> torch.Tensor: + # 使用Swish替代原始的ReLU + return x * torch.sigmoid(x) # PyTorch内置Swish实现 + +batch_size = 16 +dim = 16384 + +def get_inputs(): + x = torch.randn(batch_size, dim) + return [x] + +def get_init_inputs(): + return [] # 不需要特殊初始化输入 \ No newline at end of file diff --git a/example/002-example/prompt.txt b/example/002-example/prompt.txt new file mode 100644 index 0000000..a9f5eb1 --- /dev/null +++ b/example/002-example/prompt.txt @@ -0,0 +1,31 @@ +Write a custom CUDA kernel that fuses matrix multiplication with GELU activation. + +The original architecture performs: +1. Matrix multiplication: output = input @ weight.T + bias +2. GELU activation: gelu_output = gelu(output) + +You should fuse these two operations into a single CUDA kernel to avoid: +- Storing the intermediate matrix multiplication result to global memory +- Reading it back for the GELU operation + +The GELU activation function can be approximated as: + gelu(x) = 0.5 * x * (1 + tanh(sqrt(2/π) * (x + 0.044715 * x^3))) + +Considerations: +- Use 2D grid and block dimensions to parallelize over batch size and hidden features +- Implement efficient shared memory usage for tiling if possible +- Ensure numerical stability and precision + +You are given the following architecture: + +import torch +import torch.nn as nn + +class Model(nn.Module): + def __init__(self, in_features=16384, hidden_features=4096): + super(Model, self).__init__() + self.linear = nn.Linear(in_features, hidden_features) + + def forward(self, x): + x = self.linear(x) + return torch.nn.functional.gelu(x) \ No newline at end of file diff --git a/example/002-example/run_code.py b/example/002-example/run_code.py new file mode 100644 index 0000000..fbd9757 --- /dev/null +++ b/example/002-example/run_code.py @@ -0,0 +1,78 @@ +import torch +import time +from example_torchcode import Model, get_inputs, get_init_inputs +from example_cudacode import ModelNew + +def run_benchmark(): + if not torch.cuda.is_available(): + print("CUDA 不可用") + return + + device = torch.device("cuda") + + # 准备输入数据 + inputs = [x.cuda(device=device) for x in get_inputs()] + init_inputs = [x.cuda(device=device) if isinstance(x, torch.Tensor) else x for x in get_init_inputs()] + + # 初始化模型 + torch_model = Model(*init_inputs).cuda() + cuda_model = ModelNew(*init_inputs).cuda() + + torch_model.eval() + cuda_model.eval() + + print("-------------------- 精度对齐验证 --------------------") + with torch.no_grad(): + # 预热GPU + _ = torch_model(*inputs) + _ = cuda_model(*inputs) + + # 正式测试 + output_torch = torch_model(*inputs) + output_cuda = cuda_model(*inputs) + + # 精度验证 + abs_diff = torch.abs(output_torch - output_cuda) + max_diff = torch.max(abs_diff).item() + mean_diff = torch.mean(abs_diff).item() + + if max_diff < 1e-4 and mean_diff < 1e-5: + print(f"✅ 精度对齐:最大误差 {max_diff:.6f},平均误差 {mean_diff:.6f}") + precision_flag = True + else: + print(f"❌ 精度不一致:最大误差 {max_diff:.6f},平均误差 {mean_diff:.6f}") + precision_flag = False + + print("\n-------------------- 性能加速比测试 --------------------") + num_iterations = 100 + + # 预热GPU + for _ in range(10): + _ = torch_model(*inputs) + _ = cuda_model(*inputs) + + # PyTorch模型计时 + torch.cuda.synchronize() + start_time = time.time() + for _ in range(num_iterations): + _ = torch_model(*inputs) + torch.cuda.synchronize() + torch_time = (time.time() - start_time) / num_iterations + + # 自定义CUDA内核计时 + torch.cuda.synchronize() + start_time = time.time() + for _ in range(num_iterations): + _ = cuda_model(*inputs) + torch.cuda.synchronize() + cuda_time = (time.time() - start_time) / num_iterations + + print(f"PyTorch内置Swish平均执行时间: {torch_time:.6f}秒") + print(f"自定义CUDA Swish平均执行时间: {cuda_time:.6f}秒") + speedup = torch_time / cuda_time if cuda_time > 0 else 0 + print(f"加速比 (Speedup): {speedup:.2f}x") + + return precision_flag, speedup + +if __name__ == "__main__": + precision_flag, speedup = run_benchmark() \ No newline at end of file