From af06d0bbdaa12cfa2112e3ec65058d9e16a0a984 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E5=90=B4=E5=A7=BF=E8=8A=9D?= <16377376+zizi05@user.noreply.gitee.com> Date: Sat, 22 Nov 2025 15:44:18 +0800 Subject: [PATCH] feat:add pairwise_distance #3 --- S1/zizi05_#3/pairwise_distance_cudacode.py | 89 +++++++++++++++ S1/zizi05_#3/pairwise_distance_torchcode.py | 41 +++++++ S1/zizi05_#3/prompt.txt | 113 ++++++++++++++++++++ S1/zizi05_#3/run_code.py | 74 +++++++++++++ 4 files changed, 317 insertions(+) create mode 100644 S1/zizi05_#3/pairwise_distance_cudacode.py create mode 100644 S1/zizi05_#3/pairwise_distance_torchcode.py create mode 100644 S1/zizi05_#3/prompt.txt create mode 100644 S1/zizi05_#3/run_code.py diff --git a/S1/zizi05_#3/pairwise_distance_cudacode.py b/S1/zizi05_#3/pairwise_distance_cudacode.py new file mode 100644 index 00000000..2e299c72 --- /dev/null +++ b/S1/zizi05_#3/pairwise_distance_cudacode.py @@ -0,0 +1,89 @@ +import torch +from torch.utils.cpp_extension import load_inline + +# 高效的成对距离 CUDA 源码 +pairwise_distance_source = """ +#include +#include +#include + +// Kernel to compute pairwise Euclidean distance +// Each thread computes one element of the output distance matrix. +__global__ void pairwise_distance_kernel( + const float* __restrict__ x, + const float* __restrict__ y, + float* __restrict__ dist, + int N, int M, int D +) { + // 2D thread mapping + int row = blockIdx.y * blockDim.y + threadIdx.y; // Maps to N + int col = blockIdx.x * blockDim.x + threadIdx.x; // Maps to M + + if (row < N && col < M) { + // Pointer to the start of the vectors for this thread's computation + const float* x_vec = x + row * D; + const float* y_vec = y + col * D; + + float sum_sq = 0.0f; + // Compute the sum of squared differences + for (int k = 0; k < D; ++k) { + float diff = x_vec[k] - y_vec[k]; + sum_sq += diff * diff; + } + + // Store the final distance + dist[row * M + col] = sqrtf(sum_sq); + } +} + +torch::Tensor pairwise_distance_cuda(torch::Tensor x, torch::Tensor y) { + TORCH_CHECK(x.is_cuda() && x.is_contiguous()); + TORCH_CHECK(y.is_cuda() && y.is_contiguous()); + TORCH_CHECK(x.dim() == 2 && y.dim() == 2); + TORCH_CHECK(x.size(1) == y.size(1), "Vector dimensions must match"); + + int N = x.size(0); + int M = y.size(0); + int D = x.size(1); + + auto dist = torch::empty({N, M}, x.options()); + + // Use a 2D grid and 2D blocks + dim3 block_size(16, 16); // A 16x16 thread block is a good starting point + dim3 grid_size((M + block_size.x - 1) / block_size.x, + (N + block_size.y - 1) / block_size.y); + + pairwise_distance_kernel<<>>( + x.data_ptr(), + y.data_ptr(), + dist.data_ptr(), + N, M, D + ); + + return dist; +} +""" + +pairwise_distance_cpp_source = """ +torch::Tensor pairwise_distance_cuda(torch::Tensor x, torch::Tensor y); +""" + +# 编译模块 +pairwise_distance_module = load_inline( + name="pairwise_distance", + cpp_sources=pairwise_distance_cpp_source, + cuda_sources=pairwise_distance_source, + functions=["pairwise_distance_cuda"], + verbose=True, + extra_cuda_cflags=["-O3", "--use_fast_math"] +) + +class ModelNew(torch.nn.Module): + def __init__(self): + super(ModelNew, self).__init__() + self.pairwise_distance = pairwise_distance_module + + def forward(self, x, y): + if not x.is_contiguous(): x = x.contiguous() + if not y.is_contiguous(): y = y.contiguous() + return self.pairwise_distance.pairwise_distance_cuda(x, y) diff --git a/S1/zizi05_#3/pairwise_distance_torchcode.py b/S1/zizi05_#3/pairwise_distance_torchcode.py new file mode 100644 index 00000000..bbe02704 --- /dev/null +++ b/S1/zizi05_#3/pairwise_distance_torchcode.py @@ -0,0 +1,41 @@ +import torch +import torch.nn as nn + +class Model(nn.Module): + """ + Computes the pairwise Euclidean distance matrix between two sets of vectors. + This implementation uses broadcasting, which is memory-intensive. + """ + def __init__(self): + super(Model, self).__init__() + + def forward(self, x: torch.Tensor, y: torch.Tensor) -> torch.Tensor: + """ + Args: + x (torch.Tensor): A tensor of shape (N, D). + y (torch.Tensor): A tensor of shape (M, D). + + Returns: + torch.Tensor: A tensor of shape (N, M) where Z[i, j] is the distance + between x[i] and y[j]. + """ + # x.unsqueeze(1) -> (N, 1, D) + # y.unsqueeze(0) -> (1, M, D) + # The result of subtraction is (N, M, D) + diff = x.unsqueeze(1) - y.unsqueeze(0) + # Compute L2 norm along the last dimension + dist_matrix = torch.norm(diff, p=2, dim=-1) + return dist_matrix + +# 测试数据配置 +N = 1024 # Number of vectors in the first set +M = 1024 # Number of vectors in the second set +D = 512 # Dimensionality of each vector + +def get_inputs(): + x = torch.randn(N, D) + y = torch.randn(M, D) + return [x, y] + +def get_init_inputs(): + return [] # No special initialization inputs needed diff --git a/S1/zizi05_#3/prompt.txt b/S1/zizi05_#3/prompt.txt new file mode 100644 index 00000000..40723a2f --- /dev/null +++ b/S1/zizi05_#3/prompt.txt @@ -0,0 +1,113 @@ +You write custom CUDA kernels to replace the PyTorch operators in the given Pairwise Euclidean Distance architecture to get speedups. +You have complete freedom to choose the set of operators you want to replace. You may make the decision to replace the combined broadcasting subtraction and L2 norm operators with a custom CUDA kernel (considering operator fusion opportunities to combine the element-wise difference calculation, sum of squares, and square root into a single kernel) or adjust algorithms for better performance. You are only limited by your imagination. + +Here's an example to show you the syntax of inline embedding custom CUDA operators in PyTorch (for reference of code structure, not functional alignment): +The example given architecture (sample structure): +import torch +import torch.nn as nn +import torch.nn.functional as F +class Model(nn.Module): + def __init__(self) -> None: + super().__init__() + def forward(self, a, b): + return a + b +def get_inputs(): + # randomly generate input tensors based on the model architecture + a = torch.randn(1, 128).cuda() + b = torch.randn(1, 128).cuda() + return [a, b] +def get_init_inputs(): + # randomly generate tensors required for initialization based on the model architecture + return [] + +The example new arch with custom CUDA kernels (sample structure): +import torch +import torch.nn as nn +import torch.nn.functional as F +from torch.utils.cpp_extension import load_inline +# Define custom CUDA kernel and load it inline +custom_add_source = """ +#include +#include +__global__ void custom_add_kernel(const float* a, const float* b, float* out, int size) { + int idx = blockIdx.x * blockDim.x + threadIdx.x; + if (idx < size) { + out[idx] = a[idx] + b[idx]; + } +} +torch::Tensor custom_add_cuda(torch::Tensor a, torch::Tensor b) { + auto size = a.numel(); + auto out = torch::empty_like(a); + const int block_size = 256; + int num_blocks = (size + block_size - 1) / block_size; + custom_add_kernel<<>>(a.data_ptr(), b.data_ptr(), out.data_ptr(), size); + return out; +} +""" +custom_add_cpp_source = "torch::Tensor custom_add_cuda(torch::Tensor a, torch::Tensor b);" +custom_add = load_inline( + name="custom_add", + cpp_sources=custom_add_cpp_source, + cuda_sources=custom_add_source, + functions=["custom_add_cuda"], + verbose=True +) +class Model(nn.Module): + def __init__(self) -> None: + super().__init__() + self.custom_add = custom_add + def forward(self, a, b): + return self.custom_add.custom_add_cuda(a, b) +def get_inputs(): + # randomly generate input tensors based on the model architecture + a = torch.randn(1, 128).cuda() + b = torch.randn(1, 128).cuda() + return [a, b] +def get_init_inputs(): + # randomly generate tensors required for initialization based on the model architecture + return [] + +You are given the following Pairwise Euclidean Distance architecture (base PyTorch implementation): +import torch +import torch.nn as nn +class Model(nn.Module): + """ + Pairwise Euclidean Distance function: Mathematical formulation is ||x_i - y_j||_2 for each pair (x_i, y_j), + where x_i is the i-th vector in input x (shape (N, D)), y_j is the j-th vector in input y (shape (M, D)), + and ||·||_2 denotes the L2 norm (Euclidean distance). + """ + def __init__(self): + super(Model, self).__init__() + + def forward(self, x: torch.Tensor, y: torch.Tensor) -> torch.Tensor: + """ + Computes the pairwise Euclidean distance matrix between two sets of vectors. + Args: + x (torch.Tensor): Input tensor with fixed shape (N, D) + where N=1024 (number of vectors in first set) and D=512 (dimensionality of each vector). + y (torch.Tensor): Input tensor with fixed shape (M, D) + where M=1024 (number of vectors in second set) and D=512 (dimensionality of each vector). + Returns: + torch.Tensor: Output tensor of shape (N, M) where each element [i, j] is the Euclidean distance between x[i] and y[j]. + """ + # x.unsqueeze(1) -> (N, 1, D), y.unsqueeze(0) -> (1, M, D) + # Broadcasting subtraction results in (N, M, D) + diff = x.unsqueeze(1) - y.unsqueeze(0) + # Compute L2 norm along the last dimension to get (N, M) distance matrix + dist_matrix = torch.norm(diff, p=2, dim=-1) + return dist_matrix + +# Fixed hyperparameters for input generation +N = 1024 # Number of vectors in the first set +M = 1024 # Number of vectors in the second set +D = 512 # Dimensionality of each vector + +def get_inputs(): + # Randomly generate input tensors matching the fixed shapes (N, D) and (M, D) + x = torch.randn(N, D) + y = torch.randn(M, D) + return [x, y] + +def get_init_inputs(): + # No special initialization tensors needed (model has no trainable parameters) + return [] \ No newline at end of file diff --git a/S1/zizi05_#3/run_code.py b/S1/zizi05_#3/run_code.py new file mode 100644 index 00000000..659038f9 --- /dev/null +++ b/S1/zizi05_#3/run_code.py @@ -0,0 +1,74 @@ +########################################################### +# 性能和精度验证程序 +########################################################### +import torch +import torch.nn as nn +import time +from pairwise_distance_torchcode import Model,get_inputs,get_init_inputs +from pairwise_distance_cudacode import ModelNew + +def run_benchmark(): + # 检查 CUDA 是否可用 + if not torch.cuda.is_available(): + print("CUDA 不可用,请确保您有可用的 NVIDIA GPU 并已正确安装 PyTorch CUDA 版本。") + return + else: + device = torch.device("cuda") + + # 初始化模型 + init_inputs = get_init_inputs() + init_inputs = [ + x.cuda(device=device) if isinstance(x, torch.Tensor) else x for x in init_inputs + ] + inputs = get_inputs() + inputs = [ + x.cuda(device=device) if isinstance(x, torch.Tensor) else x for x in inputs + ] + + torch_model = Model(*init_inputs).cuda() + cuda_model = ModelNew(*init_inputs).cuda() + + torch_model.eval() + cuda_model.eval() + + print("-------------------- 精度对齐验证 --------------------") + with torch.no_grad(): + output_torch = torch_model( *inputs) + output_cuda = cuda_model(*inputs) + + precision_flag = torch.allclose(output_torch, output_cuda,rtol=1e-03) + if precision_flag: + print("✅ 精度对齐:两个模型的输出结果非常接近。") + else: + print("❌ 精度不一致!") + + print("\n-------------------- 性能加速比测试 --------------------") + num_iterations = 100 + + # PyTorch 模型计时 + torch.cuda.synchronize() + start_time = time.time() + for _ in range(num_iterations): + _ = torch_model(*inputs) + torch.cuda.synchronize() + torch_time = (time.time() - start_time) / num_iterations + + # 自定义 CUDA 内核计时 + torch.cuda.synchronize() + start_time = time.time() + for _ in range(num_iterations): + _ = cuda_model(*inputs) + torch.cuda.synchronize() + cuda_time = (time.time() - start_time) / num_iterations + + print(f"PyTorch pairwise_distance 平均执行时间: {torch_time:.6f} 秒") + print(f"自定义 CUDA 内核 平均执行时间: {cuda_time:.6f} 秒") + speedup = 0 + if cuda_time > 0: + speedup = torch_time / cuda_time + print(f"加速比 (Speedup): {speedup:.2f}x") + else: + print("CUDA 内核执行时间为0,无法计算加速比。") + return precision_flag,speedup +if __name__ == "__main__": + precision_flag,speedup = run_benchmark() \ No newline at end of file