From d10b395fcf78e9cdc6ef5c2cdb07373993d263dd Mon Sep 17 00:00:00 2001 From: hli28146 Date: Tue, 18 Nov 2025 10:35:21 +0800 Subject: [PATCH] finish lppool1d #6 --- S1/hli28146_#6/run_code.py | 86 +++++++++++++++++++------------------- 1 file changed, 42 insertions(+), 44 deletions(-) diff --git a/S1/hli28146_#6/run_code.py b/S1/hli28146_#6/run_code.py index b38bbfa..f08b83d 100644 --- a/S1/hli28146_#6/run_code.py +++ b/S1/hli28146_#6/run_code.py @@ -1,76 +1,74 @@ +########################################################### +# 性能和精度验证程序 +########################################################### import torch +import torch.nn as nn import time -from lppool1d_torch import Model, get_inputs, get_init_inputs +from lppool1d_torch import Model,get_inputs,get_init_inputs from lppool1d_cuda import ModelNew - + def run_benchmark(): + # 检查 CUDA 是否可用 if not torch.cuda.is_available(): - print("CUDA 不可用") + print("CUDA 不可用,请确保您有可用的 NVIDIA GPU 并已正确安装 PyTorch CUDA 版本。") return - - device = torch.device("cuda") - - # 准备输入数据 - inputs = [x.cuda(device=device) for x in get_inputs()] - init_inputs = [x.cuda(device=device) if isinstance(x, torch.Tensor) else x for x in get_init_inputs()] + else: + device = torch.device("cuda") + # 初始化模型 + init_inputs = get_init_inputs() + init_inputs = [ + x.cuda(device=device) if isinstance(x, torch.Tensor) else x for x in init_inputs + ] + inputs = get_inputs() + inputs = [ + x.cuda(device=device) if isinstance(x, torch.Tensor) else x for x in inputs + ] + torch_model = Model(*init_inputs).cuda() cuda_model = ModelNew(*init_inputs).cuda() - + torch_model.eval() cuda_model.eval() - + print("-------------------- 精度对齐验证 --------------------") with torch.no_grad(): - # 预热GPU - _ = torch_model(*inputs) - _ = cuda_model(*inputs) - - # 正式测试 - output_torch = torch_model(*inputs) + output_torch = torch_model( *inputs) output_cuda = cuda_model(*inputs) - # 精度验证 - abs_diff = torch.abs(output_torch - output_cuda) - max_diff = torch.max(abs_diff).item() - mean_diff = torch.mean(abs_diff).item() - - if max_diff < 1e-4 and mean_diff < 1e-5: - print(f"✅ 精度对齐:最大误差 {max_diff:.6f},平均误差 {mean_diff:.6f}") - precision_flag = True + precision_flag = torch.allclose(output_torch, output_cuda,rtol=1e-03) + if precision_flag: + print("✅ 精度对齐:两个模型的输出结果非常接近。") else: - print(f"❌ 精度不一致:最大误差 {max_diff:.6f},平均误差 {mean_diff:.6f}") - precision_flag = False - + print("❌ 精度不一致!") + print("\n-------------------- 性能加速比测试 --------------------") num_iterations = 100 - # 预热GPU - for _ in range(10): - _ = torch_model(*inputs) - _ = cuda_model(*inputs) - - # PyTorch模型计时 + # PyTorch 模型计时 torch.cuda.synchronize() start_time = time.time() for _ in range(num_iterations): - _ = torch_model(*inputs) + _ = torch_model(*inputs) torch.cuda.synchronize() torch_time = (time.time() - start_time) / num_iterations - # 自定义CUDA内核计时 + # 自定义 CUDA 内核计时 torch.cuda.synchronize() start_time = time.time() for _ in range(num_iterations): - _ = cuda_model(*inputs) + _ = cuda_model(*inputs) torch.cuda.synchronize() cuda_time = (time.time() - start_time) / num_iterations - print(f"PyTorch内置Swish平均执行时间: {torch_time:.6f}秒") - print(f"自定义CUDA Swish平均执行时间: {cuda_time:.6f}秒") - speedup = torch_time / cuda_time if cuda_time > 0 else 0 - print(f"加速比 (Speedup): {speedup:.2f}x") - - return precision_flag, speedup + print(f"PyTorch torch.relu 平均执行时间: {torch_time:.6f} 秒") + print(f"自定义 CUDA 内核 平均执行时间: {cuda_time:.6f} 秒") + speedup = 0 + if cuda_time > 0: + speedup = torch_time / cuda_time + print(f"加速比 (Speedup): {speedup:.2f}x") + else: + print("CUDA 内核执行时间为0,无法计算加速比。") + return precision_flag,speedup if __name__ == "__main__": - precision_flag, speedup = run_benchmark() \ No newline at end of file + precision_flag,speedup = run_benchmark() \ No newline at end of file