- 🍨本文为🔗365天深度学习训练营中的学习记录博客
- 🍖原作者:K同学啊
本周是学习深度学习的第15周。编译器使用的是vscode,安装的是CPU版PyTorch:torch 2.12.0+cpu。
本周的学习内容跟J1周(深度学习第J1周:ResNet-50算法学习-CSDN博客)相似,数据集一样,本次学习只给出不一样的部分。
1 模型
本周使用了ResNetV1和ResNetV2进行训练。
ResNetV1部分:
class ResNetV1(nn.Module): def __init__(self, classes=3): super(ResNetV1, self).__init__() self.conv1 = nn.Sequential( nn.Conv2d(3, 64, 7, stride=2, padding=3, bias=False, padding_mode='zeros'), nn.BatchNorm2d(64), nn.ReLU(), nn.MaxPool2d(kernel_size=3, stride=2, padding=1) ) self.conv2 = nn.Sequential( ConvBlock(64, 3, [64, 64, 256], stride=1), IdentityBlock(256, 3, [64, 64, 256]), IdentityBlock(256, 3, [64, 64, 256]) ) self.conv3 = nn.Sequential( ConvBlock(256, 3, [128, 128, 512]), IdentityBlock(512, 3, [128, 128, 512]), IdentityBlock(512, 3, [128, 128, 512]), IdentityBlock(512, 3, [128, 128, 512]) ) self.conv4 = nn.Sequential( ConvBlock(512, 3, [256, 256, 1024]), IdentityBlock(1024, 3, [256, 256, 1024]), IdentityBlock(1024, 3, [256, 256, 1024]), IdentityBlock(1024, 3, [256, 256, 1024]), IdentityBlock(1024, 3, [256, 256, 1024]), IdentityBlock(1024, 3, [256, 256, 1024]) ) self.conv5 = nn.Sequential( ConvBlock(1024, 3, [512, 512, 2048]), IdentityBlock(2048, 3, [512, 512, 2048]), IdentityBlock(2048, 3, [512, 512, 2048]) ) self.pool = nn.AvgPool2d(kernel_size=7, stride=7, padding=0) self.fc = nn.Linear(2048, classes) def forward(self, x): x = self.conv1(x) x = self.conv2(x) x = self.conv3(x) x = self.conv4(x) x = self.conv5(x) x = self.pool(x) x = torch.flatten(x, start_dim=1) x = self.fc(x) return xResNetV2 的预激活残差块:BN -> ReLU -> Conv,相加后不再使用 ReLU。模型部分:
class PreActBlock(nn.Module): def __init__(self, in_channel, kernel_size, filters, stride=1): super(PreActBlock, self).__init__() filters1, filters2, filters3 = filters self.bn1 = nn.BatchNorm2d(in_channel) self.conv1 = nn.Conv2d(in_channel, filters1, 1, stride=stride, bias=False) self.bn2 = nn.BatchNorm2d(filters1) self.conv2 = nn.Conv2d(filters1, filters2, kernel_size, padding=autopad(kernel_size), bias=False) self.bn3 = nn.BatchNorm2d(filters2) self.conv3 = nn.Conv2d(filters2, filters3, 1, bias=False) self.relu = nn.ReLU() self.shortcut = None if stride != 1 or in_channel != filters3: self.shortcut = nn.Conv2d(in_channel, filters3, 1, stride=stride, bias=False) def forward(self, x): preact = self.relu(self.bn1(x)) shortcut = x if self.shortcut is None else self.shortcut(preact) x1 = self.conv1(preact) x1 = self.conv2(self.relu(self.bn2(x1))) x1 = self.conv3(self.relu(self.bn3(x1))) return x1 + shortcut # ResNet-50 V2 class ResNetV2(nn.Module): def __init__(self, classes=3): super(ResNetV2, self).__init__() self.conv1 = nn.Sequential( nn.Conv2d(3, 64, 7, stride=2, padding=3, bias=False), nn.MaxPool2d(kernel_size=3, stride=2, padding=1) ) self.conv2 = nn.Sequential( PreActBlock(64, 3, [64, 64, 256]), PreActBlock(256, 3, [64, 64, 256]), PreActBlock(256, 3, [64, 64, 256]) ) self.conv3 = nn.Sequential( PreActBlock(256, 3, [128, 128, 512], stride=2), PreActBlock(512, 3, [128, 128, 512]), PreActBlock(512, 3, [128, 128, 512]), PreActBlock(512, 3, [128, 128, 512]) ) self.conv4 = nn.Sequential( PreActBlock(512, 3, [256, 256, 1024], stride=2), PreActBlock(1024, 3, [256, 256, 1024]), PreActBlock(1024, 3, [256, 256, 1024]), PreActBlock(1024, 3, [256, 256, 1024]), PreActBlock(1024, 3, [256, 256, 1024]), PreActBlock(1024, 3, [256, 256, 1024]), ) self.conv5 = nn.Sequential( PreActBlock(1024, 3, [512, 512, 2048], stride=2), PreActBlock(2048, 3, [512, 512, 2048]), PreActBlock(2048, 3, [512, 512, 2048]) ) self.bn = nn.BatchNorm2d(2048) self.relu = nn.ReLU() self.pool = nn.AvgPool2d(kernel_size=7, stride=7, padding=0) self.fc = nn.Linear(2048, classes) def forward(self, x): x = self.conv1(x) x = self.conv2(x) x = self.conv3(x) x = self.conv4(x) x = self.conv5(x) x = self.relu(self.bn(x)) x = self.pool(x) x = torch.flatten(x, start_dim=1) x = self.fc(x) return x2 训练过程
train和test步骤一样,绘图部分:
for model_name, model_class in [('ResNetV1', ResNetV1), ('ResNetV2', ResNetV2)]: print(f'\nTraining {model_name}') model = model_class(classes=len(total_data.classes)).to(device) optimizer = torch.optim.AdamW(model.parameters(),lr= 1e-4) loss_fn = nn.CrossEntropyLoss() epochs =10 train_loss =[] train_acc =[] test_loss =[] test_acc =[] best_acc =-1 for epoch in range(epochs): model.train() epoch_train_acc,epoch_train_loss = train(train_dl,model,loss_fn,optimizer) model.eval() epoch_test_acc,epoch_test_loss = test(test_dl,model,loss_fn) if epoch_test_acc > best_acc: best_acc = epoch_test_acc best_model = copy.deepcopy(model) train_acc.append(epoch_train_acc) train_loss.append(epoch_train_loss) test_acc.append(epoch_test_acc) test_loss.append(epoch_test_loss) lr =optimizer.state_dict()['param_groups'][0]['lr'] template =('Epoch:{:2d}, Train_acc:{:.1f}%, Train_loss:{:.3f}, Test_acc:{:.1f}%, Test_loss:{:.3f}, Lr:{:.2E}') print(template.format(epoch+1,epoch_train_acc*100,epoch_train_loss, epoch_test_acc*100,epoch_test_loss,lr)) PATH = f'./{model_name}_best_model.pth' torch.save(best_model.state_dict(), PATH) print('Done') # Plot training and validation metrics current_time = datetime.now() # 获取当前时间 epochs_range = range(epochs) plt.figure(figsize=(12, 3)) plt.subplot(1, 2, 1) plt.plot(epochs_range, train_acc, label='Training Accuracy') plt.plot(epochs_range, test_acc, label='Test Accuracy') plt.legend(loc='lower right') plt.title(f'{model_name} Accuracy') plt.xlabel(current_time) # 打卡需带出时间戳,否则代码截图无效 plt.subplot(1, 2, 2) plt.plot(epochs_range, train_loss, label='Training Loss') plt.plot(epochs_range, test_loss, label='Test Loss') plt.legend(loc='upper right') plt.title(f'{model_name} Loss') plt.tight_layout() plt.savefig(f'./{model_name}_ACC_LOSS.png') best_model.load_state_dict(torch.load(PATH,map_location=device)) epoch_test_acc,epoch_test_loss =test(test_dl,best_model,loss_fn) print(epoch_test_acc) print(epoch_test_loss) plt.show()3 训练结果
进行ResNetV1训练:
进行ResNetV2训练:
4 个人总结
ResNetV1 和 ResNetV2 均采用 ResNet-50 结构,主要区别在于残差块中归一化与激活操作的位置。V1 采用“卷积→BN→ReLU”的顺序,并在残差相加后再次进行 ReLU 激活;V2 采用“BN→ReLU→卷积”的预激活方式,相加后直接输出,使信息和梯度能够更顺畅地传播。不过,结构上的改进并不保证 V2 在所有任务中都取得更高准确率。
从本次 10 轮训练曲线来看,两个模型的测试准确率均接近 90%,训练损失总体下降,未出现明显的持续过拟合趋势。V1 最终训练准确率约为 89.5%,测试损失约为 0.26;V2 最终训练准确率约为 87%,测试损失约为 0.30,前期波动较大、后期趋于稳定。因此,本次实验中两者分类准确率接近,V1 的最终损失略低;但由于类别数量不均衡,仍需结合各类别召回率和混淆矩阵,进一步判断骨折识别效果。