One of the standard image processing examples is to use the CIFAR-10 image dataset. I am continuously refining my PyTorch skills so I decided to revisit the CIFAR-10 example.
CIFAR-10 has 60,000 images, divided into 50,000 training and 10,000 test images. Each image is 3-channel color with 32×32 pixels. Each pixel value is between 0 and 255. There are 10 kinds/classes of images: ‘plane’, ‘car’, ‘bird’, ‘cat’, ‘deer’, ‘dog’, ‘frog’, ‘horse’, ‘ship’, ‘truck’.
The PyTorch documentation gives a very good example of creating a CNN (convolutional neural network) for CIFAR-10. But that example is in a Jupyter notebook (I prefer ordinary code), and it has a lot of extras (such as analyzing accuracy by class). My goal was to write a simplified version that has just the essentials. Here “simplified” is relative — CNNs are very complicated.
My simplified CIFAR-10 demo has 130 lines of code. When I started writing this blog post, it was my intention to explain the demo program in detail. As I thought it out, I realized that even a brief explanation would require many pages. So, I’ll just pluck out a few things.
In the definition of the Net class, the demo uses two convolution layers and three linear layers:
self.conv1 = T.nn.Conv2d(3, 6, 5) # in, out, kernel self.conv2 = T.nn.Conv2d(6, 16, 5) self.pool = T.nn.MaxPool2d(2, 2) # kernel, stride self.fc1 = T.nn.Linear(16 * 5 * 5, 120) self.fc2 = T.nn.Linear(120, 84) self.fc3 = T.nn.Linear(84, 10)
There are hundreds of possible architecture designs so in practice you just have to use intuition based on experience. Dealing with the various shapes of the layers is one of the trickiest parts for me.
When I write an accuracy() function, I usually like to process one input item at a time so I can insert print() statements when things go wrong. But processing item-by-item is too slow for large datasets like CIFAR-10 so I had to use an aggregate approach:
def accuracy(model, ds):
ldr = T.utils.data.DataLoader(ds,
batch_size=len(ds), shuffle=False)
n_correct = 0
for data in ldr:
(pixels, labels) = data
with T.no_grad():
oupts = model(pixels)
(_, predicteds) = T.max(oupts, 1)
n_correct += (predicteds == labels).sum().item()
acc = (n_correct * 1.0) / len(ds)
return acc
Every CIFAR-10 example I’ve seen does a double normalization. The ToTensor() transform does an invisible divide each pixel value by 255 normalization. But then it’s standard to apply an additional Normalize() that does z-score normalization. I removed the z-score normalization that was included in the documentation example.
A few years ago, a neural network was just a neural network. But now, in my opinion, image processing problems, natural language processing problems, and tabular data processing problems have become very complex and are distinct fields of study.
Neural network image classifiers would have trouble dealing with these images.
# cifar_basic_cnn.py
# PyTorch 1.6.0-CPU Anaconda3-2020.02 Python 3.7.6
# Windows 10
import numpy as np
import torch as T
import torchvision as tv
device = T.device("cpu")
class Net(T.nn.Module):
def __init__(self):
super(Net, self).__init__()
self.conv1 = T.nn.Conv2d(3, 6, 5) # in, out, kernel
self.conv2 = T.nn.Conv2d(6, 16, 5)
self.pool = T.nn.MaxPool2d(2, 2) # kernel, stride
self.fc1 = T.nn.Linear(16 * 5 * 5, 120)
self.fc2 = T.nn.Linear(120, 84)
self.fc3 = T.nn.Linear(84, 10)
def forward(self, x):
z = T.nn.functional.relu(self.conv1(x))
z = self.pool(z)
z = T.nn.functional.relu(self.conv2(z))
z = self.pool(z)
z = z.view(-1, 16 * 5 * 5)
z = T.nn.functional.relu(self.fc1(z))
z = T.nn.functional.relu(self.fc2(z))
z = self.fc3(z)
return z
def accuracy(model, ds):
ldr = T.utils.data.DataLoader(ds,
batch_size=len(ds), shuffle=False)
n_correct = 0
for data in ldr:
(pixels, labels) = data
with T.no_grad():
oupts = model(pixels)
(_, predicteds) = T.max(oupts, 1)
n_correct += (predicteds == labels).sum().item()
acc = (n_correct * 1.0) / len(ds)
return acc
def main():
# 0. setup
print("\nBegin CIFAR-10 demo ")
np.random.seed(1)
T.manual_seed(1)
# 1. create Dataset
print("\nCreating Dataset ")
# trfm = tv.transforms.Compose([tv.transforms.ToTensor(),
# tv.transforms.Normalize((0.5, 0.5, 0.5),
# (0.5, 0.5, 0.5))]) # u, sd
trfm = tv.transforms.Compose([ tv.transforms.ToTensor() ])
train_ds = tv.datasets.CIFAR10(root=".\\Data",
train=True, download=True, transform=trfm)
bat_size = 10
train_ldr = T.utils.data.DataLoader(train_ds,
batch_size=bat_size, shuffle=True)
# 2. create network
print("\nCreating CNN with 2 conv and 400-120-84-10 ")
net = Net().to(device)
# 3. train model
max_epochs = 50
ep_log_interval = 5
lrn_rate = 0.005
loss_func = T.nn.CrossEntropyLoss() # does log-softmax()
optimizer = T.optim.SGD(net.parameters(), lr=lrn_rate)
print("\nbat_size = %3d " % bat_size)
print("loss = " + str(loss_func))
print("optimizer = SGD")
print("max_epochs = %3d " % max_epochs)
print("lrn_rate = %0.3f " % lrn_rate)
print("\nStarting training")
net = net.train()
for epoch in range(0, max_epochs):
epoch_loss = 0 # for one full epoch
num_lines_read = 0
for (batch_idx, batch) in enumerate(train_ldr):
(X, Y) = batch # X = pixels, Y = target labels
num_lines_read += bat_size # not exactly
# if num_lines_read >= 10:
# continue
optimizer.zero_grad()
oupt = net(X) # X is Size([bat_size, 3, 32, 32])
loss_obj = loss_func(oupt, Y) # a tensor
epoch_loss += loss_obj.item() # accumulate
loss_obj.backward()
optimizer.step()
if epoch % ep_log_interval == 0:
print("epoch = %4d loss = %0.4f" % (epoch, epoch_loss))
print("Done ")
# 4. evaluate model accuracy
print("\nComputing model accuracy")
net = net.eval()
acc_train = accuracy(net, train_ds) # item-by-item
print("Accuracy on training data = %0.4f" % acc_train)
# 5. make a prediction
print("\nPredicting image for random 32x32 pixels ")
cifar = ['plane', 'car', 'bird', 'cat', 'deer',
'dog', 'frog', 'horse', 'ship', 'truck']
inpt = np.random.random((1, 3, 32, 32))
inpt = T.tensor(inpt, dtype=T.float32).to(device)
with T.no_grad():
oupt = net(inpt) # like [[-0.12, 1.03, 0.55, . . ]]
am = T.argmax(oupt) # 0 to 9
print("\nPredicted class is " + cifar[am] )
# 6. save model
print("\nSaving trained model state")
fn = ".\\Models\\cifar_model.pth"
T.save(net.state_dict(), fn)
print("\nEnd CIFAR-10 demo demo")
if __name__ == "__main__":
main()



.NET Test Automation Recipes
Software Testing
SciPy Programming Succinctly
Keras Succinctly
R Programming
Visual Studio Live
Microsoft MLADS Conference
DevIntersection Conference
Machine Learning Week
Ai4 Conference
G2E Conference
iSC West Conference
Thank you for your example! I’m just learning PyTorch and this really helps.
I’m running the example on Linux and I’m using a NVIDIA GPU. I changed the device to cuda and when I run, I get an error about a mismatch in one of the layers. The model (network) class is below (I added a couple of lines but it doesn’t change anything).
class Net(T.nn.Module):
def __init__(self):
super(Net, self).__init__()
self.conv1 = T.nn.Conv2d(3, 6, 5) # in, out, kernel
self.conv2 = T.nn.Conv2d(6, 16, 5)
self.pool = T.nn.MaxPool2d(2, 2) # kernel, stride
self.fc1 = T.nn.Linear(16 * 5 * 5, 120)
self.fc2 = T.nn.Linear(120, 84)
self.fc3 = T.nn.Linear(84, 10)
# end def
def forward(self, x):
z = T.nn.functional.relu(self.conv1(x))
z = self.pool(z)
z = T.nn.functional.relu(self.conv2(z))
z = self.pool(z)
z = z.view(-1, 16 * 5 * 5)
z = T.nn.functional.relu(self.fc1(z))
z = T.nn.functional.relu(self.fc2(z))
z = self.fc3(z)
return z
# end def
# end class
It is complaining about the first line in the forward function:
z = T.nn.functional.relu(self.conv1(x))
The error is the following (the line numbers won’t match your code above).
File “pytorch-cifar10-case3.py”, line 161, in
main()
File “pytorch-cifar10-case3.py”, line 114, in main
oupt = net(X) # X is Size([bat_size, 3, 32, 32])
File “/home/laytonjb/anaconda3/lib/python3.8/site-packages/torch/nn/modules/module.py”, line 727, in _call_impl
result = self.forward(*input, **kwargs)
File “pytorch-cifar10-case3.py”, line 28, in forward
z = T.nn.functional.relu(self.conv1(x))
File “/home/laytonjb/anaconda3/lib/python3.8/site-packages/torch/nn/modules/module.py”, line 727, in _call_impl
result = self.forward(*input, **kwargs)
File “/home/laytonjb/anaconda3/lib/python3.8/site-packages/torch/nn/modules/conv.py”, line 423, in forward
return self._conv_forward(input, self.weight)
File “/home/laytonjb/anaconda3/lib/python3.8/site-packages/torch/nn/modules/conv.py”, line 419, in _conv_forward
return F.conv2d(input, weight, self.bias, self.stride,
RuntimeError: Input type (torch.FloatTensor) and weight type (torch.cuda.FloatTensor) should be the same
When I changed the device to “cpu”, then the training completes correctly.
I know that PyTorch sees the GPU (there are actually two), because I can run some other PyTorch code on the GPU without difficulty.
Could you provide any help in fixing this issue?
Thanks!
Jeff
The error indicates that the input to the NN is on cpu instead of cuda. In the demo code, the input is:
inpt = np.random.random((1, 3, 32, 32))
inpt = T.tensor(inpt, dtype=T.float32).to(device)
So, are you sure you applied .to(device) on your input?
I ran the demo on a Linux machine with GPU without any errors . . .
I’m pretty sure I do. At the top of the main routine I have:
device = T.device(“cuda”)
I hate to post the entire code, but just in case, here it is: (apologies)
# cifar_basic_cnn.py
# PyTorch 1.6.0-CPU Anaconda3-2020.02 Python 3.7.6
# Windows 10
# https://jamesmccaffrey.wordpress.com/2020/10/29/yet-another-cifar-10-example-using-pytorch/
import numpy as np
import torch as T
import torchvision as tv
class Net(T.nn.Module):
def __init__(self):
super(Net, self).__init__()
self.conv1 = T.nn.Conv2d(3, 6, 5) # in, out, kernel
self.conv2 = T.nn.Conv2d(6, 16, 5)
self.pool = T.nn.MaxPool2d(2, 2) # kernel, stride
self.fc1 = T.nn.Linear(16 * 5 * 5, 120)
self.fc2 = T.nn.Linear(120, 84)
self.fc3 = T.nn.Linear(84, 10)
# end def
def forward(self, x):
z = T.nn.functional.relu(self.conv1(x))
z = self.pool(z)
z = T.nn.functional.relu(self.conv2(z))
z = self.pool(z)
z = z.view(-1, 16 * 5 * 5)
z = T.nn.functional.relu(self.fc1(z))
z = T.nn.functional.relu(self.fc2(z))
z = self.fc3(z)
return z
# end def
# end class
def accuracy(model, ds):
ldr = T.utils.data.DataLoader(ds,
batch_size=len(ds), shuffle=False)
n_correct = 0
for data in ldr:
(pixels, labels) = data
with T.no_grad():
oupts = model(pixels)
(_, predicteds) = T.max(oupts, 1)
n_correct += (predicteds == labels).sum().item()
# end with
# end
acc = (n_correct * 1.0) / len(ds)
return acc
def main():
# cpu, cuda, mkldnn, opengl, opencl, ideep, hip, msnpu, xla
device = T.device(“cuda”)
# 0. setup
print(” “)
print(“Begin CIFAR-10 demo “)
np.random.seed(1)
T.manual_seed(1)
# 1. create Dataset
print(” “)
print(“Creating Dataset “)
# trfm = tv.transforms.Compose([tv.transforms.ToTensor(),
# tv.transforms.Normalize((0.5, 0.5, 0.5),
# (0.5, 0.5, 0.5))]) # u, sd
trfm = tv.transforms.Compose([ tv.transforms.ToTensor() ])
train_ds = tv.datasets.CIFAR10(root=”.\\Data”,
train=True, download=True, transform=trfm)
bat_size = 10
train_ldr = T.utils.data.DataLoader(train_ds,
batch_size=bat_size, shuffle=True)
# 2. create network
print(” “)
print(“Creating CNN with 2 conv and 400-120-84-10 “)
net = Net().to(device)
# 3. train model
max_epochs = 50
ep_log_interval = 5
lrn_rate = 0.005
loss_func = T.nn.CrossEntropyLoss() # does log-softmax()
optimizer = T.optim.SGD(net.parameters(), lr=lrn_rate)
print(” “)
print(“bat_size = %3d ” % bat_size)
print(“loss = ” + str(loss_func))
print(“optimizer = SGD”)
print(“max_epochs = %3d ” % max_epochs)
print(“lrn_rate = %0.3f ” % lrn_rate)
print(” “)
print(“Starting training”)
net = net.train()
for epoch in range(0, max_epochs):
epoch_loss = 0 # for one full epoch
num_lines_read = 0
for (batch_idx, batch) in enumerate(train_ldr):
(X, Y) = batch # X = pixels, Y = target labels
num_lines_read += bat_size # not exactly
# if num_lines_read >= 10:
# continue
optimizer.zero_grad()
oupt = net(X) # X is Size([bat_size, 3, 32, 32])
loss_obj = loss_func(oupt, Y) # a tensor
epoch_loss += loss_obj.item() # accumulate
loss_obj.backward()
optimizer.step()
# end for
print(“epoch = %4d loss = %0.4f” % (epoch, epoch_loss))
# end for
print(“Done “)
# 4. evaluate model accuracy
print(” “)
print(“Computing model accuracy”)
net = net.eval()
acc_train = accuracy(net, train_ds) # item-by-item
print(“Accuracy on training data = %0.4f” % acc_train)
# 5. make a prediction
print(” “)
print(“Predicting image for random 32×32 pixels “)
cifar = [‘plane’, ‘car’, ‘bird’, ‘cat’, ‘deer’,
‘dog’, ‘frog’, ‘horse’, ‘ship’, ‘truck’]
inpt = np.random.random((1, 3, 32, 32))
inpt = T.tensor(inpt, dtype=T.float32).to(device)
with T.no_grad():
oupt = net(inpt) # like [[-0.12, 1.03, 0.55, . . ]]
# end with
am = T.argmax(oupt) # 0 to 9
print(” “)
print(“Predicted class is ” + cifar[am] )
# 6. save model
print(” “)
print(“Saving trained model state”)
fn = “.\\Models\\cifar_model.pth”
T.save(net.state_dict(), fn)
print(” “)
print(“End CIFAR-10 demo demo”)
# end def
if __name__ == “__main__”:
main()
# end if