12 KiB
12 KiB
In [1]:
import torch
from torch import nnIn [2]:
net = nn.Sequential(nn.LazyLinear(8),
nn.ReLU(),
nn.LazyLinear(1))
X = torch.rand(size=(2, 4))
net(X).shapeOut [2]:
torch.Size([2, 1])
In [3]:
net[2].state_dict()Out [3]:
OrderedDict([('weight',
tensor([[-0.1649, 0.0605, 0.1694, -0.2524, 0.3526, -0.3414, -0.2322, 0.0822]])),
('bias', tensor([0.0709]))])In [4]:
type(net[2].bias), net[2].bias.dataOut [4]:
(torch.nn.parameter.Parameter, tensor([0.0709]))
In [5]:
net[2].weight.grad == NoneOut [5]:
True
In [6]:
[(name, param.shape) for name, param in net.named_parameters()]Out [6]:
[('0.weight', torch.Size([8, 4])),
('0.bias', torch.Size([8])),
('2.weight', torch.Size([1, 8])),
('2.bias', torch.Size([1]))]In [7]:
# We need to give the shared layer a name so that we can refer to its
# parameters
shared = nn.LazyLinear(8)
net = nn.Sequential(nn.LazyLinear(8), nn.ReLU(),
shared, nn.ReLU(),
shared, nn.ReLU(),
nn.LazyLinear(1))
net(X)
# Check whether the parameters are the same
print(net[2].weight.data[0] == net[4].weight.data[0])
net[2].weight.data[0, 0] = 100
# Make sure that they are actually the same object rather than just having the
# same value
print(net[2].weight.data[0] == net[4].weight.data[0])tensor([True, True, True, True, True, True, True, True]) tensor([True, True, True, True, True, True, True, True])