In [9]:
# -*- coding: utf-8 -*-
import torch

dtype = torch.float
device = torch.device("cpu")
# device = torch.device("cuda:0") # Uncomment this to run on GPU

# N is batch size; D_in is input dimension;
# H is hidden dimension; D_out is output dimension.
N, D_in, H, D_out = 64, 1000, 100, 10

# Create random Tensors to hold input and outputs.
# Setting requires_grad=False indicates that we do not need to compute gradients
# with respect to these Tensors during the backward pass.
x = torch.randn(N, D_in, device=device, dtype=dtype)
y = torch.randn(N, D_out, device=device, dtype=dtype)

# Create random Tensors for weights.
# Setting requires_grad=True indicates that we want to compute gradients with
# respect to these Tensors during the backward pass.
w1 = torch.randn(D_in, H, device=device, dtype=dtype, requires_grad=True)
w2 = torch.randn(H, D_out, device=device, dtype=dtype, requires_grad=True)

learning_rate = 1e-6
for t in range(500):
    # Forward pass: compute predicted y using operations on Tensors; these
    # are exactly the same operations we used to compute the forward pass using
    # Tensors, but we do not need to keep references to intermediate values since
    # we are not implementing the backward pass by hand.
    y_pred = x.mm(w1).clamp(min=0).mm(w2)

    # Compute and print loss using operations on Tensors.
    # Now loss is a Tensor of shape (1,)
    # loss.item() gets the a scalar value held in the loss.
    loss = (y_pred - y).pow(2).sum()
    print(t, loss.item())

    # Use autograd to compute the backward pass. This call will compute the
    # gradient of loss with respect to all Tensors with requires_grad=True.
    # After this call w1.grad and w2.grad will be Tensors holding the gradient
    # of the loss with respect to w1 and w2 respectively.
    loss.backward()

    # Manually update weights using gradient descent. Wrap in torch.no_grad()
    # because weights have requires_grad=True, but we don't need to track this
    # in autograd.
    # An alternative way is to operate on weight.data and weight.grad.data.
    # Recall that tensor.data gives a tensor that shares the storage with
    # tensor, but doesn't track history.
    # You can also use torch.optim.SGD to achieve this.
    with torch.no_grad():
        w1 -= learning_rate * w1.grad
        w2 -= learning_rate * w2.grad
        print(w1.shape,w2.shape)

        # Manually zero the gradients after updating weights
        w1.grad.zero_()
        w2.grad.zero_()

0 34270984.0
torch.Size([1000, 100]) torch.Size([100, 10])
1 37849660.0
torch.Size([1000, 100]) torch.Size([100, 10])
2 50401220.0
torch.Size([1000, 100]) torch.Size([100, 10])
3 60393836.0
torch.Size([1000, 100]) torch.Size([100, 10])
4 52560984.0
torch.Size([1000, 100]) torch.Size([100, 10])
5 28192658.0
torch.Size([1000, 100]) torch.Size([100, 10])
6 9921597.0
torch.Size([1000, 100]) torch.Size([100, 10])
7 3359692.5
torch.Size([1000, 100]) torch.Size([100, 10])
8 1664241.25
torch.Size([1000, 100]) torch.Size([100, 10])
9 1151527.375
torch.Size([1000, 100]) torch.Size([100, 10])
10 912742.3125
torch.Size([1000, 100]) torch.Size([100, 10])
11 755491.625
torch.Size([1000, 100]) torch.Size([100, 10])
12 635763.375
torch.Size([1000, 100]) torch.Size([100, 10])
13 540042.25
torch.Size([1000, 100]) torch.Size([100, 10])
14 462081.6875
torch.Size([1000, 100]) torch.Size([100, 10])
15 397924.625
torch.Size([1000, 100]) torch.Size([100, 10])
16 344597.09375
torch.Size([1000, 100]) torch.Size

In [7]:
import visdom
vis = visdom.Visdom()

trace = dict(x=[1, 2, 3], y=[4, 5, 6], mode="markers+lines", type='custom',
             marker={'color': 'red', 'symbol': 104, 'size': "10"},
             text=["one", "two", "three"], name='1st Trace')
layout = dict(title="First Plot", xaxis={'title': 'x1'}, yaxis={'title': 'x2'})

vis._send({'data': [trace], 'layout': layout, 'win': 'mywin'})

'mywin'