Tensors
import torch
torch.randn(32, 128) # random normal
torch.zeros(3, 3); torch.arange(10)
x.to("cuda"); x.shape; x.dtype
x.view(-1, 64); x.reshape(2, -1)
x.permute(0, 2, 1); x.unsqueeze(0)
torch.cat([a, b], dim=1); torch.stack([a, b])
x.requires_grad_(True); x.detach()
Layers
Core
nn.Linear(in, out)
nn.Conv2d(in_c, out_c, kernel_size=3, padding=1)
nn.MaxPool2d(2); nn.AdaptiveAvgPool2d(1)
nn.LSTM(in, hidden, num_layers=2, batch_first=True)
nn.GRU(in, hidden, batch_first=True)
nn.MultiheadAttention(embed_dim, num_heads, batch_first=True)
nn.TransformerEncoderLayer(d_model, nhead, batch_first=True)
nn.Embedding(vocab, dim)
Regularization & Norm
nn.Dropout(0.3)
nn.BatchNorm2d(channels)
nn.LayerNorm(dim)
nn.Sequential(...) # stack modules
Activations
| Name |
API |
| ReLU | nn.ReLU() |
| GELU | nn.GELU() |
| Sigmoid | nn.Sigmoid() |
| Tanh | nn.Tanh() |
| Softmax | nn.Softmax(dim=-1) |
Losses & Optimizers
nn.MSELoss() # regression
nn.L1Loss() # MAE
nn.BCEWithLogitsLoss() # binary (logits)
nn.CrossEntropyLoss() # multi-class (logits)
torch.optim.SGD(p, lr=0.1, momentum=0.9)
torch.optim.Adam(p, lr=1e-3)
torch.optim.AdamW(p, lr=3e-4, weight_decay=1e-2)
torch.optim.lr_scheduler.CosineAnnealingLR(opt, T_max=10)
torch.optim.lr_scheduler.OneCycleLR(opt, max_lr=1e-3, ...)
Training Loop
model.train()
for xb, yb in loader:
xb, yb = xb.to(device), yb.to(device)
optimizer.zero_grad()
loss = criterion(model(xb), yb)
loss.backward()
nn.utils.clip_grad_norm_(model.parameters(), 1.0)
optimizer.step()
scheduler.step()
# eval
model.eval()
with torch.no_grad():
preds = model(xb).argmax(dim=1)
Data & Persistence
from torch.utils.data import DataLoader, TensorDataset
ds = TensorDataset(X, y)
loader = DataLoader(ds, batch_size=64, shuffle=True,
num_workers=4, pin_memory=True)
torch.save(model.state_dict(), "m.pt")
model.load_state_dict(torch.load("m.pt"))
# mixed precision
scaler = torch.cuda.amp.GradScaler()
with torch.autocast("cuda"):
loss = criterion(model(xb), yb)