import torch import torch.nn as nn import torch.optim as optim # 1. Setup: Use your RTX 4070 Super device = torch.device('cuda' if torch.cuda.is_available() else 'cpu') print(f"Using device: {device}") # 2. Dummy Data: Predicting the next number (e.g., [1, 2, 3] -> 4) # In a real SLM, these would be tokenized text IDs. data = torch.tensor([[1.0, 2.0, 3.0], [2.0, 3.0, 4.0]], device=device) targets = torch.tensor([4.0, 5.0], device=device) # 3. Model: A very simple Linear Layer model = nn.Linear(3, 1).to(device) # 4. Optimizer & Loss: The "teacher" that corrects the model optimizer = optim.SGD(model.parameters(), lr=0.01) criterion = nn.MSELoss() # 5. Training Loop model.train() for epoch in range(100): optimizer.zero_grad() # Clear previous gradients outputs = model(data) # Forward pass: Make a prediction loss = criterion(outputs.flatten(), targets) # Measure how "wrong" it is loss.backward() # Backward pass: Calculate adjustments optimizer.step() # Update weights based on adjustments if (epoch+1) % 20 == 0: print(f'Epoch [{epoch+1}/100], Loss: {loss.item():.4f}') print("Training complete!")