def forward(self, idx): # idx = (B, T) token ids
tok = self.wte(idx) # (B,T,C) token embeddings
pos = self.wpe(positions) # (T,C) position embeddings
x = tok + pos # the residual stream starts here
for block in self.blocks: # × 12 identical layers
h = block.ln_1(x) # layernorm
q, k, v = block.attn.c_attn(h).split(C, dim=2)
att = (q @ k.transpose(-2,-1)) / sqrt(head_size)
att = att.masked_fill(causal_mask == 0, -inf)
att = softmax(att, dim=-1) # attention weights
y = att @ v # weighted sum of values
x = x + block.attn.c_proj(y) # residual add
h = block.ln_2(x) # layernorm
h = gelu(block.mlp.c_fc(h)) # up-project 4× + GELU
x = x + block.mlp.c_proj(h) # residual add
x = self.ln_f(x) # final layernorm
logits = x @ self.wte.weight.T # tied to the embeddings
probs = softmax(logits[:, -1]) # next-token distribution
return probs
00tokenize
1 / 18
The prompt is already token IDs — four of them.
Train guesses the next id. Serve samples it. Fine-tune nudges a few matrices.