108 def gpt(token_id, pos_id, keys, values): # ★ 輸入:目前這個字母、它的位置 ⋯ 143 logits = linear(x, state_dict['lm_head']) 144 return logits # ★ 輸出:27 個字母各自的原始分數(還不是機率)
75 n_layer = 1 # depth of the transformer …
76 n_embd = 16 # width of the network …
77 block_size = 16 # maximum context length …
78 n_head = 4 # number of attention heads
79 head_dim = n_embd // n_head # derived dimension …
# ★ 五個超參數:層數・寬度・上限格數・頭數・每頭 4 維
80 matrix = lambda nout, nin, std=0.08:
↪ [[Value(random.gauss(0, std)) for _ in range(nin)]
↪ for _ in range(nout)]
# ★ matrix:造一個矩陣;Value=數字+一張便條
81 state_dict = {'wte': matrix(vocab_size, n_embd),
↪ 'wpe': matrix(block_size, n_embd),
↪ 'lm_head': matrix(vocab_size, n_embd)}
# ★ state_dict:知識收納櫃;vocab_size=27
82 for i in range(n_layer):
83 state_dict[f'layer{i}.attn_wq'] =
↪ matrix(n_embd, n_embd)
⋯
87 state_dict[f'layer{i}.mlp_fc1'] =
↪ matrix(4 * n_embd, n_embd)
⋯
89 params = [p for mat in state_dict.values()
↪ for row in mat for p in row] # flatten …
# ★ params:攤平成一條 list 的 4,192 個數字
90 print(f"num params: {len(params)}")
111 x = [t + p for t, p in zip(tok_emb, pos_emb)] # … # ★ x、tok_emb、pos_emb:主角本人+兩張便利貼 ⋯ 134 x = [a + b for a, b in zip(x, x_residual)] # ★ 左邊是新的 x,右邊的 x_residual 是舊的 x——舊的沒被丟掉 ⋯ 141 x = [a + b for a, b in zip(x, x_residual)] # ★ 一模一樣的一行:attention 之後一次,MLP 之後一次
50 def relu(self): return Value(max(0, self.data), ↪ (self,), (float(self.data > 0),)) # ★ self.data:數值本體;括號=便條 ⋯ 139 x = [xi.relu() for xi in x] # 16 個數字各自過門檻
# relu 的 _local_grads=(1,) 或 (0,) # 0 → 這一輪梯度完全不往回傳(圖 9.5)
103 def rmsnorm(x): # ★ rmsnorm:音控台 104 ms = sum(xi * xi for xi in x) / len(x) # ms:平均音量 105 scale = (ms + 1e-5) ** -0.5 # scale:調回標準的倍率 106 return [xi * scale for xi in x] # 只動音量,不動方向
112 x = rmsnorm(x) # 進主幹前 ⋯ 117 x = rmsnorm(x) # attention 之前 ⋯ 137 x = rmsnorm(x) # MLP 之前
30 class Value: ⋯ 33 def __init__(self, data, children=(), ↪ local_grads=()): 34 self.data = data # ★ 數值本體 # 前向只用得到 .data ⋯ 39 def __add__(self, other): ⋯ 41 return Value(self.data + other.data, ↪ (self, other), (1, 1)) # ★ 加法:算出和,順手寫下便條 # 「來源為這兩個,微分值都是 1」
108 def gpt(token_id, pos_id, keys, values): # ★ gpt:模型本體,僅此一個函數 109 tok_emb = state_dict['wte'][token_id] 110 pos_emb = state_dict['wpe'][pos_id] ⋯ 144 return logits # 27 個分數,如此而已 ⋯ 194 logits = gpt(token_id, pos_id, keys, values) # ★ 外面的迴圈這樣呼叫它
189 for sample_idx in range(20): # ★ sample_idx:第幾個名字 190 keys, values = [[] for _ in range(n_layer)], ↪ [[] for _ in range(n_layer)] # ★ keys、values:KV cache 本體——只進不出(見圖 9) 191 token_id = BOS # ★ token_id:現在手上這個字母的編號 192 sample = [] # ★ sample:已經生出來的字母們 193 for pos_id in range(block_size): 194 logits = gpt(token_id, pos_id, keys, values) # ★ gpt:整個模型=一個函數;logits:27 個分數 195 probs = softmax([l / temperature for l in logits]) # ★ probs:27 個機率;temperature 見圖 8 196 token_id = random.choices(range(vocab_size), ↪ weights=[p.data for p in probs])[0] 197 if token_id == BOS: 198 break 199 sample.append(uchars[token_id])
24 uchars = sorted(set(''.join(docs))) # unique characters …
# ★ uchars:字元排序表;docs:全部 32,033 個名字
25 BOS = len(uchars) # token id for a special Beginning …
# ★ BOS:開頭兼結尾的特殊符號=26(vocab_size=27 的由來)
26 vocab_size = len(uchars) + 1 # total number of unique …
157 tokens = [BOS] + [uchars.index(ch) for ch in doc] + [BOS] # ★ tokens:一筆名字變成的編號串 [26, 4, 12, 12, 0, 26]
196 token_id = random.choices(range(vocab_size), ↪ weights=[p.data for p in probs])[0] 197 if token_id == BOS: 198 break
108 def gpt(token_id, pos_id, keys, values): 109 tok_emb = state_dict['wte'][token_id] # token embedding 110 pos_emb = state_dict['wpe'][pos_id] # position embedding # 前述兩張便利貼——此處是它們的來源 111 x = [t + p for t, p in zip(tok_emb, pos_emb)] # joint … # 「相加」正式上場 112 x = rmsnorm(x) # note: not redundant due to backward … # 音控台(rmsnorm)正式上工
77 block_size = 16 # maximum context length ↪ of the attention window ↪ (note: the longest name is 15 characters)
81 state_dict = {'wte': matrix(vocab_size, n_embd),
↪ 'wpe': matrix(block_size, n_embd),
↪ 'lm_head': matrix(vocab_size, n_embd)}
# wpe 僅有 block_size=16 列
110 pos_emb = state_dict['wpe'][pos_id] # position … # pos_id 超過 15 → 這一行拋出 IndexError
114 for li in range(n_layer):
115 # 1) Multi-head Attention block
116 x_residual = x
117 x = rmsnorm(x)
118 q = linear(x, state_dict[f'layer{li}.attn_wq'])
# q、k、v 的定義見圖 4a-0,此處先看結構
⋯
133 x = linear(x_attn, state_dict[f'layer{li}.attn_wo'])
# ★ x_attn:四個 head 的輸出接起來(16 維,拼法見圖 4b-0)
134 x = [a + b for a, b in zip(x, x_residual)]
135 # 2) MLP block
136 x_residual = x
137 x = rmsnorm(x)
138 x = linear(x, state_dict[f'layer{li}.mlp_fc1'])
139 x = [xi.relu() for xi in x]
140 x = linear(x, state_dict[f'layer{li}.mlp_fc2'])
141 x = [a + b for a, b in zip(x, x_residual)]
121 keys[li].append(k) 122 values[li].append(v) ⋯ 129 attn_logits = [sum(q_h[j] * k_h[t][j] ↪ for j in range(head_dim)) / head_dim**0.5 ↪ for t in range(len(k_h))] 130 attn_weights = softmax(attn_logits) 131 head_out = [sum(attn_weights[t] * v_h[t][j] ↪ for t in range(len(v_h))) ↪ for j in range(head_dim)]
138 x = linear(x, state_dict[f'layer{li}.mlp_fc1'])
139 x = [xi.relu() for xi in x]
140 x = linear(x, state_dict[f'layer{li}.mlp_fc2'])
118 q = linear(x, state_dict[f'layer{li}.attn_wq'])
119 k = linear(x, state_dict[f'layer{li}.attn_wk'])
120 v = linear(x, state_dict[f'layer{li}.attn_wv'])
# ★ q/k/v:搜尋詞/書背標題/書的內容
129 attn_logits = [sum(q_h[j] * k_h[t][j] ↪ for j in range(head_dim)) / head_dim**0.5 ↪ for t in range(len(k_h))] # ★ q_h、k_h:q、k 切給這個 head 的 4 維
129 attn_logits = [sum(q_h[j] * k_h[t][j] ↪ for j in range(head_dim)) / head_dim**0.5 ↪ for t in range(len(k_h))] # ★ attn_logits:對頻分數(2.7/0.5/−1.2)
130 attn_weights = softmax(attn_logits) # ★ attn_weights:比例,加總=1(0.85/0.13/0.02)
131 head_out = [sum(attn_weights[t] * v_h[t][j] ↪ for t in range(len(v_h))) ↪ for j in range(head_dim)] # ★ head_out:按比例搬回來的 4 個數字
79 head_dim = n_embd // n_head # ★ head_dim:16 ÷ 4 = 4
124 for h in range(n_head): 125 hs = h * head_dim # ★ hs:切片起點(0、4、8、12) 126 q_h = q[hs:hs+head_dim] 127 k_h = [ki[hs:hs+head_dim] for ki in keys[li]] 128 v_h = [vi[hs:hs+head_dim] for vi in values[li]] # ki/vi:cache 裡每一格前文的 k、v
132 x_attn.extend(head_out) # ★ x_attn:四個 head 接起來的 16 個數字
124 for h in range(n_head): ⋯ 132 x_attn.extend(head_out)
116 x_residual = x
117 x = rmsnorm(x)
118 q = linear(x, state_dict[f'layer{li}.attn_wq'])
119 k = linear(x, state_dict[f'layer{li}.attn_wk'])
120 v = linear(x, state_dict[f'layer{li}.attn_wv'])
121 keys[li].append(k) # 這一格的 k、v 存檔
122 values[li].append(v)
123 x_attn = []
124 for h in range(n_head):
125 hs = h * head_dim
126 q_h = q[hs:hs+head_dim] # 「切」僅這一行
⋯
132 x_attn.extend(head_out)
133 x = linear(x_attn, state_dict[f'layer{li}.attn_wo'])
# Wo:唯一讓四個 head 交流的地方
134 x = [a + b for a, b in zip(x, x_residual)]
# 旁路在這裡 ⊕ 回來
50 def relu(self): return Value(max(0, self.data), ↪ (self,), (float(self.data > 0),))
87 state_dict[f'layer{i}.mlp_fc1'] = matrix(4 * n_embd, n_embd)
88 state_dict[f'layer{i}.mlp_fc2'] = matrix(n_embd, 4 * n_embd)
# fc1=64 個偵測分數;fc2=把啟動結果寫回 16 維
136 x_residual = x
137 x = rmsnorm(x)
138 x = linear(x, state_dict[f'layer{li}.mlp_fc1'])
139 x = [xi.relu() for xi in x]
140 x = linear(x, state_dict[f'layer{li}.mlp_fc2'])
141 x = [a + b for a, b in zip(x, x_residual)]
39 def __add__(self, other): # ★ other:另一個加數 40 other = other if isinstance(other, Value) ↪ else Value(other) 41 return Value(self.data + other.data, ↪ (self, other), (1, 1)) # 便條上寫 (1, 1)——訓練時梯度得以原樣通過
114 for li in range(n_layer): # 多層=多跑一圈、多兩個加號 ⋯ 134 x = [a + b for a, b in zip(x, x_residual)] ⋯ 141 x = [a + b for a, b in zip(x, x_residual)]
81 state_dict = {'wte': matrix(vocab_size, n_embd),
↪ 'wpe': matrix(block_size, n_embd),
↪ 'lm_head': matrix(vocab_size, n_embd)}
# lm_head 與 wte 同形,但不是同一份權重
94 def linear(x, w): 95 return [sum(wi * xi for wi, xi in zip(wo, x)) for wo in w] ⋯ 143 logits = linear(x, state_dict['lm_head']) # 最後一個矩陣:27 個候選向量,各自產生一個分數 144 return logits
187 temperature = 0.5 # in (0, 1], control the …
193 for pos_id in range(block_size): 194 logits = gpt(token_id, pos_id, keys, values) 195 probs = softmax([l / temperature for l in logits]) # 溫度僅是將分數相除 196 token_id = random.choices(range(vocab_size), ↪ weights=[p.data for p in probs])[0] # 這段生成流程唯一的隨機選擇,僅此一行 197 if token_id == BOS: 198 break
121 keys[li].append(k) # 每輪只把「新的」加進 cache 122 values[li].append(v)
190 keys, values = [[] for _ in range(n_layer)], ↪ [[] for _ in range(n_layer)] # 每個名字歸零重來 191 token_id = BOS 192 sample = [] 193 for pos_id in range(block_size): 194 logits = gpt(token_id, pos_id, keys, values) 195 probs = softmax([l / temperature for l in logits]) 196 token_id = random.choices(range(vocab_size), ↪ weights=[p.data for p in probs])[0] 197 if token_id == BOS: 198 break 199 sample.append(uchars[token_id])
39 def __add__(self, other): # a + b 41 return Value(self.data + other.data, ↪ (self, other), (1, 1)) # ★ 加法:兩邊都是 1 43 def __mul__(self, other): # a * b 45 return Value(self.data * other.data, ↪ (self, other), (other.data, self.data)) # ★ 乘法:便條寫「對方的數值」 50 def relu(self): … (float(self.data > 0),) # ★ relu:1 或 0——關閉的路徑不傳遞任何值
69 self.grad = 1 # ★ 從 loss 出發 70 for v in reversed(topo): # 從後往前走一遍 71 for child, local_grad in ↪ zip(v._children, v._local_grads): 72 child.grad += local_grad * v.grad # ★ 上方那張表,即是這一行執行五次
156 doc = docs[step % len(docs)] # ★ step:第幾步 157 tokens = [BOS] + [uchars.index(ch) for ch in doc] + [BOS] 158 n = min(block_size, len(tokens) - 1) # ★ n:幾個學習訊號 ⋯ 163 for pos_id in range(n): 164 token_id, target_id = tokens[pos_id], ↪ tokens[pos_id + 1] # ★ target_id:正確答案=下一個字母 165 logits = gpt(token_id, pos_id, keys, values) 166 probs = softmax(logits) 167 loss_t = -probs[target_id].log() # ★ loss_t/loss:誤差大小(正解機率取 −log) 168 losses.append(loss_t) 169 loss = (1 / n) * sum(losses) # final average loss … ⋯ 172 loss.backward() # 求出每個參數往哪個方向可使 loss 變小 ⋯ 175 lr_t = learning_rate * (1 - step / num_steps) # … # ★ lr_t/num_steps:步幅逐步遞減/共 1,000 步 176 for i, p in enumerate(params): # ★ p:4,192 個逐一走訪 177 m[i] = beta1 * m[i] + (1 - beta1) * p.grad # ★ m/v:Adam 兩組統計量(與前述 v 無關);p.grad=梯度 178 v[i] = beta2 * v[i] + (1 - beta2) * p.grad ** 2 ⋯ 181 p.data -= lr_t * m_hat / (v_hat ** 0.5 + eps_adam) 182 p.grad = 0
解析訓練完成的 checkpoint:wpe 的長度與夾角、四個 head 在同一時刻的實際注意力分佈。
scripts/dump_position.py
scripts/dump_attention.py
一字一 token 的 tokenizer 在中文上 vocab 增至 700+,block_size 反而變小——同一份程式碼,參數分佈截然不同。
microgpt.py(金庸人名)
loss 持續下降,但模型開始記住訓練資料——生成結果直接命中訓練集原文,即 overfitting 的具體表現。
EXPERIMENTS.md