add comment

2026-04-13 14:17:23 +02:00
parent 9822cc7424
commit a3ca42a678
1 changed files with 2 additions and 0 deletions
@@ -237,6 +237,8 @@ class GPT(nn.Module):
        # Decaying x0 init: earlier layers get more input embedding blending
        for i in range(n_layer):
            self.x0_lambdas.data[i] = 0.20 - (0.15 * i / max(n_layer - 1, 1))
+
+        # Smear/backout scalars and smear gate must be explicitly initialized 
        torch.nn.init.zeros_(self.smear_lambda)
        torch.nn.init.constant_(self.backout_lambda, 0.2)
        torch.nn.init.uniform_(self.smear_gate.weight, 0.0, 0.02)