nvidia
/

NVIDIA-Nemotron-3-Nano-30B-A3B-BF16

Text Generation

Model card Files Files and versions

suhara commited on 2 days ago

Commit

4a31ff1

·

verified ·

1 Parent(s): 0d1698d

Upload modeling_nemotron_h.py

Files changed (1) hide show

modeling_nemotron_h.py +4 -3

modeling_nemotron_h.py CHANGED Viewed

@@ -623,8 +623,8 @@ class NemotronHMamba2Mixer(nn.Module):
             hidden_states = hidden_states.reshape(batch_size, seq_len, -1, self.head_dim).float()
             B = B.reshape(batch_size, seq_len, -1, self.ssm_state_size).float()
             C = C.reshape(batch_size, seq_len, -1, self.ssm_state_size).float()
-            B = B.repeat(1, 1, self.num_heads // self.n_groups, 1)
-            C = C.repeat(1, 1, self.num_heads // self.n_groups, 1)
             pad_size = (self.chunk_size - seq_len % self.chunk_size) % self.chunk_size
             D_residual = self.D[..., None] * pad_tensor_by_size(hidden_states, pad_size)
@@ -852,7 +852,8 @@ class NemotronHMOE(nn.Module):
                 final_hidden_states.index_add_(0, token_indices, weighted_output)
             else:
                 # Local empty expert: no-op compute that still marks params as used.
-                dummy_out = expert(torch.zeros_like(hidden_states[0]).unsqueeze(0).to(final_hidden_states.dtype))
                 final_hidden_states = final_hidden_states + dummy_out
         # in original deepseek, the output of the experts are gathered once we leave this module

             hidden_states = hidden_states.reshape(batch_size, seq_len, -1, self.head_dim).float()
             B = B.reshape(batch_size, seq_len, -1, self.ssm_state_size).float()
             C = C.reshape(batch_size, seq_len, -1, self.ssm_state_size).float()
+            B = B.repeat_interleave(self.num_heads // self.n_groups, dim=2, output_size=self.num_heads)
+            C = C.repeat_interleave(self.num_heads // self.n_groups, dim=2, output_size=self.num_heads)
             pad_size = (self.chunk_size - seq_len % self.chunk_size) % self.chunk_size
             D_residual = self.D[..., None] * pad_tensor_by_size(hidden_states, pad_size)
                 final_hidden_states.index_add_(0, token_indices, weighted_output)
             else:
                 # Local empty expert: no-op compute that still marks params as used.
+                expert_dtype = expert.down_proj.weight.dtype
+                dummy_out = expert(torch.zeros_like(hidden_states[0]).unsqueeze(0).to(expert_dtype))
                 final_hidden_states = final_hidden_states + dummy_out
         # in original deepseek, the output of the experts are gathered once we leave this module