Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 1 addition & 2 deletions src/transformers/models/nemotron_h/modeling_nemotron_h.py
Original file line number Diff line number Diff line change
Expand Up @@ -377,8 +377,7 @@ def __init__(self, config: NemotronHConfig, layer_idx: int | None = None, initia
self.n_groups = config.n_groups
self.head_dim = config.mamba_head_dim
self.chunk_size = config.chunk_size
# No upper limit
self.time_step_limit = (config.time_step_min, float("inf"))
self.time_step_limit = config.time_step_limit

self.conv_dim = self.intermediate_size + 2 * self.n_groups * self.ssm_state_size

Expand Down
1 change: 1 addition & 0 deletions src/transformers/models/nemotron_h/modular_nemotron_h.py
Original file line number Diff line number Diff line change
Expand Up @@ -58,6 +58,7 @@ def __init__(self, config: NemotronHConfig, layer_idx: int | None = None, initia
self.n_groups = config.n_groups
self.head_dim = config.mamba_head_dim
self.num_heads = config.mamba_num_heads
self.time_step_limit = config.time_step_limit

self.conv1d = nn.Conv1d(
in_channels=self.conv_dim,
Expand Down
44 changes: 44 additions & 0 deletions tests/models/nemotron_h/test_modeling_nemotron_h.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@
# limitations under the License.
"""Testing suite for the PyTorch NemotronH model."""

import copy
import tempfile
import unittest

Expand Down Expand Up @@ -42,6 +43,7 @@
import torch

from transformers import DynamicCache, NemotronHForCausalLM, NemotronHModel, StaticCache
from transformers.models.nemotron_h.modeling_nemotron_h import NemotronHMamba2Mixer


class NemotronHModelTester:
Expand Down Expand Up @@ -529,6 +531,48 @@ def test_model(self):
config_and_inputs = self.model_tester.prepare_config_and_inputs()
self.model_tester.create_and_check_model(*config_and_inputs)

def test_mamba_time_step_min_only_affects_initialization(self):
torch.manual_seed(0)
config = self.model_tester.get_config()
mixer = NemotronHMamba2Mixer(config, layer_idx=0).eval()
with torch.no_grad():
mixer.dt_bias.fill_(-12.0)
mixer.D.zero_()

other_config = copy.deepcopy(config)
other_config.time_step_min = 0.1
other_mixer = NemotronHMamba2Mixer(other_config, layer_idx=0).eval()
other_mixer.load_state_dict(mixer.state_dict())

hidden_states = torch.randn(2, 7, config.hidden_size, requires_grad=True)
output = mixer(hidden_states)
other_output = other_mixer(hidden_states)
torch.testing.assert_close(output, other_output)
gradient = torch.autograd.grad(output.square().sum(), hidden_states)[0]
other_gradient = torch.autograd.grad(other_output.square().sum(), hidden_states)[0]
torch.testing.assert_close(gradient, other_gradient)

def test_mamba_time_step_limit_bounds(self):
torch.manual_seed(0)
config = self.model_tester.get_config()
config.time_step_limit = (0.01, 0.02)
mixer = NemotronHMamba2Mixer(config, layer_idx=0).eval()
with torch.no_grad():
mixer.in_proj.weight[-config.mamba_num_heads :].zero_()
mixer.D.zero_()
reference_config = copy.deepcopy(config)
reference_config.time_step_limit = (0.0, float("inf"))
reference = NemotronHMamba2Mixer(reference_config, layer_idx=0).eval()
reference.load_state_dict(mixer.state_dict())
hidden_states = torch.randn(2, 7, config.hidden_size)

for dt_bias, expected_dt in [(-12.0, 0.01), (12.0, 0.02)]:
with self.subTest(dt_bias=dt_bias), torch.no_grad():
mixer.dt_bias.fill_(dt_bias)
dt = torch.tensor(expected_dt)
reference.dt_bias.copy_((dt + torch.log(-torch.expm1(-dt))).expand_as(reference.dt_bias))
torch.testing.assert_close(mixer(hidden_states), reference(hidden_states))

def test_for_causal_lm(self):
config_and_inputs = self.model_tester.prepare_config_and_inputs()
self.model_tester.create_and_check_for_causal_lm(*config_and_inputs)
Expand Down
Loading