From 6c117b4dd6106fba7200b663cfafa970788066ee Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Wed, 30 Jul 2025 06:34:56 +0200 Subject: [PATCH 01/29] fix MaskFormer --- .../models/maskformer/modeling_maskformer.py | 15 +++++---------- tests/test_modeling_common.py | 2 +- 2 files changed, 6 insertions(+), 11 deletions(-) diff --git a/src/transformers/models/maskformer/modeling_maskformer.py b/src/transformers/models/maskformer/modeling_maskformer.py index 3be1021a2c29..4aac418c9aa4 100644 --- a/src/transformers/models/maskformer/modeling_maskformer.py +++ b/src/transformers/models/maskformer/modeling_maskformer.py @@ -1431,27 +1431,22 @@ def _init_weights(self, module: nn.Module): nn.init.xavier_uniform_(module.input_projection.weight, gain=xavier_std) nn.init.constant_(module.input_projection.bias, 0) # FPN - elif isinstance(module, MaskFormerFPNModel): - nn.init.xavier_uniform_(module.stem.get_submodule("0").weight, gain=xavier_std) - elif isinstance(module, MaskFormerFPNLayer): nn.init.xavier_uniform_(module.proj[0].weight, gain=xavier_std) elif isinstance(module, MaskFormerFPNConvLayer): nn.init.xavier_uniform_(module.get_submodule("0").weight, gain=xavier_std) # The MLP head - elif isinstance(module, MaskformerMLPPredictionHead): + elif isinstance(module, PredictionBlock): # I was not able to find the correct initializer in the original implementation # we'll use xavier - for submodule in module.modules(): - if isinstance(submodule, nn.Linear): - nn.init.xavier_uniform_(submodule.weight, gain=xavier_std) - nn.init.constant_(submodule.bias, 0) - elif isinstance(module, nn.LayerNorm): + nn.init.xavier_uniform_(module.get_submodule("0").weight, gain=xavier_std) + nn.init.constant_(module.get_submodule("0").bias, 0) + elif isinstance(module, (nn.LayerNorm, nn.GroupNorm)): module.bias.data.zero_() module.weight.data.fill_(1.0) # copied from DETR - if isinstance(module, (nn.Linear, nn.Conv2d, nn.BatchNorm2d)): + if isinstance(module, (nn.Linear, nn.Conv2d)): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 module.weight.data.normal_(mean=0.0, std=std) diff --git a/tests/test_modeling_common.py b/tests/test_modeling_common.py index fcc47466a397..c84dbe36f7c8 100755 --- a/tests/test_modeling_common.py +++ b/tests/test_modeling_common.py @@ -857,7 +857,7 @@ def test_can_init_all_missing_weights(self): for model_class in self.all_model_classes: # For now, skip everything older than 2024 and "important models" (too much models to patch otherwise) # TODO: relax this as we patch more and more models - if addition_year < 2023: + if addition_year < 2022: self.skipTest(reason=f"{model_class} is not a priorited model for now.") # Monkey patch the method to add a seed (we do it on PreTrainedModel._initialize_weights, which wraps From 5c7c025e2cd3d09c3c3d673770c6dca5836bcf59 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Wed, 30 Jul 2025 09:05:59 +0200 Subject: [PATCH 02/29] fix ConditionalDetr and Detr --- .../models/conditional_detr/modeling_conditional_detr.py | 5 ++++- src/transformers/models/detr/modeling_detr.py | 5 ++++- 2 files changed, 8 insertions(+), 2 deletions(-) diff --git a/src/transformers/models/conditional_detr/modeling_conditional_detr.py b/src/transformers/models/conditional_detr/modeling_conditional_detr.py index 25eacb959aef..7e9f54914e40 100644 --- a/src/transformers/models/conditional_detr/modeling_conditional_detr.py +++ b/src/transformers/models/conditional_detr/modeling_conditional_detr.py @@ -973,7 +973,7 @@ def _init_weights(self, module): elif isinstance(module, ConditionalDetrLearnedPositionEmbedding): nn.init.uniform_(module.row_embeddings.weight) nn.init.uniform_(module.column_embeddings.weight) - if isinstance(module, (nn.Linear, nn.Conv2d, nn.BatchNorm2d)): + if isinstance(module, (nn.Linear, nn.Conv2d)): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 module.weight.data.normal_(mean=0.0, std=std) @@ -983,6 +983,9 @@ def _init_weights(self, module): module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() + elif isinstance(module, (nn.LayerNorm, nn.GroupNorm)): + module.weight.data.fill_(1.0) + module.bias.data.zero_() # Copied from transformers.models.detr.modeling_detr.DetrEncoder with Detr->ConditionalDetr,DETR->ConditionalDETR diff --git a/src/transformers/models/detr/modeling_detr.py b/src/transformers/models/detr/modeling_detr.py index d2a205fb2112..ba746baa9445 100644 --- a/src/transformers/models/detr/modeling_detr.py +++ b/src/transformers/models/detr/modeling_detr.py @@ -734,7 +734,7 @@ def _init_weights(self, module): elif isinstance(module, DetrLearnedPositionEmbedding): nn.init.uniform_(module.row_embeddings.weight) nn.init.uniform_(module.column_embeddings.weight) - if isinstance(module, (nn.Linear, nn.Conv2d, nn.BatchNorm2d)): + if isinstance(module, (nn.Linear, nn.Conv2d)): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 module.weight.data.normal_(mean=0.0, std=std) @@ -744,6 +744,9 @@ def _init_weights(self, module): module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() + elif isinstance(module, (nn.LayerNorm, nn.GroupNorm)): + module.weight.data.fill_(1.0) + module.bias.data.zero_() class DetrEncoder(DetrPreTrainedModel): From 9f020a00a5dce824c11f5a6450602639c004e7de Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Wed, 30 Jul 2025 09:59:59 +0200 Subject: [PATCH 03/29] fix Wav2Vec2Conformer and Wav2Vec2 --- .../conditional_detr/modeling_conditional_detr.py | 2 +- src/transformers/models/detr/modeling_detr.py | 2 +- src/transformers/models/wav2vec2/modeling_wav2vec2.py | 7 ++++++- .../wav2vec2_conformer/modeling_wav2vec2_conformer.py | 11 +++++++---- .../wav2vec2_conformer/modular_wav2vec2_conformer.py | 11 +++++++---- 5 files changed, 22 insertions(+), 11 deletions(-) diff --git a/src/transformers/models/conditional_detr/modeling_conditional_detr.py b/src/transformers/models/conditional_detr/modeling_conditional_detr.py index 7e9f54914e40..d00d2bec7aaf 100644 --- a/src/transformers/models/conditional_detr/modeling_conditional_detr.py +++ b/src/transformers/models/conditional_detr/modeling_conditional_detr.py @@ -961,7 +961,7 @@ class ConditionalDetrPreTrainedModel(PreTrainedModel): main_input_name = "pixel_values" _no_split_modules = [r"ConditionalDetrConvEncoder", r"ConditionalDetrEncoderLayer", r"ConditionalDetrDecoderLayer"] - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): std = self.config.init_std xavier_std = self.config.init_xavier_std diff --git a/src/transformers/models/detr/modeling_detr.py b/src/transformers/models/detr/modeling_detr.py index ba746baa9445..6fac78406557 100644 --- a/src/transformers/models/detr/modeling_detr.py +++ b/src/transformers/models/detr/modeling_detr.py @@ -722,7 +722,7 @@ class DetrPreTrainedModel(PreTrainedModel): main_input_name = "pixel_values" _no_split_modules = [r"DetrConvEncoder", r"DetrEncoderLayer", r"DetrDecoderLayer"] - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): std = self.config.init_std xavier_std = self.config.init_xavier_std diff --git a/src/transformers/models/wav2vec2/modeling_wav2vec2.py b/src/transformers/models/wav2vec2/modeling_wav2vec2.py index 2c352da1804c..492125a20249 100755 --- a/src/transformers/models/wav2vec2/modeling_wav2vec2.py +++ b/src/transformers/models/wav2vec2/modeling_wav2vec2.py @@ -1036,7 +1036,7 @@ class Wav2Vec2PreTrainedModel(PreTrainedModel): _supports_sdpa = True _supports_flex_attn = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" # Wav2Vec2ForPreTraining last 2 linear layers need standard Linear init. if isinstance(module, Wav2Vec2ForPreTraining): @@ -1074,6 +1074,11 @@ def _init_weights(self, module): if module.bias is not None: k = math.sqrt(module.groups / (module.in_channels * module.kernel_size[0])) nn.init.uniform_(module.bias, a=-k, b=k) + elif isinstance(module, AMSoftmaxLoss): + nn.init.normal_(module.weight) + + if hasattr(module, "masked_spec_embed"): + nn.init.uniform_(module.masked_spec_embed) def _get_feat_extract_output_lengths( self, input_lengths: Union[torch.LongTensor, int], add_adapter: Optional[bool] = None diff --git a/src/transformers/models/wav2vec2_conformer/modeling_wav2vec2_conformer.py b/src/transformers/models/wav2vec2_conformer/modeling_wav2vec2_conformer.py index bdc3dcddaf0b..d95b4b53b8fe 100644 --- a/src/transformers/models/wav2vec2_conformer/modeling_wav2vec2_conformer.py +++ b/src/transformers/models/wav2vec2_conformer/modeling_wav2vec2_conformer.py @@ -856,14 +856,14 @@ class Wav2Vec2ConformerPreTrainedModel(PreTrainedModel): main_input_name = "input_values" supports_gradient_checkpointing = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" # Wav2Vec2ForPreTraining last 2 linear layers need standard Linear init. if isinstance(module, Wav2Vec2ConformerForPreTraining): module.project_hid.reset_parameters() module.project_q.reset_parameters() - module.project_hid._is_hf_initialized = True - module.project_q._is_hf_initialized = True + elif isinstance(module, Wav2Vec2ConformerForXVector): + nn.init.normal_(module.objective.weight) # gumbel softmax requires special init elif isinstance(module, Wav2Vec2ConformerGumbelVectorQuantizer): module.weight_proj.weight.data.normal_(mean=0.0, std=1) @@ -890,7 +890,7 @@ def _init_weights(self, module): if module.bias is not None: module.bias.data.zero_() - elif isinstance(module, (nn.LayerNorm, nn.GroupNorm)): + elif isinstance(module, (nn.LayerNorm, nn.GroupNorm, nn.BatchNorm1d)): module.bias.data.zero_() module.weight.data.fill_(1.0) elif isinstance(module, nn.Conv1d): @@ -900,6 +900,9 @@ def _init_weights(self, module): k = math.sqrt(module.groups / (module.in_channels * module.kernel_size[0])) nn.init.uniform_(module.bias, a=-k, b=k) + if hasattr(module, "masked_spec_embed"): + nn.init.uniform_(module.masked_spec_embed) + def _get_feat_extract_output_lengths( self, input_lengths: Union[torch.LongTensor, int], add_adapter: Optional[bool] = None ): diff --git a/src/transformers/models/wav2vec2_conformer/modular_wav2vec2_conformer.py b/src/transformers/models/wav2vec2_conformer/modular_wav2vec2_conformer.py index b54e7d0259e2..69ce2369d06d 100644 --- a/src/transformers/models/wav2vec2_conformer/modular_wav2vec2_conformer.py +++ b/src/transformers/models/wav2vec2_conformer/modular_wav2vec2_conformer.py @@ -551,14 +551,14 @@ class Wav2Vec2ConformerPreTrainedModel(PreTrainedModel): main_input_name = "input_values" supports_gradient_checkpointing = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" # Wav2Vec2ForPreTraining last 2 linear layers need standard Linear init. if isinstance(module, Wav2Vec2ConformerForPreTraining): module.project_hid.reset_parameters() module.project_q.reset_parameters() - module.project_hid._is_hf_initialized = True - module.project_q._is_hf_initialized = True + elif isinstance(module, Wav2Vec2ConformerForXVector): + nn.init.normal_(module.objective.weight) # gumbel softmax requires special init elif isinstance(module, Wav2Vec2ConformerGumbelVectorQuantizer): module.weight_proj.weight.data.normal_(mean=0.0, std=1) @@ -585,7 +585,7 @@ def _init_weights(self, module): if module.bias is not None: module.bias.data.zero_() - elif isinstance(module, (nn.LayerNorm, nn.GroupNorm)): + elif isinstance(module, (nn.LayerNorm, nn.GroupNorm, nn.BatchNorm1d)): module.bias.data.zero_() module.weight.data.fill_(1.0) elif isinstance(module, nn.Conv1d): @@ -595,6 +595,9 @@ def _init_weights(self, module): k = math.sqrt(module.groups / (module.in_channels * module.kernel_size[0])) nn.init.uniform_(module.bias, a=-k, b=k) + if hasattr(module, "masked_spec_embed"): + nn.init.uniform_(module.masked_spec_embed) + def _get_feat_extract_output_lengths( self, input_lengths: Union[torch.LongTensor, int], add_adapter: Optional[bool] = None ): From 8a9dd350f1fcc5bc3cf06a43f8c4fe9abe3abb82 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Wed, 30 Jul 2025 10:10:50 +0200 Subject: [PATCH 04/29] fix Levit --- src/transformers/models/levit/modeling_levit.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/src/transformers/models/levit/modeling_levit.py b/src/transformers/models/levit/modeling_levit.py index fc275a1c4c40..a78dfde9c5ff 100644 --- a/src/transformers/models/levit/modeling_levit.py +++ b/src/transformers/models/levit/modeling_levit.py @@ -473,7 +473,7 @@ class LevitPreTrainedModel(PreTrainedModel): main_input_name = "pixel_values" _no_split_modules = ["LevitResidualLayer"] - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" if isinstance(module, (nn.Linear, nn.Conv2d)): # Slightly different from the TF version which uses truncated_normal for initialization @@ -484,6 +484,8 @@ def _init_weights(self, module): elif isinstance(module, (nn.BatchNorm1d, nn.BatchNorm2d)): module.bias.data.zero_() module.weight.data.fill_(1.0) + elif isinstance(module, (LevitAttention, LevitAttentionSubsample)): + module.attention_biases.data.zero_() @auto_docstring From d5c608a37f03208690878aa4b38d4ebeeaa2aa97 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Wed, 30 Jul 2025 12:07:58 +0200 Subject: [PATCH 05/29] fix ChineseCLIP --- .../chinese_clip/modeling_chinese_clip.py | 21 +++++++++---------- .../test_modeling_chinese_clip.py | 3 --- 2 files changed, 10 insertions(+), 14 deletions(-) diff --git a/src/transformers/models/chinese_clip/modeling_chinese_clip.py b/src/transformers/models/chinese_clip/modeling_chinese_clip.py index 6d2452242d52..ac2deaa82223 100644 --- a/src/transformers/models/chinese_clip/modeling_chinese_clip.py +++ b/src/transformers/models/chinese_clip/modeling_chinese_clip.py @@ -592,23 +592,22 @@ class ChineseCLIPPreTrainedModel(PreTrainedModel): base_model_prefix = "chinese_clip" supports_gradient_checkpointing = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" factor = self.config.initializer_factor + std = self.config.initializer_range if isinstance(module, ChineseCLIPVisionEmbeddings): - factor = self.config.initializer_factor - nn.init.normal_(module.class_embedding, mean=0.0, std=module.embed_dim**-0.5 * factor) + nn.init.normal_(module.class_embedding, std=module.embed_dim**-0.5 * factor) nn.init.normal_(module.patch_embedding.weight, std=module.config.initializer_range * factor) nn.init.normal_(module.position_embedding.weight, std=module.config.initializer_range * factor) elif isinstance(module, ChineseCLIPTextEmbeddings): - nn.init.normal_(module.word_embeddings.weight, mean=0.0, std=self.config.initializer_range) - nn.init.normal_(module.position_embeddings.weight, mean=0.0, std=self.config.initializer_range) - nn.init.normal_(module.token_type_embeddings.weight, mean=0.0, std=self.config.initializer_range) + nn.init.normal_(module.word_embeddings.weight, std=std) + nn.init.normal_(module.position_embeddings.weight, std=std) + nn.init.normal_(module.token_type_embeddings.weight, std=std) for embedding in [module.word_embeddings, module.position_embeddings, module.token_type_embeddings]: if embedding.padding_idx is not None: embedding.weight.data[embedding.padding_idx].zero_() elif isinstance(module, ChineseCLIPVisionAttention): - factor = self.config.initializer_factor in_proj_std = (module.embed_dim**-0.5) * ((2 * module.config.num_hidden_layers) ** -0.5) * factor out_proj_std = (module.embed_dim**-0.5) * factor nn.init.normal_(module.q_proj.weight, std=in_proj_std) @@ -616,7 +615,6 @@ def _init_weights(self, module): nn.init.normal_(module.v_proj.weight, std=in_proj_std) nn.init.normal_(module.out_proj.weight, std=out_proj_std) elif isinstance(module, ChineseCLIPVisionMLP): - factor = self.config.initializer_factor in_proj_std = (module.config.hidden_size**-0.5) * ((2 * module.config.num_hidden_layers) ** -0.5) * factor fc_std = (2 * module.config.hidden_size) ** -0.5 * factor nn.init.normal_(module.fc1.weight, std=fc_std) @@ -624,18 +622,19 @@ def _init_weights(self, module): elif isinstance(module, ChineseCLIPModel): nn.init.normal_( module.text_projection.weight, - std=module.text_embed_dim**-0.5 * self.config.initializer_factor, + std=module.text_embed_dim**-0.5 * factor, ) nn.init.normal_( module.visual_projection.weight, - std=module.vision_embed_dim**-0.5 * self.config.initializer_factor, + std=module.vision_embed_dim**-0.5 * factor, ) + module.logit_scale.data.fill_(self.config.logit_scale_init_value) if isinstance(module, nn.LayerNorm): module.bias.data.zero_() module.weight.data.fill_(1.0) if isinstance(module, nn.Linear): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.bias is not None: module.bias.data.zero_() diff --git a/tests/models/chinese_clip/test_modeling_chinese_clip.py b/tests/models/chinese_clip/test_modeling_chinese_clip.py index dc8e9a145b08..e4831596308e 100644 --- a/tests/models/chinese_clip/test_modeling_chinese_clip.py +++ b/tests/models/chinese_clip/test_modeling_chinese_clip.py @@ -585,9 +585,6 @@ def test_initialization(self): config, inputs_dict = self.model_tester.prepare_config_and_inputs_for_common() configs_no_init = _config_zero_init(config) - for sub_config_key in ("vision_config", "text_config"): - sub_config = getattr(configs_no_init, sub_config_key, {}) - setattr(configs_no_init, sub_config_key, _config_zero_init(sub_config)) for model_class in self.all_model_classes: model = model_class(config=configs_no_init) for name, param in model.named_parameters(): From 1a89491fe0de53ae3ae6a3b4392a6ef1565e3f0a Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Wed, 30 Jul 2025 12:52:42 +0200 Subject: [PATCH 06/29] fix CLIPSeg --- .../models/bridgetower/modeling_bridgetower.py | 16 ++++++++-------- .../models/clipseg/modeling_clipseg.py | 18 +++++++++--------- .../models/owlv2/modeling_owlv2.py | 2 +- .../models/owlvit/modeling_owlvit.py | 2 +- tests/models/clipseg/test_modeling_clipseg.py | 3 --- 5 files changed, 19 insertions(+), 22 deletions(-) diff --git a/src/transformers/models/bridgetower/modeling_bridgetower.py b/src/transformers/models/bridgetower/modeling_bridgetower.py index 644b8b0a3b6f..2084544995e9 100644 --- a/src/transformers/models/bridgetower/modeling_bridgetower.py +++ b/src/transformers/models/bridgetower/modeling_bridgetower.py @@ -938,22 +938,22 @@ class BridgeTowerPreTrainedModel(PreTrainedModel): _skip_keys_device_placement = "past_key_values" def _init_weights(self, module: nn.Module): - std = self.config.initializer_factor + factor = self.config.initializer_factor if isinstance(module, BridgeTowerVisionTransformer): proj_std = (self.config.hidden_size**-0.5) * ((2 * self.config.num_hidden_layers) ** -0.5) attn_std = self.config.hidden_size**-0.5 fc_std = (2 * self.config.hidden_size) ** -0.5 for block in module.transformer.resblocks: - nn.init.normal_(block.attn.in_proj_weight, std=attn_std * std) + nn.init.normal_(block.attn.in_proj_weight, std=attn_std * factor) block.attn.in_proj_bias.data.zero_() - nn.init.normal_(block.attn.out_proj.weight, std=proj_std * std) - nn.init.normal_(block.mlp.c_fc.weight, std=fc_std * std) - nn.init.normal_(block.mlp.c_proj.weight, std=proj_std * std) + nn.init.normal_(block.attn.out_proj.weight, std=proj_std * factor) + nn.init.normal_(block.mlp.c_fc.weight, std=fc_std * factor) + nn.init.normal_(block.mlp.c_proj.weight, std=proj_std * factor) - nn.init.normal_(module.embeddings.class_embedding, std=attn_std * std) - nn.init.normal_(module.embeddings.position_embedding.weight, std=attn_std * std) + nn.init.normal_(module.embeddings.class_embedding, std=attn_std * factor) + nn.init.normal_(module.embeddings.position_embedding.weight, std=attn_std * factor) elif isinstance(module, (nn.Linear, nn.Conv2d, nn.Embedding)): - module.weight.data.normal_(mean=0.0, std=0.05 * std) + module.weight.data.normal_(mean=0.0, std=0.05 * factor) elif isinstance(module, nn.LayerNorm): module.bias.data.zero_() module.weight.data.fill_(1.0) diff --git a/src/transformers/models/clipseg/modeling_clipseg.py b/src/transformers/models/clipseg/modeling_clipseg.py index 34d19ffaf387..d42021753df2 100644 --- a/src/transformers/models/clipseg/modeling_clipseg.py +++ b/src/transformers/models/clipseg/modeling_clipseg.py @@ -432,19 +432,17 @@ class CLIPSegPreTrainedModel(PreTrainedModel): base_model_prefix = "clip" supports_gradient_checkpointing = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" factor = self.config.initializer_factor if isinstance(module, CLIPSegTextEmbeddings): module.token_embedding.weight.data.normal_(mean=0.0, std=factor * 0.02) module.position_embedding.weight.data.normal_(mean=0.0, std=factor * 0.02) elif isinstance(module, CLIPSegVisionEmbeddings): - factor = self.config.initializer_factor - nn.init.normal_(module.class_embedding, mean=0.0, std=module.embed_dim**-0.5 * factor) + nn.init.normal_(module.class_embedding, std=module.embed_dim**-0.5 * factor) nn.init.normal_(module.patch_embedding.weight, std=module.config.initializer_range * factor) nn.init.normal_(module.position_embedding.weight, std=module.config.initializer_range * factor) elif isinstance(module, CLIPSegAttention): - factor = self.config.initializer_factor in_proj_std = (module.embed_dim**-0.5) * ((2 * module.config.num_hidden_layers) ** -0.5) * factor out_proj_std = (module.embed_dim**-0.5) * factor nn.init.normal_(module.q_proj.weight, std=in_proj_std) @@ -452,7 +450,6 @@ def _init_weights(self, module): nn.init.normal_(module.v_proj.weight, std=in_proj_std) nn.init.normal_(module.out_proj.weight, std=out_proj_std) elif isinstance(module, CLIPSegMLP): - factor = self.config.initializer_factor in_proj_std = (module.config.hidden_size**-0.5) * ((2 * module.config.num_hidden_layers) ** -0.5) * factor fc_std = (2 * module.config.hidden_size) ** -0.5 * factor nn.init.normal_(module.fc1.weight, std=fc_std) @@ -460,18 +457,21 @@ def _init_weights(self, module): elif isinstance(module, CLIPSegModel): nn.init.normal_( module.text_projection.weight, - std=module.text_embed_dim**-0.5 * self.config.initializer_factor, + std=module.text_embed_dim**-0.5 * factor, ) nn.init.normal_( module.visual_projection.weight, - std=module.vision_embed_dim**-0.5 * self.config.initializer_factor, + std=module.vision_embed_dim**-0.5 * factor, ) + module.logit_scale.data.fill_(self.config.logit_scale_init_value) if isinstance(module, nn.LayerNorm): module.bias.data.zero_() module.weight.data.fill_(1.0) - if isinstance(module, nn.Linear) and module.bias is not None: - module.bias.data.zero_() + if isinstance(module, (nn.Linear, nn.ConvTranspose2d)): + module.weight.data.normal_(mean=0.0, std=factor * 0.02) + if module.bias is not None: + module.bias.data.zero_() # Copied from transformers.models.altclip.modeling_altclip.AltCLIPEncoder with AltCLIP->CLIPSeg diff --git a/src/transformers/models/owlv2/modeling_owlv2.py b/src/transformers/models/owlv2/modeling_owlv2.py index e80292d98ce4..3c185de5e572 100644 --- a/src/transformers/models/owlv2/modeling_owlv2.py +++ b/src/transformers/models/owlv2/modeling_owlv2.py @@ -596,7 +596,7 @@ def _init_weights(self, module: nn.Module): module.bias.data.zero_() module.weight.data.fill_(1.0) if isinstance(module, nn.Linear): - module.weight.data.normal_(mean=0.0, std=factor) + module.weight.data.normal_(mean=0.0, std=factor * 0.02) if module.bias is not None: module.bias.data.zero_() diff --git a/src/transformers/models/owlvit/modeling_owlvit.py b/src/transformers/models/owlvit/modeling_owlvit.py index 482be48ec1d4..e7d38fe3f361 100644 --- a/src/transformers/models/owlvit/modeling_owlvit.py +++ b/src/transformers/models/owlvit/modeling_owlvit.py @@ -583,7 +583,7 @@ def _init_weights(self, module: nn.Module): module.bias.data.zero_() module.weight.data.fill_(1.0) if isinstance(module, nn.Linear): - module.weight.data.normal_(mean=0.0, std=factor) + module.weight.data.normal_(mean=0.0, std=factor * 0.02) if module.bias is not None: module.bias.data.zero_() diff --git a/tests/models/clipseg/test_modeling_clipseg.py b/tests/models/clipseg/test_modeling_clipseg.py index 08a21f9dcf3b..d701090f2213 100644 --- a/tests/models/clipseg/test_modeling_clipseg.py +++ b/tests/models/clipseg/test_modeling_clipseg.py @@ -510,9 +510,6 @@ def test_initialization(self): delta=1e-3, msg=f"Parameter {name} of model {model_class} seems not properly initialized", ) - elif "film" in name or "transposed_conv" in name or "reduce" in name: - # those parameters use PyTorch' default nn.Linear initialization scheme - pass else: self.assertIn( ((param.data.mean() * 1e9).round() / 1e9).item(), From 3ae1d23028645383e68912a29a384768d800619a Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Wed, 30 Jul 2025 16:33:49 +0200 Subject: [PATCH 07/29] fix Cvt --- src/transformers/models/cvt/modeling_cvt.py | 11 +++++------ .../models/mgp_str/modeling_mgp_str.py | 6 +++--- src/transformers/models/pvt/modeling_pvt.py | 14 +++----------- 3 files changed, 11 insertions(+), 20 deletions(-) diff --git a/src/transformers/models/cvt/modeling_cvt.py b/src/transformers/models/cvt/modeling_cvt.py index e838ffb3cd41..1bba8d092cb9 100644 --- a/src/transformers/models/cvt/modeling_cvt.py +++ b/src/transformers/models/cvt/modeling_cvt.py @@ -514,20 +514,19 @@ class CvtPreTrainedModel(PreTrainedModel): main_input_name = "pixel_values" _no_split_modules = ["CvtLayer"] - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" + std = self.config.initializer_range if isinstance(module, (nn.Linear, nn.Conv2d)): - module.weight.data = nn.init.trunc_normal_(module.weight.data, mean=0.0, std=self.config.initializer_range) + nn.init.trunc_normal_(module.weight, std=std) if module.bias is not None: module.bias.data.zero_() - elif isinstance(module, nn.LayerNorm): + elif isinstance(module, (nn.LayerNorm, nn.BatchNorm2d)): module.bias.data.zero_() module.weight.data.fill_(1.0) elif isinstance(module, CvtStage): if self.config.cls_token[module.stage]: - module.cls_token.data = nn.init.trunc_normal_( - module.cls_token.data, mean=0.0, std=self.config.initializer_range - ) + nn.init.trunc_normal_(module.cls_token, std=std) @auto_docstring diff --git a/src/transformers/models/mgp_str/modeling_mgp_str.py b/src/transformers/models/mgp_str/modeling_mgp_str.py index 9e6ab26a4b98..490224d070d8 100644 --- a/src/transformers/models/mgp_str/modeling_mgp_str.py +++ b/src/transformers/models/mgp_str/modeling_mgp_str.py @@ -294,10 +294,10 @@ def _init_weights(self, module: nn.Module) -> None: """Initialize the weights""" std = self.config.initializer_range if isinstance(module, MgpstrEmbeddings): - nn.init.trunc_normal_(module.pos_embed, mean=0.0, std=std) - nn.init.trunc_normal_(module.cls_token, mean=0.0, std=std) + nn.init.trunc_normal_(module.pos_embe, std=std) + nn.init.trunc_normal_(module.cls_token, std=std) elif isinstance(module, (nn.Linear, nn.Conv2d)): - nn.init.trunc_normal_(module.weight.data, mean=0.0, std=std) + nn.init.trunc_normal_(module.weight, std=std) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.LayerNorm): diff --git a/src/transformers/models/pvt/modeling_pvt.py b/src/transformers/models/pvt/modeling_pvt.py index 9e2c5a69d8de..e51721049561 100755 --- a/src/transformers/models/pvt/modeling_pvt.py +++ b/src/transformers/models/pvt/modeling_pvt.py @@ -453,24 +453,16 @@ def _init_weights(self, module: nn.Module) -> None: if isinstance(module, (nn.Linear, nn.Conv2d)): # Upcast the input in `fp32` and cast it back to desired `dtype` to avoid # `trunc_normal_cpu` not implemented in `half` issues - nn.init.trunc_normal_(module.weight.data, mean=0.0, std=std) + nn.init.trunc_normal_(module.weight, std=std) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.LayerNorm): module.bias.data.zero_() module.weight.data.fill_(1.0) elif isinstance(module, PvtPatchEmbeddings): - module.position_embeddings.data = nn.init.trunc_normal_( - module.position_embeddings.data, - mean=0.0, - std=std, - ) + nn.init.trunc_normal_(module.position_embeddings, std=std) if module.cls_token is not None: - module.cls_token.data = nn.init.trunc_normal_( - module.cls_token.data, - mean=0.0, - std=std, - ) + nn.init.trunc_normal_(module.cls_token, std=std) @auto_docstring From 62d6219e9fed853b7d91b997847d18580b55f65e Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Wed, 30 Jul 2025 16:46:37 +0200 Subject: [PATCH 08/29] fix Vilt --- src/transformers/models/vilt/modeling_vilt.py | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/src/transformers/models/vilt/modeling_vilt.py b/src/transformers/models/vilt/modeling_vilt.py index 8fc846addc6d..f0834b829520 100755 --- a/src/transformers/models/vilt/modeling_vilt.py +++ b/src/transformers/models/vilt/modeling_vilt.py @@ -550,16 +550,20 @@ class ViltPreTrainedModel(PreTrainedModel): supports_gradient_checkpointing = True _no_split_modules = ["ViltEmbeddings", "ViltSelfAttention"] - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" - if isinstance(module, (nn.Linear, nn.Conv2d)): + std = self.config.initializer_range + if isinstance(module, ViltEmbeddings): + nn.init.trunc_normal_(module.cls_token, std=std) + nn.init.trunc_normal_(module.position_embeddings, std=std) + elif isinstance(module, (nn.Linear, nn.Conv2d)): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.Embedding): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() elif isinstance(module, nn.LayerNorm): From de3d6561f32e347ff2a0938918727a95a1e55c27 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Thu, 31 Jul 2025 05:11:50 +0200 Subject: [PATCH 09/29] fix Yolos --- src/transformers/models/yolos/modeling_yolos.py | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/src/transformers/models/yolos/modeling_yolos.py b/src/transformers/models/yolos/modeling_yolos.py index b201ba72e485..8e82bbe6d764 100755 --- a/src/transformers/models/yolos/modeling_yolos.py +++ b/src/transformers/models/yolos/modeling_yolos.py @@ -533,9 +533,16 @@ class YolosPreTrainedModel(PreTrainedModel): _supports_flex_attn = True _supports_attention_backend = True - def _init_weights(self, module: Union[nn.Linear, nn.Conv2d, nn.LayerNorm]) -> None: + def _init_weights(self, module: nn.Module) -> None: """Initialize the weights""" - if isinstance(module, (nn.Linear, nn.Conv2d)): + std = self.config.initializer_range + if isinstance(module, YolosEmbeddings): + nn.init.trunc_normal_(module.cls_token, std=std) + nn.init.trunc_normal_(module.position_embeddings, std=std) + nn.init.trunc_normal_(module.detection_tokens, std=std) + elif isinstance(module, YolosEncoder): + nn.init.trunc_normal_(module.mid_position_embeddings, std=std) + elif isinstance(module, (nn.Linear, nn.Conv2d)): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) From 38c9853bb8363d996f3202072da74673cd908552 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Thu, 31 Jul 2025 05:56:59 +0200 Subject: [PATCH 10/29] fix GroupViT --- .../models/groupvit/modeling_groupvit.py | 27 ++++++++++++------- 1 file changed, 17 insertions(+), 10 deletions(-) diff --git a/src/transformers/models/groupvit/modeling_groupvit.py b/src/transformers/models/groupvit/modeling_groupvit.py index c9673a128fa8..47388cdae86b 100644 --- a/src/transformers/models/groupvit/modeling_groupvit.py +++ b/src/transformers/models/groupvit/modeling_groupvit.py @@ -748,26 +748,22 @@ class GroupViTPreTrainedModel(PreTrainedModel): base_model_prefix = "groupvit" supports_gradient_checkpointing = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" - init_range = self.config.initializer_range + factor = self.config.initializer_factor if isinstance(module, (nn.Linear, nn.Conv2d)): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 module.weight.data.normal_(mean=0.0, std=init_range) if module.bias is not None: module.bias.data.zero_() - elif isinstance(module, nn.LayerNorm): - module.bias.data.zero_() - module.weight.data.fill_(1.0) - - factor = self.config.initializer_factor - if isinstance(module, GroupViTTextEmbeddings): + elif isinstance(module, GroupViTTextEmbeddings): module.token_embedding.weight.data.normal_(mean=0.0, std=factor * 0.02) module.position_embedding.weight.data.normal_(mean=0.0, std=factor * 0.02) + elif isinstance(module, GroupViTVisionEmbeddings): + nn.init.trunc_normal_(module.position_embeddings, std=factor * 0.02) elif isinstance(module, GroupViTAttention): - factor = self.config.initializer_factor in_proj_std = (module.embed_dim**-0.5) * ((2 * module.config.num_hidden_layers) ** -0.5) * factor out_proj_std = (module.embed_dim**-0.5) * factor nn.init.normal_(module.q_proj.weight, std=in_proj_std) @@ -775,11 +771,22 @@ def _init_weights(self, module): nn.init.normal_(module.v_proj.weight, std=in_proj_std) nn.init.normal_(module.out_proj.weight, std=out_proj_std) elif isinstance(module, GroupViTMLP): - factor = self.config.initializer_factor in_proj_std = (module.config.hidden_size**-0.5) * ((2 * module.config.num_hidden_layers) ** -0.5) * factor fc_std = (2 * module.config.hidden_size) ** -0.5 * factor nn.init.normal_(module.fc1.weight, std=fc_std) nn.init.normal_(module.fc2.weight, std=in_proj_std) + elif isinstance(module, GroupViTStage): + if module.num_group_token > 0: + # only zero init group token if we have a projection + if module.group_projector is not None: + module.group_token.data.zero_() + else: + nn.init.trunc_normal_(module.group_token, std=factor * 0.02) + elif isinstance(module, GroupViTModel): + module.logit_scale.data.fill_(self.config.logit_scale_init_value) + elif isinstance(module, (nn.LayerNorm, nn.BatchNorm1d)): + module.bias.data.zero_() + module.weight.data.fill_(1.0) class GroupViTVisionEncoder(nn.Module): From 0f70e77b8b824384073121a83b47e9b4ba171115 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Thu, 31 Jul 2025 06:24:57 +0200 Subject: [PATCH 11/29] fix Swin2SR --- src/transformers/models/swin2sr/modeling_swin2sr.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/src/transformers/models/swin2sr/modeling_swin2sr.py b/src/transformers/models/swin2sr/modeling_swin2sr.py index be92ac9e70df..91dd65b91928 100644 --- a/src/transformers/models/swin2sr/modeling_swin2sr.py +++ b/src/transformers/models/swin2sr/modeling_swin2sr.py @@ -728,12 +728,14 @@ class Swin2SRPreTrainedModel(PreTrainedModel): main_input_name = "pixel_values" supports_gradient_checkpointing = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" if isinstance(module, (nn.Linear, nn.Conv2d)): - torch.nn.init.trunc_normal_(module.weight.data, std=self.config.initializer_range) + nn.init.trunc_normal_(module.weight, std=self.config.initializer_range) if module.bias is not None: module.bias.data.zero_() + elif isinstance(module, Swin2SRSelfAttention): + module.logit_scale.data.fill_(math.log(10)) elif isinstance(module, nn.LayerNorm): module.bias.data.zero_() module.weight.data.fill_(1.0) From 0a8b3bf37ce99a50f7d55b39eca038a82612ad4e Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Thu, 31 Jul 2025 06:29:34 +0200 Subject: [PATCH 12/29] fix Mvp --- src/transformers/models/mvp/modeling_mvp.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/src/transformers/models/mvp/modeling_mvp.py b/src/transformers/models/mvp/modeling_mvp.py index 52da33565d1a..276339980970 100644 --- a/src/transformers/models/mvp/modeling_mvp.py +++ b/src/transformers/models/mvp/modeling_mvp.py @@ -489,7 +489,7 @@ class MvpPreTrainedModel(PreTrainedModel): base_model_prefix = "model" supports_gradient_checkpointing = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): std = self.config.init_std if isinstance(module, nn.Linear): module.weight.data.normal_(mean=0.0, std=std) @@ -499,6 +499,9 @@ def _init_weights(self, module): module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() + elif isinstance(module, nn.LayerNorm): + module.bias.data.zero_() + module.weight.data.fill_(1.0) @property def dummy_inputs(self): From e72fac723b2be01bb5131c0c493c116d759a8695 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Thu, 31 Jul 2025 08:57:34 +0200 Subject: [PATCH 13/29] fix DeformableDetr --- .../modeling_deformable_detr.py | 30 +++++++++++-------- 1 file changed, 17 insertions(+), 13 deletions(-) diff --git a/src/transformers/models/deformable_detr/modeling_deformable_detr.py b/src/transformers/models/deformable_detr/modeling_deformable_detr.py index db74c715b565..6d75ec3a6070 100755 --- a/src/transformers/models/deformable_detr/modeling_deformable_detr.py +++ b/src/transformers/models/deformable_detr/modeling_deformable_detr.py @@ -926,14 +926,14 @@ class DeformableDetrPreTrainedModel(PreTrainedModel): r"DeformableDetrDecoderLayer", ] - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): std = self.config.init_std if isinstance(module, DeformableDetrLearnedPositionEmbedding): nn.init.uniform_(module.row_embeddings.weight) nn.init.uniform_(module.column_embeddings.weight) elif isinstance(module, DeformableDetrMultiscaleDeformableAttention): - nn.init.constant_(module.sampling_offsets.weight.data, 0.0) + nn.init.constant_(module.sampling_offsets.weight, 0.0) default_dtype = torch.get_default_dtype() thetas = torch.arange(module.n_heads, dtype=torch.int64).to(default_dtype) * ( 2.0 * math.pi / module.n_heads @@ -946,15 +946,15 @@ def _init_weights(self, module): ) for i in range(module.n_points): grid_init[:, :, i, :] *= i + 1 - with torch.no_grad(): - module.sampling_offsets.bias = nn.Parameter(grid_init.view(-1)) - nn.init.constant_(module.attention_weights.weight.data, 0.0) - nn.init.constant_(module.attention_weights.bias.data, 0.0) - nn.init.xavier_uniform_(module.value_proj.weight.data) - nn.init.constant_(module.value_proj.bias.data, 0.0) - nn.init.xavier_uniform_(module.output_proj.weight.data) - nn.init.constant_(module.output_proj.bias.data, 0.0) - elif isinstance(module, (nn.Linear, nn.Conv2d, nn.BatchNorm2d)): + + module.sampling_offsets.bias = nn.Parameter(grid_init.view(-1)) + nn.init.constant_(module.attention_weights.weight, 0.0) + nn.init.constant_(module.attention_weights.bias, 0.0) + nn.init.xavier_uniform_(module.value_proj.weight) + nn.init.constant_(module.value_proj.bias, 0.0) + nn.init.xavier_uniform_(module.output_proj.weight) + nn.init.constant_(module.output_proj.bias, 0.0) + elif isinstance(module, (nn.Linear, nn.Conv2d)): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 module.weight.data.normal_(mean=0.0, std=std) @@ -964,9 +964,13 @@ def _init_weights(self, module): module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() + elif isinstance(module, (nn.LayerNorm, nn.GroupNorm)): + module.bias.data.zero_() + module.weight.data.fill_(1.0) + if hasattr(module, "reference_points") and not self.config.two_stage: - nn.init.xavier_uniform_(module.reference_points.weight.data, gain=1.0) - nn.init.constant_(module.reference_points.bias.data, 0.0) + nn.init.xavier_uniform_(module.reference_points.weight, gain=1.0) + nn.init.constant_(module.reference_points.bias, 0.0) if hasattr(module, "level_embed"): nn.init.normal_(module.level_embed) From 198942da7b4017982e7920ab66a688aa0062c6b7 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Thu, 31 Jul 2025 09:10:52 +0200 Subject: [PATCH 14/29] fix Nystromformer --- .../models/nystromformer/modeling_nystromformer.py | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/src/transformers/models/nystromformer/modeling_nystromformer.py b/src/transformers/models/nystromformer/modeling_nystromformer.py index 45e69b6b4693..779ec36028eb 100755 --- a/src/transformers/models/nystromformer/modeling_nystromformer.py +++ b/src/transformers/models/nystromformer/modeling_nystromformer.py @@ -447,21 +447,24 @@ class NystromformerPreTrainedModel(PreTrainedModel): base_model_prefix = "nystromformer" supports_gradient_checkpointing = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" + std = self.config.initializer_range if isinstance(module, (nn.Linear, nn.Conv2d)): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.Embedding): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() elif isinstance(module, nn.LayerNorm): module.bias.data.zero_() module.weight.data.fill_(1.0) + elif isinstance(module, NystromformerLMPredictionHead): + module.bias.data.zero_() @auto_docstring From b1f0b4747efa5606ebf193b391dad6c5a57e727e Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Thu, 31 Jul 2025 09:38:27 +0200 Subject: [PATCH 15/29] fix RegNet and ResNet --- src/transformers/models/regnet/modeling_regnet.py | 6 ++++-- src/transformers/models/resnet/modeling_resnet.py | 6 ++++-- src/transformers/models/rt_detr/modeling_rt_detr_resnet.py | 6 ++++-- tests/models/regnet/test_modeling_regnet.py | 2 +- tests/models/resnet/test_modeling_resnet.py | 2 +- 5 files changed, 14 insertions(+), 8 deletions(-) diff --git a/src/transformers/models/regnet/modeling_regnet.py b/src/transformers/models/regnet/modeling_regnet.py index c9cdda640b60..fb9cf2be4d9d 100644 --- a/src/transformers/models/regnet/modeling_regnet.py +++ b/src/transformers/models/regnet/modeling_regnet.py @@ -266,9 +266,11 @@ class RegNetPreTrainedModel(PreTrainedModel): _no_split_modules = ["RegNetYLayer"] # Copied from transformers.models.resnet.modeling_resnet.ResNetPreTrainedModel._init_weights - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): if isinstance(module, nn.Conv2d): nn.init.kaiming_normal_(module.weight, mode="fan_out", nonlinearity="relu") + if module.bias is not None: + module.bias.data.zero_() # copied from the `reset_parameters` method of `class Linear(Module)` in `torch`. elif isinstance(module, nn.Linear): nn.init.kaiming_uniform_(module.weight, a=math.sqrt(5)) @@ -276,7 +278,7 @@ def _init_weights(self, module): fan_in, _ = nn.init._calculate_fan_in_and_fan_out(module.weight) bound = 1 / math.sqrt(fan_in) if fan_in > 0 else 0 nn.init.uniform_(module.bias, -bound, bound) - elif isinstance(module, (nn.BatchNorm2d, nn.GroupNorm)): + elif isinstance(module, nn.BatchNorm2d): nn.init.constant_(module.weight, 1) nn.init.constant_(module.bias, 0) diff --git a/src/transformers/models/resnet/modeling_resnet.py b/src/transformers/models/resnet/modeling_resnet.py index 266d148bcc48..b9d3c005906b 100644 --- a/src/transformers/models/resnet/modeling_resnet.py +++ b/src/transformers/models/resnet/modeling_resnet.py @@ -251,9 +251,11 @@ class ResNetPreTrainedModel(PreTrainedModel): main_input_name = "pixel_values" _no_split_modules = ["ResNetConvLayer", "ResNetShortCut"] - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): if isinstance(module, nn.Conv2d): nn.init.kaiming_normal_(module.weight, mode="fan_out", nonlinearity="relu") + if module.bias is not None: + module.bias.data.zero_() # copied from the `reset_parameters` method of `class Linear(Module)` in `torch`. elif isinstance(module, nn.Linear): nn.init.kaiming_uniform_(module.weight, a=math.sqrt(5)) @@ -261,7 +263,7 @@ def _init_weights(self, module): fan_in, _ = nn.init._calculate_fan_in_and_fan_out(module.weight) bound = 1 / math.sqrt(fan_in) if fan_in > 0 else 0 nn.init.uniform_(module.bias, -bound, bound) - elif isinstance(module, (nn.BatchNorm2d, nn.GroupNorm)): + elif isinstance(module, nn.BatchNorm2d): nn.init.constant_(module.weight, 1) nn.init.constant_(module.bias, 0) diff --git a/src/transformers/models/rt_detr/modeling_rt_detr_resnet.py b/src/transformers/models/rt_detr/modeling_rt_detr_resnet.py index 770a9b032660..b644e31f53ff 100644 --- a/src/transformers/models/rt_detr/modeling_rt_detr_resnet.py +++ b/src/transformers/models/rt_detr/modeling_rt_detr_resnet.py @@ -302,9 +302,11 @@ class RTDetrResNetPreTrainedModel(PreTrainedModel): main_input_name = "pixel_values" _no_split_modules = ["RTDetrResNetConvLayer", "RTDetrResNetShortCut"] - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): if isinstance(module, nn.Conv2d): nn.init.kaiming_normal_(module.weight, mode="fan_out", nonlinearity="relu") + if module.bias is not None: + module.bias.data.zero_() # copied from the `reset_parameters` method of `class Linear(Module)` in `torch`. elif isinstance(module, nn.Linear): nn.init.kaiming_uniform_(module.weight, a=math.sqrt(5)) @@ -312,7 +314,7 @@ def _init_weights(self, module): fan_in, _ = nn.init._calculate_fan_in_and_fan_out(module.weight) bound = 1 / math.sqrt(fan_in) if fan_in > 0 else 0 nn.init.uniform_(module.bias, -bound, bound) - elif isinstance(module, (nn.BatchNorm2d, nn.GroupNorm)): + elif isinstance(module, nn.BatchNorm2d): nn.init.constant_(module.weight, 1) nn.init.constant_(module.bias, 0) diff --git a/tests/models/regnet/test_modeling_regnet.py b/tests/models/regnet/test_modeling_regnet.py index 59cc13d4898c..bef959721fc7 100644 --- a/tests/models/regnet/test_modeling_regnet.py +++ b/tests/models/regnet/test_modeling_regnet.py @@ -168,7 +168,7 @@ def test_initialization(self): for model_class in self.all_model_classes: model = model_class(config=config) for name, module in model.named_modules(): - if isinstance(module, (nn.BatchNorm2d, nn.GroupNorm)): + if isinstance(module, nn.BatchNorm2d): self.assertTrue( torch.all(module.weight == 1), msg=f"Parameter {name} of model {model_class} seems not properly initialized", diff --git a/tests/models/resnet/test_modeling_resnet.py b/tests/models/resnet/test_modeling_resnet.py index 3778bd400544..ba27363a013a 100644 --- a/tests/models/resnet/test_modeling_resnet.py +++ b/tests/models/resnet/test_modeling_resnet.py @@ -213,7 +213,7 @@ def test_initialization(self): for model_class in self.all_model_classes: model = model_class(config=config) for name, module in model.named_modules(): - if isinstance(module, (nn.BatchNorm2d, nn.GroupNorm)): + if isinstance(module, nn.BatchNorm2d): self.assertTrue( torch.all(module.weight == 1), msg=f"Parameter {name} of model {model_class} seems not properly initialized", From 8746788e67cd3029df1f693bea2268107381b4d2 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Thu, 31 Jul 2025 10:42:31 +0200 Subject: [PATCH 16/29] fix Timesformer --- .../models/timesformer/modeling_timesformer.py | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/src/transformers/models/timesformer/modeling_timesformer.py b/src/transformers/models/timesformer/modeling_timesformer.py index c0110b379aac..06adc7e74c33 100644 --- a/src/transformers/models/timesformer/modeling_timesformer.py +++ b/src/transformers/models/timesformer/modeling_timesformer.py @@ -460,18 +460,20 @@ class TimesformerPreTrainedModel(PreTrainedModel): supports_gradient_checkpointing = True _no_split_modules = ["TimesformerLayer"] - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): + std = self.config.initializer_range if isinstance(module, (nn.Linear, nn.Conv2d)): - nn.init.trunc_normal_(module.weight, std=self.config.initializer_range) + nn.init.trunc_normal_(module.weight, std=std) if module.bias is not None: nn.init.constant_(module.bias, 0) elif isinstance(module, nn.LayerNorm): nn.init.constant_(module.bias, 0) nn.init.constant_(module.weight, 1.0) elif isinstance(module, TimesformerEmbeddings): - nn.init.trunc_normal_(module.cls_token, std=self.config.initializer_range) - nn.init.trunc_normal_(module.position_embeddings, std=self.config.initializer_range) - module.patch_embeddings.apply(self._init_weights) + nn.init.trunc_normal_(module.cls_token, std=std) + nn.init.trunc_normal_(module.position_embeddings, std=std) + if self.config.attention_type != "space_only": + module.time_embeddings.data.zero_() @auto_docstring From e63912b774394e765cc9d623cf5c47f30b9ee927 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Thu, 31 Jul 2025 11:29:18 +0200 Subject: [PATCH 17/29] fix AltCLIP and CLIP --- .../models/altclip/modeling_altclip.py | 18 +++++++----------- src/transformers/models/clip/modeling_clip.py | 18 ++++++++---------- 2 files changed, 15 insertions(+), 21 deletions(-) diff --git a/src/transformers/models/altclip/modeling_altclip.py b/src/transformers/models/altclip/modeling_altclip.py index 6581e2f18c44..94374eed1719 100755 --- a/src/transformers/models/altclip/modeling_altclip.py +++ b/src/transformers/models/altclip/modeling_altclip.py @@ -815,16 +815,14 @@ class AltCLIPPreTrainedModel(PreTrainedModel): supports_gradient_checkpointing = True _no_split_module = [] - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" factor = self.config.initializer_factor if isinstance(module, AltCLIPVisionEmbeddings): - factor = self.config.initializer_factor - nn.init.normal_(module.class_embedding, mean=0.0, std=module.embed_dim**-0.5 * factor) + nn.init.normal_(module.class_embedding, std=module.embed_dim**-0.5 * factor) nn.init.normal_(module.patch_embedding.weight, std=module.config.initializer_range * factor) nn.init.normal_(module.position_embedding.weight, std=module.config.initializer_range * factor) elif isinstance(module, AltCLIPAttention): - factor = self.config.initializer_factor in_proj_std = (module.embed_dim**-0.5) * ((2 * module.config.num_hidden_layers) ** -0.5) * factor out_proj_std = (module.embed_dim**-0.5) * factor nn.init.normal_(module.q_proj.weight, std=in_proj_std) @@ -832,7 +830,6 @@ def _init_weights(self, module): nn.init.normal_(module.v_proj.weight, std=in_proj_std) nn.init.normal_(module.out_proj.weight, std=out_proj_std) elif isinstance(module, AltCLIPMLP): - factor = self.config.initializer_factor in_proj_std = (module.config.hidden_size**-0.5) * ((2 * module.config.num_hidden_layers) ** -0.5) * factor fc_std = (2 * module.config.hidden_size) ** -0.5 * factor nn.init.normal_(module.fc1.weight, std=fc_std) @@ -840,23 +837,22 @@ def _init_weights(self, module): elif isinstance(module, AltCLIPModel): nn.init.normal_( module.text_projection.weight, - std=module.text_embed_dim**-0.5 * self.config.initializer_factor, + std=module.text_embed_dim**-0.5 * factor, ) - module.text_projection._is_hf_initialized = True nn.init.normal_( module.visual_projection.weight, - std=module.vision_embed_dim**-0.5 * self.config.initializer_factor, + std=module.vision_embed_dim**-0.5 * factor, ) - module.visual_projection._is_hf_initialized = True + module.logit_scale.data.fill_(self.config.logit_scale_init_value) elif isinstance(module, nn.LayerNorm): module.bias.data.zero_() module.weight.data.fill_(1.0) elif isinstance(module, nn.Linear): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_factor) + module.weight.data.normal_(mean=0.0, std=factor * 0.02) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.Embedding): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_factor) + module.weight.data.normal_(mean=0.0, std=factor * 0.02) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() diff --git a/src/transformers/models/clip/modeling_clip.py b/src/transformers/models/clip/modeling_clip.py index a187bdaa635e..bf63ad0bbea4 100644 --- a/src/transformers/models/clip/modeling_clip.py +++ b/src/transformers/models/clip/modeling_clip.py @@ -432,19 +432,17 @@ class CLIPPreTrainedModel(PreTrainedModel): _supports_flex_attn = True _supports_attention_backend = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" factor = self.config.initializer_factor if isinstance(module, CLIPTextEmbeddings): module.token_embedding.weight.data.normal_(mean=0.0, std=factor * 0.02) module.position_embedding.weight.data.normal_(mean=0.0, std=factor * 0.02) elif isinstance(module, CLIPVisionEmbeddings): - factor = self.config.initializer_factor - nn.init.normal_(module.class_embedding, mean=0.0, std=module.embed_dim**-0.5 * factor) + nn.init.normal_(module.class_embedding, std=module.embed_dim**-0.5 * factor) nn.init.normal_(module.patch_embedding.weight, std=module.config.initializer_range * factor) nn.init.normal_(module.position_embedding.weight, std=module.config.initializer_range * factor) elif isinstance(module, CLIPAttention): - factor = self.config.initializer_factor in_proj_std = (module.embed_dim**-0.5) * ((2 * module.config.num_hidden_layers) ** -0.5) * factor out_proj_std = (module.embed_dim**-0.5) * factor nn.init.normal_(module.q_proj.weight, std=in_proj_std) @@ -452,7 +450,6 @@ def _init_weights(self, module): nn.init.normal_(module.v_proj.weight, std=in_proj_std) nn.init.normal_(module.out_proj.weight, std=out_proj_std) elif isinstance(module, CLIPMLP): - factor = self.config.initializer_factor in_proj_std = (module.config.hidden_size**-0.5) * ((2 * module.config.num_hidden_layers) ** -0.5) * factor fc_std = (2 * module.config.hidden_size) ** -0.5 * factor nn.init.normal_(module.fc1.weight, std=fc_std) @@ -460,26 +457,27 @@ def _init_weights(self, module): elif isinstance(module, CLIPModel): nn.init.normal_( module.text_projection.weight, - std=module.text_embed_dim**-0.5 * self.config.initializer_factor, + std=module.text_embed_dim**-0.5 * factor, ) nn.init.normal_( module.visual_projection.weight, - std=module.vision_embed_dim**-0.5 * self.config.initializer_factor, + std=module.vision_embed_dim**-0.5 * factor, ) + module.logit_scale.data.fill_(self.config.logit_scale_init_value) elif isinstance(module, CLIPVisionModelWithProjection): nn.init.normal_( module.visual_projection.weight, - std=self.config.hidden_size**-0.5 * self.config.initializer_factor, + std=self.config.hidden_size**-0.5 * factor, ) elif isinstance(module, CLIPTextModelWithProjection): nn.init.normal_( module.text_projection.weight, - std=self.config.hidden_size**-0.5 * self.config.initializer_factor, + std=self.config.hidden_size**-0.5 * factor, ) elif isinstance(module, CLIPForImageClassification): nn.init.normal_( module.classifier.weight, - std=self.config.vision_config.hidden_size**-0.5 * self.config.initializer_factor, + std=self.config.vision_config.hidden_size**-0.5 * factor, ) if isinstance(module, nn.LayerNorm): From 07a64430d7132688da4ab0eb1778c5bc3e0f16d0 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Thu, 31 Jul 2025 15:40:32 +0200 Subject: [PATCH 18/29] fix XCLIP --- .../models/x_clip/modeling_x_clip.py | 17 ++++++++--------- 1 file changed, 8 insertions(+), 9 deletions(-) diff --git a/src/transformers/models/x_clip/modeling_x_clip.py b/src/transformers/models/x_clip/modeling_x_clip.py index f6c5c51a27c5..bc6f19cefb92 100644 --- a/src/transformers/models/x_clip/modeling_x_clip.py +++ b/src/transformers/models/x_clip/modeling_x_clip.py @@ -513,19 +513,17 @@ class XCLIPPreTrainedModel(PreTrainedModel): base_model_prefix = "x_clip" supports_gradient_checkpointing = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" factor = self.config.initializer_factor if isinstance(module, XCLIPTextEmbeddings): module.token_embedding.weight.data.normal_(mean=0.0, std=factor * 0.02) module.position_embedding.weight.data.normal_(mean=0.0, std=factor * 0.02) elif isinstance(module, XCLIPVisionEmbeddings): - factor = self.config.initializer_factor - nn.init.normal_(module.class_embedding, mean=0.0, std=module.embed_dim**-0.5 * factor) + nn.init.normal_(module.class_embedding, std=module.embed_dim**-0.5 * factor) nn.init.normal_(module.patch_embedding.weight, std=module.config.initializer_range * factor) nn.init.normal_(module.position_embedding.weight, std=module.config.initializer_range * factor) elif isinstance(module, XCLIPAttention): - factor = self.config.initializer_factor in_proj_std = (module.embed_dim**-0.5) * ((2 * module.config.num_hidden_layers) ** -0.5) * factor out_proj_std = (module.embed_dim**-0.5) * factor nn.init.normal_(module.q_proj.weight, std=in_proj_std) @@ -533,13 +531,11 @@ def _init_weights(self, module): nn.init.normal_(module.v_proj.weight, std=in_proj_std) nn.init.normal_(module.out_proj.weight, std=out_proj_std) elif isinstance(module, XCLIPMLP): - factor = self.config.initializer_factor in_proj_std = (module.config.hidden_size**-0.5) * ((2 * module.config.num_hidden_layers) ** -0.5) * factor fc_std = (2 * module.config.hidden_size) ** -0.5 * factor nn.init.normal_(module.fc1.weight, std=fc_std) nn.init.normal_(module.fc2.weight, std=in_proj_std) elif isinstance(module, XCLIPModel): - factor = self.config.initializer_factor nn.init.normal_( module.text_projection.weight, std=module.text_embed_dim**-0.5 * factor, @@ -548,15 +544,18 @@ def _init_weights(self, module): module.visual_projection.weight, std=module.vision_embed_dim**-0.5 * factor, ) - nn.init.normal_(module.prompts_visual_projection, mean=0.0, std=module.vision_embed_dim**-0.5 * factor) + nn.init.normal_(module.prompts_visual_projection, std=module.vision_embed_dim**-0.5 * factor) + module.logit_scale.data.fill_(self.config.logit_scale_init_value) elif isinstance(module, XCLIPMultiframeIntegrationTransformer): - nn.init.normal_(module.position_embedding, std=self.config.initializer_factor) + nn.init.normal_(module.position_embedding, std=factor * 0.02) + elif isinstance(module, XCLIPPromptGenerator): + module.alpha.data.fill_(self.config.prompt_alpha) if isinstance(module, nn.LayerNorm): module.bias.data.zero_() module.weight.data.fill_(1.0) if isinstance(module, nn.Linear): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_factor) + module.weight.data.normal_(mean=0.0, std=factor * 0.02) if module.bias is not None: module.bias.data.zero_() From c58c3c8741b7e17a1d67cf1626543556e5618e99 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Thu, 31 Jul 2025 15:58:50 +0200 Subject: [PATCH 19/29] fix VideoMAE --- src/transformers/models/videomae/modeling_videomae.py | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/src/transformers/models/videomae/modeling_videomae.py b/src/transformers/models/videomae/modeling_videomae.py index 7b22eb9a4944..a7251977e6f7 100755 --- a/src/transformers/models/videomae/modeling_videomae.py +++ b/src/transformers/models/videomae/modeling_videomae.py @@ -471,14 +471,21 @@ class VideoMAEPreTrainedModel(PreTrainedModel): _supports_flex_attn = True _supports_attention_backend = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" + std = self.config.initializer_range if isinstance(module, (nn.Linear, nn.Conv3d)): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.bias is not None: module.bias.data.zero_() + elif isinstance(module, VideoMAESelfAttention): + if self.config.qkv_bias: + module.q_bias.data.zero_() + module.v_bias.data.zero_() + elif isinstance(module, VideoMAEForPreTraining): + module.mask_token.data.normal_(mean=0.0, std=std) elif isinstance(module, nn.LayerNorm): module.bias.data.zero_() module.weight.data.fill_(1.0) From b954115396a7ca392d7499bd10821e72130fc2b2 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Thu, 31 Jul 2025 16:05:38 +0200 Subject: [PATCH 20/29] fix TimeSeriesTransformer --- .../modeling_time_series_transformer.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/src/transformers/models/time_series_transformer/modeling_time_series_transformer.py b/src/transformers/models/time_series_transformer/modeling_time_series_transformer.py index 84d5de004c51..94c227914eba 100644 --- a/src/transformers/models/time_series_transformer/modeling_time_series_transformer.py +++ b/src/transformers/models/time_series_transformer/modeling_time_series_transformer.py @@ -636,7 +636,7 @@ class TimeSeriesTransformerPreTrainedModel(PreTrainedModel): _supports_sdpa = False _supports_flex_attn = False - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): std = self.config.init_std if isinstance(module, nn.Linear): module.weight.data.normal_(mean=0.0, std=std) @@ -648,6 +648,9 @@ def _init_weights(self, module): module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() + elif isinstance(module, nn.LayerNorm): + module.bias.data.zero_() + module.weight.data.fill_(1.0) # Copied from transformers.models.bart.modeling_bart.BartPreTrainedModel._update_full_mask def _update_full_mask( From 40c72405891c7ec1efd5e385c52ab74f9b970e5a Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Thu, 31 Jul 2025 16:44:42 +0200 Subject: [PATCH 21/29] fix Blip --- src/transformers/models/blip/modeling_blip.py | 22 +++++-------------- .../models/blip/modeling_blip_text.py | 4 ++-- 2 files changed, 8 insertions(+), 18 deletions(-) diff --git a/src/transformers/models/blip/modeling_blip.py b/src/transformers/models/blip/modeling_blip.py index 267b0ffcb0ca..2030150fe626 100644 --- a/src/transformers/models/blip/modeling_blip.py +++ b/src/transformers/models/blip/modeling_blip.py @@ -440,29 +440,19 @@ class BlipPreTrainedModel(PreTrainedModel): _no_split_modules = ["BlipEncoderLayer", "BlipTextEmbeddings"] _skip_keys_device_placement = ["past_key_value"] - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" - factor = self.config.initializer_range + std = self.config.initializer_range if isinstance(module, (nn.Conv2d, nn.Embedding, nn.Linear)): - module.weight.data.normal_(mean=0.0, std=factor) + module.weight.data.normal_(mean=0.0, std=std) if hasattr(module, "bias") and module.bias is not None: module.bias.data.zero_() if isinstance(module, BlipVisionEmbeddings): if hasattr(self.config, "vision_config"): - factor = self.config.vision_config.initializer_range - nn.init.trunc_normal_( - module.position_embedding, - mean=0.0, - std=factor, - ) - - nn.init.trunc_normal_( - module.class_embedding, - mean=0.0, - std=factor, - ) - + std = self.config.vision_config.initializer_range + nn.init.trunc_normal_(module.position_embedding, std=std) + nn.init.trunc_normal_(module.class_embedding, std=std) elif isinstance(module, nn.LayerNorm): module.bias.data.zero_() module.weight.data.fill_(1.0) diff --git a/src/transformers/models/blip/modeling_blip_text.py b/src/transformers/models/blip/modeling_blip_text.py index 0c6e8777fd05..3c8269514507 100644 --- a/src/transformers/models/blip/modeling_blip_text.py +++ b/src/transformers/models/blip/modeling_blip_text.py @@ -579,7 +579,7 @@ class BlipTextPreTrainedModel(PreTrainedModel): base_model_prefix = "bert" _no_split_modules = [] - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" if isinstance(module, (nn.Linear, nn.Embedding)): # Slightly different from the TF version which uses truncated_normal for initialization @@ -588,7 +588,7 @@ def _init_weights(self, module): elif isinstance(module, nn.LayerNorm): module.bias.data.zero_() module.weight.data.fill_(1.0) - if isinstance(module, nn.Linear) and module.bias is not None: + if isinstance(module, (nn.Linear, BlipTextLMPredictionHead)) and module.bias is not None: module.bias.data.zero_() From da184dc4acea2dd7f34dfa5a4c6ffcfe456dfab6 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Thu, 31 Jul 2025 16:51:52 +0200 Subject: [PATCH 22/29] fix Ernie --- src/transformers/models/ernie/modeling_ernie.py | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/src/transformers/models/ernie/modeling_ernie.py b/src/transformers/models/ernie/modeling_ernie.py index f2aaa897237e..c8a552ebc2a6 100644 --- a/src/transformers/models/ernie/modeling_ernie.py +++ b/src/transformers/models/ernie/modeling_ernie.py @@ -622,21 +622,24 @@ class ErniePreTrainedModel(PreTrainedModel): base_model_prefix = "ernie" supports_gradient_checkpointing = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" + std = self.config.initializer_range if isinstance(module, nn.Linear): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.Embedding): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() elif isinstance(module, nn.LayerNorm): module.bias.data.zero_() module.weight.data.fill_(1.0) + elif isinstance(module, ErnieLMPredictionHead): + module.bias.data.zero_() @dataclass From 4f637155320a8fb4c878592d542338e09bc2be35 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Fri, 1 Aug 2025 10:11:25 +0200 Subject: [PATCH 23/29] fix OwlViTForObjectDetection --- src/transformers/models/owlvit/modeling_owlvit.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/transformers/models/owlvit/modeling_owlvit.py b/src/transformers/models/owlvit/modeling_owlvit.py index e7d38fe3f361..93f7968586c3 100644 --- a/src/transformers/models/owlvit/modeling_owlvit.py +++ b/src/transformers/models/owlvit/modeling_owlvit.py @@ -1203,6 +1203,9 @@ def __init__(self, config: OwlViTConfig): self.num_patches_width = self.config.vision_config.image_size // self.config.vision_config.patch_size self.box_bias = self.compute_box_bias(self.num_patches_height, self.num_patches_width) + # Initialize weights and apply final processing + self.post_init() + @staticmethod def normalize_grid_corner_coordinates(num_patches_height: int, num_patches_width: int) -> torch.Tensor: # Create grid coordinates using torch From c6f3b32e13a90338fcc0afa2e66555b136fc97e8 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Fri, 1 Aug 2025 10:26:04 +0200 Subject: [PATCH 24/29] fix Blip --- src/transformers/models/blip/modeling_blip.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/src/transformers/models/blip/modeling_blip.py b/src/transformers/models/blip/modeling_blip.py index 2030150fe626..6076c097c350 100644 --- a/src/transformers/models/blip/modeling_blip.py +++ b/src/transformers/models/blip/modeling_blip.py @@ -453,6 +453,8 @@ def _init_weights(self, module: nn.Module): std = self.config.vision_config.initializer_range nn.init.trunc_normal_(module.position_embedding, std=std) nn.init.trunc_normal_(module.class_embedding, std=std) + elif isinstance(module, BlipModel): + module.logit_scale.data.fill_(self.config.logit_scale_init_value) elif isinstance(module, nn.LayerNorm): module.bias.data.zero_() module.weight.data.fill_(1.0) From a586d9951acb5b6a10c3f442e6f06c5b4dfa29a5 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Fri, 1 Aug 2025 10:56:40 +0200 Subject: [PATCH 25/29] fix MaskFormerSwin --- .../models/maskformer/modeling_maskformer_swin.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/src/transformers/models/maskformer/modeling_maskformer_swin.py b/src/transformers/models/maskformer/modeling_maskformer_swin.py index 22e91c0970bf..5727f8e62c59 100644 --- a/src/transformers/models/maskformer/modeling_maskformer_swin.py +++ b/src/transformers/models/maskformer/modeling_maskformer_swin.py @@ -740,7 +740,7 @@ class MaskFormerSwinPreTrainedModel(PreTrainedModel): supports_gradient_checkpointing = True _no_split_modules = ["MaskFormerSwinStage"] - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" if isinstance(module, (nn.Linear, nn.Conv2d)): # Slightly different from the TF version which uses truncated_normal for initialization @@ -771,6 +771,9 @@ def __init__(self, config, add_pooling_layer=True): self.layernorm = nn.LayerNorm(self.num_features, eps=config.layer_norm_eps) self.pooler = nn.AdaptiveAvgPool1d(1) if add_pooling_layer else None + # Initialize weights and apply final processing + self.post_init() + def get_input_embeddings(self): return self.embeddings.patch_embeddings From b7d93f77d5781a745f359c4c455ecebd03a6f8d8 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Fri, 1 Aug 2025 11:11:32 +0200 Subject: [PATCH 26/29] fix Mgpstr --- src/transformers/models/mgp_str/modeling_mgp_str.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/transformers/models/mgp_str/modeling_mgp_str.py b/src/transformers/models/mgp_str/modeling_mgp_str.py index 490224d070d8..be89317972ec 100644 --- a/src/transformers/models/mgp_str/modeling_mgp_str.py +++ b/src/transformers/models/mgp_str/modeling_mgp_str.py @@ -294,7 +294,7 @@ def _init_weights(self, module: nn.Module) -> None: """Initialize the weights""" std = self.config.initializer_range if isinstance(module, MgpstrEmbeddings): - nn.init.trunc_normal_(module.pos_embe, std=std) + nn.init.trunc_normal_(module.pos_embed, std=std) nn.init.trunc_normal_(module.cls_token, std=std) elif isinstance(module, (nn.Linear, nn.Conv2d)): nn.init.trunc_normal_(module.weight, std=std) From 48b81feb5537e72cd781929a8824aa71c7993ff9 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Fri, 1 Aug 2025 11:26:39 +0200 Subject: [PATCH 27/29] fix Vilt --- src/transformers/models/vilt/modeling_vilt.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/src/transformers/models/vilt/modeling_vilt.py b/src/transformers/models/vilt/modeling_vilt.py index f0834b829520..f0f0bf829f85 100755 --- a/src/transformers/models/vilt/modeling_vilt.py +++ b/src/transformers/models/vilt/modeling_vilt.py @@ -569,6 +569,8 @@ def _init_weights(self, module: nn.Module): elif isinstance(module, nn.LayerNorm): module.bias.data.zero_() module.weight.data.fill_(1.0) + elif isinstance(module, ViltMLMHead): + module.bias.data.zero_() @auto_docstring From 0220b9db084f546f589b4adc3f08ea0d4faef7ca Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Fri, 1 Aug 2025 15:49:07 +0200 Subject: [PATCH 28/29] fix Esm --- src/transformers/models/bert/modeling_bert.py | 7 +- .../models/camembert/modeling_camembert.py | 7 +- src/transformers/models/esm/modeling_esm.py | 7 +- .../models/esm/modeling_esmfold.py | 88 +++++++++++-------- .../models/markuplm/modeling_markuplm.py | 7 +- .../models/roberta/modeling_roberta.py | 7 +- .../modeling_roberta_prelayernorm.py | 7 +- .../models/tapas/modeling_tapas.py | 7 +- .../xlm_roberta/modeling_xlm_roberta.py | 7 +- .../xlm_roberta_xl/modeling_xlm_roberta_xl.py | 7 +- src/transformers/models/xmod/modeling_xmod.py | 7 +- 11 files changed, 91 insertions(+), 67 deletions(-) diff --git a/src/transformers/models/bert/modeling_bert.py b/src/transformers/models/bert/modeling_bert.py index 90ed959176ad..745c7b79590a 100755 --- a/src/transformers/models/bert/modeling_bert.py +++ b/src/transformers/models/bert/modeling_bert.py @@ -789,16 +789,17 @@ class BertPreTrainedModel(PreTrainedModel): supports_gradient_checkpointing = True _supports_sdpa = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" + std = self.config.initializer_range if isinstance(module, nn.Linear): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.Embedding): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() elif isinstance(module, nn.LayerNorm): diff --git a/src/transformers/models/camembert/modeling_camembert.py b/src/transformers/models/camembert/modeling_camembert.py index bb983794ab74..97eafd28fadc 100644 --- a/src/transformers/models/camembert/modeling_camembert.py +++ b/src/transformers/models/camembert/modeling_camembert.py @@ -675,16 +675,17 @@ class CamembertPreTrainedModel(PreTrainedModel): _supports_sdpa = True # Copied from transformers.models.bert.modeling_bert.BertPreTrainedModel._init_weights with BertLMPredictionHead->CamembertLMHead - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" + std = self.config.initializer_range if isinstance(module, nn.Linear): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.Embedding): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() elif isinstance(module, nn.LayerNorm): diff --git a/src/transformers/models/esm/modeling_esm.py b/src/transformers/models/esm/modeling_esm.py index 9acc3625bd69..905a56105b29 100755 --- a/src/transformers/models/esm/modeling_esm.py +++ b/src/transformers/models/esm/modeling_esm.py @@ -733,16 +733,17 @@ class EsmPreTrainedModel(PreTrainedModel): _supports_flash_attn = True # Copied from transformers.models.bert.modeling_bert.BertPreTrainedModel._init_weights with BertLMPredictionHead->EsmLMHead - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" + std = self.config.initializer_range if isinstance(module, nn.Linear): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.Embedding): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() elif isinstance(module, nn.LayerNorm): diff --git a/src/transformers/models/esm/modeling_esmfold.py b/src/transformers/models/esm/modeling_esmfold.py index 8c74afdc7c7a..b3c3dd0ac119 100644 --- a/src/transformers/models/esm/modeling_esmfold.py +++ b/src/transformers/models/esm/modeling_esmfold.py @@ -915,47 +915,58 @@ class EsmFoldPreTrainedModel(EsmPreTrainedModel): """ # Subclass `EsMPreTrainedModel` to deal with special init - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" if isinstance(module, EsmFoldLinear): - with torch.no_grad(): - if module.init_fn is not None: - module.init_fn(module.weight, module.bias) - elif module.init == "default": - trunc_normal_init_(module.weight, scale=1.0) - elif module.init == "relu": - trunc_normal_init_(module.weight, scale=2.0) - elif module.init == "glorot": - nn.init.xavier_uniform_(module.weight, gain=1) - elif module.init == "gating": - module.weight.fill_(0.0) - if module.bias: - module.bias.fill_(1.0) - elif module.init == "normal": - torch.nn.init.kaiming_normal_(module.weight, nonlinearity="linear") - elif module.init == "final": - module.weight.fill_(0.0) + if module.init_fn is not None: + module.init_fn(module.weight, module.bias) + elif module.init == "default": + trunc_normal_init_(module.weight, scale=1.0) + if module.bias is not None: + module.bias.data.zero_() + elif module.init == "relu": + trunc_normal_init_(module.weight, scale=2.0) + if module.bias is not None: + module.bias.data.zero_() + elif module.init == "glorot": + nn.init.xavier_uniform_(module.weight, gain=1) + if module.bias is not None: + module.bias.data.zero_() + elif module.init == "gating": + module.weight.fill_(0.0) + if module.bias is not None: + module.bias.fill_(1.0) + elif module.init == "normal": + torch.nn.init.kaiming_normal_(module.weight, nonlinearity="linear") + if module.bias is not None: + module.bias.data.zero_() + elif module.init == "final": + module.weight.fill_(0.0) + if module.bias is not None: + module.bias.data.zero_() elif isinstance(module, EsmFoldInvariantPointAttention): ipa_point_weights_init_(module.head_weights) elif isinstance(module, EsmFoldTriangularSelfAttentionBlock): - torch.nn.init.zeros_(module.tri_mul_in.linear_z.weight) - torch.nn.init.zeros_(module.tri_mul_in.linear_z.bias) - torch.nn.init.zeros_(module.tri_mul_out.linear_z.weight) - torch.nn.init.zeros_(module.tri_mul_out.linear_z.bias) - torch.nn.init.zeros_(module.tri_att_start.mha.linear_o.weight) - torch.nn.init.zeros_(module.tri_att_start.mha.linear_o.bias) - torch.nn.init.zeros_(module.tri_att_end.mha.linear_o.weight) - torch.nn.init.zeros_(module.tri_att_end.mha.linear_o.bias) - - torch.nn.init.zeros_(module.sequence_to_pair.o_proj.weight) - torch.nn.init.zeros_(module.sequence_to_pair.o_proj.bias) - torch.nn.init.zeros_(module.pair_to_sequence.linear.weight) - torch.nn.init.zeros_(module.seq_attention.o_proj.weight) - torch.nn.init.zeros_(module.seq_attention.o_proj.bias) - torch.nn.init.zeros_(module.mlp_seq.mlp[-2].weight) - torch.nn.init.zeros_(module.mlp_seq.mlp[-2].bias) - torch.nn.init.zeros_(module.mlp_pair.mlp[-2].weight) - torch.nn.init.zeros_(module.mlp_pair.mlp[-2].bias) + nn.init.zeros_(module.tri_mul_in.linear_z.weight) + nn.init.zeros_(module.tri_mul_in.linear_z.bias) + nn.init.zeros_(module.tri_mul_out.linear_z.weight) + nn.init.zeros_(module.tri_mul_out.linear_z.bias) + nn.init.zeros_(module.tri_att_start.mha.linear_o.weight) + nn.init.zeros_(module.tri_att_start.mha.linear_o.bias) + nn.init.zeros_(module.tri_att_end.mha.linear_o.weight) + nn.init.zeros_(module.tri_att_end.mha.linear_o.bias) + + nn.init.zeros_(module.sequence_to_pair.o_proj.weight) + nn.init.zeros_(module.sequence_to_pair.o_proj.bias) + nn.init.zeros_(module.pair_to_sequence.linear.weight) + nn.init.zeros_(module.seq_attention.o_proj.weight) + nn.init.zeros_(module.seq_attention.o_proj.bias) + nn.init.zeros_(module.mlp_seq.mlp[-2].weight) + nn.init.zeros_(module.mlp_seq.mlp[-2].bias) + nn.init.zeros_(module.mlp_pair.mlp[-2].weight) + nn.init.zeros_(module.mlp_pair.mlp[-2].bias) + elif isinstance(module, EsmForProteinFolding): + module.esm_s_combine.data.zero_() else: super()._init_weights(module) @@ -1988,7 +1999,7 @@ def distogram(coords, min_bin, max_bin, num_bins): protein(s). """ ) -class EsmForProteinFolding(EsmPreTrainedModel): +class EsmForProteinFolding(EsmFoldPreTrainedModel): _no_split_modules = ["EsmFoldStructureModule", "EsmFoldTriangularSelfAttentionBlock"] _supports_flash_attn = False @@ -2047,6 +2058,9 @@ def __init__(self, config): nn.Linear(self.config.esmfold_config.lddt_head_hid_dim, 37 * self.lddt_bins), ) + # Initialize weights and apply final processing + self.post_init() + @staticmethod def _af2_to_esm_from_vocab_list(vocab_list: list[str]) -> torch.Tensor: # Remember that t is shifted from residue_constants by 1 (0 is padding). diff --git a/src/transformers/models/markuplm/modeling_markuplm.py b/src/transformers/models/markuplm/modeling_markuplm.py index 0dd845ecff3c..e01e0c1a2af2 100755 --- a/src/transformers/models/markuplm/modeling_markuplm.py +++ b/src/transformers/models/markuplm/modeling_markuplm.py @@ -556,16 +556,17 @@ class MarkupLMPreTrainedModel(PreTrainedModel): base_model_prefix = "markuplm" # Copied from transformers.models.bert.modeling_bert.BertPreTrainedModel._init_weights with Bert->MarkupLM - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" + std = self.config.initializer_range if isinstance(module, nn.Linear): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.Embedding): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() elif isinstance(module, nn.LayerNorm): diff --git a/src/transformers/models/roberta/modeling_roberta.py b/src/transformers/models/roberta/modeling_roberta.py index 26a10273a5fe..6f39a606e207 100644 --- a/src/transformers/models/roberta/modeling_roberta.py +++ b/src/transformers/models/roberta/modeling_roberta.py @@ -675,16 +675,17 @@ class RobertaPreTrainedModel(PreTrainedModel): _supports_sdpa = True # Copied from transformers.models.bert.modeling_bert.BertPreTrainedModel._init_weights with BertLMPredictionHead->RobertaLMHead - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" + std = self.config.initializer_range if isinstance(module, nn.Linear): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.Embedding): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() elif isinstance(module, nn.LayerNorm): diff --git a/src/transformers/models/roberta_prelayernorm/modeling_roberta_prelayernorm.py b/src/transformers/models/roberta_prelayernorm/modeling_roberta_prelayernorm.py index c36f029cf35e..77037602632f 100644 --- a/src/transformers/models/roberta_prelayernorm/modeling_roberta_prelayernorm.py +++ b/src/transformers/models/roberta_prelayernorm/modeling_roberta_prelayernorm.py @@ -556,16 +556,17 @@ class RobertaPreLayerNormPreTrainedModel(PreTrainedModel): _no_split_modules = ["RobertaPreLayerNormEmbeddings", "RobertaPreLayerNormSelfAttention"] # Copied from transformers.models.bert.modeling_bert.BertPreTrainedModel._init_weights with BertLMPredictionHead->RobertaPreLayerNormLMHead - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" + std = self.config.initializer_range if isinstance(module, nn.Linear): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.Embedding): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() elif isinstance(module, nn.LayerNorm): diff --git a/src/transformers/models/tapas/modeling_tapas.py b/src/transformers/models/tapas/modeling_tapas.py index b8a681a0aacd..0916d1e834ec 100644 --- a/src/transformers/models/tapas/modeling_tapas.py +++ b/src/transformers/models/tapas/modeling_tapas.py @@ -692,16 +692,17 @@ class TapasPreTrainedModel(PreTrainedModel): _supports_param_buffer_assignment = False # Copied from transformers.models.bert.modeling_bert.BertPreTrainedModel._init_weights with Bert->Tapas - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" + std = self.config.initializer_range if isinstance(module, nn.Linear): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.Embedding): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() elif isinstance(module, nn.LayerNorm): diff --git a/src/transformers/models/xlm_roberta/modeling_xlm_roberta.py b/src/transformers/models/xlm_roberta/modeling_xlm_roberta.py index d1c70b602899..f0cc8eec3af2 100644 --- a/src/transformers/models/xlm_roberta/modeling_xlm_roberta.py +++ b/src/transformers/models/xlm_roberta/modeling_xlm_roberta.py @@ -677,16 +677,17 @@ class XLMRobertaPreTrainedModel(PreTrainedModel): _supports_sdpa = True # Copied from transformers.models.bert.modeling_bert.BertPreTrainedModel._init_weights with BertLMPredictionHead->XLMRobertaLMHead - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" + std = self.config.initializer_range if isinstance(module, nn.Linear): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.Embedding): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() elif isinstance(module, nn.LayerNorm): diff --git a/src/transformers/models/xlm_roberta_xl/modeling_xlm_roberta_xl.py b/src/transformers/models/xlm_roberta_xl/modeling_xlm_roberta_xl.py index 62793c38bca8..030344dd63f1 100644 --- a/src/transformers/models/xlm_roberta_xl/modeling_xlm_roberta_xl.py +++ b/src/transformers/models/xlm_roberta_xl/modeling_xlm_roberta_xl.py @@ -670,16 +670,17 @@ class XLMRobertaXLPreTrainedModel(PreTrainedModel): _supports_sdpa = True # Copied from transformers.models.bert.modeling_bert.BertPreTrainedModel._init_weights with BertLMPredictionHead->XLMRobertaXLLMHead - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" + std = self.config.initializer_range if isinstance(module, nn.Linear): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.Embedding): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() elif isinstance(module, nn.LayerNorm): diff --git a/src/transformers/models/xmod/modeling_xmod.py b/src/transformers/models/xmod/modeling_xmod.py index 6b4ac64f4e96..191eabafc956 100644 --- a/src/transformers/models/xmod/modeling_xmod.py +++ b/src/transformers/models/xmod/modeling_xmod.py @@ -621,16 +621,17 @@ class XmodPreTrainedModel(PreTrainedModel): supports_gradient_checkpointing = True # Copied from transformers.models.bert.modeling_bert.BertPreTrainedModel._init_weights with BertLMPredictionHead->XmodLMHead - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" + std = self.config.initializer_range if isinstance(module, nn.Linear): # Slightly different from the TF version which uses truncated_normal for initialization # cf https://github.com/pytorch/pytorch/pull/5617 - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.bias is not None: module.bias.data.zero_() elif isinstance(module, nn.Embedding): - module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) + module.weight.data.normal_(mean=0.0, std=std) if module.padding_idx is not None: module.weight.data[module.padding_idx].zero_() elif isinstance(module, nn.LayerNorm): From 91fa7d2ec9939dc54a628723f650be6f96aa1b12 Mon Sep 17 00:00:00 2001 From: BUI Van Tuan Date: Thu, 14 Aug 2025 09:45:12 +0200 Subject: [PATCH 29/29] fix Data2VecAudio --- .../models/data2vec/modeling_data2vec_audio.py | 11 +++++++---- .../models/data2vec/modular_data2vec_audio.py | 11 +++++++---- 2 files changed, 14 insertions(+), 8 deletions(-) diff --git a/src/transformers/models/data2vec/modeling_data2vec_audio.py b/src/transformers/models/data2vec/modeling_data2vec_audio.py index c9b3f01f42d4..e28c6a75c56b 100755 --- a/src/transformers/models/data2vec/modeling_data2vec_audio.py +++ b/src/transformers/models/data2vec/modeling_data2vec_audio.py @@ -506,7 +506,7 @@ class Data2VecAudioPreTrainedModel(PreTrainedModel): _supports_sdpa = True _supports_flex_attn = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" if isinstance(module, Data2VecAudioFeatureProjection): k = math.sqrt(1 / module.projection.in_features) @@ -516,21 +516,24 @@ def _init_weights(self, module): nn.init.constant_(module.conv.bias, 0) elif isinstance(module, nn.Linear): module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) - if module.bias is not None: module.bias.data.zero_() - elif isinstance(module, (nn.LayerNorm, nn.GroupNorm)): + elif isinstance(module, nn.LayerNorm): if module.bias is not None: module.bias.data.zero_() if module.weight is not None: module.weight.data.fill_(1.0) elif isinstance(module, nn.Conv1d): nn.init.kaiming_normal_(module.weight) - if module.bias is not None: k = math.sqrt(module.groups / (module.in_channels * module.kernel_size[0])) nn.init.uniform_(module.bias, a=-k, b=k) + if hasattr(module, "objective"): + nn.init.normal_(module.objective.weight) + if hasattr(module, "masked_spec_embed"): + nn.init.uniform_(module.masked_spec_embed) + def _get_feat_extract_output_lengths( self, input_lengths: Union[torch.LongTensor, int], add_adapter: Optional[bool] = None ): diff --git a/src/transformers/models/data2vec/modular_data2vec_audio.py b/src/transformers/models/data2vec/modular_data2vec_audio.py index 0be5019c016c..10159fb21017 100644 --- a/src/transformers/models/data2vec/modular_data2vec_audio.py +++ b/src/transformers/models/data2vec/modular_data2vec_audio.py @@ -143,7 +143,7 @@ class Data2VecAudioPreTrainedModel(PreTrainedModel, Wav2Vec2PreTrainedModel): _supports_sdpa = True _supports_flex_attn = True - def _init_weights(self, module): + def _init_weights(self, module: nn.Module): """Initialize the weights""" if isinstance(module, Data2VecAudioFeatureProjection): k = math.sqrt(1 / module.projection.in_features) @@ -153,21 +153,24 @@ def _init_weights(self, module): nn.init.constant_(module.conv.bias, 0) elif isinstance(module, nn.Linear): module.weight.data.normal_(mean=0.0, std=self.config.initializer_range) - if module.bias is not None: module.bias.data.zero_() - elif isinstance(module, (nn.LayerNorm, nn.GroupNorm)): + elif isinstance(module, nn.LayerNorm): if module.bias is not None: module.bias.data.zero_() if module.weight is not None: module.weight.data.fill_(1.0) elif isinstance(module, nn.Conv1d): nn.init.kaiming_normal_(module.weight) - if module.bias is not None: k = math.sqrt(module.groups / (module.in_channels * module.kernel_size[0])) nn.init.uniform_(module.bias, a=-k, b=k) + if hasattr(module, "objective"): + nn.init.normal_(module.objective.weight) + if hasattr(module, "masked_spec_embed"): + nn.init.uniform_(module.masked_spec_embed) + def _get_adapters(self): raise AttributeError("Not needed for Data2VecAudio")