huggingface
diff --git a/‎docs/source/en/_toctree.yml‎
Lines changed: 2 additions & 0 deletions b/‎docs/source/en/_toctree.yml‎
Lines changed: 2 additions & 0 deletions
diff --git a/‎docs/source/en/model_doc/jais2.md‎
Lines changed: 55 additions & 0 deletions b/‎docs/source/en/model_doc/jais2.md‎
Lines changed: 55 additions & 0 deletions
diff --git a/‎src/transformers/models/auto/configuration_auto.py‎
Lines changed: 2 additions & 0 deletions b/‎src/transformers/models/auto/configuration_auto.py‎
Lines changed: 2 additions & 0 deletions
diff --git a/‎src/transformers/models/auto/modeling_auto.py‎
Lines changed: 5 additions & 0 deletions b/‎src/transformers/models/auto/modeling_auto.py‎
Lines changed: 5 additions & 0 deletions
diff --git a/‎src/transformers/models/jais2/__init__.py‎
Lines changed: 47 additions & 0 deletions b/‎src/transformers/models/jais2/__init__.py‎
Lines changed: 47 additions & 0 deletions
diff --git a/‎src/transformers/models/jais2/configuration_jais2.py‎
Lines changed: 169 additions & 0 deletions b/‎src/transformers/models/jais2/configuration_jais2.py‎
Lines changed: 169 additions & 0 deletions
@@ -547,6 +547,8 @@
         title: HunYuanMoEV1
       - local: model_doc/ibert
         title: I-BERT
+      - local: model_doc/jais2
+        title: Jais2
       - local: model_doc/jamba
         title: Jamba
       - local: model_doc/jetmoe
 
@@ -0,0 +1,55 @@
+<!--Copyright 2024 The HuggingFace Team. All rights reserved.
+
+Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file except in compliance with
+the License. You may obtain a copy of the License at
+
+http://www.apache.org/licenses/LICENSE-2.0
+
+Unless required by applicable law or agreed to in writing, software distributed under the License is distributed on
+an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the License for the
+specific language governing permissions and limitations under the License.
+
+⚠️ Note that this file is in Markdown but contain specific syntax for our doc-builder (similar to MDX) that may not be
+rendered properly in your Markdown viewer.
+
+-->
+*This model was released on {release_date} and added to Hugging Face Transformers on 2025-12-09.*
+
+# Jais2
+
+## Overview
+
+Jais2 is a large language model developed by MBZUAI, Inception and Cerebras Systems. It is based on the transformer architecture with several modifications including:
+
+- LayerNorm instead of RMSNorm
+- ReLU² activation function
+- Rotary Position Embeddings (RoPE)
+
+## Jais2Config
+
+[[autodoc]] Jais2Config
+
+## Jais2Model
+
+[[autodoc]] Jais2Model
+    - forward
+
+## Jais2ForCausalLM
+
+[[autodoc]] Jais2ForCausalLM
+    - forward
+
+## Jais2ForSequenceClassification
+
+[[autodoc]] Jais2ForSequenceClassification
+    - forward
+
+## Jais2ForTokenClassification
+
+[[autodoc]] Jais2ForTokenClassification
+    - forward
+
+## Jais2ForQuestionAnswering
+
+[[autodoc]] Jais2ForQuestionAnswering
+    - forward
@@ -215,6 +215,7 @@
         ("instructblipvideo", "InstructBlipVideoConfig"),
         ("internvl", "InternVLConfig"),
         ("internvl_vision", "InternVLVisionConfig"),
+        ("jais2", "Jais2Config"),
         ("jamba", "JambaConfig"),
         ("janus", "JanusConfig"),
         ("jetmoe", "JetMoeConfig"),
@@ -658,6 +659,7 @@
         ("instructblipvideo", "InstructBlipVideo"),
         ("internvl", "InternVL"),
         ("internvl_vision", "InternVLVision"),
+        ("jais2", "Jais2"),
         ("jamba", "Jamba"),
         ("janus", "Janus"),
         ("jetmoe", "JetMoe"),
 
@@ -216,6 +216,7 @@ class _BaseModelWithGenerate(PreTrainedModel, GenerationMixin):
         ("instructblipvideo", "InstructBlipVideoModel"),
         ("internvl", "InternVLModel"),
         ("internvl_vision", "InternVLVisionModel"),
+        ("jais2", "Jais2Model"),
         ("jamba", "JambaModel"),
         ("janus", "JanusModel"),
         ("jetmoe", "JetMoeModel"),
@@ -689,6 +690,7 @@ class _BaseModelWithGenerate(PreTrainedModel, GenerationMixin):
         ("helium", "HeliumForCausalLM"),
         ("hunyuan_v1_dense", "HunYuanDenseV1ForCausalLM"),
         ("hunyuan_v1_moe", "HunYuanMoEV1ForCausalLM"),
+        ("jais2", "Jais2ForCausalLM"),
         ("jamba", "JambaForCausalLM"),
         ("jetmoe", "JetMoeForCausalLM"),
         ("lfm2", "Lfm2ForCausalLM"),
@@ -1245,6 +1247,7 @@ class _BaseModelWithGenerate(PreTrainedModel, GenerationMixin):
         ("hunyuan_v1_dense", "HunYuanDenseV1ForSequenceClassification"),
         ("hunyuan_v1_moe", "HunYuanMoEV1ForSequenceClassification"),
         ("ibert", "IBertForSequenceClassification"),
+        ("jais2", "Jais2ForSequenceClassification"),
         ("jamba", "JambaForSequenceClassification"),
         ("jetmoe", "JetMoeForSequenceClassification"),
         ("layoutlm", "LayoutLMForSequenceClassification"),
@@ -1343,6 +1346,7 @@ class _BaseModelWithGenerate(PreTrainedModel, GenerationMixin):
         ("gpt_neox", "GPTNeoXForQuestionAnswering"),
         ("gptj", "GPTJForQuestionAnswering"),
         ("ibert", "IBertForQuestionAnswering"),
+        ("jais2", "Jais2ForQuestionAnswering"),
         ("layoutlmv2", "LayoutLMv2ForQuestionAnswering"),
         ("layoutlmv3", "LayoutLMv3ForQuestionAnswering"),
         ("led", "LEDForQuestionAnswering"),
@@ -1458,6 +1462,7 @@ class _BaseModelWithGenerate(PreTrainedModel, GenerationMixin):
         ("gpt_oss", "GptOssForTokenClassification"),
         ("helium", "HeliumForTokenClassification"),
         ("ibert", "IBertForTokenClassification"),
+        ("jais2", "Jais2ForTokenClassification"),
         ("layoutlm", "LayoutLMForTokenClassification"),
         ("layoutlmv2", "LayoutLMv2ForTokenClassification"),
         ("layoutlmv3", "LayoutLMv3ForTokenClassification"),
 
@@ -0,0 +1,47 @@
+from typing import TYPE_CHECKING
+
+from ...utils import OptionalDependencyNotAvailable, _LazyModule, is_torch_available
+
+
+_import_structure = {
+    "configuration_jais2": ["Jais2Config"],
+}
+
+try:
+    if not is_torch_available():
+        raise OptionalDependencyNotAvailable()
+except OptionalDependencyNotAvailable:
+    pass
+else:
+    _import_structure["modeling_jais2"] = [
+        "Jais2ForCausalLM",
+        "Jais2ForQuestionAnswering",
+        "Jais2ForSequenceClassification",
+        "Jais2ForTokenClassification",
+        "Jais2Model",
+        "Jais2PreTrainedModel",
+    ]
+
+
+if TYPE_CHECKING:
+    from .configuration_jais2 import Jais2Config
+
+    try:
+        if not is_torch_available():
+            raise OptionalDependencyNotAvailable()
+    except OptionalDependencyNotAvailable:
+        pass
+    else:
+        from .modeling_jais2 import (
+            Jais2ForCausalLM,
+            Jais2ForQuestionAnswering,
+            Jais2ForSequenceClassification,
+            Jais2ForTokenClassification,
+            Jais2Model,
+            Jais2PreTrainedModel,
+        )
+
+else:
+    import sys
+
+    sys.modules[__name__] = _LazyModule(__name__, globals()["__file__"], _import_structure, module_spec=__spec__)
@@ -0,0 +1,169 @@
+#                🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨
+#           This file was automatically generated from src/transformers/models/jais2/modular_jais2.py.
+#               Do NOT edit this file manually as any edits will be overwritten by the generation of
+#             the file from the modular. If any change should be done, please apply the change to the
+#                          modular_jais2.py file directly. One of our CI enforces this.
+#                🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨
+# coding=utf-8
+# Copyright 2025 the HuggingFace Team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from typing import Optional
+
+from ...configuration_utils import PreTrainedConfig
+from ...modeling_rope_utils import RopeParameters
+
+
+class Jais2Config(PreTrainedConfig):
+    r"""
+    This is the configuration class to store the configuration of a [`Jais2Model`].
+    It inherits from [`LlamaConfig`] and can be used to control the model outputs.
+
+    Read more from the [inceptionai/Jais-2-8B-Chat](https://huggingface.co/inceptionai/Jais-2-8B-Chat).
+
+    Args:
+        vocab_size (`int`, *optional*, defaults to 150272):
+            Vocabulary size of the Jais2 model.
+        hidden_size (`int`, *optional*, defaults to 3328):
+            Dimension of the hidden representations.
+        intermediate_size (`int`, *optional*, defaults to 26624):
+            Dimension of the MLP representations.
+        num_hidden_layers (`int`, *optional*, defaults to 32):
+            Number of hidden layers in the Transformer decoder.
+        num_attention_heads (`int`, *optional*, defaults to 26):
+            Number of attention heads for each attention layer.
+        num_key_value_heads (`int`, *optional*):
+            Number of key_value heads for Grouped Query Attention.
+        hidden_act (`str`, *optional*, defaults to `"relu2"`):
+            The non-linear activation function in the decoder.
+        max_position_embeddings (`int`, *optional*, defaults to 8192):
+            The maximum sequence length.
+        initializer_range (`float`, *optional*, defaults to 0.02):
+            The standard deviation of the truncated_normal_initializer.
+        layer_norm_eps (`float`, *optional*, defaults to 1e-05):
+            The epsilon used by the normalization layers.
+        use_cache (`bool`, *optional*, defaults to `True`):
+            Whether to return last key/values attentions.
+        pad_token_id (`int`, *optional*):
+            Padding token id.
+        bos_token_id (`int`, *optional*, defaults to 0):
+            Beginning of stream token id.
+        eos_token_id (`int`, *optional*, defaults to 150024):
+            End of stream token id.
+        pretraining_tp (`int`, *optional*, defaults to 1):
+            Tensor parallelism rank used during pretraining.
+        tie_word_embeddings (`bool`, *optional*, defaults to `False`):
+            Whether to tie weight embeddings.
+        attention_bias (`bool`, *optional*, defaults to `True`):
+            Whether to use a bias in the query, key, value and output projection layers.
+        attention_dropout (`float`, *optional*, defaults to 0.0):
+            The dropout ratio for the attention probabilities.
+        mlp_bias (`bool`, *optional*, defaults to `True`):
+            Whether to use a bias in up_proj, down_proj and gate_proj layers.
+        head_dim (`int`, *optional*):
+            The attention head dimension.
+        rope_theta (`float`, *optional*, defaults to 500000.0):
+            The base period of the RoPE embeddings.
+        rope_parameters (`dict`, *optional*):
+            The RoPE parameters.
+    """
+
+    model_type = "jais2"
+    keys_to_ignore_at_inference = ["past_key_values"]
+
+    base_model_tp_plan = {
+        "layers.*.self_attn.q_proj": "colwise",
+        "layers.*.self_attn.k_proj": "colwise",
+        "layers.*.self_attn.v_proj": "colwise",
+        "layers.*.self_attn.o_proj": "rowwise",
+        "layers.*.mlp.up_proj": "colwise",
+        "layers.*.mlp.down_proj": "rowwise",
+    }
+    base_model_pp_plan = {
+        "embed_tokens": (["input_ids"], ["inputs_embeds"]),
+        "layers": (["hidden_states", "attention_mask"], ["hidden_states"]),
+        "norm": (["hidden_states"], ["hidden_states"]),
+    }
+
+    def __init__(
+        self,
+        vocab_size: Optional[int] = 150272,
+        hidden_size: Optional[int] = 3328,
+        intermediate_size: Optional[int] = 26624,
+        num_hidden_layers: Optional[int] = 32,
+        num_attention_heads: Optional[int] = 26,
+        num_key_value_heads: Optional[int] = None,
+        hidden_act: Optional[str] = "relu2",
+        max_position_embeddings: Optional[int] = 8192,
+        initializer_range: Optional[float] = 0.02,
+        layer_norm_eps: Optional[float] = 1e-5,
+        use_cache: Optional[bool] = True,
+        pad_token_id: Optional[int] = None,
+        bos_token_id: Optional[int] = 0,
+        eos_token_id: Optional[int] = 150024,
+        pretraining_tp: Optional[int] = 1,
+        tie_word_embeddings: Optional[bool] = False,
+        attention_bias: Optional[bool] = True,
+        attention_dropout: Optional[float] = 0.0,
+        mlp_bias: Optional[bool] = True,
+        head_dim: Optional[int] = None,
+        rope_theta: Optional[float] = 500000.0,
+        rope_parameters: Optional[RopeParameters | dict[str, RopeParameters]] = None,
+        **kwargs,
+    ):
+        # If rope_parameters not provided, create default with rope_theta
+        if rope_parameters is None:
+            rope_parameters = RopeParameters(rope_theta=rope_theta)
+
+        # Define rms_norm_eps for the parent init to use
+        rms_norm_eps = layer_norm_eps
+        self.vocab_size = vocab_size
+        self.max_position_embeddings = max_position_embeddings
+        self.hidden_size = hidden_size
+        self.intermediate_size = intermediate_size
+        self.num_hidden_layers = num_hidden_layers
+        self.num_attention_heads = num_attention_heads
+
+        # for backward compatibility
+        if num_key_value_heads is None:
+            num_key_value_heads = num_attention_heads
+
+        self.num_key_value_heads = num_key_value_heads
+        self.hidden_act = hidden_act
+        self.initializer_range = initializer_range
+        self.rms_norm_eps = rms_norm_eps
+        self.pretraining_tp = pretraining_tp
+        self.use_cache = use_cache
+        self.attention_bias = attention_bias
+        self.attention_dropout = attention_dropout
+        self.mlp_bias = mlp_bias
+        self.head_dim = head_dim if head_dim is not None else self.hidden_size // self.num_attention_heads
+        self.rope_parameters = rope_parameters
+
+        super().__init__(
+            pad_token_id=pad_token_id,
+            bos_token_id=bos_token_id,
+            eos_token_id=eos_token_id,
+            tie_word_embeddings=tie_word_embeddings,
+            **kwargs,
+        )
+        # Rename the attribute from rms_norm_eps to layer_norm_eps
+        self.layer_norm_eps = self.rms_norm_eps
+
+        # Validate and standardize RoPE parameters
+        self.standardize_rope_params()
+        self.validate_rope()
+
+
+__all__ = ["Jais2Config"]