#                🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨
#           This file was automatically generated from src/transformers/models/cohere_compass/modular_cohere_compass.py.
#               Do NOT edit this file manually as any edits will be overwritten by the generation of
#             the file from the modular. If any change should be done, please apply the change to the
#                          modular_cohere_compass.py file directly. One of our CI enforces this.
#                🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨
# Copyright 2026 Cohere Inc. and the HuggingFace Inc. team. All rights reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
#     http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
from huggingface_hub.dataclasses import strict

from ...configuration_utils import PreTrainedConfig
from ...modeling_rope_utils import RopeParameters
from ...utils import auto_docstring


@auto_docstring(checkpoint="CohereLabs/North-Micro-Vision-Instruct")
@strict
class CohereCompassVisionConfig(PreTrainedConfig):
    r"""
    out_hidden_size (`int`, *optional*, defaults to 3584):
        The output hidden size of the vision model.
    num_position_embeddings (`int`, *optional*, defaults to 2304):
        The maximum sequence length that this model might ever be used with
    deepstack_visual_indexes (`list[int]`, *optional*, defaults to `[8, 16, 24]`):
        Indexed of layers for deepstack embeddings.
    """

    model_type = "cohere_compass_vision"
    base_config_key = "vision_config"

    depth: int = 27
    hidden_size: int = 1152
    hidden_act: str = "gelu_pytorch_tanh"
    intermediate_size: int = 4304
    num_heads: int = 16
    in_channels: int = 3
    patch_size: int | list[int] | tuple[int, int] = 16
    spatial_merge_size: int = 2
    temporal_patch_size: int | list[int] | tuple[int, int] = 2
    out_hidden_size: int = 3584
    num_position_embeddings: int = 2304
    deepstack_visual_indexes: list[int] | tuple[int, ...] = (8, 16, 24)
    initializer_range: float = 0.02


@auto_docstring(checkpoint="CohereLabs/North-Micro-Vision-Instruct")
@strict
class CohereCompassTextConfig(PreTrainedConfig):
    r"""
    logit_scale (`float`, *optional*):
        Scale applied to language-model logits.
    pooling (`str`, *optional*):
        The pooling strategy (`bos` | `eos` | `mean`); `None` defaults to `eos`.
    """

    model_type = "cohere_compass_text"
    keys_to_ignore_at_inference = ["past_key_values"]
    base_model_tp_plan = {
        "layers.*.self_attn.q_proj": "colwise",
        "layers.*.self_attn.k_proj": "colwise",
        "layers.*.self_attn.v_proj": "colwise",
        "layers.*.self_attn.o_proj": "rowwise",
        "layers.*.mlp.gate_proj": "colwise",
        "layers.*.mlp.up_proj": "colwise",
        "layers.*.mlp.down_proj": "rowwise",
    }
    base_model_pp_plan = {
        "embed_tokens": (["input_ids"], ["inputs_embeds"]),
        "layers": (["hidden_states", "attention_mask"], ["hidden_states"]),
        "norm": (["hidden_states"], ["hidden_states"]),
    }

    vocab_size: int = 256000
    hidden_size: int = 8192
    intermediate_size: int = 22528
    logit_scale: float | None = None
    num_hidden_layers: int = 40
    num_attention_heads: int = 64
    num_key_value_heads: int | None = None
    hidden_act: str = "silu"
    max_position_embeddings: int = 8192
    initializer_range: float = 0.02
    layer_norm_eps: float = 1e-5
    use_cache: bool = True
    pad_token_id: int | None = 0
    bos_token_id: int | None = 5
    eos_token_id: int | list[int] | None = 255001
    tie_word_embeddings: bool = True

    rope_parameters: dict[str, RopeParameters | dict | float | str | int | None] | None = None
    attention_bias: bool = False
    attention_dropout: float | int = 0.0
    sliding_window: int | None = 4096
    layer_types: list[str] | None = None
    base_config_key = "text_config"
    ignore_keys_at_rope_validation = {"mrope_section", "mrope_interleaved"}
    pooling: str | None = None

    def __post_init__(self, **kwargs):
        if self.layer_types is None:
            self.layer_types = ["full_attention"] * self.num_hidden_layers
        if self.num_key_value_heads is None:
            self.num_key_value_heads = self.num_attention_heads

        # Need to specify head_dim in the config so it can be used in the attention forward functions
        self.head_dim = self.hidden_size // self.num_attention_heads

        # BC -> the pattern used to be a simple int, and it's still present in configs on the Hub
        if self.layer_types is None:
            # BC -> the pattern used to be a simple int, and it's still present in configs on the Hub
            _sliding_window_pattern = kwargs.pop("sliding_window_pattern", 4)
            self.layer_types = [
                "sliding_attention" if bool((i + 1) % _sliding_window_pattern) else "full_attention"
                for i in range(self.num_hidden_layers)
            ]

        super().__post_init__(**kwargs)

    def convert_rope_params_to_dict(self, **kwargs):
        # allow per layer rope with optional NoPE layers
        self.rope_parameters = self.rope_parameters if self.rope_parameters is not None else {}
        self.standardize_rope_params()
        return kwargs


@auto_docstring(checkpoint="CohereLabs/North-Micro-Vision-Instruct")
@strict
class CohereCompassConfig(PreTrainedConfig):
    r"""
    Example:

    ```python
    >>> from transformers import CohereCompassForConditionalGeneration, CohereCompassConfig

    >>> # Initializing a "CohereLabs/North-Micro-Vision-Instruct" style configuration
    >>> configuration = CohereCompassConfig()

    >>> # Initializing a model from the "CohereLabs/North-Micro-Vision-Instruct" style configuration
    >>> model = CohereCompassForConditionalGeneration(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```"""

    model_type = "cohere_compass"
    sub_configs = {
        "text_config": CohereCompassTextConfig,
        "vision_config": CohereCompassVisionConfig,
    }
    keys_to_ignore_at_inference = ["past_key_values"]

    text_config: dict | PreTrainedConfig | None = None
    vision_config: dict | PreTrainedConfig | None = None

    image_token_id: int = 255031
    video_token_id: int = 255032
    vision_start_token_id: int = 255028
    vision_end_token_id: int = 255029
    tie_word_embeddings: bool = False

    def __post_init__(self, **kwargs):
        if isinstance(self.vision_config, dict):
            # old ckpt with incorrect model type -> override manually
            if self.vision_config.get("model_type") == "cohere_compass":
                self.vision_config["model_type"] = "cohere_compass_vision"
            self.vision_config = self.sub_configs["vision_config"](**self.vision_config)
        elif self.vision_config is None:
            self.vision_config = self.sub_configs["vision_config"]()

        if isinstance(self.text_config, dict):
            self.text_config = self.sub_configs["text_config"](**self.text_config)
        elif self.text_config is None:
            self.text_config = self.sub_configs["text_config"]()

        super().__post_init__(**kwargs)


__all__ = ["CohereCompassConfig", "CohereCompassTextConfig", "CohereCompassVisionConfig"]
