ó
    qyüi  ã                   óò   • S r SSKJr  SSKJr  SSKJrJr  \R                  " \	5      r
\" SS9\ " S S	\5      5       5       r\" SS9\ " S
 S\5      5       5       r\" SS9\ " S S\5      5       5       r/ SQrg)zMllama model configurationé    )Ústricté   )ÚPreTrainedConfig)Úauto_docstringÚloggingzmeta-llama/Llama-3.2-11B-Vision)Ú
checkpointc                   ó„  ^ • \ rS rSr% SrSrSrSS0rSr\	\
S'   S	r\\
S
'   Sr\	\
S'   Sr\	\
S'   Sr\	\
S'   Sr\	\
S'   Sr\	\
S'   Sr\	\
S'   Sr\	\\	   -  \\	\	4   -  \
S'   Sr\	\\	   -  \\	\	4   -  \
S'   Sr\\
S'   Sr\	\
S'   Sr\\	   S-  \
S'   Sr\\\	      S-  \
S '   S!r\\
S"'   U 4S# jrS$ r\S%\	4S& j5       r S'r!U =r"$ )(ÚMllamaVisionConfigé   aV  
num_global_layers (`int`, *optional*, defaults to 8):
    Number of global layers in the Transformer encoder. Vision model has a second transformer encoder, called global.
vision_output_dim (`int`, *optional*, defaults to 7680):
    Dimensionality of the vision model output. Includes output of transformer
    encoder with intermediate layers and global transformer encoder.
max_num_tiles (`int`, *optional*, defaults to 4):
    Maximum number of tiles for image splitting.
intermediate_layers_indices (`list[int]`, *optional*, defaults to [3, 7, 15, 23, 30]):
    Indices of intermediate layers of transformer encoder from which to extract and output features.
    These output features are concatenated with final hidden state of transformer encoder.
supported_aspect_ratios (`list[list[int]]`, *optional*):
    List of supported aspect ratios for image splitting. If not specified, the default supported aspect ratios
    are [[1, 1], [1, 2], [1, 3], [1, 4], [2, 1], [2, 2], [3, 1], [4, 1]] for `max_num_tiles=4`.

Example:

```python
>>> from transformers import MllamaVisionConfig, MllamaVisionModel

>>> # Initializing a Llama config
>>> config = MllamaVisionConfig()

>>> # Initializing a vision model from the mllama-11b style configuration
>>> model = MllamaVisionModel(config)

>>> # Accessing the model configuration
>>> configuration = model.config
```Úmllama_vision_modelÚvision_configÚnum_attention_headsÚattention_headsi   Úhidden_sizeÚgeluÚ
hidden_acté    Únum_hidden_layersé   Únum_global_layersé   r   Únum_channelsi   Úintermediate_sizei   Úvision_output_dimiÀ  Ú
image_sizeé   Ú
patch_sizeçñhãˆµøä>Únorm_epsé   Úmax_num_tilesNÚintermediate_layers_indicesÚsupported_aspect_ratiosç{®Gáz”?Úinitializer_rangec           	      óª   >• U R                   c  SS/SS/SS/SS/SS/SS/SS/SS//U l         U R                  c	  / SQU l        [        TU ]  " S0 UD6  g )Né   é   r   r    )r   é   é   é   é   © )r#   r"   ÚsuperÚ__post_init__©ÚselfÚkwargsÚ	__class__s     €Úl/home/mande/repo/quber/.venv/lib/python3.13/site-packages/transformers/models/mllama/configuration_mllama.pyr/   Ú MllamaVisionConfig.__post_init__M   sv   ø€ Ø×'Ñ'Ñ/Ø-.°¨F°Q¸°F¸QÀ¸FÀQÈÀFÈQÐPQÈFÐUVÐXYÐTZÐ]^Ð`aÐ\bÐefÐhiÐdjÐ+kˆDÔ(à×+Ñ+Ñ3Ú/AˆDÔ,Ü‰ÒÑ' Ó'ó    c           
      óŒ   • U R                   SS/SS/SS/SS/SS/SS/SS/SS//:X  a  U R                  S:w  a  [        S5      egg)zOPart of `@strict`-powered validation. Validates the architecture of the config.r'   r(   r   r    z;max_num_tiles must be 4 for default supported aspect ratiosN)r#   r!   Ú
ValueError©r1   s    r4   Úvalidate_architectureÚ(MllamaVisionConfig.validate_architectureU   sr   € ð ×(Ñ(¨a°¨V°a¸°V¸aÀ¸VÀaÈÀVÈaÐQRÈVÐVWÐYZÐU[Ð^_ÐabÐ]cÐfgÐijÐekÐ,lÓlØ×"Ñ" aÓ'äÐZÓ[Ð[ð (ð mr6   Úreturnc                 ó,   • [        U R                  5      $ )N)Úlenr#   r9   s    r4   Úmax_aspect_ratio_idÚ&MllamaVisionConfig.max_aspect_ratio_id]   s   € ä�4×/Ñ/Ó0Ð0r6   )r"   r#   )#Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__Ú
model_typeÚbase_config_keyÚattribute_mapr   ÚintÚ__annotations__r   Ústrr   r   r   r   r   r   r   ÚlistÚtupler   r   Úfloatr!   r"   r#   r%   r/   r:   Úpropertyr?   Ú__static_attributes__Ú__classcell__©r3   s   @r4   r
   r
      s"  ø‡ ñð< '€JØ%€OØ*Ð,=Ð>€Mà€K�ÓØ€J�ÓØÐ�sÓØÐ�sÓØ€O�SÓØ€L�#ÓØ!Ð�sÓ!Ø!Ð�sÓ!Ø47€J��d˜3‘i‘ %¨¨S¨¡/Ñ1Ó7Ø46€J��d˜3‘i‘ %¨¨S¨¡/Ñ1Ó6Ø€HˆeÓØ€M�3ÓØ48Ð  c¡¨TÑ!1Ó8Ø6:Ð˜T $ s¡)™_¨tÑ3Ó:Ø#Ð�uÓ#õ(ò\ð ð1 Só 1ó ö1r6   r
   c                   óf  ^ • \ rS rSr% SrSrSrSrSr\	\
S'   Sr\	\
S	'   S
r\\
S'   Sr\	\
S'   Sr\	\
S'   Sr\	\
S'   Sr\	\
S'   Sr\S-  \
S'   Sr\\
S'   Sr\	\
S'   Sr\\
S'   Sr\\
S'   Sr\\
S'   Sr\\	   S-  \
S '   S!r\\	-  \
S"'   S#r\	\
S$'   S%r\	\\	   -  S-  \
S&'   S'r \	S-  \
S('   U 4S) jr!S*r"U =r#$ )+ÚMllamaTextConfigéb   aí  
cross_attention_layers (`list[int]`, *optional*):
    Indices of the cross attention layers. If not specified, will default to [3, 8, 13, 18, 23, 28, 33, 38].

Example:

```python
>>> from transformers import MllamaTextModel, MllamaTextConfig

>>> # Initializing a Mllama text config
>>> config = MllamaTextConfig()

>>> # Initializing a model from the Mllama text configuration
>>> model = MllamaTextModel(config)

>>> # Accessing the model configuration
>>> configuration = model.config
```Úmllama_text_modelÚtext_configg    €„Aé õ Ú
vocab_sizei   r   Úsilur   é(   r   r   r   r   Únum_key_value_headsi 8  r   NÚrope_parametersr   Úrms_norm_epsi   Úmax_position_embeddingsr$   r%   TÚ	use_cacheFÚtie_word_embeddingsÚcross_attention_layersg        Údropouti ô Úbos_token_idiô Úeos_token_idiô Úpad_token_idc                 óR   >• U R                   c	  / SQU l         [        TU ]  " S0 UD6  g )N)r   r   é   é   r+   é   é!   é&   r-   )rb   r.   r/   r0   s     €r4   r/   ÚMllamaTextConfig.__post_init__�   s'   ø€ Ø×&Ñ&Ñ.Ú*HˆDÔ'Ü‰ÒÑ' Ó'r6   )rb   )$rA   rB   rC   rD   rE   rF   rG   Údefault_thetarY   rI   rJ   r   r   rK   r   r   r\   r   r]   Údictr^   rN   r_   r%   r`   Úboolra   rb   rL   rc   rd   re   rf   r/   rP   rQ   rR   s   @r4   rT   rT   b   s  ø‡ ñð& %€JØ#€OØ€Mà€J�ÓØ€K�ÓØ€J�ÓØÐ�sÓØ!Ð˜Ó!Ø Ð˜Ó Ø#Ð�sÓ#Ø#'€O�T˜D‘[Ó'Ø€L�%ÓØ#*Ð˜SÓ*Ø#Ð�uÓ#Ø€IˆtÓØ %Ð˜Ó%Ø/3Ð˜D ™I¨Ñ,Ó3Ø€GˆU�S‰[ÓØ€L�#ÓØ+1€L�#˜˜S™	‘/ DÑ(Ó1Ø%€L�#˜‘*Ó%÷(ó (r6   rT   c                   ó†   ^ • \ rS rSr% SrSrSS0r\\S.r	Sr
\\-  S-  \S'   Sr\\-  S-  \S	'   S
r\\S'   U 4S jrSrU =r$ )ÚMllamaConfigé•   a\  
Example:

```python
>>> from transformers import MllamaForConditionalGeneration, MllamaConfig, MllamaVisionConfig, MllamaTextConfig

>>> # Initializing a CLIP-vision config
>>> vision_config = MllamaVisionConfig()

>>> # Initializing a Llama config
>>> text_config = MllamaTextConfig()

>>> # Initializing a mllama-11b style configuration
>>> configuration = MllamaConfig(vision_config, text_config)

>>> # Initializing a model from the mllama-11b style configuration
>>> model = MllamaForConditionalGeneration(configuration)

>>> # Accessing the model configuration
>>> configuration = model.config
```ÚmllamaÚimage_token_idÚimage_token_index)rW   r   Nr   rW   rX   c                 óÒ  >• U R                   c%  [        5       U l         [        R                  S5        O9[	        U R                   [
        5      (       a  [        S0 U R                   D6U l         U R                  c%  [        5       U l        [        R                  S5        O9[	        U R                  [
        5      (       a  [        S0 U R                  D6U l        [        TU ]$  " S0 UD6  g )Nz9vision_config is None, using default mllama vision configz5text_config is None, using default mllama text configr-   )
r   r
   ÚloggerÚinfoÚ
isinstancero   rW   rT   r.   r/   r0   s     €r4   r/   ÚMllamaConfig.__post_init__¸   s­   ø€ Ø×ÑÑ%Ü!3Ó!5ˆDÔÜ�K‰KÐSÕTÜ˜×*Ñ*¬D×1Ñ1Ü!3Ñ!I°d×6HÑ6HÑ!IˆDÔà×ÑÑ#Ü/Ó1ˆDÔÜ�K‰KÐOÕPÜ˜×(Ñ(¬$×/Ñ/Ü/ÑC°$×2BÑ2BÑCˆDÔä‰ÒÑ' Ó'r6   )rA   rB   rC   rD   rE   rF   rH   rT   r
   Úsub_configsr   ro   r   rJ   rW   rv   rI   r/   rP   rQ   rR   s   @r4   rr   rr   •   sh   ø‡ ñð, €JàÐ-ð€Mð #3ÐEWÑX€Kà48€M�4Ð*Ñ*¨TÑ1Ó8Ø26€K�Ð(Ñ(¨4Ñ/Ó6Ø#Ð�sÓ#÷(ó (r6   rr   )rr   rT   r
   N)rE   Úhuggingface_hub.dataclassesr   Úconfiguration_utilsr   Úutilsr   r   Ú
get_loggerrA   rx   r
   rT   rr   Ú__all__r-   r6   r4   Ú<module>r‚      s±   ðñ !å .å 3ß ,ð 
×	Ò	˜HÓ	%€ñ Ð<Ñ=ØôE1Ð)ó E1ó ó >ðE1ñP Ð<Ñ=Øô.(Ð'ó .(ó ó >ð.(ñb Ð<Ñ=Øô.(Ð#ó .(ó ó >ð.(òb E�r6   