ó
    qyüi>+  ã                   óú   • S SK Jr  SSKJr  SSKJr  SSKJrJr  \R                  " \
5      r\" SS9\ " S S	\5      5       5       r\" SS9\ " S
 S\5      5       5       r\" SS9\ " S S\5      5       5       r/ SQrg)é    )Ústricté   )ÚPreTrainedConfig)ÚRopeParameters)Úauto_docstringÚloggingz meta-llama/Llama-4-Scout-17B-16E)Ú
checkpointc                   ó�  • \ rS rSr% SrSSSSSSSS.rSrSrS	r\	\
S
'   Sr\\
S'   Sr\	\
S'   Sr\	\
S'   Sr\	\
S'   Sr\	\
S'   Sr\	\
S'   Sr\	\\	   -  \\	\	4   -  \
S'   Sr\	\\	   -  \\	\	4   -  \
S'   Sr\\
S'   Sr\\
S'   Sr\\
S '   S!r\\
S"'   S#r\	\
S$'   S#r\	\
S%'   S&r\\
S''   S(r\\	-  \
S)'   S(r \\	-  \
S*'   S+r!\"\#-  S+-  \
S,'   S-r$g+).ÚLlama4VisionConfigé   aG  
vision_output_dim (`int`, *optional*, defaults to 7680):
    Dimensionality of the vision model output. Includes output of transformer
    encoder with intermediate layers and global transformer encoder.
pixel_shuffle_ratio (`float`, *optional*, defaults to 0.5):
    Pixel-shuffle ratio for downsampling patch tokens. Smaller values produce fewer tokens (more downsampling).
projector_input_dim (`int`, *optional*, defaults to 4096):
    Width of the vision adapter MLP before pixel shuffle. Larger value increases capacity and compute.
projector_output_dim (`int`, *optional*, defaults to 4096):
    Output width of the vision adapter. Larger value yields higher-dimensional image features.
projector_dropout (`float`, *optional*, defaults to 0.0):
    Dropout rate inside the vision adapter MLP. Higher value adds more regularization.
ÚcolwiseÚrowwiseÚcolwise_gather_output)zmodel.layers.*.self_attn.q_projzmodel.layers.*.self_attn.k_projzmodel.layers.*.self_attn.v_projzmodel.layers.*.self_attn.o_projzvision_adapter.mlp.fc1zvision_adapter.mlp.fc2zpatch_embedding.linearÚllama4_vision_modelÚvision_configi   Úhidden_sizeÚgeluÚ
hidden_acté"   Únum_hidden_layersé   Únum_attention_headsr   Únum_channelsi   Úintermediate_sizei   Úvision_output_dimiÀ  Ú
image_sizeé   Ú
patch_sizeçñhãˆµøä>Únorm_epsÚdefaultÚvision_feature_select_strategyç{®Gáz”?Úinitializer_rangeg      à?Úpixel_shuffle_ratioi   Úprojector_input_dimÚprojector_output_dimFÚmulti_modal_projector_biasç        Úprojector_dropoutÚattention_dropoutNÚrope_parameters© )%Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__Úbase_model_tp_planÚ
model_typeÚbase_config_keyr   ÚintÚ__annotations__r   Ústrr   r   r   r   r   r   ÚlistÚtupler   r    Úfloatr"   r$   r%   r&   r'   r(   Úboolr*   r+   r,   r   ÚdictÚ__static_attributes__r-   ó    Úl/home/mande/repo/quber/.venv/lib/python3.13/site-packages/transformers/models/llama4/configuration_llama4.pyr   r      s8  ‡ ñð ,5Ø+4Ø+4Ø+4Ø"+Ø"+Ø"9ñÐð '€JØ%€Oà€K�ÓØ€J�ÓØÐ�sÓØ!Ð˜Ó!Ø€L�#ÓØ!Ð�sÓ!Ø!Ð�sÓ!Ø47€J��d˜3‘i‘ %¨¨S¨¡/Ñ1Ó7Ø46€J��d˜3‘i‘ %¨¨S¨¡/Ñ1Ó6Ø€HˆeÓØ*3Ð" CÓ3Ø#Ð�uÓ#Ø!$Ð˜Ó$Ø#Ð˜Ó#Ø $Ð˜#Ó$Ø',Ð Ó,Ø%(Ð�u˜s‘{Ó(Ø%(Ð�u˜s‘{Ó(Ø48€O�^ dÑ*¨TÑ1Ö8r?   r   c                   ó¸  ^ • \ rS rSr% SrSrS/rSrSSSSSSSSSSSSS	.rSSSSS
S
SSSSS.
r	Sr
\\S'   Sr\\S'   Sr\\S'   Sr\\S'   Sr\\S'   Sr\\S'   Sr\\S'   Sr\\S'   Sr\\S'   Sr\\S '   S!r\\S"'   S#r\\S$'   S%r\\S&'   S'r\S'-  \S('   S)r\S'-  \S*'   S+r\\\   -  S'-  \S,'   S-r \\S.'   S/r!\\-  \S0'   S)r"\\S1'   S2r#\\S3'   S'r$\\   S'-  \S4'   S)r%\\S5'   S%r&\\S6'   S-r'\\S7'   S8r(\\S9'   S/r)\\S:'   S'r*\+\,-  S'-  \S;'   S'r-\\   S'-  \S<'   S=r.\\S>'   Sr/\S'-  \S?'   S'r0\\   S'-  \S@'   S%r1\\SA'   Sr2\\SB'   SCr3\\SD'   S-r4\\SE'   U 4SF jr5SGr6U =r7$ )HÚLlama4TextConfigéM   aç  
intermediate_size_mlp (`int`, *optional*, defaults to 16384):
    Intermediate size of dense MLP layers. Larger value increases FFN capacity and compute.
moe_layers (`list[int]`, *optional*):
    List of layer indices that use MoE. Overrides `interleave_moe_layer_step` when set.
interleave_moe_layer_step (`int`, *optional*, defaults to 1):
    Spacing between MoE layers when `moe_layers` is `None`. Larger value means fewer MoE layers.
use_qk_norm (`bool`, *optional*, defaults to `True`):
    Whether to L2-normalize queries/keys on RoPE layers. Can stabilize attention when enabled.
no_rope_layers (`list[int]`, *optional*):
    List with at least the same length as the number of layers in the model.
    A `1` at an index position indicates that the corresponding layer will use RoPE,
    while a `0` indicates that it's a NoPE layer.
no_rope_layer_interval (`int`, *optional*, defaults to 4):
    If `no_rope_layers` is `None`, it will be created using a NoPE layer every
    `no_rope_layer_interval` layers.
attention_chunk_size (`int`, *optional*, defaults to 8192):
    Chunk size for the attention computation. Smaller value enforces more local attention and lowers memory.
attn_temperature_tuning (`bool`, *optional*, defaults to `True`):
    Whether to dynamically scale the attention temperature for each query token based on sequence length.
    Recommended for long sequences (e.g., >32k tokens) to maintain stable output results.
floor_scale (`int`, *optional*, defaults to 8192):
    Base scale (in tokens) for attention temperature tuning. Larger value delays scaling to longer positions.
attn_scale (`float`, *optional*, defaults to 0.1):
    Strength of attention temperature tuning. Larger value increases scaling at long positions.

Example:
Úllama4_textÚpast_key_valuesg    €„Ar   r   Úpacked_rowwise)úlayers.*.self_attn.q_projúlayers.*.self_attn.k_projúlayers.*.self_attn.v_projúlayers.*.self_attn.o_projz-layers.*.feed_forward.shared_expert.gate_projz+layers.*.feed_forward.shared_expert.up_projz-layers.*.feed_forward.shared_expert.down_projú*layers.*.feed_forward.experts.gate_up_projú'layers.*.feed_forward.experts.down_projúlayers.*.feed_forward.gate_projúlayers.*.feed_forward.up_projúlayers.*.feed_forward.down_projÚgrouped_gemmÚ	ep_router)
rG   rH   rI   rJ   rK   rL   rM   rN   rO   zlayers.*.feed_forward.routeri@ Ú
vocab_sizei   r   i    r   i @  Úintermediate_size_mlpé0   r   é(   r   é   Únum_key_value_headsé€   Úhead_dimÚsilur   i   Úmax_position_embeddingsr#   r$   r   Úrms_norm_epsTÚ	use_cacheNÚpad_token_idé   Úbos_token_idé   Úeos_token_idFÚtie_word_embeddingsr)   r+   Únum_experts_per_tokr   Únum_local_expertsÚ
moe_layersÚinterleave_moe_layer_stepÚuse_qk_normÚoutput_router_logitsgü©ñÒMbP?Úrouter_aux_loss_coefÚrouter_jitter_noiser,   Úno_rope_layersé   Úno_rope_layer_intervalÚattention_chunk_sizeÚlayer_typesÚattn_temperature_tuningÚfloor_scalegš™™™™™¹?Ú
attn_scaleÚattention_biasc                 óÆ  >• U R                   c  U R                  U l         [        U R                  5       Vs/ s H!  n[	        US-   U R
                  -  S:g  5      PM#     nnU R                  (       a  U R                  OUU l        U R                  b  U R                  OU R                  U R                  -  U l        U R                  b  U R                  O6[        [        U R                  S-
  U R                  U R                  5      5      U l	        U R                  c*  U R                   Vs/ s H  oD(       a  SOSPM     snU l        [        TU ]8  " S0 UD6  g s  snf s  snf )Nr_   r   Úchunked_attentionÚfull_attentionr-   )rW   r   Úranger   r6   rn   rl   rY   r   rf   r9   rg   rp   ÚsuperÚ__post_init__)ÚselfÚkwargsÚ	layer_idxÚdefault_no_rope_layersÚno_ropeÚ	__class__s        €r@   rz   ÚLlama4TextConfig.__post_init__¯   sH  ø€ Ø×#Ñ#Ñ+Ø'+×'?Ñ'?ˆDÔ$ô V[Ð[_×[qÑ[qÔUró"
ÚUrÈ	ŒC�˜Q‘ $×"=Ñ"=Ñ=ÀÑBÖCÑUrð 	ð "
ð 6:×5H×5H˜d×1Ò1ÐNdˆÔØ)-¯©Ñ)B˜ŸšÈ×HXÑHXÐ\`×\tÑ\tÑHtˆŒð �‰Ñ*ð �OŠOäÜØ×2Ñ2°QÑ6Ø×*Ñ*Ø×2Ñ2óóð 	Œð ×ÑÑ#àTX×TgÒTgó ÚTgÈ¥wÑ#Ð4DÒDÑTgñ ˆDÔô 	‰ÒÑ' Ó'ùò/"
ùò& s   ·(EÄ,E)rY   rp   rf   rl   rW   )8r.   r/   r0   r1   r2   r4   Úkeys_to_ignore_at_inferenceÚdefault_thetar3   Úbase_model_ep_planrR   r6   r7   r   r   rS   r   r   rW   rY   r   r8   r[   r$   r;   r\   r]   r<   r^   r`   rb   r9   rc   r+   rd   re   rf   rg   rh   ri   rj   rk   r,   r   r=   rl   rn   ro   rp   rq   rr   rs   rt   rz   r>   Ú__classcell__©r€   s   @r@   rB   rB   M   s/  ø‡ ñð: €JØ#4Ð"5ÐØ€Mà%.Ø%.Ø%.Ø%.Ø9BØ7@Ø9BØ6FØ3<Ø+4Ø)2Ø+4ñÐð &/Ø%.Ø%.Ø%.Ø6DØ3AØ+4Ø)2Ø+4Ø(3ñÐð €J�ÓØ€K�ÓØ!Ð�sÓ!Ø!&Ð˜3Ó&ØÐ�sÓØ!Ð˜Ó!Ø Ð˜Ó Ø€HˆcÓØ€J�ÓØ#,Ð˜SÓ,Ø#Ð�uÓ#Ø€L�%ÓØ€IˆtÓØ#€L�#˜‘*Ó#Ø €L�#˜‘*Ó Ø+,€L�#˜˜S™	‘/ DÑ(Ó,Ø %Ð˜Ó%Ø%(Ð�u˜s‘{Ó(Ø Ð˜Ó ØÐ�sÓØ#'€J��S‘	˜DÑ Ó'Ø%&Ð˜sÓ&Ø€K�ÓØ!&Ð˜$Ó&Ø"'Ð˜%Ó'Ø!$Ð˜Ó$Ø48€O�^ dÑ*¨TÑ1Ó8Ø'+€N�D˜‘I Ñ$Ó+Ø"#Ð˜CÓ#Ø'+Ð˜# ™*Ó+Ø$(€K��c‘˜TÑ!Ó(Ø$(Ð˜TÓ(Ø€K�ÓØ€J�ÓØ €N�DÓ ÷(ó (r?   rB   c                   ó¼   ^ • \ rS rSr% SrSrSSSS.r\\S.r	S	S
0r
Sr\\-  S-  \S'   Sr\\-  S-  \S'   Sr\\S'   Sr\\S'   Sr\\S'   Sr\\S'   U 4S jrSrU =r$ )ÚLlama4ConfigéÍ   a<  
boi_token_index (`int`, *optional*, defaults to 200080):
    The begin-of-image token index to wrap the image prompt.
eoi_token_index (`int`, *optional*, defaults to 200081):
    The end-of-image token index to wrap the image prompt.

```python
>>> from transformers import Llama4Model, Llama4Config

>>> # Initializing a Llama4 7B style configuration
>>> configuration = Llama4Config()

>>> # Initializing a model from the Llama4 7B style configuration
>>> model = Llama4Model(configuration)

>>> # Accessing the model configuration
>>> configuration = model.config
```
Úllama4Úimage_token_indexÚboi_token_indexÚeoi_token_index)Úimage_token_idÚboi_token_idÚeoi_token_id)Útext_configr   zmulti_modal_projector.linear_1Úcolwise_repNr   r‘   i� i‘ iœ Frc   c                 óÒ  >• U R                   c%  [        5       U l         [        R                  S5        O9[	        U R                   [
        5      (       a  [        S0 U R                   D6U l         U R                  c%  [        5       U l        [        R                  S5        O9[	        U R                  [
        5      (       a  [        S0 U R                  D6U l        [        TU ]$  " S0 UD6  g )Nz9vision_config is None, using default llama4 vision configz5text_config is None, using default llama4 text configr-   )
r   r   ÚloggerÚinfoÚ
isinstancer=   r‘   rB   ry   rz   )r{   r|   r€   s     €r@   rz   ÚLlama4Config.__post_init__ö   s­   ø€ Ø×ÑÑ%Ü!3Ó!5ˆDÔÜ�K‰KÐSÕTÜ˜×*Ñ*¬D×1Ñ1Ü!3Ñ!I°d×6HÑ6HÑ!IˆDÔà×ÑÑ#Ü/Ó1ˆDÔÜ�K‰KÐOÕPÜ˜×(Ñ(¬$×/Ñ/Ü/ÑC°$×2BÑ2BÑCˆDÔÜ‰ÒÑ' Ó'r?   )r.   r/   r0   r1   r2   r4   Úattribute_maprB   r   Úsub_configsr3   r   r=   r   r7   r‘   rŒ   r6   r�   r‹   rc   r<   rz   r>   r…   r†   s   @r@   rˆ   rˆ   Í   s™   ø‡ ñð( €Jà-Ø)Ø)ñ€Mð
 #3ÐEWÑX€Kà(¨-ðÐð 59€M�4Ð*Ñ*¨TÑ1Ó8Ø26€K�Ð(Ñ(¨4Ñ/Ó6Ø!€O�SÓ!Ø!€O�SÓ!Ø#Ð�sÓ#Ø %Ð˜Ó%÷(ó (r?   rˆ   )rˆ   rB   r   N)Úhuggingface_hub.dataclassesr   Úconfiguration_utilsr   Úmodeling_rope_utilsr   Úutilsr   r   Ú
get_loggerr.   r”   r   rB   rˆ   Ú__all__r-   r?   r@   Ú<module>r       s±   ðõ" /å 3Ý 1ß ,ð 
×	Ò	˜HÓ	%€ñ Ð=Ñ>Øô-9Ð)ó -9ó ó ?ð-9ñ` Ð=Ñ>Øô{(Ð'ó {(ó ó ?ð{(ñ| Ð=Ñ>Øô3(Ð#ó 3(ó ó ?ð3(òl E�r?   