ó
    qyüirœ  ã                   ó6  • S r SSKJr  SSKJr  SSKrSSKJr  SSKJr  SSK	J
r
Jr  SS	KJr  SS
KJr  SSKJr  SSKJr  SSKJrJrJr  SSKJrJr  SSKJr  SSKJrJrJ r J!r!  SSK"J#r#  SSK$J%r%  SSK&J'r'  SSK(J)r)J*r*  \!RV                  " \,5      r-\" SS9\ " S S\5      5       5       r.\" SS9\ " S S\5      5       5       r/ " S S\R`                  5      r1 SFS \R`                  S!\Rd                  S"\Rd                  S#\Rd                  S$\Rd                  S-  S%\3S&\34S' jjr4 " S( S)\R`                  5      r5 " S* S+\R`                  5      r6 " S, S-\R`                  5      r7 " S. S/\5      r8 " S0 S1\R`                  5      r9S2\Rd                  S3\:S4\Rd                  4S5 jr; " S6 S7\R`                  5      r< " S8 S9\R`                  5      r=\ " S: S;\5      5       r>\" S<S9 " S= S>\>5      5       r?\" S?S9 " S@ SA\>5      5       r@\" SBS9 " SC SD\>\5      5       rA/ SEQrBg)GzPyTorch Idefics3 model.é    )ÚCallable)Ú	dataclassN)Únné   )ÚACT2FN)ÚCacheÚDynamicCache)ÚGenerationMixin)Úcreate_bidirectional_mask)ÚFlashAttentionKwargs)ÚGradientCheckpointingLayer)ÚBaseModelOutputÚBaseModelOutputWithPoolingÚModelOutput)ÚALL_ATTENTION_FUNCTIONSÚPreTrainedModel)ÚUnpack)ÚTransformersKwargsÚauto_docstringÚcan_return_tupleÚlogging)Úmerge_with_config_defaults)Úcapture_outputsé   )Ú	AutoModelé   )ÚIdefics3ConfigÚIdefics3VisionConfigz|
    Base class for Idefics3 model's outputs that may also contain a past key/values (to speed up sequential decoding).
    ©Úcustom_introc                   óà   • \ rS rSr% SrSr\R                  S-  \S'   Sr	\
S-  \S'   Sr\\R                     S-  \S'   Sr\\R                     S-  \S'   Sr\\R                     S-  \S'   S	rg)
ÚIdefics3BaseModelOutputWithPasté)   a>  
last_hidden_state (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`):
    Sequence of hidden-states at the output of the last layer of the model.
    If `past_key_values` is used only the last hidden-state of the sequences of shape `(batch_size, 1,
    hidden_size)` is output.
past_key_values (`Cache`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
    It is a [`~cache_utils.Cache`] instance. For more details, see our [kv cache guide](https://huggingface.co/docs/transformers/en/kv_cache).

    Contains pre-computed hidden-states (key and values in the self-attention blocks and optionally if
    `config.is_encoder_decoder=True` in the cross-attention blocks) that can be used (see `past_key_values`
    input) to speed up sequential decoding.
image_hidden_states (`tuple(torch.FloatTensor)`, *optional*):
    Tuple of `torch.FloatTensor` (one for the output of the image embeddings, `(batch_size, num_images,
    sequence_length, hidden_size)`.
    image_hidden_states of the model produced by the vision encoder
NÚlast_hidden_stateÚpast_key_valuesÚhidden_statesÚ
attentionsÚimage_hidden_states© )Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__r$   ÚtorchÚFloatTensorÚ__annotations__r%   r   r&   Útupler'   r(   Ú__static_attributes__r)   ó    Úk/home/mande/repo/quber/.venv/lib/python3.13/site-packages/transformers/models/idefics3/modeling_idefics3.pyr"   r"   )   s|   ‡ ñð" 37Ð�u×(Ñ(¨4Ñ/Ó6Ø$(€O�U˜T‘\Ó(Ø59€M�5˜×*Ñ*Ñ+¨dÑ2Ó9Ø26€J��e×'Ñ'Ñ(¨4Ñ/Ó6Ø;?Ð˜˜u×0Ñ0Ñ1°DÑ8Ö?r4   r"   zS
    Base class for Idefics causal language model (or autoregressive) outputs.
    c                   ó  • \ rS rSr% SrSr\R                  S-  \S'   Sr	\R                  S-  \S'   Sr
\S-  \S'   Sr\\R                     S-  \S'   Sr\\R                     S-  \S'   Sr\\R                     S-  \S	'   S
rg)ÚIdefics3CausalLMOutputWithPastéH   a  
loss (`torch.FloatTensor` of shape `(1,)`, *optional*, returned when `labels` is provided):
    Language modeling loss (for next-token prediction).
logits (`torch.FloatTensor` of shape `(batch_size, sequence_length, config.vocab_size)`):
    Prediction scores of the language modeling head (scores for each vocabulary token before SoftMax).
past_key_values (`Cache`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
    It is a [`~cache_utils.Cache`] instance. For more details, see our [kv cache guide](https://huggingface.co/docs/transformers/en/kv_cache).

    Contains pre-computed hidden-states (key and values in the self-attention blocks) that can be used (see
    `past_key_values` input) to speed up sequential decoding.
image_hidden_states (`tuple(torch.FloatTensor)`, *optional*):
    Tuple of `torch.FloatTensor` (one for the output of the image embeddings, `(batch_size, num_images,
    sequence_length, hidden_size)`.
    image_hidden_states of the model produced by the vision encoder
NÚlossÚlogitsr%   r&   r'   r(   r)   )r*   r+   r,   r-   r.   r9   r/   r0   r1   r:   r%   r   r&   r2   r'   r(   r3   r)   r4   r5   r7   r7   H   s�   ‡ ñð  &*€Dˆ%×
Ñ
˜dÑ
"Ó)Ø'+€FˆE×Ñ Ñ$Ó+Ø$(€O�U˜T‘\Ó(Ø59€M�5˜×*Ñ*Ñ+¨dÑ2Ó9Ø26€J��e×'Ñ'Ñ(¨4Ñ/Ó6Ø;?Ð˜˜u×0Ñ0Ñ1°DÑ8Ö?r4   r7   c                   ó†   ^ • \ rS rSrSrS\4U 4S jjrS\R                  S\R                  S\R                  4S jrS	rU =r$ )
ÚIdefics3VisionEmbeddingséh   a<  
This is a modified version of `siglip.modelign_siglip.SiglipVisionEmbeddings` to enable images of variable
resolution.

The modifications are adapted from [Patch n' Pack: NaViT, a Vision Transformer for any Aspect Ratio and Resolution](https://huggingface.co/papers/2307.06304)
which allows treating images in their native aspect ratio and without the need to resize them to the same
fixed size. In particular, we start from the original pre-trained SigLIP model
(which uses images of fixed-size square images) and adapt it by training on images of variable resolutions.
Úconfigc                 óø  >• [         TU ]  5         UR                  U l        UR                  U l        UR
                  U l        [        R                  " UR                  U R                  U R
                  U R
                  SS9U l	        U R                  U R
                  -  U l
        U R                  S-  U l        U R                  U l        [        R                  " U R                  U R                  5      U l        g )NÚvalid)Úin_channelsÚout_channelsÚkernel_sizeÚstrideÚpaddingr   )ÚsuperÚ__init__Úhidden_sizeÚ	embed_dimÚ
image_sizeÚ
patch_sizer   ÚConv2dÚnum_channelsÚpatch_embeddingÚnum_patches_per_sideÚnum_patchesÚnum_positionsÚ	EmbeddingÚposition_embedding©Úselfr>   Ú	__class__s     €r5   rG   Ú!Idefics3VisionEmbeddings.__init__s   s¼   ø€ Ü‰ÑÔØ×+Ñ+ˆŒØ ×+Ñ+ˆŒØ ×+Ñ+ˆŒä!ŸyšyØ×+Ñ+ØŸ™ØŸ™Ø—?‘?Øñ 
ˆÔð %)§O¡O°t·±Ñ$FˆÔ!Ø×4Ñ4°aÑ7ˆÔØ!×-Ñ-ˆÔÜ"$§,¢,¨t×/AÑ/AÀ4Ç>Á>Ó"RˆÕr4   Úpixel_valuesÚpatch_attention_maskÚreturnc                 ó8  • UR                   u  p4pVU R                  U5      nUR                  S5      R                  SS5      nXPR                  -  X`R                  -  p©[
        R                  " SU R                  -  SSU R                  -  UR                  S9n[
        R                  " X9U
-  4SUR                  S9nUS S 2S S 2S4   R                  SS9nUS S 2SS S 24   R                  SS9nSU-  nSU-  nUR                  S5      nUR                  S5      n[
        R                  " UUR                  [
        R                  S9n[
        R                  " UUR                  [
        R                  S9nUS S S 24   US S 2S 4   -  nUS S S 24   US S 2S 4   -  n[
        R                  " US	S
9n[
        R                  " US	S
9nUR                  UR                  5      nUR                  UR                  5      n[
        R                   " UUSS9n[
        R                   " UUSS9nUS S 2S S 2S 4   U R                  -  US S 2S S S 24   -   nUR#                  US5      nUUR%                  US5         XÂR%                  US5      '   X€R'                  U5      -   nU$ )Nr   r   g      ð?)Údevicer   )ÚsizeÚ
fill_valuer\   ©Údim)r\   Údtypegé!çýÿï?)ÚmaxT)Úrightéÿÿÿÿ)ÚshaperN   ÚflattenÚ	transposerK   r/   ÚarangerO   r\   ÚfullÚsumr]   Úfloat32ÚclampÚtora   Ú	bucketizeÚreshapeÚviewrS   )rU   rX   rY   Ú
batch_sizeÚ_Úmax_im_hÚmax_im_wÚpatch_embedsÚ
embeddingsÚmax_nb_patches_hÚmax_nb_patches_wÚ
boundariesÚposition_idsÚnb_patches_hÚnb_patches_wÚstep_hÚstep_wÚmax_patches_hÚmax_patches_wÚ	h_indicesÚ	w_indicesÚfractional_coords_hÚfractional_coords_wÚbucket_coords_hÚbucket_coords_wÚpos_idss                             r5   ÚforwardÚ Idefics3VisionEmbeddings.forward†   s–  € Ø,8×,>Ñ,>Ñ)ˆ
�xà×+Ñ+¨LÓ9ˆØ!×)Ñ)¨!Ó,×6Ñ6°q¸!Ó<ˆ
à-5¿¹Ñ-HÈ(×VeÑVeÑJeÐ*Ü—\’\Ø�×)Ñ)Ñ)¨3°°D×4MÑ4MÑ0MÐVb×ViÑViñ
ˆ
ô —z’zØÐ1AÑAÐBÈqÐYe×YlÑYlñ
ˆð ,ªAªq°!¨GÑ4×8Ñ8¸QÐ8Ð?ˆØ+ªA¨q²!¨GÑ4×8Ñ8¸QÐ8Ð?ˆà�|Ñ#ˆØ�|Ñ#ˆà,×1Ñ1°!Ó4ˆØ,×1Ñ1°!Ó4ˆÜ—L’L °|×7JÑ7JÔRW×R_ÑR_Ñ`ˆ	Ü—L’L °|×7JÑ7JÔRW×R_ÑR_Ñ`ˆ	à'¨ªa¨Ñ0°6º!¸T¸'±?ÑBÐØ'¨ªa¨Ñ0°6º!¸T¸'±?ÑBÐä#ŸkškÐ*=ÀJÑPÐÜ#ŸkškÐ*=ÀJÑPÐà1×4Ñ4°\×5GÑ5GÓHÐØ1×4Ñ4°\×5GÑ5GÓHÐäŸ/š/Ð*=¸zÐQUÑVˆÜŸ/š/Ð*=¸zÐQUÑVˆà!¢!¢Q¨ *Ñ-°×0IÑ0IÑIÈOÒ\]Ð_cÒefÐ\fÑLgÑgˆØ—/‘/ *¨bÓ1ˆàBIÐJ^×JcÑJcÐdnÐprÓJsÑBtˆ×.Ñ.¨z¸2Ó>Ñ?à×"9Ñ"9¸,Ó"GÑGˆ
ØÐr4   )rI   rJ   rP   rO   rQ   rN   rK   rS   )r*   r+   r,   r-   r.   r   rG   r/   r0   Ú
BoolTensorÚTensorrˆ   r3   Ú__classcell__©rV   s   @r5   r<   r<   h   sI   ø† ñðSÐ3÷ Sð&+ E×$5Ñ$5ð +ÈU×M]ÑM]ð +Ðbg×bnÑbn÷ +ò +r4   r<   ÚmoduleÚqueryÚkeyÚvalueÚattention_maskÚscalingÚdropoutc                 ó°  • [         R                  " XR                  SS5      5      U-  nUb  X„-   n[        R                  R                  US[         R                  S9R                  UR                  5      n[        R                  R                  X†U R                  S9n[         R                  " Xƒ5      n	U	R                  SS5      R                  5       n	X˜4$ )Nrd   éþÿÿÿ)r`   ra   )ÚpÚtrainingr   r   )r/   Úmatmulrg   r   Ú
functionalÚsoftmaxrk   rm   ra   r”   r˜   Ú
contiguous)
rŽ   r�   r�   r‘   r’   r“   r”   ÚkwargsÚattn_weightsÚattn_outputs
             r5   Úeager_attention_forwardr    µ   s°   € ô —<’< §}¡}°R¸Ó'<Ó=ÀÑG€LØÑ!Ø#Ñ4ˆä—=‘=×(Ñ(¨¸2ÄUÇ]Á]Ð(ÐS×VÑVÐW\×WbÑWbÓc€LÜ—=‘=×(Ñ(¨È6Ï?É?Ð(Ð[€Lä—,’,˜|Ó3€KØ×'Ñ'¨¨1Ó-×8Ñ8Ó:€KàÐ$Ð$r4   c            
       ó®   ^ • \ rS rSrSrU 4S jr S
S\R                  S\R                  S-  S\\R                  \R                  S-  4   4S jjr	S	r
U =r$ )ÚIdefics3VisionAttentionéÍ   z=Multi-headed attention from 'Attention Is All You Need' paperc                 ó   >• [         TU ]  5         Xl        UR                  U l        UR
                  U l        U R                  U R                  -  U l        U R                  U R                  -  U R                  :w  a&  [        SU R                   SU R                   S35      eU R                  S-  U l	        UR                  U l        [        R                  " U R                  U R                  5      U l        [        R                  " U R                  U R                  5      U l        [        R                  " U R                  U R                  5      U l        [        R                  " U R                  U R                  5      U l        SU l        g )Nz;embed_dim must be divisible by num_heads (got `embed_dim`: z and `num_heads`: z).g      à¿F)rF   rG   r>   rH   rI   Únum_attention_headsÚ	num_headsÚhead_dimÚ
ValueErrorÚscaleÚattention_dropoutr”   r   ÚLinearÚk_projÚv_projÚq_projÚout_projÚ	is_causalrT   s     €r5   rG   Ú Idefics3VisionAttention.__init__Ñ   s  ø€ Ü‰ÑÔØŒØ×+Ñ+ˆŒØ×3Ñ3ˆŒØŸ™¨$¯.©.Ñ8ˆŒØ�=‰=˜4Ÿ>™>Ñ)¨T¯^©^Ó;ÜØMÈdÏnÉnÐM]ð ^Ø—N‘NÐ# 2ð'óð ð —]‘] DÑ(ˆŒ
Ø×/Ñ/ˆŒä—i’i §¡°·±Ó?ˆŒÜ—i’i §¡°·±Ó?ˆŒÜ—i’i §¡°·±Ó?ˆŒÜŸ	š	 $§.¡.°$·.±.ÓAˆŒð ˆ�r4   Nr&   r’   rZ   c                 ó°  • UR                   SS n/ UQSPU R                  P7nU R                  U5      R                  U5      R	                  SS5      nU R                  U5      R                  U5      R	                  SS5      nU R                  U5      R                  U5      R	                  SS5      n[        R                  " U R                  R                  [        5      n	U	" U UUUUU R                  U R                  U R                  (       d  SOU R                  S9u  p«U
R                   " / UQSP76 R#                  5       n
U R%                  U
5      n
X«4$ )z#Input shape: Batch x Time x ChannelNrd   r   r   ç        )r°   r“   r”   )re   r§   r®   rp   rg   r¬   r­   r   Úget_interfacer>   Ú_attn_implementationr    r°   r©   r˜   r”   ro   rœ   r¯   )rU   r&   r’   r�   Úinput_shapeÚhidden_shapeÚqueriesÚkeysÚvaluesÚattention_interfacerŸ   rž   s               r5   rˆ   ÚIdefics3VisionAttention.forwardç   s6  € ð $×)Ñ)¨#¨2Ð.ˆØ8˜Ð8 bÐ8¨$¯-©-Ñ8ˆØ—+‘+˜mÓ,×1Ñ1°,Ó?×IÑIÈ!ÈQÓOˆØ�{‰{˜=Ó)×.Ñ.¨|Ó<×FÑFÀqÈ!ÓLˆØ—‘˜]Ó+×0Ñ0°Ó>×HÑHÈÈAÓNˆä(?×(MÒ(MØ�K‰K×,Ñ,Ô.Eó)
Ðñ %8ØØØØØØ—n‘nØ—J‘JØ#Ÿ}Ÿ}‘C°$·,±,ñ	%
Ñ!ˆð "×)Ò)Ð;¨;Ð;¸Ò;×FÑFÓHˆØ—m‘m KÓ0ˆàÐ(Ð(r4   )r>   r”   rI   r§   r°   r¬   r¦   r¯   r®   r©   r­   ©N)r*   r+   r,   r-   r.   rG   r/   r‹   r2   rˆ   r3   rŒ   r�   s   @r5   r¢   r¢   Í   sZ   ø† ÙGõð2 /3ñ)à—|‘|ð)ð Ÿ™ tÑ+ð)ð
 
ˆu�|‰|˜UŸ\™\¨DÑ0Ð0Ñ	1÷)ó )r4   r¢   c                   ób   ^ • \ rS rSrU 4S jrS\R                  S\R                  4S jrSrU =r	$ )ÚIdefics3VisionMLPi
  c                 ó  >• [         TU ]  5         Xl        [        UR                     U l        [        R                  " UR                  UR                  5      U l
        [        R                  " UR                  UR                  5      U l        g r½   )rF   rG   r>   r   Ú
hidden_actÚactivation_fnr   r«   rH   Úintermediate_sizeÚfc1Úfc2rT   s     €r5   rG   ÚIdefics3VisionMLP.__init__  sb   ø€ Ü‰ÑÔØŒÜ# F×$5Ñ$5Ñ6ˆÔÜ—9’9˜V×/Ñ/°×1IÑ1IÓJˆŒÜ—9’9˜V×5Ñ5°v×7IÑ7IÓJˆ�r4   r&   rZ   c                 ól   • U R                  U5      nU R                  U5      nU R                  U5      nU$ r½   )rÄ   rÂ   rÅ   )rU   r&   s     r5   rˆ   ÚIdefics3VisionMLP.forward  s4   € ØŸ™ Ó/ˆØ×*Ñ*¨=Ó9ˆØŸ™ Ó/ˆØÐr4   )rÂ   r>   rÄ   rÅ   )
r*   r+   r,   r-   rG   r/   r‹   rˆ   r3   rŒ   r�   s   @r5   r¿   r¿   
  s)   ø† õKð U§\¡\ð °e·l±l÷ ò r4   r¿   c                   ó.   ^ • \ rS rSrU 4S jrS rSrU =r$ )ÚIdefics3SimpleMLPi  c                 óÎ   >• [         TU ]  5         UR                  R                  UR                  S-  -  nUR
                  R                  n[        R                  " X#SS9U l        g )Nr   F©Úbias)	rF   rG   Úvision_configrH   Úscale_factorÚtext_configr   r«   Úproj)rU   r>   Ú
input_sizeÚoutput_sizerV   s       €r5   rG   ÚIdefics3SimpleMLP.__init__  sR   ø€ Ü‰ÑÔØ×)Ñ)×5Ñ5¸×9LÑ9LÈaÑ9OÑPˆ
Ø×(Ñ(×4Ñ4ˆÜ—I’I˜j¸EÑBˆ�	r4   c                 ó$   • U R                  U5      $ r½   ©rÑ   )rU   Úxs     r5   rˆ   ÚIdefics3SimpleMLP.forward   s   € Ø�y‰y˜‹|Ðr4   rÖ   )r*   r+   r,   r-   rG   rˆ   r3   rŒ   r�   s   @r5   rÊ   rÊ     s   ø† õC÷ð r4   rÊ   c            	       ó–   ^ • \ rS rSrS\4U 4S jjr\S\R                  S\R                  S\	\
   S\R                  4S j5       rS	rU =r$ )
ÚIdefics3EncoderLayeri%  r>   c                 ó<  >• [         TU ]  5         UR                  U l        [	        U5      U l        [        R                  " U R                  UR                  S9U l	        [        U5      U l        [        R                  " U R                  UR                  S9U l        g ©N)Úeps)rF   rG   rH   rI   r¢   Ú	self_attnr   Ú	LayerNormÚlayer_norm_epsÚlayer_norm1r¿   ÚmlpÚlayer_norm2rT   s     €r5   rG   ÚIdefics3EncoderLayer.__init__&  sm   ø€ Ü‰ÑÔØ×+Ñ+ˆŒÜ0°Ó8ˆŒÜŸ<š<¨¯©¸F×<QÑ<QÑRˆÔÜ$ VÓ,ˆŒÜŸ<š<¨¯©¸F×<QÑ<QÑRˆÕr4   r&   r’   r�   rZ   c                 ó²   • UnU R                  U5      nU R                  " SUUS.UD6u  pXA-   nUnU R                  U5      nU R                  U5      nXA-   nU$ )N)r&   r’   r)   )rá   rÞ   rã   râ   )rU   r&   r’   r�   Úresidualrr   s         r5   rˆ   ÚIdefics3EncoderLayer.forward.  sz   € ð !ˆà×(Ñ(¨Ó7ˆØŸ>š>ð 
Ø'Ø)ñ
ð ñ
Ñˆð
 !Ñ0ˆà ˆØ×(Ñ(¨Ó7ˆØŸ™ Ó/ˆØ Ñ0ˆàÐr4   )rI   rá   rã   râ   rÞ   )r*   r+   r,   r-   r   rG   r   r/   r‹   r   r   r0   rˆ   r3   rŒ   r�   s   @r5   rÚ   rÚ   %  s`   ø† ðSÐ3÷ Sð ðà—|‘|ðð Ÿ™ðð Ð+Ñ,ð	ð
 
×	Ñ	óó ör4   rÚ   c                   óv   ^ • \ rS rSrSrS\4U 4S jjr\ S
S\R                  S-  S\
\-  4S jj5       rS	rU =r$ )ÚIdefics3EncoderiI  z¡
Transformer encoder consisting of `config.num_hidden_layers` self attention layers. Each layer is a
[`Idefics3EncoderLayer`].

Args:
    config: Idefics3Config
r>   c                 óÖ   >• [         TU ]  5         Xl        [        R                  " [        UR                  5       Vs/ s H  n[        U5      PM     sn5      U l        SU l	        g s  snf )NF)
rF   rG   r>   r   Ú
ModuleListÚrangeÚnum_hidden_layersrÚ   ÚlayersÚgradient_checkpointing)rU   r>   rr   rV   s      €r5   rG   ÚIdefics3Encoder.__init__R  sT   ø€ Ü‰ÑÔØŒÜ—m’mÌ5ÐQW×QiÑQiÔKjÓ$kÒKjÀaÔ%9¸&Ö%AÑKjÑ$kÓlˆŒØ&+ˆÕ#ùò %ls   ½A&Nr’   rZ   c                 óT   • UnU R                    H  nU" UU5      nUnM     [        US9$ )N©r$   )rî   r   )rU   Úinputs_embedsr’   r&   Úencoder_layerÚlayer_outputss         r5   rˆ   ÚIdefics3Encoder.forwardY  s;   € ð &ˆØ!Ÿ[œ[ˆMÙ)ØØóˆMð
 *ŠMñ )ô °Ñ?Ð?r4   )r>   rï   rî   r½   )r*   r+   r,   r-   r.   r   rG   r   r/   r‹   r2   r   rˆ   r3   rŒ   r�   s   @r5   ré   ré   I  sS   ø† ñð,˜~÷ ,ð ð /3ñ@ð Ÿ™ tÑ+ð@ð 
�Ñ	 ô	@ó ö@r4   ré   r&   Ún_reprZ   c                 ó    • U R                   u  p#pEUS:X  a  U $ U SS2SS2SSS2SS24   R                  X#XU5      n U R                  X#U-  XE5      $ )zÈ
This is the equivalent of torch.repeat_interleave(x, dim=1, repeats=n_rep). The hidden states go from (batch,
num_key_value_heads, seqlen, head_dim) to (batch, num_attention_heads, seqlen, head_dim)
r   N)re   Úexpandro   )r&   r÷   ÚbatchÚnum_key_value_headsÚslenr§   s         r5   Ú	repeat_kvrý   l  s_   € ð
 2?×1DÑ1DÑ.€E Ø�ƒzØÐØ!¢!¢Q¨ªa²Ð"2Ñ3×:Ñ:¸5ÐW\ÐdlÓm€MØ× Ñ  ¸eÑ(CÀTÓTÐTr4   c                   óx   ^ • \ rS rSrS
S\SS4U 4S jjjrS\R                  S\R                  4S jrS r	S	r
U =r$ )ÚIdefics3RMSNormiy  rÝ   rZ   Nc                 óŒ   >• [         TU ]  5         [        R                  " [        R
                  " U5      5      U l        X l        g)z.
Idefics3RMSNorm is equivalent to T5LayerNorm
N)rF   rG   r   Ú	Parameterr/   ÚonesÚweightÚvariance_epsilon)rU   rH   rÝ   rV   s      €r5   rG   ÚIdefics3RMSNorm.__init__z  s/   ø€ ô 	‰ÑÔÜ—l’l¤5§:¢:¨kÓ#:Ó;ˆŒØ #Õr4   r&   c                 ó  • UR                   nUR                  [        R                  5      nUR	                  S5      R                  SSS9nU[        R                  " X0R                  -   5      -  nU R                  UR                  U5      -  $ )Nr   rd   T)Úkeepdim)	ra   rm   r/   rk   ÚpowÚmeanÚrsqrtr  r  )rU   r&   Úinput_dtypeÚvariances       r5   rˆ   ÚIdefics3RMSNorm.forward‚  sw   € Ø#×)Ñ)ˆØ%×(Ñ(¬¯©Ó7ˆØ ×$Ñ$ QÓ'×,Ñ,¨R¸Ð,Ð>ˆØ%¬¯ª°H×?TÑ?TÑ4TÓ(UÑUˆØ�{‰{˜]×-Ñ-¨kÓ:Ñ:Ð:r4   c                 ó^   • [        U R                  R                  5       SU R                   3$ )Nz, eps=)r2   r  re   r  ©rU   s    r5   Ú
extra_reprÚIdefics3RMSNorm.extra_repr‰  s*   € Ü˜Ÿ™×)Ñ)Ó*Ð+¨6°$×2GÑ2GÐ1HÐIÐIr4   )r  r  )g�íµ ÷Æ°>)r*   r+   r,   r-   ÚfloatrG   r/   r‹   rˆ   r  r3   rŒ   r�   s   @r5   rÿ   rÿ   y  sB   ø† ñ$¨ð $¸$÷ $ð $ð; U§\¡\ð ;°e·l±lô ;÷Jð Jr4   rÿ   c                   ó8   ^ • \ rS rSrU 4S jrSS jrS rSrU =r$ )ÚIdefics3Connectori�  c                 ód   >• [         TU ]  5         UR                  U l        [        U5      U l        g r½   )rF   rG   rÏ   rÊ   Úmodality_projectionrT   s     €r5   rG   ÚIdefics3Connector.__init__Ž  s)   ø€ Ü‰ÑÔØ"×/Ñ/ˆÔÜ#4°VÓ#<ˆÕ r4   c                 ó¨  • UR                  5       u  p4n[        US-  5      =pgUR                  X6Xu5      nUR                  X6[        Xr-  5      XR-  5      nUR                  SSSS5      nUR	                  U[        Xr-  5      [        Xb-  5      XRS-  -  5      nUR                  SSSS5      nUR	                  U[        XBS-  -  5      XRS-  -  5      nU$ )Ng      à?r   r   r   r   )r]   Úintrp   Úpermutero   )rU   r×   rÏ   ÚbszÚseqrI   ÚheightÚwidths           r5   Úpixel_shuffleÚIdefics3Connector.pixel_shuffle“  sÐ   € ØŸf™f›hÑˆ�)Ü˜S #™X›Ð&ˆØ�F‰F�3 Ó1ˆØ�F‰F�3¤ EÑ$8Ó 9¸9Ñ;SÓTˆØ�I‰I�a˜˜A˜qÓ!ˆØ�I‰I�cœ3˜uÑ3Ó4´c¸&Ñ:OÓ6PÐR[ÐmnÑ_nÑRoÓpˆØ�I‰I�a˜˜A˜qÓ!ˆØ�I‰I�cœ3˜s°A¡oÑ6Ó7¸ÐTUÁoÑ9VÓWˆØˆr4   c                 ó^   • U R                  XR                  5      nU R                  U5      nU$ r½   )r  rÏ   r  )rU   r(   s     r5   rˆ   ÚIdefics3Connector.forwardž  s2   € Ø"×0Ñ0Ð1D×FWÑFWÓXÐØ"×6Ñ6Ð7JÓKÐØ"Ð"r4   )r  rÏ   )r   )	r*   r+   r,   r-   rG   r  rˆ   r3   rŒ   r�   s   @r5   r  r  �  s   ø† õ=ô
	÷#ð #r4   r  c                   óH   • \ rS rSr% \\S'   SrSrSrSS/r	Sr
SrSrSrSrS	rg
)ÚIdefics3PreTrainedModeli¤  r>   Úmodel)ÚimageÚtextTr¢   ÚIdefics3DecoderLayerr%   r)   N)r*   r+   r,   r-   r   r1   Úbase_model_prefixÚinput_modalitiesÚsupports_gradient_checkpointingÚ_no_split_modulesÚ_skip_keys_device_placementÚ_supports_flash_attnÚ_supports_sdpaÚ_supports_flex_attnÚ_supports_attention_backendr3   r)   r4   r5   r$  r$  ¤  sC   ‡ àÓØÐØ(ÐØ&*Ð#Ø2Ð4JÐKÐØ"3ÐØÐØ€NØÐØ"&Ór4   r$  zO
    The Idefics3 Vision Transformer Model outputting raw image embedding.
    c            
       ó´   ^ • \ rS rSr% \\S'   Sr\\S.r	S\4U 4S jjr
S rS r\\" SS	9 SS\R                   S
-  S\\   S\\-  4S jj5       5       rSrU =r$ )ÚIdefics3VisionTransformeri²  r>   )r&  )r&   r'   c                 ó  >• [         TU ]  U5        UR                  n[        U5      U l        [        U5      U l        UR                  U l        [        R                  " X!R                  S9U l        U R                  5         g rÜ   )rF   rG   rH   r<   rv   ré   ÚencoderrK   r   rß   rà   Úpost_layernormÚ	post_init)rU   r>   rI   rV   s      €r5   rG   Ú"Idefics3VisionTransformer.__init__¿  sa   ø€ Ü‰Ñ˜Ô Ø×&Ñ&ˆ	ä2°6Ó:ˆŒÜ& vÓ.ˆŒØ ×+Ñ+ˆŒÜ Ÿlšl¨9×:OÑ:OÑPˆÔà�‰Õr4   c                 ó   • U R                   $ r½   ©rv   r  s    r5   Úget_input_embeddingsÚ.Idefics3VisionTransformer.get_input_embeddingsË  s   € Ø�‰Ðr4   c                 ó   • Xl         g r½   r:  ©rU   r‘   s     r5   Úset_input_embeddingsÚ.Idefics3VisionTransformer.set_input_embeddingsÏ  s   € Ø�r4   F)Útie_last_hidden_statesNrY   r�   rZ   c                 óä  • UR                  S5      nUcq  U R                  n[        R                  " UUR                  S5      U-  UR                  S5      U-  45      nUR	                  [        R
                  UR                  S9nU R                  XS9nUR                  US5      n[        U R                  UUS9nU R                  UUS9nUR                  nU R                  U5      n[        US	9$ )
Nr   r   r   ©ra   r\   )rX   rY   rd   )r>   ró   r’   )ró   r’   rò   )r]   rK   r/   r  rm   Úboolr\   rv   rp   r   r>   r5  r$   r6  r   )	rU   rX   rY   r�   rq   rK   r&   Úencoder_outputsr$   s	            r5   rˆ   Ú!Idefics3VisionTransformer.forwardÒ  s	  € ð "×&Ñ& qÓ)ˆ
ØÑ'ØŸ™ˆJÜ#(§:¢:àØ ×%Ñ% aÓ(¨JÑ6Ø ×%Ñ% aÓ(¨JÑ6ðó$Ð ð $8×#:Ñ#:ÄÇÁÐT`×TgÑTgÐ#:Ð#hÐ àŸ™°\˜Ðmˆà3×8Ñ8¸ÀRÓHÐä8Ø—;‘;Ø'Ø/ñ 
Ðð ,0¯<©<Ø'Ø/ð ,8ð ,
ˆð
 ,×=Ñ=ÐØ ×/Ñ/Ð0AÓBÐäØ/ñ
ð 	
r4   )rv   r5  rK   r6  r½   )r*   r+   r,   r-   r   r1   r*  rÚ   r¢   Ú_can_record_outputsrG   r;  r?  r   r   r/   rŠ   r   r   r2   r   rˆ   r3   rŒ   r�   s   @r5   r3  r3  ²  s�   ø‡ ð !Ó Ø!Ðà-Ø-ñÐð
	Ð3÷ 	òò ð  Ù¨EÑ2ð 9=ñ&
ð $×.Ñ.°Ñ5ð&
ð Ð+Ñ,ð	&
ð
 
�Ñ	 ô&
ó 3ó  ö&
r4   r3  zZ
    Idefics3 model consisting of a SIGLIP vision encoder and Llama3 language decoder
    c                   ó>  ^ • \ rS rSrS\4U 4S jjrS rS rS\R                  S\R                  S-  S	\R                  S-  4S
 jr\\ SS\R                  S\R                  S-  S\\   S\\-  4S jj5       5       r\\" SS9         SS\R                  S-  S\R                  S-  S\R                  S-  S\S-  S\R                  S-  S\R                  S-  S\R*                  S-  S	\R                  S-  S\S-  S\\   S\\-  4S jj5       5       rSrU =r$ )ÚIdefics3Modeliý  r>   c                 ó\  >• [         TU ]  U5        U R                  R                  R                  U l        U R                  R                  R                  U l        [        R                  UR                  5      U l
        [        U5      U l        [        R                  " UR                  5      U l        [!        UR                  R"                  UR                  R$                  -  S-  UR&                  S-  -  5      U l        U R                  R*                  U l        U R-                  5         g )Nr   )rF   rG   r>   rÐ   Úpad_token_idÚpadding_idxÚ
vocab_sizer3  Ú_from_configrÎ   Úvision_modelr  Ú	connectorr   Úfrom_configÚ
text_modelr  rJ   rK   rÏ   Úimage_seq_lenÚimage_token_idr7  rT   s     €r5   rG   ÚIdefics3Model.__init__  sß   ø€ Ü‰Ñ˜Ô ØŸ;™;×2Ñ2×?Ñ?ˆÔØŸ+™+×1Ñ1×<Ñ<ˆŒä5×BÑBÀ6×CWÑCWÓXˆÔÜ*¨6Ó2ˆŒÜ#×/Ò/°×0BÑ0BÓCˆŒä Ø×"Ñ"×-Ñ-°×1EÑ1E×1PÑ1PÑPÐUVÑVÐ[a×[nÑ[nÐpqÑ[qÑró
ˆÔð #Ÿk™k×8Ñ8ˆÔà�‰Õr4   c                 ó6   • U R                   R                  5       $ r½   )rR  r;  r  s    r5   r;  Ú"Idefics3Model.get_input_embeddings  s   € Ø�‰×3Ñ3Ó5Ð5r4   c                 ó:   • U R                   R                  U5        g r½   )rR  r?  r>  s     r5   r?  Ú"Idefics3Model.set_input_embeddings  s   € Ø�‰×,Ñ,¨UÕ3r4   Ú	input_idsró   Nr(   c           	      óð  • Ucj  X R                  5       " [        R                  " U R                  R                  [        R
                  UR                  S95      :H  nUR                  S5      nOXR                  R                  :H  nUR                  S5      R                  U5      R                  UR                  5      nUR                  UR                  UR                  5      nUR                  XC5      nU$ )a3  
This method aims at merging the token embeddings with the image hidden states into one single sequence of vectors that are fed to the transformer LM.
The merging happens as follows:
- The text token sequence is: `tok_1 tok_2 tok_3 <fake_token_around_image> <image> <image> ... <image> <fake_token_around_image> tok_4`.
- We get the image hidden states for the image through the vision encoder and that hidden state, after a pixel shuffle operation, is then projected into the text embedding space.
We thus have a sequence of image hidden states of size (1, image_seq_len, hidden_dim), where 1 is for batch_size of 1 image and hidden_dim is the hidden_dim of the LM transformer.
- The merging happens so that we obtain the following sequence: `vector_tok_1 vector_tok_2 vector_tok_3 vector_fake_tok_around_image {sequence of image_seq_len image hidden states} vector_fake_toke_around_image vector_tok_4`. That sequence is fed to the LM.
- To fit the format of that sequence, `input_ids`, `inputs_embeds`, `attention_mask` are all 3 adapted to insert the image hidden states.
rC  rd   )r;  r/   Útensorr>   rT  Úlongr\   ÚallÚ	unsqueezeÚ	expand_asrm   ra   Úmasked_scatter)rU   rZ  ró   r(   Úspecial_image_masks        r5   Úinputs_mergerÚIdefics3Model.inputs_merger  sÒ   € ð ÑØ!.×2KÑ2KÔ2MÜ—’˜TŸ[™[×7Ñ7¼u¿z¹zÐR_×RfÑRfÑgó3ñ "Ðð "4×!7Ñ!7¸Ó!;Ñà!*¯k©k×.HÑ.HÑ!HÐà/×9Ñ9¸"Ó=×GÑGÈÓV×YÑYÐZg×ZnÑZnÓoÐØ1×4Ñ4°]×5IÑ5IÈ=×K^ÑK^Ó_ÐØ%×4Ñ4Ð5GÓ]ˆØÐr4   rX   Úpixel_attention_maskr�   rZ   c                 ó‚  • UR                   u  pEpgnUR                  U R                  S9nUR                  " XE-  /UR                   SS Q76 nUR                   SS R	                  5       n	US:H  R                  SS9U	:g  n
X   R                  5       nUc_  [        R                  " UR                  S5      UR                  S5      UR                  S	5      4[        R                  UR                  S
9nO4UR                  " XE-  /UR                   SS Q76 nX*   R                  5       nU R                  R                  R                  nUR                  SX»S9nUR                  SX»S9nUR                  SS9S:„  R                  5       nU R                   " SXSS.UD6nUR"                  nU R%                  U5      nUUl        U$ )á  
pixel_values (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
    The tensors corresponding to the input images.
pixel_attention_mask (`torch.LongTensor`, *optional*):
    The attention mask indicating padded regions in the image.
)ra   r   Nr   r³   )rd   r–   éýÿÿÿr_   r   r   )r]   ra   r\   )Ú	dimensionr]   Ústep)rd   r–   T)rX   rY   Úreturn_dictr)   )re   rm   ra   rp   Únumelrj   rœ   r/   r  r]   rD  r\   r>   rÎ   rK   ÚunfoldrO  r$   rP  Úpooler_output)rU   rX   re  r�   rq   Ú
num_imagesrM   r  r  Únb_values_per_imageÚreal_images_indsrK   Úpatches_subgridrY   Úimage_outputsr(   Úimage_featuress                    r5   Úget_image_featuresÚ Idefics3Model.get_image_features7  sã  € ð ?K×>PÑ>PÑ;ˆ
 °eØ#—‘¨T¯Z©Z�Ð8ˆØ#×(Ò(¨Ñ)@ÐZÀ<×CUÑCUÐVWÐVXÐCYÒZˆð +×0Ñ0°°Ð4×:Ñ:Ó<ÐØ(¨CÑ/×4Ñ4¸Ð4ÐFÐJ]Ñ]ÐØ#Ñ5×@Ñ@ÓBˆð  Ñ'Ü#(§:¢:Ø"×'Ñ'¨Ó*¨L×,=Ñ,=¸aÓ,@À,×BSÑBSÐTUÓBVÐWÜ—j‘jØ#×*Ñ*ñ$Ñ ð $8×#<Ò#<¸ZÑ=TÐ#vÐWk×WqÑWqÐrsÐrtÐWuÒ#vÐ Ø#7Ñ#I×#TÑ#TÓ#VÐ à—[‘[×.Ñ.×9Ñ9ˆ
Ø.×5Ñ5ÀÈ
Ð5ÐdˆØ)×0Ñ0¸1À:Ð0Ð_ˆØ /× 3Ñ 3¸Ð 3Ð AÀAÑ E×KÑKÓMÐð ×)Ò)ð 
Ø%Ð^bñ
Øflñ
ˆð ,×=Ñ=Ðð Ÿ™Ð(;Ó<ˆØ&4ˆÔ#àÐr4   aØ  
        Inputs fed to the model can have an arbitrary number of images. To account for this, pixel_values fed to
        the model have image padding -> (batch_size, max_num_images, 3, max_heights, max_widths) where
        max_num_images is the maximum number of images among the batch_size samples in the batch.
        Padding images are not needed beyond padding the pixel_values at the entrance of the model.
        For efficiency, we only pass through the vision_model's forward the real images by
        discarding the padding images i.e. pixel_values of size (image_batch_size, 3, height, width) where
        image_batch_size would be 7 when num_images_per_sample=[1, 3, 1, 2] and max_num_images would be 3.
        r   r’   rz   r%   Ú	use_cachec
           	      ó  • U R                   (       a9  U R                  R                  (       a  U	(       a  [        R	                  S5        Sn	Ub  UR
                  u  p¼OUb  UR
                  u  p¼nO[        S5      eU	(       a  Uc  [        U R                  S9nUc9  U R                  R                  5       " U5      R                  U R                  5      nUb  Ub  [        S5      eUb  U R                  XgSS9R                  nO'Ub$  UR                  U R                  UR                  S9nUb  U R                  UUUS	9nU R                  " SUUUUU	S
.U
D6n[!        UR"                  UR$                  UR&                  UR(                  US9$ )aT  
pixel_attention_mask (`torch.Tensor` of shape `(batch_size, image_size, image_size)`, *optional*):
    Mask to avoid performing attention on padding pixel indices.
image_hidden_states (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
    The hidden states of the image encoder after modality projection.
zZ`use_cache=True` is incompatible with gradient checkpointing. Setting `use_cache=False`...Fz5You have to specify either input_ids or inputs_embeds)r>   zMYou cannot specify both pixel_values and image_hidden_states at the same timeT)rk  rC  )rZ  ró   r(   )ró   r’   rz   r%   rw  )r$   r%   r&   r'   r(   r)   )r˜   rR  rï   ÚloggerÚwarning_oncere   r¨   r	   r>   r;  rm   r\   ru  rn  ra   rc  r"   r$   r%   r&   r'   )rU   rZ  r’   rz   r%   ró   rX   re  r(   rw  r�   rq   Ú
seq_lengthrr   Úoutputss                  r5   rˆ   ÚIdefics3Model.forwardk  s¤  € ð@ �=�=˜TŸ_™_×C×CÎ	Ü×ÑØlôð ˆIð Ñ Ø%.§_¡_Ñ"ˆJ˜
ØÑ&Ø(5×(;Ñ(;Ñ%ˆJ¡AäÐTÓUÐUæ˜Ñ0Ü*°$·+±+Ñ>ˆOàÑ Ø ŸO™O×@Ñ@ÔBÀ9ÓM×PÑPÐQU×Q\ÑQ\Ó]ˆMð Ñ#Ð(;Ñ(GÜÐlÓmÐmØÑ%Ø"&×"9Ñ"9ØÀð #:ð #ç‰mñ  ð !Ñ,Ø"5×"8Ñ"8¸t¿z¹zÐR[×RbÑRbÐ"8Ð"cÐàÑ*ð !×.Ñ.Ø#Ø+Ø$7ð /ð ˆMð —/’/ð 
Ø'Ø)Ø%Ø+Øñ
ð ñ
ˆô /Ø%×7Ñ7Ø#×3Ñ3Ø!×/Ñ/Ø×)Ñ)Ø 3ñ
ð 	
r4   )rP  rS  rT  rL  rR  rO  rM  r½   )	NNNNNNNNN)r*   r+   r,   r-   r   rG   r;  r?  r/   Ú
LongTensorr‹   rc  r   r   r0   r   r   r2   r   ru  r   rŠ   rD  r   r"   rˆ   r3   rŒ   r�   s   @r5   rI  rI  ý  sÒ  ø† ð˜~÷ ò"6ò4ðà×#Ñ#ðð —|‘| dÑ*ðð #Ÿ\™\¨DÑ0ô	ð8 Øð 9=ñ0à×'Ñ'ð0ð $×.Ñ.°Ñ5ð0ð Ð+Ñ,ð	0ð
 
Ð+Ñ	+ô0ó ó ð0ðd Ùðñ
ð .2Ø.2Ø04Ø(,Ø26Ø15Ø8<Ø8<Ø!%ñJ
à×#Ñ# dÑ*ðJ
ð Ÿ™ tÑ+ðJ
ð ×&Ñ&¨Ñ-ð	J
ð
  ™ðJ
ð ×(Ñ(¨4Ñ/ðJ
ð ×'Ñ'¨$Ñ.ðJ
ð $×.Ñ.°Ñ5ðJ
ð #×.Ñ.°Ñ5ðJ
ð ˜$‘;ðJ
ð Ð-Ñ.ðJ
ð 
Ð0Ñ	0ôJ
ó
ó öJ
r4   rI  zˆ
    The Idefics3 Model with a language modeling head. It is made up a SigLIP vision encoder, with a language modeling head on top.
    c                   ó0  ^ • \ rS rSrSS0rU 4S jrS rS r\ SS\	R                  S	\	R                  S-  S
\\   S\\-  4S jj5       r\\           SS\	R                  S-  S\	R$                  S-  S\	R                  S-  S\S-  S\	R                  S-  S\	R                  S-  S	\	R(                  S-  S\	R                  S-  S\	R                  S-  S\S-  S\\	R$                  -  S
\\   S\\-  4S jj5       5       r         SU 4S jjrSrU =r$ )Ú Idefics3ForConditionalGenerationiÄ  zlm_head.weightz$model.text_model.embed_tokens.weightc                 óV  >• [         TU ]  U5        [        U5      U l        U R                  R
                  U l        [        R                  " UR                  R                  UR                  R                  SS9U l        UR                  R                  U l
        U R                  5         g )NFrÌ   )rF   rG   rI  r%  r>   rT  r   r«   rÐ   rH   rM  Úlm_headr7  rT   s     €r5   rG   Ú)Idefics3ForConditionalGeneration.__init__Í  sz   ø€ Ü‰Ñ˜Ô Ü" 6Ó*ˆŒ
Ø"Ÿk™k×8Ñ8ˆÔä—y’y ×!3Ñ!3×!?Ñ!?À×ASÑAS×A^ÑA^ÐejÑkˆŒØ ×,Ñ,×7Ñ7ˆŒð 	�‰Õr4   c                 óJ   • U R                   R                  R                  5       $ r½   )r%  rR  r;  r  s    r5   r;  Ú5Idefics3ForConditionalGeneration.get_input_embeddingsÙ  s   € Ø�z‰z×$Ñ$×9Ñ9Ó;Ð;r4   c                 óN   • U R                   R                  R                  U5        g r½   )r%  rR  r?  r>  s     r5   r?  Ú5Idefics3ForConditionalGeneration.set_input_embeddingsÝ  s   € Ø�
‰
×Ñ×2Ñ2°5Õ9r4   NrX   re  r�   rZ   c                 ó>   • U R                   R                  " SXS.UD6$ )rg  )rX   re  r)   )r%  ru  )rU   rX   re  r�   s       r5   ru  Ú3Idefics3ForConditionalGeneration.get_image_featuresà  s+   € ð �z‰z×,Ò,ð 
Ø%ñ
ØTZñ
ð 	
r4   rZ  r’   rz   r%   ró   r(   Úlabelsrw  Úlogits_to_keepc                 ó   • U R                   " SUUUUUUUUU
SS.
UD6nUS   n[        U[        5      (       a  [        U* S5      OUnU R	                  USS2USS24   5      nSnU	b3  U R
                  " SUX�R                  R                  R                  S.UD6n[        UUUR                  UR                  UR                  UR                  S9$ )a^  
pixel_attention_mask (`torch.Tensor` of shape `(batch_size, image_size, image_size)`, *optional*):
    Mask to avoid performing attention on padding pixel indices.
image_hidden_states (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
    The hidden states of the image encoder after modality projection.
labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
    Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
    config.vocab_size]` or `model.image_token_id` (where `model` is your instance of `Idefics3ForConditionalGeneration`).
    Tokens with indices set to `model.image_token_id` are ignored (masked), the loss is only
    computed for the tokens with labels in `[0, ..., config.vocab_size]`.

Example:

```python
>>> import torch
>>> from PIL import Image
>>> from io import BytesIO

>>> from transformers import AutoProcessor, AutoModelForImageTextToText
>>> from transformers.image_utils import load_image

>>> # Note that passing the image urls (instead of the actual pil images) to the processor is also possible
>>> image1 = load_image("https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg")
>>> image2 = load_image("https://cdn.britannica.com/59/94459-050-DBA42467/Skyline-Chicago.jpg")
>>> image3 = load_image("https://cdn.britannica.com/68/170868-050-8DDE8263/Golden-Gate-Bridge-San-Francisco.jpg")

>>> processor = AutoProcessor.from_pretrained("HuggingFaceM4/Idefics3-8B-Llama3")
>>> model = AutoModelForImageTextToText.from_pretrained("HuggingFaceM4/Idefics3-8B-Llama3", dtype=torch.bfloat16, device_map="auto")

>>> # Create inputs
>>> messages = [
...     {
...         "role": "user",
...         "content": [
...             {"type": "image"},
...             {"type": "text", "text": "In this image, we can see the city of New York, and more specifically the Statue of Liberty."},
...             {"type": "image"},
...             {"type": "text", "text": "What can we see in this image?"},
...         ]
...     },
...     {
...         "role": "user",
...         "content": [
...             {"type": "image"},
...             {"type": "text", "text": "In which city is that bridge located?"},
...         ]
...     }
... ]

>>> prompts = [processor.apply_chat_template([message], add_generation_prompt=True) for message in messages]
>>> images = [[image1, image2], [image3]]
>>> inputs = processor(text=prompts, images=images, padding=True, return_tensors="pt").to(model.device)

>>> # Generate
>>> generated_ids = model.generate(**inputs, max_new_tokens=256)
>>> generated_texts = processor.batch_decode(generated_ids, skip_special_tokens=True)

>>> print(generated_texts[0])
Assistant: There are buildings, trees, lights, and water visible in this image.

>>> print(generated_texts[1])
Assistant: The bridge is in San Francisco.
```T)
rZ  r’   rz   r%   ró   rX   re  r(   rw  rk  r   N)r:   rŠ  rM  )r9   r:   r%   r&   r'   r(   r)   )r%  Ú
isinstancer  Úslicer‚  Úloss_functionr>   rÐ   rM  r7   r%   r&   r'   r(   )rU   rZ  r’   rz   r%   ró   rX   re  r(   rŠ  rw  r‹  r�   r|  r&   Úslice_indicesr:   r9   s                     r5   rˆ   Ú(Idefics3ForConditionalGeneration.forwardñ  sù   € ðb —*’*ð 
ØØ)Ø%Ø+Ø'Ø%Ø!5Ø 3ØØñ
ð ñ
ˆð   ™
ˆä8BÀ>ÔSV×8WÑ8Wœ˜~˜o¨tÔ4Ð]kˆØ—‘˜mªA¨}ºaÐ,?Ñ@ÓAˆàˆØÑØ×%Ò%ð Ø f¿¹×9PÑ9P×9[Ñ9[ñØ_eñˆDô .ØØØ#×3Ñ3Ø!×/Ñ/Ø×)Ñ)Ø '× ;Ñ ;ñ
ð 	
r4   c                 ót   >• [         TU ]  " U4UUUUUUUU	U
S.	UD6nUc  U
(       a  U	(       d
  S US'   S US'   U$ )N)	r%   r’   ró   rX   re  r(   r‹  Úis_first_iterationrw  rX   re  )rF   Úprepare_inputs_for_generation)rU   rZ  r%   r’   ró   rX   re  r(   r‹  r“  rw  r�   Úmodel_inputsrV   s                €r5   r”  Ú>Idefics3ForConditionalGeneration.prepare_inputs_for_generatione  si   ø€ ô" ‘wÒ<Øð
à+Ø)Ø'Ø%Ø!5Ø 3Ø)Ø1Øñ
ð ñ
ˆð Ñ*®yÖASØ+/ˆL˜Ñ(Ø37ˆLÐ/Ñ0àÐr4   )rT  r‚  r%  rM  r½   )NNNNNNNNNNr   )	NNNNNNNFF)r*   r+   r,   r-   Ú_tied_weights_keysrG   r;  r?  r   r/   r0   r~  r   r   r2   r   ru  r   r‹   r   rŠ   rD  r  r7   rˆ   r”  r3   rŒ   r�   s   @r5   r€  r€  Ä  så  ø† ð +Ð,RÐSÐõ	ò<ò:ð ð 9=ñ
à×'Ñ'ð
ð $×.Ñ.°Ñ5ð
ð Ð+Ñ,ð	
ð
 
Ð+Ñ	+ô
ó ð
ð  Øð .2Ø.2Ø04Ø(,Ø26Ø15Ø8<Ø8<Ø*.Ø!%Ø-.ño
à×#Ñ# dÑ*ðo
ð Ÿ™ tÑ+ðo
ð ×&Ñ&¨Ñ-ð	o
ð
  ™ðo
ð ×(Ñ(¨4Ñ/ðo
ð ×'Ñ'¨$Ñ.ðo
ð $×.Ñ.°Ñ5ðo
ð #×.Ñ.°Ñ5ðo
ð × Ñ  4Ñ'ðo
ð ˜$‘;ðo
ð ˜eŸl™lÑ*ðo
ð Ð+Ñ,ðo
ð 
Ð/Ñ	/ôo
ó ó ðo
ðj ØØØØ!Ø ØØ Ø÷#õ #r4   r€  )r€  r$  rI  r3  )r³   )Cr.   Úcollections.abcr   Údataclassesr   r/   r   Úactivationsr   Úcache_utilsr   r	   Ú
generationr
   Úmasking_utilsr   Úmodeling_flash_attention_utilsr   Úmodeling_layersr   Úmodeling_outputsr   r   r   Úmodeling_utilsr   r   Úprocessing_utilsr   Úutilsr   r   r   r   Úutils.genericr   Úutils.output_capturingr   Úautor   Úconfiguration_idefics3r   r   Ú
get_loggerr*   ry  r"   r7   ÚModuler<   r‹   r  r    r¢   r¿   rÊ   rÚ   ré   r  rý   rÿ   r  r$  r3  rI  r€  Ú__all__r)   r4   r5   Ú<module>r«     sq  ðñ å $Ý !ã Ý å !ß .Ý )Ý 6Ý BÝ 9ß XÑ Xß FÝ &ß RÓ RÝ 7Ý 5Ý ß Hð 
×	Ò	˜HÓ	%€ñ ðñð
 ô@ kó @ó óð@ñ2 ðñð
 ô@ [ó @ó óð@ô4I˜rŸy™yô Iðh ñ%Ø�I‰Ið%à�<‰<ð%ð 
�‰ð%ð �<‰<ð	%ð
 —L‘L 4Ñ'ð%ð ð%ð õ%ô09)˜bŸi™iô 9)ôz˜Ÿ	™	ô ô˜Ÿ	™	ô ô Ð5ô  ôH@�b—i‘iô @ðF	U˜UŸ\™\ð 	U°#ð 	U¸%¿,¹,ô 	UôJ�b—i‘iô Jô(#˜Ÿ	™	ô #ð. ô
'˜oó 
'ó ð
'ñ ðñô
C
Ð 7ó C
óð
C
ñL ðñô

Ð+ó 
óð

ñD ðñô
Ð'>Àó óð
òD x�r4   