ó
    qyüi~1  ã                   ó>  • S r SSKJr  SSKJr  SSKJr  SSKJrJ	r	  \" SS	9\ " S
 S\5      5       5       r
\" SS	9\ " S S\5      5       5       r\" SS	9\ " S S\5      5       5       r\" SS	9\ " S S\5      5       5       r\" SS	9\ " S S\5      5       5       r/ SQrg)zSAM2 model configurationé    )Ústricté   )ÚPreTrainedConfig)Úauto_docstringé   )ÚCONFIG_MAPPINGÚ
AutoConfigzfacebook/sam2.1-hiera-tiny)Ú
checkpointc                   óâ  ^ • \ rS rSr% SrSrSrSr\\	S'   Sr
\\	S'   S	r\\	S
'   Sr\\\   -  S-  \	S'   Sr\\\   -  S-  \	S'   Sr\\\   -  S-  \	S'   Sr\\\   -  S-  \	S'   Sr\\\   -  S-  \	S'   Sr\\   S-  \	S'   S	r\\	S'   Sr\\   S-  \	S'   Sr\\   S-  \	S'   Sr\\   S-  \	S'   Sr\\   S-  \	S'   Sr\\   S-  \	S'   Sr\\	S'   Sr\\	S'   Sr\\	S'   Sr\\	S'   U 4S  jrS!r U =r!$ )"ÚSam2HieraDetConfigé   aÐ  
patch_kernel_size (`list[int]`, *optional*, defaults to `[7, 7]`):
    The kernel size of the patch.
patch_stride (`list[int]`, *optional*, defaults to `[4, 4]`):
    The stride of the patch.
patch_padding (`list[int]`, *optional*, defaults to `[3, 3]`):
    The padding of the patch.
query_stride (`list[int]`, *optional*, defaults to `[2, 2]`):
    The downsample stride between stages.
window_positional_embedding_background_size (`list[int]`, *optional*, defaults to `[7, 7]`):
    The window size per stage when not using global attention.
num_query_pool_stages (`int`, *optional*, defaults to 3):
    The number of query pool stages.
blocks_per_stage (`list[int]`, *optional*, defaults to `[1, 2, 7, 2]`):
    The number of blocks per stage.
embed_dim_per_stage (`list[int]`, *optional*, defaults to `[96, 192, 384, 768]`):
    The embedding dimension per stage.
num_attention_heads_per_stage (`list[int]`, *optional*, defaults to `[1, 2, 4, 8]`):
    The number of attention heads per stage.
window_size_per_stage (`list[int]`, *optional*, defaults to `[8, 4, 14, 7]`):
    The window size per stage.
global_attention_blocks (`list[int]`, *optional*, defaults to `[5, 7, 9]`):
    The blocks where global attention is used.
Úbackbone_configÚsam2_hiera_det_modelé`   Úhidden_sizeé   Únum_attention_headsr   Únum_channelsNÚ
image_sizeÚpatch_kernel_sizeÚpatch_strideÚpatch_paddingÚquery_strideÚ+window_positional_embedding_background_sizeÚnum_query_pool_stagesÚblocks_per_stageÚembed_dim_per_stageÚnum_attention_heads_per_stageÚwindow_size_per_stageÚglobal_attention_blocksg      @Ú	mlp_ratioÚgeluÚ
hidden_actç�íµ ÷Æ°>Úlayer_norm_epsç{®Gáz”?Úinitializer_rangec                 ó  >• U R                   b  U R                   OSS/U l         U R                  b  U R                  OSS/U l        U R                  b  U R                  OSS/U l        U R                  b  U R                  OSS/U l        U R                  b  U R                  OSS/U l        U R
                  b  U R
                  OSS/U l        U R                  b  U R                  O/ SQU l        U R                  b  U R                  O/ SQU l        U R                  b  U R                  O/ SQU l        U R                  b  U R                  O/ S	QU l	        U R                  b  U R                  O/ S
QU l
        [        TU ]0  " S0 UD6  g )Né   é   é   r   r   )r   r   r*   r   )r   éÀ   é€  é   )r   r   r+   é   )r/   r+   é   r*   )é   r*   é	   © )r   r   r   r   r   r   r   r   r   r   r    ÚsuperÚ__post_init__©ÚselfÚkwargsÚ	__class__s     €Úh/home/mande/repo/quber/.venv/lib/python3.13/site-packages/transformers/models/sam2/configuration_sam2.pyr5   Ú Sam2HieraDetConfig.__post_init__J   s€  ø€ Ø-1¯_©_Ñ-H˜$Ÿ/š/ÈtÐUYÈlˆŒØ;?×;QÑ;QÑ;] ×!7Ò!7ÐdeÐghÐciˆÔØ15×1BÑ1BÑ1N˜D×-Ò-ÐUVÐXYÐTZˆÔØ37×3EÑ3EÑ3Q˜T×/Ò/ÐXYÐ[\ÐW]ˆÔØ15×1BÑ1BÑ1N˜D×-Ò-ÐUVÐXYÐTZˆÔð ×?Ñ?ÑKð ×<Ò<à�Q�ð 	Ô8ð
 :>×9NÑ9NÑ9Z × 5Ò 5Ò`lˆÔà(,×(@Ñ(@Ñ(LˆD×$Ò$ÒReð 	Ô ð 37×2TÑ2TÑ2`ˆD×.Ò.Òfrð 	Ô*ð +/×*DÑ*DÑ*PˆD×&Ò&ÒVcð 	Ô"ð -1×,HÑ,HÑ,TˆD×(Ò(ÒZcð 	Ô$ô 	‰ÒÑ' Ó'ó    )r   r   r    r   r   r   r   r   r   r   r   )"Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__Úbase_config_keyÚ
model_typer   ÚintÚ__annotations__r   r   r   Úlistr   r   r   r   r   r   r   r   r   r   r    r!   Úfloatr#   Ústrr%   r'   r5   Ú__static_attributes__Ú__classcell__©r9   s   @r:   r   r      s]  ø‡ ñð2 (€OØ'€Jà€K�ÓØ Ð˜Ó Ø€L�#ÓØ)-€J��d˜3‘i‘ $Ñ&Ó-Ø04Ð�s˜T #™Y‘¨Ñ-Ó4Ø+/€L�#˜˜S™	‘/ DÑ(Ó/Ø,0€M�3˜˜c™‘? TÑ)Ó0Ø+/€L�#˜˜S™	‘/ DÑ(Ó/ØDHÐ/°°c±¸TÑ1AÓHØ!"Ð˜3Ó"Ø)-Ð�d˜3‘i $Ñ&Ó-Ø,0Ð˜˜c™ TÑ)Ó0Ø6:Ð! 4¨¡9¨tÑ#3Ó:Ø.2Ð˜4 ™9 tÑ+Ó2Ø04Ð˜T #™Y¨Ñ-Ó4Ø€IˆuÓØ€J�ÓØ €N�EÓ Ø#Ð�uÓ#÷(ó (r<   r   c                   ó  ^ • \ rS rSr% SrSrSrS\0rSr	\
\-  S-  \S'   Sr\\   S-  \S'   Sr\S-  \S'   S	r\\S
'   Sr\\S'   Sr\\S'   Sr\\S'   Sr\\   S-  \S'   Sr\\S'   Sr\\S'   Sr\\S'   Sr\\S'   U 4S jrSrU =r$ )ÚSam2VisionConfigée   a›  
backbone_channel_list (`List[int]`, *optional*, defaults to `[768, 384, 192, 96]`):
    The list of channel dimensions for the backbone.
backbone_feature_sizes (`List[List[int]]`, *optional*, defaults to `[[256, 256], [128, 128], [64, 64]]`):
    The spatial sizes of the feature maps from the backbone.
fpn_hidden_size (`int`, *optional*, defaults to 256):
    The hidden dimension of the FPN.
fpn_kernel_size (`int`, *optional*, defaults to 1):
    The kernel size for the convolutions in the neck.
fpn_stride (`int`, *optional*, defaults to 1):
    The stride for the convolutions in the neck.
fpn_padding (`int`, *optional*, defaults to 0):
    The padding for the convolutions in the neck.
fpn_top_down_levels (`List[int]`, *optional*, defaults to `[2, 3]`):
    The levels for the top-down FPN connections.
num_feature_levels (`int`, *optional*, defaults to 3):
    The number of feature levels from the FPN to use.
Úvision_configÚsam2_vision_modelr   NÚbackbone_channel_listÚbackbone_feature_sizesé   Úfpn_hidden_sizer   Úfpn_kernel_sizeÚ
fpn_strider   Úfpn_paddingÚfpn_top_down_levelsr   Únum_feature_levelsr"   r#   r$   r%   r&   r'   c                 ó   >• U R                   c  / SQOU R                   U l         U R                  c  SS/SS/SS//OU R                  U l        U R                  c  SS/OU R                  U l        [        U R                  [
        5      (       aU  U R                  R                  SS5      U R                  S'   [        U R                  S      " S	0 U R                  D6U l        OU R                  c  [        5       U l        [        TU ](  " S	0 UD6  g )
N)r.   r-   r,   r   rS   é€   é@   r   r   rC   r   r3   )rQ   rR   rX   Ú
isinstancer   ÚdictÚgetr   r   r4   r5   r6   s     €r:   r5   ÚSam2VisionConfig.__post_init__Ž   sý   ø€ à#'×#=Ñ#=Ñ#EÓÈ4×KeÑKeð 	Ô"ð 37×2MÑ2MÑ2Uˆc�3ˆZ˜#˜s˜ b¨" XÑ.Ð[_×[vÑ[vð 	Ô#ð .2×-EÑ-EÑ-M A q¡6ÐSW×SkÑSkˆÔ ä�d×*Ñ*¬D×1Ñ1Ø15×1EÑ1E×1IÑ1IÈ,ÐXnÓ1oˆD× Ñ  Ñ.Ü#1°$×2FÑ2FÀ|Ñ2TÒ#UÑ#mÐX\×XlÑXlÑ#mˆDÕ Ø×!Ñ!Ñ)Ü#5Ó#7ˆDÔ ä‰ÒÑ' Ó'r<   )rQ   r   rR   rX   )r=   r>   r?   r@   rA   rB   rC   r	   Úsub_configsr   r^   r   rE   rQ   rF   rD   rR   rT   rU   rV   rW   rX   rY   r#   rH   r%   rG   r'   r5   rI   rJ   rK   s   @r:   rM   rM   e   sÊ   ø‡ ñð& &€OØ$€Jà˜:ð€Kð 7;€O�TÐ,Ñ,¨tÑ3Ó:Ø.2Ð˜4 ™9 tÑ+Ó2Ø*.Ð˜D 4™KÓ.Ø€O�SÓØ€O�SÓØ€J�ÓØ€K�ÓØ,0Ð˜˜c™ TÑ)Ó0ØÐ˜ÓØ€J�ÓØ €N�EÓ Ø#Ð�uÓ#÷(ó (r<   rM   c                   óÆ   • \ rS rSr% SrSrSr\\S'   Sr	\\
\   -  \\\4   -  \S'   Sr\\
\   -  \\\4   -  \S	'   Sr\\S
'   Sr\\S'   Sr\\S'   Sr\\S'   Sr\\S'   Srg)ÚSam2PromptEncoderConfigé    a=  
mask_input_channels (`int`, *optional*, defaults to 16):
    The number of channels to be fed to the `MaskDecoder` module.
num_point_embeddings (`int`, *optional*, defaults to 4):
    The number of point embeddings to be used.
scale (`float`, *optional*, defaults to 1):
    The scale factor for the prompt encoder.
Úprompt_encoder_configrS   r   r)   r   é   Ú
patch_sizeÚmask_input_channelsr+   Únum_point_embeddingsr"   r#   r$   r%   r   Úscaler3   N)r=   r>   r?   r@   rA   rB   r   rD   rE   r   rF   Útuplerg   rh   ri   r#   rH   r%   rG   rj   rI   r3   r<   r:   rc   rc       s‰   ‡ ñð .€Oà€K�ÓØ48€J��d˜3‘i‘ %¨¨S¨¡/Ñ1Ó8Ø46€J��d˜3‘i‘ %¨¨S¨¡/Ñ1Ó6Ø!Ð˜Ó!Ø !Ð˜#Ó!Ø€J�ÓØ €N�EÓ Ø€Eˆ3†Nr<   rc   c                   óÆ   • \ rS rSr% SrSrSr\\S'   Sr	\
\S'   Sr\\S	'   S
r\\S'   Sr\\S'   S
r\\S'   Sr\\S'   Sr\\S'   Sr\\S'   Sr\\S'   Sr\\S'   Sr\\S'   Srg)ÚSam2MaskDecoderConfigé¸   am  
mlp_dim (`int`, *optional*, defaults to 2048):
    The dimension of the MLP in the two-way transformer.
attention_downsample_rate (`int`, *optional*, defaults to 2):
    The downsample rate for the attention layers.
num_multimask_outputs (`int`, *optional*, defaults to 3):
    The number of multimask outputs.
iou_head_depth (`int`, *optional*, defaults to 3):
    The depth of the IoU head.
iou_head_hidden_dim (`int`, *optional*, defaults to 256):
    The hidden dimension of the IoU head.
dynamic_multimask_via_stability (`bool`, *optional*, defaults to `True`):
    Whether to use dynamic multimask via stability.
dynamic_multimask_stability_delta (`float`, *optional*, defaults to 0.05):
    The stability delta for the dynamic multimask.
dynamic_multimask_stability_thresh (`float`, *optional*, defaults to 0.98):
    The stability threshold for the dynamic multimask.
Úmask_decoder_configrS   r   r"   r#   i   Úmlp_dimr   Únum_hidden_layersr/   r   Úattention_downsample_rater   Únum_multimask_outputsÚiou_head_depthÚiou_head_hidden_dimTÚdynamic_multimask_via_stabilitygš™™™™™©?Ú!dynamic_multimask_stability_deltag\�Âõ(\ï?Ú"dynamic_multimask_stability_threshr3   N)r=   r>   r?   r@   rA   rB   r   rD   rE   r#   rH   rp   rq   r   rr   rs   rt   ru   rv   Úboolrw   rG   rx   rI   r3   r<   r:   rm   rm   ¸   sŽ   ‡ ñð& ,€Oà€K�ÓØ€J�ÓØ€GˆSÓØÐ�sÓØ Ð˜Ó Ø%&Ð˜sÓ&Ø!"Ð˜3Ó"Ø€N�CÓØ"Ð˜Ó"Ø,0Ð# TÓ0Ø/3Ð% uÓ3Ø04Ð&¨Ö4r<   rm   c                   óš   ^ • \ rS rSr% SrSr\\\S.r	Sr
\\-  S-  \S'   Sr\\-  S-  \S'   Sr\\-  S-  \S'   S	r\\S
'   U 4S jrSrU =r$ )Ú
Sam2ConfigéÞ   a  
prompt_encoder_config (Union[`dict`, `Sam2PromptEncoderConfig`], *optional*):
    Dictionary of configuration options used to initialize [`Sam2PromptEncoderConfig`].
mask_decoder_config (Union[`dict`, `Sam2MaskDecoderConfig`], *optional*):
    Dictionary of configuration options used to initialize [`Sam2MaskDecoderConfig`].

Example:

```python
>>> from transformers import (
...     Sam2VisionConfig,
...     Sam2PromptEncoderConfig,
...     Sam2MaskDecoderConfig,
...     Sam2Model,
... )

>>> # Initializing a Sam2Config with `"facebook/sam2.1_hiera_tiny"` style configuration
>>> configuration = Sam2Config()

>>> # Initializing a Sam2Model (with random weights) from the `"facebook/sam2.1_hiera_tiny"` style configuration
>>> model = Sam2Model(configuration)

>>> # Accessing the model configuration
>>> configuration = model.config

>>> # We can also initialize a Sam2Config from a Sam2VisionConfig, Sam2PromptEncoderConfig, and Sam2MaskDecoderConfig

>>> # Initializing SAM2 vision encoder, memory attention, and memory encoder configurations
>>> vision_config = Sam2VisionConfig()
>>> prompt_encoder_config = Sam2PromptEncoderConfig()
>>> mask_decoder_config = Sam2MaskDecoderConfig()

>>> config = Sam2Config(vision_config, prompt_encoder_config, mask_decoder_config)
```Úsam2)rO   re   ro   NrO   re   ro   r&   r'   c                 ó¦  >• [        U R                  [        5      (       aU  U R                  R                  SS5      U R                  S'   [        U R                  S      " S0 U R                  D6U l        O U R                  c  [        S   " 5       U l        [        U R
                  [        5      (       a  [        S0 U R
                  D6U l        OU R
                  c  [        5       U l        [        U R                  [        5      (       a  [        S0 U R                  D6U l        OU R                  c  [        5       U l        [        TU ](  " S0 UD6  g )NrC   rP   r3   )r]   rO   r^   r_   r   re   rc   ro   rm   r4   r5   r6   s     €r:   r5   ÚSam2Config.__post_init__  s  ø€ Ü�d×(Ñ(¬$×/Ñ/Ø/3×/AÑ/A×/EÑ/EÀlÐTgÓ/hˆD×Ñ˜|Ñ,Ü!/°×0BÑ0BÀ<Ñ0PÒ!QÑ!gÐTX×TfÑTfÑ!gˆDÕØ×ÑÑ'Ü!/Ð0CÒ!DÓ!FˆDÔä�d×0Ñ0´$×7Ñ7Ü)@Ñ)^À4×C]ÑC]Ñ)^ˆDÕ&Ø×'Ñ'Ñ/Ü)@Ó)BˆDÔ&ä�d×.Ñ.´×5Ñ5Ü'<Ñ'X¸t×?WÑ?WÑ'XˆDÕ$Ø×%Ñ%Ñ-Ü'<Ó'>ˆDÔ$ä‰ÒÑ' Ó'r<   )ro   re   rO   )r=   r>   r?   r@   rA   rC   r	   rc   rm   ra   rO   r^   r   rE   re   ro   r'   rG   r5   rI   rJ   rK   s   @r:   r{   r{   Þ   sx   ø‡ ñ!ðF €Jà#Ø!8Ø4ñ€Kð 59€M�4Ð*Ñ*¨TÑ1Ó8Ø<@Ð˜4Ð"2Ñ2°TÑ9Ó@Ø:>Ð˜Ð 0Ñ0°4Ñ7Ó>Ø#Ð�uÓ#÷(ó (r<   r{   )r{   r   rM   rc   rm   N)rA   Úhuggingface_hub.dataclassesr   Úconfiguration_utilsr   Úutilsr   Úautor   r	   r   rM   rc   rm   r{   Ú__all__r3   r<   r:   Ú<module>r…      sù   ðñ å .å 3Ý #ß -ñ Ð7Ñ8ØôI(Ð)ó I(ó ó 9ðI(ñX Ð7Ñ8Øô6(Ð'ó 6(ó ó 9ð6(ñr Ð7Ñ8ØôÐ.ó ó ó 9ðñ, Ð7Ñ8Øô!5Ð,ó !5ó ó 9ð!5ñH Ð7Ñ8ØôA(Ð!ó A(ó ó 9ðA(òH�r<   