ó
    qyüi!  ã                   ól   • S SK Jr  SSKJr  SSKJr  SSKJr  \" SS9\ " S S	\5      5       5       rS	/r	g
)é    )Ústricté   )ÚPreTrainedConfig)ÚRopeParameters)Úauto_docstringzUsefulSensors/moonshine-tiny)Ú
checkpointc                   óØ  ^ • \ rS rSr% SrSrS/rSSSSS	.rS
r\	\
S'   Sr\	\
S'   Sr\	\
S'   Sr\	\
S'   Sr\	\
S'   Sr\	\
S'   Sr\	\
S'   Sr\	S-  \
S'   Sr\	S-  \
S'   Sr\	S-  \
S'   Sr\\
S'   Sr\\
S'   Sr\	\
S'   Sr\\
S'   Sr\	\
S'   S r\\
S!'   Sr\\-  S-  \
S"'   S r \\
S#'   S$r!\\
S%'   S&r"\\	-  \
S''   Sr#\	S-  \
S('   S)r$\	\%\	   -  S-  \
S*'   Sr&\	S-  \
S+'   S r'\\
S,'   U 4S- jr(S.r)U =r*$ )/ÚMoonshineConfigé   a	  
encoder_num_key_value_heads (`int`, *optional*):
    This is the number of key_value heads that should be used to implement Grouped Query Attention. If
    `encoder_num_key_value_heads=encoder_num_attention_heads`, the model will use Multi Head Attention (MHA), if
    `encoder_num_key_value_heads=1` the model will use Multi Query Attention (MQA) otherwise GQA is used. When
    converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed
    by meanpooling all the original heads within that group. For more details, check out [this
    paper](https://huggingface.co/papers/2305.13245). If it is not specified, will default to
    `num_attention_heads`.
decoder_num_key_value_heads (`int`, *optional*):
    This is the number of key_value heads that should be used to implement Grouped Query Attention. If
    `decoder_num_key_value_heads=decoder_num_attention_heads`, the model will use Multi Head Attention (MHA), if
    `decoder_num_key_value_heads=1` the model will use Multi Query Attention (MQA) otherwise GQA is used. When
    converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed
    by meanpooling all the original heads within that group. For more details, check out [this
    paper](https://huggingface.co/papers/2305.13245). If it is not specified, will default to
    `decoder_num_attention_heads`.
pad_head_dim_to_multiple_of (`int`, *optional*):
    Pad head dimension in encoder and decoder to the next multiple of this value. Necessary for using certain
    optimized attention implementations.
encoder_hidden_act (`str` or `function`, *optional*, defaults to `"gelu"`):
    The non-linear activation function (function or string) in the encoder.
decoder_hidden_act (`str` or `function`, *optional*, defaults to `"silu"`):
    The non-linear activation function (function or string) in the decoder.

Example:

```python
>>> from transformers import MoonshineModel, MoonshineConfig

>>> # Initializing a Moonshine style configuration
>>> configuration = MoonshineConfig().from_pretrained("UsefulSensors/moonshine-tiny")

>>> # Initializing a model from the configuration
>>> model = MoonshineModel(configuration)

>>> # Accessing the model configuration
>>> configuration = model.config
```Ú	moonshineÚpast_key_valuesÚdecoder_num_key_value_headsÚdecoder_num_attention_headsÚdecoder_num_hidden_layersÚdecoder_hidden_act)Únum_key_value_headsÚnum_attention_headsÚnum_hidden_layersÚ
hidden_acti €  Ú
vocab_sizei   Úhidden_sizei€  Úintermediate_sizeé   Úencoder_num_hidden_layersé   Úencoder_num_attention_headsNÚencoder_num_key_value_headsÚpad_head_dim_to_multiple_ofÚgeluÚencoder_hidden_actÚsilui   Úmax_position_embeddingsg{®Gáz”?Úinitializer_rangeé   Údecoder_start_token_idTÚ	use_cacheÚrope_parametersÚis_encoder_decoderFÚattention_biasg        Úattention_dropoutÚbos_token_idé   Úeos_token_idÚpad_token_idÚtie_word_embeddingsc                 óÂ   >• U R                   c  U R                  U l         U R                  c  U R                  U l        UR	                  SS5        [
        TU ]  " S0 UD6  g )NÚpartial_rotary_factorgÍÌÌÌÌÌì?© )r   r   r   r   Ú
setdefaultÚsuperÚ__post_init__)ÚselfÚkwargsÚ	__class__s     €Úr/home/mande/repo/quber/.venv/lib/python3.13/site-packages/transformers/models/moonshine/configuration_moonshine.pyr5   ÚMoonshineConfig.__post_init__i   sX   ø€ Ø×+Ñ+Ñ3Ø/3×/OÑ/OˆDÔ,à×+Ñ+Ñ3Ø/3×/OÑ/OˆDÔ,à×ÑÐ1°3Ô7Ü‰ÒÑ' Ó'ó    )r   r   )+Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__Ú
model_typeÚkeys_to_ignore_at_inferenceÚattribute_mapr   ÚintÚ__annotations__r   r   r   r   r   r   r   r   r   r    Ústrr   r"   r#   Úfloatr%   r&   Úboolr'   r   Údictr(   r)   r*   r+   r-   Úlistr.   r/   r5   Ú__static_attributes__Ú__classcell__)r8   s   @r9   r
   r
      sg  ø‡ ñ&ðP €JØ#4Ð"5Ðà<Ø<Ø8Ø*ñ	€Mð €J�ÓØ€K�ÓØ!Ð�sÓ!Ø%&Ð˜sÓ&Ø%&Ð˜sÓ&Ø'(Ð Ó(Ø'(Ð Ó(Ø.2Ð  t¡Ó2Ø.2Ð  t¡Ó2Ø.2Ð  t¡Ó2Ø$Ð˜Ó$Ø$Ð˜Ó$Ø#&Ð˜SÓ&Ø#Ð�uÓ#Ø"#Ð˜CÓ#Ø€IˆtÓØ48€O�^ dÑ*¨TÑ1Ó8Ø#Ð˜Ó#Ø €N�DÓ Ø%(Ð�u˜s‘{Ó(Ø €L�#˜‘*Ó Ø+,€L�#˜˜S™	‘/ DÑ(Ó,Ø#€L�#˜‘*Ó#Ø $Ð˜Ó$÷(ó (r;   r
   N)
Úhuggingface_hub.dataclassesr   Úconfiguration_utilsr   Úmodeling_rope_utilsr   Úutilsr   r
   Ú__all__r2   r;   r9   Ú<module>rR      sK   ðõ* /å 3Ý 1Ý #ñ Ð9Ñ:ØôS(Ð&ó S(ó ó ;ðS(ðl Ð
�r;   