ó
    �®žj<  ã                   ó  • S r SSKJr  SSKJrJrJrJrJr  SSK	J
r
JrJrJr  SSKJr  SSKJrJr  SSKJr   " S S	\5      r " S
 S\\5      r " S S\5      r " S S\5      r " S S\5      r\" 1 Sk5      r\S1-  r " S S\5      rg)a  Configuration models for docling's automatic speech recognition (ASR) backends.

This module defines the option classes that configure how audio is transcribed.
``InlineAsrOptions`` is the shared base for locally-run models, specialized by:

- ``InlineAsrNativeWhisperOptions``: OpenAI's native ``whisper`` (PyTorch), the
  default backend, supported on CPU and CUDA.
- ``InlineAsrMlxWhisperOptions``: ``mlx-whisper``, optimized for Apple Silicon.
- ``InlineAsrWhisperS2TOptions``: WhisperS2T (CTranslate2), an optional,
  experimental high-throughput backend installed via the ``format-audio`` extra.

The concrete model presets, the ``AsrModelType`` enum, and the hardware-based
auto-selection live in ``docling.datamodel.asr_model_specs``. The auto-selecting
``WHISPER_*`` presets (the ASR pipeline default) use MLX Whisper on Apple Silicon
when available and native Whisper otherwise. WhisperS2T is never auto-selected;
use an explicit ``WHISPER_*_S2T`` preset to opt in (the ``*_MLX`` and ``*_NATIVE``
presets likewise force those backends).
é    )ÚEnum)Ú	AnnotatedÚAnyÚLiteralÚOptionalÚUnion)ÚAnyUrlÚ	BaseModelÚFieldÚmodel_validator)Ú
deprecated)ÚAcceleratorDeviceÚAcceleratorOptions)ÚTransformersModelTypec                   ó6   • \ rS rSr% Sr\\\" SS94   \S'   Sr	g)ÚBaseAsrOptionsé$   z;Base configuration for automatic speech recognition models.zbType identifier for the ASR options. Used for discriminating between different ASR configurations.©ÚdescriptionÚkind© N)
Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__r   Ústrr   Ú__annotations__Ú__static_attributes__r   ó    Úi/home/mande/repo/quber/.venv/lib/python3.13/site-packages/docling/datamodel/pipeline_options_asr_model.pyr   r   $   s'   ‡ ÙEà
ØÙð8ñ	
ð	ñö r    r   c                   ó    • \ rS rSrSrSrSrSrg)ÚInferenceAsrFrameworké2   ÚmlxÚwhisperÚwhisper_s2tr   N)r   r   r   r   ÚMLXÚWHISPERÚWHISPER_S2Tr   r   r    r!   r#   r#   2   s   † Ø
€Cà€GØƒKr    r#   c                   ó¬  • \ rS rSr% SrSr\S   \S'   \\	\
" SSS/S94   \S	'   S
r\\\
" SS94   \S'   Sr\\\
" SS94   \S'   Sr\\\
" SS94   \S'   Sr\\\
" SS94   \S'   Sr\\\
" SS94   \S'   Sr\\\	   \
" SS94   \S'   \R,                  \R.                  \R0                  \R2                  /r\\\   \
" SS94   \S'   \S\	4S  j5       rS!rg)"ÚInlineAsrOptionsé9   z4Configuration for inline ASR models running locally.Úinline_model_optionsr   zwHuggingFace model repository ID for the ASR model. Must be a Whisper-compatible model for automatic speech recognition.zopenai/whisper-tinyzopenai/whisper-base©r   ÚexamplesÚrepo_idFzHEnable verbose logging output from the ASR model for debugging purposes.r   ÚverboseTz˜Generate timestamps for transcribed segments. When enabled, each transcribed segment includes start and end times for temporal alignment with the audio.Ú
timestampsg        z¶Sampling temperature for text generation. 0.0 uses greedy decoding (deterministic), higher values (e.g., 0.7-1.0) increase randomness. Recommended: 0.0 for consistent transcriptions.Útemperatureé   zŸMaximum number of tokens to generate per transcription segment. Limits output length to prevent runaway generation. Adjust based on expected transcript length.Úmax_new_tokensg      >@z±Maximum duration in seconds for each audio chunk processed by the model. Audio longer than this is split into chunks. Whisper models are typically trained on 30-second segments.Úmax_time_chunkNz¹PyTorch data type for model weights. Options: `float32`, `float16`, `bfloat16`. Lower precision (float16/bfloat16) reduces memory usage and increases speed. If None, uses model default.Útorch_dtypezHList of hardware accelerators supported by this ASR model configuration.Úsupported_devicesÚreturnc                 ó:   • U R                   R                  SS5      $ )NÚ/z--)r1   Úreplace©Úselfs    r!   Úrepo_cache_folderÚ"InlineAsrOptions.repo_cache_folder“   s   € à�|‰|×#Ñ# C¨Ó.Ð.r    r   )r   r   r   r   r   r   r   r   r   r   r   r2   Úboolr3   r4   Úfloatr6   Úintr7   r8   r   r   ÚCPUÚCUDAÚMPSÚXPUr9   ÚlistÚpropertyr@   r   r   r    r!   r,   r,   9   s¼  ‡ Ù>à,B€Dˆ'Ð(Ñ
)ÓBØØÙðMð ,Ð-BÐCñ	
ð	ñ	ó 	ð$ 	ð ˆYØÙðñ	
ð	ñó ð$ 	ð �	ØÙð5ñ	
ð	ñ	ó 	ð( 	ð �ØÙð"ñ	
ð	ñ
ó 
ð( 	ð �IØÙð7ñ	
ð	ñ	ó 	ð& 	ð �IØÙðNñ	
ð	ñ	ó 	ð( 	ð �Ø�‰Ùðñ	
ð	ñ
ó 
ð( 	×ÑØ×ÑØ×ÑØ×Ñð		ð �yØÐÑÙð!ñ	
ð	ñó ð ð/ 3ó /ó ó/r    r,   c                   ó8  • \ rS rSr% Sr\R                  r\\\	" SS94   \
S'   Sr\\\   \	" S/ SQS	94   \
S
'   \R                  \R                   /r\\\   \	" SS94   \
S'   Sr\\\	" SS94   \
S'   Sr\\\   \	" SS94   \
S'   Sr\\\   \	" SS94   \
S'   Srg)ÚInlineAsrNativeWhisperOptionsé˜   z4Configuration for native Whisper ASR implementation.zXInference framework for ASR. Uses native Whisper implementation for optimal performance.r   Úinference_frameworkNúöLanguage code for transcription. When `None` (the default), Whisper auto-detects the language from the first 30 seconds of audio. Specifying the correct language skips detection and improves accuracy. Use ISO 639-1 codes (e.g., `en`, `es`, `fr`).©ÚenÚesÚfrÚder/   ÚlanguagezNHardware accelerators supported by native Whisper. Supports CPU and CUDA only.r9   TúŽGenerate word-level timestamps in addition to segment timestamps. Provides fine-grained temporal alignment for each word in the transcription.Úword_timestampszÉBeam size for beam search decoding. When `None` (the default), Whisper uses greedy decoding. Set to a positive integer (e.g. `5`) to enable beam search, which can improve accuracy at the cost of speed.Ú	beam_sizea  Whether to feed the model's previous output as a prompt for the next window. When `None` (the default), Whisper's default of `True` is used. Set to `False` to reduce repetition loops on long or difficult audio, at the risk of inconsistent text across windows.Úcondition_on_previous_textr   )r   r   r   r   r   r#   r)   rN   r   r   r   rU   r   r   r   rE   rF   r9   rI   rW   rB   rX   rD   rY   r   r   r    r!   rL   rL   ˜   s:  ‡ Ù>ð 	×%Ñ%ð ˜ØÙð:ñ	
ð	ñó &ð* 	ð ˆiØ�‰Ùðò .ñ		
ð
	ñó ð, 	×ÑØ×Ñð	ð �yØÐÑÙð%ñ	
ð	ñó ð* 	ð �YØÙð-ñ	
ð	ñ	ó 	ð& 	ð ˆyØ�‰ÙðZñ	
ð	ñ	ó 	ð( 	ð  	Ø�‰ÙðCñ	
ð	ñ
!ö 
r    rL   c                   ó\  • \ rS rSr% Sr\R                  r\\\	" SS94   \
S'   Sr\\\   \	" S/ SQS	94   \
S
'   Sr\\\	" SSS/S	94   \
S'   \R                   /r\\\   \	" SS94   \
S'   Sr\\\	" SS94   \
S'   Sr\\\	" SS94   \
S'   Sr\\\	" SS94   \
S'   Sr\\\	" SS94   \
S'   Srg)ÚInlineAsrMlxWhisperOptionséÞ   z€MLX Whisper options for Apple Silicon optimization.

Uses mlx-whisper library for efficient inference on Apple Silicon devices.
z\Inference framework for ASR. Uses MLX for optimized performance on Apple Silicon (M1/M2/M3).r   rN   NrO   rP   r/   rU   Ú
transcribez“ASR task type. `transcribe` converts speech to text in the same language. `translate` converts speech to English text regardless of input language.Ú	translateÚtaskzWHardware accelerators supported by MLX Whisper. Optimized for Apple Silicon (MPS) only.r9   TrV   rW   g333333ã?zÃThreshold for detecting speech vs. silence. Segments with no-speech probability above this threshold are considered silent. Range: 0.0-1.0. Higher values are more aggressive in filtering silence.Úno_speech_thresholdg      ð¿z½Log probability threshold for filtering low-confidence transcriptions. Segments with average log probability below this threshold are filtered out. More negative values are more permissive.Úlogprob_thresholdg333333@z¹Compression ratio threshold for detecting repetitive or low-quality transcriptions. Segments with compression ratio above this threshold are filtered. Higher values are more permissive.Úcompression_ratio_thresholdr   )r   r   r   r   r   r#   r(   rN   r   r   r   rU   r   r   r_   r   rG   r9   rI   rW   rB   r`   rC   ra   rb   r   r   r    r!   r[   r[   Þ   s‡  ‡ ñð 	×!Ñ!ð ˜ØÙð;ñ	
ð	ñó "ð* 	ð ˆiØ�‰Ùðò .ñ		
ð
	ñó ð. 	ð 	ˆ)ØÙð0ð # KÐ0ñ	
ð	ñ
ó 
ð& 
×	Ñ	Ðð �yØÐÑÙð,ñ	
ð	ñó  ð$ 	ð �YØÙð-ñ	
ð	ñ	ó 	ð( 	ð ˜ØÙð%ñ	
ð	ñ
ó 
ð* 	ð �yØÙðñ	
ð	ñ
ó 
ð* 	ð   ØÙðñ	
ð	ñ
"ö 
r    r[   >   úbase.enútiny.enúsmall.enú	medium.enúdistil-large-v3údistil-small.enúdistil-medium.enúdistil-large-v3.5zlarge-v3-turboc                   óø  • \ rS rSr% Sr\R                  r\\\	" SS94   \
S'   Sr\\\	" S/ SQS	94   \
S
'   Sr\\\	" SSS/S	94   \
S'   Sr\\\   \	" S/ SQS	94   \
S'   Sr\\\	" SS94   \
S'   Sr\\\	" SS94   \
S'   Sr\\\	" SS94   \
S'   \" 5       R,                  r\\\	" SS94   \
S'   Sr\\\   \	" SS94   \
S '   \R2                  \R4                  /r\\\   \	" S!S94   \
S"'   \" S#S$9S'S% j5       rS&rg)(ÚInlineAsrWhisperS2TOptionsiM  zÅConfiguration for WhisperS2T (CTranslate2-based) high-speed ASR.

Uses whisper_s2t library with CTranslate2 backend for fast inference
on CPU and CUDA devices. Requires whisper-s2t-reborn package.
ziInference framework for ASR. Uses WhisperS2T with CTranslate2 backend for optimized high-speed inference.r   rN   rQ   zNLanguage code for transcription. Use ISO 639-1 codes (e.g., `en`, `es`, `fr`).)rQ   rR   rS   rT   ÚjaÚzhr/   rU   r]   zvASR task type. `transcribe` converts speech to text in the same language. `translate` converts speech to English text.r^   r_   Úfloat16z²Computation precision for CTranslate2. Options: `float32`, `float16`, `bfloat16`. Lower precision increases speed and reduces memory. bfloat16 requires compute capability >= 8.6.)Úfloat32ro   Úbfloat16r8   é   ziNumber of audio segments to process in parallel. Higher values increase throughput but require more VRAM.Ú
batch_sizeé   z�Beam size for beam search decoding. 1 = greedy decoding (fastest), higher values (e.g., 5) may improve accuracy at cost of speed.rX   FzeGenerate word-level timestamps. Requires an additional alignment model and increases processing time.rW   zBNumber of CPU threads for inference. Only used when device is CPU.Únum_threadsNztOptional text prompt to condition the transcription style or provide context. Useful for domain-specific vocabulary.Úinitial_promptz.Hardware accelerators supported by WhisperS2T.r9   Úafter)Úmodec                 ó  • U R                   [        ;   a6  U R                  S:w  a&  [        SU R                    SU R                   S35      eU R                   [        ;   a)  U R
                  S:X  a  [        SU R                    S35      eU $ )NrQ   zModel `z1` is English-only and does not support language `zj`. Set language='en' or choose a multilingual model (e.g., `tiny`, `base`, `small`, `medium`, `large-v3`).r^   z�` does not support the `translate` task. Set task='transcribe' or choose a multilingual model with translation capability (e.g., `large-v3`).)r1   Ú_ENGLISH_ONLY_S2T_REPOSrU   Ú
ValueErrorÚ_NO_TRANSLATE_S2T_REPOSr_   r>   s    r!   Ú_validate_repo_capabilitiesÚ6InlineAsrWhisperS2TOptions._validate_repo_capabilities°  s�   € ð �<‰<Ô2Ó2°t·}±}ÈÓ7LÜØ˜$Ÿ,™,˜ð (Ø!Ÿ]™]˜Oð ,ð óð ð �<‰<Ô2Ó2°t·y±yÀKÓ7OÜØ˜$Ÿ,™,˜ð (=ð >óð ð
 ˆr    r   )r:   rl   ) r   r   r   r   r   r#   r*   rN   r   r   r   rU   r   r_   r8   r   rs   rD   rX   rW   rB   r   ru   rv   r   rE   rF   r9   rI   r   r}   r   r   r    r!   rl   rl   M  s  ‡ ñð 	×)Ñ)ð ˜ØÙð>ñ	
ð	ñó *ð$ 	ð ˆiØÙð,ò :ñ	
ð	ñ	ó 	ð& 	ð 	ˆ)ØÙðNð # KÐ0ñ	
ð	ñ	ó 	ð( 	ð �Ø�‰ÙðOò 8ñ	
ð	ñ
ó 
ð& 	
ð �	ØÙð=ñ	
ð	ñó 
ð" 	
ð ˆyØÙðQñ	
ð	ñó 
ð" 	ð �YØÙð7ñ	
ð	ñó ñ  	Ó×(Ñ(ð �ØÙàTñ	
ð	ñó )ð  	ð �IØ�‰ÙðJñ	
ð	ñó ð 	×ÑØ×Ñð	ð �yØÐÑÙÐKÑMð	Oñó ñ ˜'Ñ"óó #ór    rl   N)r   Úenumr   Útypingr   r   r   r   r   Úpydanticr	   r
   r   r   Útyping_extensionsr   Ú%docling.datamodel.accelerator_optionsr   r   Ú,docling.datamodel.pipeline_options_vlm_modelr   r   r   r#   r,   rL   r[   Ú	frozensetrz   r|   rl   r   r    r!   Ú<module>r†      s¡   ðñõ& ß ;Õ ;ç >Ó >Ý (ç Wõô�Yô ô ˜C ô  ô\/�~ô \/ô~CÐ$4ô CôLZÐ!1ô Zñ| $ò	óÐ ð 2Ð5EÐ4FÑFÐ ôtÐ!1õ tr    