ó
    �®žj,:  ã                   ó¦  • S SK Jr  S SKJrJrJrJrJrJr  S SK	J
r
  S SKJrJrJrJr  S SKJr   S SKJr  S S	KJr  S S
KJr  \(       a  S SK	J
r
  S SKJr   " S S\5      r " S S\\5      r " S S\\5      r " S S\\5      r " S S\\5      r  " S S\5      r!\" S5       " S S\!5      5       r" " S S\5      r#g! \ a     " S S5      r N™f = f)é    )ÚEnum)ÚTYPE_CHECKINGÚ	AnnotatedÚAnyÚLiteralÚOptionalÚUnion)ÚSegmentedPage)ÚAnyUrlÚ	BaseModelÚ
ConfigDictÚField)Ú
deprecated)ÚStoppingCriteriac                   ó   • \ rS rSrSrg)r   é   © N©Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__static_attributes__r   ó    Úi/home/mande/repo/quber/.venv/lib/python3.13/site-packages/docling/datamodel/pipeline_options_vlm_model.pyr   r      s   † Úr   r   )ÚAcceleratorDevice)ÚGenerationStopper)ÚPagec                   óô   • \ rS rSr% Sr\\\" SS94   \S'   \\\" SS94   \S'   Sr	\\
\" S	S94   \S
'   Sr\\\   \" SS94   \S'   Sr\\
\" SS94   \S'   SS.S\S   S\S   S\4S jjrS\S\4S jrSrg)ÚBaseVlmOptionsé   z.Base configuration for vision-language models.zbType identifier for the VLM options. Used for discriminating between different VLM configurations.©ÚdescriptionÚkindzbPrompt template for the vision-language model. Guides the model's output format and content focus.Úpromptg       @zŸScaling factor for image resolution before processing. Higher values provide more detail but increase processing time and memory usage. Range: 0.5-4.0 typical.ÚscaleNzœMaximum image dimension (width or height) in pixels. Images larger than this are resized while maintaining aspect ratio. If None, no size limit is enforced.Úmax_sizeg        z¯Sampling temperature for text generation. 0.0 uses greedy decoding (deterministic), higher values (e.g., 0.7-1.0) increase randomness. Recommended: 0.0 for consistent outputs.Útemperature)Ú_internal_pageÚpager
   r)   r   Úreturnc                ó   • U R                   $ )a  Build the prompt for VLM inference.

Args:
    page: The parsed/segmented page to process.
    _internal_page: Internal parameter for experimental layout-aware pipelines.
        Do not rely on this in user code - subject to change.

Returns:
    The formatted prompt string.
)r%   )Úselfr*   r)   s      r   Úbuild_promptÚBaseVlmOptions.build_promptR   s   € ð  �{‰{Ðr   Útextc                 ó   • U$ )Nr   )r-   r0   s     r   Údecode_responseÚBaseVlmOptions.decode_responsed   s   € Øˆr   r   )r   r   r   r   Ú__doc__r   Ústrr   Ú__annotations__r&   Úfloatr'   r   Úintr(   r.   r2   r   r   r   r   r    r       s  ‡ Ù8à
ØÙð8ñ	
ð	ñó ð ØÙð;ñ	
ð	ñó ð$ 	ð 
ˆ9ØÙð8ñ	
ð	ñ	ó 	ð& 	ð ˆiØ�‰Ùð6ñ	
ð	ñ	ó 	ð& 	ð �ØÙðPñ	
ð	ñ	ó 	ð ,0ò	à�Ñ'ðð ! Ñ(ð	ð
 
õð$ Cð ¨C÷ r   r    c                   ó<   • \ rS rSrSrSrSrSrSrSr	Sr
S	rS
rSrSrg)ÚResponseFormatéh   ÚdoctagsÚdoclangÚmarkdownÚdeepseekocr_markdownÚunlimited_ocr_markdownÚhtmlÚotslÚ	plaintextÚchandra_htmlÚ	dots_jsonr   N)r   r   r   r   ÚDOCTAGSÚDOCLANGÚMARKDOWNÚDEEPSEEKOCR_MARKDOWNÚUNLIMITED_OCR_MARKDOWNÚHTMLÚOTSLÚ	PLAINTEXTÚCHANDRA_HTMLÚ	DOTS_JSONr   r   r   r   r:   r:   h   s6   † Ø€GØ€GØ€HØ1ÐØ5ÐØ€DØ€DØ€IØ!€LØƒIr   r:   c                   ó    • \ rS rSrSrSrSrSrg)ÚInferenceFrameworkéu   ÚmlxÚtransformersÚvllmr   N)r   r   r   r   ÚMLXÚTRANSFORMERSÚVLLMr   r   r   r   rQ   rQ   u   s   † Ø
€CØ!€LØƒDr   rQ   c                   ó    • \ rS rSrSrSrSrSrg)ÚTransformersModelTypeé{   Ú	automodelzautomodel-causallmzautomodel-imagetexttotextr   N)r   r   r   r   Ú	AUTOMODELÚAUTOMODEL_CAUSALLMÚAUTOMODEL_IMAGETEXTTOTEXTr   r   r   r   rZ   rZ   {   s   † Ø€IØ-ÐØ ;Ór   rZ   c                   ó    • \ rS rSrSrSrSrSrg)ÚTransformersPromptStyleé�   ÚchatÚrawÚnoner   N)r   r   r   r   ÚCHATÚRAWÚNONEr   r   r   r   ra   ra   �   s   † Ø€DØ
€CØƒDr   ra   c                   óˆ  • \ rS rSr% Sr\" SS9rSr\S   \	S'   \
\\" SSS	/S
94   \	S'   Sr\
\\" SSS/S
94   \	S'   Sr\
\\" SS94   \	S'   Sr\
\\" SS94   \	S'   Sr\
\\" SS94   \	S'   Sr\
\\" SS94   \	S'   \
\\" SS94   \	S'   \R,                  r\
\\" SS94   \	S'   \R2                  r\
\\" SS94   \	S '   \
\\" S!S94   \	S"'   S#r\
\\   \" S$S94   \	S%'   \R>                  \R@                  \RB                  \RD                  /r#\
\$\   \" S&S94   \	S''   / r%\
\$\   \" S(S94   \	S)'   / r&\
\$\'\(\)4      \" S*S94   \	S+'   0 r*\
\+\\,4   \" S,S94   \	S-'   0 r-\
\+\\,4   \" S.S94   \	S/'   Sr.\
\\" S0S94   \	S1'   S2r/\
\0\" S3S94   \	S4'   Sr1\
\\" S5S94   \	S6'   Sr2\
\\" S7S94   \	S8'   \3S9\4S: j5       r4S;r5g#)<ÚInlineVlmOptionsé‡   z@Configuration for inline vision-language models running locally.T©Úarbitrary_types_allowedÚinline_model_optionsr$   z€HuggingFace model repository ID for the vision-language model. Must be a model capable of processing images and generating text.zQwen/Qwen2-VL-2B-Instructz!ibm-granite/granite-vision-3.3-2b©r#   ÚexamplesÚrepo_idÚmainz‚Git revision (branch, tag, or commit hash) of the model repository. Allows pinning to specific model versions for reproducibility.zv1.0.0ÚrevisionFz«Allow execution of custom code from the model repository. Required for some models with custom architectures. Enable only for trusted sources due to security implications.r"   Útrust_remote_codez§Load model weights in 8-bit precision using bitsandbytes quantization. Reduces memory usage by ~50% with minimal accuracy loss. Requires bitsandbytes library and CUDA.Úload_in_8bitg      @zÀThreshold for LLM.int8() quantization outlier detection. Values with magnitude above this threshold are kept in float16 for accuracy. Lower values increase quantization but may reduce quality.Úllm_int8_thresholdz¡Indicates if the model is pre-quantized (e.g., GGUF, AWQ). When True, skips runtime quantization. Use for models already quantized during training or conversion.Ú	quantizedzˆInference framework for running the VLM. Options: `transformers` (HuggingFace), `mlx` (Apple Silicon), `vllm` (high-throughput serving).Úinference_frameworkzÑHuggingFace Transformers model class to use. Options: `automodel` (auto-detect), `automodel-vision2seq` (vision-to-sequence), `automodel-causallm` (causal LM), `automodel-imagetexttotext` (image+text to text).Útransformers_model_typez¤Prompt formatting style for Transformers models. Options: `chat` (chat template), `raw` (raw text), `none` (no formatting). Use `chat` for instruction-tuned models.Útransformers_prompt_stylez»Expected output format from the VLM. Options: `doctags` (structured tags), `doclang` (Doclang XML), `markdown`, `html`, `otsl` (table structure), `plaintext`. Guides model output parsing.Úresponse_formatNz PyTorch data type for model weights. Options: `float32`, `float16`, `bfloat16`. Lower precision reduces memory and increases speed. If None, uses model default.Útorch_dtypezBList of hardware accelerators supported by this VLM configuration.Úsupported_deviceszŽList of strings that trigger generation stopping when encountered. Used to prevent the model from generating beyond desired output boundaries.Ústop_stringsz Custom stopping criteria objects for fine-grained control over generation termination. Allows implementing complex stopping logic beyond simple string matching.Úcustom_stopping_criteriazžAdditional generation configuration parameters passed to the model. Overrides or extends default generation settings (e.g., top_p, top_k, repetition_penalty).Úextra_generation_configz�Additional keyword arguments passed to the image processor. Used for model-specific preprocessing options not covered by standard parameters.Úextra_processor_kwargszµEnable key-value caching for transformer attention. Significantly speeds up generation by caching attention computations. Disable only for debugging or memory-constrained scenarios.Úuse_kv_cachei   z–Maximum number of tokens to generate. Limits output length to prevent runaway generation. Adjust based on expected output size and memory constraints.Úmax_new_tokensz’Track and store generated tokens during inference. Useful for debugging, analysis, or implementing custom post-processing. Increases memory usage.Útrack_generated_tokensz‚Track and store the input prompt sent to the model. Useful for debugging, logging, or auditing. May contain sensitive information.Útrack_input_promptr+   c                 ó:   • U R                   R                  SS5      $ )NÚ/z--)rq   Úreplace)r-   s    r   Úrepo_cache_folderÚ"InlineVlmOptions.repo_cache_folder^  s   € à�|‰|×#Ñ# C¨Ó.Ð.r   r   )6r   r   r   r   r4   r   Úmodel_configr$   r   r6   r   r5   r   rs   rt   Úboolru   rv   r7   rw   rQ   rZ   r]   ry   ra   rf   rz   r:   r|   r   r   ÚCPUÚCUDAÚMPSÚXPUr}   Úlistr~   r   r	   r   r   r€   Údictr   r�   r‚   rƒ   r8   r„   r…   Úpropertyr‰   r   r   r   r   rj   rj   ‡   sù  ‡ ÙJá°dÑ;€LØ,B€Dˆ'Ð(Ñ
)ÓBØØÙð#ð 2Ð3VÐWñ	
ð	ñ
ó 
ð* 	ð ˆiØÙð#ð ˜hÐ'ñ	
ð	ñ
ó 
ð( 	ð �yØÙðIñ	
ð	ñ	ó 	ð& 	ð �)ØÙðIñ	
ð	ñ	ó 	ð( 	ð ˜	ØÙð&ñ	
ð	ñ
ó 
ð( 	ð ˆyØÙð;ñ	
ð	ñ	ó 	ð #ØÙð-ñ	
ð	ñ	ó 	ð( 	×'Ñ'ð ˜YØÙðDñ	
ð	ñ
ó 
(ð( 	 ×$Ñ$ð ˜yØÙðHñ	
ð	ñ	 ó 	%ð ØÙð"ñ	
ð	ñ
ó 
ð( 	ð �Ø�‰Ùð@ñ	
ð	ñ	ó 	ð$ 	×ÑØ×ÑØ×ÑØ×Ñð		ð �yØÐÑÙàTñ	
ð	ñó ð, 	ð �)ØˆS‰	Ùð-ñ	
ð	ñ	ó 	ð& 	ð ˜iØˆUÐ#Ð%6Ð6Ñ7Ñ8Ùð@ñ	
ð	ñ	ó 	ð& 	ð ˜YØˆS�#ˆX‰Ùð5ñ	
ð	ñ	ó 	ð& 	ð ˜IØˆS�#ˆX‰Ùð'ñ	
ð	ñ	ó 	ð( 	ð �)ØÙð0ñ	
ð	ñ
ó 
ð( 	ð �IØÙð/ñ	
ð	ñ	ó 	ð& 	ð ˜IØÙð*ñ	
ð	ñ	ó 	ð& 	ð ˜	ØÙðñ	
ð	ñ	ó 	ð ð/ 3ó /ó ó/r   rj   zUse InlineVlmOptions instead.c                   ó   • \ rS rSrSrg)ÚHuggingFaceVlmOptionsic  r   Nr   r   r   r   r•   r•   c  s   † âr   r•   c                   óŠ  • \ rS rSr% Sr\" SS9rSr\S   \	S'   \
" S5      r\\
\" SS	94   \	S
'   0 r\\\\4   \" SSS0/S94   \	S'   0 r\\\\4   \" SS	94   \	S'   Sr\\\" SS	94   \	S'   Sr\\\" SS	94   \	S'   \\\" SS	94   \	S'   / r\\\   \" SS	94   \	S'   / r\\\   \" SS	94   \	S'   Sr\\\" SS	94   \	S '   S!rg")#ÚApiVlmOptionsih  z;Configuration for API-based vision-language model services.Trl   Úapi_model_optionsr$   z*http://localhost:11434/v1/chat/completionsz®API endpoint URL for VLM service. Must be OpenAI-compatible chat completions endpoint. Default points to local Ollama server; update for cloud services or custom deployments.r"   ÚurlzoHTTP headers to include in API requests. Use for authentication or custom headers required by your API service.ÚAuthorizationzBearer TOKENro   Úheadersz‰Additional query parameters to include in API requests. Service-specific parameters for customizing API behavior beyond standard options.Úparamsg      N@z”Maximum time in seconds to wait for API response before timing out. Increase for slow networks or complex vision tasks. Recommended: 30-120 seconds.Útimeouté   z¡Number of concurrent API requests allowed. Higher values improve throughput but may hit API rate limits. Adjust based on API service quotas and network capacity.Úconcurrencyz»Expected output format from the VLM API. Options: `doctags` (structured tags), `doclang` (Doclang XML), `markdown`, `html`, `otsl` (table structure), `plaintext`. Guides response parsing.r{   z•List of strings that trigger generation stopping when encountered. Sent to API to prevent the model from generating beyond desired output boundaries.r~   z™Custom stopping criteria objects for client-side generation control. Applied after receiving API responses for additional filtering or termination logic.r   Fz€Track and store the input prompt sent to the API. Useful for debugging, logging, or auditing. May contain sensitive information.r…   r   N)r   r   r   r   r4   r   r‹   r$   r   r6   r   r™   r   r   r›   r’   r5   rœ   r   r�   r7   rŸ   r8   r:   r~   r‘   r   r   r…   rŒ   r   r   r   r   r—   r—   h  sÊ  ‡ ÙEá°dÑ;€LØ)<€Dˆ'Ð%Ñ
&Ó<ñ 	Ð;Ó<ð ˆØÙðKñ	
ð	ñ	
ó 	=ð& 	ð ˆYØˆS�#ˆX‰ÙðQð '¨Ð7Ð8ñ	
ð	ñ	ó 	ð& 	ð ˆIØˆS�#ˆX‰Ùð+ñ	
ð	ñ	ó 	ð& 	ð ˆYØÙð6ñ	
ð	ñ	ó 	ð& 	
ð �ØÙð>ñ	
ð	ñ	ó 	
ð ØÙð$ñ	
ð	ñ
ó 
ð( 	ð �)ØˆS‰	Ùð4ñ	
ð	ñ	ó 	ð& 	ð ˜iØÐÑÙð2ñ	
ð	ñ	ó 	ð& 	ð ˜	ØÙðñ	
ð	ñ	ö 	r   r—   N)$Úenumr   Útypingr   r   r   r   r   r	   Údocling_core.types.doc.pager
   Úpydanticr   r   r   r   Útyping_extensionsr   rT   r   ÚImportErrorÚ%docling.datamodel.accelerator_optionsr   Ú%docling.models.utils.generation_utilsr   Údocling.datamodel.base_modelsr   r    r5   r:   rQ   rZ   ra   rj   r•   r—   r   r   r   Ú<module>r©      sØ   ðõ ß J× Jå 5ß 9Ó 9Ý (ðÝ-õ DÝ CæÝ9å2ôG�Yô GôT
�S˜$ô 
ô˜˜dô ô<˜C ô <ô˜c 4ô ôY/�~ô Y/ñx Ð+Ó,ô	Ð,ó 	ó -ð	ô_�Nõ _øðw
 ó ÷ó ð	ús   °B= Â=CÃC