ó
    �®žj²L ã                   óâ  • S SK r S SKrS SKrS SKJr  S SKJr  S SKJr  S SKJ	r	J
r
JrJr  S SKJr  S SKJr  S SKJrJrJrJrJrJrJrJr  S S	KJr  S S
KJrJrJr  S SK J!r!J"r"  S SK#J$r$J%r%  S SK&J'r'  S SK(J)r)  S SK*J+r+J,r,J-r-J.r.J/r/J0r0J1r1  S SK2J3r3  S SK4J5r5  S SK6J7r7  S SK8J9r9J:r:J;r;J<r<  S SK=J>r>J?r?J@r@JArA  S SKBJCrC  S SKDJErEJFrGJHrIJJrJJKrLJMrNJOrO  S SKPJQrQ  S SKRJSrS  \ R¨                  " \U5      rV " S S\5      rW " S S\X\5      rY " S S\X\5      rZ " S S \W5      r[ " S! S"\[5      r\ " S# S$\[5      r] " S% S&\[5      r^ " S' S(\W5      r_ " S) S*\_5      r` " S+ S,\_5      ra " S- S.\_5      rb " S/ S0\_5      rc " S1 S2\_5      rd " S3 S4\_5      re " S5 S6\_5      rf " S7 S8\_\)5      rg " S9 S:\W5      rh " S; S<\h5      ri " S= S>\h5      rj " S? S@\@\S\h5      rk\j" SASB9rl \j" SCSDSE9rm  " SF SG\@\S\5      rn " SH SI\@\S\5      ro\nRá                  \Râ                  5        \nRá                  \Rä                  5        \nRá                  \Ræ                  5        \nRá                  \Rè                  5        \nRá                  \Rê                  5        \nRá                  \Rì                  5        \nRá                  \Rî                  5        \nRá                  \Rð                  5        \nRá                  \Rò                  5        \nRá                  \Rô                  5        \nRá                  \Rö                  5        \nRá                  \Rø                  5        \nRá                  \Rú                  5        \nRá                  \Rü                  5        \nRá                  \Rþ                  5        \nRá                  \GR                   5        \nRá                  \GR                  5        \nRá                  \GR                  5        \nRá                  \GR                  5        \kRá                  \GR                  5        \kRá                  \GR
                  5        \kRá                  \GR                  5        \kRá                  \GR                  5        \oRá                  \GR                  5        \oRá                  \GR                  5        \nGR                  SJ5      r‹ \kGR                  SK5      rŒ \5GR                  " SL5      r� \oGR                  SM5      rŽ  " SN SO\X\5      r�SP\�SQ\�4SR jr�\" SS5       " ST SU\X\5      5       r‘ " SV SW\W5      r’ " SX SY\’5      r“ " SZ S[\“5      r” " S\ S]\”5      r• " S^ S_\W5      r– " S` Sa\–5      r— " Sb Sc\?\Q\–5      r˜\˜Rá                  \GR2                  5        \˜Rá                  \GR4                  5        \˜Rá                  \GR6                  5        \˜Rá                  \GR8                  5        \˜Rá                  \GR:                  5         " Sd Se\W5      rž " Sf Sg\ž5      rŸ " Sh Si\’5      r S SjK¡J¢r¢   " Sk Sl\’5      r£ " Sm Sn\’5      r¤ " So Sp\5      r¥ " Sq Sr\”5      r¦ " Ss St\X\5      r§ " Su Sv\¦5      r¨SQ\©4Sw jrª " Sx Sy\”5      r«g)zé    N)Údatetime)ÚEnum)ÚPath)Ú	AnnotatedÚAnyÚClassVarÚLiteral)ÚPictureClassificationLabel)ÚTextCellUnit)ÚAnyUrlÚ	BaseModelÚ
ConfigDictÚFieldÚPositiveIntÚcomputed_fieldÚfield_validatorÚmodel_validator©Ú
deprecated)Úasr_model_specsÚstage_model_specsÚvlm_model_specs)ÚAcceleratorDeviceÚAcceleratorOptions)ÚChartExtractionModelKindÚChartExtractionModelOptions)ÚExtractionPromptStyle)ÚKserveV2OptionsMixin)ÚDOCLING_LAYOUT_EGRET_LARGEÚDOCLING_LAYOUT_EGRET_MEDIUMÚDOCLING_LAYOUT_EGRET_XLARGEÚDOCLING_LAYOUT_HERONÚDOCLING_LAYOUT_HERON_101ÚDOCLING_LAYOUT_V2ÚLayoutModelConfig)Ú BaseObjectDetectionEngineOptions)Ú DocumentPictureClassifierOptions)ÚInlineAsrOptions)ÚApiVlmOptionsÚInferenceFrameworkÚInlineVlmOptionsÚResponseFormat)ÚObjectDetectionModelSpecÚObjectDetectionStagePresetMixinÚStagePresetMixinÚVlmModelSpec)ÚBaseVlmEngineOptions)ÚGRANITE_VISION_4_1_TRANSFORMERSÚGRANITE_VISION_OLLAMAÚGRANITE_VISION_TRANSFORMERSÚNU_EXTRACT_2B_TRANSFORMERSÚSMOLDOCLING_MLXÚSMOLDOCLING_TRANSFORMERSÚVlmModelType)Ú!ObjectDetectionEngineOptionsMixin)ÚVlmEngineOptionsMixinc                   ó*   • \ rS rSr% Sr\\   \S'   Srg)ÚBaseOptionséV   a»  Base class for all pipeline option models.

Every option class in the pipeline configuration hierarchy inherits from
`BaseOptions`. Subclasses must declare a `kind` ClassVar that serves as
a discriminator for polymorphic deserialization in Pydantic unions.

Attributes:
    kind: String discriminator identifying the concrete option type.
        Must be declared as a ``ClassVar[str]`` or
        ``ClassVar[Literal[...]]`` in each subclass.
Úkind© N)	Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__r   ÚstrÚ__annotations__Ú__static_attributes__r?   ó    Ú_/home/mande/repo/quber/.venv/lib/python3.13/site-packages/docling/datamodel/pipeline_options.pyr<   r<   V   s   ‡ ñ
ð �3‰-ÖrH   r<   c                   ó(   • \ rS rSrSrSrSrSrSrSr	g)	ÚOcrModeéf   z-
How to generate the input for the OCR model
Ú	full_pageÚlayout_regionsÚpdf_aware_layout_regionsÚdefaultr?   N)
r@   rA   rB   rC   rD   Ú	FULL_PAGEÚLAYOUT_REGIONSÚPDF_AWARE_LAYOUT_REGIONSÚDEFAULTrG   r?   rH   rI   rK   rK   f   s$   † ñð
 €Ið &€Nð  :Ðð ƒGrH   rK   c                   ó    • \ rS rSrSrSrSrSrg)ÚTableFormerModeéx   aã  Operating modes for TableFormer table structure extraction model.

Controls the trade-off between processing speed and extraction accuracy.
Choose based on your performance requirements and document complexity.

Attributes:
    FAST: Fast mode prioritizes speed over precision. Suitable for simple tables or high-volume
        processing.
    ACCURATE: Accurate mode provides higher quality results with slower processing. Recommended for complex
        tables and production use.
ÚfastÚaccurater?   N)r@   rA   rB   rC   rD   ÚFASTÚACCURATErG   r?   rH   rI   rV   rV   x   s   † ñ
ð €DØƒHrH   rV   c                   ó   • \ rS rSrSrSrg)ÚBaseTableStructureOptionsé‰   aS  Base options for table structure extraction models.

Serves as the abstract base for all table structure backends. Concrete
implementations (e.g., `TableStructureOptions` for TableFormer) inherit
from this class and register their own `kind` discriminator.

See Also:
    `TableStructureOptions`: Default TableFormer-based implementation.
r?   N©r@   rA   rB   rC   rD   rG   r?   rH   rI   r]   r]   ‰   s   † ôrH   r]   c                   ó‚   • \ rS rSr% SrSr\\   \S'   Sr	\
\\" SS94   \S'   \R                  r\
\\" S	S94   \S
'   Srg)ÚTableStructureOptionsé•   z1Options for the table structure (TableFormer V1).Údocling_tableformerr>   Tz¼Enable cell matching to align detected table cells with their content. When enabled, the model attempts to match table structure predictions with actual cell content for improved accuracy.©ÚdescriptionÚdo_cell_matchingz¾Table structure extraction mode. `accurate` provides higher quality results with slower processing, while `fast` prioritizes speed over precision. Recommended: `accurate` for production use.Úmoder?   N)r@   rA   rB   rC   rD   r>   r   rE   rF   rf   r   Úboolr   rV   r[   rg   rG   r?   rH   rI   ra   ra   •   sq   ‡ Ù;à/€Dˆ(�3‰-Ó/ð 	ð �iØÙðpñ	
ð	ñó ð" 	× Ñ ð 	ˆ)ØÙðmñ	
ð	ñö !rH   ra   c                   ó<   • \ rS rSr% SrSr\\   \S'   Sr	\
\S'   Srg)	ÚTableStructureV2Optionsé­   z1Options for the table structure (TableFormer V2).Údocling_tableformer_v2r>   Trf   r?   N)r@   rA   rB   rC   rD   r>   r   rE   rF   rf   rh   rG   r?   rH   rI   rj   rj   ­   s"   ‡ Ù;à2€Dˆ(�3‰-Ó2àð �dö rH   rj   c                   ó.   • \ rS rSr% SrSr\\   \S'   Sr	g)Ú"GraniteVisionTableStructureOptionsé¹   zGOptions for the table structure model using Granite Vision (VLM-based).Úgranite_vision_tabler>   r?   N)
r@   rA   rB   rC   rD   r>   r   rE   rF   rG   r?   rH   rI   rn   rn   ¹   s   ‡ ÙQà0€Dˆ(�3‰-Ö0rH   rn   c            	       óˆ  • \ rS rSr% Sr\R                  r\\\	" S\R                  \R                  \R                  \R                  /S94   \S'   \\\   \	" SSS//S94   \S	'   S
r\\\	" SSS
/SS94   \S'   \" SS9\S\S\4S j5       5       r\" SSS/S9\S\4S j5       5       r\R4                  S\SS4S j5       rSrg)Ú
OcrOptionsé¿   a  Base configuration for Optical Character Recognition engines.

Defines the common interface shared by all OCR engine implementations.
Subclasses provide engine-specific parameters while inheriting the shared
language selection, full-page OCR toggle, and bitmap area threshold.

See Also:
    `OcrAutoOptions`: Automatic engine selection based on availability.
    `EasyOcrOptions`, `TesseractCliOcrOptions`, `TesseractOcrOptions`,
    `RapidOcrOptions`, `OcrMacOptions`, `NemotronOcrOptions`: Engine-specific
    configurations.
z2Which document regions to feed as input to the OCR©re   Úexamplesrg   z[List of OCR languages to use. The format must match the values of the OCR engine of choice.ÚdeuÚengÚlangg      @zãImage scale multiplier applied before running OCR. The page is rendered at 72 DPI times this factor, so the default 3 yields 216 DPI. Lower it when the source image is already high resolution and upscaling degrades recognition.ç      ð?ç        )re   ru   ÚgtÚscaleÚbefore)rg   ÚdataÚreturnc                 ó„   • [        U[        5      (       a*  UR                  SS5      (       a  [        R                  US'   U$ )zy
Accept the deprecated `force_full_page_ocr` constructor keyword and
translate it into the `mode` it is an old name for.
Úforce_full_page_ocrFrg   )Ú
isinstanceÚdictÚpoprK   rQ   )Úclsr~   s     rI   Ú_accept_force_full_page_ocrÚ&OcrOptions._accept_force_full_page_ocrð   s6   € ô �dœD×!Ñ! d§h¡hÐ/DÀe×&LÑ&LÜ"×,Ñ,ˆD�‰LØˆrH   zJ`force_full_page_ocr` is deprecated; set `mode=OcrMode.FULL_PAGE` instead.z.If enabled, a full-page OCR is always applied.F)r   re   ru   c                 ó:   • U R                   [        R                  L $ ©N)rg   rK   rQ   ©Úselfs    rI   r�   ÚOcrOptions.force_full_page_ocrý   s   € ð �y‰yœG×-Ñ-Ð-Ð-rH   ÚvalueNc                 ó>   • U(       a  [         R                  U l        g g r‰   )rK   rQ   rg   )r‹   r�   s     rI   r�   rŒ     s   € æÜ×)Ñ)ˆD�Ið rH   )r@   rA   rB   rC   rD   rK   rT   rg   r   r   rQ   rR   rS   rF   ÚlistrE   r|   Úfloatr   Úclassmethodr   r†   r   Úpropertyrh   r�   ÚsetterrG   r?   rH   rI   rr   rr   ¿   sO  ‡ ñð0 	�‰ð 	ˆ)ØÙØLà×!Ñ!Ø×&Ñ&Ø×0Ñ0Ø—‘ð	ñ	
ð		ñó ð ØˆS‰	ÙØuØ˜e�nÐ%ñ	
ð	ñó ð( 	ð 
ˆ9ØÙðAð
 ˜3�ZØñ		
ð
	ñó ñ ˜(Ñ#Øð¨sð °só ó ó $ðñ àXàDØ�ñð ð. Tó .ó óð.ð ×Ñð*¨ð *°$ó *ó  ó*rH   rr   c                   óZ   • \ rS rSr% SrSr\\S      \S'   / r	\
\\   \" SS94   \S'   Srg	)
ÚOcrAutoOptionsi  aâ  Automatic OCR engine selection based on system availability.

When this option is used, Docling probes the runtime environment at
pipeline initialization and selects the best available OCR engine
(e.g., EasyOCR if GPU is present, Tesseract otherwise). Language
settings are deferred to the chosen engine's defaults.

Notes:
    The `lang` field is intentionally defaulted to an empty list.
    To control language selection, specify an explicit OCR engine
    option class instead.
Úautor>   zŠThe automatic OCR engine will use the default values of the engine. Please specify the engine explicitly to change the language selection.rd   rx   r?   N)r@   rA   rB   rC   rD   r>   r   r	   rF   rx   r   r�   rE   r   rG   r?   rH   rI   r•   r•     sK   ‡ ñð '-€Dˆ(�7˜6‘?Ñ
#Ó,ð 	ð 	ˆ)ØˆS‰	Ùð?ñ	
ð	ñö rH   r•   c                   óP  • \ rS rSr% SrSr\\S      \S'   S/r	\
\\   \" SS94   \S'   S	r\
\S
   \" SS94   \S'   Sr\
\\" SS94   \S'   Sr\
\S-  \" SS94   \S'   Sr\
\S-  \" SS94   \S'   Sr\
\S-  \" SS94   \S'   Sr\
\\" SS94   \S'   Sr\
\S-  \" SS94   \S'   Sr\
\S-  \" SS94   \S'   Sr\
\S-  \" SS94   \S'   Sr\
\S-  \" S S94   \S!'   Sr\
\S-  \" S"S#S$94   \S%'   Sr\
\S-  \" S&S94   \S''   0 r\
\\\4   \" S(S94   \S)'   \" S*S+9r S,r!g)-ÚRapidOcrOptionsi(  zòConfiguration for RapidOCR engine with multiple backend support.

See Also:
    - https://rapidai.github.io/RapidOCRDocs/install_usage/api/RapidOCR/
    - https://rapidai.github.io/RapidOCRDocs/main/install_usage/rapidocr/usage/#__tabbed_3_4
Úrapidocrr>   Úchinesea²  Recognition language. RapidOCR uses a single language per run; if more than one value is given only the first is used. Accepted values resolve to a PP-OCR recognizer: PP-OCRv6 covers ~52 language codes (e.g. 'ch', 'en', 'de', 'fr', 'japan'; the docling defaults 'chinese'/'english' map to 'ch'/'en'). Script-family names route to PP-OCRv5 on the onnxruntime/openvino/paddle backends ('arabic', 'ch', 'cyrillic', 'devanagari', 'el', 'en', 'eslav', 'korean', 'latin', 'ta', 'te', 'th') or to PP-OCRv4 on the torch backend ('arabic', 'cyrillic', 'devanagari', 'ka', 'korean', 'latin', 'ta', 'te'). A language the resolved backend cannot serve raises an error rather than falling back silently.rd   rx   Úonnxruntime)r›   ÚopenvinoÚpaddleÚtorchai  Inference backend for RapidOCR. Options: `onnxruntime` (default, cross-platform), `openvino` (Intel), `paddle` (PaddlePaddle), `torch` (PyTorch). Choose based on your hardware and available libraries. Note: for languages outside the PP-OCRv6 set, `torch` is limited to the PP-OCRv4 script models while the other backends use the wider PP-OCRv5 set (see `lang`).Úbackendç      à?z»Minimum confidence score for text detection. Text regions with scores below this threshold are filtered out. Range: 0.0-1.0. Lower values detect more text but may include false positives.Ú
text_scoreNzEEnable text detection stage. If None, uses RapidOCR default behavior.Úuse_detzTEnable text direction classification stage. If None, uses RapidOCR default behavior.Úuse_clszGEnable text recognition stage. If None, uses RapidOCR default behavior.Úuse_recFzCEnable verbose logging output from RapidOCR for debugging purposes.Úprint_verbosezJCustom path to text detection model. If None, uses default RapidOCR model.Údet_model_pathzOCustom path to text classification model. If None, uses default RapidOCR model.Úcls_model_pathzLCustom path to text recognition model. If None, uses default RapidOCR model.Úrec_model_pathzJCustom path to recognition keys file. If None, uses default RapidOCR keys.Úrec_keys_pathz"Deprecated. Use font_path instead.T)re   r   Úrec_font_pathz=Custom path to font file for text rendering in visualization.Ú	font_pathz•Additional parameters to pass through to RapidOCR engine. Use this to override or extend default RapidOCR configuration with engine-specific options.Úrapidocr_paramsÚforbid©Úextrar?   )"r@   rA   rB   rC   rD   r>   r   r	   rF   rx   r   r�   rE   r   rŸ   r¡   r�   r¢   rh   r£   r¤   r¥   r¦   r§   r¨   r©   rª   r«   r¬   rƒ   r   r   Úmodel_configrG   r?   rH   rI   r˜   r˜   (  s±  ‡ ñð +5€Dˆ(�7˜:Ñ&Ñ
'Ó4ð  
ˆð 	ˆ)ØˆS‰	ÙðZñ	
ð	ñó ð4 	ð ˆYØÐ<Ñ=ÙðNñ	
ð	ñ
ó 
ð& 	ð �	ØÙðoñ	
ð	ñó ð 	ð ˆYØˆt‰ÙØ_ñ	
ð	ñó ð 	ð ˆYØˆt‰ÙØnñ	
ð	ñó ð 	ð ˆYØˆt‰ÙØañ	
ð	ñó ð 	ð �9ØÙØ]ñ	
ð	ñó ð 	ð �IØˆd‰
ÙØdñ	
ð	ñó ð 	ð �IØˆd‰
ÙØiñ	
ð	ñó ð 	ð �IØˆd‰
ÙØfñ	
ð	ñó ð 	ð �9Øˆd‰
ÙØdñ	
ð	ñó ð 	ð �9Øˆd‰
ÙØ<Øñ	
ð	ñó ð 	ð ˆyØˆd‰
ÙØWñ	
ð	ñó ð 	ð �YØˆS�#ˆX‰ÙðOñ	
ð	ñó ñ ØñƒLrH   r˜   c                   ó¬   • \ rS rSr% SrSr\\S      \S'   / r	\
\\   \" SS94   \S'   Sr\
\S	   \" S
S94   \S'   \" SS9rSr\
\\" SS94   \S'   Srg)ÚNemotronOcrOptionsi   zŒConfiguration for NVIDIA Nemotron OCR.

Notes:
    Use the pipeline-level `artifacts_path` to point to pre-downloaded checkpoint artifacts.
znemotron-ocrr>   zLList of OCR languages. nemotron-OCR-v2 supports 'english' and 'multilingual'rd   rx   Úsentence)Úwordr³   Ú	paragraphzvGranularity requested from Nemotron OCR. `sentence` is the default because it maps most directly to Docling OCR cells.Úmerge_levelr­   r®   é   z~Number of images within the same page to process. In practice a batch>1 happens only with PDF inputs with many OCR rectangles.Ú
batch_sizer?   N)r@   rA   rB   rC   rD   r>   r   r	   rF   rx   r   r�   rE   r   r¶   r   r°   r¸   ÚintrG   r?   rH   rI   r²   r²      s·   ‡ ñð /=€Dˆ(�7˜>Ñ*Ñ
+Ó<ð 	ð 	ˆ)ØˆS‰	Ùà^ñ	
ð	ñó ð  	ð �ØÐ/Ñ0ÙðFñ	
ð	ñó ñ Øñ€Lð 	
ð �	ØÙð_ñ	
ð	ñö 
rH   r²   c                   ó>  • \ rS rSr% SrSr\\S      \S'   / SQr	\
\\   \" SS94   \S'   S	r\
\S	-  \" S
S94   \S'   Sr\
\\" SS94   \S'   S	r\
\S	-  \" SS94   \S'   Sr\
\S	-  \" SS94   \S'   Sr\
\\" SS94   \S'   Sr\
\\" SS94   \S'   \" SSS9rSrg	)ÚEasyOcrOptionsiÇ  z!Configuration for EasyOCR engine.Úeasyocrr>   )ÚfrÚdeÚesÚenz­List of language codes for OCR. EasyOCR supports 80+ languages. Use ISO 639-1 codes (e.g., `en`, `fr`, `de`). Multiple languages can be specified for multilingual documents.rd   rx   Nz‰Enable GPU acceleration for EasyOCR. If None, automatically detects and uses GPU if available. Set to False to force CPU-only processing.Úuse_gpur    z±Minimum confidence score for text recognition. Text with confidence below this threshold is filtered out. Range: 0.0-1.0. Lower values include more text but may reduce accuracy.Úconfidence_thresholdzŸDirectory path for storing downloaded EasyOCR models. If None, uses default EasyOCR cache location. Useful for offline environments or custom model management.Úmodel_storage_directoryÚstandardz®Recognition network architecture to use. Options: `standard` (default, balanced), `craft` (higher accuracy). Different networks may perform better on specific document types.Úrecog_networkTz}Allow automatic download of EasyOCR models on first use. Disable for offline environments where models must be pre-installed.Údownload_enabledz}Suppress Metal Performance Shaders (MPS) warnings on macOS. Reduces console noise when using Apple Silicon GPUs with EasyOCR.Úsuppress_mps_warningsr­   r?   )r¯   Úprotected_namespaces)r@   rA   rB   rC   rD   r>   r   r	   rF   rx   r   r�   rE   r   rÁ   rh   rÂ   r�   rÃ   rÅ   rÆ   rÇ   r   r°   rG   r?   rH   rI   r»   r»   Ç  sl  ‡ Ù+à)2€Dˆ(�7˜9Ñ%Ñ
&Ó2ò 	!ð 	ˆ)ØˆS‰	Ùðlñ	
ð	ñó !ð" 	ð ˆYØˆt‰Ùð=ñ	
ð	ñó ð" 	ð ˜)ØÙðZñ	
ð	ñó ð" 	ð ˜YØˆd‰
ÙðNñ	
ð	ñó ð" 	ð �9Øˆd‰
Ùð_ñ	
ð	ñó ð" 	ð �iØÙð6ñ	
ð	ñó ð" 	ð ˜9ØÙð3ñ	
ð	ñó ñ ØØñƒLrH   r»   c                   óÖ   • \ rS rSr% SrSr\\S      \S'   / SQr	\
\\   \" SS94   \S'   Sr\
\\" S	S94   \S
'   Sr\
\S-  \" SS94   \S'   Sr\
\S-  \" SS94   \S'   \" SS9rSrg)ÚTesseractCliOcrOptionsi  z;Configuration for Tesseract OCR via command-line interface.Ú	tesseractr>   ©Úfrarv   Úsparw   ú½List of Tesseract language codes. Use 3-letter ISO 639-2 codes (e.g., `eng`, `fra`, `deu`). Multiple languages enable multilingual OCR. Requires corresponding Tesseract language data files.rd   rx   z�Command or path to Tesseract executable. Use `tesseract` if in system PATH, or provide full path for custom installations (e.g., `/usr/local/bin/tesseract`).Útesseract_cmdNúwPath to Tesseract data directory containing language files. If None, uses Tesseract's default TESSDATA_PREFIX location.Úpathú¹Page Segmentation Mode for Tesseract. Values 0-13 control how Tesseract segments the page. Common values: 3 (auto), 6 (uniform block), 11 (sparse text). If None, uses Tesseract default.Úpsmr­   r®   r?   )r@   rA   rB   rC   rD   r>   r   r	   rF   rx   r   r�   rE   r   rÐ   rÒ   rÔ   r¹   r   r°   rG   r?   rH   rI   rÊ   rÊ     sã   ‡ ÙEà+6€Dˆ(�7˜;Ñ'Ñ
(Ó6ò 	%ð 	ˆ)ØˆS‰	Ùðtñ	
ð	ñó %ð" 	ð �9ØÙðOñ	
ð	ñó ð" 	ð 	ˆ)Øˆd‰
Ùð,ñ	
ð	ñó ð" 	ð ˆØˆd‰
Ùðqñ	
ð	ñ
ó ñ ØñƒLrH   rÊ   c                   ó¶   • \ rS rSr% SrSr\\S      \S'   / SQr	\
\\   \" SS94   \S'   S	r\
\S	-  \" S
S94   \S'   S	r\
\S	-  \" SS94   \S'   \" SS9rSrg	)ÚTesseractOcrOptionsi=  z@Configuration for Tesseract OCR via Python bindings (tesserocr).Ú	tesserocrr>   rÌ   rÏ   rd   rx   NrÑ   rÒ   rÓ   rÔ   r­   r®   r?   )r@   rA   rB   rC   rD   r>   r   r	   rF   rx   r   r�   rE   r   rÒ   rÔ   r¹   r   r°   rG   r?   rH   rI   rÖ   rÖ   =  s·   ‡ ÙJà+6€Dˆ(�7˜;Ñ'Ñ
(Ó6ò 	%ð 	ˆ)ØˆS‰	Ùðtñ	
ð	ñó %ð" 	ð 	ˆ)Øˆd‰
Ùð,ñ	
ð	ñó ð" 	ð ˆØˆd‰
Ùðqñ	
ð	ñ
ó ñ ØñƒLrH   rÖ   c                   óª   • \ rS rSr% SrSr\\S      \S'   / SQr	\
\\   \" SS94   \S'   S	r\
\\" S
S94   \S'   Sr\
\\" SS94   \S'   \" SS9rSrg)ÚOcrMacOptionsia  z:Configuration for native macOS OCR using Vision framework.Úocrmacr>   )zfr-FRzde-DEzes-ESzen-USz§List of language locale codes for macOS OCR. Use format `language-REGION` (e.g., `en-US`, `fr-FR`). Leverages native macOS Vision framework for OCR on Apple platforms.rd   rx   rY   zœRecognition accuracy level. Options: `accurate` (higher quality, slower) or `fast` (lower quality, faster). Choose based on speed vs. accuracy requirements.ÚrecognitionÚvisionzˆmacOS framework to use for OCR. Currently supports `vision` (Apple Vision framework). Future versions may support additional frameworks.Ú	frameworkr­   r®   r?   N)r@   rA   rB   rC   rD   r>   r   r	   rF   rx   r   r�   rE   r   rÛ   rÝ   r   r°   rG   r?   rH   rI   rÙ   rÙ   a  s°   ‡ ÙDà(0€Dˆ(�7˜8Ñ$Ñ
%Ó0ò 	-ð 	ˆ)ØˆS‰	ÙðVñ	
ð	ñó -ð" 	ð �ØÙðLñ	
ð	ñó ð" 	ð ˆyØÙðEñ	
ð	ñó ñ ØñƒLrH   rÙ   c                   ó¤   • \ rS rSr% SrSr\\S      \S'   \	" SSS9r
\\S'   S	S
/r\\\   \	" SS94   \S'   Sr\\\	" SSS94   \S'   \" SS9rSrg)ÚKserveV2OcrOptionsi…  ag  Configuration for KServe v2-based OCR (e.g., Triton Inference Server).

This OCR engine connects to a remote KServe v2-compatible inference server
(such as Triton) to perform OCR via gRPC or HTTP. It combines standard OCR
options with KServe v2 connection settings inherited from KserveV2OptionsMixin.

The engine handles custom preprocessing (RGB conversion, transpose, batching)
to match the expected input format of typical OCR models deployed on KServe v2
endpoints.

See Also:
    `KserveV2OptionsMixin`: Provides all KServe v2 connection configuration.
    `RapidOcrOptions`: Local OCR engine for comparison.
Úkserve_v2_ocrr>   Úocrz7Remote model name registered in the KServe v2 endpoint.©rP   re   Ú
model_nameÚenglishrš   z˜List of OCR languages. Note: Language selection depends on the deployed model. This parameter is passed to the server but may not be used by all models.rd   rx   ç       @z‘Image scale multiplier for OCR processing. Higher values increase resolution for better text recognition. Default 2.0 converts 72 DPI to 144 DPI.rz   )re   r{   r|   r­   r®   r?   N)r@   rA   rB   rC   rD   r>   r   r	   rF   r   rã   rE   rx   r   r�   r|   r�   r   r°   rG   r?   rH   rI   rß   rß   …  s©   ‡ ñð 0?€Dˆ(�7˜?Ñ+Ñ
,Ó>áØØMñ€J�ó ð 
�IÐð 	ˆ)ØˆS‰	Ùð\ñ	
ð	ñó ð& 	ð 
ˆ9ØÙðWð ñ	
ð	ñ	ó 	ñ ØñƒLrH   rß   c                   ó  • \ rS rSr% SrSr\\S'   Sr\	\
\" SSS94   \S	'   S
r\	\\" SSS94   \S'   Sr\	\\" SS94   \S'   Sr\	\\   S-  \" SS94   \S'   Sr\	\\   S-  \" SS94   \S'   Sr\	\\" SS94   \S'   Srg)ÚPictureDescriptionBaseOptionsi¶  a8  Base configuration for picture description models.

Provides shared parameters for all picture description backends,
including batch processing, image scaling, area thresholds, and
classification-based filtering (allow/deny lists). Concrete
implementations supply the actual model integration.

See Also:
    `PictureDescriptionApiOptions`: OpenAI-compatible API backend.
    `PictureDescriptionVlmOptions`: Legacy HuggingFace Transformers
        backend.
    `PictureDescriptionVlmEngineOptions`: New runtime-based backend
        with preset support (recommended).
TÚ_keep_deprecated_annotationsr·   é   z¯Number of images to process in a single batch during picture description. Higher values improve throughput but increase memory usage. Adjust based on available GPU/CPU memory.©Úgere   r¸   rå   r   zºScaling factor for image resolution before processing. Higher values (e.g., 2.0) provide more detail for the vision model but increase processing time and memory. Range: 0.5-4.0 typical.©r{   re   r|   çš™™™™™©?z¹Minimum picture area as fraction of page area (0.0-1.0) to trigger description. Pictures smaller than this threshold are skipped. Use lower values (e.g., 0.01) to describe small images.rd   Úpicture_area_thresholdNa	  List of picture classification labels to allow for description. Only pictures classified with these labels will be processed. If None, all picture types are allowed unless explicitly denied. Use to focus description on specific image types (e.g., diagrams, charts).Úclassification_allowzþList of picture classification labels to exclude from description. Pictures classified with these labels will be skipped. If None, no picture types are denied unless not in allow list. Use to exclude unwanted image types (e.g., decorative images, logos).Úclassification_denyrz   a!  Minimum classification confidence score (0.0-1.0) required for a picture to be processed. Pictures with classification confidence below this threshold are skipped. Higher values ensure only confidently classified images are described. Range: 0.0 (no filtering) to 1.0 (maximum confidence).Úclassification_min_confidencer?   )r@   rA   rB   rC   rD   rè   rh   rF   r¸   r   r¹   r   r|   r�   rî   rï   r�   r
   rð   rñ   rG   r?   rH   rI   rç   rç   ¶  s8  ‡ ñð$ *.Ð  $Ó-ð 	
ð �	ØÙØðbñ	
ð	ñ	ó 	
ð& 	ð 
ˆ9ØÙØðhñ	
ð	ñ	ó 	ð$ 	ð ˜IØÙðfñ	
ð	ñó ð$ 	ð ˜)ØÐ'Ñ(¨4Ñ/ÙðVñ	
ð	ñ	ó 	ð& 	ð ˜ØÐ'Ñ(¨4Ñ/ÙðQñ	
ð	ñ	ó 	ð& 	ð " 9ØÙðvñ	
ð	ñ	$ö 	rH   rç   c                   ól  • \ rS rSr% SrSr\\S      \S'   \	" S5      r
\\	\" SS94   \S'   0 r\\\\4   \" S	S
S0/S94   \S'   0 r\\\\4   \" SS94   \S'   Sr\\\" SS94   \S'   Sr\\\" SS94   \S'   Sr\\\" SS/S94   \S'   Sr\\\" SS94   \S'   Sr\\S-  \" S/ S QS94   \S!'   S"rg)#ÚPictureDescriptionApiOptionsi  ao  Configuration for API-based picture description services.

Sends images to an OpenAI-compatible chat completions endpoint for
description generation. Supports custom headers for authentication,
configurable timeouts, and concurrent request control.

Notes:
    Requires ``enable_remote_services=True`` on the parent pipeline
    options to permit external API calls.
Úapir>   z)http://localhost:8000/v1/chat/completionsz·API endpoint URL for picture description service. Must be OpenAI-compatible chat completions endpoint. Default points to local server; update for cloud services or custom deployments.rd   ÚurlzoHTTP headers to include in API requests. Use for authentication or custom headers required by your API service.ÚAuthorizationzBearer TOKENrt   Úheadersz‰Additional query parameters to include in API requests. Service-specific parameters for customizing API behavior beyond standard options.Úparamsç      4@z™Maximum time in seconds to wait for API response before timing out. Increase for slow networks or complex image descriptions. Recommended: 10-60 seconds.Útimeoutré   z¡Number of concurrent API requests allowed. Higher values improve throughput but may hit API rate limits. Adjust based on API service quotas and network capacity.Úconcurrencyú'Describe this image in a few sentences.z„Prompt template sent to the vision model for image description. Customize to guide the model's output style, detail level, or focus.z/Provide a technical description of this diagramÚpromptÚ z�Provenance information to track the source or method of picture descriptions. Used for metadata and auditing purposes in the output document.Ú
provenanceÚusageNzâResponse JSON key, or dotted path, whose value should be preserved as the raw usage payload on picture description metadata. The default captures OpenAI-compatible `usage` objects. Set to None to disable usage payload capture.)r   ÚproviderUsagez
meta.usageÚusage_response_keyr?   )r@   rA   rB   rC   rD   r>   r   r	   rF   r   rõ   r   r   r÷   rƒ   rE   rø   r   rú   r�   rû   r¹   rý   rÿ   r  rG   r?   rH   rI   ró   ró     s®  ‡ ñ	ð &+€Dˆ(�7˜5‘>Ñ
"Ó*ñ 	Ð:Ó;ð ˆØÙðcñ	
ð	ñ
ó <ð$ 	ð ˆYØˆS�#ˆX‰Ùðð '¨Ð7Ð8ñ	
ð	ñ	ó 	ð$ 	ð ˆIØˆS�#ˆX‰Ùð8ñ	
ð	ñó ð" 	ð ˆYØÙðJñ	
ð	ñó ð" 	
ð �ØÙðKñ	
ð	ñó 
ð$ 	2ð ˆIØÙð1ð HÐHñ	
ð	ñ	ó 	2ð$ 	ð �	ØÙð@ñ	
ð	ñó ð& 	ð ˜	Øˆd‰
Ùð@ò >ñ	
ð	ñ
ö 
rH   ró   c                   óê   • \ rS rSr% SrSr\\S      \S'   \	\
\" SSS/S94   \S	'   S
r\	\
\" SSS/S94   \S'   SSS.r\	\\
\4   \" SS94   \S'   Sr\	\S   \" SS94   \S'   \S\
4S j5       rSrg)ÚPictureDescriptionVlmOptionsic  a  Configuration for inline vision-language models for picture description.

This is the legacy implementation that uses direct HuggingFace Transformers integration.
For the new runtime-based system with preset support, use PictureDescriptionVlmEngineOptions.
Úvlmr>   zŒHuggingFace model repository ID for the vision-language model. Must be a model capable of image-to-text generation for picture descriptions.ú#HuggingFaceTB/SmolVLM-256M-Instructú!ibm-granite/granite-vision-3.3-2brt   Úrepo_idrü   úePrompt template for the vision model. Customize to control description style, detail level, or focus.úWhat is shown in this image?ú(Provide a detailed technical descriptionrý   éÈ   F©Úmax_new_tokensÚ	do_samplezâHuggingFace generation configuration for text generation. Controls output length, sampling strategy, temperature, etc. See: https://huggingface.co/docs/transformers/en/main_classes/text_generation#transformers.GenerationConfigrd   Úgeneration_configÚleft)r  Úrightz¢Tokenizer padding side used for batched generation. Defaults to left to preserve the legacy behavior, but can be overridden for models that require right padding.Úpadding_sider   c                 ó:   • U R                   R                  SS5      $ )z¼Return the local cache folder name derived from the HuggingFace repo ID.

Converts the ``repo_id`` (e.g., ``"org/model"``) to a filesystem-safe
folder name by replacing ``/`` with ``--``.
Ú/z--)r  ÚreplacerŠ   s    rI   Úrepo_cache_folderÚ.PictureDescriptionVlmOptions.repo_cache_folder˜  s   € ð �|‰|×#Ñ# C¨Ó.Ð.rH   r?   N)r@   rA   rB   rC   rD   r>   r   r	   rF   r   rE   r   rý   r  rƒ   r   r  r’   r  rG   r?   rH   rI   r  r  c  s  ‡ ñð &+€Dˆ(�7˜5‘>Ñ
"Ó*ØØÙð`ð 6Ø3ðñ		
ð
	ñó ð0 	2ð ˆIØÙàwð /Ø:ðñ		
ð		ñó 2ð* ¨UÑ3ð �yØˆS�#ˆX‰Ùðyñ	
ð	ñ	ó 	4ð$ 	ð �)Ø�Ñ ÙðYñ	
ð	ñó ð ð/ 3ó /ó ó/rH   r  c                   ó    • \ rS rSr% SrSr\\S      \S'   \	" SS9r
\\S'   Sr\\\	" S	S
S/S94   \S'   SSS.r\\\\4   \	" SS94   \S'   Srg)Ú"PictureDescriptionVlmEngineOptionsi¢  aÒ  Configuration for VLM runtime-based picture description.

This is the new implementation that uses the pluggable runtime system with preset support.
Supports all runtime types (Transformers, MLX, API, etc.) through the unified runtime interface.

Use `from_preset()` to create instances from registered presets.

Examples:
    # Use preset with default runtime
    options = PictureDescriptionVlmEngineOptions.from_preset("smolvlm")

    # Use preset with runtime override
    from docling.datamodel.vlm_engine_options import MlxVlmEngineOptions, VlmEngineType
    options = PictureDescriptionVlmEngineOptions.from_preset(
        "smolvlm",
        engine_options=MlxVlmEngineOptions(engine_type=VlmEngineType.MLX)
    )
Úpicture_description_vlm_enginer>   ú3Model specification with runtime-specific overridesrd   Ú
model_specrü   r	  r
  r  rt   rý   r  Fr  zjGeneration configuration for text generation. Controls output length, sampling strategy, temperature, etc.r  r?   N)r@   rA   rB   rC   rD   r>   r   r	   rF   r   r  r0   rý   r   rE   r  rƒ   r   rG   r?   rH   rI   r  r  ¢  s§   ‡ ñð( 	)ð 	ˆ(�7Ð;Ñ<Ñ
=ó ñ  %ØIñ €J�ó ð 	2ð ˆIØÙàwð /Ø:ðñ		
ð		ñó 2ð( ¨UÑ3ð �yØˆS�#ˆX‰Ùð$ñ	
ð	ñö 4rH   r  r  )r  r  r
  )r  rý   c                   ó–   • \ rS rSr% Sr\" SS9r\\S'   \" SSS9r	\
\S	'   \" S
SS9r\S
-  \S'   \" SSS9r\\S'   \" SSS9r\\S'   Srg
)ÚVlmConvertOptionsiì  ae  Configuration for VLM-based document conversion.

This stage uses vision-language models to convert document pages to
structured formats (DocTags, Markdown, etc.). Supports preset-based
configuration via StagePresetMixin.

Examples:
    # Use preset with default runtime
    options = VlmConvertOptions.from_preset("smoldocling")

    # Use preset with runtime override
    from docling.datamodel.vlm_engine_options import ApiVlmEngineOptions, VlmEngineType
    options = VlmConvertOptions.from_preset(
        "smoldocling",
        engine_options=ApiVlmEngineOptions(engine_type=VlmEngineType.API_OLLAMA)
    )
r  rd   r  rå   ú&Image scaling factor for preprocessingrâ   r|   Nú)Maximum image dimension (width or height)Úmax_sizeré   z(Batch size for processing multiple pagesr¸   Fz3Force use of backend text extraction instead of VLMÚforce_backend_textr?   )r@   rA   rB   rC   rD   r   r  r0   rF   r|   r�   r"  r¹   r¸   r#  rh   rG   r?   rH   rI   r  r  ì  s†   ‡ ññ$  %ØIñ €J�ó ñ ØÐ!Iñ€Eˆ5ó ñ !ØÐ"Mñ€Hˆc�D‰jó ñ ØÐIñ€J�ó ñ  %ØÐ#Xñ Ð˜ö rH   r  c                   ó–   • \ rS rSr% Sr\" SS9r\\S'   \" SSS9r	\
\S	'   \" S
SS9r\S
-  \S'   \" SSS9r\\S'   \" SSS9r\\S'   Srg
)ÚCodeFormulaVlmOptionsi  a²  Configuration for VLM-based code and formula extraction.

This stage uses vision-language models to extract code blocks and
mathematical formulas from document images. Supports preset-based
configuration via StagePresetMixin.

Examples:
    # Use CodeFormulaV2 preset
    options = CodeFormulaVlmOptions.from_preset("codeformulav2")

    # Use Granite Docling preset
    options = CodeFormulaVlmOptions.from_preset("granite_docling")
r  rd   r  rå   r   râ   r|   Nr!  r"  TzExtract code blocksÚextract_codezExtract mathematical formulasÚextract_formulasr?   )r@   rA   rB   rC   rD   r   r  r0   rF   r|   r�   r"  r¹   r&  rh   r'  rG   r?   rH   rI   r%  r%    s   ‡ ññ  %ØIñ €J�ó ñ ØÐ!Iñ€Eˆ5ó ñ !ØÐ"Mñ€Hˆc�D‰jó ñ  tÐ9NÑO€L�$ÓOá"ØÐ"AñÐ�dö rH   r%  Úgranite_doclingÚsmolvlmÚdocument_figure_classifier_v2Úcodeformulav2c                   ó0   • \ rS rSrSrSrSrSrSrSr	Sr
S	rg
)Ú
PdfBackendi{  aÌ  Available PDF parsing backends for document processing.

Different backends offer varying levels of text extraction quality, layout
preservation, and processing speed. Choose based on your document complexity
and quality requirements.

Attributes:
    PYPDFIUM2: Standard PDF parser using PyPDFium2 library. Fast and
        reliable for basic text extraction.
    DOCLING_PARSE: Docling Parse backend providing enhanced layout
        analysis, structure preservation, and advanced table detection.
        Single-threaded; use `THREADED_DOCLING_PARSE` unless serialized
        page parsing is required.
    THREADED_DOCLING_PARSE: Threaded Docling Parse backend optimized for
        concurrent page parsing in the standard PDF pipeline. This is the
        default and recommended backend for most use cases.
    DLPARSE_V1: Deprecated. Maps to `DOCLING_PARSE`.
    DLPARSE_V2: Deprecated. Maps to `DOCLING_PARSE`.
    DLPARSE_V4: Deprecated. Maps to `DOCLING_PARSE`.
Ú	pypdfium2Údocling_parseÚthreaded_docling_parseÚ
dlparse_v1Ú
dlparse_v2Ú
dlparse_v4r?   N)r@   rA   rB   rC   rD   Ú	PYPDFIUM2ÚDOCLING_PARSEÚTHREADED_DOCLING_PARSEÚ
DLPARSE_V1Ú
DLPARSE_V2Ú
DLPARSE_V4rG   r?   rH   rI   r-  r-  {  s*   † ñð* €IØ#€MØ5Ðð €JØ€JØƒJrH   r-  rŸ   r   c                 ó   • SSK n[        R                  [        R                  [        R                  [        R                  [        R
                  [        R                  0nX;   a(  UR                  " SU R                   S3[        SS9  X    $ U $ )zðNormalize deprecated backend enum values to current ones.

Args:
    backend: The PDF backend enum value to normalize.

Returns:
    The normalized backend enum value.

Raises:
    DeprecationWarning: If a deprecated backend value is used.
r   NzPdfBackend.zh was previously deprecated and removed in this docling version. Using PdfBackend.DOCLING_PARSE instead. é   ©Ú
stacklevel)	Úwarningsr-  r7  r5  r8  r9  ÚwarnÚnameÚDeprecationWarning)rŸ   r>  Údeprecated_mappings      rI   Únormalize_pdf_backendrC  ›  s…   € ó ô 	×Ñœz×7Ñ7Ü×Ñœz×7Ñ7Ü×Ñœz×7Ñ7ðÐð Ó$Ø�ŠØ˜'Ÿ,™,˜ð  (Pð  QÜØò	
ð
 "Ñ*Ð*à€NrH   zNUse get_ocr_factory().registered_kind to get a list of registered OCR engines.c                   ó0   • \ rS rSrSrSrSrSrSrSr	Sr
S	rg
)Ú	OcrEnginei»  a  Available OCR (Optical Character Recognition) engines for text extraction from images.

Each engine has different characteristics in terms of accuracy, speed, language support,
and platform compatibility. Choose based on your specific requirements.

Attributes:
    AUTO: Automatically select the best available OCR engine based on platform and installed libraries.
    EASYOCR: Deep learning-based OCR supporting 80+ languages with GPU acceleration.
    TESSERACT_CLI: Tesseract OCR via command-line interface (requires system installation).
    TESSERACT: Tesseract OCR via Python bindings (tesserocr library).
    OCRMAC: Native macOS Vision framework OCR (Apple platforms only).
    RAPIDOCR: Lightweight OCR with multiple backend options (ONNX, OpenVINO, PaddlePaddle).
r–   r¼   Útesseract_clirË   rÚ   r™   r?   N)r@   rA   rB   rC   rD   ÚAUTOÚEASYOCRÚTESSERACT_CLIÚ	TESSERACTÚOCRMACÚRAPIDOCRrG   r?   rH   rI   rE  rE  »  s'   † ñð €DØ€GØ#€MØ€IØ€FØƒHrH   rE  c                   óê   • \ rS rSr% SrSr\\S-  \" SSS/S94   \	S'   \
" 5       r\\
\" S	S
94   \	S'   Sr\\\" SS/S94   \	S'   Sr\\\" SS/S94   \	S'   Sr\\\-  S-  \" SSS/S94   \	S'   Srg)ÚPipelineOptionsiÕ  a  Base configuration for document processing pipelines.

Provides the foundational settings shared by every pipeline type:
document-level timeout, hardware accelerator selection, remote service
permissions, external plugin control, and model artifact paths. All
specialized pipeline option classes inherit from this base.

See Also:
    `ConvertPipelineOptions`: Adds picture classification and description.
    `AsrPipelineOptions`: Audio/speech recognition pipeline.
    `VlmExtractionPipelineOptions`: VLM-based structured extraction.
Na­  Maximum processing time in seconds before aborting document conversion. When exceeded, the pipeline stops processing and returns partial results with PARTIAL_SUCCESS status. Timeout errors are recorded in ConversionResult.errors with category=TIMEOUT and descriptive error messages. Use ConversionResult.has_timeout_errors() to detect timeouts. If None, no timeout is enforced. Recommended: 90-120 seconds for production systems.ç      $@rù   rt   Údocument_timeoutz»Hardware acceleration configuration for model inference. Controls GPU device selection, memory management, and execution optimization settings for layout, OCR, and table structure models.rd   Úaccelerator_optionsFz´Allow pipeline to call external APIs or cloud services during processing. Required for API-based picture description models. Disabled by default for security and offline operation.Úenable_remote_serviceszÅAllow loading external third-party plugins for OCR, layout, table structure, or picture description models. Enables custom model implementations via plugin system. Disabled by default for security.Úallow_external_pluginszöLocal directory containing pre-downloaded model artifacts (weights, configs). If None, models are fetched from remote sources on first use. Use `docling-tools models download` to pre-fetch artifacts for offline operation or faster initialization.z./artifactsz/tmp/docling_outputsÚartifacts_pathr?   )r@   rA   rB   rC   rD   rP  r   r�   r   rF   r   rQ  rR  rh   rS  rT  r   rE   rG   r?   rH   rI   rN  rN  Õ  s  ‡ ñð2 	ð �iØ�‰ÙðFð ˜D�\ñ		
ð
	ñó ñ* 	Óð ˜ØÙðoñ	
ð	ñó ð$ 	ð ˜IØÙðfð �Wñ	
ð	ñ	ó 	ð& 	ð ˜IØÙðtð �Wñ	
ð	ñ	ó 	ð( 	ð �IØˆs‰
�TÑÙðBð $Ð%;Ð<ñ	
ð	ñ
ö 
rH   rN  c                   óä   • \ rS rSr% SrSr\\\" SS94   \	S'   \
r\\\" SS94   \	S'   Sr\\\" S	S94   \	S
'   \r\\\" SS94   \	S'   Sr\\\" SS94   \	S'   \" 5       r\\\" SS94   \	S'   Srg)ÚConvertPipelineOptionsi  a�  Base configuration for document conversion pipelines.

Extends `PipelineOptions` with picture-related features: classification
(categorizing images by type) and description (generating textual
captions via vision-language models). Also supports chart data extraction
from bar, pie, and line charts.

See Also:
    `PaginatedPipelineOptions`: Adds page image generation for paginated
        formats.
FzžEnable picture classification to categorize images by type (photo, diagram, chart, etc.). Useful for downstream processing that requires image type awareness.rd   Údo_picture_classificationz�Configuration for picture classification model/runtime. Supports selecting transformers, onnxruntime, or remote api_kserve_v2 inference engines.Úpicture_classification_optionszªEnable automatic generation of textual descriptions for pictures using vision-language models. Descriptions are added to the document for accessibility and searchability.Údo_picture_descriptionzåConfiguration for picture description model. Uses new preset system (recommended). Default: 'smolvlm' preset. Only applicable when `do_picture_description=True`. Example: PictureDescriptionVlmOptions.from_preset('granite_vision')Úpicture_description_optionsz¾Enable chart data extraction to convert bar, pie, and line charts into structured tabular data. Automatically enables picture classification. Only applicable when `do_chart_extraction=True`.Údo_chart_extractionz�Configuration for the chart extraction model, including which model variant to use and which output formats to generate (CSV, code, summary).Úchart_extraction_optionsr?   N)r@   rA   rB   rC   rD   rW  r   rh   r   rF   Ú'_default_picture_classification_optionsrX  r'   rY  Ú$_default_picture_description_optionsrZ  rç   r[  r   r\  rG   r?   rH   rI   rV  rV    s  ‡ ñ
ð( 	ð ˜yØÙðWñ	
ð	ñ ó ð" 	0ð # IØ(Ùðkñ	
ð	ñ%ó 0ð" 	ð ˜IØÙð^ñ	
ð	ñó ð$ 	-ð   Ø%ÙðVñ	
ð	ñ	"ó 	-ð( 	ð ˜ØÙðCñ	
ð	ñ	ó 	ñ$ 	$Ó%ð ˜iØ#ÙðMñ	
ð	ñö &rH   rV  c                   óz   • \ rS rSr% SrSr\\\" SS94   \	S'   Sr
\\\" SS94   \	S	'   Sr\\\" S
S94   \	S'   Srg)ÚPaginatedPipelineOptionsib  aÊ  Configuration for pipelines processing paginated documents.

Extends `ConvertPipelineOptions` with page-level image generation
controls for formats that have a concept of discrete pages (PDF, PPTX,
images). Controls the resolution scaling and whether page/picture images
are generated during conversion.

See Also:
    `PdfPipelineOptions`: Full PDF pipeline with OCR, layout, and tables.
    `VlmPipelineOptions`: VLM-based document understanding pipeline.
ry   úëScaling factor for generated images. Higher values produce higher resolution but increase processing time and storage requirements. Recommended values: 1.0 (standard quality), 2.0 (high resolution), 0.5 (lower resolution for previews).rd   Úimages_scaleFú«Generate rendered page images during extraction. Creates PNG representations of each page for visual preview, validation, or downstream image-based machine learning tasks.Úgenerate_page_imagesz³Extract and save embedded images from the document. Exports individual images (figures, photos, diagrams, charts) found in the document as separate image files for downstream use.Úgenerate_picture_imagesr?   N)r@   rA   rB   rC   rD   rb  r   r�   r   rF   rd  rh   re  rG   r?   rH   rI   r`  r`  b  sŠ   ‡ ñ
ð* 	ð �)ØÙð,ñ	
ð	ñ	ó 	ð$ 	ð ˜)ØÙðYñ	
ð	ñó ð" 	ð ˜YØÙð\ñ	
ð	ñö rH   r`  c                   ó†   • \ rS rSr% SrSr\\\" SS94   \	S'   Sr
\\\" SS94   \	S	'   \r\\\-  \-  \" S
S94   \	S'   Srg)ÚVlmPipelineOptionsi�  a  Pipeline configuration for vision-language model based document processing.

Uses a VLM to understand document pages holistically from rendered page
images rather than composing results from separate layout, OCR, and
table-structure models. Page image generation is enabled by default
since the VLM requires visual input.

Notes:
    Unlike `PdfPipelineOptions`, this pipeline does not run separate
    layout analysis or OCR stages. Set ``force_backend_text=True`` to
    use the PDF backend's native text instead of VLM-predicted text.
TzŽGenerate page images for VLM processing. Required for vision-language models to analyze document pages. Automatically enabled in VLM pipeline.rd   rd  Fz¦Force use of backend's native text extraction instead of VLM predictions. When enabled, bypasses VLM text detection and uses embedded text from the document directly.r#  a  Vision-Language Model configuration for document understanding. Uses new VlmConvertOptions with preset system (recommended). Legacy InlineVlmOptions/ApiVlmOptions still supported. Default: 'granite_docling' preset. Example: VlmConvertOptions.from_preset('smoldocling')Úvlm_optionsr?   N)r@   rA   rB   rC   rD   rd  r   rh   r   rF   r#  Ú_default_vlm_convert_optionsrh  r  r+   r)   rG   r?   rH   rI   rg  rg  �  s•   ‡ ñð* 	ð ˜)ØÙð9ñ	
ð	ñó ð" 	ð ˜	ØÙðTñ	
ð	ñó ð$ 	%ð �ØÐ,Ñ,¨}Ñ<Ùðkñ	
ð	ñ	ö 	%rH   rg  c                   óz   • \ rS rSr% SrSr\\\" SS94   \	S'   Sr
\\\" SS94   \	S'   S	r\\\" S
S94   \	S'   Srg)ÚBaseLayoutOptionsi¹  aû  Base options for document layout analysis models.

Layout analysis detects the structural regions of a document page
(text blocks, tables, figures, headers, etc.) and assigns content
cells to those regions. This base class provides the shared controls
for empty-cluster retention and cell-assignment skipping.

See Also:
    `LayoutObjectDetectionOptions`: Default layout options; object-detection
        runtime with preset support.
    `LayoutOptions`: Deprecated predecessor, translated onto the above.
FzªRetain empty clusters in layout analysis results. When False, clusters without content are removed. Enable for debugging or when empty regions are semantically important.rd   Úkeep_empty_clusterszÇSkip assignment of cells to table structures during layout analysis. When True, cells are detected but not associated with tables. Use for performance optimization when table structure is not needed.Úskip_cell_assignmentTzºCreate clusters for orphaned elements not assigned to any structure. When True, isolated text or elements are grouped into their own clusters. Recommended for complete document coverage.Úcreate_orphan_clustersr?   N©r@   rA   rB   rC   rD   rl  r   rh   r   rF   rm  rn  rG   r?   rH   rI   rk  rk  ¹  s‹   ‡ ñð* 	ð ˜ØÙðYñ	
ð	ñó ð" 	ð ˜)ØÙðwñ	
ð	ñó ð" 	ð ˜IØÙðlñ	
ð	ñö rH   rk  c                   ón   ^ • \ rS rSr% SrSr\\   \S'   \	r
\\\" SS94   \S'   S\S	S
4U 4S jjrSrU =r$ )ÚLayoutOptionsiä  a©  Deprecated. Use `LayoutObjectDetectionOptions` instead.

Retained so existing code keeps working: it still constructs, still
selects any of the supported layout models, and is translated onto
`LayoutObjectDetectionOptions` by the `LayoutModel` shim.

Notes:
    ``DOCLING_LAYOUT_V2`` is no longer supported and falls back to
    ``DOCLING_LAYOUT_HERON`` with a warning.

    Removing this class also retires `layout_model_specs` (including every
    ``DOCLING_LAYOUT_*`` constant and `LayoutModelConfig`) and the
    `models/stages/layout/layout_model.py` shim, which exist solely to
    serve it.

Example:
    >>> LayoutObjectDetectionOptions.from_preset("layout_heron_default")
Údocling_layout_defaultr>   z¿Layout model configuration specifying which model to use for document layout analysis. Options include DOCLING_LAYOUT_HERON (default, balanced), DOCLING_LAYOUT_EGRET_* (higher accuracy), etc.rd   r  Úcontextr   Nc                óX   >• [         TU ]  U5        [        R                  " S[        SS9  g )Nz­LayoutOptions is deprecated and will be removed in a future release. Use LayoutObjectDetectionOptions, e.g. LayoutObjectDetectionOptions.from_preset("layout_heron_default").é   r<  )ÚsuperÚmodel_post_initr>  r?  rA  )r‹   rs  Ú	__class__s     €rI   rw  ÚLayoutOptions.model_post_init  s*   ø€ Ü‰Ñ Ô(Ü�ŠðPô Øó	
rH   r?   )r@   rA   rB   rC   rD   r>   r   rE   rF   r"   r  r   r%   r   r   rw  rG   Ú__classcell__)rx  s   @rI   rq  rq  ä  s\   ø‡ ñð& 3€Dˆ(�3‰-Ó2ð 	ð �	ØÙðkñ	
ð	ñó ð
 sð 
°$÷ 
õ 
rH   rq  c                   óH   • \ rS rSr% SrSr\\   \S'   \	" S SS9r
\\S'   S	rg
)ÚLayoutObjectDetectionOptionsi  a  Options for layout detection using object-detection runtimes.

The default layout options. Uses the pluggable object-detection engine
system with preset support via `ObjectDetectionStagePresetMixin`; use
``from_preset()`` to create instances from registered model presets.

Notes:
    The default model is ``layout_heron_default``. For higher accuracy on
    complex documents, consider the ``layout_egret_large`` or
    ``layout_egret_xlarge`` presets.

Example:
    >>> LayoutObjectDetectionOptions.from_preset("layout_egret_large")
Úlayout_object_detectionr>   c                  óP   • [         R                  R                  R                  SS9$ )NT)Údeep)r   ÚOBJECT_DETECTION_LAYOUT_HERONr  Ú
model_copyr?   rH   rI   Ú<lambda>Ú%LayoutObjectDetectionOptions.<lambda>&  s%   € Ü×;Ñ;×FÑF×QÑQØð Rñ rH   z8Object-detection model specification for layout analysis)Údefault_factoryre   r  r?   N)r@   rA   rB   rC   rD   r>   r   rE   rF   r   r  r-   rG   r?   rH   rI   r|  r|    s4   ‡ ñ
ð 4€Dˆ(�3‰-Ó3á+0ñ
ð
 Oñ,€JÐ(ö rH   r|  c                   óz   • \ rS rSr% SrSr\\\" SS94   \	S'   Sr
\\\" SS94   \	S'   S	r\\\" S
S94   \	S'   Srg)ÚBaseLayoutPostprocessorOptionsi@  aG  Algorithm parameters consumed by ``LayoutPostprocessor``.

These controls drive the post-processing of raw layout clusters
(cell assignment, empty-cluster handling, orphan-cluster creation).
They are decoupled from the layout (prediction) options so the
post-processing stage and the predictor models can evolve
independently.
FzcRetain empty clusters in layout analysis results. When False, clusters without content are removed.rd   rl  zƒSkip assignment of cells to clusters during layout post-processing. When True, cells are detected but not associated with clusters.rm  TzDCreate clusters for orphaned elements not assigned to any structure.rn  r?   Nro  r?   rH   rI   r†  r†  @  s„   ‡ ñð  	ð ˜ØÙàuñ	
ð	ñó ð  	ð ˜)ØÙð4ñ	
ð	ñó ð  	ð ˜IØÙàVñ	
ð	ñö rH   r†  c                   óN   • \ rS rSr% SrSr\\   \S'   Sr	\
\\" SS94   \S'   S	rg
)ÚLayoutPostprocessorOptionsie  a  Stage options for ``LayoutPostprocessingModel``.

Extends the algorithm parameters with the stage-level toggle
``run_postprocessor``. When disabled, the stage only computes the
layout confidence score and leaves the raw clusters untouched
(used by the table-crops layout model).
Úlayout_postprocessorr>   Tz†Run the layout post-processor. When False, raw clusters are passed through unchanged and only the layout confidence score is computed.rd   Úrun_postprocessorr?   N)r@   rA   rB   rC   rD   r>   r   rE   rF   rŠ  r   rh   r   rG   r?   rH   rI   rˆ  rˆ  e  sB   ‡ ñð 1€Dˆ(�3‰-Ó0ð 	ð �yØÙð7ñ	
ð	ñö rH   rˆ  c                   óN   • \ rS rSr% Sr\R                  r\\	\
" SS94   \S'   Srg)ÚAsrPipelineOptionsiz  a  Configuration options for the Automatic Speech Recognition (ASR) pipeline.

This pipeline processes audio files and converts speech to text using Whisper-based models.
Supports various audio formats (MP3, WAV, FLAC, etc.) and video files with audio tracks.
zÆAutomatic Speech Recognition (ASR) model configuration for audio transcription. Specifies which ASR model to use (e.g., Whisper variants) and model-specific parameters for speech-to-text conversion.rd   Úasr_optionsr?   N)r@   rA   rB   rC   rD   r   ÚWHISPER_TINYr�  r   r(   r   rF   rG   r?   rH   rI   rŒ  rŒ  z  s9   ‡ ñð 	×$Ñ$ð �ØÙðyñ	
ð	ñö %rH   rŒ  )ÚVideoFrameSamplingModec                   óÎ  • \ rS rSr% Sr\R                  r\\	\
" SS94   \S'   \R                  r\\\
" SS94   \S'   Sr\\\
" S	S
S94   \S'   Sr\\S-  \
" SS	SS94   \S'   Sr\\\
" S	SS94   \S'   Sr\\\
" S	SS94   \S'   Sr\\S-  \
" SS	SS94   \S'   Sr\\\
" SS	SS94   \S'   Sr\\S-  \
" SS	SS94   \S'   S r\\\
" S S!S"94   \S#'   S$r\\\
" S$S%S"94   \S&'   S'rg)(ÚVideoPipelineOptionsi�  a·  Configuration options for the video pipeline.

Controls ASR transcription, frame sampling strategy, and optional
scene description for video documents.

Recommended configs by use case:
  - Business meetings:  frame_sampling_mode=SCENE_CHANGE, scene_change_prominence=0.03
  - Lecture recordings: frame_sampling_mode=SCENE_CHANGE, cuts_per_minute=2.0
  - General video:      frame_sampling_mode=FIXED_INTERVAL, frame_interval_seconds=10.0
z2ASR model configuration for the video audio track.rd   r�  z-How representative video frames are selected.Úframe_sampling_moderO  r   z)Fixed frame sampling interval in seconds.rì   Úframe_interval_secondsNz;Prominence for local peak detection. None = auto-calibrate.)rP   rë   re   Úscene_change_prominencery   z-Low frame rate used for scene-change probing.Úscene_change_probe_fpsrå   z.Minimum duration before accepting a new scene.rê   Úmin_scene_duration_secondszOptional cap on sampled frames.)rP   r{   re   Úmax_sampled_framesru  zqSmoothing window (in frames) applied when detecting scene-change peaks. Higher values produce smoother detection.Úscene_change_smooth_windowz–Optional target density of cuts per minute for scene-change sampling. If set, the sampler will aim to produce approximately this many cuts per minute.Úcuts_per_minuteTziWhen True, representative frames are sampled and embedded in the output DoclingDocument as picture items.râ   Úgenerate_frame_imagesFz:Enable speaker diarization on audio tracks when available.Úenable_diarizationr?   )r@   rA   rB   rC   rD   r   rŽ  r�  r   r(   r   rF   r�  ÚFIXED_INTERVALr’  r“  r�   r”  r•  r–  r—  r¹   r˜  r™  rš  rh   r›  rG   r?   rH   rI   r‘  r‘  �  s  ‡ ñ	ð 	×$Ñ$ð �ØÙÐNÑOð	Qñó %ð 	×-Ñ-ð ˜ØÙÐIÑJð	Lñó .ð 	ð ˜IØÙ�Ð KÑLð	Nñó ð 	ð ˜YØ�‰ÙØØØUñ	
ð	ñó ð 	ð ˜IØÙ�Ð OÑPð	Rñó ð 	ð  	ØÙ�Ð PÑQð	Sñ!ó ð 	ð ˜	Øˆd‰
Ù�d˜qÐ.OÑPð	Rñó ð 	
ð  	ØÙØØð<ñ		
ð	ñ
!ó 

ð, 	ð �YØ�‰ÙØØðcñ		
ð	ñ
ó 
ð* 	ð ˜9ØÙØð;ñ	
ð	ñ	ó 	ð" 	ð ˜	ØÙØØUñ	
ð	ñö rH   r‘  c                   ón   • \ rS rSr% Sr\r\\\	" SS94   \
S'   \R                  r\S\	" SS94   \
S'   S	rg
)ÚVlmExtractionPipelineOptionsiî  a  Options for VLM-based structured information extraction pipeline.

Configures a pipeline that uses a vision-language model (default:
NuExtract-2B) to extract structured data fields from document images.
Unlike `VlmPipelineOptions` which converts pages to document format,
this pipeline targets extraction of specific entities or key-value pairs.

Supported models:
    - ``NU_EXTRACT_2B_TRANSFORMERS`` (default) with ``ExtractionPromptStyle.NUEXTRACT``
    - ``GRANITE_VISION_4_1_TRANSFORMERS`` with ``ExtractionPromptStyle.GRANITE_VISION``
zÁVision-Language Model (VLM) configuration for structured information extraction. Specifies which VLM to use and its parameters for extracting structured data from documents using vision models.rd   rh  r   zePrompt style to use for extraction. Determines how the template is formatted and passed to the model.Úextraction_prompt_styler?   N)r@   rA   rB   rC   rD   r5   rh  r   r+   r   rF   r   Ú	NUEXTRACTrŸ  rG   r?   rH   rI   rž  rž  î  sd   ‡ ñ
ð( 	#ð �ØÙðoñ	
ð	ñó #ð$ 	×'Ñ'ð ˜YØÙð8ñ	
ð	ñö (rH   rž  c                   óR  • \ rS rSr% SrSr\\\" SS94   \	S'   Sr
\\\" SS94   \	S	'   Sr\\\" S
S94   \	S'   Sr\\\" SS94   \	S'   Sr\\\" SS94   \	S'   Sr\\\" SSSS94   \	S'   Sr\\\   S-  \" SS94   \	S'   Sr\\\" SSSS94   \	S'   Sr\\\" SSSS94   \	S '   S!rg)"ÚHeadingHierarchyOptionsi  aþ  Options for inferring section-header levels in the PDF/image pipeline.

The layout model only flags regions as ``SECTION_HEADER`` without a level, so every
heading produced by the PDF path defaults to ``level=1`` and the document hierarchy is
flattened. When ``enabled``, :class:`HeadingHierarchyModel` runs right after the
reading-order model and assigns ``SectionHeaderItem.level`` from (in precedence order)
PDF bookmarks/ToC, numbering and font style. The step changes heading levels and may
promote a heading mis-classified as a list-item when it confidently matches a bookmark;
otherwise it never adds, removes or reorders items, and headings for which no signal
applies keep their current level.

Notes:
    - ``use_bookmarks`` reads the PDF outline surfaced on ``ConversionResult._pdf_outline``.
      When a bookmark confidently matches a detected heading it is authoritative; entries
      that match nothing fall back to numbering/style, so partial/noisy outlines never
      degrade the numbering result.
    - ``use_style`` requires the parsed PDF cells to still be available when the
      heading-hierarchy step runs, i.e.
      ``PdfPipelineOptions.generate_parsed_pages=True``. Without them, style inference is
      silently skipped (numbering still applies).
FzœEnable inference of section-header levels for the PDF/image pipeline. When disabled (default), all detected headings remain at level 1 (unchanged behavior).rd   ÚenabledTaA  Use the PDF bookmarks / table-of-contents (when present) as the authoritative heading signal. Bookmarks are fuzzily matched to detected headings by title and page; confident matches win over numbering and style, and a confidently matched list-item is promoted to a heading. Unmatched entries fall back to numbering/style.Úuse_bookmarksz›Use legal/outline numbering (e.g. PART I -> 1. -> 1.1 -> (a) -> (i), Roman vs Arabic numerals) as the primary signal for headings without a bookmark match.Úuse_numberingzÏUse the visual style of the heading (font size, and with `use_font_style` also weight, slant and letter case) as a fallback for headings without recognizable numbering. Requires `generate_parsed_pages=True`.Ú	use_styleag  Refine the style fallback with the font weight and slant read from the embedded PDF font names, plus all-caps detection, so that headings sharing a font size are still ranked (bold above regular, upright above italic, all-caps above mixed case). Ignored when `use_style` is disabled; font names that carry no recognizable styling fall back to font size alone.Úuse_font_stylerí   rz   ry   aU  Relative difference below which two heading font sizes are treated as one size by the style fallback. The size of a heading is measured from its cells, so the same font measures a little taller on a heading that has descenders; without this tolerance such headings would land on different levels. Higher = more sizes collapse into one level.)rë   Úlere   Ústyle_size_toleranceNzæOptional override of the numbering-scheme precedence (highest level first). Known schemes: 'part', 'chapter', 'article', 'roman_u', 'arabic', 'alpha_u', 'alpha_l', 'roman_l'. When None, a default legal/regulatory ordering is used.Únumbering_schemesé   ré   éd   z;Maximum heading level to assign. Deeper levels are clamped.Ú	max_levelgš™™™™™é?zÙMinimum normalized title-similarity (0..1) for a bookmark to be considered a match to a detected heading/list-item. Below this, the bookmark is ignored and the heading falls back to numbering/style. Higher = stricter.Úbookmark_match_thresholdr?   )r@   rA   rB   rC   rD   r£  r   rh   r   rF   r¤  r¥  r¦  r§  r©  r�   rª  r�   rE   r­  r¹   r®  rG   r?   rH   rI   r¢  r¢    s§  ‡ ñð> 	ð ˆYØÙðñ	
ð	ñ	ó 	ð* 	ð �9ØÙð#ñ	
ð		ñó ð( 	ð �9ØÙðcñ	
ð	ñó ð$ 	ð ˆyØÙðDñ	
ð	ñ	ó 	ð* 	ð �IØÙðEñ	
ð		ñó ð2 	ð ˜)ØÙØØð1ñ	
	
ð	ñó ð0 	ð �yØˆS‰	�DÑÙð$ñ	
ð	ñ
ó 
ð$ 	
ð ˆyØÙØØØUñ	
ð	ñó 
ð& 	ð ˜iØÙØØðPñ		
ð		ñö rH   r¢  c                   óà  • \ rS rSr% SrSr\\\" SS94   \	S'   Sr
\\\" SS94   \	S'   S	r\\\" S
S94   \	S'   S	r\\\" SS94   \	S'   S	r\\\" SS94   \	S'   \" 5       r\\\" SS94   \	S'   \" 5       r\\\" SS94   \	S'   \" \S9r\\\" SS94   \	S'   \r\\\" SS94   \	S'   Sr\\\" SS94   \	S'   S	r\\\" SS94   \	S'   S	r\\\" SS94   \	S'   S	r\\\" S S!94   \	S"'   S	r\\\" S#S94   \	S$'   \ " 5       r!\\ \" S%S94   \	S&'   S'r"\\#\" S(S94   \	S)'   S'r$\\#\" S*S94   \	S+'   S'r%\\#\" S,S94   \	S-'   S.r&\\\" S/S94   \	S0'   S1r'\\#\" S2S94   \	S3'   S4r(\\\" S5S94   \	S6'   S7r)g8)9ÚPdfPipelineOptionsi‹  a%  Configuration options for the PDF document processing pipeline.

Notes:
    - Enabling multiple features (OCR, table structure, formulas) increases the processing time significantly.
        Enable only necessary features for your use case.
    - For production systems processing large document volumes, implement a timeout protection (for instance, 90-120
        seconds via `document_timeout` parameter).
    - OCR requires a system installation of engines (Tesseract, EasyOCR). Verify the installation before enabling
        OCR via `do_ocr=True`.
    - RapidOCR has known issues with read-only filesystems (e.g., Databricks). Consider Tesseract or alternative
        backends for distributed systems.

See Also:
    - `examples/pipeline_options_advanced.py`: Comprehensive configuration examples.
TzÉEnable table structure extraction and reconstruction. Detects table regions, extracts cell content with row/column relationships, and reconstructs the logical table structure for downstream processing.rd   Údo_table_structurea  Enable Optical Character Recognition for scanned or image-based PDFs. Replaces or supplements programmatic text extraction with OCR-detected text. Required for scanned documents with no embedded text layer. Note: OCR significantly increases processing time.Údo_ocrFz¸Enable specialized processing for code blocks. Applies code-aware OCR and formatting to improve accuracy of programming language snippets, terminal output, and structured code content.Údo_code_enrichmentzÂEnable mathematical formula recognition and LaTeX conversion. Uses specialized models to detect and extract mathematical expressions, converting them to LaTeX format for accurate representation.Údo_formula_enrichmentzþForce use of PDF backend's native text extraction instead of layout model predictions. When enabled, bypasses the layout model's text detection and uses the embedded text from the PDF file directly. Useful for PDFs with reliable programmatic text layers.r#  z®Configuration for table structure extraction. Controls table detection accuracy, cell matching behavior, and table formatting. Only applicable when `do_table_structure=True`.Útable_structure_optionsz¦Configuration for OCR engine. Specifies which OCR engine to use (Tesseract, EasyOCR, RapidOCR, etc.) and engine-specific settings. Only applicable when `do_ocr=True`.Úocr_options©r„  a   Configuration for document layout analysis model. Controls layout detection behavior including cluster creation for orphaned elements, cell assignment to table structures, and handling of empty regions. Specifies which layout model to use (default: Heron).Úlayout_optionsa  Configuration for code and formula extraction using VLM. Uses new preset system (recommended). Default: 'default' preset. Only applicable when `do_code_enrichment=True` or `do_formula_enrichment=True`. Example: CodeFormulaVlmOptions.from_preset('granite_vision')Úcode_formula_optionsry   ra  rb  rc  rd  z®Extract and save embedded images from the PDF. Exports individual images (figures, photos, diagrams, charts) found in the document as separate image files for downstream use.re  z„This field is deprecated. Use `generate_page_images=True` and call `TableItem.get_image()` to extract table images from page images.r   Úgenerate_table_imagesa  Retain intermediate parsed page representations after processing. When enabled, keeps detailed page-level parsing data structures for debugging or advanced post-processing. Increases memory usage. Automatically disabled after document assembly unless explicitly enabled.Úgenerate_parsed_pageszçConfiguration for inferring section-header levels from PDF bookmarks, numbering and font style. Disabled by default; when enabled, the reading-order stage assigns SectionHeaderItem.level instead of leaving every heading at level 1.Úheading_hierarchy_optionsé   zñBatch size for OCR processing stage in threaded pipeline. Pages are grouped and processed together to improve throughput. Higher values increase GPU/CPU utilization but require more memory. Only used by `StandardPdfPipeline` (threaded mode).Úocr_batch_sizezèBatch size for layout analysis stage in threaded pipeline. Pages are grouped and processed together by the layout model. Higher values improve throughput but increase memory usage. Only used by `StandardPdfPipeline` (threaded mode).Úlayout_batch_sizezèBatch size for table structure extraction stage in threaded pipeline. Tables from multiple pages are processed together. Higher values improve throughput but increase memory usage. Only used by `StandardPdfPipeline` (threaded mode).Útable_batch_sizer    a  Polling interval in seconds for batch collection in threaded pipeline stages. Each stage waits up to this duration to accumulate items before processing. Lower values reduce latency but may decrease batching efficiency. Only used by `StandardPdfPipeline` (threaded mode).Úbatch_polling_interval_secondsr¬  a  Maximum queue size for inter-stage communication in threaded pipeline. Limits the number of items buffered between processing stages to prevent memory overflow. When full, upstream stages block until space is available. Only used by `StandardPdfPipeline` (threaded mode).Úqueue_max_sizeg      .@zãSeconds to wait for each pipeline stage thread to terminate during shutdown before it is abandoned as stuck (its resources may then leak for the rest of the process lifetime). Only used by `StandardPdfPipeline` (threaded mode).Ústage_shutdown_timeout_secondsr?   N)*r@   rA   rB   rC   rD   r±  r   rh   r   rF   r²  r³  r´  r#  ra   rµ  r]   r•   r¶  rr   r|  r¸  rk  Ú_default_code_formula_optionsr¹  r%  rb  r�   rd  re  rº  r»  r¢  r¼  r¾  r¹   r¿  rÀ  rÁ  rÂ  rÃ  rG   r?   rH   rI   r°  r°  ‹  s¬  ‡ ñð0 	ð ˜	ØÙðtñ	
ð	ñó ð$ 	ð ˆIØÙðQñ	
ð	ñ	ó 	ð$ 	ð ˜	ØÙðbñ	
ð	ñó ð" 	ð ˜9ØÙðqñ	
ð	ñó ð$ 	ð ˜	ØÙðCñ	
ð	ñ	ó 	ñ$ 	Óð ˜YØ!ÙðXñ	
ð	ñó  ñ" 	Óð �ØÙðTñ	
ð	ñó ñ$ 	Ð:Ñ;ð �IØÙðHñ	
ð	ñ	ó 	<ð& 	&ð ˜)ØÙðOñ	
ð	ñ	ó 	&ð& 	ð �)ØÙð,ñ	
ð	ñ	ó 	ð$ 	ð ˜)ØÙðYñ	
ð	ñó ð" 	ð ˜YØÙð\ñ	
ð	ñó ð" 	ð ˜9ØÙð1ñ	
ð	ñó ð$ 	ð ˜9ØÙðNñ	
ð	ñ	ó 	ñ( 	 Ó!ð ˜yØÙð&ñ	
ð	ñ
 ó 
"ð0 	
ð �IØÙð9ñ	
ð	ñ	ó 	
ð& 	
ð �yØÙð9ñ	
ð	ñ	ó 	
ð& 	
ð �iØÙð9ñ	
ð	ñ	ó 	
ð* 	ð # IØÙð[ñ	
ð	ñ	%ó 	ð( 	ð �IØÙðZñ	
ð	ñ	ó 	ð( 	ð # IØÙðAñ	
ð	ñ	%ö 	rH   r°  c                   ó,   • \ rS rSrSrSrSrSrSrSr	Sr
g	)
ÚProcessingPipelineiq  aí  Available document processing pipeline types for different use cases.

Each pipeline is optimized for specific document types and processing requirements.
Select the appropriate pipeline based on your input format and desired output.

Attributes:
    LEGACY: Legacy pipeline for backward compatibility with older document processing workflows.
    STANDARD: Standard pipeline for general document processing (PDF, DOCX, images, etc.) with layout analysis.
    NATIVE: Model-free pipeline extracting the native text and images of a PDF with docling-parse.
    VLM: Vision-Language Model pipeline for advanced document understanding using multimodal AI models.
    ASR: Automatic Speech Recognition pipeline for audio and video transcription to text.
ÚlegacyrÄ   Únativer  Úasrr?   N)r@   rA   rB   rC   rD   ÚLEGACYÚSTANDARDÚNATIVEÚVLMÚASRrG   r?   rH   rI   rÆ  rÆ  q  s"   † ñð €FØ€HØ€FØ
€CØ
ƒCrH   rÆ  c                   ó   • \ rS rSrSrSrg)ÚThreadedPdfPipelineOptionsi†  aû  Pipeline options for the threaded PDF pipeline with batching and backpressure control.

Inherits all settings from `PdfPipelineOptions`. The threaded pipeline
processes pages through concurrent stages (OCR, layout analysis, table
structure extraction) connected by bounded queues, enabling pipelined
parallelism within a single document. Batch sizes, polling intervals,
and queue limits are inherited from the parent class.

See Also:
    `PdfPipelineOptions`: Base class with all batch and queue settings.
r?   Nr_   r?   rH   rI   rÐ  rÐ  †  s   † ô
rH   rÐ  c                  óX   • [        S[        R                  " 5       =(       d    SS-
  5      $ )zJAll but one of the machine's CPU threads, so the machine stays responsive.ré   ru  )ÚmaxÚosÚ	cpu_countr?   rH   rI   Údefault_parser_threadsrÕ  ”  s   € äˆq”2—<’<“>×& Q¨!Ñ+Ó,Ð,rH   c                   ó¶   • \ rS rSr% Sr\R                  r\\\	" SS94   \
S'   \	" \S9r\\\	" SS94   \
S'   S	r\\\	" S
S94   \
S'   S	r\\\	" SS94   \
S'   Srg)ÚNativePdfPipelineOptionsi™  aØ  Pipeline options for the native (model-free) PDF pipeline.

The native pipeline reads what is already encoded in the PDF: the text cells
and the embedded bitmap images reported by docling-parse. It runs no layout,
OCR or table-structure model, so conversion is fast but the resulting
`DoclingDocument` carries one plain `TextItem` per text cell in the parser's
order, without reading order, headings or tables.

Note:
    Native picture images additionally require the PDF backend to decode the
    embedded bitmaps (`PdfBackendOptions.include_bitmap_images=True`);
    without it, pictures are still emitted, but only with their bounding box.

See Also:
    `PdfPipelineOptions`: Full PDF pipeline with layout, OCR and tables.
zæGranularity of the native text cells emitted as text items: one item per line (default), per word, or per character. The PDF backend must materialize the requested cell unit; the docling-parse backends materialize words and lines.rd   Útext_cell_unitr·  zäNumber of PDF parser worker threads. This is the only parallelism this pipeline has, since it runs no model: `accelerator_options` governs model inference and is unused here. Defaults to all but one of the machine's CPU threads.Úparser_threadsTz‡Attach the embedded bitmap images of the PDF to the picture items. Requires a PDF backend configured with `include_bitmap_images=True`.re  zçAttach a rendered image of every page to the document. Page images are produced by rasterizing the page, so the pipeline parses *and* renders each page, at `images_scale` pixels per point. Disable it to parse only, which is faster.rd  r?   N)r@   rA   rB   rC   rD   r   ÚLINErØ  r   r   rF   rÕ  rÙ  r   re  rh   rd  rG   r?   rH   rI   r×  r×  ™  sÂ   ‡ ñð4 	×Ñð �IØÙð_ñ	
ð	ñ	ó 	ñ( 	Ð4Ñ5ð �IØÙðñ	
ð	ñ
ó 
6ð& 	ð ˜YØÙðLñ	
ð	ñó ð$ 	ð ˜)ØÙð^ñ	
ð	ñ	ö 	rH   r×  )¬ÚloggingrÓ  r>  r   Úenumr   Úpathlibr   Útypingr   r   r   r	   Údocling_core.types.docr
   Údocling_core.types.doc.pager   Úpydanticr   r   r   r   r   r   r   r   Útyping_extensionsr   Údocling.datamodelr   r   r   Ú%docling.datamodel.accelerator_optionsr   r   Ú*docling.datamodel.chart_extraction_optionsr   r   Ú$docling.datamodel.extraction_optionsr   Ú#docling.datamodel.kserve_v2_optionsr   Ú$docling.datamodel.layout_model_specsr   r    r!   r"   r#   r$   r%   Ú1docling.datamodel.object_detection_engine_optionsr&   Ú0docling.datamodel.picture_classification_optionsr'   Ú,docling.datamodel.pipeline_options_asr_modelr(   Ú,docling.datamodel.pipeline_options_vlm_modelr)   r*   r+   r,   Ú#docling.datamodel.stage_model_specsr-   r.   r/   r0   Ú$docling.datamodel.vlm_engine_optionsr1   Ú!docling.datamodel.vlm_model_specsr2   r3   Ú,granite_vision_vlm_ollama_conversion_optionsr4   Ú%granite_vision_vlm_conversion_optionsr5   r6   Ú&smoldocling_vlm_mlx_conversion_optionsr7   Ú"smoldocling_vlm_conversion_optionsr8   Ú6docling.models.inference_engines.object_detection.baser9   Ú)docling.models.inference_engines.vlm.baser:   Ú	getLoggerr@   Ú_logr<   rE   rK   rV   r]   ra   rj   rn   rr   r•   r˜   r²   r»   rÊ   rÖ   rÙ   rß   rç   ró   r  r  Úsmolvlm_picture_descriptionÚgranite_picture_descriptionr  r%  Úregister_presetÚVLM_CONVERT_SMOLDOCLINGÚVLM_CONVERT_GRANITE_DOCLINGÚVLM_CONVERT_DEEPSEEK_OCRÚVLM_CONVERT_GRANITE_VISIONÚVLM_CONVERT_PIXTRALÚVLM_CONVERT_GOT_OCRÚVLM_CONVERT_PHI4ÚVLM_CONVERT_QWENÚVLM_CONVERT_NANONETS_OCR2ÚVLM_CONVERT_GEMMA_12BÚVLM_CONVERT_GEMMA_27BÚVLM_CONVERT_DOLPHINÚVLM_CONVERT_GLMOCRÚVLM_CONVERT_LIGHTONOCRÚVLM_CONVERT_FALCON_OCRÚVLM_CONVERT_CHANDRA_OCR2ÚVLM_CONVERT_UNLIMITED_OCRÚVLM_CONVERT_DOTS_OCRÚVLM_CONVERT_DOTS_MOCRÚPICTURE_DESC_SMOLVLMÚPICTURE_DESC_GRANITE_VISIONÚPICTURE_DESC_PIXTRALÚPICTURE_DESC_QWENÚCODE_FORMULA_CODEFORMULAV2ÚCODE_FORMULA_GRANITE_DOCLINGÚfrom_presetri  r^  r]  rÄ  r-  rC  rE  rN  rV  r`  rg  rk  rq  r|  r€  Ú!OBJECT_DETECTION_LAYOUT_HERON_101Ú$OBJECT_DETECTION_LAYOUT_EGRET_MEDIUMÚ#OBJECT_DETECTION_LAYOUT_EGRET_LARGEÚ$OBJECT_DETECTION_LAYOUT_EGRET_XLARGEr†  rˆ  rŒ  Ú"docling.utils.video_frame_samplingr�  r‘  rž  r¢  r°  rÆ  rÐ  r¹   rÕ  r×  r?   rH   rI   Ú<module>r     su  ðó Û 	Û Ý Ý Ý ß 4Ó 4å =Ý 4÷	÷ 	ó 	õ )÷ñ ÷ X÷õ GÝ D÷÷ ñ õõõ J÷ó ÷ó õ F÷÷ ñ õõ Là×Ò˜Ó"€ô�)ô ô ˆc�4ô ô$�c˜4ô ô"	 ô 	ô!Ð5ô !ô0	Ð7ô 	ô1Ð)Bô 1ôL*�ô L*ô^�Zô ô4u�jô uôp$
˜ô $
ôNF�Zô FôR*˜Zô *ôZ!˜*ô !ôH!�Jô !ôH.˜Ð%9ô .ôbO Kô OôdXÐ#@ô Xôv</Ð#@ô </ô~14ØÐ+Ð-Jô14ñj ;Ø1ñÐ ðñ ;Ø/Ø)ñÐ ðô%Ð(Ð*?Àô %ôPÐ,Ð.CÀYô ðN × !Ñ !Ð"3×"KÑ"KÔ LØ × !Ñ !Ð"3×"OÑ"OÔ PØ × !Ñ !Ð"3×"LÑ"LÔ MØ × !Ñ !Ð"3×"NÑ"NÔ OØ × !Ñ !Ð"3×"GÑ"GÔ HØ × !Ñ !Ð"3×"GÑ"GÔ HØ × !Ñ !Ð"3×"DÑ"DÔ EØ × !Ñ !Ð"3×"DÑ"DÔ EØ × !Ñ !Ð"3×"MÑ"MÔ NØ × !Ñ !Ð"3×"IÑ"IÔ JØ × !Ñ !Ð"3×"IÑ"IÔ JØ × !Ñ !Ð"3×"GÑ"GÔ HØ × !Ñ !Ð"3×"FÑ"FÔ GØ × !Ñ !Ð"3×"JÑ"JÔ KØ × !Ñ !Ð"3×"JÑ"JÔ KØ × !Ñ !Ð"3×"LÒ"LÔ MØ × !Ñ !Ð"3×"MÒ"MÔ NØ × !Ñ !Ð"3×"HÒ"HÔ IØ × !Ñ !Ð"3×"IÒ"IÔ Jð #× 2Ñ 2Ø×*Ò*ôð #× 2Ñ 2Ø×1Ò1ôð #× 2Ñ 2Ø×*Ò*ôð #× 2Ñ 2Ð3D×3VÒ3VÔ Wð × %Ñ %Ð&7×&RÒ&RÔ SØ × %Ñ %Ð&7×&TÒ&TÔ Uð  1×<Ò<Ð=NÓOÐ Ø Xð (J×'UÒ'UØó(Ð $ð Yð +K×*VÓ*VØ#ó+Ð 'ð Yð !6× AÒ AÀ/Ó RÐ Ø Wô��dô ð@ :ð °*ô ñ@ ØTóô��Tó óðô.B�kô BôJE&˜_ô E&ôP(Ð5ô (ôV)%Ð1ô )%ôX(˜ô (ôV(
Ð%ô (
ôVØ#Ø%Øôð@ × ,Ñ ,Ø×3Ò3ôð × ,Ñ ,Ø×7Ò7ôð × ,Ñ ,Ø×:Ò:ôð × ,Ñ ,Ø×9Ò9ôð × ,Ñ ,Ø×:Ò:ôô
" [ô "ôJÐ!?ô ô*%˜ô %õ$ Fô\˜?ô \ô~( ?ô (ôDx˜iô xôvcÐ1ô côL˜˜dô ô*Ð!3ô ð- ô -ô
9Ð7õ 9rH   