ó
    ôMóiÇ  ã                  ó>  • S SK Jr  S SKJrJrJr  S SKJr  S SKJ	r	J
r
  S SKJr  S SKJr  S SKJr  S SKJr  S S	KJr  \
" 5       \\S
S
SSS
S
S\R*                  S
S
S
SS
S
SSSSS
4                                         SS jj5       5       5       rg
)é    )Úannotations)ÚIOÚAnyÚOptional)Úadd_chunking_strategy)ÚElementÚprocess_metadata)Úadd_metadata)Úexactly_one)Úcheck_language_args)Úpartition_pdf_or_image)ÚPartitionStrategyNFé   Tc                óÀ   • [        XS9  [        U=(       d    / U5      n[        S0 SU _SU_SS_SU_SU_SU_S	U_S
U_SU_SU
_SU_SU_SU_SU_SU_SU_SU_SU_UD6$ )aö  Parses an image into a list of interpreted elements.

Parameters
----------
filename
    A string defining the target filename path.
file
    A file-like object as bytes --> open(filename, "rb").
include_page_breaks
    If True, includes page breaks at the end of each page in the document.
infer_table_structure
    Only applicable if `strategy=hi_res`.
    If True, any Table elements that are extracted will also have a metadata field
    named "text_as_html" where the table's text content is rendered into an html string.
    I.e., rows and cells are preserved.
    Whether True or False, the "text" field is always present in any Table element
    and is the text content of the table (no structure).
languages
    The languages present in the document, for use in partitioning and/or OCR. To use a language
    with Tesseract, you'll first need to install the appropriate Tesseract language pack.
strategy
    The strategy to use for partitioning the image. Valid strategies are "hi_res" and
    "ocr_only". When using the "hi_res" strategy, the function uses a layout detection
    model if to identify document elements. When using the "ocr_only" strategy,
    partition_image simply extracts the text from the document using OCR and processes it.
    The default strategy is `hi_res`.
metadata_last_modified
    The last modified date for the document.
hi_res_model_name
    The layout detection model used when partitioning strategy is set to `hi_res`.
extract_images_in_pdf
    Only applicable if `strategy=hi_res`.
    If True, any detected images will be saved in the path specified by
    'extract_image_block_output_dir' or stored as base64 encoded data within metadata fields.
    Deprecation Note: This parameter is marked for deprecation. Future versions will use
    'extract_image_block_types' for broader extraction capabilities.
extract_image_block_types
    Only applicable if `strategy=hi_res`.
    Images of the element type(s) specified in this list (e.g., ["Image", "Table"]) will be
    saved in the path specified by 'extract_image_block_output_dir' or stored as base64 encoded
    data within metadata fields.
extract_image_block_to_payload
    Only applicable if `strategy=hi_res`.
    If True, images of the element type(s) defined in 'extract_image_block_types' will be
    encoded as base64 data and stored in two metadata fields: 'image_base64' and
    'image_mime_type'.
    This parameter facilitates the inclusion of element data directly within the payload,
    especially for web-based applications or APIs.
extract_image_block_output_dir
    Only applicable if `strategy=hi_res` and `extract_image_block_to_payload=False`.
    The filesystem path for saving images of the element type(s)
    specified in 'extract_image_block_types'.
extract_forms
    Whether the form extraction logic should be run
    (results in adding FormKeysValues elements to output).
form_extraction_skip_tables
    Whether the form extraction logic should ignore regions designated as Tables.
password
    The password to decrypt the PDF file.
)ÚfilenameÚfiler   r   Úis_imageTÚinclude_page_breaksÚinfer_table_structureÚ	languagesÚdetect_language_per_elementÚstrategyÚmetadata_last_modifiedÚhi_res_model_nameÚextract_images_in_pdfÚextract_image_block_typesÚextract_image_block_output_dirÚextract_image_block_to_payloadÚstarting_page_numberÚextract_formsÚform_extraction_skip_tablesÚpassword© )r   r   r   )r   r   r   r   Úocr_languagesr   r   r   r   Úchunking_strategyr   r   r   r   r   r   r    r!   r"   Úkwargss                       Úv/var/www/eduai.edurigo.com/storigo/production/storigo_env/lib/python3.13/site-packages/unstructured/partition/image.pyÚpartition_imager(      sá   € ôj ˜Ò-ä# I§O°°]ÓC€Iä!ò Ùðáðñ ðñ 0ð	ñ
 4ðñ ðñ %@ðñ ðñ  6ðñ ,ðñ 4ðñ #<ðñ (Fðñ (Fðñ 2ðñ  $ð!ñ" %@ð#ñ$ Ø
ñ'ð ó    )*r   úOptional[str]r   zOptional[IO[bytes]]r   Úboolr   r+   r$   r*   r   úOptional[list[str]]r   r+   r   Ústrr   r*   r%   r*   r   r*   r   r+   r   r,   r   r*   r   r+   r   Úintr    r+   r!   r+   r"   r*   r&   r   Úreturnzlist[Element])Ú
__future__r   Útypingr   r   r   Úunstructured.chunkingr   Úunstructured.documents.elementsr   r	   Ú unstructured.file_utils.filetyper
   Ú$unstructured.partition.common.commonr   Ú"unstructured.partition.common.langr   Úunstructured.partition.pdfr   Ú&unstructured.partition.utils.constantsr   ÚHI_RESr(   r#   r)   r'   Ú<module>r:      sn  ðÝ "ç $Ñ $å 7ß EÝ 9Ý <Ý BÝ =Ý Dñ ÓØØà"Ø $Ø %Ø"'Ø#'Ø%)Ø(-Ø%×,Ñ,Ø,0Ø'+Ø'+Ø"'Ø59Ø48Ø+0Ø !ØØ(,Ø"ð'jØðjà
ðjð ðjð  ð	jð
 !ðjð #ðjð "&ðjð ðjð *ðjð %ðjð %ðjð  ðjð  3ðjð %2ðjð %)ðjð  ð!jð" ð#jð$ "&ð%jð& ð'jð( ð)jð* ô+jó ó ó ñjr)   