
    j5j                    d   d Z ddlmZ ddlZddlZddlZddlZddlmZm	Z	m
Z
 ddlmZ ddlmZ ddlmZ ddlmZ dd	lmZmZmZmZmZ erdd
lmZ ddlZddlZddlmZmZm Z  ddl!m"Z" ddl#m$Z$ ddl%m&Z&m'Z' ddl(m)Z) ddl*m+Z+m,Z,m-Z- ddl.m/Z/m0Z0 ddl1m2Z2m3Z3m4Z4m5Z5m6Z6m7Z7m8Z8 ddl9m:Z:m;Z;m<Z< ddl=m>Z>m?Z?m@Z@ ddlAmBZBmCZCmDZDmEZEmFZFmGZG ddlHmIZImJZJ 	 ddlKmLZL  ed      ZN ej                  eP      ZQdZR eL        dGdZSdHdIdZT	 	 	 	 	 	 	 	 	 	 dJdZUdddddd 	 	 	 	 	 	 	 	 	 	 	 dKd!ZVdLd"ZWdMd#ZX	 dN	 	 	 	 	 dOd$ZY	 dN	 	 	 	 	 dOd%ZZdPd&Z[dQd'Z\	 	 	 	 	 	 	 	 dRd(Z]dSd)Z^dTd*Z_dUd+Z`dVd,Za	 	 	 dW	 	 	 	 	 	 	 	 	 	 	 dXd-ZbdQd.ZcdQd/ZddQd0ZedYd1ZfdZd2Zg	 	 	 	 	 	 d[d3Zhd\d4ZidYd5Zj	 	 	 	 	 	 	 	 d]d6Zk	 	 	 	 	 	 d^d7Zld_d8Zm	 	 	 	 	 	 	 	 	 	 	 	 d`d9Zn	 da	 	 	 	 	 	 	 	 	 dbd:Zodcd;Zpddd<Zqded=Zrdfd>Zsdgd?Ztdhd@ZudidAZv	 	 	 	 	 	 djdBZw	 	 	 	 	 	 	 	 dkdCZx	 	 	 	 dldDZydmdEZz	 	 	 	 	 	 	 	 dndFZ{y# eM$ r d ZLY Zw xY w)oz,OCRmyPDF page processing pipeline functions.    )annotationsN)IterableIteratorSequence)suppress)BytesIO)Path)copyfileobj)TYPE_CHECKINGAnyBinaryIOTypeVarcast)
OcrElement)Image
ImageColor	ImageDraw)Executor)unpaper)PageContext
PdfContext)repair_docinfo_nuls)
OcrOptionsProcessingModeTaggedPdfMode)log_box_repairsrepair_page_boxes)DigitalSignatureErrorDpiErrorEncryptedPdfErrorInputFileErrorPriorOcrFoundErrorTaggedPDFErrorUnsupportedImageFormatError)IMG2PDF_KWARGS
Resolutionsafe_symlink)file_claims_pdfagenerate_pdfa_psspeculative_pdfa_conversion)
ColorspaceEncoding	FloatRectInkPageInfoPdfInfo)GhostscriptRasterDeviceOrientationConfidence)register_heif_openerc                      y N r6       H/var/www/html/qr/venv/lib/python3.12/site-packages/ocrmypdf/_pipeline.pyr3   r3   7   s    r7   Ti  c           	        t         j                  d       	 t        j                  |       }|5  t         j                  d       d	|j                  v rl|j                  d	   d
k  r|j                  st        j                  dg|j                     t        j                  dg|j                  d	     t#        d      |j                  s+t        j                  dg|j                     t#        d      |j$                  dv rt        d      d|j                  vr?|j$                  dk(  rt         j                  d       n|j$                  dk(  rt        d      ddd       	 t         j                  d       t&        j(                  }|j                  r3t'        j*                  t-        |j                  |j                              }t        |d      5 }t'        j.                  t1        j2                  |       f||dt4         ddd       t         j                  d       y# t        $ r}t         j                  t        |      j                  t        |       t        |j                                     | j                         st         j                  d|        | j                         rt         j                  d|        | j                         rt         j                  d|        | j                         j                  dk(  rt         j                  d|        t               |d}~ww xY w# 1 sw Y   xY w# 1 sw Y   8xY w# t&        j6                  $ r}t               |d}~ww xY w)a  Triage the input image file.

    If the input file is an image, check its resolution and convert it to PDF.

    Args:
        input_file: The path to the input file.
        output_file: The path to the output file.
        options: An object containing the options passed to the OCRmyPDF command.

    Raises:
        UnsupportedImageFormatError: If the input file is not a supported image format.
        DpiError: If the input image has no resolution (DPI) in its metadata or if the
            resolution is not credible.
    z6Input file is not a PDF, checking if it is an image...zInput file does not exist: %szInput file is a directory: %szInput file is a file: %sr   zInput file is empty: %sNzInput file is an imagedpi)`   r<   zImage size: (%d, %d)zImage resolution: (%d, %d)zInput file is an image, but the resolution (DPI) is not credible.  Estimate the resolution at which the image was scanned and specify it using --image-dpi.zInput file is an image, but has no resolution (DPI) in its metadata.  Estimate the resolution at which image was scanned and specify it using --image-dpi.)RGBALAzEThe input image has an alpha channel. Remove the alpha channel first.
iccprofileRGBz-Input image has no ICC profile, assuming sRGBCMYKz/Input CMYK image has no ICC profile, not usablez+Image seems valid. Try converting to PDF...wb)
layout_funoutputstreamz,Successfully converted to PDF, processing...)loginfor   openOSErrorerrorstrreplace
input_fileexistsis_diris_filestatst_sizer$   	image_dpisizer   modeimg2pdfdefault_layout_funget_fixed_dpi_layout_funr&   convertosfspathr%   ImageOpenError)rL   output_fileoptionsimerC   outfs          r8   triage_image_filera   D   s    HHEF3ZZ
# 
 )*BGGwwu~)'2C2C/:"'':5GGJ 
 ""HH+6bgg6F  77n$-W  rww&ww%HIF"1E 9@3>?//
 997,,g.?.?@J +t$ 	OO		*%%! !		 	?@w  3		#a&..Z#g6H6H2IJK  "II5zBII5zBII0*=??$$)II/<)+23 N	 	 !! 3)+23sU   H# D-L.$A0M 1L;M #	L+,C:L&&L+.L8;M M M+M&&M+c                    t        | d      5 }|j                  |      }ddd       t        j                  d      }|r |j	                  d      j                  d      S y# 1 sw Y   BxY w)zTry to find version signature at start of file.

    Not robust enough to deal with appended files.

    Returns empty string if not found, indicating file is probably not PDF.
    rbNs   %PDF-(\d\.\d)   ascii )rG   readresearchgroupdecode)rL   search_windowf	signaturems        r8   _pdf_guess_versionrp      sc     
j$	 *1FF=)	*
		#Y/Awwqz  ))* *s   A  A)c                   	 t        |      r|j                  rt        j                  d       	 t	        j
                  |      5 }t        |j                        D ci c]  \  }}t        |      x}r|| }}}t        |       |j                  |       ddd       |S 	 t'        |||       |S c c}}w # 1 sw Y   |S xY w# t        j                  $ r}	t               |	d}	~	wt        j                  $ r}	t               |	d}	~	ww xY w# t        $ rM}	t        j!                  d|        t#        |	      j%                  t#        |      |       }
t        |
      |	d}	~	ww xY w)z5Triage the input file. We can handle PDFs and images.zTArgument --image-dpi is being ignored because the input file is a PDF, not an image.NzTemporary file was at: )rp   rR   rE   warningpikepdfrG   	enumeratepagesr   r   savePdfErrorr!   PasswordErrorr    rH   debugrJ   rK   ra   )original_filenamerL   r\   r]   pdfnpagerepairsrepairs_by_pager_   msgs              r8   triager      sW   )j)  91\\*- * (1';'#At'8'>>G> 7
'O '
 $O4HH[)* ' *2 j+w7%'* 	 ## .$&A-(( 1')q01  )		+J<89!fnnS_.?@S!q()s}   ,D
 C B6B07B6C D
 0B66C ;C >D
  C DC!!D7DDD
 
	E AEE FT)detailed_analysisprogbarmax_workersuse_threadscheck_pagesc          	         	 t        | ||||||      S # t        j                  $ r}t               |d}~wt        j                  $ r}t               |d}~ww xY w)zGet the PDF info.)r   r   r   r   r   executorN)r0   rs   rx   r    rw   r!   )rL   r   r   r   r   r   r   r_   s           r8   get_pdfinfor      sf    &/###
 	
    )!q( &A%&s    A2AAAc                   | j                   }| j                  }|j                  rt        d      |j                  r,|j
                  rt        j                  d       n
t               |j                  ro|j                  t        j                  k(  rt        d      t        j                  d       |j                  t        j                  k7  rt        j                  d       |j                  s|j                   rnt        j                  d       |j"                  t$        j&                  k(  r<|j                  t        j&                  k(  rt        j                  d       t)               | j*                  j-                  ||       y	)
zValidate the PDF info options.z~This PDF contains dynamic XFA forms created by Adobe LiveCycle Designer and can only be read by Adobe Acrobat or Adobe Reader.z*All digital signatures will be invalidatedzgThis PDF has a user fillable form. --redo-ocr (or --mode redo) is not currently possible on such files.z_This PDF has a fillable form. Chances are it is a pure digital document that does not need OCR.zUse the option --force-ocr (or --mode force) to produce an image of the form and all filled form fields. The output PDF will be 'flattened' and will no longer be fillable.au  This PDF contains structural markup (it is a Tagged PDF or carries a logical structure tree). This often indicates that the PDF was generated from an office document or is otherwise born digital, and does not need OCR. OCRmyPDF cannot rebuild this structure to match new text, so any page it re-OCRs with --force-ocr or --redo-ocr will have its structural markup discarded.z3Use --tagged-pdf-mode ignore to ignore Tagged PDFs.)pdfinfor]   N)r   r]   needs_renderingr!   has_signatureinvalidate_digital_signaturesrE   rr   r   has_acroformrT   r   redoforcerF   	is_taggedhas_structure_treetagged_pdf_moder   defaultr#   plugin_managervalidate)contextr   r]   s      r8   validate_pdfinfo_optionsr      s;   ooGooGN
 	
 00KKDE'))<<>... ; 
 KK3
 ||~333J
 G66	
 ##}'<'<< 6 66HHJK ""##GW#Er7   c                B    | j                   s| j                  rt        S dS )zBGet a DPI to use for vector pages, if the page has vector content.r   )
has_vectorhas_textVECTOR_PAGE_DPIpageinfos    r8   _vector_page_dpir     s    &11X5F5F?MAMr7   c           	     `   | j                   }| j                  }|s|j                  }|j                  xs d}|j                  xs d}t        |j                        xs d}t        t        ||z  xs t        ||z  xs t        t        |      |j                  xs d            }t        ||      S )zqGet the DPI when we require xres == yres, scaled to physical units.

    Page DPI includes UserUnit scaling.
            g      ?)r   r]   r;   xyfloatuserunitmaxr   r   
oversampler&   )page_contextrR   r   r]   xresyresr   unitss           r8   get_page_square_dpir     s     $$H""GLL	;;#D;;#DX&&'.3HH_0H_0X&%#		
E eU##r7   c           	     
   | j                   }| j                  }|s|j                  }t        t	        |j
                  xs t        |j                  xs t        t        |      |j                  xs d            }t        ||      S )zGet the DPI when we require xres == yres, in Postscript units.

    Canvas DPI is independent of PDF UserUnit scaling, which is
    used to describe situations where the PDF user space is not 1:1 with
    the physical units of the page.
    r   )r   r]   r;   r   r   r   r   r   r   r   r&   )r   rR   r   r]   r   s        r8   get_canvas_square_dpir   4  su     $$H""GLL	KK*?KK*?X&%#		
E eU##r7   c                b   | j                   }| j                  }|j                  t        j                  k(  ryd}|j
                  rK|j                  |j
                  vr3t        j                  d|j                   d|j
                          d}n|j                  r|j                  t        j                  k(  rt        d      |j                  t        j                  k(  rt        j                  d       d}nC|j                  t        j                  k(  r:|j                  rt        j!                  d       nt        j                  d       d}n|j                  t        j"                  k(  rt        j                  d	       d}n|j$                  s|j&                  s|j                  t        j                  k(  r0|j(                  r$t        j                  d
|j(                   d       nR|j                  t        j                  k(  rt        j!                  dt*         d       nt        j                  d       d}|rp|j,                  rd|j$                  rX|j.                  |j0                  z  }||j,                  dz  kD  r-d}t        j!                  d|dz  dd|j,                  dd       |S )z$Check if the page needs to be OCR'd.FTzskipped z as requested by --pages zpage already has text! - aborting (use --force-ocr or --mode force to force OCR; see also help for --skip-text, --redo-ocr, and --mode)z@page already has text! - rasterizing text and running OCR anywayzksome text on this page cannot be mapped to characters: consider using --force-ocr (or --mode force) insteadzredoing OCRz$skipping all processing on this pagez$page has no images - rasterizing at zR DPI because --force-ocr --oversample (or --mode force --oversample) was specifiedz>page has no images - all vector content will be rasterized at za DPI, losing some resolution and likely increasing file size. Use --oversample to adjust the DPI.zpage has no images - skipping all processing on this page to avoid losing detail. Use --force-ocr (or --mode force) if you wish to perform OCR on pages that have vector content.@B zpage too big, skipping OCR (z.1fz MPixels > z MPixels --skip-big))r   r]   rT   r   
strip_textru   pagenorE   ry   r   r   r"   r   rF   r   has_corrupt_textrr   skipimageslossless_reconstructionr   r   skip_bigwidth_pixelsheight_pixels)r   r   r]   ocr_requiredpixel_counts        r8   is_ocr_requiredr   L  sJ   $$H""G||~000 L}}=		HX__--Fw}}oVW			<<>111$W  \\^111HHWXL\\^000((K
 'L\\^000HH;< L__W%D%D <<>///G4F4FHH"")"4"4!5 6XX
 \\^111KK!!0 1 2 HH2 !L((X__++h.D.DD'**Y67 LKK 9,c2+##C((<>
 r7   c                   |j                  d      }t        dd      j                  t        |      g      }t        dd      j                  t	        |      g      }|j
                  j                  | |t        j                  ||j                  j                  dz   |dd|j                  j                   |j                  d       |S )z'Generate a lower quality preview image.zrasterize_preview.jpgg     r@rd   r   F)rL   r\   raster_device
raster_dpir   page_dpirotationfilter_vectorstop_on_soft_errorr]   use_cropbox)get_pathr&   take_minr   r   r   rasterize_pdf_pager1   JPEGGRAYr   r   r]   continue_on_soft_render_error)rL   r   r\   
canvas_dpir   s        r8   rasterize_previewr     s    ''(?@KE5)22	|	,-J %'002El2S1TUH22-66$$++a/+33QQQ$$ 3  r7   c                p   ddddd}dddd	d}| j                   j                  }d
}|j                  | j                  j                  k\  r|dk7  r	d||   z   }nd}n	|dk7  rdnd}d
}|dk7  rd|j                  |d       d}|d|j                  |j                  d       z  }| d|j                  dd| S )zDDescribe the page rotation we are going to perform (or not perform).u   ⇧u   ⇨u   ⇩u   ⇦)r   Z      i   u   ⬏u   ↻u   ⬑rf   r   zwill rotate zrotation appears correctzconfidence too low to rotatez	no changezwith existing rotation ?z, zpage is facing z, confidence .2fz - )r   r   
confidencer]   rotate_pages_thresholdgetangle)r   orient_conf
correction	directionturnsexisting_rotationactionfacings           r8   describe_rotationr     s     u5u=IU7E$--66F!5!5!L!LL?#eJ&77F/F3=?/FA*9==9JC+P*QQST
	k.?.? EFGGFX];#9#9#">c&JJr7   c                :   |j                   j                  |j                        }|j                  | |j                        }|j                  dz  }t
        j                  t        |||             |j                  |j                  j                  k\  r|dk7  r|S y)a  Work out orientation correction for each page.

    We ask Ghostscript to draw a preview page, which will rasterize with the
    current /Rotate applied, and then ask OCR which way the page is
    oriented. If the value of /Rotate is correct (e.g., a user already
    manually fixed rotation), then OCR will say the page is pointing
    up and the correction is zero. Otherwise, the orientation found by
    OCR represents the clockwise rotation, or the counterclockwise
    correction to rotation.

    When we draw the real page for OCR, we rotate it by the CCW correction,
    which points it (hopefully) upright. _graft.py takes care of the orienting
    the image and text layers.
    r]   h  r   )
r   get_ocr_enginer]   get_orientationr   rE   rF   r   r   r   )previewr   
ocr_enginer   r   s        r8   get_orientation_correctionr     s     ,,;;$$ < J ,,Wl6J6JKK""S(JHH|[*EF,"6"6"M"MM!Or7   c                    | j                   }|j                         }|r1|j                  dk  r"t        |j                  |j                        }|S |j
                  }|S )z%Calculate the DPI for the page image.皙?)r   page_dpi_profileaverage_to_max_dpi_ratior&   weighted_dpir;   )r   r   dpi_profilerR   s       r8   calculate_image_dpir     s\    $$H++-K{;;cA{779Q9QR	  LL	r7   c                   t        |       }| j                  j                         }t        | |      }t	        | |      }|rI|j
                  dk  r:t        j                  d|j                  |j                  |j                                ||fS )z$Calculate the DPI for rasterization.r   zWeighted average image DPI is %0.1f, max DPI is %0.1f. The discrepancy may indicate a high detail region on this page, but could also indicate a problem with the input PDF file. Page image will be rendered at %0.1f DPI.)r   r   r   r   r   r   rE   rr   r   max_dpi	to_scalar)r   rR   r   r   r   s        r8   calculate_raster_dpir     s     $L1I''88:K&|Y?J"<;H{;;cA8 $$  "	
 xr7   c                \   t         j                  t         j                  t         j                  t         j                  gdfd}| j
                  D ]  }|j                  dk(  rh|j                  t        j                  k(  r |t         j                        n3|j                  t        j                  k(  r |t         j                        {|j                  dkD  s|j                  t        j                  k(  r |t         j                        |j                  t        j                  k(  r |t         j                         |t         j                         | j                  r<t        j!                  dt         j                           |t         j                           S )a  Choose the minimum raster device that preserves the page's color depth.

    The device escalates from 1-bit mono through grayscale, indexed, and full
    color as required by the page's images, image masks, and vector content.
    Image masks are painted with the current fill color, so a mask painted in
    gray or color escalates the device even though the mask itself is 1-bit.
    r   c                :    t        j                  |             S r5   )r   index)
colorspacecolorspaces
device_idxs    r8   at_leastz'_select_raster_device.<locals>.at_least  s    :{00<==r7   stencilrd   zPage has vector content, using )r1   PNGMONODPNGGRAYPNG256PNG16Mr   type_inkr.   colorgraybpcr+   r   r   rE   ry   )r   r   imager   r   s      @@r8   _select_raster_devicer    sI    	 ((''&&&&	K J>  F;;)# yyCII%%&=&D&DE
chh&%&=&E&EF
99q={{j...%&=&D&DE

/%&=&E&EF
%&=&D&DE
F" 		34K4R4R3STU5<<=
z""r7   c                   ||j                   j                  }|j                  d| d      }|j                  }t	        |      }t
        j                  d| d| d|j                          t        |      \  }}	|j                  j                  | ||||	|j                  dz   |||j                   j                   |j                   d       |S )	a(  Rasterize a PDF page to a PNG image.

    Args:
        input_file: The input PDF file path.
        page_context: The page context object.
        correction: The orientation correction angle. Defaults to 0.
        output_tag: The output tag. Defaults to ''.
        remove_vectors: Whether to remove vectors. Defaults to None, which means
            the value from the page context options will be used. If the value
            is True or False, it will override the page context options.

    Returns:
        Path: The output PNG file path.
    	rasterizez.pngzRasterize with z, rotation z, mediabox rd   F)rL   r\   r   r   r   r   r   r   r   r]   r   )r]   remove_vectorsr   r   r  rE   ry   mediaboxr   r   r   r   r   )
rL   r   r   
output_tagr  r\   r   devicer   r   s
             r8   r  r  8  s    * %--<<'')J<t(DEK$$H"8,FII
&ZLHDUDUCVW 0=J22"$+33QQQ$$ 3  r7   c                    t        d |j                  j                  D              rt        d      t        j                  d       | S )zBRemove the background from the input image (temporarily disabled).c              3  :   K   | ]  }|j                   d kD    yw)rd   N)r  ).0r  s     r8   	<genexpr>z/preprocess_remove_background.<locals>.<genexpr>m  s     
CU599q=
Cs   z2--remove-background is temporarily not implementedz'background removal skipped on mono page)anyr   r   NotImplementedErrorrE   rF   )rL   r   s     r8   preprocess_remove_backgroundr  k  s=    

Cl&;&;&B&B
CC!"VWW HH67r7   c           	        |j                  d      }t        |t        |            }|j                  j	                  |j
                        }|j                  | |j
                        }t        j                  |       5 }|j                  |t        j                  j                  t        j                  d|j                              }|j                  ||       ddd       |S # 1 sw Y   |S xY w)a  Deskews the input image using the OCR engine and saves the output to a file.

    Args:
        input_file: The input image file to deskew.
        page_context: The context of the page being processed.

    Returns:
        Path: The path to the deskewed image file.
    zpp_deskew.pngr   white)rT   )resample	fillcolorr;   N)r   r   r   r   r   r]   
get_deskewr   rG   rotate
ResamplingBICUBICr   getcolorrT   rv   )rL   r   r\   r;   r   deskew_angle_degreesr^   deskeweds           r8   preprocess_deskewr  v  s     ''8K
l,?,M
NC,,;;$$ < J &00\=Q=QR	J	 ,2 99 %%-- ))'@  

 	ks+, , s   >AC''C1c                    |j                  d      }t        |t        |            }t        j                  | ||j                         |j                  j                        S )z$Clean the input image using unpaper.zpp_clean.png)r;   unpaper_args)r   r   r   r   cleanr   r]   r   )rL   r   r\   r;   s       r8   preprocess_cleanr"    sS    ''7K
l,?,M
NC==MMO!))66	 r7   c           	        |j                  d      }|j                  }t        j                  |       5 }t        j                  d|j                  d          |j                  t        j                  k7  rd}|j                  t        j                  k(  rd}t        j                  |      }|j                  j                  |d      D ]  }|D cg c]  }t        |       }	}t        d |j                  d   D              }
|	d   |
d   z  |j                   |	d	   |
d
   z  z
  |	d   |
d   z  |j                   |	d
   |
d
   z  z
  f}t        j                  d|       |j#                  |d        |j$                  j'                  ||      }||}t        d |j                  d   D              }|j)                  ||       ddd       |S c c}w # 1 sw Y   |S xY w)zCreate the image we send for OCR.

    Might not be the same as the display image depending on preprocessing.
    This image will never be shown to the user.
    zocr.pngzresolution %rr;   NT)visiblecorruptc              3  8   K   | ]  }t        |      d z    yw)      R@N)r   r  coords     r8   r  z#create_ocr_image.<locals>.<genexpr>  s     Pet 3Ps   r      rd      zblanking %rr  )fill)r}   r  c              3  2   K   | ]  }t        |        y wr5   )roundr(  s     r8   r  z#create_ocr_image.<locals>.<genexpr>  s     =UE%L=s   r  )r   r]   r   rG   rE   ry   rF   rT   r   r   r   r   r   get_textareasr   tupleheight	rectangler   filter_ocr_imagerv   )r  r   r\   r]   r^   maskdrawtextareavbboxxyscale	pixcoords	filter_imr;   s                 r8   create_ocr_imager<    s    ''	2K""G	E	 %&b		/2775>2<<>/// D||~222&&r*D(11??d @  8 +33Qa33PPPGgaj(IIQ'!* 44Ggaj(IIQ'!* 44		 		-3yw78$ !//@@R A 
	  B =bggen==
%K%&L + 4#%&L s   BGG C#GGGc                    |j                  d      }|j                  d      }|j                  }|j                  j                  |      }|j	                  | |||       ||fS )z,Run the OCR engine and generate hOCR output.zocr_hocr.hocrzocr_hocr.txtr   )rL   output_hocroutput_textr]   )r   r]   r   r   generate_hocr)rL   r   hocr_outhocr_text_outr]   r   s         r8   ocr_engine_hocrrC    sr    $$_5H )).9M""G,,;;G;LJ!	   ]""r7   c                    |j                  d      }|j                  }|j                  j                  |      }|j	                  | ||j
                        \  }}|j                  |d       ||fS )a  Run the OCR engine and return OcrElement tree directly.

    This is the modern path for OCR engines that support the generate_ocr() API.
    It bypasses hOCR file generation for better performance and richer data.

    Args:
        input_file: The image file to OCR.
        page_context: The page context with options and path utilities.

    Returns:
        A tuple of (OcrElement tree, path to text sidecar file).
    zocr_direct.txtr   )rL   r]   page_numberutf-8encoding)r   r]   r   r   generate_ocrr   
write_text)rL   r   text_outr]   r   ocr_treetext_contents          r8   ocr_engine_directrN    s     $$%56H""G,,;;G;LJ'44 '' 5 Hl w7Xr7   c                h    t        | j                        xr t        d | j                  D              S )ao  Determines whether the visible page image should be saved as a JPEG.

    If all images were JPEGs originally (including FlateDecode+DCTDecode),
    permit a JPEG as output.

    Args:
        pageinfo: The PageInfo object containing information about the page.

    Returns:
        A boolean indicating whether the visible page image should be saved as a JPEG.
    c              3  t   K   | ]0  }|j                   t        j                  t        j                  fv  2 y wr5   )encr,   jpeg
flate_jpeg)r  r^   s     r8   r  z4should_visible_page_image_use_jpg.<locals>.<genexpr>  s-      );=8==("5"566)s   68)boolr   allr   s    r8   !should_visible_page_image_use_jpgrV    s2       S )AI) & r7   c                4   |j                  d      }t        j                  |       5 }d|j                  v rt	        |j                  d    }nt        |t        |            }|j                  |d|j                                ddd       |S # 1 sw Y   |S xY w)zCreate a visible page image in JPEG format.

    This is intended to be used when all images on the page were originally JPEGs.
    zvisible.jpgr;   JPEG)formatr;   N)	r   r   rG   rF   r&   r   r   rv   to_int)r  r   r\   r^   r;   s        r8   create_visible_page_jpgr[    s    
 ''6K	E	 >b BGGbggen-C &l4G4UVC 	F

=> > s   ABBc                   |j                  d      }|j                  }dt        |j                        z  dt        |j                        z  f}|j
                  |z
  dz  }|dz  dk(  }|r
|d   |d   f}t               }t        | d      5 }	t        j                  d	       t        j                  |      }
t        j                  |	|
|t        j                  j                  t        j                  j                   
       t        j                  d       ddd       |j#                  d       t%        ||||       |j&                  j)                  || |      }|S # 1 sw Y   IxY w)z$Create a PDF page from a page image.zvisible.pdfr'  r   r   r   rd   r   rc   rX   )rC   rD   enginer   zconvert doneN)	swap_axis)r}   image_filename
output_pdf)r   r   r   width_inchesheight_inchesr   r   rG   rE   ry   rU   get_layout_funrX   Enginers   Rotationifvalidseekfix_pagepdf_boxesr   filter_pdf_page)r  r   orientation_correctionr\   r   pagesizeeffective_rotationr^  bioimfilerC   s              r8   create_pdf_page_from_imagero  )  sN    ''6K$$HeH1122D5AWAW;X4XXH"++.DDK"S(B.IA;+ )C	eT	 "f		)++H5
!>>))%%--	
 			.!" HHQKc;	J--==%K > K )" "s   
B
EE%c                    |j                  d      }|j                  d      }|j                  }|j                  j                  |      }|j	                  | |||       ||fS )zBRun the OCR engine and generate a text-only PDF (will look blank).zocr_tess.pdfzocr_tess.txtr   )rL   r`  r?  r]   )r   r]   r   r   generate_pdf)input_imager   r`  r?  r]   r   s         r8   ocr_engine_textonly_pdfrs  U  st     &&~6J''7K""G,,;;G;LJ	   {""r7   c                V    | d   |d   z   | d   |d   z   | d   |d   z   | d   |d   z   fS )z%Offset a rectangle by a given amount.r   rd   r+  r*  r6   )rectoffsets     r8   _offset_rectrw  g  sN     	Q&)Q&)Q&)Q&)	 r7   c                    ||k(  ry t        ||      }|r|d   |d   |d   |d   f}|| |<   t        j                  t        |       d|        y )Nrd   r   r*  r+  z = )rw  rE   ry   rJ   )r}   	media_boxname
target_boxrv  r^  boxs          r8   _adjust_pageboxr}  q  s`     J
z6
*C!fc!fc!fc!f,DJIIT3zl+,r7   c                $   t        j                  |       5 }|j                  D ]  }t        j	                  d|j
                   d|j                  j                          |j                  j                  }|d    |d    f}|r|d   |d   |d   |d   f}g d}|D ]J  }	t        ||t        j                  d|	       t        |j                  |	j                               ||       L  |j                  |       d	d	d	       |S # 1 sw Y   |S xY w)
a  Fix the bounding boxes in a single page PDF.

    The single page PDF is created with a normal MediaBox with its lower left corner
    at (0, 0). infile is the single page PDF. page_context.mediabox has the original
    file's mediabox, which may have a different origin. We need to adjust the other
    boxes in the single page PDF to match the effect they had on the original page.

    When correcting page rotation, we create a single page PDF that is correctly
    rotated instead of an incorrectly rotated and then setting page.Rotate on it.
    If rotation is either 90 or 270 degrees, then this function can be called
    with swap_axis to swap the X and Y coordinates of all the boxes.

    We are not concerned with solving degenerate cases where the boxes overlap or
    or express invalid rectangles. We merely pass the boxes, producing a
    transformation equivalent to the change made by constructing a new page image.
    zinitial mediabox=z and pageinfo mediabox=r   rd   r*  r+  )CropBoxTrimBoxArtBoxBleedBox/N)rs   rG   ru   rE   ry   MediaBoxr   r  r}  Namegetattrlowerrv   )
infileout_filer   r^  r{   r}   r  rv  boxesbox_names
             r8   rh  rh    s   , 
f	 II 	DII#DMM? 3(11::;= $,,55Hqk\HQK</F#A;Xa[(1+M@E! LL1XJ0L118>>3CD	( 	+, O-, Os   C%DDc                >    | j                  d      }t        |       |S )zGenerates a PostScript file stub for the given PDF context.

    Args:
        context: The PDF context to generate the PostScript file stub for.

    Returns:
        Path: The path to the generated PostScript file stub.
    zpdfa.ps)r   r)   )r   r\   s     r8   generate_postscript_stubr    s"     ""9-K[!r7   c           
     z   |j                   }|j                  }|j                  d      }|j                  d      }t        j                  |       5 }t        |      r|j                  |       nt        | |       ddd       |j                  j                  d      r1|j                  dk(  rd}n!|j                  j                  d      d   }nd}|j                  j                  |j                  |g|||||j                  r|j                  j                         nd|j                           |S # 1 sw Y   xY w)	a  Converts the given PDF to PDF/A.

    Args:
        input_pdf: The input PDF file path (presumably not PDF/A).
        input_ps_stub: The input PostScript file path, containing instructions
            for the PDF/A generator to use.
        context: The PDF context.
    zfix_docinfo.pdfzpdfa.pdfNpdfa2-)pdf_version	pdf_pagespdfmarkr\   r   	pdfa_partprogressbar_classr   )r]   r   r   rs   rG   r   rv   r'   output_type
startswithsplitr   generate_pdfamin_versionprogress_barget_progressbar_classr   )		input_pdfinput_ps_stubr   r]   input_pdfinfofix_docinfo_filer\   pdf_filer  s	            r8   convert_to_pdfar    s6    ooGOOM''(9:"":.K 
i	  6Hx(MM*+$45	6 %%f-&(I++11#6I 	((!--#$ ## ""88:&DDD )  A6 6s   *D11D:c                n   ddl m} |j                  }t        |dd      }|)t        |dd      }|dk7  rt        j                  d|       y|j                         st        j                  d       y|j                  d	      }	 t        | ||j                         |j                  |j                        }|j                  ||      }|j                  rt        j                  d
       |S t        j                  d|j                         y# t        $ r }	t        j                  d|	       Y d}	~	yd}	~	ww xY w)a  Try speculative PDF/A conversion with verapdf validation.

    This attempts a fast PDF/A conversion by adding PDF/A structures
    directly with pikepdf, then validating with verapdf. If validation
    passes, returns the converted file. If it fails or verapdf is not
    available, returns None to signal that Ghostscript should be used.

    Args:
        input_pdf: Path to the PDF to convert
        context: The PDF context

    Returns:
        Path to valid PDF/A file, or None if speculative conversion failed
    r   verapdfghostscriptNpdfa_image_compressionautozLSkipping speculative PDF/A: --pdfa-image-compression=%s requires Ghostscriptz<verapdf not available, skipping speculative PDF/A conversionzspeculative_pdfa.pdfz=Speculative PDF/A conversion succeeded - skipping GhostscriptzUSpeculative PDF/A validation failed (%d rule violations), falling back to Ghostscriptz'Speculative PDF/A conversion failed: %s)ocrmypdf._execr  r]   r  rE   ry   	availabler   r*   r  output_type_to_flavourr   validrF   failed_rules	Exception)
r  r   r  r]   gs_optscompressionr\   flavourresultr_   s
             r8   try_speculative_pdfar    s    'ooG g}d3Gg'?H& II
 		PQ""#9:K#I{G<O<OP001D1DE!!+w7<<HHTUII.##
  		;Q?s   A&D * D 	D4D//D4c                   ddl m} |j                         r+t        | |      }||dfS t        j                  d       | dfS t        | |j                        rt        j                  d       | dfS t        j                  d       | dfS )a\  Best-effort PDF/A for 'auto' output type.

    This function attempts to produce PDF/A without requiring Ghostscript:
    1. If verapdf is available, tries speculative conversion with validation
    2. Without verapdf, passes through as PDF/A if safe (input already PDF/A
       or force-ocr was used)
    3. Falls back to regular PDF if neither condition is met

    Args:
        input_pdf: Path to the PDF to convert
        context: The PDF context

    Returns:
        Tuple of (output_path, actual_output_type) where actual_output_type
        is 'pdfa' if PDF/A was achieved, 'pdf' otherwise
    r   r  r  zFAuto mode: speculative PDF/A validation failed, outputting regular PDFr{   z=Auto mode: passing through as PDF/A (input already compliant)zFAuto mode: no verapdf available and input is not PDF/A, outputting PDF)r  r  r  r  rE   rF   _is_safe_pdfar]   )r  r   r  r  s       r8   try_auto_pdfar  0  s    " ' %i9F##T	
 5!! Y0PQ6"" HHUVur7   c                ^    t        |       }|d   ry|j                  t        j                  k(  S )a  Check if file can be considered PDF/A without validation.

    These are cases where our modifications don't break PDF/A compliance:
    1. Input already claims PDF/A (we just grafted OCR text onto it)
    2. We used force-ocr (we rewrote the entire PDF from scratch)

    Args:
        input_pdf: Path to the PDF to check
        options: OCR options

    Returns:
        True if file can safely be considered PDF/A
    passT)r(   rT   r   r   )r  r]   pdfa_statuss      r8   r  r  Y  s0     #9-K6 <<>////r7   c                x    t        j                  |       j                  }||j                  j                  dz  kD  S )zsDetermine whether the PDF should be linearized.

    For smaller files, linearization is not worth the effort.
    r   )rY   rP   rQ   r]   fast_web_view)working_filer   filesizes      r8   should_linearizer  p  s2    
 ww|$,,Hw44y@AAr7   c                    | dk(  r?t        ddt        j                  j                  t        j                  j
                        S t        ddt        j                  j                        S )zGet pikepdf.Pdf.save settings for the given output type.

    Essentially, don't use features that are incompatible with a given
    PDF/A specification.
    zpdfa-1T)preserve_pdfacompress_streamsstream_decode_levelobject_stream_mode)r  r  r  )dictrs   StreamDecodeLevelgeneralizedObjectStreamModedisablegenerate)r  s    r8   get_pdf_save_settingsr  y  sc     h ! ' 9 9 E E&77??	
 	
 ! ' 8 8 A A
 	
r7   c                    | j                         j                  }|j                         j                  }|dk(  ry||z  }d||z  z
  }||fS )a  Calculate ratio of input to output file sizes and percentage savings.

    Args:
        input_file (Path): The path to the input file.
        output_file (Path): The path to the output file.

    Returns:
        tuple[float | None, float | None]: A tuple containing the file size
        ratio and the percentage savings achieved by the output file size
        compared to the input file size.
    r   NNrd   )rP   rQ   )rL   r\   
input_sizeoutput_sizeratiosavingss         r8   _file_size_ratior    sX     "**J""$,,Ka$E+
**G'>r7   c           
     R   |j                  d      }|j                  j                  | |||t        | |            \  }}t	        | |      \  }}|rt
        j                  d|dd|d       t	        |j                  |      \  }}|rt
        j                  d|dd|d       ||fS )zOptimize the given PDF file.zoptimize.pdf)r  r`  r   r   	linearizezImage optimization ratio: r   z
 savings: z.1%zTotal file size ratio: )r   r   optimize_pdfr  r  rE   rF   origin)rL   r   r   r\   r`  messagesr  r  s           r8   r  r    s     "">2K"11>>":w7 ? J &j+>NE7-eC[
GS/RS%gnnkBNE7*5+Z#OPxr7   c              #     K   d\  }}t        |       D ])  \  }}|dz  }|r|||dz
  fdf d}||f|f %|(|}+ |	||fdf yyw)ar  Enumerate the ranges of non-empty elements in an iterable.

    Compresses consecutive ranges of length 1 into single elements.

    Args:
        iterable: An iterable of elements to enumerate.

    Yields:
        A tuple containing a range of indices and the corresponding element.
        If the element is None, the range represents a skipped range of indices.
    r  rd   N)rt   )iterableskipped_fromr   txt_files       r8   enumerate_compress_rangesr    s      %L%$X. 	%x
'#UQY/55#%.(**#$	% U#T))  s
   8AAc                |   |j                  d      }t        |dd      5 }t        |       D ]w  \  \  }}}|dk7  r|j                  d       |r3|j	                  d      }|j                  |j                  d             T||k7  r| d| n| }|j                  d| d	       y 	 d
d
d
       |S # 1 sw Y   |S xY w)a   Merge the page sidecar files into a single file.

    Sidecar files are created by the OCR engine and contain the text for each
    page in the PDF. This function merges the sidecar files into a single file
    and returns the path to the merged file.
    zsidecar.txtwrF  rG  rd   r  z[OCR skipped on page(s) ]N)r   rG   r  write	read_textremovesuffix)		txt_filesr   r\   streamfrom_to_r  txtru   s	            r8   merge_sidecarsr    s     ""=1K	k3	1 BV&?	&J 
	B"LUC(zT"(('(: S--d34,1SL5'3%(7wa@A
	BB B s   BB11B;c                "   t         j                  d| |       | j                  d      5 }|dk(  rCt        |t        j
                  j                         t        j
                  j                          nrt        |d      rEt        t        |      }t        ||       t        t              5  |j                          ddd       n!t        |d      5 }t        ||       ddd       ddd       y# 1 sw Y   xY w# 1 sw Y   xY w# 1 sw Y   yxY w)a.  Copy the final temporary file to the output destination.

    Args:
        input_file (Path): The intermediate input file to copy.
        output_file (str | Path | BinaryIO): The output file to copy to.
        original_file: The original file to copy attributes from.

    Returns:
        None
    z%s -> %src   r  writableNzw+b)rE   ry   rG   r
   sysstdoutbufferflushhasattrr   r   r   AttributeError)rL   r\   original_fileinput_streamoutput_streams        r8   
copy_finalr    s     IIj*k2		 9,#cjj&7&78JJ[*- ;7Mm4.) &##%& & k5) 9]L-899 9& &9 99 9s<   B D)C-:DC9D-C6	2D9D	>DD)rL   r	   r\   r	   r]   r   returnNone)i   )rL   r	   r  rJ   )
rz   rJ   rL   r	   r\   r	   r]   r   r  r	   )r   r   r   rT  r   rT  r   z
int | Noner   rT  r  r0   )r   r   r  r  )r   r/   r  intr5   )r   r   rR   zResolution | Noner  r&   )r   r   r  rT  )rL   r	   r   r   r  r	   )r   r   r   r2   r   r  r  rJ   )r   r	   r   r   r  r  )r   r   r  r&   )r   r   )r   r/   r  r1   )r   rf   N)rL   r	   r   r   r   r  r	  rJ   r  zbool | Noner  r	   )r  r	   r   r   r  r	   )rL   r	   r   r   r  tuple[Path, Path])rL   r	   r   r   r  ztuple[OcrElement, Path])r   r/   r  rT  )r  r	   r   r   rj  r  r  r	   )rr  r	   r   r   r  r  )ru  z!tuple[float, float, float, float]rv  tuple[float, float])r}   zpikepdf.Pagery  r-   rz  zpikepdf.Namer{  r-   rv  r  r^  rT  )F)
r  zPath | BinaryIOr  r	   r   r   r^  rT  r  r	   )r   r   r  r	   )r  r	   r  r	   r   r   r  r	   )r  r	   r   r   r  Path | None)r  r	   r   r   r  ztuple[Path, str])r  r	   r  rT  )r  r	   r   r   r  rT  )r  rJ   r  zdict[str, Any])rL   r	   r\   r	   r  z!tuple[float | None, float | None])rL   r	   r   r   r   r   r  ztuple[Path, Sequence[str]])r  zIterable[T]r  z*Iterator[tuple[tuple[int, int], T | None]])r  zIterable[Path | None]r   r   r  r	   )rL   r	   r\   zstr | Path | BinaryIOr  r  r  r  )|__doc__
__future__r   loggingrY   rh   r  collections.abcr   r   r   
contextlibr   ior   pathlibr	   shutilr
   typingr   r   r   r   r   ocrmypdf.hocrtransformr   rU   rs   PILr   r   r   ocrmypdf._concurrentr   r  r   ocrmypdf._jobcontextr   r   ocrmypdf._metadatar   ocrmypdf._optionsr   r   r   ocrmypdf._pageboxesr   r   ocrmypdf.exceptionsr   r   r    r!   r"   r#   r$   ocrmypdf.helpersr%   r&   r'   ocrmypdf.pdfar(   r)   r*   ocrmypdf.pdfinfor+   r,   r-   r.   r/   r0   ocrmypdf.pluginspecr1   r2   pi_heifr3   ImportErrorr9   	getLogger__name__rE   r   ra   rp   r   r   r   r   r   r   r   r   r   r   r   r   r  r  r  r  r"  r<  rC  rN  rV  r[  ro  rs  rw  r}  rh  r  r  r  r  r  r  r  r  r  r  r  r  r6   r7   r8   <module>r     s  
 3 "  	 	 
 8 8     > >1   , , ) " 8 2 G G B   F E 
 U T N, CLg!  O3d(,;?JT	L $"& & 	&
 & & & &61FhN ?C$$*;$$4 ?C$$*;$$0Nb.KK,AKORKK4> *(#\ "&000 0 	0
  0 
0f>	.b# $/>".))*)DG)	)X##%0##$-
-- - 	-
  - -* 	,,, , 	,
 
,^3l8v&R0.B
.#'&.  ) 5=  ,**/*:.99#89IT9	9y%  s   H# #H/.H/