
    jj                         S r SSKrSSKrSSKrSSKJr  SSKJr  SSKJ	r	  SSK
JrJrJrJrJr  S\S\4S	 jrS
\S\4S jrS\4S jrS rS r\S:X  a  \" 5         gg)a  
training_ppt.py
===============
Training module for PPT/PPTX presentation formats.

Handles text extraction from PowerPoint presentations (.ppt, .pptx),
semantic chunking, and FAISS embedding creation.

- .pptx: Extracted directly using python-pptx
- .ppt:  Converted to .pptx via LibreOffice, then extracted

Shared utilities (semantic chunking, embedding creation, FAISS merging)
are imported from training.py to maintain DRY principles.
    N)Presentation)Document)OllamaEmbeddings) split_text_with_semantic_chunkercount_tokens_in_documentssave_documents_to_txtcreate_and_save_embeddingsmerge_all_faiss	file_pathreturnc                    [        U 5      n/ n[        UR                  SS9 GH  u  p4/ nUR                   H  n[	        US5      (       aH  UR
                  R                  5       (       a)  UR                  UR
                  R                  5       5        UR                  (       d  Mo  UR                  R                   HJ  nUR                   H7  nUR
                  R                  5       n	U	(       d  M&  UR                  U	5        M9     ML     M     UR                  (       ad  UR                  R                  (       aI  UR                  R                  R
                  R                  5       n
U
(       a  UR                  SU
 35        U(       d  GMj  UR                  SU S35        UR                  U5        GM     SR!                  U5      $ )z
Extract all text content from a .pptx file using python-pptx.

Iterates through all slides and shapes to capture text from:
- Text boxes and titles
- Tables within slides
- Grouped shapes (recursive extraction)
- Notes sections
   )starttextz[Notes] z
--- Slide z ---
)r   	enumerateslidesshapeshasattrr   stripappend	has_tabletablerowscellshas_notes_slidenotes_slidenotes_text_frameextendjoin)r   prs
text_parts	slide_idxslideslide_textsshaperowcell	cell_text
notes_texts              D/var/www/eduai.edurigo.com/question_generate/staging/training_ppt.py_extract_text_from_pptxr,   &   sT    y
!CJ%cjj:	\\Euf%%%***:*:*<*<""5::#3#3#56  ;;++C #		$(IIOO$5	$9'..y9 !* , "   U%6%6%G%G**;;@@FFHJ""Xj\#:;;
9+T:;k*3 ;6 99Z      ppt_file_pathc           	      X   [         R                  R                  U 5      =(       d    Sn [        R                  " SSSSSUU /SSSS	9nUR
                  S
:w  a%  [        SUR
                   SUR                   35      e [         R                  R                  [         R                  R                  U 5      5      S
   n[         R                  R                  X S35      n[         R                  R                  U5      (       d  [        SU 35      eU$ ! [         a    [        S5      ef = f)z
Convert a legacy .ppt file to .pptx using LibreOffice headless mode.

Returns the path to the newly created .pptx file.
Raises RuntimeError if LibreOffice conversion fails.
.libreofficez
--headlessz--convert-topptxz--outdirTx   )capture_outputr   timeoutr   z)LibreOffice conversion failed (exit code z): zYLibreOffice is not installed or not in PATH. It is required to convert legacy .ppt files..pptxz:LibreOffice conversion produced no output file. Expected: )ospathdirname
subprocessrun
returncodeRuntimeErrorstderrFileNotFoundErrorsplitextbasenamer    exists)r.   
output_dirresult	base_name	pptx_paths        r+   _convert_ppt_to_pptxrG   Q   s.    /63J
J  
 !;F<M<M;Nc==/#  "   !1!1-!@A!DIZ;e)<=I77>>)$$"%
 	

 !  
;
 	

s   AD D)c                 r   [         R                  R                  U 5      S   R                  5       nSn US:X  a  [	        U 5      nO,US:X  a  [        U 5      n[	        U5      nO[        SU S35      eUR                  5       (       da  [        SU  35        / U(       aJ  [         R                  R                  U5      (       a%  [         R                  " U5        [        SU 35        $ $ $ [        UXS	.S
9/n[        SU S[        U5       S35        UU(       aJ  [         R                  R                  U5      (       a%  [         R                  " U5        [        SU 35        $ $ $ ! U(       aJ  [         R                  R                  U5      (       a%  [         R                  " U5        [        SU 35        f f f = f)a0  
Load a PPT or PPTX file and return a list of LangChain Document objects.

For .pptx files, text is extracted directly using python-pptx.
For .ppt files, the presentation is first converted to .pptx via LibreOffice.

Returns:
    list[Document]: List of LangChain Document objects with page_content set.
r   Nr6   z.pptzUnsupported file type: z$. Only .ppt and .pptx are supported.z(Warning: No text content extracted from z%Cleaned up temporary converted file: )sourceformat)page_contentmetadatazSuccessfully loaded z presentation: z characters extracted.)r7   r8   r@   lowerr,   rG   
ValueErrorr   printrB   remover   len)r   file_extensionconverted_pptx_pathr   	documentss        r+   load_pptrU      s    WW%%i0399;N#QW$*95Dv%"6y"A*+>?D).)9 :5 6 
 zz||<YKHI$ 277>>2E#F#FII)*9:M9NOP $G !$-H
	 	">"2/4yk/1	
  277>>2E#F#FII)*9:M9NOP $G277>>2E#F#FII)*9:M9NOP $Gs   A"E# ()E# #AF6c                 l    SnU  H+  nUR                   R                  5       nU[        U5      -  nM-     U$ )zx
Count total tokens across all Document objects.
Uses whitespace tokenization for consistency with the existing system.
r   )rK   splitrQ   )rT   total_tokensdoctokenss       r+   count_tokens_in_ppt_documentsr[      s=    
 L!!'')F#  r-   c                     Sn SnSnSU SU 3n[         R                   " 5       n[        SU 35        [        U 5      nU(       d  [        S5        g[        U5      n[        S	U 35        [	        S
S9n[        XW5      n[        X5        [        XU5        [        X5        [         R                   " 5       n	[        SX-
  S S35        [        SU 35        g)z/Test the PPT/PPTX training pipeline standalone.ztest_presentation.pptxtest_clienttest_refztemp/_zStart Time: zNo documents loaded. Exiting.NzTotal tokens: znomic-embed-text)modelzTraining process took z.2fz secondszTotal tokens processed: )	timerO   rU   r[   r   r   r   r	   r
   )
r   	client_idreference_idrC   
start_timedocstoken_count
embeddingssplit_documentsend_times
             r+   mainrj      s    (IIL1\N3JJ	L
%& ID-./5K	N;-
() "(:;J6tHO /6 <H I,yy{H	"8#8"=X
FG	$[M
23r-   __main__)__doc__r7   r:   ra   r2   r   langchain_core.documentsr   langchain_ollamar   trainingr   r   r   r	   r
   strr,   rG   rU   r[   rj   __name__ r-   r+   <module>rs      s~    
    - - (!s (!s (!V- - -h0Q 0Qn	 "4J zF r-   