
    ZiM                     &   S SK rS SKrS SKrS SKrS SKJr  S SKJr  S SK	r
S SKJr  SSKJr  SrSrS\\   4S	 jrS\4S
 jr\R(                  S:X  a  \" 5       (       a  SrO S SKrSrS\
R0                  S\4S jr " S S\5      rg! \ a     N)f = f)    N)Path)Optional)UnknownMethod   )ShellParserFreturnc                      [         R                  R                  S5      n U b  U R                  (       d  g U R                   H4  n[	        [        U5      R                  S5      5      =n(       d  M/  US   s  $    g )Npocketsphinxz!_pocketsphinx.cpython-*-darwin.sor   )	importlibutil	find_specsubmodule_search_locationslistr   glob)speclocmatchess      p/var/www/eduai.edurigo.com/question_generate/ques_gen_env/lib/python3.13/site-packages/textract/parsers/audio.py_find_pocketsphinx_sor      sc    >>##N3D|4::..49>>*MNOO7O1: /     c                  z    [        5       n U c  g[        R                  " SS[        U 5      /SS9R                  S:g  $ )NFcodesignz-vT)capture_outputr   )r   
subprocessrunstr
returncode)sos    r   _pocketsphinx_signature_invalidr      s<    		 B	z
D#b'24HSSWXXr   darwinTa  pocketsphinx has an invalid macOS code signature. This happens when the package is compiled from source (no prebuilt wheel for this Python version) and scikit-build-core's post-build tools invalidate the signature.

Fix with:
  make sync         # if you have the source checked out
  python -c "import subprocess, sys; from pathlib import Path; import importlib.util; spec = importlib.util.find_spec('pocketsphinx'); [subprocess.run(['codesign', '-s', '-', '-f', str(so)]) for loc in (spec.submodule_search_locations or []) for so in Path(loc).glob('_pocketsphinx.cpython-*-darwin.so')]"

See: https://github.com/scikit-build/scikit-build-core/issues for upstream status.audioc                 b   [         c%  [        (       a  [        [        5      e[	        S5      eU R                  SSS9n[         R                  " [        R                  S9nUR                  5         UR                  USSS9  UR                  5         UR                  5       nUb  UR                  $ S	$ )
zBTranscribe audio using pocketsphinx 5.x with bundled en-US models.sphinxi>     )convert_rateconvert_width)logfnFT)	no_searchfull_utt )_pocketsphinx_POCKETSPHINX_CODESIGN_ISSUERuntimeError_CODESIGN_FIX_MSGr   get_raw_dataDecoderosdevnull	start_uttprocess_rawend_utthyphypstr)r!   raw_datadecoderr6   s       r   _sphinx_transcriber:   >   s    ''011H%%!!uA!FH##"**5GEDAOO
++-C3::0b0r   c                   (    \ rS rSrSrSS jrS rSrg)ParserM   a:  
Extract text (i.e. speech) from an audio file, using SpeechRecognition.

Since SpeechRecognition expects a .wav file, with 1 channel,
the audio file has to be converted, via sox, if not compliant

Note: for testing, use -
http://www2.research.att.com/~ttsweb/tts/demo.php,
with Rich (US English) for best results
c                    Sn[         R                  R                  U5      u  pVUS:w  a@  U R                  U5      n U R                  " Xr40 UD6n[        U5      R                  5         U$ [        R                  " 5       n[        R                  " U5       n	UR                  U	5      n
S S S 5         US;   a  UR                  W
5      nOUS:X  a  [        W
5      nO[        U5      eUS-  nU$ ! [        U5      R                  5         f = f! , (       d  f       Ni= f! [         a    Sn ND[        R                   a    Sn N[[        R                    a    Sn Nrf = f)Nr*   .wav>   r*   googler#   
)r1   pathsplitextconvert_to_wavextractr   unlinksr
RecognizerWavFilerecordrecognize_googler:   r   LookupErrorUnknownValueErrorRequestError)selffilenamemethodkwargsspeechbaseexttemp_filenamersourcer!   s              r   rE   Parser.extractY   s5    GG$$X.	&= //9M-mFvF]#**,0 - AH%( &^+//6Fx'/6F'// dNF1 ]#**, &%  '' ?? sA   C, D
0D D D ,D

DE)E EEc                 X    U R                  5        S3nU R                  SSSSX/5        U$ )z
Uses sox cmdline tool, to convert audio file to .wav

Note: for testing, use -
http://www.text2speech.org/,
with American Male 2 for best results
r?   soxz-Gz-c1)rV   r   )rO   rP   rV   s      r   rD   Parser.convert_to_wav}   s7      --/05%tS(BCr    N)r*   )__name__
__module____qualname____firstlineno____doc__rE   rD   __static_attributes__r^   r   r   r<   r<   M   s    	"H
r   r<   )importlib.utilr   r1   r   syspathlibr   typingr   speech_recognitionrG   textract.exceptionsr   utilsr   r+   r,   r   boolr   platformr
   ImportErrorr.   	AudioDatar   r:   r<   r^   r   r   <module>rp      s     	  
    - $ x~   <<8 ? A A#' ,
 1bll 1s 1:[ :?  s   "B BB