a
    d                  	   @   s   d dl Z d dlZd dlZd dlmZmZmZ d dlZe	e
ejdddZde
eeddd	Zde
eee
 eeeeef ef  ed
ddZde
ee
e
f edddZe
dddZdS )    N)OptionalTupleUnion)bpayloadsampling_ratereturnc                 C   s   | }d}d}dddd|d|d|d	d
ddg}zBt j|t jt jd}|| }W d   n1 sb0    Y  W n. ty } ztd|W Y d}~n
d}~0 0 |d }	t|	tj}
|
j	d dkrtd|
S )z?
    Helper function to read an audio file through ffmpeg.
    1f32leffmpeg-izpipe:0-ac-ar-f-hide_banner	-loglevelquietpipe:1)stdinstdoutNzFffmpeg was not found but is required to load audio files from filenamer   zMalformed soundfile)

subprocessPopenPIPEcommunicateFileNotFoundError
ValueErrornp
frombufferfloat32shape)r   r   aracformat_for_conversionffmpeg_commandffmpeg_processZoutput_streamerror	out_bytesaudio r'   k/var/www/html/stable-diffusion-webui/venv/lib/python3.9/site-packages/transformers/pipelines/audio_utils.pyffmpeg_read
   s6    , r)   r	   )r   chunk_length_sr!   c                 c   s   |  }d}|dkrd}n|dkr&d}nt d| dt }|dkrPd	}d
}n"|dkrbd}d}n|dkrrd}d
}dd|d|d|d|d|ddddddg}	tt| | | }
t|	|
}|D ]
}|V  qdS )z6
    Helper function ro read raw microphone data.
    r   s16le   r	      Unhandled format ` `. Please use `s16le` or `f32le`LinuxZalsadefaultDarwinZavfoundationz:0WindowsZdshowr
   r   r   r   r   z-fflagsZnobufferr   r   r   r   N)r   platformsystemintround_ffmpeg_stream)r   r*   r!   r   r    size_of_sampler5   Zformat_Zinput_r"   	chunk_leniteratoritemr'   r'   r(   ffmpeg_microphone-   sN    
r=   )r   r*   stream_chunk_sstride_length_sr!   c                 c   s`  |dur|}n|}t | ||d}|dkr4tj}d}n$|dkrHtj}d}ntd| d|du rh|d	 }tt| | | }	t|ttfr||g}tt| |d
  | }
tt| |d  | }t	j	
 }t	j|d}t||	|
|fddD ]n}tj|d |d|d< |d d
 | |d d | f|d< | |d< ||7 }t	j	
 |d|  krTq|V  qdS )a  
    Helper function to read audio from the microphone file through ffmpeg. This will output `partial` overlapping
    chunks starting from `stream_chunk_s` (if it is defined) until `chunk_length_s` is reached. It will make use of
    striding to avoid errors on the "sides" of the various chunks.

    Arguments:
        sampling_rate (`int`):
            The sampling_rate to use when reading the data from the microphone. Try using the model's sampling_rate to
            avoid resampling later.
        chunk_length_s (`float` or `int`):
            The length of the maximum chunk of audio to be sent returned. This includes the eventual striding.
        stream_chunk_s (`float` or `int`)
            The length of the minimal temporary audio to be returned.
        stride_length_s (`float` or `int` or `(float, float)`, *optional*, defaults to `None`)
            The length of the striding to be used. Stride is used to provide context to a model on the (left, right) of
            an audio sample but without using that part to actually make the prediction. Setting this does not change
            the length of the chunk.
        format_for_conversion (`str`, defalts to `f32le`)
            The name of the format of the audio samples to be returned by ffmpeg. The standard is `f32le`, `s16le`
            could also be used.
    Return:
        A generator yielding dictionaries of the following form

        `{"sampling_rate": int, "raw": np.array(), "partial" bool}` With optionnally a `"stride" (int, int)` key if
        `stride_length_s` is defined.

        `stride` and `raw` are all expressed in `samples`, and `partial` is a boolean saying if the current yield item
        is a whole chunk, or a partial temporary result to be later replaced by another larger chunk.


    N)r!   r+   r,   r	   r-   r.   r/      r      )secondsT)stridestreamraw)dtyperC   r   
   )r=   r   int16r   r   r6   r7   
isinstancefloatdatetimenow	timedeltachunk_bytes_iterr   )r   r*   r>   r?   r!   Zchunk_sZ
microphonerF   r9   r:   stride_leftstride_rightZ
audio_timedeltar<   r'   r'   r(   ffmpeg_microphone_liveb   s<    &
rR   F)r:   rC   rD   c           
      c   s   d}|\}}|| |kr2t d| d| d| d}| D ]}||7 }|rvt||k rv|df}|d| |ddV  q:t||kr:||f}|d| |d	}	|rd
|	d< |	V  |}||| | d }qvq:t||kr||dfd	}	|rd
|	d< |	V  dS )z
    Reads raw bytes from an iterator and does chunks of length `chunk_len`. Optionally adds `stride` to each chunks to
    get overlaps. `stream` is used to return partial results even if a full `chunk_len` is not yet available.
        z5Stride needs to be strictly smaller than chunk_len: (z, z) vs r   NT)rE   rC   partial)rE   rC   FrT   )r   len)
r;   r:   rC   rD   accrO   rP   Z_stride_leftrE   r<   r'   r'   r(   rN      s2    rN   )buflenc              
   c   s   d}zTt j| t j|d.}|j|}|dkr0q8|V  qW d   n1 sL0    Y  W n. ty } ztd|W Y d}~n
d}~0 0 dS )zJ
    Internal function to create the generator of data through ffmpeg
    i   )r   bufsizerS   NzHffmpeg was not found but is required to stream audio files from filename)r   r   r   r   readr   r   )r"   rW   rX   r#   rE   r$   r'   r'   r(   r8      s    *r8   )r	   )NNr	   )F)rK   r4   r   typingr   r   r   numpyr   bytesr6   arrayr)   rJ   strr=   rR   boolrN   r8   r'   r'   r'   r(   <module>   s.   & 8   N#