simple_ffmpeg_batch_io.AudioIO

Read/write audio frames or batches of audio frames from (compressed) file, including video file with audio stream(s), using FFmpeg backend.

This module defines the main AudioIO class used to open audio streams, read audio frames or batches of frames, and write processed outputs.

Authors

Dominique Vaufreydaz (inspired from original C++ code: https://github.com/Vaufreyd/ReadWriteVideosWithOpenCV)

  1"""
  2Read/write audio frames or batches of audio frames from (compressed) file, including video file with audio stream(s), using FFmpeg backend.
  3
  4This module defines the main `AudioIO` class used to open audio streams,
  5read audio frames or batches of frames, and write processed outputs.
  6
  7Authors
  8-------
  9Dominique Vaufreydaz (inspired from original C++ code: https://github.com/Vaufreyd/ReadWriteVideosWithOpenCV)
 10
 11"""
 12
 13__authors__ = ("Dominique Vaufreydaz")
 14
 15import sys
 16import subprocess as sp
 17import re
 18from enum import Enum
 19from typing import Union
 20
 21import math
 22
 23import numpy as np
 24
 25from .FrameCounter import FrameCounter
 26from .FrameContainer import FrameContainer
 27from .PipeMode import PipeMode
 28
 29# init static_ffmpeg at import time, first time it will download ffmpeg executables
 30import static_ffmpeg
 31static_ffmpeg.add_paths()
 32
 33class AudioIO:
 34    # "static" variables  to ffmpeg, ffprobe executables
 35    audioProgram, paramProgram = static_ffmpeg.run.get_or_fetch_platform_executables_else_raise()
 36
 37    class AudioIOException(Exception):
 38        """
 39        Dedicated exception class for AudioIO class.
 40        """
 41        def __init__(self, message="Error while reading/writing video occurs"):
 42            self.message = message
 43            super().__init__(self.message)
 44
 45    class AudioFormat(Enum):
 46        """
 47        Enum class for supported input video type: 32-bit float is the only supported type for the moment.
 48        """
 49        PCM32LE = 'pcm_f32le' # default format (unique mode for the moment)
 50
 51    @classmethod
 52    def reader(cls, filename, *, loglevel = 16, debug = False, **kwargs):
 53        """
 54        Create and open an AudioIO object in reader mode
 55
 56        See ``AudioIO.open`` for the full list of accepted parameters.
 57        """
 58        reader = cls(logLevel=loglevel, debug=debug)
 59        reader.open(filename, **kwargs)
 60        return reader
 61
 62    @classmethod
 63    def writer(cls, filename, sample_rate, channels, *, loglevel = 16, debug = False, **kwargs):
 64        """
 65        Create and open an AudioIO object in writer mode
 66
 67        See ``AudioIO.create`` for the full list of accepted parameters.
 68        """
 69        writer = cls(logLevel=loglevel, debug=debug)
 70        writer.create(filename, sample_rate, channels, **kwargs)
 71        return writer
 72
 73    # standard method
 74    def get_corresponding_writer(self, filename, **kwargs):
 75        """
 76        Method to get writer for an audio file with same sample_rate, channels as the current one AudioIO object
 77
 78        See `AudioIO.create` for the full list
 79        of accepted parameters.
 80        """
 81        return AudioIO.writer(filename, self.sample_rate, self.channels, **kwargs)
 82
 83    # To use with context manager "with AudioIO.reader(...) as f:' for instance
 84    def __enter__(self):
 85        """
 86        Method call at initialisation of a context manager like "with AudioIO.reader/writer(...) as f:' for instance
 87        """
 88        # simply return myself
 89        return self
 90
 91    def __exit__(self, exc_type, exc_val, exc_tb):
 92        """
 93        Method call when existing of a context manager like "with AudioIO.reader/writer(...) as f:' for instance
 94        """
 95        # close AudioIO
 96        self.close()
 97        return False
 98
 99    @staticmethod
100    def get_time_in_sec(filename, *, debug=False, logLevel=16):
101        """
102        Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals).
103
104        Parameters
105        ----------
106        filename : str or path. 
107            Raw audio waveform as a 1D array.
108
109        debug : bool (default False).
110            Show debug info.
111
112        logLevel: int (default 16).
113            Log level to pass to the underlying ffmpeg/ffprobe command.
114        
115        Returns
116        ----------
117        float
118            Length in seconds of video file (including milliseconds as decimal part with 3 decimals)
119        """
120        
121        cmd = [AudioIO.paramProgram, # ffprobe
122                    '-hide_banner',
123                    '-loglevel', str(logLevel),
124                    '-show_entries', 'format=duration',
125                    '-of', 'default=noprint_wrappers=1:nokey=1',
126                    str(filename)
127                    ]
128
129        if debug == True:
130            print(' '.join(cmd))
131
132        # call ffprobe and get params in one single line
133        lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg
134        output = lpipe.stdout.readlines()
135        lpipe.terminate()
136        # transform Bytes output to one single string
137        output = ''.join( [element.decode('utf-8') for element in output])
138
139        try:
140            return float(output)
141        except (ValueError, TypeError):
142            return None
143
144    @staticmethod
145    def get_params(filename, *, debug=False, logLevel=16):
146        """
147        Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds.
148
149        Parameters
150        ----------
151        filename : str or path.
152            Raw audio waveform as a 1D array.
153
154        debug : bool (default (False).
155            Show debug info.
156
157        log_level: int (default 16).
158            Log level to pass to the underlying ffmpeg/ffprobe command.
159
160        Returns
161        ----------
162        tuple
163            Tuple containing (channels,sample_rate) of the file
164        """
165        cmd = [AudioIO.paramProgram, # ffprobe
166                    '-hide_banner',
167                    '-loglevel', str(logLevel),
168                    '-show_entries', 'stream=channels,sample_rate',
169                    str(filename)
170                    ]
171
172        if debug == True:
173            print(' '.join(cmd))
174
175        # call ffprobe and get params in one single line
176        lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg
177        output = lpipe.stdout.readlines()
178        lpipe.terminate()
179        # transform Bytes output to one single string
180        output = ''.join( [element.decode('utf-8') for element in output])
181
182        pattern_sample_rate = r'sample_rate=(\d+)'
183        pattern_channels = r'channels=(\d+)'
184
185        # Search for values in the ffprobe output
186        match_sample_rate = re.search(pattern_sample_rate, output, flags=re.MULTILINE)
187        match_channels = re.search(pattern_channels, output, flags=re.MULTILINE)
188
189        # Extraction des valeurs
190        if match_sample_rate:
191            sample_rate = int(match_sample_rate.group(1))
192        else:
193            raise AudioIO.AudioIOException("Unable to get audio sample_rate of '" + str(filename) + "'")
194
195        if match_channels:
196            channels = int(match_channels.group(1))
197        else:
198            raise AudioIO.AudioIOException("Unable to get audio channels of '" + str(filename) + "'")
199
200        return (channels,sample_rate)
201
202        # Attributes
203        mode: PipeMode
204        """ Pipemode of the current object (default PipeMode.UNK_MODE)"""
205
206        loglevel: int
207        """ loglevel of the underlying ffmpeg backend for this object (default 16)"""
208
209        debug: bool
210        """ debug flag for this object (print debut info, default False)"""
211
212        channels: int
213        """ Number of channels of images (default -1) """
214
215        sample_rate: int
216        """ sample_rate of images (default -1) """
217
218        plannar: bool
219        """ Read/write data as plannar, i.e. not interleaved (default True) """
220
221        pipe: sp.Popen
222        """ pipe object to ffmpeg/ffprobe (default None)"""
223
224        frame_size: int
225        """ Weight in bytes of one image (default -1)"""
226
227        filename: str
228        """ Filename of the file (default None)"""
229
230        frame_counter: FrameCounter
231        """ `Framecounter` object to count ellapsed time (default None)"""
232
233    def __init__(self, *, logLevel = 16, debug = False):
234        """
235        Create a AudioIO object giving ffmpeg/ffrobe loglevel and defining debug mode
236
237        Parameters
238        ----------
239        log_level: int (default 16)
240            Log level to pass to the underlying ffmpeg/ffprobe command.
241
242        debug: bool (default (False)
243            Show debug info. while processing video
244        """
245
246        self.mode = PipeMode.UNK_MODE
247        self.logLevel = logLevel
248        self.debug = debug
249
250        # Call init() method
251        self.init()
252
253    def init(self):
254        """
255        Init or reinit a AudioIO object.
256        """
257        self.channels  = -1
258        self.sample_rate = -1
259        self.plannar = True
260        self.pipe = None
261        self.frame_size = -1
262        self.filename = None
263        self.frame_counter = None
264        self.float_size = None
265
266    _repr_exclude = {"pipe"}
267    """ List of excluded attribute for string conversion. """
268
269    # converting the object to a string representation
270    def __repr__(self):
271        """
272        Convert object (excluding attributes in _repr_exclude) to string representation.
273        """
274        attrs = ", ".join(
275            f"{k}={v!r}"
276            for k, v in self.__dict__.items()
277            if k not in self._repr_exclude
278        )
279        return f"{self.__class__.__name__}({attrs})"
280
281    __str__ = __repr__
282    """ String representation """
283
284    def get_elapsed_time_as_str(self) -> str:
285        """
286        Method to get elapsed time (float value represented) as str.
287
288        Returns
289        ----------
290        str or None
291            Elapsed time (float value) as str, "15.500" for instance for 15 secondes and 500 milliseconds
292            None if no frame counter are available.
293        """
294        if self.frame_counter is None:
295            return None
296        return self.frame_counter.get_elapsed_time_as_str()
297
298    def get_formated_elapsed_time_as_str(self,show_ms=True) -> str:
299        """
300        Method to get elapsed time (hour format) as str.
301
302        Returns
303        ----------
304        str or None
305            Elapsed time (float value) as str, "00:00:15.500" for instance for 15 secondes and 500 milliseconds
306            None if no frame counter are available.
307        """
308        if self.frame_counter is None:
309            return None
310        return self.frame_counter.get_formated_elapsed_time_as_str()
311
312    def get_elapsed_time(self) -> float:
313        """
314        Method to get elapsed time as float value rounded to 3 decimals.
315
316        Returns
317        ----------
318        float or None
319            Elapsed time (float value) as str, 15.500 for instance for 15 secondes and 500 milliseconds
320            None if no frame counter are available.
321        """
322        if self.frame_counter is None:
323            return None
324        return self.frame_counter.get_elapsed_time()
325
326    def is_opened(self) -> bool:
327        """
328        Method to get status of the underlying pipe to ffmpeg.
329
330        Returns
331        ----------
332        bool
333            True if pipe is opened (reading or writing mode), False if not.
334        """
335        # is the pip opened?
336        if self.pipe is not None and self.pipe.poll() is None:
337            return True
338
339        return False
340
341    def close(self):
342        """
343        Method to close current pipe to ffmpeg (if any). Ffmpeg/ffprobe  will be terminated. Object can be reused using open or create methods.
344        """
345        if self.pipe is not None:
346            if self.mode == PipeMode.WRITE_MODE:
347                # killing will make ffmpeg not finish properly the job, close the pipe
348                # to let it know that no more data are comming
349                self.pipe.stdin.close()
350            else: # self.mode == PipeMode.READ_MODE
351                # in read mode, no need to be nice, send SIGTERM on Linux,/Kill it on windows
352                self.pipe.kill()
353
354            # wait for subprocess to end
355            self.pipe.wait()
356
357        # reinit object for later use
358        self.init()
359
360    def create( self, filename, sample_rate, channels, *, writeOverExistingFile = False,
361                outputEncoding = AudioFormat.PCM32LE, encodingParams = None, plannar = True ):
362        """
363        Method to create a audio file using parametrized access through ffmpeg. Importante note: calling create
364        on a AudioIO will close any former open video.
365
366        Parameters
367        ----------
368        filename: str or path
369            filename of path to the file (mp4, avi, ...)
370
371        sample_rate: int
372            If defined as a positive value, sample_rates of the output file will be set to this value.
373
374        channels: int
375            If defined as a positive value, number of channels of output file will be set to this value.
376
377        fps:
378            If defined as a positive value, fps of input video will be set to this value.
379
380        outputEncoding: AudioFormat optional (default AudioFormat.PCM32LE)
381            Define audio format for samples. Possible value is AudioFormat.PCM32LE.
382
383        encodingParams: str optional (default None)
384            Parameter to pass to ffmpeg to encode video like audio filters.
385
386        plannar : bool optionnal (default True)
387            Input data to write are grouped by channel if True, interleaved instead.
388
389        Returns
390        ----------
391        bool
392            Was the creation successfull
393        """
394
395        # Close if already opened
396        self.close()
397
398        # Set geometry/fps of the video stream from params
399        self.sample_rate = int(sample_rate)
400        self.channels = int(channels)
401        self.plannar = plannar
402
403        # Compute size of float 32, usefull if we want later to use other floating point convention
404        self.float_size = int(np.dtype(np.float32).itemsize)
405
406        # Check params
407        if self.sample_rate <= 0 or self.channels <= 0:
408            raise self.AudioIOException("Bad parameters: sample_rate={}, channels={}".format(self.sample_rate,self.channels))
409
410        # To write audio, we do not need to know in advance frame size, we will write x values of n bytes
411        self.frame_size = None
412
413        # Video params are set, open the video
414        cmd = [self.audioProgram] # ffmpeg
415
416        if writeOverExistingFile == True:
417            cmd.extend(['-y'])
418
419        cmd.extend(['-hide_banner',
420            '-nostats',
421            '-loglevel', str(self.logLevel),
422            '-f', 'f32le', '-acodec', outputEncoding.value, # input expected coding
423            '-ar', f"{self.sample_rate}",
424            '-ac', f"{self.channels}",
425            '-i', '-'])
426
427        if encodingParams is not None:
428            cmd.extend(encodingParams.split())
429
430        # Audio filename converted to str (for Path values)
431        cmd.extend( ['-vn', str(filename) ] )
432
433        if self.debug == True:
434            print( ' '.join(cmd), file=sys.stderr )
435
436        # store filename and set mode
437        self.filename = str(filename)
438        self.mode = PipeMode.WRITE_MODE
439
440        # call ffmpeg in write mode
441        try:
442            self.pipe = sp.Popen(cmd, stdin=sp.PIPE)
443            self.frame_counter = FrameCounter(self.sample_rate)
444        except Exception as e:
445            # if pipe failed, reinit object and raise exception
446            self.init()
447            raise
448
449        return True
450
451    def open( self, filename, *, sample_rate = -1, channels = -1, inputEncoding = AudioFormat.PCM32LE,
452                    decodingParams = None, frame_size = 1.0, plannar = True, start_time = 0.0 ):
453        """
454        Method to read (video file containing) audio using parametrized access through ffmpeg. Importante note: calling open
455        on a AudioIO will close any former open file.
456
457        Parameters
458        ----------
459        filename: str or path
460            filename of path to the file (mp4, avi, ...)
461
462        sample_rate: int optional (default -1)
463            If defined as a positive value, sample rate of the input audio will be converted to this value.
464
465        channels: int optional (default -1)
466            If defined as a positive value, number of channels of the input audio will converted to this value.
467
468        inputEncoding: AudioFormat optional (default AudioFormat.PCM32LE)
469            Define audio format for samples. Possible value is AudioFormat.PCM32LE.
470
471        decodingParams: str optional (default None)
472            Parameter to pass to ffmpeg to decode video like audio filters.
473
474        plannar: bool optionnal (default True)
475            Group audio samples per channel if True. Else, samples are interleaved.
476
477        frame_size: int or float (default 1.0)
478            If frame_size is an int, it is the number of expected samples in each frame, for instance 8000 for 8000 samples.
479            if frame_size is a float, it is considered as a time in seconds for each audio frame, for instance 1.0 for 1 second, 0.010 for 10 ms.
480            Number of samples in this case is computed using frame_size and sample_rate as int(frame_size * sample_rate)
481
482        start_time: float optional (default 0.0)
483            Define the reading start time. If not set, reading at beginning of the file.
484
485        Returns
486        ----------
487        bool
488            Was the opening successfull
489        """
490
491        # Close if already opened
492        self.close()
493
494        # Force conversion of parameters
495        channels = int(channels)
496        sample_rate = float(sample_rate)
497
498        self.plannar = plannar
499
500        # Compute size of float 32, usefull if we want later to use other floating point convention
501        self.float_size = int(np.dtype(np.float32).itemsize)
502
503        # get parameters from file if needed:
504        if sample_rate <= 0 or channels <= 0:
505            self.channels, self.sample_rate = self.getAudioParams(filename)
506
507        # check if parameters ask to overide video parameters
508        if channels > 0:
509            self.channels = channels
510        if sample_rate > 0:
511            self.sample_rate = sample_rate
512
513        # check parameters
514
515        if isinstance(frame_size,float):
516            # time in seconds
517            self.frame_size = int(frame_size*self.sample_rate)
518        elif isinstance(frame_size,int):
519            # number of samples
520            self.frame_size = frame_size
521        else:
522            # to do
523            pass
524
525        # Video params are set, open the video
526        cmd = [self.audioProgram, # ffmpeg
527                    '-hide_banner',
528                    '-nostats',
529                    '-loglevel', str(self.logLevel)]
530
531        if decodingParams is not None:
532            cmd.extend([decodingParams.split()])
533
534        if start_time < 0.0:
535            pass
536        elif start_time > 0.0:
537            cmd.extend(["-ss", f"{start_time}"])            
538
539        cmd.extend( ['-i', str(filename),
540                     '-f', 'f32le', '-acodec', inputEncoding.value, # input expected coding
541                     '-ar', f"{self.sample_rate}",
542                     '-ac', f"{self.channels}",
543                     '-' # output to stdout
544                    ]
545                )
546
547        if self.debug == True:
548            print( ' '.join(cmd) )
549
550        # store filename and set mode to READ_MODE
551        self.filename = str(filename)
552        self.mode = PipeMode.READ_MODE
553
554        # call ffmpeg in read mode
555        try:
556            self.pipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg/ffprobe
557            self.frame_counter = FrameCounter(self.sample_rate)
558            if start_time > 0.0:
559                self.frame_counter += start_time # adding with float means adding time
560        except Exception as e:
561            # if pipe failed, reinit object and raise exception
562            self.init()
563            raise
564
565        return True
566
567    def read_frame(self, with_timestamps = False):
568        """
569        Read next frame from the audio file
570
571        Parameters
572        ----------
573        with_timestamps: bool optional (default False)
574            If set to True, the method returns a ``FrameContainer`` with the audio and an array containing the associated timestamp(s)
575
576        Returns
577        ----------
578        nparray or FrameContainer
579            A frame of shape (self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A frame
580            of shape (self.channels*self.frame_size) with interleaved data if self.plannar is False.
581            if with_timestamps is True, the return object is a FrameContainer with the audio data in ``FrameContainer.data`` and
582            the associated timestamp in ``FrameContainer.timestamps`` as an array (one element).
583        """
584
585        if self.pipe is None:
586            raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading a frame.".format(self.audioProgram))
587        # - pipe is in write mode
588        if self.mode != PipeMode.READ_MODE:
589            raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename))
590
591        if with_timestamps:
592            # get elapsed time in video, it is time of next frame(s)
593            current_elapsed_time = self.get_elapsed_time()
594
595        # read rgb image from pipe
596        toread = self.frame_size*self.float_size
597        buffer = self.pipe.stdout.read(toread)
598
599        if buffer == b"":
600            # not considered as an error, no more frame, no exception
601            return None
602
603        # get numpy UINT8 array from buffer
604        audio = np.frombuffer(buffer, dtype = np.float32).reshape(len(buffer)//self.float_size, self.channels)
605
606        # make it plannar (or not)
607        if self.plannar:
608            #transpose it
609            audio = audio.T
610
611        # increase frame_counter
612        self.frame_counter.frame_count += (self.frame_size * self.channels)
613
614        # say to gc that this buffer is no longer needed
615        del buffer
616
617        if with_timestamps:
618            return FrameContainer(1, audio, self.frame_size/self.sample_rate, current_elapsed_time)
619        
620        return audio
621
622    def read_batch(self, numberOfFrames, with_timestamps = False):
623        """
624        Read next batch of audio from the file
625
626        Parameters
627        ----------
628        number_of_frames: int
629            Number of desired images within the batch. The last batch from the file may have less images.
630            
631        with_timestamps: bool optional (default False)
632            If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames
633
634        Returns
635        ----------
636        nparray or FrameContainer
637            A batch of shape (n, self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A batch
638            of shape (n, self.channels*self.frame_size) with interleaved data if self.plannar is False.
639            if with_timestamps is True, the return object is a FrameContainer with the audio batch in ``FrameContainer.data`` and
640            the associated timestamp in ``FrameContainer.timestamps`` as an array (one element for each audio frame).
641        """
642
643        if self.pipe is None:
644            raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading frames.".format(self.audioProgram))
645        # - pipe is in write mode
646        if self.mode != PipeMode.READ_MODE:
647            raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename))
648
649        if with_timestamps:
650            # get elapsed time in video, it is time of next frame(s)
651            current_elapsed_time = self.get_elapsed_time()
652
653        # try to read complete batch
654        toread = self.frame_size*self.float_size*self.channels*numberOfFrames
655        buffer = self.pipe.stdout.read(toread)
656
657        # check if we are at the end of the buffer
658        if buffer == b"":
659            # not considered as an error, no more frame, no exception
660            return None
661
662        # compute actual number of Frames
663        # do we have a full batch ? Computer standard division
664        actualNbFrames = len(buffer)/(self.frame_size*self.float_size*self.channels)
665
666        if actualNbFrames.is_integer():
667            actualNbFrames = int(actualNbFrames)
668            # get and reshape batch from buffer
669            batch = np.frombuffer(buffer, dtype = np.float32).reshape((actualNbFrames, self.frame_size, self.channels,))
670        else:
671            # We do not have a full batch, we are at end of the stream or ffmpeg has been stopped/killed
672            # Compute frame size in samples knowing len of buffer, self.float_size, self.channels and numberOfFrames
673            l_frame_size = len(buffer)//(numberOfFrames*self.float_size*self.channels)
674            if l_frame_size <= 0:
675                # not considered as an error, no more frame, no exception
676                return None
677            # get and reshape batch from buffer
678            batch = np.frombuffer(buffer, dtype = np.float32).reshape((numberOfFrames, l_frame_size, self.channels,))
679
680        if self.plannar:
681            batch = batch.transpose(0, 2, 1)
682
683        # increase frame_counter
684        self.frame_counter.frame_count += (actualNbFrames * self.frame_size * self.channels)
685        
686        # say to gc that this buffer is no longer needed
687        del buffer
688
689        if with_timestamps:
690            return FrameContainer( actualNbFrames, batch, self.frame_size/self.sample_rate, current_elapsed_time)
691        
692        return batch
693
694    def write_frame(self, audio) -> bool:
695        """
696        Write an audio frame to the file
697
698        Parameters
699        ----------
700        audio: nparray
701            The audio frame to write to the video file of shape (self.channels,nb_samples_per_channel) if plannar is True else (self.channels*nb_samples_per_channel).
702
703        Returns
704        ----------
705        bool
706            Writing was successful or not.
707        """
708        # Check params
709        # - pipe exists
710        if self.pipe is None:
711            raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram))
712        # - pipe is in write mode
713        if self.mode != PipeMode.WRITE_MODE:
714            raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename))
715        # - shape of image is fine, thus we have pixels for a full compatible frame
716        if audio.shape[0] != self.channels:
717            raise self.AudioIOException("Wong audio shape: {} expected ({}, nb_samples_to_write).".format(audio.shape,self.channels))
718        # - type of data is Float32
719        if audio.dtype != np.float32:
720            raise self.AudioIOException("Wong audio type: {} expected np.float32.".format(audio.dtype))
721
722        # array must have a shape (channels, samples), reshape it it to (samples, channels) if plannar
723        if not self.plannar:
724            audio = audio.reshape(-1)
725
726        # print( audio.shape )
727
728        # garantee to have a C continuous array
729        if not audio.flags['C_CONTIGUOUS']:
730            a = np.ascontiguousarray(a) 
731
732        # write frame
733        buffer = audio.tobytes()
734        if self.pipe.stdin.write( buffer ) < len(buffer):
735            print( f"Error writing frame to {self.filename}" )
736            return False
737
738        # increase frame_counter
739        self.frame_counter.frame_count += (audio.shape[1] * self.channels)
740
741        # say to gc that this buffer is no longer needed 
742        del buffer
743
744        return True
745
746    def write_batch(self, batch):
747        """
748        Write a batch of audio frame to the file
749
750        Parameters
751        ----------
752        batch: nparray
753            The batch of audio frames to write to the video file of shape (n,self.channels,nb_samples_per_channel) if plannar is True else (n,self.channels*nb_samples_per_channel) of interleaved audio data.
754
755        Returns
756        ----------
757        bool
758            Writing was successful or not.
759        """
760        # Check params
761        # - pipe exists
762        if self.pipe is None:
763            raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram))
764        # - pipe is in write mode
765        if self.mode != PipeMode.WRITE_MODE:
766            raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename))
767        # batch is 3D (n, channels, nb samples)
768        if batch.ndim !=3:
769            raise self.AudioIOException("Wrong batch shape: {} expected 3 dimensions (n, n_channels, n_samples_per_channel).".format(batch.shape))
770        # - shape of images in batch is fine
771        if batch.shape[2] != self.channels:
772            raise self.AudioIOException("Wrong audio channels in batch: {} expected {} {}.".format(batch.shape[2], self.channels, batch.shape))
773
774        # array must have a shape (n * n_channels * n_samples_per_channel) before writing them to pipe
775        # reshape it it to (n * n_channels * n_samples_per_channel) if plannar is False
776        if not self.plannar:
777            # goes from (n, n_channels, n_samples_per_channel) to (n * n_channels * n_samples_per_channel)
778            batch = batch.transpose(0, 2, 1) # first go to (n, n_samples_per_channel, n_channels)
779            batch = batch.reshape(-1) # then to 1D array (n * n_channels * n_samples_per_channel)
780
781        # garantee to have a C continuous array
782        if not batch.flags['C_CONTIGUOUS']:
783            batch = np.ascontiguousarray(batch)
784
785        # write frame
786        buffer = batch.tobytes()
787        if self.pipe.stdin.write( buffer ) < len(buffer):
788            # say to gc that this buffer is no longer needed
789            del buffer
790            raise self.AudioIOException("Error writing batch to '{}'.".format(self.filename))
791
792        # increase frame_counter
793        self.frame_counter.frame_count += int(batch.shape[0]/self.channels) # int conversion is mandatory to avoid confusion with time as float
794              
795        # say to gc that this buffer is no longer needed
796        del buffer
797
798        return True
799
800    def iter_frames(self, with_timestamps = False):
801        """
802        Method to iterate on audio frames using AudioIO obj.
803        for audio_frame in obj.iter_frames():
804            ....
805
806        Parameters
807        ----------
808        with_timestamps: bool optional (default False)
809            If set to True, the method returns a FrameContainer object with the batch and an array containing the associated timestamps to frames
810
811        Returns
812        ----------
813        nparray or FrameContainer
814            A batch of images of shape ()
815        """
816
817        try:
818            if self.mode == PipeMode.READ_MODE:
819                while self.isOpened():
820                    frame = self.readFrame(with_timestamps)
821                    if frame is not None:
822                        yield frame
823        finally:
824            self.close()
825
826    def iter_batches(self, batch_size : int, with_timestamps = False ):
827        """
828        Method to iterate on batch ofaudio  frames using AudioIO obj.
829        for audio_batch in obj.iter_batches():
830            ....
831
832        Parameters
833        ----------
834        with_timestamps: bool optional (default False)
835            If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames
836        """
837        try:
838            if self.mode == PipeMode.READ_MODE:
839                while self.isOpened():
840                    batch = self.readBatch(batch_size, with_timestamps)
841                    if batch is not None:
842                        yield batch
843        finally:
844            self.close()
845
846    # function aliases to be compliant with original C++ version
847    getAudioTimeInSec = get_time_in_sec
848    getAudioParams = get_params
849    get_audio_time_in_sec = get_time_in_sec
850    get_audio_params = get_params
851    isOpened = is_opened
852    readFrame = read_frame
853    readBatch = read_batch
854    writeFrame = write_frame
855    writeBatch = write_batch
class AudioIO:
 34class AudioIO:
 35    # "static" variables  to ffmpeg, ffprobe executables
 36    audioProgram, paramProgram = static_ffmpeg.run.get_or_fetch_platform_executables_else_raise()
 37
 38    class AudioIOException(Exception):
 39        """
 40        Dedicated exception class for AudioIO class.
 41        """
 42        def __init__(self, message="Error while reading/writing video occurs"):
 43            self.message = message
 44            super().__init__(self.message)
 45
 46    class AudioFormat(Enum):
 47        """
 48        Enum class for supported input video type: 32-bit float is the only supported type for the moment.
 49        """
 50        PCM32LE = 'pcm_f32le' # default format (unique mode for the moment)
 51
 52    @classmethod
 53    def reader(cls, filename, *, loglevel = 16, debug = False, **kwargs):
 54        """
 55        Create and open an AudioIO object in reader mode
 56
 57        See ``AudioIO.open`` for the full list of accepted parameters.
 58        """
 59        reader = cls(logLevel=loglevel, debug=debug)
 60        reader.open(filename, **kwargs)
 61        return reader
 62
 63    @classmethod
 64    def writer(cls, filename, sample_rate, channels, *, loglevel = 16, debug = False, **kwargs):
 65        """
 66        Create and open an AudioIO object in writer mode
 67
 68        See ``AudioIO.create`` for the full list of accepted parameters.
 69        """
 70        writer = cls(logLevel=loglevel, debug=debug)
 71        writer.create(filename, sample_rate, channels, **kwargs)
 72        return writer
 73
 74    # standard method
 75    def get_corresponding_writer(self, filename, **kwargs):
 76        """
 77        Method to get writer for an audio file with same sample_rate, channels as the current one AudioIO object
 78
 79        See `AudioIO.create` for the full list
 80        of accepted parameters.
 81        """
 82        return AudioIO.writer(filename, self.sample_rate, self.channels, **kwargs)
 83
 84    # To use with context manager "with AudioIO.reader(...) as f:' for instance
 85    def __enter__(self):
 86        """
 87        Method call at initialisation of a context manager like "with AudioIO.reader/writer(...) as f:' for instance
 88        """
 89        # simply return myself
 90        return self
 91
 92    def __exit__(self, exc_type, exc_val, exc_tb):
 93        """
 94        Method call when existing of a context manager like "with AudioIO.reader/writer(...) as f:' for instance
 95        """
 96        # close AudioIO
 97        self.close()
 98        return False
 99
100    @staticmethod
101    def get_time_in_sec(filename, *, debug=False, logLevel=16):
102        """
103        Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals).
104
105        Parameters
106        ----------
107        filename : str or path. 
108            Raw audio waveform as a 1D array.
109
110        debug : bool (default False).
111            Show debug info.
112
113        logLevel: int (default 16).
114            Log level to pass to the underlying ffmpeg/ffprobe command.
115        
116        Returns
117        ----------
118        float
119            Length in seconds of video file (including milliseconds as decimal part with 3 decimals)
120        """
121        
122        cmd = [AudioIO.paramProgram, # ffprobe
123                    '-hide_banner',
124                    '-loglevel', str(logLevel),
125                    '-show_entries', 'format=duration',
126                    '-of', 'default=noprint_wrappers=1:nokey=1',
127                    str(filename)
128                    ]
129
130        if debug == True:
131            print(' '.join(cmd))
132
133        # call ffprobe and get params in one single line
134        lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg
135        output = lpipe.stdout.readlines()
136        lpipe.terminate()
137        # transform Bytes output to one single string
138        output = ''.join( [element.decode('utf-8') for element in output])
139
140        try:
141            return float(output)
142        except (ValueError, TypeError):
143            return None
144
145    @staticmethod
146    def get_params(filename, *, debug=False, logLevel=16):
147        """
148        Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds.
149
150        Parameters
151        ----------
152        filename : str or path.
153            Raw audio waveform as a 1D array.
154
155        debug : bool (default (False).
156            Show debug info.
157
158        log_level: int (default 16).
159            Log level to pass to the underlying ffmpeg/ffprobe command.
160
161        Returns
162        ----------
163        tuple
164            Tuple containing (channels,sample_rate) of the file
165        """
166        cmd = [AudioIO.paramProgram, # ffprobe
167                    '-hide_banner',
168                    '-loglevel', str(logLevel),
169                    '-show_entries', 'stream=channels,sample_rate',
170                    str(filename)
171                    ]
172
173        if debug == True:
174            print(' '.join(cmd))
175
176        # call ffprobe and get params in one single line
177        lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg
178        output = lpipe.stdout.readlines()
179        lpipe.terminate()
180        # transform Bytes output to one single string
181        output = ''.join( [element.decode('utf-8') for element in output])
182
183        pattern_sample_rate = r'sample_rate=(\d+)'
184        pattern_channels = r'channels=(\d+)'
185
186        # Search for values in the ffprobe output
187        match_sample_rate = re.search(pattern_sample_rate, output, flags=re.MULTILINE)
188        match_channels = re.search(pattern_channels, output, flags=re.MULTILINE)
189
190        # Extraction des valeurs
191        if match_sample_rate:
192            sample_rate = int(match_sample_rate.group(1))
193        else:
194            raise AudioIO.AudioIOException("Unable to get audio sample_rate of '" + str(filename) + "'")
195
196        if match_channels:
197            channels = int(match_channels.group(1))
198        else:
199            raise AudioIO.AudioIOException("Unable to get audio channels of '" + str(filename) + "'")
200
201        return (channels,sample_rate)
202
203        # Attributes
204        mode: PipeMode
205        """ Pipemode of the current object (default PipeMode.UNK_MODE)"""
206
207        loglevel: int
208        """ loglevel of the underlying ffmpeg backend for this object (default 16)"""
209
210        debug: bool
211        """ debug flag for this object (print debut info, default False)"""
212
213        channels: int
214        """ Number of channels of images (default -1) """
215
216        sample_rate: int
217        """ sample_rate of images (default -1) """
218
219        plannar: bool
220        """ Read/write data as plannar, i.e. not interleaved (default True) """
221
222        pipe: sp.Popen
223        """ pipe object to ffmpeg/ffprobe (default None)"""
224
225        frame_size: int
226        """ Weight in bytes of one image (default -1)"""
227
228        filename: str
229        """ Filename of the file (default None)"""
230
231        frame_counter: FrameCounter
232        """ `Framecounter` object to count ellapsed time (default None)"""
233
234    def __init__(self, *, logLevel = 16, debug = False):
235        """
236        Create a AudioIO object giving ffmpeg/ffrobe loglevel and defining debug mode
237
238        Parameters
239        ----------
240        log_level: int (default 16)
241            Log level to pass to the underlying ffmpeg/ffprobe command.
242
243        debug: bool (default (False)
244            Show debug info. while processing video
245        """
246
247        self.mode = PipeMode.UNK_MODE
248        self.logLevel = logLevel
249        self.debug = debug
250
251        # Call init() method
252        self.init()
253
254    def init(self):
255        """
256        Init or reinit a AudioIO object.
257        """
258        self.channels  = -1
259        self.sample_rate = -1
260        self.plannar = True
261        self.pipe = None
262        self.frame_size = -1
263        self.filename = None
264        self.frame_counter = None
265        self.float_size = None
266
267    _repr_exclude = {"pipe"}
268    """ List of excluded attribute for string conversion. """
269
270    # converting the object to a string representation
271    def __repr__(self):
272        """
273        Convert object (excluding attributes in _repr_exclude) to string representation.
274        """
275        attrs = ", ".join(
276            f"{k}={v!r}"
277            for k, v in self.__dict__.items()
278            if k not in self._repr_exclude
279        )
280        return f"{self.__class__.__name__}({attrs})"
281
282    __str__ = __repr__
283    """ String representation """
284
285    def get_elapsed_time_as_str(self) -> str:
286        """
287        Method to get elapsed time (float value represented) as str.
288
289        Returns
290        ----------
291        str or None
292            Elapsed time (float value) as str, "15.500" for instance for 15 secondes and 500 milliseconds
293            None if no frame counter are available.
294        """
295        if self.frame_counter is None:
296            return None
297        return self.frame_counter.get_elapsed_time_as_str()
298
299    def get_formated_elapsed_time_as_str(self,show_ms=True) -> str:
300        """
301        Method to get elapsed time (hour format) as str.
302
303        Returns
304        ----------
305        str or None
306            Elapsed time (float value) as str, "00:00:15.500" for instance for 15 secondes and 500 milliseconds
307            None if no frame counter are available.
308        """
309        if self.frame_counter is None:
310            return None
311        return self.frame_counter.get_formated_elapsed_time_as_str()
312
313    def get_elapsed_time(self) -> float:
314        """
315        Method to get elapsed time as float value rounded to 3 decimals.
316
317        Returns
318        ----------
319        float or None
320            Elapsed time (float value) as str, 15.500 for instance for 15 secondes and 500 milliseconds
321            None if no frame counter are available.
322        """
323        if self.frame_counter is None:
324            return None
325        return self.frame_counter.get_elapsed_time()
326
327    def is_opened(self) -> bool:
328        """
329        Method to get status of the underlying pipe to ffmpeg.
330
331        Returns
332        ----------
333        bool
334            True if pipe is opened (reading or writing mode), False if not.
335        """
336        # is the pip opened?
337        if self.pipe is not None and self.pipe.poll() is None:
338            return True
339
340        return False
341
342    def close(self):
343        """
344        Method to close current pipe to ffmpeg (if any). Ffmpeg/ffprobe  will be terminated. Object can be reused using open or create methods.
345        """
346        if self.pipe is not None:
347            if self.mode == PipeMode.WRITE_MODE:
348                # killing will make ffmpeg not finish properly the job, close the pipe
349                # to let it know that no more data are comming
350                self.pipe.stdin.close()
351            else: # self.mode == PipeMode.READ_MODE
352                # in read mode, no need to be nice, send SIGTERM on Linux,/Kill it on windows
353                self.pipe.kill()
354
355            # wait for subprocess to end
356            self.pipe.wait()
357
358        # reinit object for later use
359        self.init()
360
361    def create( self, filename, sample_rate, channels, *, writeOverExistingFile = False,
362                outputEncoding = AudioFormat.PCM32LE, encodingParams = None, plannar = True ):
363        """
364        Method to create a audio file using parametrized access through ffmpeg. Importante note: calling create
365        on a AudioIO will close any former open video.
366
367        Parameters
368        ----------
369        filename: str or path
370            filename of path to the file (mp4, avi, ...)
371
372        sample_rate: int
373            If defined as a positive value, sample_rates of the output file will be set to this value.
374
375        channels: int
376            If defined as a positive value, number of channels of output file will be set to this value.
377
378        fps:
379            If defined as a positive value, fps of input video will be set to this value.
380
381        outputEncoding: AudioFormat optional (default AudioFormat.PCM32LE)
382            Define audio format for samples. Possible value is AudioFormat.PCM32LE.
383
384        encodingParams: str optional (default None)
385            Parameter to pass to ffmpeg to encode video like audio filters.
386
387        plannar : bool optionnal (default True)
388            Input data to write are grouped by channel if True, interleaved instead.
389
390        Returns
391        ----------
392        bool
393            Was the creation successfull
394        """
395
396        # Close if already opened
397        self.close()
398
399        # Set geometry/fps of the video stream from params
400        self.sample_rate = int(sample_rate)
401        self.channels = int(channels)
402        self.plannar = plannar
403
404        # Compute size of float 32, usefull if we want later to use other floating point convention
405        self.float_size = int(np.dtype(np.float32).itemsize)
406
407        # Check params
408        if self.sample_rate <= 0 or self.channels <= 0:
409            raise self.AudioIOException("Bad parameters: sample_rate={}, channels={}".format(self.sample_rate,self.channels))
410
411        # To write audio, we do not need to know in advance frame size, we will write x values of n bytes
412        self.frame_size = None
413
414        # Video params are set, open the video
415        cmd = [self.audioProgram] # ffmpeg
416
417        if writeOverExistingFile == True:
418            cmd.extend(['-y'])
419
420        cmd.extend(['-hide_banner',
421            '-nostats',
422            '-loglevel', str(self.logLevel),
423            '-f', 'f32le', '-acodec', outputEncoding.value, # input expected coding
424            '-ar', f"{self.sample_rate}",
425            '-ac', f"{self.channels}",
426            '-i', '-'])
427
428        if encodingParams is not None:
429            cmd.extend(encodingParams.split())
430
431        # Audio filename converted to str (for Path values)
432        cmd.extend( ['-vn', str(filename) ] )
433
434        if self.debug == True:
435            print( ' '.join(cmd), file=sys.stderr )
436
437        # store filename and set mode
438        self.filename = str(filename)
439        self.mode = PipeMode.WRITE_MODE
440
441        # call ffmpeg in write mode
442        try:
443            self.pipe = sp.Popen(cmd, stdin=sp.PIPE)
444            self.frame_counter = FrameCounter(self.sample_rate)
445        except Exception as e:
446            # if pipe failed, reinit object and raise exception
447            self.init()
448            raise
449
450        return True
451
452    def open( self, filename, *, sample_rate = -1, channels = -1, inputEncoding = AudioFormat.PCM32LE,
453                    decodingParams = None, frame_size = 1.0, plannar = True, start_time = 0.0 ):
454        """
455        Method to read (video file containing) audio using parametrized access through ffmpeg. Importante note: calling open
456        on a AudioIO will close any former open file.
457
458        Parameters
459        ----------
460        filename: str or path
461            filename of path to the file (mp4, avi, ...)
462
463        sample_rate: int optional (default -1)
464            If defined as a positive value, sample rate of the input audio will be converted to this value.
465
466        channels: int optional (default -1)
467            If defined as a positive value, number of channels of the input audio will converted to this value.
468
469        inputEncoding: AudioFormat optional (default AudioFormat.PCM32LE)
470            Define audio format for samples. Possible value is AudioFormat.PCM32LE.
471
472        decodingParams: str optional (default None)
473            Parameter to pass to ffmpeg to decode video like audio filters.
474
475        plannar: bool optionnal (default True)
476            Group audio samples per channel if True. Else, samples are interleaved.
477
478        frame_size: int or float (default 1.0)
479            If frame_size is an int, it is the number of expected samples in each frame, for instance 8000 for 8000 samples.
480            if frame_size is a float, it is considered as a time in seconds for each audio frame, for instance 1.0 for 1 second, 0.010 for 10 ms.
481            Number of samples in this case is computed using frame_size and sample_rate as int(frame_size * sample_rate)
482
483        start_time: float optional (default 0.0)
484            Define the reading start time. If not set, reading at beginning of the file.
485
486        Returns
487        ----------
488        bool
489            Was the opening successfull
490        """
491
492        # Close if already opened
493        self.close()
494
495        # Force conversion of parameters
496        channels = int(channels)
497        sample_rate = float(sample_rate)
498
499        self.plannar = plannar
500
501        # Compute size of float 32, usefull if we want later to use other floating point convention
502        self.float_size = int(np.dtype(np.float32).itemsize)
503
504        # get parameters from file if needed:
505        if sample_rate <= 0 or channels <= 0:
506            self.channels, self.sample_rate = self.getAudioParams(filename)
507
508        # check if parameters ask to overide video parameters
509        if channels > 0:
510            self.channels = channels
511        if sample_rate > 0:
512            self.sample_rate = sample_rate
513
514        # check parameters
515
516        if isinstance(frame_size,float):
517            # time in seconds
518            self.frame_size = int(frame_size*self.sample_rate)
519        elif isinstance(frame_size,int):
520            # number of samples
521            self.frame_size = frame_size
522        else:
523            # to do
524            pass
525
526        # Video params are set, open the video
527        cmd = [self.audioProgram, # ffmpeg
528                    '-hide_banner',
529                    '-nostats',
530                    '-loglevel', str(self.logLevel)]
531
532        if decodingParams is not None:
533            cmd.extend([decodingParams.split()])
534
535        if start_time < 0.0:
536            pass
537        elif start_time > 0.0:
538            cmd.extend(["-ss", f"{start_time}"])            
539
540        cmd.extend( ['-i', str(filename),
541                     '-f', 'f32le', '-acodec', inputEncoding.value, # input expected coding
542                     '-ar', f"{self.sample_rate}",
543                     '-ac', f"{self.channels}",
544                     '-' # output to stdout
545                    ]
546                )
547
548        if self.debug == True:
549            print( ' '.join(cmd) )
550
551        # store filename and set mode to READ_MODE
552        self.filename = str(filename)
553        self.mode = PipeMode.READ_MODE
554
555        # call ffmpeg in read mode
556        try:
557            self.pipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg/ffprobe
558            self.frame_counter = FrameCounter(self.sample_rate)
559            if start_time > 0.0:
560                self.frame_counter += start_time # adding with float means adding time
561        except Exception as e:
562            # if pipe failed, reinit object and raise exception
563            self.init()
564            raise
565
566        return True
567
568    def read_frame(self, with_timestamps = False):
569        """
570        Read next frame from the audio file
571
572        Parameters
573        ----------
574        with_timestamps: bool optional (default False)
575            If set to True, the method returns a ``FrameContainer`` with the audio and an array containing the associated timestamp(s)
576
577        Returns
578        ----------
579        nparray or FrameContainer
580            A frame of shape (self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A frame
581            of shape (self.channels*self.frame_size) with interleaved data if self.plannar is False.
582            if with_timestamps is True, the return object is a FrameContainer with the audio data in ``FrameContainer.data`` and
583            the associated timestamp in ``FrameContainer.timestamps`` as an array (one element).
584        """
585
586        if self.pipe is None:
587            raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading a frame.".format(self.audioProgram))
588        # - pipe is in write mode
589        if self.mode != PipeMode.READ_MODE:
590            raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename))
591
592        if with_timestamps:
593            # get elapsed time in video, it is time of next frame(s)
594            current_elapsed_time = self.get_elapsed_time()
595
596        # read rgb image from pipe
597        toread = self.frame_size*self.float_size
598        buffer = self.pipe.stdout.read(toread)
599
600        if buffer == b"":
601            # not considered as an error, no more frame, no exception
602            return None
603
604        # get numpy UINT8 array from buffer
605        audio = np.frombuffer(buffer, dtype = np.float32).reshape(len(buffer)//self.float_size, self.channels)
606
607        # make it plannar (or not)
608        if self.plannar:
609            #transpose it
610            audio = audio.T
611
612        # increase frame_counter
613        self.frame_counter.frame_count += (self.frame_size * self.channels)
614
615        # say to gc that this buffer is no longer needed
616        del buffer
617
618        if with_timestamps:
619            return FrameContainer(1, audio, self.frame_size/self.sample_rate, current_elapsed_time)
620        
621        return audio
622
623    def read_batch(self, numberOfFrames, with_timestamps = False):
624        """
625        Read next batch of audio from the file
626
627        Parameters
628        ----------
629        number_of_frames: int
630            Number of desired images within the batch. The last batch from the file may have less images.
631            
632        with_timestamps: bool optional (default False)
633            If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames
634
635        Returns
636        ----------
637        nparray or FrameContainer
638            A batch of shape (n, self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A batch
639            of shape (n, self.channels*self.frame_size) with interleaved data if self.plannar is False.
640            if with_timestamps is True, the return object is a FrameContainer with the audio batch in ``FrameContainer.data`` and
641            the associated timestamp in ``FrameContainer.timestamps`` as an array (one element for each audio frame).
642        """
643
644        if self.pipe is None:
645            raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading frames.".format(self.audioProgram))
646        # - pipe is in write mode
647        if self.mode != PipeMode.READ_MODE:
648            raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename))
649
650        if with_timestamps:
651            # get elapsed time in video, it is time of next frame(s)
652            current_elapsed_time = self.get_elapsed_time()
653
654        # try to read complete batch
655        toread = self.frame_size*self.float_size*self.channels*numberOfFrames
656        buffer = self.pipe.stdout.read(toread)
657
658        # check if we are at the end of the buffer
659        if buffer == b"":
660            # not considered as an error, no more frame, no exception
661            return None
662
663        # compute actual number of Frames
664        # do we have a full batch ? Computer standard division
665        actualNbFrames = len(buffer)/(self.frame_size*self.float_size*self.channels)
666
667        if actualNbFrames.is_integer():
668            actualNbFrames = int(actualNbFrames)
669            # get and reshape batch from buffer
670            batch = np.frombuffer(buffer, dtype = np.float32).reshape((actualNbFrames, self.frame_size, self.channels,))
671        else:
672            # We do not have a full batch, we are at end of the stream or ffmpeg has been stopped/killed
673            # Compute frame size in samples knowing len of buffer, self.float_size, self.channels and numberOfFrames
674            l_frame_size = len(buffer)//(numberOfFrames*self.float_size*self.channels)
675            if l_frame_size <= 0:
676                # not considered as an error, no more frame, no exception
677                return None
678            # get and reshape batch from buffer
679            batch = np.frombuffer(buffer, dtype = np.float32).reshape((numberOfFrames, l_frame_size, self.channels,))
680
681        if self.plannar:
682            batch = batch.transpose(0, 2, 1)
683
684        # increase frame_counter
685        self.frame_counter.frame_count += (actualNbFrames * self.frame_size * self.channels)
686        
687        # say to gc that this buffer is no longer needed
688        del buffer
689
690        if with_timestamps:
691            return FrameContainer( actualNbFrames, batch, self.frame_size/self.sample_rate, current_elapsed_time)
692        
693        return batch
694
695    def write_frame(self, audio) -> bool:
696        """
697        Write an audio frame to the file
698
699        Parameters
700        ----------
701        audio: nparray
702            The audio frame to write to the video file of shape (self.channels,nb_samples_per_channel) if plannar is True else (self.channels*nb_samples_per_channel).
703
704        Returns
705        ----------
706        bool
707            Writing was successful or not.
708        """
709        # Check params
710        # - pipe exists
711        if self.pipe is None:
712            raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram))
713        # - pipe is in write mode
714        if self.mode != PipeMode.WRITE_MODE:
715            raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename))
716        # - shape of image is fine, thus we have pixels for a full compatible frame
717        if audio.shape[0] != self.channels:
718            raise self.AudioIOException("Wong audio shape: {} expected ({}, nb_samples_to_write).".format(audio.shape,self.channels))
719        # - type of data is Float32
720        if audio.dtype != np.float32:
721            raise self.AudioIOException("Wong audio type: {} expected np.float32.".format(audio.dtype))
722
723        # array must have a shape (channels, samples), reshape it it to (samples, channels) if plannar
724        if not self.plannar:
725            audio = audio.reshape(-1)
726
727        # print( audio.shape )
728
729        # garantee to have a C continuous array
730        if not audio.flags['C_CONTIGUOUS']:
731            a = np.ascontiguousarray(a) 
732
733        # write frame
734        buffer = audio.tobytes()
735        if self.pipe.stdin.write( buffer ) < len(buffer):
736            print( f"Error writing frame to {self.filename}" )
737            return False
738
739        # increase frame_counter
740        self.frame_counter.frame_count += (audio.shape[1] * self.channels)
741
742        # say to gc that this buffer is no longer needed 
743        del buffer
744
745        return True
746
747    def write_batch(self, batch):
748        """
749        Write a batch of audio frame to the file
750
751        Parameters
752        ----------
753        batch: nparray
754            The batch of audio frames to write to the video file of shape (n,self.channels,nb_samples_per_channel) if plannar is True else (n,self.channels*nb_samples_per_channel) of interleaved audio data.
755
756        Returns
757        ----------
758        bool
759            Writing was successful or not.
760        """
761        # Check params
762        # - pipe exists
763        if self.pipe is None:
764            raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram))
765        # - pipe is in write mode
766        if self.mode != PipeMode.WRITE_MODE:
767            raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename))
768        # batch is 3D (n, channels, nb samples)
769        if batch.ndim !=3:
770            raise self.AudioIOException("Wrong batch shape: {} expected 3 dimensions (n, n_channels, n_samples_per_channel).".format(batch.shape))
771        # - shape of images in batch is fine
772        if batch.shape[2] != self.channels:
773            raise self.AudioIOException("Wrong audio channels in batch: {} expected {} {}.".format(batch.shape[2], self.channels, batch.shape))
774
775        # array must have a shape (n * n_channels * n_samples_per_channel) before writing them to pipe
776        # reshape it it to (n * n_channels * n_samples_per_channel) if plannar is False
777        if not self.plannar:
778            # goes from (n, n_channels, n_samples_per_channel) to (n * n_channels * n_samples_per_channel)
779            batch = batch.transpose(0, 2, 1) # first go to (n, n_samples_per_channel, n_channels)
780            batch = batch.reshape(-1) # then to 1D array (n * n_channels * n_samples_per_channel)
781
782        # garantee to have a C continuous array
783        if not batch.flags['C_CONTIGUOUS']:
784            batch = np.ascontiguousarray(batch)
785
786        # write frame
787        buffer = batch.tobytes()
788        if self.pipe.stdin.write( buffer ) < len(buffer):
789            # say to gc that this buffer is no longer needed
790            del buffer
791            raise self.AudioIOException("Error writing batch to '{}'.".format(self.filename))
792
793        # increase frame_counter
794        self.frame_counter.frame_count += int(batch.shape[0]/self.channels) # int conversion is mandatory to avoid confusion with time as float
795              
796        # say to gc that this buffer is no longer needed
797        del buffer
798
799        return True
800
801    def iter_frames(self, with_timestamps = False):
802        """
803        Method to iterate on audio frames using AudioIO obj.
804        for audio_frame in obj.iter_frames():
805            ....
806
807        Parameters
808        ----------
809        with_timestamps: bool optional (default False)
810            If set to True, the method returns a FrameContainer object with the batch and an array containing the associated timestamps to frames
811
812        Returns
813        ----------
814        nparray or FrameContainer
815            A batch of images of shape ()
816        """
817
818        try:
819            if self.mode == PipeMode.READ_MODE:
820                while self.isOpened():
821                    frame = self.readFrame(with_timestamps)
822                    if frame is not None:
823                        yield frame
824        finally:
825            self.close()
826
827    def iter_batches(self, batch_size : int, with_timestamps = False ):
828        """
829        Method to iterate on batch ofaudio  frames using AudioIO obj.
830        for audio_batch in obj.iter_batches():
831            ....
832
833        Parameters
834        ----------
835        with_timestamps: bool optional (default False)
836            If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames
837        """
838        try:
839            if self.mode == PipeMode.READ_MODE:
840                while self.isOpened():
841                    batch = self.readBatch(batch_size, with_timestamps)
842                    if batch is not None:
843                        yield batch
844        finally:
845            self.close()
846
847    # function aliases to be compliant with original C++ version
848    getAudioTimeInSec = get_time_in_sec
849    getAudioParams = get_params
850    get_audio_time_in_sec = get_time_in_sec
851    get_audio_params = get_params
852    isOpened = is_opened
853    readFrame = read_frame
854    readBatch = read_batch
855    writeFrame = write_frame
856    writeBatch = write_batch
AudioIO(*, logLevel=16, debug=False)
234    def __init__(self, *, logLevel = 16, debug = False):
235        """
236        Create a AudioIO object giving ffmpeg/ffrobe loglevel and defining debug mode
237
238        Parameters
239        ----------
240        log_level: int (default 16)
241            Log level to pass to the underlying ffmpeg/ffprobe command.
242
243        debug: bool (default (False)
244            Show debug info. while processing video
245        """
246
247        self.mode = PipeMode.UNK_MODE
248        self.logLevel = logLevel
249        self.debug = debug
250
251        # Call init() method
252        self.init()

Create a AudioIO object giving ffmpeg/ffrobe loglevel and defining debug mode

Parameters

log_level: int (default 16) Log level to pass to the underlying ffmpeg/ffprobe command.

debug: bool (default (False) Show debug info. while processing video

@classmethod
def reader(cls, filename, *, loglevel=16, debug=False, **kwargs):
52    @classmethod
53    def reader(cls, filename, *, loglevel = 16, debug = False, **kwargs):
54        """
55        Create and open an AudioIO object in reader mode
56
57        See ``AudioIO.open`` for the full list of accepted parameters.
58        """
59        reader = cls(logLevel=loglevel, debug=debug)
60        reader.open(filename, **kwargs)
61        return reader

Create and open an AudioIO object in reader mode

See AudioIO.open for the full list of accepted parameters.

@classmethod
def writer( cls, filename, sample_rate, channels, *, loglevel=16, debug=False, **kwargs):
63    @classmethod
64    def writer(cls, filename, sample_rate, channels, *, loglevel = 16, debug = False, **kwargs):
65        """
66        Create and open an AudioIO object in writer mode
67
68        See ``AudioIO.create`` for the full list of accepted parameters.
69        """
70        writer = cls(logLevel=loglevel, debug=debug)
71        writer.create(filename, sample_rate, channels, **kwargs)
72        return writer

Create and open an AudioIO object in writer mode

See AudioIO.create for the full list of accepted parameters.

def get_corresponding_writer(self, filename, **kwargs):
75    def get_corresponding_writer(self, filename, **kwargs):
76        """
77        Method to get writer for an audio file with same sample_rate, channels as the current one AudioIO object
78
79        See `AudioIO.create` for the full list
80        of accepted parameters.
81        """
82        return AudioIO.writer(filename, self.sample_rate, self.channels, **kwargs)

Method to get writer for an audio file with same sample_rate, channels as the current one AudioIO object

See AudioIO.create for the full list of accepted parameters.

@staticmethod
def get_time_in_sec(filename, *, debug=False, logLevel=16):
100    @staticmethod
101    def get_time_in_sec(filename, *, debug=False, logLevel=16):
102        """
103        Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals).
104
105        Parameters
106        ----------
107        filename : str or path. 
108            Raw audio waveform as a 1D array.
109
110        debug : bool (default False).
111            Show debug info.
112
113        logLevel: int (default 16).
114            Log level to pass to the underlying ffmpeg/ffprobe command.
115        
116        Returns
117        ----------
118        float
119            Length in seconds of video file (including milliseconds as decimal part with 3 decimals)
120        """
121        
122        cmd = [AudioIO.paramProgram, # ffprobe
123                    '-hide_banner',
124                    '-loglevel', str(logLevel),
125                    '-show_entries', 'format=duration',
126                    '-of', 'default=noprint_wrappers=1:nokey=1',
127                    str(filename)
128                    ]
129
130        if debug == True:
131            print(' '.join(cmd))
132
133        # call ffprobe and get params in one single line
134        lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg
135        output = lpipe.stdout.readlines()
136        lpipe.terminate()
137        # transform Bytes output to one single string
138        output = ''.join( [element.decode('utf-8') for element in output])
139
140        try:
141            return float(output)
142        except (ValueError, TypeError):
143            return None

Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals).

Parameters

filename : str or path. Raw audio waveform as a 1D array.

debug : bool (default False). Show debug info.

logLevel: int (default 16). Log level to pass to the underlying ffmpeg/ffprobe command.

Returns

float Length in seconds of video file (including milliseconds as decimal part with 3 decimals)

@staticmethod
def get_params(filename, *, debug=False, logLevel=16):
145    @staticmethod
146    def get_params(filename, *, debug=False, logLevel=16):
147        """
148        Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds.
149
150        Parameters
151        ----------
152        filename : str or path.
153            Raw audio waveform as a 1D array.
154
155        debug : bool (default (False).
156            Show debug info.
157
158        log_level: int (default 16).
159            Log level to pass to the underlying ffmpeg/ffprobe command.
160
161        Returns
162        ----------
163        tuple
164            Tuple containing (channels,sample_rate) of the file
165        """
166        cmd = [AudioIO.paramProgram, # ffprobe
167                    '-hide_banner',
168                    '-loglevel', str(logLevel),
169                    '-show_entries', 'stream=channels,sample_rate',
170                    str(filename)
171                    ]
172
173        if debug == True:
174            print(' '.join(cmd))
175
176        # call ffprobe and get params in one single line
177        lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg
178        output = lpipe.stdout.readlines()
179        lpipe.terminate()
180        # transform Bytes output to one single string
181        output = ''.join( [element.decode('utf-8') for element in output])
182
183        pattern_sample_rate = r'sample_rate=(\d+)'
184        pattern_channels = r'channels=(\d+)'
185
186        # Search for values in the ffprobe output
187        match_sample_rate = re.search(pattern_sample_rate, output, flags=re.MULTILINE)
188        match_channels = re.search(pattern_channels, output, flags=re.MULTILINE)
189
190        # Extraction des valeurs
191        if match_sample_rate:
192            sample_rate = int(match_sample_rate.group(1))
193        else:
194            raise AudioIO.AudioIOException("Unable to get audio sample_rate of '" + str(filename) + "'")
195
196        if match_channels:
197            channels = int(match_channels.group(1))
198        else:
199            raise AudioIO.AudioIOException("Unable to get audio channels of '" + str(filename) + "'")
200
201        return (channels,sample_rate)
202
203        # Attributes
204        mode: PipeMode
205        """ Pipemode of the current object (default PipeMode.UNK_MODE)"""
206
207        loglevel: int
208        """ loglevel of the underlying ffmpeg backend for this object (default 16)"""
209
210        debug: bool
211        """ debug flag for this object (print debut info, default False)"""
212
213        channels: int
214        """ Number of channels of images (default -1) """
215
216        sample_rate: int
217        """ sample_rate of images (default -1) """
218
219        plannar: bool
220        """ Read/write data as plannar, i.e. not interleaved (default True) """
221
222        pipe: sp.Popen
223        """ pipe object to ffmpeg/ffprobe (default None)"""
224
225        frame_size: int
226        """ Weight in bytes of one image (default -1)"""
227
228        filename: str
229        """ Filename of the file (default None)"""
230
231        frame_counter: FrameCounter
232        """ `Framecounter` object to count ellapsed time (default None)"""

Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds.

Parameters

filename : str or path. Raw audio waveform as a 1D array.

debug : bool (default (False). Show debug info.

log_level: int (default 16). Log level to pass to the underlying ffmpeg/ffprobe command.

Returns

tuple Tuple containing (channels,sample_rate) of the file

mode
logLevel
debug
def init(self):
254    def init(self):
255        """
256        Init or reinit a AudioIO object.
257        """
258        self.channels  = -1
259        self.sample_rate = -1
260        self.plannar = True
261        self.pipe = None
262        self.frame_size = -1
263        self.filename = None
264        self.frame_counter = None
265        self.float_size = None

Init or reinit a AudioIO object.

def get_elapsed_time_as_str(self) -> str:
285    def get_elapsed_time_as_str(self) -> str:
286        """
287        Method to get elapsed time (float value represented) as str.
288
289        Returns
290        ----------
291        str or None
292            Elapsed time (float value) as str, "15.500" for instance for 15 secondes and 500 milliseconds
293            None if no frame counter are available.
294        """
295        if self.frame_counter is None:
296            return None
297        return self.frame_counter.get_elapsed_time_as_str()

Method to get elapsed time (float value represented) as str.

Returns

str or None Elapsed time (float value) as str, "15.500" for instance for 15 secondes and 500 milliseconds None if no frame counter are available.

def get_formated_elapsed_time_as_str(self, show_ms=True) -> str:
299    def get_formated_elapsed_time_as_str(self,show_ms=True) -> str:
300        """
301        Method to get elapsed time (hour format) as str.
302
303        Returns
304        ----------
305        str or None
306            Elapsed time (float value) as str, "00:00:15.500" for instance for 15 secondes and 500 milliseconds
307            None if no frame counter are available.
308        """
309        if self.frame_counter is None:
310            return None
311        return self.frame_counter.get_formated_elapsed_time_as_str()

Method to get elapsed time (hour format) as str.

Returns

str or None Elapsed time (float value) as str, "00:00:15.500" for instance for 15 secondes and 500 milliseconds None if no frame counter are available.

def get_elapsed_time(self) -> float:
313    def get_elapsed_time(self) -> float:
314        """
315        Method to get elapsed time as float value rounded to 3 decimals.
316
317        Returns
318        ----------
319        float or None
320            Elapsed time (float value) as str, 15.500 for instance for 15 secondes and 500 milliseconds
321            None if no frame counter are available.
322        """
323        if self.frame_counter is None:
324            return None
325        return self.frame_counter.get_elapsed_time()

Method to get elapsed time as float value rounded to 3 decimals.

Returns

float or None Elapsed time (float value) as str, 15.500 for instance for 15 secondes and 500 milliseconds None if no frame counter are available.

def is_opened(self) -> bool:
327    def is_opened(self) -> bool:
328        """
329        Method to get status of the underlying pipe to ffmpeg.
330
331        Returns
332        ----------
333        bool
334            True if pipe is opened (reading or writing mode), False if not.
335        """
336        # is the pip opened?
337        if self.pipe is not None and self.pipe.poll() is None:
338            return True
339
340        return False

Method to get status of the underlying pipe to ffmpeg.

Returns

bool True if pipe is opened (reading or writing mode), False if not.

def close(self):
342    def close(self):
343        """
344        Method to close current pipe to ffmpeg (if any). Ffmpeg/ffprobe  will be terminated. Object can be reused using open or create methods.
345        """
346        if self.pipe is not None:
347            if self.mode == PipeMode.WRITE_MODE:
348                # killing will make ffmpeg not finish properly the job, close the pipe
349                # to let it know that no more data are comming
350                self.pipe.stdin.close()
351            else: # self.mode == PipeMode.READ_MODE
352                # in read mode, no need to be nice, send SIGTERM on Linux,/Kill it on windows
353                self.pipe.kill()
354
355            # wait for subprocess to end
356            self.pipe.wait()
357
358        # reinit object for later use
359        self.init()

Method to close current pipe to ffmpeg (if any). Ffmpeg/ffprobe will be terminated. Object can be reused using open or create methods.

def create( self, filename, sample_rate, channels, *, writeOverExistingFile=False, outputEncoding=<AudioFormat.PCM32LE: 'pcm_f32le'>, encodingParams=None, plannar=True):
361    def create( self, filename, sample_rate, channels, *, writeOverExistingFile = False,
362                outputEncoding = AudioFormat.PCM32LE, encodingParams = None, plannar = True ):
363        """
364        Method to create a audio file using parametrized access through ffmpeg. Importante note: calling create
365        on a AudioIO will close any former open video.
366
367        Parameters
368        ----------
369        filename: str or path
370            filename of path to the file (mp4, avi, ...)
371
372        sample_rate: int
373            If defined as a positive value, sample_rates of the output file will be set to this value.
374
375        channels: int
376            If defined as a positive value, number of channels of output file will be set to this value.
377
378        fps:
379            If defined as a positive value, fps of input video will be set to this value.
380
381        outputEncoding: AudioFormat optional (default AudioFormat.PCM32LE)
382            Define audio format for samples. Possible value is AudioFormat.PCM32LE.
383
384        encodingParams: str optional (default None)
385            Parameter to pass to ffmpeg to encode video like audio filters.
386
387        plannar : bool optionnal (default True)
388            Input data to write are grouped by channel if True, interleaved instead.
389
390        Returns
391        ----------
392        bool
393            Was the creation successfull
394        """
395
396        # Close if already opened
397        self.close()
398
399        # Set geometry/fps of the video stream from params
400        self.sample_rate = int(sample_rate)
401        self.channels = int(channels)
402        self.plannar = plannar
403
404        # Compute size of float 32, usefull if we want later to use other floating point convention
405        self.float_size = int(np.dtype(np.float32).itemsize)
406
407        # Check params
408        if self.sample_rate <= 0 or self.channels <= 0:
409            raise self.AudioIOException("Bad parameters: sample_rate={}, channels={}".format(self.sample_rate,self.channels))
410
411        # To write audio, we do not need to know in advance frame size, we will write x values of n bytes
412        self.frame_size = None
413
414        # Video params are set, open the video
415        cmd = [self.audioProgram] # ffmpeg
416
417        if writeOverExistingFile == True:
418            cmd.extend(['-y'])
419
420        cmd.extend(['-hide_banner',
421            '-nostats',
422            '-loglevel', str(self.logLevel),
423            '-f', 'f32le', '-acodec', outputEncoding.value, # input expected coding
424            '-ar', f"{self.sample_rate}",
425            '-ac', f"{self.channels}",
426            '-i', '-'])
427
428        if encodingParams is not None:
429            cmd.extend(encodingParams.split())
430
431        # Audio filename converted to str (for Path values)
432        cmd.extend( ['-vn', str(filename) ] )
433
434        if self.debug == True:
435            print( ' '.join(cmd), file=sys.stderr )
436
437        # store filename and set mode
438        self.filename = str(filename)
439        self.mode = PipeMode.WRITE_MODE
440
441        # call ffmpeg in write mode
442        try:
443            self.pipe = sp.Popen(cmd, stdin=sp.PIPE)
444            self.frame_counter = FrameCounter(self.sample_rate)
445        except Exception as e:
446            # if pipe failed, reinit object and raise exception
447            self.init()
448            raise
449
450        return True

Method to create a audio file using parametrized access through ffmpeg. Importante note: calling create on a AudioIO will close any former open video.

Parameters

filename: str or path filename of path to the file (mp4, avi, ...)

sample_rate: int If defined as a positive value, sample_rates of the output file will be set to this value.

channels: int If defined as a positive value, number of channels of output file will be set to this value.

fps: If defined as a positive value, fps of input video will be set to this value.

outputEncoding: AudioFormat optional (default AudioFormat.PCM32LE) Define audio format for samples. Possible value is AudioFormat.PCM32LE.

encodingParams: str optional (default None) Parameter to pass to ffmpeg to encode video like audio filters.

plannar : bool optionnal (default True) Input data to write are grouped by channel if True, interleaved instead.

Returns

bool Was the creation successfull

def open( self, filename, *, sample_rate=-1, channels=-1, inputEncoding=<AudioFormat.PCM32LE: 'pcm_f32le'>, decodingParams=None, frame_size=1.0, plannar=True, start_time=0.0):
452    def open( self, filename, *, sample_rate = -1, channels = -1, inputEncoding = AudioFormat.PCM32LE,
453                    decodingParams = None, frame_size = 1.0, plannar = True, start_time = 0.0 ):
454        """
455        Method to read (video file containing) audio using parametrized access through ffmpeg. Importante note: calling open
456        on a AudioIO will close any former open file.
457
458        Parameters
459        ----------
460        filename: str or path
461            filename of path to the file (mp4, avi, ...)
462
463        sample_rate: int optional (default -1)
464            If defined as a positive value, sample rate of the input audio will be converted to this value.
465
466        channels: int optional (default -1)
467            If defined as a positive value, number of channels of the input audio will converted to this value.
468
469        inputEncoding: AudioFormat optional (default AudioFormat.PCM32LE)
470            Define audio format for samples. Possible value is AudioFormat.PCM32LE.
471
472        decodingParams: str optional (default None)
473            Parameter to pass to ffmpeg to decode video like audio filters.
474
475        plannar: bool optionnal (default True)
476            Group audio samples per channel if True. Else, samples are interleaved.
477
478        frame_size: int or float (default 1.0)
479            If frame_size is an int, it is the number of expected samples in each frame, for instance 8000 for 8000 samples.
480            if frame_size is a float, it is considered as a time in seconds for each audio frame, for instance 1.0 for 1 second, 0.010 for 10 ms.
481            Number of samples in this case is computed using frame_size and sample_rate as int(frame_size * sample_rate)
482
483        start_time: float optional (default 0.0)
484            Define the reading start time. If not set, reading at beginning of the file.
485
486        Returns
487        ----------
488        bool
489            Was the opening successfull
490        """
491
492        # Close if already opened
493        self.close()
494
495        # Force conversion of parameters
496        channels = int(channels)
497        sample_rate = float(sample_rate)
498
499        self.plannar = plannar
500
501        # Compute size of float 32, usefull if we want later to use other floating point convention
502        self.float_size = int(np.dtype(np.float32).itemsize)
503
504        # get parameters from file if needed:
505        if sample_rate <= 0 or channels <= 0:
506            self.channels, self.sample_rate = self.getAudioParams(filename)
507
508        # check if parameters ask to overide video parameters
509        if channels > 0:
510            self.channels = channels
511        if sample_rate > 0:
512            self.sample_rate = sample_rate
513
514        # check parameters
515
516        if isinstance(frame_size,float):
517            # time in seconds
518            self.frame_size = int(frame_size*self.sample_rate)
519        elif isinstance(frame_size,int):
520            # number of samples
521            self.frame_size = frame_size
522        else:
523            # to do
524            pass
525
526        # Video params are set, open the video
527        cmd = [self.audioProgram, # ffmpeg
528                    '-hide_banner',
529                    '-nostats',
530                    '-loglevel', str(self.logLevel)]
531
532        if decodingParams is not None:
533            cmd.extend([decodingParams.split()])
534
535        if start_time < 0.0:
536            pass
537        elif start_time > 0.0:
538            cmd.extend(["-ss", f"{start_time}"])            
539
540        cmd.extend( ['-i', str(filename),
541                     '-f', 'f32le', '-acodec', inputEncoding.value, # input expected coding
542                     '-ar', f"{self.sample_rate}",
543                     '-ac', f"{self.channels}",
544                     '-' # output to stdout
545                    ]
546                )
547
548        if self.debug == True:
549            print( ' '.join(cmd) )
550
551        # store filename and set mode to READ_MODE
552        self.filename = str(filename)
553        self.mode = PipeMode.READ_MODE
554
555        # call ffmpeg in read mode
556        try:
557            self.pipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg/ffprobe
558            self.frame_counter = FrameCounter(self.sample_rate)
559            if start_time > 0.0:
560                self.frame_counter += start_time # adding with float means adding time
561        except Exception as e:
562            # if pipe failed, reinit object and raise exception
563            self.init()
564            raise
565
566        return True

Method to read (video file containing) audio using parametrized access through ffmpeg. Importante note: calling open on a AudioIO will close any former open file.

Parameters

filename: str or path filename of path to the file (mp4, avi, ...)

sample_rate: int optional (default -1) If defined as a positive value, sample rate of the input audio will be converted to this value.

channels: int optional (default -1) If defined as a positive value, number of channels of the input audio will converted to this value.

inputEncoding: AudioFormat optional (default AudioFormat.PCM32LE) Define audio format for samples. Possible value is AudioFormat.PCM32LE.

decodingParams: str optional (default None) Parameter to pass to ffmpeg to decode video like audio filters.

plannar: bool optionnal (default True) Group audio samples per channel if True. Else, samples are interleaved.

frame_size: int or float (default 1.0) If frame_size is an int, it is the number of expected samples in each frame, for instance 8000 for 8000 samples. if frame_size is a float, it is considered as a time in seconds for each audio frame, for instance 1.0 for 1 second, 0.010 for 10 ms. Number of samples in this case is computed using frame_size and sample_rate as int(frame_size * sample_rate)

start_time: float optional (default 0.0) Define the reading start time. If not set, reading at beginning of the file.

Returns

bool Was the opening successfull

def read_frame(self, with_timestamps=False):
568    def read_frame(self, with_timestamps = False):
569        """
570        Read next frame from the audio file
571
572        Parameters
573        ----------
574        with_timestamps: bool optional (default False)
575            If set to True, the method returns a ``FrameContainer`` with the audio and an array containing the associated timestamp(s)
576
577        Returns
578        ----------
579        nparray or FrameContainer
580            A frame of shape (self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A frame
581            of shape (self.channels*self.frame_size) with interleaved data if self.plannar is False.
582            if with_timestamps is True, the return object is a FrameContainer with the audio data in ``FrameContainer.data`` and
583            the associated timestamp in ``FrameContainer.timestamps`` as an array (one element).
584        """
585
586        if self.pipe is None:
587            raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading a frame.".format(self.audioProgram))
588        # - pipe is in write mode
589        if self.mode != PipeMode.READ_MODE:
590            raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename))
591
592        if with_timestamps:
593            # get elapsed time in video, it is time of next frame(s)
594            current_elapsed_time = self.get_elapsed_time()
595
596        # read rgb image from pipe
597        toread = self.frame_size*self.float_size
598        buffer = self.pipe.stdout.read(toread)
599
600        if buffer == b"":
601            # not considered as an error, no more frame, no exception
602            return None
603
604        # get numpy UINT8 array from buffer
605        audio = np.frombuffer(buffer, dtype = np.float32).reshape(len(buffer)//self.float_size, self.channels)
606
607        # make it plannar (or not)
608        if self.plannar:
609            #transpose it
610            audio = audio.T
611
612        # increase frame_counter
613        self.frame_counter.frame_count += (self.frame_size * self.channels)
614
615        # say to gc that this buffer is no longer needed
616        del buffer
617
618        if with_timestamps:
619            return FrameContainer(1, audio, self.frame_size/self.sample_rate, current_elapsed_time)
620        
621        return audio

Read next frame from the audio file

Parameters

with_timestamps: bool optional (default False) If set to True, the method returns a FrameContainer with the audio and an array containing the associated timestamp(s)

Returns

nparray or FrameContainer A frame of shape (self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A frame of shape (self.channels*self.frame_size) with interleaved data if self.plannar is False. if with_timestamps is True, the return object is a FrameContainer with the audio data in FrameContainer.data and the associated timestamp in FrameContainer.timestamps as an array (one element).

def read_batch(self, numberOfFrames, with_timestamps=False):
623    def read_batch(self, numberOfFrames, with_timestamps = False):
624        """
625        Read next batch of audio from the file
626
627        Parameters
628        ----------
629        number_of_frames: int
630            Number of desired images within the batch. The last batch from the file may have less images.
631            
632        with_timestamps: bool optional (default False)
633            If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames
634
635        Returns
636        ----------
637        nparray or FrameContainer
638            A batch of shape (n, self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A batch
639            of shape (n, self.channels*self.frame_size) with interleaved data if self.plannar is False.
640            if with_timestamps is True, the return object is a FrameContainer with the audio batch in ``FrameContainer.data`` and
641            the associated timestamp in ``FrameContainer.timestamps`` as an array (one element for each audio frame).
642        """
643
644        if self.pipe is None:
645            raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading frames.".format(self.audioProgram))
646        # - pipe is in write mode
647        if self.mode != PipeMode.READ_MODE:
648            raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename))
649
650        if with_timestamps:
651            # get elapsed time in video, it is time of next frame(s)
652            current_elapsed_time = self.get_elapsed_time()
653
654        # try to read complete batch
655        toread = self.frame_size*self.float_size*self.channels*numberOfFrames
656        buffer = self.pipe.stdout.read(toread)
657
658        # check if we are at the end of the buffer
659        if buffer == b"":
660            # not considered as an error, no more frame, no exception
661            return None
662
663        # compute actual number of Frames
664        # do we have a full batch ? Computer standard division
665        actualNbFrames = len(buffer)/(self.frame_size*self.float_size*self.channels)
666
667        if actualNbFrames.is_integer():
668            actualNbFrames = int(actualNbFrames)
669            # get and reshape batch from buffer
670            batch = np.frombuffer(buffer, dtype = np.float32).reshape((actualNbFrames, self.frame_size, self.channels,))
671        else:
672            # We do not have a full batch, we are at end of the stream or ffmpeg has been stopped/killed
673            # Compute frame size in samples knowing len of buffer, self.float_size, self.channels and numberOfFrames
674            l_frame_size = len(buffer)//(numberOfFrames*self.float_size*self.channels)
675            if l_frame_size <= 0:
676                # not considered as an error, no more frame, no exception
677                return None
678            # get and reshape batch from buffer
679            batch = np.frombuffer(buffer, dtype = np.float32).reshape((numberOfFrames, l_frame_size, self.channels,))
680
681        if self.plannar:
682            batch = batch.transpose(0, 2, 1)
683
684        # increase frame_counter
685        self.frame_counter.frame_count += (actualNbFrames * self.frame_size * self.channels)
686        
687        # say to gc that this buffer is no longer needed
688        del buffer
689
690        if with_timestamps:
691            return FrameContainer( actualNbFrames, batch, self.frame_size/self.sample_rate, current_elapsed_time)
692        
693        return batch

Read next batch of audio from the file

Parameters

number_of_frames: int Number of desired images within the batch. The last batch from the file may have less images.

with_timestamps: bool optional (default False) If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames

Returns

nparray or FrameContainer A batch of shape (n, self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A batch of shape (n, self.channels*self.frame_size) with interleaved data if self.plannar is False. if with_timestamps is True, the return object is a FrameContainer with the audio batch in FrameContainer.data and the associated timestamp in FrameContainer.timestamps as an array (one element for each audio frame).

def write_frame(self, audio) -> bool:
695    def write_frame(self, audio) -> bool:
696        """
697        Write an audio frame to the file
698
699        Parameters
700        ----------
701        audio: nparray
702            The audio frame to write to the video file of shape (self.channels,nb_samples_per_channel) if plannar is True else (self.channels*nb_samples_per_channel).
703
704        Returns
705        ----------
706        bool
707            Writing was successful or not.
708        """
709        # Check params
710        # - pipe exists
711        if self.pipe is None:
712            raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram))
713        # - pipe is in write mode
714        if self.mode != PipeMode.WRITE_MODE:
715            raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename))
716        # - shape of image is fine, thus we have pixels for a full compatible frame
717        if audio.shape[0] != self.channels:
718            raise self.AudioIOException("Wong audio shape: {} expected ({}, nb_samples_to_write).".format(audio.shape,self.channels))
719        # - type of data is Float32
720        if audio.dtype != np.float32:
721            raise self.AudioIOException("Wong audio type: {} expected np.float32.".format(audio.dtype))
722
723        # array must have a shape (channels, samples), reshape it it to (samples, channels) if plannar
724        if not self.plannar:
725            audio = audio.reshape(-1)
726
727        # print( audio.shape )
728
729        # garantee to have a C continuous array
730        if not audio.flags['C_CONTIGUOUS']:
731            a = np.ascontiguousarray(a) 
732
733        # write frame
734        buffer = audio.tobytes()
735        if self.pipe.stdin.write( buffer ) < len(buffer):
736            print( f"Error writing frame to {self.filename}" )
737            return False
738
739        # increase frame_counter
740        self.frame_counter.frame_count += (audio.shape[1] * self.channels)
741
742        # say to gc that this buffer is no longer needed 
743        del buffer
744
745        return True

Write an audio frame to the file

Parameters

audio: nparray The audio frame to write to the video file of shape (self.channels,nb_samples_per_channel) if plannar is True else (self.channels*nb_samples_per_channel).

Returns

bool Writing was successful or not.

def write_batch(self, batch):
747    def write_batch(self, batch):
748        """
749        Write a batch of audio frame to the file
750
751        Parameters
752        ----------
753        batch: nparray
754            The batch of audio frames to write to the video file of shape (n,self.channels,nb_samples_per_channel) if plannar is True else (n,self.channels*nb_samples_per_channel) of interleaved audio data.
755
756        Returns
757        ----------
758        bool
759            Writing was successful or not.
760        """
761        # Check params
762        # - pipe exists
763        if self.pipe is None:
764            raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram))
765        # - pipe is in write mode
766        if self.mode != PipeMode.WRITE_MODE:
767            raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename))
768        # batch is 3D (n, channels, nb samples)
769        if batch.ndim !=3:
770            raise self.AudioIOException("Wrong batch shape: {} expected 3 dimensions (n, n_channels, n_samples_per_channel).".format(batch.shape))
771        # - shape of images in batch is fine
772        if batch.shape[2] != self.channels:
773            raise self.AudioIOException("Wrong audio channels in batch: {} expected {} {}.".format(batch.shape[2], self.channels, batch.shape))
774
775        # array must have a shape (n * n_channels * n_samples_per_channel) before writing them to pipe
776        # reshape it it to (n * n_channels * n_samples_per_channel) if plannar is False
777        if not self.plannar:
778            # goes from (n, n_channels, n_samples_per_channel) to (n * n_channels * n_samples_per_channel)
779            batch = batch.transpose(0, 2, 1) # first go to (n, n_samples_per_channel, n_channels)
780            batch = batch.reshape(-1) # then to 1D array (n * n_channels * n_samples_per_channel)
781
782        # garantee to have a C continuous array
783        if not batch.flags['C_CONTIGUOUS']:
784            batch = np.ascontiguousarray(batch)
785
786        # write frame
787        buffer = batch.tobytes()
788        if self.pipe.stdin.write( buffer ) < len(buffer):
789            # say to gc that this buffer is no longer needed
790            del buffer
791            raise self.AudioIOException("Error writing batch to '{}'.".format(self.filename))
792
793        # increase frame_counter
794        self.frame_counter.frame_count += int(batch.shape[0]/self.channels) # int conversion is mandatory to avoid confusion with time as float
795              
796        # say to gc that this buffer is no longer needed
797        del buffer
798
799        return True

Write a batch of audio frame to the file

Parameters

batch: nparray The batch of audio frames to write to the video file of shape (n,self.channels,nb_samples_per_channel) if plannar is True else (n,self.channels*nb_samples_per_channel) of interleaved audio data.

Returns

bool Writing was successful or not.

def iter_frames(self, with_timestamps=False):
801    def iter_frames(self, with_timestamps = False):
802        """
803        Method to iterate on audio frames using AudioIO obj.
804        for audio_frame in obj.iter_frames():
805            ....
806
807        Parameters
808        ----------
809        with_timestamps: bool optional (default False)
810            If set to True, the method returns a FrameContainer object with the batch and an array containing the associated timestamps to frames
811
812        Returns
813        ----------
814        nparray or FrameContainer
815            A batch of images of shape ()
816        """
817
818        try:
819            if self.mode == PipeMode.READ_MODE:
820                while self.isOpened():
821                    frame = self.readFrame(with_timestamps)
822                    if frame is not None:
823                        yield frame
824        finally:
825            self.close()

Method to iterate on audio frames using AudioIO obj. for audio_frame in obj.iter_frames(): ....

Parameters

with_timestamps: bool optional (default False) If set to True, the method returns a FrameContainer object with the batch and an array containing the associated timestamps to frames

Returns

nparray or FrameContainer A batch of images of shape ()

def iter_batches(self, batch_size: int, with_timestamps=False):
827    def iter_batches(self, batch_size : int, with_timestamps = False ):
828        """
829        Method to iterate on batch ofaudio  frames using AudioIO obj.
830        for audio_batch in obj.iter_batches():
831            ....
832
833        Parameters
834        ----------
835        with_timestamps: bool optional (default False)
836            If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames
837        """
838        try:
839            if self.mode == PipeMode.READ_MODE:
840                while self.isOpened():
841                    batch = self.readBatch(batch_size, with_timestamps)
842                    if batch is not None:
843                        yield batch
844        finally:
845            self.close()

Method to iterate on batch ofaudio frames using AudioIO obj. for audio_batch in obj.iter_batches(): ....

Parameters

with_timestamps: bool optional (default False) If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames

@staticmethod
def getAudioTimeInSec(filename, *, debug=False, logLevel=16):
100    @staticmethod
101    def get_time_in_sec(filename, *, debug=False, logLevel=16):
102        """
103        Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals).
104
105        Parameters
106        ----------
107        filename : str or path. 
108            Raw audio waveform as a 1D array.
109
110        debug : bool (default False).
111            Show debug info.
112
113        logLevel: int (default 16).
114            Log level to pass to the underlying ffmpeg/ffprobe command.
115        
116        Returns
117        ----------
118        float
119            Length in seconds of video file (including milliseconds as decimal part with 3 decimals)
120        """
121        
122        cmd = [AudioIO.paramProgram, # ffprobe
123                    '-hide_banner',
124                    '-loglevel', str(logLevel),
125                    '-show_entries', 'format=duration',
126                    '-of', 'default=noprint_wrappers=1:nokey=1',
127                    str(filename)
128                    ]
129
130        if debug == True:
131            print(' '.join(cmd))
132
133        # call ffprobe and get params in one single line
134        lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg
135        output = lpipe.stdout.readlines()
136        lpipe.terminate()
137        # transform Bytes output to one single string
138        output = ''.join( [element.decode('utf-8') for element in output])
139
140        try:
141            return float(output)
142        except (ValueError, TypeError):
143            return None

Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals).

Parameters

filename : str or path. Raw audio waveform as a 1D array.

debug : bool (default False). Show debug info.

logLevel: int (default 16). Log level to pass to the underlying ffmpeg/ffprobe command.

Returns

float Length in seconds of video file (including milliseconds as decimal part with 3 decimals)

@staticmethod
def getAudioParams(filename, *, debug=False, logLevel=16):
145    @staticmethod
146    def get_params(filename, *, debug=False, logLevel=16):
147        """
148        Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds.
149
150        Parameters
151        ----------
152        filename : str or path.
153            Raw audio waveform as a 1D array.
154
155        debug : bool (default (False).
156            Show debug info.
157
158        log_level: int (default 16).
159            Log level to pass to the underlying ffmpeg/ffprobe command.
160
161        Returns
162        ----------
163        tuple
164            Tuple containing (channels,sample_rate) of the file
165        """
166        cmd = [AudioIO.paramProgram, # ffprobe
167                    '-hide_banner',
168                    '-loglevel', str(logLevel),
169                    '-show_entries', 'stream=channels,sample_rate',
170                    str(filename)
171                    ]
172
173        if debug == True:
174            print(' '.join(cmd))
175
176        # call ffprobe and get params in one single line
177        lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg
178        output = lpipe.stdout.readlines()
179        lpipe.terminate()
180        # transform Bytes output to one single string
181        output = ''.join( [element.decode('utf-8') for element in output])
182
183        pattern_sample_rate = r'sample_rate=(\d+)'
184        pattern_channels = r'channels=(\d+)'
185
186        # Search for values in the ffprobe output
187        match_sample_rate = re.search(pattern_sample_rate, output, flags=re.MULTILINE)
188        match_channels = re.search(pattern_channels, output, flags=re.MULTILINE)
189
190        # Extraction des valeurs
191        if match_sample_rate:
192            sample_rate = int(match_sample_rate.group(1))
193        else:
194            raise AudioIO.AudioIOException("Unable to get audio sample_rate of '" + str(filename) + "'")
195
196        if match_channels:
197            channels = int(match_channels.group(1))
198        else:
199            raise AudioIO.AudioIOException("Unable to get audio channels of '" + str(filename) + "'")
200
201        return (channels,sample_rate)
202
203        # Attributes
204        mode: PipeMode
205        """ Pipemode of the current object (default PipeMode.UNK_MODE)"""
206
207        loglevel: int
208        """ loglevel of the underlying ffmpeg backend for this object (default 16)"""
209
210        debug: bool
211        """ debug flag for this object (print debut info, default False)"""
212
213        channels: int
214        """ Number of channels of images (default -1) """
215
216        sample_rate: int
217        """ sample_rate of images (default -1) """
218
219        plannar: bool
220        """ Read/write data as plannar, i.e. not interleaved (default True) """
221
222        pipe: sp.Popen
223        """ pipe object to ffmpeg/ffprobe (default None)"""
224
225        frame_size: int
226        """ Weight in bytes of one image (default -1)"""
227
228        filename: str
229        """ Filename of the file (default None)"""
230
231        frame_counter: FrameCounter
232        """ `Framecounter` object to count ellapsed time (default None)"""

Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds.

Parameters

filename : str or path. Raw audio waveform as a 1D array.

debug : bool (default (False). Show debug info.

log_level: int (default 16). Log level to pass to the underlying ffmpeg/ffprobe command.

Returns

tuple Tuple containing (channels,sample_rate) of the file

@staticmethod
def get_audio_time_in_sec(filename, *, debug=False, logLevel=16):
100    @staticmethod
101    def get_time_in_sec(filename, *, debug=False, logLevel=16):
102        """
103        Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals).
104
105        Parameters
106        ----------
107        filename : str or path. 
108            Raw audio waveform as a 1D array.
109
110        debug : bool (default False).
111            Show debug info.
112
113        logLevel: int (default 16).
114            Log level to pass to the underlying ffmpeg/ffprobe command.
115        
116        Returns
117        ----------
118        float
119            Length in seconds of video file (including milliseconds as decimal part with 3 decimals)
120        """
121        
122        cmd = [AudioIO.paramProgram, # ffprobe
123                    '-hide_banner',
124                    '-loglevel', str(logLevel),
125                    '-show_entries', 'format=duration',
126                    '-of', 'default=noprint_wrappers=1:nokey=1',
127                    str(filename)
128                    ]
129
130        if debug == True:
131            print(' '.join(cmd))
132
133        # call ffprobe and get params in one single line
134        lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg
135        output = lpipe.stdout.readlines()
136        lpipe.terminate()
137        # transform Bytes output to one single string
138        output = ''.join( [element.decode('utf-8') for element in output])
139
140        try:
141            return float(output)
142        except (ValueError, TypeError):
143            return None

Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals).

Parameters

filename : str or path. Raw audio waveform as a 1D array.

debug : bool (default False). Show debug info.

logLevel: int (default 16). Log level to pass to the underlying ffmpeg/ffprobe command.

Returns

float Length in seconds of video file (including milliseconds as decimal part with 3 decimals)

@staticmethod
def get_audio_params(filename, *, debug=False, logLevel=16):
145    @staticmethod
146    def get_params(filename, *, debug=False, logLevel=16):
147        """
148        Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds.
149
150        Parameters
151        ----------
152        filename : str or path.
153            Raw audio waveform as a 1D array.
154
155        debug : bool (default (False).
156            Show debug info.
157
158        log_level: int (default 16).
159            Log level to pass to the underlying ffmpeg/ffprobe command.
160
161        Returns
162        ----------
163        tuple
164            Tuple containing (channels,sample_rate) of the file
165        """
166        cmd = [AudioIO.paramProgram, # ffprobe
167                    '-hide_banner',
168                    '-loglevel', str(logLevel),
169                    '-show_entries', 'stream=channels,sample_rate',
170                    str(filename)
171                    ]
172
173        if debug == True:
174            print(' '.join(cmd))
175
176        # call ffprobe and get params in one single line
177        lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg
178        output = lpipe.stdout.readlines()
179        lpipe.terminate()
180        # transform Bytes output to one single string
181        output = ''.join( [element.decode('utf-8') for element in output])
182
183        pattern_sample_rate = r'sample_rate=(\d+)'
184        pattern_channels = r'channels=(\d+)'
185
186        # Search for values in the ffprobe output
187        match_sample_rate = re.search(pattern_sample_rate, output, flags=re.MULTILINE)
188        match_channels = re.search(pattern_channels, output, flags=re.MULTILINE)
189
190        # Extraction des valeurs
191        if match_sample_rate:
192            sample_rate = int(match_sample_rate.group(1))
193        else:
194            raise AudioIO.AudioIOException("Unable to get audio sample_rate of '" + str(filename) + "'")
195
196        if match_channels:
197            channels = int(match_channels.group(1))
198        else:
199            raise AudioIO.AudioIOException("Unable to get audio channels of '" + str(filename) + "'")
200
201        return (channels,sample_rate)
202
203        # Attributes
204        mode: PipeMode
205        """ Pipemode of the current object (default PipeMode.UNK_MODE)"""
206
207        loglevel: int
208        """ loglevel of the underlying ffmpeg backend for this object (default 16)"""
209
210        debug: bool
211        """ debug flag for this object (print debut info, default False)"""
212
213        channels: int
214        """ Number of channels of images (default -1) """
215
216        sample_rate: int
217        """ sample_rate of images (default -1) """
218
219        plannar: bool
220        """ Read/write data as plannar, i.e. not interleaved (default True) """
221
222        pipe: sp.Popen
223        """ pipe object to ffmpeg/ffprobe (default None)"""
224
225        frame_size: int
226        """ Weight in bytes of one image (default -1)"""
227
228        filename: str
229        """ Filename of the file (default None)"""
230
231        frame_counter: FrameCounter
232        """ `Framecounter` object to count ellapsed time (default None)"""

Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds.

Parameters

filename : str or path. Raw audio waveform as a 1D array.

debug : bool (default (False). Show debug info.

log_level: int (default 16). Log level to pass to the underlying ffmpeg/ffprobe command.

Returns

tuple Tuple containing (channels,sample_rate) of the file

def isOpened(self) -> bool:
327    def is_opened(self) -> bool:
328        """
329        Method to get status of the underlying pipe to ffmpeg.
330
331        Returns
332        ----------
333        bool
334            True if pipe is opened (reading or writing mode), False if not.
335        """
336        # is the pip opened?
337        if self.pipe is not None and self.pipe.poll() is None:
338            return True
339
340        return False

Method to get status of the underlying pipe to ffmpeg.

Returns

bool True if pipe is opened (reading or writing mode), False if not.

def readFrame(self, with_timestamps=False):
568    def read_frame(self, with_timestamps = False):
569        """
570        Read next frame from the audio file
571
572        Parameters
573        ----------
574        with_timestamps: bool optional (default False)
575            If set to True, the method returns a ``FrameContainer`` with the audio and an array containing the associated timestamp(s)
576
577        Returns
578        ----------
579        nparray or FrameContainer
580            A frame of shape (self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A frame
581            of shape (self.channels*self.frame_size) with interleaved data if self.plannar is False.
582            if with_timestamps is True, the return object is a FrameContainer with the audio data in ``FrameContainer.data`` and
583            the associated timestamp in ``FrameContainer.timestamps`` as an array (one element).
584        """
585
586        if self.pipe is None:
587            raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading a frame.".format(self.audioProgram))
588        # - pipe is in write mode
589        if self.mode != PipeMode.READ_MODE:
590            raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename))
591
592        if with_timestamps:
593            # get elapsed time in video, it is time of next frame(s)
594            current_elapsed_time = self.get_elapsed_time()
595
596        # read rgb image from pipe
597        toread = self.frame_size*self.float_size
598        buffer = self.pipe.stdout.read(toread)
599
600        if buffer == b"":
601            # not considered as an error, no more frame, no exception
602            return None
603
604        # get numpy UINT8 array from buffer
605        audio = np.frombuffer(buffer, dtype = np.float32).reshape(len(buffer)//self.float_size, self.channels)
606
607        # make it plannar (or not)
608        if self.plannar:
609            #transpose it
610            audio = audio.T
611
612        # increase frame_counter
613        self.frame_counter.frame_count += (self.frame_size * self.channels)
614
615        # say to gc that this buffer is no longer needed
616        del buffer
617
618        if with_timestamps:
619            return FrameContainer(1, audio, self.frame_size/self.sample_rate, current_elapsed_time)
620        
621        return audio

Read next frame from the audio file

Parameters

with_timestamps: bool optional (default False) If set to True, the method returns a FrameContainer with the audio and an array containing the associated timestamp(s)

Returns

nparray or FrameContainer A frame of shape (self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A frame of shape (self.channels*self.frame_size) with interleaved data if self.plannar is False. if with_timestamps is True, the return object is a FrameContainer with the audio data in FrameContainer.data and the associated timestamp in FrameContainer.timestamps as an array (one element).

def readBatch(self, numberOfFrames, with_timestamps=False):
623    def read_batch(self, numberOfFrames, with_timestamps = False):
624        """
625        Read next batch of audio from the file
626
627        Parameters
628        ----------
629        number_of_frames: int
630            Number of desired images within the batch. The last batch from the file may have less images.
631            
632        with_timestamps: bool optional (default False)
633            If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames
634
635        Returns
636        ----------
637        nparray or FrameContainer
638            A batch of shape (n, self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A batch
639            of shape (n, self.channels*self.frame_size) with interleaved data if self.plannar is False.
640            if with_timestamps is True, the return object is a FrameContainer with the audio batch in ``FrameContainer.data`` and
641            the associated timestamp in ``FrameContainer.timestamps`` as an array (one element for each audio frame).
642        """
643
644        if self.pipe is None:
645            raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading frames.".format(self.audioProgram))
646        # - pipe is in write mode
647        if self.mode != PipeMode.READ_MODE:
648            raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename))
649
650        if with_timestamps:
651            # get elapsed time in video, it is time of next frame(s)
652            current_elapsed_time = self.get_elapsed_time()
653
654        # try to read complete batch
655        toread = self.frame_size*self.float_size*self.channels*numberOfFrames
656        buffer = self.pipe.stdout.read(toread)
657
658        # check if we are at the end of the buffer
659        if buffer == b"":
660            # not considered as an error, no more frame, no exception
661            return None
662
663        # compute actual number of Frames
664        # do we have a full batch ? Computer standard division
665        actualNbFrames = len(buffer)/(self.frame_size*self.float_size*self.channels)
666
667        if actualNbFrames.is_integer():
668            actualNbFrames = int(actualNbFrames)
669            # get and reshape batch from buffer
670            batch = np.frombuffer(buffer, dtype = np.float32).reshape((actualNbFrames, self.frame_size, self.channels,))
671        else:
672            # We do not have a full batch, we are at end of the stream or ffmpeg has been stopped/killed
673            # Compute frame size in samples knowing len of buffer, self.float_size, self.channels and numberOfFrames
674            l_frame_size = len(buffer)//(numberOfFrames*self.float_size*self.channels)
675            if l_frame_size <= 0:
676                # not considered as an error, no more frame, no exception
677                return None
678            # get and reshape batch from buffer
679            batch = np.frombuffer(buffer, dtype = np.float32).reshape((numberOfFrames, l_frame_size, self.channels,))
680
681        if self.plannar:
682            batch = batch.transpose(0, 2, 1)
683
684        # increase frame_counter
685        self.frame_counter.frame_count += (actualNbFrames * self.frame_size * self.channels)
686        
687        # say to gc that this buffer is no longer needed
688        del buffer
689
690        if with_timestamps:
691            return FrameContainer( actualNbFrames, batch, self.frame_size/self.sample_rate, current_elapsed_time)
692        
693        return batch

Read next batch of audio from the file

Parameters

number_of_frames: int Number of desired images within the batch. The last batch from the file may have less images.

with_timestamps: bool optional (default False) If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames

Returns

nparray or FrameContainer A batch of shape (n, self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A batch of shape (n, self.channels*self.frame_size) with interleaved data if self.plannar is False. if with_timestamps is True, the return object is a FrameContainer with the audio batch in FrameContainer.data and the associated timestamp in FrameContainer.timestamps as an array (one element for each audio frame).

def writeFrame(self, audio) -> bool:
695    def write_frame(self, audio) -> bool:
696        """
697        Write an audio frame to the file
698
699        Parameters
700        ----------
701        audio: nparray
702            The audio frame to write to the video file of shape (self.channels,nb_samples_per_channel) if plannar is True else (self.channels*nb_samples_per_channel).
703
704        Returns
705        ----------
706        bool
707            Writing was successful or not.
708        """
709        # Check params
710        # - pipe exists
711        if self.pipe is None:
712            raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram))
713        # - pipe is in write mode
714        if self.mode != PipeMode.WRITE_MODE:
715            raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename))
716        # - shape of image is fine, thus we have pixels for a full compatible frame
717        if audio.shape[0] != self.channels:
718            raise self.AudioIOException("Wong audio shape: {} expected ({}, nb_samples_to_write).".format(audio.shape,self.channels))
719        # - type of data is Float32
720        if audio.dtype != np.float32:
721            raise self.AudioIOException("Wong audio type: {} expected np.float32.".format(audio.dtype))
722
723        # array must have a shape (channels, samples), reshape it it to (samples, channels) if plannar
724        if not self.plannar:
725            audio = audio.reshape(-1)
726
727        # print( audio.shape )
728
729        # garantee to have a C continuous array
730        if not audio.flags['C_CONTIGUOUS']:
731            a = np.ascontiguousarray(a) 
732
733        # write frame
734        buffer = audio.tobytes()
735        if self.pipe.stdin.write( buffer ) < len(buffer):
736            print( f"Error writing frame to {self.filename}" )
737            return False
738
739        # increase frame_counter
740        self.frame_counter.frame_count += (audio.shape[1] * self.channels)
741
742        # say to gc that this buffer is no longer needed 
743        del buffer
744
745        return True

Write an audio frame to the file

Parameters

audio: nparray The audio frame to write to the video file of shape (self.channels,nb_samples_per_channel) if plannar is True else (self.channels*nb_samples_per_channel).

Returns

bool Writing was successful or not.

def writeBatch(self, batch):
747    def write_batch(self, batch):
748        """
749        Write a batch of audio frame to the file
750
751        Parameters
752        ----------
753        batch: nparray
754            The batch of audio frames to write to the video file of shape (n,self.channels,nb_samples_per_channel) if plannar is True else (n,self.channels*nb_samples_per_channel) of interleaved audio data.
755
756        Returns
757        ----------
758        bool
759            Writing was successful or not.
760        """
761        # Check params
762        # - pipe exists
763        if self.pipe is None:
764            raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram))
765        # - pipe is in write mode
766        if self.mode != PipeMode.WRITE_MODE:
767            raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename))
768        # batch is 3D (n, channels, nb samples)
769        if batch.ndim !=3:
770            raise self.AudioIOException("Wrong batch shape: {} expected 3 dimensions (n, n_channels, n_samples_per_channel).".format(batch.shape))
771        # - shape of images in batch is fine
772        if batch.shape[2] != self.channels:
773            raise self.AudioIOException("Wrong audio channels in batch: {} expected {} {}.".format(batch.shape[2], self.channels, batch.shape))
774
775        # array must have a shape (n * n_channels * n_samples_per_channel) before writing them to pipe
776        # reshape it it to (n * n_channels * n_samples_per_channel) if plannar is False
777        if not self.plannar:
778            # goes from (n, n_channels, n_samples_per_channel) to (n * n_channels * n_samples_per_channel)
779            batch = batch.transpose(0, 2, 1) # first go to (n, n_samples_per_channel, n_channels)
780            batch = batch.reshape(-1) # then to 1D array (n * n_channels * n_samples_per_channel)
781
782        # garantee to have a C continuous array
783        if not batch.flags['C_CONTIGUOUS']:
784            batch = np.ascontiguousarray(batch)
785
786        # write frame
787        buffer = batch.tobytes()
788        if self.pipe.stdin.write( buffer ) < len(buffer):
789            # say to gc that this buffer is no longer needed
790            del buffer
791            raise self.AudioIOException("Error writing batch to '{}'.".format(self.filename))
792
793        # increase frame_counter
794        self.frame_counter.frame_count += int(batch.shape[0]/self.channels) # int conversion is mandatory to avoid confusion with time as float
795              
796        # say to gc that this buffer is no longer needed
797        del buffer
798
799        return True

Write a batch of audio frame to the file

Parameters

batch: nparray The batch of audio frames to write to the video file of shape (n,self.channels,nb_samples_per_channel) if plannar is True else (n,self.channels*nb_samples_per_channel) of interleaved audio data.

Returns

bool Writing was successful or not.

audioProgram = '/usr/local/lib/python3.12/site-packages/static_ffmpeg/bin/linux/ffmpeg'
paramProgram = '/usr/local/lib/python3.12/site-packages/static_ffmpeg/bin/linux/ffprobe'
class AudioIO.AudioIOException(builtins.Exception):
38    class AudioIOException(Exception):
39        """
40        Dedicated exception class for AudioIO class.
41        """
42        def __init__(self, message="Error while reading/writing video occurs"):
43            self.message = message
44            super().__init__(self.message)

Dedicated exception class for AudioIO class.

AudioIO.AudioIOException(message='Error while reading/writing video occurs')
42        def __init__(self, message="Error while reading/writing video occurs"):
43            self.message = message
44            super().__init__(self.message)
message
class AudioIO.AudioFormat(enum.Enum):
46    class AudioFormat(Enum):
47        """
48        Enum class for supported input video type: 32-bit float is the only supported type for the moment.
49        """
50        PCM32LE = 'pcm_f32le' # default format (unique mode for the moment)

Enum class for supported input video type: 32-bit float is the only supported type for the moment.

PCM32LE = <AudioFormat.PCM32LE: 'pcm_f32le'>