simple_ffmpeg_batch_io.AudioIO
Read/write audio frames or batches of audio frames from (compressed) file, including video file with audio stream(s), using FFmpeg backend.
This module defines the main AudioIO class used to open audio streams,
read audio frames or batches of frames, and write processed outputs.
Authors
Dominique Vaufreydaz (inspired from original C++ code: https://github.com/Vaufreyd/ReadWriteVideosWithOpenCV)
1""" 2Read/write audio frames or batches of audio frames from (compressed) file, including video file with audio stream(s), using FFmpeg backend. 3 4This module defines the main `AudioIO` class used to open audio streams, 5read audio frames or batches of frames, and write processed outputs. 6 7Authors 8------- 9Dominique Vaufreydaz (inspired from original C++ code: https://github.com/Vaufreyd/ReadWriteVideosWithOpenCV) 10 11""" 12 13__authors__ = ("Dominique Vaufreydaz") 14 15import sys 16import subprocess as sp 17import re 18from enum import Enum 19from typing import Union 20 21import math 22 23import numpy as np 24 25from .FrameCounter import FrameCounter 26from .FrameContainer import FrameContainer 27from .PipeMode import PipeMode 28 29# init static_ffmpeg at import time, first time it will download ffmpeg executables 30import static_ffmpeg 31static_ffmpeg.add_paths() 32 33class AudioIO: 34 # "static" variables to ffmpeg, ffprobe executables 35 audioProgram, paramProgram = static_ffmpeg.run.get_or_fetch_platform_executables_else_raise() 36 37 class AudioIOException(Exception): 38 """ 39 Dedicated exception class for AudioIO class. 40 """ 41 def __init__(self, message="Error while reading/writing video occurs"): 42 self.message = message 43 super().__init__(self.message) 44 45 class AudioFormat(Enum): 46 """ 47 Enum class for supported input video type: 32-bit float is the only supported type for the moment. 48 """ 49 PCM32LE = 'pcm_f32le' # default format (unique mode for the moment) 50 51 @classmethod 52 def reader(cls, filename, *, loglevel = 16, debug = False, **kwargs): 53 """ 54 Create and open an AudioIO object in reader mode 55 56 See ``AudioIO.open`` for the full list of accepted parameters. 57 """ 58 reader = cls(logLevel=loglevel, debug=debug) 59 reader.open(filename, **kwargs) 60 return reader 61 62 @classmethod 63 def writer(cls, filename, sample_rate, channels, *, loglevel = 16, debug = False, **kwargs): 64 """ 65 Create and open an AudioIO object in writer mode 66 67 See ``AudioIO.create`` for the full list of accepted parameters. 68 """ 69 writer = cls(logLevel=loglevel, debug=debug) 70 writer.create(filename, sample_rate, channels, **kwargs) 71 return writer 72 73 # standard method 74 def get_corresponding_writer(self, filename, **kwargs): 75 """ 76 Method to get writer for an audio file with same sample_rate, channels as the current one AudioIO object 77 78 See `AudioIO.create` for the full list 79 of accepted parameters. 80 """ 81 return AudioIO.writer(filename, self.sample_rate, self.channels, **kwargs) 82 83 # To use with context manager "with AudioIO.reader(...) as f:' for instance 84 def __enter__(self): 85 """ 86 Method call at initialisation of a context manager like "with AudioIO.reader/writer(...) as f:' for instance 87 """ 88 # simply return myself 89 return self 90 91 def __exit__(self, exc_type, exc_val, exc_tb): 92 """ 93 Method call when existing of a context manager like "with AudioIO.reader/writer(...) as f:' for instance 94 """ 95 # close AudioIO 96 self.close() 97 return False 98 99 @staticmethod 100 def get_time_in_sec(filename, *, debug=False, logLevel=16): 101 """ 102 Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals). 103 104 Parameters 105 ---------- 106 filename : str or path. 107 Raw audio waveform as a 1D array. 108 109 debug : bool (default False). 110 Show debug info. 111 112 logLevel: int (default 16). 113 Log level to pass to the underlying ffmpeg/ffprobe command. 114 115 Returns 116 ---------- 117 float 118 Length in seconds of video file (including milliseconds as decimal part with 3 decimals) 119 """ 120 121 cmd = [AudioIO.paramProgram, # ffprobe 122 '-hide_banner', 123 '-loglevel', str(logLevel), 124 '-show_entries', 'format=duration', 125 '-of', 'default=noprint_wrappers=1:nokey=1', 126 str(filename) 127 ] 128 129 if debug == True: 130 print(' '.join(cmd)) 131 132 # call ffprobe and get params in one single line 133 lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg 134 output = lpipe.stdout.readlines() 135 lpipe.terminate() 136 # transform Bytes output to one single string 137 output = ''.join( [element.decode('utf-8') for element in output]) 138 139 try: 140 return float(output) 141 except (ValueError, TypeError): 142 return None 143 144 @staticmethod 145 def get_params(filename, *, debug=False, logLevel=16): 146 """ 147 Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds. 148 149 Parameters 150 ---------- 151 filename : str or path. 152 Raw audio waveform as a 1D array. 153 154 debug : bool (default (False). 155 Show debug info. 156 157 log_level: int (default 16). 158 Log level to pass to the underlying ffmpeg/ffprobe command. 159 160 Returns 161 ---------- 162 tuple 163 Tuple containing (channels,sample_rate) of the file 164 """ 165 cmd = [AudioIO.paramProgram, # ffprobe 166 '-hide_banner', 167 '-loglevel', str(logLevel), 168 '-show_entries', 'stream=channels,sample_rate', 169 str(filename) 170 ] 171 172 if debug == True: 173 print(' '.join(cmd)) 174 175 # call ffprobe and get params in one single line 176 lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg 177 output = lpipe.stdout.readlines() 178 lpipe.terminate() 179 # transform Bytes output to one single string 180 output = ''.join( [element.decode('utf-8') for element in output]) 181 182 pattern_sample_rate = r'sample_rate=(\d+)' 183 pattern_channels = r'channels=(\d+)' 184 185 # Search for values in the ffprobe output 186 match_sample_rate = re.search(pattern_sample_rate, output, flags=re.MULTILINE) 187 match_channels = re.search(pattern_channels, output, flags=re.MULTILINE) 188 189 # Extraction des valeurs 190 if match_sample_rate: 191 sample_rate = int(match_sample_rate.group(1)) 192 else: 193 raise AudioIO.AudioIOException("Unable to get audio sample_rate of '" + str(filename) + "'") 194 195 if match_channels: 196 channels = int(match_channels.group(1)) 197 else: 198 raise AudioIO.AudioIOException("Unable to get audio channels of '" + str(filename) + "'") 199 200 return (channels,sample_rate) 201 202 # Attributes 203 mode: PipeMode 204 """ Pipemode of the current object (default PipeMode.UNK_MODE)""" 205 206 loglevel: int 207 """ loglevel of the underlying ffmpeg backend for this object (default 16)""" 208 209 debug: bool 210 """ debug flag for this object (print debut info, default False)""" 211 212 channels: int 213 """ Number of channels of images (default -1) """ 214 215 sample_rate: int 216 """ sample_rate of images (default -1) """ 217 218 plannar: bool 219 """ Read/write data as plannar, i.e. not interleaved (default True) """ 220 221 pipe: sp.Popen 222 """ pipe object to ffmpeg/ffprobe (default None)""" 223 224 frame_size: int 225 """ Weight in bytes of one image (default -1)""" 226 227 filename: str 228 """ Filename of the file (default None)""" 229 230 frame_counter: FrameCounter 231 """ `Framecounter` object to count ellapsed time (default None)""" 232 233 def __init__(self, *, logLevel = 16, debug = False): 234 """ 235 Create a AudioIO object giving ffmpeg/ffrobe loglevel and defining debug mode 236 237 Parameters 238 ---------- 239 log_level: int (default 16) 240 Log level to pass to the underlying ffmpeg/ffprobe command. 241 242 debug: bool (default (False) 243 Show debug info. while processing video 244 """ 245 246 self.mode = PipeMode.UNK_MODE 247 self.logLevel = logLevel 248 self.debug = debug 249 250 # Call init() method 251 self.init() 252 253 def init(self): 254 """ 255 Init or reinit a AudioIO object. 256 """ 257 self.channels = -1 258 self.sample_rate = -1 259 self.plannar = True 260 self.pipe = None 261 self.frame_size = -1 262 self.filename = None 263 self.frame_counter = None 264 self.float_size = None 265 266 _repr_exclude = {"pipe"} 267 """ List of excluded attribute for string conversion. """ 268 269 # converting the object to a string representation 270 def __repr__(self): 271 """ 272 Convert object (excluding attributes in _repr_exclude) to string representation. 273 """ 274 attrs = ", ".join( 275 f"{k}={v!r}" 276 for k, v in self.__dict__.items() 277 if k not in self._repr_exclude 278 ) 279 return f"{self.__class__.__name__}({attrs})" 280 281 __str__ = __repr__ 282 """ String representation """ 283 284 def get_elapsed_time_as_str(self) -> str: 285 """ 286 Method to get elapsed time (float value represented) as str. 287 288 Returns 289 ---------- 290 str or None 291 Elapsed time (float value) as str, "15.500" for instance for 15 secondes and 500 milliseconds 292 None if no frame counter are available. 293 """ 294 if self.frame_counter is None: 295 return None 296 return self.frame_counter.get_elapsed_time_as_str() 297 298 def get_formated_elapsed_time_as_str(self,show_ms=True) -> str: 299 """ 300 Method to get elapsed time (hour format) as str. 301 302 Returns 303 ---------- 304 str or None 305 Elapsed time (float value) as str, "00:00:15.500" for instance for 15 secondes and 500 milliseconds 306 None if no frame counter are available. 307 """ 308 if self.frame_counter is None: 309 return None 310 return self.frame_counter.get_formated_elapsed_time_as_str() 311 312 def get_elapsed_time(self) -> float: 313 """ 314 Method to get elapsed time as float value rounded to 3 decimals. 315 316 Returns 317 ---------- 318 float or None 319 Elapsed time (float value) as str, 15.500 for instance for 15 secondes and 500 milliseconds 320 None if no frame counter are available. 321 """ 322 if self.frame_counter is None: 323 return None 324 return self.frame_counter.get_elapsed_time() 325 326 def is_opened(self) -> bool: 327 """ 328 Method to get status of the underlying pipe to ffmpeg. 329 330 Returns 331 ---------- 332 bool 333 True if pipe is opened (reading or writing mode), False if not. 334 """ 335 # is the pip opened? 336 if self.pipe is not None and self.pipe.poll() is None: 337 return True 338 339 return False 340 341 def close(self): 342 """ 343 Method to close current pipe to ffmpeg (if any). Ffmpeg/ffprobe will be terminated. Object can be reused using open or create methods. 344 """ 345 if self.pipe is not None: 346 if self.mode == PipeMode.WRITE_MODE: 347 # killing will make ffmpeg not finish properly the job, close the pipe 348 # to let it know that no more data are comming 349 self.pipe.stdin.close() 350 else: # self.mode == PipeMode.READ_MODE 351 # in read mode, no need to be nice, send SIGTERM on Linux,/Kill it on windows 352 self.pipe.kill() 353 354 # wait for subprocess to end 355 self.pipe.wait() 356 357 # reinit object for later use 358 self.init() 359 360 def create( self, filename, sample_rate, channels, *, writeOverExistingFile = False, 361 outputEncoding = AudioFormat.PCM32LE, encodingParams = None, plannar = True ): 362 """ 363 Method to create a audio file using parametrized access through ffmpeg. Importante note: calling create 364 on a AudioIO will close any former open video. 365 366 Parameters 367 ---------- 368 filename: str or path 369 filename of path to the file (mp4, avi, ...) 370 371 sample_rate: int 372 If defined as a positive value, sample_rates of the output file will be set to this value. 373 374 channels: int 375 If defined as a positive value, number of channels of output file will be set to this value. 376 377 fps: 378 If defined as a positive value, fps of input video will be set to this value. 379 380 outputEncoding: AudioFormat optional (default AudioFormat.PCM32LE) 381 Define audio format for samples. Possible value is AudioFormat.PCM32LE. 382 383 encodingParams: str optional (default None) 384 Parameter to pass to ffmpeg to encode video like audio filters. 385 386 plannar : bool optionnal (default True) 387 Input data to write are grouped by channel if True, interleaved instead. 388 389 Returns 390 ---------- 391 bool 392 Was the creation successfull 393 """ 394 395 # Close if already opened 396 self.close() 397 398 # Set geometry/fps of the video stream from params 399 self.sample_rate = int(sample_rate) 400 self.channels = int(channels) 401 self.plannar = plannar 402 403 # Compute size of float 32, usefull if we want later to use other floating point convention 404 self.float_size = int(np.dtype(np.float32).itemsize) 405 406 # Check params 407 if self.sample_rate <= 0 or self.channels <= 0: 408 raise self.AudioIOException("Bad parameters: sample_rate={}, channels={}".format(self.sample_rate,self.channels)) 409 410 # To write audio, we do not need to know in advance frame size, we will write x values of n bytes 411 self.frame_size = None 412 413 # Video params are set, open the video 414 cmd = [self.audioProgram] # ffmpeg 415 416 if writeOverExistingFile == True: 417 cmd.extend(['-y']) 418 419 cmd.extend(['-hide_banner', 420 '-nostats', 421 '-loglevel', str(self.logLevel), 422 '-f', 'f32le', '-acodec', outputEncoding.value, # input expected coding 423 '-ar', f"{self.sample_rate}", 424 '-ac', f"{self.channels}", 425 '-i', '-']) 426 427 if encodingParams is not None: 428 cmd.extend(encodingParams.split()) 429 430 # Audio filename converted to str (for Path values) 431 cmd.extend( ['-vn', str(filename) ] ) 432 433 if self.debug == True: 434 print( ' '.join(cmd), file=sys.stderr ) 435 436 # store filename and set mode 437 self.filename = str(filename) 438 self.mode = PipeMode.WRITE_MODE 439 440 # call ffmpeg in write mode 441 try: 442 self.pipe = sp.Popen(cmd, stdin=sp.PIPE) 443 self.frame_counter = FrameCounter(self.sample_rate) 444 except Exception as e: 445 # if pipe failed, reinit object and raise exception 446 self.init() 447 raise 448 449 return True 450 451 def open( self, filename, *, sample_rate = -1, channels = -1, inputEncoding = AudioFormat.PCM32LE, 452 decodingParams = None, frame_size = 1.0, plannar = True, start_time = 0.0 ): 453 """ 454 Method to read (video file containing) audio using parametrized access through ffmpeg. Importante note: calling open 455 on a AudioIO will close any former open file. 456 457 Parameters 458 ---------- 459 filename: str or path 460 filename of path to the file (mp4, avi, ...) 461 462 sample_rate: int optional (default -1) 463 If defined as a positive value, sample rate of the input audio will be converted to this value. 464 465 channels: int optional (default -1) 466 If defined as a positive value, number of channels of the input audio will converted to this value. 467 468 inputEncoding: AudioFormat optional (default AudioFormat.PCM32LE) 469 Define audio format for samples. Possible value is AudioFormat.PCM32LE. 470 471 decodingParams: str optional (default None) 472 Parameter to pass to ffmpeg to decode video like audio filters. 473 474 plannar: bool optionnal (default True) 475 Group audio samples per channel if True. Else, samples are interleaved. 476 477 frame_size: int or float (default 1.0) 478 If frame_size is an int, it is the number of expected samples in each frame, for instance 8000 for 8000 samples. 479 if frame_size is a float, it is considered as a time in seconds for each audio frame, for instance 1.0 for 1 second, 0.010 for 10 ms. 480 Number of samples in this case is computed using frame_size and sample_rate as int(frame_size * sample_rate) 481 482 start_time: float optional (default 0.0) 483 Define the reading start time. If not set, reading at beginning of the file. 484 485 Returns 486 ---------- 487 bool 488 Was the opening successfull 489 """ 490 491 # Close if already opened 492 self.close() 493 494 # Force conversion of parameters 495 channels = int(channels) 496 sample_rate = float(sample_rate) 497 498 self.plannar = plannar 499 500 # Compute size of float 32, usefull if we want later to use other floating point convention 501 self.float_size = int(np.dtype(np.float32).itemsize) 502 503 # get parameters from file if needed: 504 if sample_rate <= 0 or channels <= 0: 505 self.channels, self.sample_rate = self.getAudioParams(filename) 506 507 # check if parameters ask to overide video parameters 508 if channels > 0: 509 self.channels = channels 510 if sample_rate > 0: 511 self.sample_rate = sample_rate 512 513 # check parameters 514 515 if isinstance(frame_size,float): 516 # time in seconds 517 self.frame_size = int(frame_size*self.sample_rate) 518 elif isinstance(frame_size,int): 519 # number of samples 520 self.frame_size = frame_size 521 else: 522 # to do 523 pass 524 525 # Video params are set, open the video 526 cmd = [self.audioProgram, # ffmpeg 527 '-hide_banner', 528 '-nostats', 529 '-loglevel', str(self.logLevel)] 530 531 if decodingParams is not None: 532 cmd.extend([decodingParams.split()]) 533 534 if start_time < 0.0: 535 pass 536 elif start_time > 0.0: 537 cmd.extend(["-ss", f"{start_time}"]) 538 539 cmd.extend( ['-i', str(filename), 540 '-f', 'f32le', '-acodec', inputEncoding.value, # input expected coding 541 '-ar', f"{self.sample_rate}", 542 '-ac', f"{self.channels}", 543 '-' # output to stdout 544 ] 545 ) 546 547 if self.debug == True: 548 print( ' '.join(cmd) ) 549 550 # store filename and set mode to READ_MODE 551 self.filename = str(filename) 552 self.mode = PipeMode.READ_MODE 553 554 # call ffmpeg in read mode 555 try: 556 self.pipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg/ffprobe 557 self.frame_counter = FrameCounter(self.sample_rate) 558 if start_time > 0.0: 559 self.frame_counter += start_time # adding with float means adding time 560 except Exception as e: 561 # if pipe failed, reinit object and raise exception 562 self.init() 563 raise 564 565 return True 566 567 def read_frame(self, with_timestamps = False): 568 """ 569 Read next frame from the audio file 570 571 Parameters 572 ---------- 573 with_timestamps: bool optional (default False) 574 If set to True, the method returns a ``FrameContainer`` with the audio and an array containing the associated timestamp(s) 575 576 Returns 577 ---------- 578 nparray or FrameContainer 579 A frame of shape (self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A frame 580 of shape (self.channels*self.frame_size) with interleaved data if self.plannar is False. 581 if with_timestamps is True, the return object is a FrameContainer with the audio data in ``FrameContainer.data`` and 582 the associated timestamp in ``FrameContainer.timestamps`` as an array (one element). 583 """ 584 585 if self.pipe is None: 586 raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading a frame.".format(self.audioProgram)) 587 # - pipe is in write mode 588 if self.mode != PipeMode.READ_MODE: 589 raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename)) 590 591 if with_timestamps: 592 # get elapsed time in video, it is time of next frame(s) 593 current_elapsed_time = self.get_elapsed_time() 594 595 # read rgb image from pipe 596 toread = self.frame_size*self.float_size 597 buffer = self.pipe.stdout.read(toread) 598 599 if buffer == b"": 600 # not considered as an error, no more frame, no exception 601 return None 602 603 # get numpy UINT8 array from buffer 604 audio = np.frombuffer(buffer, dtype = np.float32).reshape(len(buffer)//self.float_size, self.channels) 605 606 # make it plannar (or not) 607 if self.plannar: 608 #transpose it 609 audio = audio.T 610 611 # increase frame_counter 612 self.frame_counter.frame_count += (self.frame_size * self.channels) 613 614 # say to gc that this buffer is no longer needed 615 del buffer 616 617 if with_timestamps: 618 return FrameContainer(1, audio, self.frame_size/self.sample_rate, current_elapsed_time) 619 620 return audio 621 622 def read_batch(self, numberOfFrames, with_timestamps = False): 623 """ 624 Read next batch of audio from the file 625 626 Parameters 627 ---------- 628 number_of_frames: int 629 Number of desired images within the batch. The last batch from the file may have less images. 630 631 with_timestamps: bool optional (default False) 632 If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames 633 634 Returns 635 ---------- 636 nparray or FrameContainer 637 A batch of shape (n, self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A batch 638 of shape (n, self.channels*self.frame_size) with interleaved data if self.plannar is False. 639 if with_timestamps is True, the return object is a FrameContainer with the audio batch in ``FrameContainer.data`` and 640 the associated timestamp in ``FrameContainer.timestamps`` as an array (one element for each audio frame). 641 """ 642 643 if self.pipe is None: 644 raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading frames.".format(self.audioProgram)) 645 # - pipe is in write mode 646 if self.mode != PipeMode.READ_MODE: 647 raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename)) 648 649 if with_timestamps: 650 # get elapsed time in video, it is time of next frame(s) 651 current_elapsed_time = self.get_elapsed_time() 652 653 # try to read complete batch 654 toread = self.frame_size*self.float_size*self.channels*numberOfFrames 655 buffer = self.pipe.stdout.read(toread) 656 657 # check if we are at the end of the buffer 658 if buffer == b"": 659 # not considered as an error, no more frame, no exception 660 return None 661 662 # compute actual number of Frames 663 # do we have a full batch ? Computer standard division 664 actualNbFrames = len(buffer)/(self.frame_size*self.float_size*self.channels) 665 666 if actualNbFrames.is_integer(): 667 actualNbFrames = int(actualNbFrames) 668 # get and reshape batch from buffer 669 batch = np.frombuffer(buffer, dtype = np.float32).reshape((actualNbFrames, self.frame_size, self.channels,)) 670 else: 671 # We do not have a full batch, we are at end of the stream or ffmpeg has been stopped/killed 672 # Compute frame size in samples knowing len of buffer, self.float_size, self.channels and numberOfFrames 673 l_frame_size = len(buffer)//(numberOfFrames*self.float_size*self.channels) 674 if l_frame_size <= 0: 675 # not considered as an error, no more frame, no exception 676 return None 677 # get and reshape batch from buffer 678 batch = np.frombuffer(buffer, dtype = np.float32).reshape((numberOfFrames, l_frame_size, self.channels,)) 679 680 if self.plannar: 681 batch = batch.transpose(0, 2, 1) 682 683 # increase frame_counter 684 self.frame_counter.frame_count += (actualNbFrames * self.frame_size * self.channels) 685 686 # say to gc that this buffer is no longer needed 687 del buffer 688 689 if with_timestamps: 690 return FrameContainer( actualNbFrames, batch, self.frame_size/self.sample_rate, current_elapsed_time) 691 692 return batch 693 694 def write_frame(self, audio) -> bool: 695 """ 696 Write an audio frame to the file 697 698 Parameters 699 ---------- 700 audio: nparray 701 The audio frame to write to the video file of shape (self.channels,nb_samples_per_channel) if plannar is True else (self.channels*nb_samples_per_channel). 702 703 Returns 704 ---------- 705 bool 706 Writing was successful or not. 707 """ 708 # Check params 709 # - pipe exists 710 if self.pipe is None: 711 raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram)) 712 # - pipe is in write mode 713 if self.mode != PipeMode.WRITE_MODE: 714 raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename)) 715 # - shape of image is fine, thus we have pixels for a full compatible frame 716 if audio.shape[0] != self.channels: 717 raise self.AudioIOException("Wong audio shape: {} expected ({}, nb_samples_to_write).".format(audio.shape,self.channels)) 718 # - type of data is Float32 719 if audio.dtype != np.float32: 720 raise self.AudioIOException("Wong audio type: {} expected np.float32.".format(audio.dtype)) 721 722 # array must have a shape (channels, samples), reshape it it to (samples, channels) if plannar 723 if not self.plannar: 724 audio = audio.reshape(-1) 725 726 # print( audio.shape ) 727 728 # garantee to have a C continuous array 729 if not audio.flags['C_CONTIGUOUS']: 730 a = np.ascontiguousarray(a) 731 732 # write frame 733 buffer = audio.tobytes() 734 if self.pipe.stdin.write( buffer ) < len(buffer): 735 print( f"Error writing frame to {self.filename}" ) 736 return False 737 738 # increase frame_counter 739 self.frame_counter.frame_count += (audio.shape[1] * self.channels) 740 741 # say to gc that this buffer is no longer needed 742 del buffer 743 744 return True 745 746 def write_batch(self, batch): 747 """ 748 Write a batch of audio frame to the file 749 750 Parameters 751 ---------- 752 batch: nparray 753 The batch of audio frames to write to the video file of shape (n,self.channels,nb_samples_per_channel) if plannar is True else (n,self.channels*nb_samples_per_channel) of interleaved audio data. 754 755 Returns 756 ---------- 757 bool 758 Writing was successful or not. 759 """ 760 # Check params 761 # - pipe exists 762 if self.pipe is None: 763 raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram)) 764 # - pipe is in write mode 765 if self.mode != PipeMode.WRITE_MODE: 766 raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename)) 767 # batch is 3D (n, channels, nb samples) 768 if batch.ndim !=3: 769 raise self.AudioIOException("Wrong batch shape: {} expected 3 dimensions (n, n_channels, n_samples_per_channel).".format(batch.shape)) 770 # - shape of images in batch is fine 771 if batch.shape[2] != self.channels: 772 raise self.AudioIOException("Wrong audio channels in batch: {} expected {} {}.".format(batch.shape[2], self.channels, batch.shape)) 773 774 # array must have a shape (n * n_channels * n_samples_per_channel) before writing them to pipe 775 # reshape it it to (n * n_channels * n_samples_per_channel) if plannar is False 776 if not self.plannar: 777 # goes from (n, n_channels, n_samples_per_channel) to (n * n_channels * n_samples_per_channel) 778 batch = batch.transpose(0, 2, 1) # first go to (n, n_samples_per_channel, n_channels) 779 batch = batch.reshape(-1) # then to 1D array (n * n_channels * n_samples_per_channel) 780 781 # garantee to have a C continuous array 782 if not batch.flags['C_CONTIGUOUS']: 783 batch = np.ascontiguousarray(batch) 784 785 # write frame 786 buffer = batch.tobytes() 787 if self.pipe.stdin.write( buffer ) < len(buffer): 788 # say to gc that this buffer is no longer needed 789 del buffer 790 raise self.AudioIOException("Error writing batch to '{}'.".format(self.filename)) 791 792 # increase frame_counter 793 self.frame_counter.frame_count += int(batch.shape[0]/self.channels) # int conversion is mandatory to avoid confusion with time as float 794 795 # say to gc that this buffer is no longer needed 796 del buffer 797 798 return True 799 800 def iter_frames(self, with_timestamps = False): 801 """ 802 Method to iterate on audio frames using AudioIO obj. 803 for audio_frame in obj.iter_frames(): 804 .... 805 806 Parameters 807 ---------- 808 with_timestamps: bool optional (default False) 809 If set to True, the method returns a FrameContainer object with the batch and an array containing the associated timestamps to frames 810 811 Returns 812 ---------- 813 nparray or FrameContainer 814 A batch of images of shape () 815 """ 816 817 try: 818 if self.mode == PipeMode.READ_MODE: 819 while self.isOpened(): 820 frame = self.readFrame(with_timestamps) 821 if frame is not None: 822 yield frame 823 finally: 824 self.close() 825 826 def iter_batches(self, batch_size : int, with_timestamps = False ): 827 """ 828 Method to iterate on batch ofaudio frames using AudioIO obj. 829 for audio_batch in obj.iter_batches(): 830 .... 831 832 Parameters 833 ---------- 834 with_timestamps: bool optional (default False) 835 If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames 836 """ 837 try: 838 if self.mode == PipeMode.READ_MODE: 839 while self.isOpened(): 840 batch = self.readBatch(batch_size, with_timestamps) 841 if batch is not None: 842 yield batch 843 finally: 844 self.close() 845 846 # function aliases to be compliant with original C++ version 847 getAudioTimeInSec = get_time_in_sec 848 getAudioParams = get_params 849 get_audio_time_in_sec = get_time_in_sec 850 get_audio_params = get_params 851 isOpened = is_opened 852 readFrame = read_frame 853 readBatch = read_batch 854 writeFrame = write_frame 855 writeBatch = write_batch
34class AudioIO: 35 # "static" variables to ffmpeg, ffprobe executables 36 audioProgram, paramProgram = static_ffmpeg.run.get_or_fetch_platform_executables_else_raise() 37 38 class AudioIOException(Exception): 39 """ 40 Dedicated exception class for AudioIO class. 41 """ 42 def __init__(self, message="Error while reading/writing video occurs"): 43 self.message = message 44 super().__init__(self.message) 45 46 class AudioFormat(Enum): 47 """ 48 Enum class for supported input video type: 32-bit float is the only supported type for the moment. 49 """ 50 PCM32LE = 'pcm_f32le' # default format (unique mode for the moment) 51 52 @classmethod 53 def reader(cls, filename, *, loglevel = 16, debug = False, **kwargs): 54 """ 55 Create and open an AudioIO object in reader mode 56 57 See ``AudioIO.open`` for the full list of accepted parameters. 58 """ 59 reader = cls(logLevel=loglevel, debug=debug) 60 reader.open(filename, **kwargs) 61 return reader 62 63 @classmethod 64 def writer(cls, filename, sample_rate, channels, *, loglevel = 16, debug = False, **kwargs): 65 """ 66 Create and open an AudioIO object in writer mode 67 68 See ``AudioIO.create`` for the full list of accepted parameters. 69 """ 70 writer = cls(logLevel=loglevel, debug=debug) 71 writer.create(filename, sample_rate, channels, **kwargs) 72 return writer 73 74 # standard method 75 def get_corresponding_writer(self, filename, **kwargs): 76 """ 77 Method to get writer for an audio file with same sample_rate, channels as the current one AudioIO object 78 79 See `AudioIO.create` for the full list 80 of accepted parameters. 81 """ 82 return AudioIO.writer(filename, self.sample_rate, self.channels, **kwargs) 83 84 # To use with context manager "with AudioIO.reader(...) as f:' for instance 85 def __enter__(self): 86 """ 87 Method call at initialisation of a context manager like "with AudioIO.reader/writer(...) as f:' for instance 88 """ 89 # simply return myself 90 return self 91 92 def __exit__(self, exc_type, exc_val, exc_tb): 93 """ 94 Method call when existing of a context manager like "with AudioIO.reader/writer(...) as f:' for instance 95 """ 96 # close AudioIO 97 self.close() 98 return False 99 100 @staticmethod 101 def get_time_in_sec(filename, *, debug=False, logLevel=16): 102 """ 103 Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals). 104 105 Parameters 106 ---------- 107 filename : str or path. 108 Raw audio waveform as a 1D array. 109 110 debug : bool (default False). 111 Show debug info. 112 113 logLevel: int (default 16). 114 Log level to pass to the underlying ffmpeg/ffprobe command. 115 116 Returns 117 ---------- 118 float 119 Length in seconds of video file (including milliseconds as decimal part with 3 decimals) 120 """ 121 122 cmd = [AudioIO.paramProgram, # ffprobe 123 '-hide_banner', 124 '-loglevel', str(logLevel), 125 '-show_entries', 'format=duration', 126 '-of', 'default=noprint_wrappers=1:nokey=1', 127 str(filename) 128 ] 129 130 if debug == True: 131 print(' '.join(cmd)) 132 133 # call ffprobe and get params in one single line 134 lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg 135 output = lpipe.stdout.readlines() 136 lpipe.terminate() 137 # transform Bytes output to one single string 138 output = ''.join( [element.decode('utf-8') for element in output]) 139 140 try: 141 return float(output) 142 except (ValueError, TypeError): 143 return None 144 145 @staticmethod 146 def get_params(filename, *, debug=False, logLevel=16): 147 """ 148 Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds. 149 150 Parameters 151 ---------- 152 filename : str or path. 153 Raw audio waveform as a 1D array. 154 155 debug : bool (default (False). 156 Show debug info. 157 158 log_level: int (default 16). 159 Log level to pass to the underlying ffmpeg/ffprobe command. 160 161 Returns 162 ---------- 163 tuple 164 Tuple containing (channels,sample_rate) of the file 165 """ 166 cmd = [AudioIO.paramProgram, # ffprobe 167 '-hide_banner', 168 '-loglevel', str(logLevel), 169 '-show_entries', 'stream=channels,sample_rate', 170 str(filename) 171 ] 172 173 if debug == True: 174 print(' '.join(cmd)) 175 176 # call ffprobe and get params in one single line 177 lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg 178 output = lpipe.stdout.readlines() 179 lpipe.terminate() 180 # transform Bytes output to one single string 181 output = ''.join( [element.decode('utf-8') for element in output]) 182 183 pattern_sample_rate = r'sample_rate=(\d+)' 184 pattern_channels = r'channels=(\d+)' 185 186 # Search for values in the ffprobe output 187 match_sample_rate = re.search(pattern_sample_rate, output, flags=re.MULTILINE) 188 match_channels = re.search(pattern_channels, output, flags=re.MULTILINE) 189 190 # Extraction des valeurs 191 if match_sample_rate: 192 sample_rate = int(match_sample_rate.group(1)) 193 else: 194 raise AudioIO.AudioIOException("Unable to get audio sample_rate of '" + str(filename) + "'") 195 196 if match_channels: 197 channels = int(match_channels.group(1)) 198 else: 199 raise AudioIO.AudioIOException("Unable to get audio channels of '" + str(filename) + "'") 200 201 return (channels,sample_rate) 202 203 # Attributes 204 mode: PipeMode 205 """ Pipemode of the current object (default PipeMode.UNK_MODE)""" 206 207 loglevel: int 208 """ loglevel of the underlying ffmpeg backend for this object (default 16)""" 209 210 debug: bool 211 """ debug flag for this object (print debut info, default False)""" 212 213 channels: int 214 """ Number of channels of images (default -1) """ 215 216 sample_rate: int 217 """ sample_rate of images (default -1) """ 218 219 plannar: bool 220 """ Read/write data as plannar, i.e. not interleaved (default True) """ 221 222 pipe: sp.Popen 223 """ pipe object to ffmpeg/ffprobe (default None)""" 224 225 frame_size: int 226 """ Weight in bytes of one image (default -1)""" 227 228 filename: str 229 """ Filename of the file (default None)""" 230 231 frame_counter: FrameCounter 232 """ `Framecounter` object to count ellapsed time (default None)""" 233 234 def __init__(self, *, logLevel = 16, debug = False): 235 """ 236 Create a AudioIO object giving ffmpeg/ffrobe loglevel and defining debug mode 237 238 Parameters 239 ---------- 240 log_level: int (default 16) 241 Log level to pass to the underlying ffmpeg/ffprobe command. 242 243 debug: bool (default (False) 244 Show debug info. while processing video 245 """ 246 247 self.mode = PipeMode.UNK_MODE 248 self.logLevel = logLevel 249 self.debug = debug 250 251 # Call init() method 252 self.init() 253 254 def init(self): 255 """ 256 Init or reinit a AudioIO object. 257 """ 258 self.channels = -1 259 self.sample_rate = -1 260 self.plannar = True 261 self.pipe = None 262 self.frame_size = -1 263 self.filename = None 264 self.frame_counter = None 265 self.float_size = None 266 267 _repr_exclude = {"pipe"} 268 """ List of excluded attribute for string conversion. """ 269 270 # converting the object to a string representation 271 def __repr__(self): 272 """ 273 Convert object (excluding attributes in _repr_exclude) to string representation. 274 """ 275 attrs = ", ".join( 276 f"{k}={v!r}" 277 for k, v in self.__dict__.items() 278 if k not in self._repr_exclude 279 ) 280 return f"{self.__class__.__name__}({attrs})" 281 282 __str__ = __repr__ 283 """ String representation """ 284 285 def get_elapsed_time_as_str(self) -> str: 286 """ 287 Method to get elapsed time (float value represented) as str. 288 289 Returns 290 ---------- 291 str or None 292 Elapsed time (float value) as str, "15.500" for instance for 15 secondes and 500 milliseconds 293 None if no frame counter are available. 294 """ 295 if self.frame_counter is None: 296 return None 297 return self.frame_counter.get_elapsed_time_as_str() 298 299 def get_formated_elapsed_time_as_str(self,show_ms=True) -> str: 300 """ 301 Method to get elapsed time (hour format) as str. 302 303 Returns 304 ---------- 305 str or None 306 Elapsed time (float value) as str, "00:00:15.500" for instance for 15 secondes and 500 milliseconds 307 None if no frame counter are available. 308 """ 309 if self.frame_counter is None: 310 return None 311 return self.frame_counter.get_formated_elapsed_time_as_str() 312 313 def get_elapsed_time(self) -> float: 314 """ 315 Method to get elapsed time as float value rounded to 3 decimals. 316 317 Returns 318 ---------- 319 float or None 320 Elapsed time (float value) as str, 15.500 for instance for 15 secondes and 500 milliseconds 321 None if no frame counter are available. 322 """ 323 if self.frame_counter is None: 324 return None 325 return self.frame_counter.get_elapsed_time() 326 327 def is_opened(self) -> bool: 328 """ 329 Method to get status of the underlying pipe to ffmpeg. 330 331 Returns 332 ---------- 333 bool 334 True if pipe is opened (reading or writing mode), False if not. 335 """ 336 # is the pip opened? 337 if self.pipe is not None and self.pipe.poll() is None: 338 return True 339 340 return False 341 342 def close(self): 343 """ 344 Method to close current pipe to ffmpeg (if any). Ffmpeg/ffprobe will be terminated. Object can be reused using open or create methods. 345 """ 346 if self.pipe is not None: 347 if self.mode == PipeMode.WRITE_MODE: 348 # killing will make ffmpeg not finish properly the job, close the pipe 349 # to let it know that no more data are comming 350 self.pipe.stdin.close() 351 else: # self.mode == PipeMode.READ_MODE 352 # in read mode, no need to be nice, send SIGTERM on Linux,/Kill it on windows 353 self.pipe.kill() 354 355 # wait for subprocess to end 356 self.pipe.wait() 357 358 # reinit object for later use 359 self.init() 360 361 def create( self, filename, sample_rate, channels, *, writeOverExistingFile = False, 362 outputEncoding = AudioFormat.PCM32LE, encodingParams = None, plannar = True ): 363 """ 364 Method to create a audio file using parametrized access through ffmpeg. Importante note: calling create 365 on a AudioIO will close any former open video. 366 367 Parameters 368 ---------- 369 filename: str or path 370 filename of path to the file (mp4, avi, ...) 371 372 sample_rate: int 373 If defined as a positive value, sample_rates of the output file will be set to this value. 374 375 channels: int 376 If defined as a positive value, number of channels of output file will be set to this value. 377 378 fps: 379 If defined as a positive value, fps of input video will be set to this value. 380 381 outputEncoding: AudioFormat optional (default AudioFormat.PCM32LE) 382 Define audio format for samples. Possible value is AudioFormat.PCM32LE. 383 384 encodingParams: str optional (default None) 385 Parameter to pass to ffmpeg to encode video like audio filters. 386 387 plannar : bool optionnal (default True) 388 Input data to write are grouped by channel if True, interleaved instead. 389 390 Returns 391 ---------- 392 bool 393 Was the creation successfull 394 """ 395 396 # Close if already opened 397 self.close() 398 399 # Set geometry/fps of the video stream from params 400 self.sample_rate = int(sample_rate) 401 self.channels = int(channels) 402 self.plannar = plannar 403 404 # Compute size of float 32, usefull if we want later to use other floating point convention 405 self.float_size = int(np.dtype(np.float32).itemsize) 406 407 # Check params 408 if self.sample_rate <= 0 or self.channels <= 0: 409 raise self.AudioIOException("Bad parameters: sample_rate={}, channels={}".format(self.sample_rate,self.channels)) 410 411 # To write audio, we do not need to know in advance frame size, we will write x values of n bytes 412 self.frame_size = None 413 414 # Video params are set, open the video 415 cmd = [self.audioProgram] # ffmpeg 416 417 if writeOverExistingFile == True: 418 cmd.extend(['-y']) 419 420 cmd.extend(['-hide_banner', 421 '-nostats', 422 '-loglevel', str(self.logLevel), 423 '-f', 'f32le', '-acodec', outputEncoding.value, # input expected coding 424 '-ar', f"{self.sample_rate}", 425 '-ac', f"{self.channels}", 426 '-i', '-']) 427 428 if encodingParams is not None: 429 cmd.extend(encodingParams.split()) 430 431 # Audio filename converted to str (for Path values) 432 cmd.extend( ['-vn', str(filename) ] ) 433 434 if self.debug == True: 435 print( ' '.join(cmd), file=sys.stderr ) 436 437 # store filename and set mode 438 self.filename = str(filename) 439 self.mode = PipeMode.WRITE_MODE 440 441 # call ffmpeg in write mode 442 try: 443 self.pipe = sp.Popen(cmd, stdin=sp.PIPE) 444 self.frame_counter = FrameCounter(self.sample_rate) 445 except Exception as e: 446 # if pipe failed, reinit object and raise exception 447 self.init() 448 raise 449 450 return True 451 452 def open( self, filename, *, sample_rate = -1, channels = -1, inputEncoding = AudioFormat.PCM32LE, 453 decodingParams = None, frame_size = 1.0, plannar = True, start_time = 0.0 ): 454 """ 455 Method to read (video file containing) audio using parametrized access through ffmpeg. Importante note: calling open 456 on a AudioIO will close any former open file. 457 458 Parameters 459 ---------- 460 filename: str or path 461 filename of path to the file (mp4, avi, ...) 462 463 sample_rate: int optional (default -1) 464 If defined as a positive value, sample rate of the input audio will be converted to this value. 465 466 channels: int optional (default -1) 467 If defined as a positive value, number of channels of the input audio will converted to this value. 468 469 inputEncoding: AudioFormat optional (default AudioFormat.PCM32LE) 470 Define audio format for samples. Possible value is AudioFormat.PCM32LE. 471 472 decodingParams: str optional (default None) 473 Parameter to pass to ffmpeg to decode video like audio filters. 474 475 plannar: bool optionnal (default True) 476 Group audio samples per channel if True. Else, samples are interleaved. 477 478 frame_size: int or float (default 1.0) 479 If frame_size is an int, it is the number of expected samples in each frame, for instance 8000 for 8000 samples. 480 if frame_size is a float, it is considered as a time in seconds for each audio frame, for instance 1.0 for 1 second, 0.010 for 10 ms. 481 Number of samples in this case is computed using frame_size and sample_rate as int(frame_size * sample_rate) 482 483 start_time: float optional (default 0.0) 484 Define the reading start time. If not set, reading at beginning of the file. 485 486 Returns 487 ---------- 488 bool 489 Was the opening successfull 490 """ 491 492 # Close if already opened 493 self.close() 494 495 # Force conversion of parameters 496 channels = int(channels) 497 sample_rate = float(sample_rate) 498 499 self.plannar = plannar 500 501 # Compute size of float 32, usefull if we want later to use other floating point convention 502 self.float_size = int(np.dtype(np.float32).itemsize) 503 504 # get parameters from file if needed: 505 if sample_rate <= 0 or channels <= 0: 506 self.channels, self.sample_rate = self.getAudioParams(filename) 507 508 # check if parameters ask to overide video parameters 509 if channels > 0: 510 self.channels = channels 511 if sample_rate > 0: 512 self.sample_rate = sample_rate 513 514 # check parameters 515 516 if isinstance(frame_size,float): 517 # time in seconds 518 self.frame_size = int(frame_size*self.sample_rate) 519 elif isinstance(frame_size,int): 520 # number of samples 521 self.frame_size = frame_size 522 else: 523 # to do 524 pass 525 526 # Video params are set, open the video 527 cmd = [self.audioProgram, # ffmpeg 528 '-hide_banner', 529 '-nostats', 530 '-loglevel', str(self.logLevel)] 531 532 if decodingParams is not None: 533 cmd.extend([decodingParams.split()]) 534 535 if start_time < 0.0: 536 pass 537 elif start_time > 0.0: 538 cmd.extend(["-ss", f"{start_time}"]) 539 540 cmd.extend( ['-i', str(filename), 541 '-f', 'f32le', '-acodec', inputEncoding.value, # input expected coding 542 '-ar', f"{self.sample_rate}", 543 '-ac', f"{self.channels}", 544 '-' # output to stdout 545 ] 546 ) 547 548 if self.debug == True: 549 print( ' '.join(cmd) ) 550 551 # store filename and set mode to READ_MODE 552 self.filename = str(filename) 553 self.mode = PipeMode.READ_MODE 554 555 # call ffmpeg in read mode 556 try: 557 self.pipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg/ffprobe 558 self.frame_counter = FrameCounter(self.sample_rate) 559 if start_time > 0.0: 560 self.frame_counter += start_time # adding with float means adding time 561 except Exception as e: 562 # if pipe failed, reinit object and raise exception 563 self.init() 564 raise 565 566 return True 567 568 def read_frame(self, with_timestamps = False): 569 """ 570 Read next frame from the audio file 571 572 Parameters 573 ---------- 574 with_timestamps: bool optional (default False) 575 If set to True, the method returns a ``FrameContainer`` with the audio and an array containing the associated timestamp(s) 576 577 Returns 578 ---------- 579 nparray or FrameContainer 580 A frame of shape (self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A frame 581 of shape (self.channels*self.frame_size) with interleaved data if self.plannar is False. 582 if with_timestamps is True, the return object is a FrameContainer with the audio data in ``FrameContainer.data`` and 583 the associated timestamp in ``FrameContainer.timestamps`` as an array (one element). 584 """ 585 586 if self.pipe is None: 587 raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading a frame.".format(self.audioProgram)) 588 # - pipe is in write mode 589 if self.mode != PipeMode.READ_MODE: 590 raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename)) 591 592 if with_timestamps: 593 # get elapsed time in video, it is time of next frame(s) 594 current_elapsed_time = self.get_elapsed_time() 595 596 # read rgb image from pipe 597 toread = self.frame_size*self.float_size 598 buffer = self.pipe.stdout.read(toread) 599 600 if buffer == b"": 601 # not considered as an error, no more frame, no exception 602 return None 603 604 # get numpy UINT8 array from buffer 605 audio = np.frombuffer(buffer, dtype = np.float32).reshape(len(buffer)//self.float_size, self.channels) 606 607 # make it plannar (or not) 608 if self.plannar: 609 #transpose it 610 audio = audio.T 611 612 # increase frame_counter 613 self.frame_counter.frame_count += (self.frame_size * self.channels) 614 615 # say to gc that this buffer is no longer needed 616 del buffer 617 618 if with_timestamps: 619 return FrameContainer(1, audio, self.frame_size/self.sample_rate, current_elapsed_time) 620 621 return audio 622 623 def read_batch(self, numberOfFrames, with_timestamps = False): 624 """ 625 Read next batch of audio from the file 626 627 Parameters 628 ---------- 629 number_of_frames: int 630 Number of desired images within the batch. The last batch from the file may have less images. 631 632 with_timestamps: bool optional (default False) 633 If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames 634 635 Returns 636 ---------- 637 nparray or FrameContainer 638 A batch of shape (n, self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A batch 639 of shape (n, self.channels*self.frame_size) with interleaved data if self.plannar is False. 640 if with_timestamps is True, the return object is a FrameContainer with the audio batch in ``FrameContainer.data`` and 641 the associated timestamp in ``FrameContainer.timestamps`` as an array (one element for each audio frame). 642 """ 643 644 if self.pipe is None: 645 raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading frames.".format(self.audioProgram)) 646 # - pipe is in write mode 647 if self.mode != PipeMode.READ_MODE: 648 raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename)) 649 650 if with_timestamps: 651 # get elapsed time in video, it is time of next frame(s) 652 current_elapsed_time = self.get_elapsed_time() 653 654 # try to read complete batch 655 toread = self.frame_size*self.float_size*self.channels*numberOfFrames 656 buffer = self.pipe.stdout.read(toread) 657 658 # check if we are at the end of the buffer 659 if buffer == b"": 660 # not considered as an error, no more frame, no exception 661 return None 662 663 # compute actual number of Frames 664 # do we have a full batch ? Computer standard division 665 actualNbFrames = len(buffer)/(self.frame_size*self.float_size*self.channels) 666 667 if actualNbFrames.is_integer(): 668 actualNbFrames = int(actualNbFrames) 669 # get and reshape batch from buffer 670 batch = np.frombuffer(buffer, dtype = np.float32).reshape((actualNbFrames, self.frame_size, self.channels,)) 671 else: 672 # We do not have a full batch, we are at end of the stream or ffmpeg has been stopped/killed 673 # Compute frame size in samples knowing len of buffer, self.float_size, self.channels and numberOfFrames 674 l_frame_size = len(buffer)//(numberOfFrames*self.float_size*self.channels) 675 if l_frame_size <= 0: 676 # not considered as an error, no more frame, no exception 677 return None 678 # get and reshape batch from buffer 679 batch = np.frombuffer(buffer, dtype = np.float32).reshape((numberOfFrames, l_frame_size, self.channels,)) 680 681 if self.plannar: 682 batch = batch.transpose(0, 2, 1) 683 684 # increase frame_counter 685 self.frame_counter.frame_count += (actualNbFrames * self.frame_size * self.channels) 686 687 # say to gc that this buffer is no longer needed 688 del buffer 689 690 if with_timestamps: 691 return FrameContainer( actualNbFrames, batch, self.frame_size/self.sample_rate, current_elapsed_time) 692 693 return batch 694 695 def write_frame(self, audio) -> bool: 696 """ 697 Write an audio frame to the file 698 699 Parameters 700 ---------- 701 audio: nparray 702 The audio frame to write to the video file of shape (self.channels,nb_samples_per_channel) if plannar is True else (self.channels*nb_samples_per_channel). 703 704 Returns 705 ---------- 706 bool 707 Writing was successful or not. 708 """ 709 # Check params 710 # - pipe exists 711 if self.pipe is None: 712 raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram)) 713 # - pipe is in write mode 714 if self.mode != PipeMode.WRITE_MODE: 715 raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename)) 716 # - shape of image is fine, thus we have pixels for a full compatible frame 717 if audio.shape[0] != self.channels: 718 raise self.AudioIOException("Wong audio shape: {} expected ({}, nb_samples_to_write).".format(audio.shape,self.channels)) 719 # - type of data is Float32 720 if audio.dtype != np.float32: 721 raise self.AudioIOException("Wong audio type: {} expected np.float32.".format(audio.dtype)) 722 723 # array must have a shape (channels, samples), reshape it it to (samples, channels) if plannar 724 if not self.plannar: 725 audio = audio.reshape(-1) 726 727 # print( audio.shape ) 728 729 # garantee to have a C continuous array 730 if not audio.flags['C_CONTIGUOUS']: 731 a = np.ascontiguousarray(a) 732 733 # write frame 734 buffer = audio.tobytes() 735 if self.pipe.stdin.write( buffer ) < len(buffer): 736 print( f"Error writing frame to {self.filename}" ) 737 return False 738 739 # increase frame_counter 740 self.frame_counter.frame_count += (audio.shape[1] * self.channels) 741 742 # say to gc that this buffer is no longer needed 743 del buffer 744 745 return True 746 747 def write_batch(self, batch): 748 """ 749 Write a batch of audio frame to the file 750 751 Parameters 752 ---------- 753 batch: nparray 754 The batch of audio frames to write to the video file of shape (n,self.channels,nb_samples_per_channel) if plannar is True else (n,self.channels*nb_samples_per_channel) of interleaved audio data. 755 756 Returns 757 ---------- 758 bool 759 Writing was successful or not. 760 """ 761 # Check params 762 # - pipe exists 763 if self.pipe is None: 764 raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram)) 765 # - pipe is in write mode 766 if self.mode != PipeMode.WRITE_MODE: 767 raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename)) 768 # batch is 3D (n, channels, nb samples) 769 if batch.ndim !=3: 770 raise self.AudioIOException("Wrong batch shape: {} expected 3 dimensions (n, n_channels, n_samples_per_channel).".format(batch.shape)) 771 # - shape of images in batch is fine 772 if batch.shape[2] != self.channels: 773 raise self.AudioIOException("Wrong audio channels in batch: {} expected {} {}.".format(batch.shape[2], self.channels, batch.shape)) 774 775 # array must have a shape (n * n_channels * n_samples_per_channel) before writing them to pipe 776 # reshape it it to (n * n_channels * n_samples_per_channel) if plannar is False 777 if not self.plannar: 778 # goes from (n, n_channels, n_samples_per_channel) to (n * n_channels * n_samples_per_channel) 779 batch = batch.transpose(0, 2, 1) # first go to (n, n_samples_per_channel, n_channels) 780 batch = batch.reshape(-1) # then to 1D array (n * n_channels * n_samples_per_channel) 781 782 # garantee to have a C continuous array 783 if not batch.flags['C_CONTIGUOUS']: 784 batch = np.ascontiguousarray(batch) 785 786 # write frame 787 buffer = batch.tobytes() 788 if self.pipe.stdin.write( buffer ) < len(buffer): 789 # say to gc that this buffer is no longer needed 790 del buffer 791 raise self.AudioIOException("Error writing batch to '{}'.".format(self.filename)) 792 793 # increase frame_counter 794 self.frame_counter.frame_count += int(batch.shape[0]/self.channels) # int conversion is mandatory to avoid confusion with time as float 795 796 # say to gc that this buffer is no longer needed 797 del buffer 798 799 return True 800 801 def iter_frames(self, with_timestamps = False): 802 """ 803 Method to iterate on audio frames using AudioIO obj. 804 for audio_frame in obj.iter_frames(): 805 .... 806 807 Parameters 808 ---------- 809 with_timestamps: bool optional (default False) 810 If set to True, the method returns a FrameContainer object with the batch and an array containing the associated timestamps to frames 811 812 Returns 813 ---------- 814 nparray or FrameContainer 815 A batch of images of shape () 816 """ 817 818 try: 819 if self.mode == PipeMode.READ_MODE: 820 while self.isOpened(): 821 frame = self.readFrame(with_timestamps) 822 if frame is not None: 823 yield frame 824 finally: 825 self.close() 826 827 def iter_batches(self, batch_size : int, with_timestamps = False ): 828 """ 829 Method to iterate on batch ofaudio frames using AudioIO obj. 830 for audio_batch in obj.iter_batches(): 831 .... 832 833 Parameters 834 ---------- 835 with_timestamps: bool optional (default False) 836 If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames 837 """ 838 try: 839 if self.mode == PipeMode.READ_MODE: 840 while self.isOpened(): 841 batch = self.readBatch(batch_size, with_timestamps) 842 if batch is not None: 843 yield batch 844 finally: 845 self.close() 846 847 # function aliases to be compliant with original C++ version 848 getAudioTimeInSec = get_time_in_sec 849 getAudioParams = get_params 850 get_audio_time_in_sec = get_time_in_sec 851 get_audio_params = get_params 852 isOpened = is_opened 853 readFrame = read_frame 854 readBatch = read_batch 855 writeFrame = write_frame 856 writeBatch = write_batch
234 def __init__(self, *, logLevel = 16, debug = False): 235 """ 236 Create a AudioIO object giving ffmpeg/ffrobe loglevel and defining debug mode 237 238 Parameters 239 ---------- 240 log_level: int (default 16) 241 Log level to pass to the underlying ffmpeg/ffprobe command. 242 243 debug: bool (default (False) 244 Show debug info. while processing video 245 """ 246 247 self.mode = PipeMode.UNK_MODE 248 self.logLevel = logLevel 249 self.debug = debug 250 251 # Call init() method 252 self.init()
Create a AudioIO object giving ffmpeg/ffrobe loglevel and defining debug mode
Parameters
log_level: int (default 16) Log level to pass to the underlying ffmpeg/ffprobe command.
debug: bool (default (False) Show debug info. while processing video
52 @classmethod 53 def reader(cls, filename, *, loglevel = 16, debug = False, **kwargs): 54 """ 55 Create and open an AudioIO object in reader mode 56 57 See ``AudioIO.open`` for the full list of accepted parameters. 58 """ 59 reader = cls(logLevel=loglevel, debug=debug) 60 reader.open(filename, **kwargs) 61 return reader
Create and open an AudioIO object in reader mode
See AudioIO.open for the full list of accepted parameters.
63 @classmethod 64 def writer(cls, filename, sample_rate, channels, *, loglevel = 16, debug = False, **kwargs): 65 """ 66 Create and open an AudioIO object in writer mode 67 68 See ``AudioIO.create`` for the full list of accepted parameters. 69 """ 70 writer = cls(logLevel=loglevel, debug=debug) 71 writer.create(filename, sample_rate, channels, **kwargs) 72 return writer
Create and open an AudioIO object in writer mode
See AudioIO.create for the full list of accepted parameters.
75 def get_corresponding_writer(self, filename, **kwargs): 76 """ 77 Method to get writer for an audio file with same sample_rate, channels as the current one AudioIO object 78 79 See `AudioIO.create` for the full list 80 of accepted parameters. 81 """ 82 return AudioIO.writer(filename, self.sample_rate, self.channels, **kwargs)
Method to get writer for an audio file with same sample_rate, channels as the current one AudioIO object
See AudioIO.create for the full list
of accepted parameters.
100 @staticmethod 101 def get_time_in_sec(filename, *, debug=False, logLevel=16): 102 """ 103 Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals). 104 105 Parameters 106 ---------- 107 filename : str or path. 108 Raw audio waveform as a 1D array. 109 110 debug : bool (default False). 111 Show debug info. 112 113 logLevel: int (default 16). 114 Log level to pass to the underlying ffmpeg/ffprobe command. 115 116 Returns 117 ---------- 118 float 119 Length in seconds of video file (including milliseconds as decimal part with 3 decimals) 120 """ 121 122 cmd = [AudioIO.paramProgram, # ffprobe 123 '-hide_banner', 124 '-loglevel', str(logLevel), 125 '-show_entries', 'format=duration', 126 '-of', 'default=noprint_wrappers=1:nokey=1', 127 str(filename) 128 ] 129 130 if debug == True: 131 print(' '.join(cmd)) 132 133 # call ffprobe and get params in one single line 134 lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg 135 output = lpipe.stdout.readlines() 136 lpipe.terminate() 137 # transform Bytes output to one single string 138 output = ''.join( [element.decode('utf-8') for element in output]) 139 140 try: 141 return float(output) 142 except (ValueError, TypeError): 143 return None
Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals).
Parameters
filename : str or path. Raw audio waveform as a 1D array.
debug : bool (default False). Show debug info.
logLevel: int (default 16). Log level to pass to the underlying ffmpeg/ffprobe command.
Returns
float Length in seconds of video file (including milliseconds as decimal part with 3 decimals)
145 @staticmethod 146 def get_params(filename, *, debug=False, logLevel=16): 147 """ 148 Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds. 149 150 Parameters 151 ---------- 152 filename : str or path. 153 Raw audio waveform as a 1D array. 154 155 debug : bool (default (False). 156 Show debug info. 157 158 log_level: int (default 16). 159 Log level to pass to the underlying ffmpeg/ffprobe command. 160 161 Returns 162 ---------- 163 tuple 164 Tuple containing (channels,sample_rate) of the file 165 """ 166 cmd = [AudioIO.paramProgram, # ffprobe 167 '-hide_banner', 168 '-loglevel', str(logLevel), 169 '-show_entries', 'stream=channels,sample_rate', 170 str(filename) 171 ] 172 173 if debug == True: 174 print(' '.join(cmd)) 175 176 # call ffprobe and get params in one single line 177 lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg 178 output = lpipe.stdout.readlines() 179 lpipe.terminate() 180 # transform Bytes output to one single string 181 output = ''.join( [element.decode('utf-8') for element in output]) 182 183 pattern_sample_rate = r'sample_rate=(\d+)' 184 pattern_channels = r'channels=(\d+)' 185 186 # Search for values in the ffprobe output 187 match_sample_rate = re.search(pattern_sample_rate, output, flags=re.MULTILINE) 188 match_channels = re.search(pattern_channels, output, flags=re.MULTILINE) 189 190 # Extraction des valeurs 191 if match_sample_rate: 192 sample_rate = int(match_sample_rate.group(1)) 193 else: 194 raise AudioIO.AudioIOException("Unable to get audio sample_rate of '" + str(filename) + "'") 195 196 if match_channels: 197 channels = int(match_channels.group(1)) 198 else: 199 raise AudioIO.AudioIOException("Unable to get audio channels of '" + str(filename) + "'") 200 201 return (channels,sample_rate) 202 203 # Attributes 204 mode: PipeMode 205 """ Pipemode of the current object (default PipeMode.UNK_MODE)""" 206 207 loglevel: int 208 """ loglevel of the underlying ffmpeg backend for this object (default 16)""" 209 210 debug: bool 211 """ debug flag for this object (print debut info, default False)""" 212 213 channels: int 214 """ Number of channels of images (default -1) """ 215 216 sample_rate: int 217 """ sample_rate of images (default -1) """ 218 219 plannar: bool 220 """ Read/write data as plannar, i.e. not interleaved (default True) """ 221 222 pipe: sp.Popen 223 """ pipe object to ffmpeg/ffprobe (default None)""" 224 225 frame_size: int 226 """ Weight in bytes of one image (default -1)""" 227 228 filename: str 229 """ Filename of the file (default None)""" 230 231 frame_counter: FrameCounter 232 """ `Framecounter` object to count ellapsed time (default None)"""
Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds.
Parameters
filename : str or path. Raw audio waveform as a 1D array.
debug : bool (default (False). Show debug info.
log_level: int (default 16). Log level to pass to the underlying ffmpeg/ffprobe command.
Returns
tuple Tuple containing (channels,sample_rate) of the file
254 def init(self): 255 """ 256 Init or reinit a AudioIO object. 257 """ 258 self.channels = -1 259 self.sample_rate = -1 260 self.plannar = True 261 self.pipe = None 262 self.frame_size = -1 263 self.filename = None 264 self.frame_counter = None 265 self.float_size = None
Init or reinit a AudioIO object.
285 def get_elapsed_time_as_str(self) -> str: 286 """ 287 Method to get elapsed time (float value represented) as str. 288 289 Returns 290 ---------- 291 str or None 292 Elapsed time (float value) as str, "15.500" for instance for 15 secondes and 500 milliseconds 293 None if no frame counter are available. 294 """ 295 if self.frame_counter is None: 296 return None 297 return self.frame_counter.get_elapsed_time_as_str()
Method to get elapsed time (float value represented) as str.
Returns
str or None Elapsed time (float value) as str, "15.500" for instance for 15 secondes and 500 milliseconds None if no frame counter are available.
299 def get_formated_elapsed_time_as_str(self,show_ms=True) -> str: 300 """ 301 Method to get elapsed time (hour format) as str. 302 303 Returns 304 ---------- 305 str or None 306 Elapsed time (float value) as str, "00:00:15.500" for instance for 15 secondes and 500 milliseconds 307 None if no frame counter are available. 308 """ 309 if self.frame_counter is None: 310 return None 311 return self.frame_counter.get_formated_elapsed_time_as_str()
Method to get elapsed time (hour format) as str.
Returns
str or None Elapsed time (float value) as str, "00:00:15.500" for instance for 15 secondes and 500 milliseconds None if no frame counter are available.
313 def get_elapsed_time(self) -> float: 314 """ 315 Method to get elapsed time as float value rounded to 3 decimals. 316 317 Returns 318 ---------- 319 float or None 320 Elapsed time (float value) as str, 15.500 for instance for 15 secondes and 500 milliseconds 321 None if no frame counter are available. 322 """ 323 if self.frame_counter is None: 324 return None 325 return self.frame_counter.get_elapsed_time()
Method to get elapsed time as float value rounded to 3 decimals.
Returns
float or None Elapsed time (float value) as str, 15.500 for instance for 15 secondes and 500 milliseconds None if no frame counter are available.
327 def is_opened(self) -> bool: 328 """ 329 Method to get status of the underlying pipe to ffmpeg. 330 331 Returns 332 ---------- 333 bool 334 True if pipe is opened (reading or writing mode), False if not. 335 """ 336 # is the pip opened? 337 if self.pipe is not None and self.pipe.poll() is None: 338 return True 339 340 return False
Method to get status of the underlying pipe to ffmpeg.
Returns
bool True if pipe is opened (reading or writing mode), False if not.
342 def close(self): 343 """ 344 Method to close current pipe to ffmpeg (if any). Ffmpeg/ffprobe will be terminated. Object can be reused using open or create methods. 345 """ 346 if self.pipe is not None: 347 if self.mode == PipeMode.WRITE_MODE: 348 # killing will make ffmpeg not finish properly the job, close the pipe 349 # to let it know that no more data are comming 350 self.pipe.stdin.close() 351 else: # self.mode == PipeMode.READ_MODE 352 # in read mode, no need to be nice, send SIGTERM on Linux,/Kill it on windows 353 self.pipe.kill() 354 355 # wait for subprocess to end 356 self.pipe.wait() 357 358 # reinit object for later use 359 self.init()
Method to close current pipe to ffmpeg (if any). Ffmpeg/ffprobe will be terminated. Object can be reused using open or create methods.
361 def create( self, filename, sample_rate, channels, *, writeOverExistingFile = False, 362 outputEncoding = AudioFormat.PCM32LE, encodingParams = None, plannar = True ): 363 """ 364 Method to create a audio file using parametrized access through ffmpeg. Importante note: calling create 365 on a AudioIO will close any former open video. 366 367 Parameters 368 ---------- 369 filename: str or path 370 filename of path to the file (mp4, avi, ...) 371 372 sample_rate: int 373 If defined as a positive value, sample_rates of the output file will be set to this value. 374 375 channels: int 376 If defined as a positive value, number of channels of output file will be set to this value. 377 378 fps: 379 If defined as a positive value, fps of input video will be set to this value. 380 381 outputEncoding: AudioFormat optional (default AudioFormat.PCM32LE) 382 Define audio format for samples. Possible value is AudioFormat.PCM32LE. 383 384 encodingParams: str optional (default None) 385 Parameter to pass to ffmpeg to encode video like audio filters. 386 387 plannar : bool optionnal (default True) 388 Input data to write are grouped by channel if True, interleaved instead. 389 390 Returns 391 ---------- 392 bool 393 Was the creation successfull 394 """ 395 396 # Close if already opened 397 self.close() 398 399 # Set geometry/fps of the video stream from params 400 self.sample_rate = int(sample_rate) 401 self.channels = int(channels) 402 self.plannar = plannar 403 404 # Compute size of float 32, usefull if we want later to use other floating point convention 405 self.float_size = int(np.dtype(np.float32).itemsize) 406 407 # Check params 408 if self.sample_rate <= 0 or self.channels <= 0: 409 raise self.AudioIOException("Bad parameters: sample_rate={}, channels={}".format(self.sample_rate,self.channels)) 410 411 # To write audio, we do not need to know in advance frame size, we will write x values of n bytes 412 self.frame_size = None 413 414 # Video params are set, open the video 415 cmd = [self.audioProgram] # ffmpeg 416 417 if writeOverExistingFile == True: 418 cmd.extend(['-y']) 419 420 cmd.extend(['-hide_banner', 421 '-nostats', 422 '-loglevel', str(self.logLevel), 423 '-f', 'f32le', '-acodec', outputEncoding.value, # input expected coding 424 '-ar', f"{self.sample_rate}", 425 '-ac', f"{self.channels}", 426 '-i', '-']) 427 428 if encodingParams is not None: 429 cmd.extend(encodingParams.split()) 430 431 # Audio filename converted to str (for Path values) 432 cmd.extend( ['-vn', str(filename) ] ) 433 434 if self.debug == True: 435 print( ' '.join(cmd), file=sys.stderr ) 436 437 # store filename and set mode 438 self.filename = str(filename) 439 self.mode = PipeMode.WRITE_MODE 440 441 # call ffmpeg in write mode 442 try: 443 self.pipe = sp.Popen(cmd, stdin=sp.PIPE) 444 self.frame_counter = FrameCounter(self.sample_rate) 445 except Exception as e: 446 # if pipe failed, reinit object and raise exception 447 self.init() 448 raise 449 450 return True
Method to create a audio file using parametrized access through ffmpeg. Importante note: calling create on a AudioIO will close any former open video.
Parameters
filename: str or path filename of path to the file (mp4, avi, ...)
sample_rate: int If defined as a positive value, sample_rates of the output file will be set to this value.
channels: int If defined as a positive value, number of channels of output file will be set to this value.
fps: If defined as a positive value, fps of input video will be set to this value.
outputEncoding: AudioFormat optional (default AudioFormat.PCM32LE) Define audio format for samples. Possible value is AudioFormat.PCM32LE.
encodingParams: str optional (default None) Parameter to pass to ffmpeg to encode video like audio filters.
plannar : bool optionnal (default True) Input data to write are grouped by channel if True, interleaved instead.
Returns
bool Was the creation successfull
452 def open( self, filename, *, sample_rate = -1, channels = -1, inputEncoding = AudioFormat.PCM32LE, 453 decodingParams = None, frame_size = 1.0, plannar = True, start_time = 0.0 ): 454 """ 455 Method to read (video file containing) audio using parametrized access through ffmpeg. Importante note: calling open 456 on a AudioIO will close any former open file. 457 458 Parameters 459 ---------- 460 filename: str or path 461 filename of path to the file (mp4, avi, ...) 462 463 sample_rate: int optional (default -1) 464 If defined as a positive value, sample rate of the input audio will be converted to this value. 465 466 channels: int optional (default -1) 467 If defined as a positive value, number of channels of the input audio will converted to this value. 468 469 inputEncoding: AudioFormat optional (default AudioFormat.PCM32LE) 470 Define audio format for samples. Possible value is AudioFormat.PCM32LE. 471 472 decodingParams: str optional (default None) 473 Parameter to pass to ffmpeg to decode video like audio filters. 474 475 plannar: bool optionnal (default True) 476 Group audio samples per channel if True. Else, samples are interleaved. 477 478 frame_size: int or float (default 1.0) 479 If frame_size is an int, it is the number of expected samples in each frame, for instance 8000 for 8000 samples. 480 if frame_size is a float, it is considered as a time in seconds for each audio frame, for instance 1.0 for 1 second, 0.010 for 10 ms. 481 Number of samples in this case is computed using frame_size and sample_rate as int(frame_size * sample_rate) 482 483 start_time: float optional (default 0.0) 484 Define the reading start time. If not set, reading at beginning of the file. 485 486 Returns 487 ---------- 488 bool 489 Was the opening successfull 490 """ 491 492 # Close if already opened 493 self.close() 494 495 # Force conversion of parameters 496 channels = int(channels) 497 sample_rate = float(sample_rate) 498 499 self.plannar = plannar 500 501 # Compute size of float 32, usefull if we want later to use other floating point convention 502 self.float_size = int(np.dtype(np.float32).itemsize) 503 504 # get parameters from file if needed: 505 if sample_rate <= 0 or channels <= 0: 506 self.channels, self.sample_rate = self.getAudioParams(filename) 507 508 # check if parameters ask to overide video parameters 509 if channels > 0: 510 self.channels = channels 511 if sample_rate > 0: 512 self.sample_rate = sample_rate 513 514 # check parameters 515 516 if isinstance(frame_size,float): 517 # time in seconds 518 self.frame_size = int(frame_size*self.sample_rate) 519 elif isinstance(frame_size,int): 520 # number of samples 521 self.frame_size = frame_size 522 else: 523 # to do 524 pass 525 526 # Video params are set, open the video 527 cmd = [self.audioProgram, # ffmpeg 528 '-hide_banner', 529 '-nostats', 530 '-loglevel', str(self.logLevel)] 531 532 if decodingParams is not None: 533 cmd.extend([decodingParams.split()]) 534 535 if start_time < 0.0: 536 pass 537 elif start_time > 0.0: 538 cmd.extend(["-ss", f"{start_time}"]) 539 540 cmd.extend( ['-i', str(filename), 541 '-f', 'f32le', '-acodec', inputEncoding.value, # input expected coding 542 '-ar', f"{self.sample_rate}", 543 '-ac', f"{self.channels}", 544 '-' # output to stdout 545 ] 546 ) 547 548 if self.debug == True: 549 print( ' '.join(cmd) ) 550 551 # store filename and set mode to READ_MODE 552 self.filename = str(filename) 553 self.mode = PipeMode.READ_MODE 554 555 # call ffmpeg in read mode 556 try: 557 self.pipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg/ffprobe 558 self.frame_counter = FrameCounter(self.sample_rate) 559 if start_time > 0.0: 560 self.frame_counter += start_time # adding with float means adding time 561 except Exception as e: 562 # if pipe failed, reinit object and raise exception 563 self.init() 564 raise 565 566 return True
Method to read (video file containing) audio using parametrized access through ffmpeg. Importante note: calling open on a AudioIO will close any former open file.
Parameters
filename: str or path filename of path to the file (mp4, avi, ...)
sample_rate: int optional (default -1) If defined as a positive value, sample rate of the input audio will be converted to this value.
channels: int optional (default -1) If defined as a positive value, number of channels of the input audio will converted to this value.
inputEncoding: AudioFormat optional (default AudioFormat.PCM32LE) Define audio format for samples. Possible value is AudioFormat.PCM32LE.
decodingParams: str optional (default None) Parameter to pass to ffmpeg to decode video like audio filters.
plannar: bool optionnal (default True) Group audio samples per channel if True. Else, samples are interleaved.
frame_size: int or float (default 1.0) If frame_size is an int, it is the number of expected samples in each frame, for instance 8000 for 8000 samples. if frame_size is a float, it is considered as a time in seconds for each audio frame, for instance 1.0 for 1 second, 0.010 for 10 ms. Number of samples in this case is computed using frame_size and sample_rate as int(frame_size * sample_rate)
start_time: float optional (default 0.0) Define the reading start time. If not set, reading at beginning of the file.
Returns
bool Was the opening successfull
568 def read_frame(self, with_timestamps = False): 569 """ 570 Read next frame from the audio file 571 572 Parameters 573 ---------- 574 with_timestamps: bool optional (default False) 575 If set to True, the method returns a ``FrameContainer`` with the audio and an array containing the associated timestamp(s) 576 577 Returns 578 ---------- 579 nparray or FrameContainer 580 A frame of shape (self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A frame 581 of shape (self.channels*self.frame_size) with interleaved data if self.plannar is False. 582 if with_timestamps is True, the return object is a FrameContainer with the audio data in ``FrameContainer.data`` and 583 the associated timestamp in ``FrameContainer.timestamps`` as an array (one element). 584 """ 585 586 if self.pipe is None: 587 raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading a frame.".format(self.audioProgram)) 588 # - pipe is in write mode 589 if self.mode != PipeMode.READ_MODE: 590 raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename)) 591 592 if with_timestamps: 593 # get elapsed time in video, it is time of next frame(s) 594 current_elapsed_time = self.get_elapsed_time() 595 596 # read rgb image from pipe 597 toread = self.frame_size*self.float_size 598 buffer = self.pipe.stdout.read(toread) 599 600 if buffer == b"": 601 # not considered as an error, no more frame, no exception 602 return None 603 604 # get numpy UINT8 array from buffer 605 audio = np.frombuffer(buffer, dtype = np.float32).reshape(len(buffer)//self.float_size, self.channels) 606 607 # make it plannar (or not) 608 if self.plannar: 609 #transpose it 610 audio = audio.T 611 612 # increase frame_counter 613 self.frame_counter.frame_count += (self.frame_size * self.channels) 614 615 # say to gc that this buffer is no longer needed 616 del buffer 617 618 if with_timestamps: 619 return FrameContainer(1, audio, self.frame_size/self.sample_rate, current_elapsed_time) 620 621 return audio
Read next frame from the audio file
Parameters
with_timestamps: bool optional (default False)
If set to True, the method returns a FrameContainer with the audio and an array containing the associated timestamp(s)
Returns
nparray or FrameContainer
A frame of shape (self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A frame
of shape (self.channels*self.frame_size) with interleaved data if self.plannar is False.
if with_timestamps is True, the return object is a FrameContainer with the audio data in FrameContainer.data and
the associated timestamp in FrameContainer.timestamps as an array (one element).
623 def read_batch(self, numberOfFrames, with_timestamps = False): 624 """ 625 Read next batch of audio from the file 626 627 Parameters 628 ---------- 629 number_of_frames: int 630 Number of desired images within the batch. The last batch from the file may have less images. 631 632 with_timestamps: bool optional (default False) 633 If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames 634 635 Returns 636 ---------- 637 nparray or FrameContainer 638 A batch of shape (n, self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A batch 639 of shape (n, self.channels*self.frame_size) with interleaved data if self.plannar is False. 640 if with_timestamps is True, the return object is a FrameContainer with the audio batch in ``FrameContainer.data`` and 641 the associated timestamp in ``FrameContainer.timestamps`` as an array (one element for each audio frame). 642 """ 643 644 if self.pipe is None: 645 raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading frames.".format(self.audioProgram)) 646 # - pipe is in write mode 647 if self.mode != PipeMode.READ_MODE: 648 raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename)) 649 650 if with_timestamps: 651 # get elapsed time in video, it is time of next frame(s) 652 current_elapsed_time = self.get_elapsed_time() 653 654 # try to read complete batch 655 toread = self.frame_size*self.float_size*self.channels*numberOfFrames 656 buffer = self.pipe.stdout.read(toread) 657 658 # check if we are at the end of the buffer 659 if buffer == b"": 660 # not considered as an error, no more frame, no exception 661 return None 662 663 # compute actual number of Frames 664 # do we have a full batch ? Computer standard division 665 actualNbFrames = len(buffer)/(self.frame_size*self.float_size*self.channels) 666 667 if actualNbFrames.is_integer(): 668 actualNbFrames = int(actualNbFrames) 669 # get and reshape batch from buffer 670 batch = np.frombuffer(buffer, dtype = np.float32).reshape((actualNbFrames, self.frame_size, self.channels,)) 671 else: 672 # We do not have a full batch, we are at end of the stream or ffmpeg has been stopped/killed 673 # Compute frame size in samples knowing len of buffer, self.float_size, self.channels and numberOfFrames 674 l_frame_size = len(buffer)//(numberOfFrames*self.float_size*self.channels) 675 if l_frame_size <= 0: 676 # not considered as an error, no more frame, no exception 677 return None 678 # get and reshape batch from buffer 679 batch = np.frombuffer(buffer, dtype = np.float32).reshape((numberOfFrames, l_frame_size, self.channels,)) 680 681 if self.plannar: 682 batch = batch.transpose(0, 2, 1) 683 684 # increase frame_counter 685 self.frame_counter.frame_count += (actualNbFrames * self.frame_size * self.channels) 686 687 # say to gc that this buffer is no longer needed 688 del buffer 689 690 if with_timestamps: 691 return FrameContainer( actualNbFrames, batch, self.frame_size/self.sample_rate, current_elapsed_time) 692 693 return batch
Read next batch of audio from the file
Parameters
number_of_frames: int Number of desired images within the batch. The last batch from the file may have less images.
with_timestamps: bool optional (default False) If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames
Returns
nparray or FrameContainer
A batch of shape (n, self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A batch
of shape (n, self.channels*self.frame_size) with interleaved data if self.plannar is False.
if with_timestamps is True, the return object is a FrameContainer with the audio batch in FrameContainer.data and
the associated timestamp in FrameContainer.timestamps as an array (one element for each audio frame).
695 def write_frame(self, audio) -> bool: 696 """ 697 Write an audio frame to the file 698 699 Parameters 700 ---------- 701 audio: nparray 702 The audio frame to write to the video file of shape (self.channels,nb_samples_per_channel) if plannar is True else (self.channels*nb_samples_per_channel). 703 704 Returns 705 ---------- 706 bool 707 Writing was successful or not. 708 """ 709 # Check params 710 # - pipe exists 711 if self.pipe is None: 712 raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram)) 713 # - pipe is in write mode 714 if self.mode != PipeMode.WRITE_MODE: 715 raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename)) 716 # - shape of image is fine, thus we have pixels for a full compatible frame 717 if audio.shape[0] != self.channels: 718 raise self.AudioIOException("Wong audio shape: {} expected ({}, nb_samples_to_write).".format(audio.shape,self.channels)) 719 # - type of data is Float32 720 if audio.dtype != np.float32: 721 raise self.AudioIOException("Wong audio type: {} expected np.float32.".format(audio.dtype)) 722 723 # array must have a shape (channels, samples), reshape it it to (samples, channels) if plannar 724 if not self.plannar: 725 audio = audio.reshape(-1) 726 727 # print( audio.shape ) 728 729 # garantee to have a C continuous array 730 if not audio.flags['C_CONTIGUOUS']: 731 a = np.ascontiguousarray(a) 732 733 # write frame 734 buffer = audio.tobytes() 735 if self.pipe.stdin.write( buffer ) < len(buffer): 736 print( f"Error writing frame to {self.filename}" ) 737 return False 738 739 # increase frame_counter 740 self.frame_counter.frame_count += (audio.shape[1] * self.channels) 741 742 # say to gc that this buffer is no longer needed 743 del buffer 744 745 return True
Write an audio frame to the file
Parameters
audio: nparray The audio frame to write to the video file of shape (self.channels,nb_samples_per_channel) if plannar is True else (self.channels*nb_samples_per_channel).
Returns
bool Writing was successful or not.
747 def write_batch(self, batch): 748 """ 749 Write a batch of audio frame to the file 750 751 Parameters 752 ---------- 753 batch: nparray 754 The batch of audio frames to write to the video file of shape (n,self.channels,nb_samples_per_channel) if plannar is True else (n,self.channels*nb_samples_per_channel) of interleaved audio data. 755 756 Returns 757 ---------- 758 bool 759 Writing was successful or not. 760 """ 761 # Check params 762 # - pipe exists 763 if self.pipe is None: 764 raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram)) 765 # - pipe is in write mode 766 if self.mode != PipeMode.WRITE_MODE: 767 raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename)) 768 # batch is 3D (n, channels, nb samples) 769 if batch.ndim !=3: 770 raise self.AudioIOException("Wrong batch shape: {} expected 3 dimensions (n, n_channels, n_samples_per_channel).".format(batch.shape)) 771 # - shape of images in batch is fine 772 if batch.shape[2] != self.channels: 773 raise self.AudioIOException("Wrong audio channels in batch: {} expected {} {}.".format(batch.shape[2], self.channels, batch.shape)) 774 775 # array must have a shape (n * n_channels * n_samples_per_channel) before writing them to pipe 776 # reshape it it to (n * n_channels * n_samples_per_channel) if plannar is False 777 if not self.plannar: 778 # goes from (n, n_channels, n_samples_per_channel) to (n * n_channels * n_samples_per_channel) 779 batch = batch.transpose(0, 2, 1) # first go to (n, n_samples_per_channel, n_channels) 780 batch = batch.reshape(-1) # then to 1D array (n * n_channels * n_samples_per_channel) 781 782 # garantee to have a C continuous array 783 if not batch.flags['C_CONTIGUOUS']: 784 batch = np.ascontiguousarray(batch) 785 786 # write frame 787 buffer = batch.tobytes() 788 if self.pipe.stdin.write( buffer ) < len(buffer): 789 # say to gc that this buffer is no longer needed 790 del buffer 791 raise self.AudioIOException("Error writing batch to '{}'.".format(self.filename)) 792 793 # increase frame_counter 794 self.frame_counter.frame_count += int(batch.shape[0]/self.channels) # int conversion is mandatory to avoid confusion with time as float 795 796 # say to gc that this buffer is no longer needed 797 del buffer 798 799 return True
Write a batch of audio frame to the file
Parameters
batch: nparray The batch of audio frames to write to the video file of shape (n,self.channels,nb_samples_per_channel) if plannar is True else (n,self.channels*nb_samples_per_channel) of interleaved audio data.
Returns
bool Writing was successful or not.
801 def iter_frames(self, with_timestamps = False): 802 """ 803 Method to iterate on audio frames using AudioIO obj. 804 for audio_frame in obj.iter_frames(): 805 .... 806 807 Parameters 808 ---------- 809 with_timestamps: bool optional (default False) 810 If set to True, the method returns a FrameContainer object with the batch and an array containing the associated timestamps to frames 811 812 Returns 813 ---------- 814 nparray or FrameContainer 815 A batch of images of shape () 816 """ 817 818 try: 819 if self.mode == PipeMode.READ_MODE: 820 while self.isOpened(): 821 frame = self.readFrame(with_timestamps) 822 if frame is not None: 823 yield frame 824 finally: 825 self.close()
Method to iterate on audio frames using AudioIO obj. for audio_frame in obj.iter_frames(): ....
Parameters
with_timestamps: bool optional (default False) If set to True, the method returns a FrameContainer object with the batch and an array containing the associated timestamps to frames
Returns
nparray or FrameContainer A batch of images of shape ()
827 def iter_batches(self, batch_size : int, with_timestamps = False ): 828 """ 829 Method to iterate on batch ofaudio frames using AudioIO obj. 830 for audio_batch in obj.iter_batches(): 831 .... 832 833 Parameters 834 ---------- 835 with_timestamps: bool optional (default False) 836 If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames 837 """ 838 try: 839 if self.mode == PipeMode.READ_MODE: 840 while self.isOpened(): 841 batch = self.readBatch(batch_size, with_timestamps) 842 if batch is not None: 843 yield batch 844 finally: 845 self.close()
Method to iterate on batch ofaudio frames using AudioIO obj. for audio_batch in obj.iter_batches(): ....
Parameters
with_timestamps: bool optional (default False) If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames
100 @staticmethod 101 def get_time_in_sec(filename, *, debug=False, logLevel=16): 102 """ 103 Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals). 104 105 Parameters 106 ---------- 107 filename : str or path. 108 Raw audio waveform as a 1D array. 109 110 debug : bool (default False). 111 Show debug info. 112 113 logLevel: int (default 16). 114 Log level to pass to the underlying ffmpeg/ffprobe command. 115 116 Returns 117 ---------- 118 float 119 Length in seconds of video file (including milliseconds as decimal part with 3 decimals) 120 """ 121 122 cmd = [AudioIO.paramProgram, # ffprobe 123 '-hide_banner', 124 '-loglevel', str(logLevel), 125 '-show_entries', 'format=duration', 126 '-of', 'default=noprint_wrappers=1:nokey=1', 127 str(filename) 128 ] 129 130 if debug == True: 131 print(' '.join(cmd)) 132 133 # call ffprobe and get params in one single line 134 lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg 135 output = lpipe.stdout.readlines() 136 lpipe.terminate() 137 # transform Bytes output to one single string 138 output = ''.join( [element.decode('utf-8') for element in output]) 139 140 try: 141 return float(output) 142 except (ValueError, TypeError): 143 return None
Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals).
Parameters
filename : str or path. Raw audio waveform as a 1D array.
debug : bool (default False). Show debug info.
logLevel: int (default 16). Log level to pass to the underlying ffmpeg/ffprobe command.
Returns
float Length in seconds of video file (including milliseconds as decimal part with 3 decimals)
145 @staticmethod 146 def get_params(filename, *, debug=False, logLevel=16): 147 """ 148 Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds. 149 150 Parameters 151 ---------- 152 filename : str or path. 153 Raw audio waveform as a 1D array. 154 155 debug : bool (default (False). 156 Show debug info. 157 158 log_level: int (default 16). 159 Log level to pass to the underlying ffmpeg/ffprobe command. 160 161 Returns 162 ---------- 163 tuple 164 Tuple containing (channels,sample_rate) of the file 165 """ 166 cmd = [AudioIO.paramProgram, # ffprobe 167 '-hide_banner', 168 '-loglevel', str(logLevel), 169 '-show_entries', 'stream=channels,sample_rate', 170 str(filename) 171 ] 172 173 if debug == True: 174 print(' '.join(cmd)) 175 176 # call ffprobe and get params in one single line 177 lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg 178 output = lpipe.stdout.readlines() 179 lpipe.terminate() 180 # transform Bytes output to one single string 181 output = ''.join( [element.decode('utf-8') for element in output]) 182 183 pattern_sample_rate = r'sample_rate=(\d+)' 184 pattern_channels = r'channels=(\d+)' 185 186 # Search for values in the ffprobe output 187 match_sample_rate = re.search(pattern_sample_rate, output, flags=re.MULTILINE) 188 match_channels = re.search(pattern_channels, output, flags=re.MULTILINE) 189 190 # Extraction des valeurs 191 if match_sample_rate: 192 sample_rate = int(match_sample_rate.group(1)) 193 else: 194 raise AudioIO.AudioIOException("Unable to get audio sample_rate of '" + str(filename) + "'") 195 196 if match_channels: 197 channels = int(match_channels.group(1)) 198 else: 199 raise AudioIO.AudioIOException("Unable to get audio channels of '" + str(filename) + "'") 200 201 return (channels,sample_rate) 202 203 # Attributes 204 mode: PipeMode 205 """ Pipemode of the current object (default PipeMode.UNK_MODE)""" 206 207 loglevel: int 208 """ loglevel of the underlying ffmpeg backend for this object (default 16)""" 209 210 debug: bool 211 """ debug flag for this object (print debut info, default False)""" 212 213 channels: int 214 """ Number of channels of images (default -1) """ 215 216 sample_rate: int 217 """ sample_rate of images (default -1) """ 218 219 plannar: bool 220 """ Read/write data as plannar, i.e. not interleaved (default True) """ 221 222 pipe: sp.Popen 223 """ pipe object to ffmpeg/ffprobe (default None)""" 224 225 frame_size: int 226 """ Weight in bytes of one image (default -1)""" 227 228 filename: str 229 """ Filename of the file (default None)""" 230 231 frame_counter: FrameCounter 232 """ `Framecounter` object to count ellapsed time (default None)"""
Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds.
Parameters
filename : str or path. Raw audio waveform as a 1D array.
debug : bool (default (False). Show debug info.
log_level: int (default 16). Log level to pass to the underlying ffmpeg/ffprobe command.
Returns
tuple Tuple containing (channels,sample_rate) of the file
100 @staticmethod 101 def get_time_in_sec(filename, *, debug=False, logLevel=16): 102 """ 103 Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals). 104 105 Parameters 106 ---------- 107 filename : str or path. 108 Raw audio waveform as a 1D array. 109 110 debug : bool (default False). 111 Show debug info. 112 113 logLevel: int (default 16). 114 Log level to pass to the underlying ffmpeg/ffprobe command. 115 116 Returns 117 ---------- 118 float 119 Length in seconds of video file (including milliseconds as decimal part with 3 decimals) 120 """ 121 122 cmd = [AudioIO.paramProgram, # ffprobe 123 '-hide_banner', 124 '-loglevel', str(logLevel), 125 '-show_entries', 'format=duration', 126 '-of', 'default=noprint_wrappers=1:nokey=1', 127 str(filename) 128 ] 129 130 if debug == True: 131 print(' '.join(cmd)) 132 133 # call ffprobe and get params in one single line 134 lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg 135 output = lpipe.stdout.readlines() 136 lpipe.terminate() 137 # transform Bytes output to one single string 138 output = ''.join( [element.decode('utf-8') for element in output]) 139 140 try: 141 return float(output) 142 except (ValueError, TypeError): 143 return None
Static method to get length of an audio file (or video file containing audio) in seconds including milliseconds as decimal part (3 decimals).
Parameters
filename : str or path. Raw audio waveform as a 1D array.
debug : bool (default False). Show debug info.
logLevel: int (default 16). Log level to pass to the underlying ffmpeg/ffprobe command.
Returns
float Length in seconds of video file (including milliseconds as decimal part with 3 decimals)
145 @staticmethod 146 def get_params(filename, *, debug=False, logLevel=16): 147 """ 148 Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds. 149 150 Parameters 151 ---------- 152 filename : str or path. 153 Raw audio waveform as a 1D array. 154 155 debug : bool (default (False). 156 Show debug info. 157 158 log_level: int (default 16). 159 Log level to pass to the underlying ffmpeg/ffprobe command. 160 161 Returns 162 ---------- 163 tuple 164 Tuple containing (channels,sample_rate) of the file 165 """ 166 cmd = [AudioIO.paramProgram, # ffprobe 167 '-hide_banner', 168 '-loglevel', str(logLevel), 169 '-show_entries', 'stream=channels,sample_rate', 170 str(filename) 171 ] 172 173 if debug == True: 174 print(' '.join(cmd)) 175 176 # call ffprobe and get params in one single line 177 lpipe = sp.Popen(cmd, stdout=sp.PIPE, stdin=sp.PIPE) # stdin=sp.PIPE to prevent manipulation of shell echo mode by ffmpeg 178 output = lpipe.stdout.readlines() 179 lpipe.terminate() 180 # transform Bytes output to one single string 181 output = ''.join( [element.decode('utf-8') for element in output]) 182 183 pattern_sample_rate = r'sample_rate=(\d+)' 184 pattern_channels = r'channels=(\d+)' 185 186 # Search for values in the ffprobe output 187 match_sample_rate = re.search(pattern_sample_rate, output, flags=re.MULTILINE) 188 match_channels = re.search(pattern_channels, output, flags=re.MULTILINE) 189 190 # Extraction des valeurs 191 if match_sample_rate: 192 sample_rate = int(match_sample_rate.group(1)) 193 else: 194 raise AudioIO.AudioIOException("Unable to get audio sample_rate of '" + str(filename) + "'") 195 196 if match_channels: 197 channels = int(match_channels.group(1)) 198 else: 199 raise AudioIO.AudioIOException("Unable to get audio channels of '" + str(filename) + "'") 200 201 return (channels,sample_rate) 202 203 # Attributes 204 mode: PipeMode 205 """ Pipemode of the current object (default PipeMode.UNK_MODE)""" 206 207 loglevel: int 208 """ loglevel of the underlying ffmpeg backend for this object (default 16)""" 209 210 debug: bool 211 """ debug flag for this object (print debut info, default False)""" 212 213 channels: int 214 """ Number of channels of images (default -1) """ 215 216 sample_rate: int 217 """ sample_rate of images (default -1) """ 218 219 plannar: bool 220 """ Read/write data as plannar, i.e. not interleaved (default True) """ 221 222 pipe: sp.Popen 223 """ pipe object to ffmpeg/ffprobe (default None)""" 224 225 frame_size: int 226 """ Weight in bytes of one image (default -1)""" 227 228 filename: str 229 """ Filename of the file (default None)""" 230 231 frame_counter: FrameCounter 232 """ `Framecounter` object to count ellapsed time (default None)"""
Static method to get params (channels,sample_rate) of a (video containing) audio file in seconds.
Parameters
filename : str or path. Raw audio waveform as a 1D array.
debug : bool (default (False). Show debug info.
log_level: int (default 16). Log level to pass to the underlying ffmpeg/ffprobe command.
Returns
tuple Tuple containing (channels,sample_rate) of the file
327 def is_opened(self) -> bool: 328 """ 329 Method to get status of the underlying pipe to ffmpeg. 330 331 Returns 332 ---------- 333 bool 334 True if pipe is opened (reading or writing mode), False if not. 335 """ 336 # is the pip opened? 337 if self.pipe is not None and self.pipe.poll() is None: 338 return True 339 340 return False
Method to get status of the underlying pipe to ffmpeg.
Returns
bool True if pipe is opened (reading or writing mode), False if not.
568 def read_frame(self, with_timestamps = False): 569 """ 570 Read next frame from the audio file 571 572 Parameters 573 ---------- 574 with_timestamps: bool optional (default False) 575 If set to True, the method returns a ``FrameContainer`` with the audio and an array containing the associated timestamp(s) 576 577 Returns 578 ---------- 579 nparray or FrameContainer 580 A frame of shape (self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A frame 581 of shape (self.channels*self.frame_size) with interleaved data if self.plannar is False. 582 if with_timestamps is True, the return object is a FrameContainer with the audio data in ``FrameContainer.data`` and 583 the associated timestamp in ``FrameContainer.timestamps`` as an array (one element). 584 """ 585 586 if self.pipe is None: 587 raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading a frame.".format(self.audioProgram)) 588 # - pipe is in write mode 589 if self.mode != PipeMode.READ_MODE: 590 raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename)) 591 592 if with_timestamps: 593 # get elapsed time in video, it is time of next frame(s) 594 current_elapsed_time = self.get_elapsed_time() 595 596 # read rgb image from pipe 597 toread = self.frame_size*self.float_size 598 buffer = self.pipe.stdout.read(toread) 599 600 if buffer == b"": 601 # not considered as an error, no more frame, no exception 602 return None 603 604 # get numpy UINT8 array from buffer 605 audio = np.frombuffer(buffer, dtype = np.float32).reshape(len(buffer)//self.float_size, self.channels) 606 607 # make it plannar (or not) 608 if self.plannar: 609 #transpose it 610 audio = audio.T 611 612 # increase frame_counter 613 self.frame_counter.frame_count += (self.frame_size * self.channels) 614 615 # say to gc that this buffer is no longer needed 616 del buffer 617 618 if with_timestamps: 619 return FrameContainer(1, audio, self.frame_size/self.sample_rate, current_elapsed_time) 620 621 return audio
Read next frame from the audio file
Parameters
with_timestamps: bool optional (default False)
If set to True, the method returns a FrameContainer with the audio and an array containing the associated timestamp(s)
Returns
nparray or FrameContainer
A frame of shape (self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A frame
of shape (self.channels*self.frame_size) with interleaved data if self.plannar is False.
if with_timestamps is True, the return object is a FrameContainer with the audio data in FrameContainer.data and
the associated timestamp in FrameContainer.timestamps as an array (one element).
623 def read_batch(self, numberOfFrames, with_timestamps = False): 624 """ 625 Read next batch of audio from the file 626 627 Parameters 628 ---------- 629 number_of_frames: int 630 Number of desired images within the batch. The last batch from the file may have less images. 631 632 with_timestamps: bool optional (default False) 633 If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames 634 635 Returns 636 ---------- 637 nparray or FrameContainer 638 A batch of shape (n, self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A batch 639 of shape (n, self.channels*self.frame_size) with interleaved data if self.plannar is False. 640 if with_timestamps is True, the return object is a FrameContainer with the audio batch in ``FrameContainer.data`` and 641 the associated timestamp in ``FrameContainer.timestamps`` as an array (one element for each audio frame). 642 """ 643 644 if self.pipe is None: 645 raise self.AudioIOException("No pipe opened to {}. Call open(...) before reading frames.".format(self.audioProgram)) 646 # - pipe is in write mode 647 if self.mode != PipeMode.READ_MODE: 648 raise self.AudioIOException("Pipe to {} for '{}' not opened in read mode.".format(self.audioProgram, self.filename)) 649 650 if with_timestamps: 651 # get elapsed time in video, it is time of next frame(s) 652 current_elapsed_time = self.get_elapsed_time() 653 654 # try to read complete batch 655 toread = self.frame_size*self.float_size*self.channels*numberOfFrames 656 buffer = self.pipe.stdout.read(toread) 657 658 # check if we are at the end of the buffer 659 if buffer == b"": 660 # not considered as an error, no more frame, no exception 661 return None 662 663 # compute actual number of Frames 664 # do we have a full batch ? Computer standard division 665 actualNbFrames = len(buffer)/(self.frame_size*self.float_size*self.channels) 666 667 if actualNbFrames.is_integer(): 668 actualNbFrames = int(actualNbFrames) 669 # get and reshape batch from buffer 670 batch = np.frombuffer(buffer, dtype = np.float32).reshape((actualNbFrames, self.frame_size, self.channels,)) 671 else: 672 # We do not have a full batch, we are at end of the stream or ffmpeg has been stopped/killed 673 # Compute frame size in samples knowing len of buffer, self.float_size, self.channels and numberOfFrames 674 l_frame_size = len(buffer)//(numberOfFrames*self.float_size*self.channels) 675 if l_frame_size <= 0: 676 # not considered as an error, no more frame, no exception 677 return None 678 # get and reshape batch from buffer 679 batch = np.frombuffer(buffer, dtype = np.float32).reshape((numberOfFrames, l_frame_size, self.channels,)) 680 681 if self.plannar: 682 batch = batch.transpose(0, 2, 1) 683 684 # increase frame_counter 685 self.frame_counter.frame_count += (actualNbFrames * self.frame_size * self.channels) 686 687 # say to gc that this buffer is no longer needed 688 del buffer 689 690 if with_timestamps: 691 return FrameContainer( actualNbFrames, batch, self.frame_size/self.sample_rate, current_elapsed_time) 692 693 return batch
Read next batch of audio from the file
Parameters
number_of_frames: int Number of desired images within the batch. The last batch from the file may have less images.
with_timestamps: bool optional (default False) If set to True, the method returns a FrameContainer with the batch and the an array containing the associated timestamps to frames
Returns
nparray or FrameContainer
A batch of shape (n, self.channels,self.frame_size) as defined in the reader/open call if self.plannar is True. A batch
of shape (n, self.channels*self.frame_size) with interleaved data if self.plannar is False.
if with_timestamps is True, the return object is a FrameContainer with the audio batch in FrameContainer.data and
the associated timestamp in FrameContainer.timestamps as an array (one element for each audio frame).
695 def write_frame(self, audio) -> bool: 696 """ 697 Write an audio frame to the file 698 699 Parameters 700 ---------- 701 audio: nparray 702 The audio frame to write to the video file of shape (self.channels,nb_samples_per_channel) if plannar is True else (self.channels*nb_samples_per_channel). 703 704 Returns 705 ---------- 706 bool 707 Writing was successful or not. 708 """ 709 # Check params 710 # - pipe exists 711 if self.pipe is None: 712 raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram)) 713 # - pipe is in write mode 714 if self.mode != PipeMode.WRITE_MODE: 715 raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename)) 716 # - shape of image is fine, thus we have pixels for a full compatible frame 717 if audio.shape[0] != self.channels: 718 raise self.AudioIOException("Wong audio shape: {} expected ({}, nb_samples_to_write).".format(audio.shape,self.channels)) 719 # - type of data is Float32 720 if audio.dtype != np.float32: 721 raise self.AudioIOException("Wong audio type: {} expected np.float32.".format(audio.dtype)) 722 723 # array must have a shape (channels, samples), reshape it it to (samples, channels) if plannar 724 if not self.plannar: 725 audio = audio.reshape(-1) 726 727 # print( audio.shape ) 728 729 # garantee to have a C continuous array 730 if not audio.flags['C_CONTIGUOUS']: 731 a = np.ascontiguousarray(a) 732 733 # write frame 734 buffer = audio.tobytes() 735 if self.pipe.stdin.write( buffer ) < len(buffer): 736 print( f"Error writing frame to {self.filename}" ) 737 return False 738 739 # increase frame_counter 740 self.frame_counter.frame_count += (audio.shape[1] * self.channels) 741 742 # say to gc that this buffer is no longer needed 743 del buffer 744 745 return True
Write an audio frame to the file
Parameters
audio: nparray The audio frame to write to the video file of shape (self.channels,nb_samples_per_channel) if plannar is True else (self.channels*nb_samples_per_channel).
Returns
bool Writing was successful or not.
747 def write_batch(self, batch): 748 """ 749 Write a batch of audio frame to the file 750 751 Parameters 752 ---------- 753 batch: nparray 754 The batch of audio frames to write to the video file of shape (n,self.channels,nb_samples_per_channel) if plannar is True else (n,self.channels*nb_samples_per_channel) of interleaved audio data. 755 756 Returns 757 ---------- 758 bool 759 Writing was successful or not. 760 """ 761 # Check params 762 # - pipe exists 763 if self.pipe is None: 764 raise self.AudioIOException("No pipe opened to {}. Call create(...) before writing frames.".format(self.audioProgram)) 765 # - pipe is in write mode 766 if self.mode != PipeMode.WRITE_MODE: 767 raise self.AudioIOException("Pipe to {} for '{}' not opened in write mode.".format(self.audioProgram, self.filename)) 768 # batch is 3D (n, channels, nb samples) 769 if batch.ndim !=3: 770 raise self.AudioIOException("Wrong batch shape: {} expected 3 dimensions (n, n_channels, n_samples_per_channel).".format(batch.shape)) 771 # - shape of images in batch is fine 772 if batch.shape[2] != self.channels: 773 raise self.AudioIOException("Wrong audio channels in batch: {} expected {} {}.".format(batch.shape[2], self.channels, batch.shape)) 774 775 # array must have a shape (n * n_channels * n_samples_per_channel) before writing them to pipe 776 # reshape it it to (n * n_channels * n_samples_per_channel) if plannar is False 777 if not self.plannar: 778 # goes from (n, n_channels, n_samples_per_channel) to (n * n_channels * n_samples_per_channel) 779 batch = batch.transpose(0, 2, 1) # first go to (n, n_samples_per_channel, n_channels) 780 batch = batch.reshape(-1) # then to 1D array (n * n_channels * n_samples_per_channel) 781 782 # garantee to have a C continuous array 783 if not batch.flags['C_CONTIGUOUS']: 784 batch = np.ascontiguousarray(batch) 785 786 # write frame 787 buffer = batch.tobytes() 788 if self.pipe.stdin.write( buffer ) < len(buffer): 789 # say to gc that this buffer is no longer needed 790 del buffer 791 raise self.AudioIOException("Error writing batch to '{}'.".format(self.filename)) 792 793 # increase frame_counter 794 self.frame_counter.frame_count += int(batch.shape[0]/self.channels) # int conversion is mandatory to avoid confusion with time as float 795 796 # say to gc that this buffer is no longer needed 797 del buffer 798 799 return True
Write a batch of audio frame to the file
Parameters
batch: nparray The batch of audio frames to write to the video file of shape (n,self.channels,nb_samples_per_channel) if plannar is True else (n,self.channels*nb_samples_per_channel) of interleaved audio data.
Returns
bool Writing was successful or not.
38 class AudioIOException(Exception): 39 """ 40 Dedicated exception class for AudioIO class. 41 """ 42 def __init__(self, message="Error while reading/writing video occurs"): 43 self.message = message 44 super().__init__(self.message)
Dedicated exception class for AudioIO class.
46 class AudioFormat(Enum): 47 """ 48 Enum class for supported input video type: 32-bit float is the only supported type for the moment. 49 """ 50 PCM32LE = 'pcm_f32le' # default format (unique mode for the moment)
Enum class for supported input video type: 32-bit float is the only supported type for the moment.