-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathpre_processing.py
More file actions
413 lines (320 loc) · 15.6 KB
/
Copy pathpre_processing.py
File metadata and controls
413 lines (320 loc) · 15.6 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
"""
This module is used to preprocess audio files for use in machine learning(DL).
This module contains 3 main classes namely:
- LoadAudioFile:
- PadAudioArray: Pads audio array with zeros at the end of file with missing frames.
- Normalize: Contains function for normalizing given data.
"""
import os
from typing import Union, Literal
import librosa
import librosa.feature
import numpy as np
import pandas as pd
from audioread import NoBackendError
from colorama import Fore
from tensorflow import reduce_min, reduce_max
import tensorflow as tf
np.set_printoptions(precision=4)
class PadDataArray:
def __init__(self,
mode: Literal["constant", "edge", "linear_ramp", "maximum", "mean",
"median", "minimum", "reflect", "symmetric", "wrap", "empty"] = 'constant'):
"""
Initializes the class with the specified padding mode.
:param mode:
"""
self.mode = mode
def zero_pad(self, data: np.ndarray,
num_zeros: int = 0,
side: Literal['right', 'left', 'both'] = "right"):
"""
Pads array with zeros to the target length
:param data: np.ndarray to be padded with zeros.
:param num_zeros: how many zeros are needed to pad the audio file with?
:param side: either "left (start of the array)" or "right(end of the array)" or "both".
:raises ValueError: if the target length is less than the original audio length
:return: np.ndarray padded with zeros.
"""
if side.lower() == 'left': # left pad
return np.pad(data, (num_zeros, 0), mode=self.mode)
elif side.lower() == 'right': # right pad
return np.pad(data, (0, num_zeros), mode=self.mode)
elif side.lower() == 'both': # pad both
return np.pad(data, (num_zeros, num_zeros), mode=self.mode)
else:
raise ValueError("Invalid padding side. Choose 'left' or 'right' or 'both'.")
class LoadAudioFile:
def __init__(self, sr: int = 16000,
start_time: float = 0.0,
duration=3, mono: bool = True,
pad: bool = True):
"""
Initializes the class with the given parameters.
:param sr: sampling rate of the audio file
:param duration: duration need for audio file to be loaded.
:param mono: boolean flag whether audio file is mono.
:param pad: boolean flag whether audio file is padded.
"""
self.sr = sr
self.start_time = start_time # offset in librosa.load(...)
self.duration = duration
self.mono = mono
self.pad = pad
self.total_frames = int(self.sr * self.duration)
self.Padder = PadDataArray()
def load_audio(self, audio_path, verbose: bool = False) -> Union[np.ndarray, None]:
"""
Loads audio files from a given audio path using librosa.load(...).
Checks for missing frames in the audio from the total frames and
pads the audio array with zeros at the end of file with missing frames using class PadDataArray.
:param audio_path: path to audio file.
:param verbose: boolean flag whether errors are printed or not.
:return: np.ndarray
"""
try:
# loads the audio according to the param with librosa.
audio, _ = librosa.load(audio_path,
sr=self.sr,
offset=self.start_time,
duration=self.duration,
mono=self.mono)
if verbose:
print(f"{Fore.BLUE}Loaded audio -- {Fore.GREEN}{audio_path} {Fore.BLUE}--", Fore.RESET)
except EOFError:
if verbose:
print(f"\nError Loading file, not enough frames in file: {os.path.basename(audio_path)}.\tSkipping...")
return None
except (NoBackendError, librosa.LibrosaError):
if verbose:
print(f"\nError Loading file: {os.path.basename(audio_path)}.\tSkipping...")
return None
except Exception as e:
raise Exception(f"Error name: {type(e).__name__} - {e}, while loading the audio:{audio_path}.")
# Pads the audio with class: PadDataArray
frames_to_pad = (self.total_frames - audio.size)
if self.pad and (frames_to_pad > 0):
audio = self.Padder.zero_pad(audio, frames_to_pad, 'right')
return audio
class TFMinMaxNormalize:
"""
Implements min-max normalization and de-normalization functions for tensorflow tensors.
### ### ### ### ### ### ### Important: In conclusion, while padded zeros won't be drastically affected,
the choice of normalization range (-1 to 1 vs. 0 to 1) can influence the overall scaling of your audio data.
:param new_min: Minimum value of the new range
:param new_max: Maximum value of the new range
"""
def __init__(self, new_min=-1, new_max=1):
self.new_min = new_min
self.new_max = new_max
def normalize(self, array):
stft_min = reduce_min(array, axis=-1, keepdims=True)
stft_max = reduce_max(array, axis=-1, keepdims=True)
return (((array - stft_min) / ((stft_max - stft_min) + 1e-10)) *
(self.new_max - self.new_min) + self.new_min)
class MinMaxNormalize:
"""
Implements min-max normalization and de-normalization functions.
Automatically calls fit before normalization if data hasn't been fitted yet.
### ### ### ### ### ### ### Important: In conclusion, while padded zeros won't be drastically affected,
the choice of normalization range (-1 to 1 vs. 0 to 1) can influence the overall scaling of your audio data.
:param new_min: Minimum value of the new range
:param new_max: Maximum value of the new range
"""
def __init__(self, new_min=0, new_max=1):
self.new_min = new_min
self.new_max = new_max
def normalize(self, array: np.ndarray):
"""
Normalizes the input array between self.new_min and self.new_max.
Automatically calls fit if original min/max haven't been calculated.
:param array: np.ndarray to be normalized.
"""
return ((array - array.min()) / (array.max() - array.min() + 1e-10)
* (self.new_max - self.new_min) + self.new_min)
def denormalize(self, norm_array: np.ndarray, original_min, original_max):
"""
Denormalizes the input array back to its original scale.
Performs de-normalization in one step for efficiency.
:param norm_array: np.ndarray to be denormalized.
:param original_min:
:param original_max:
"""
if original_min is None or original_max is None:
raise ValueError("De-normalization cannot be performed when original min and max are not set.")
if original_min == original_max:
raise ValueError("De-normalization cannot be performed when original min and max are equal.")
return ((norm_array - self.new_min) / (self.new_max - self.new_min + 1e-10)
* (original_max - original_min) + original_min)
# TODO: fix the duplicated code fragment
class LoadDirectoryPathSet:
"""
Class for loading the directory paths when in a directory structure of:
- dataset
- real
- fake
"""
def __init__(self):
self.RealPaths = None
self.FakePaths = None
# above lists concatenated to form below list
def load_audio_path(self, dataset_path: Union[str, os.PathLike],
is_audio_paths: bool = False):
"""
This function loads teh audio paths in self.RealPaths and self.FakePaths.
and concatenates them into a numpy array named, self.AudioPaths.
:param dataset_path: path to directory of dataset
:param is_audio_paths: whether to return the self.AudioPaths or not
:return: np.ndarray with audio paths if self.AudioPaths is set.
"""
# creating a list of all the paths with real files
self.RealPaths = [os.path.join(dataset_path, 'real', file_name)
for folder in os.listdir(dataset_path)
if folder == 'real'
for file_name in os.listdir(os.path.join(dataset_path, folder))
if file_name.endswith('.wav') or file_name.endswith('.flac')]
if not self.RealPaths:
print(f"No 'real' audio paths found in the given directory.{dataset_path}")
else:
self.RealPaths = np.array(self.RealPaths, dtype=str)
self.RealPaths = np.expand_dims(self.RealPaths, axis=1)
ones = np.ones((self.RealPaths.shape[0], 1), dtype=int)
self.RealPaths = np.concatenate((self.RealPaths, ones), axis=1)
# creating a list of all the paths with real files
self.FakePaths = [os.path.join(dataset_path, 'fake', file_name)
for folder in os.listdir(dataset_path)
if folder == 'fake'
for file_name in os.listdir(os.path.join(dataset_path, folder))
if file_name.endswith('.wav') or file_name.endswith('.flac')]
if not self.FakePaths:
print(f"No 'fake' audio paths found in the given directory.{dataset_path}")
else:
self.FakePaths = np.array(self.FakePaths, dtype=str)
self.FakePaths = np.expand_dims(self.FakePaths, axis=1)
zeros = np.zeros((self.FakePaths.shape[0], 1), dtype=int)
self.FakePaths = np.concatenate((self.FakePaths, zeros), axis=1)
if is_audio_paths and self.FakePaths and self.RealPaths:
return np.concatenate((self.RealPaths, self.FakePaths), axis=0)
class LoadCSVPathSet:
"""
This functino is only for ASV2019 Dataset, not generalized yet
"""
def __init__(self):
self.AudiosetDF = None
self.RealPaths = []
self.FakePaths = []
def load_audio_path(self, csv_file_path: Union[str, os.PathLike[str]],
dataset_path: Union[str, os.PathLike]):
audioset = np.array(self._read_csv(csv_file_path))
for i in range(audioset.shape[0]):
if audioset[i, 1] == 'bonafide':
audioset[i, 1] = 1
self.RealPaths.append(os.path.join(dataset_path, str(audioset[i, 0]) + '.flac'))
else:
audioset[i, 1] = 0
self.FakePaths.append(os.path.join(dataset_path, str(audioset[i, 0]) + '.flac'))
if not self.RealPaths:
print("No 'real' audio paths found in the given dataset.")
else:
self.RealPaths = np.array(self.RealPaths, dtype=str)
self.RealPaths = np.expand_dims(self.RealPaths, axis=1)
ones = np.ones((self.RealPaths.shape[0], 1), dtype=int)
self.RealPaths = np.concatenate((self.RealPaths, ones), axis=1)
if not self.FakePaths:
print("No 'fake' audio paths found in the given dataset.")
else:
self.FakePaths = np.array(self.FakePaths, dtype=str)
self.FakePaths = np.expand_dims(self.FakePaths, axis=1)
zeros = np.zeros((self.FakePaths.shape[0], 1), dtype=int)
self.FakePaths = np.concatenate((self.FakePaths, zeros), axis=1)
def _read_csv(self, csv_file_path: Union[str, os.PathLike], ):
dataframe = pd.read_csv(csv_file_path, header=None, sep=' ')
dataframe.columns = ['SPEAKER_ID', "AUDIO_FILE_NAME", "Environment_ID", "SYSTEM_ID", "KEY", ]
columns_to_keep = ['AUDIO_FILE_NAME', 'KEY']
self.AudiosetDF = dataframe[columns_to_keep]
return self.AudiosetDF
class AudioAugmentation:
def __init__(self, val_t: tuple = (10, 50), val_f: tuple = (5, 40)):
if val_t[0] >= val_t[1] or val_f[0] >= val_f[1]:
raise ValueError(f"min value should be less than max value.")
self.min_val_t = val_t[0]
self.max_val_t = val_t[1]
self.min_val_f = val_f[0]
self.max_val_f = val_f[1]
# just a wrapper around all the time augmentation functions
def time_aug(self, audio: np.ndarray, /, *, aug_prob) -> np.ndarray:
required_keys = ['noise_prob', 'gain_prob', 'time_shift_prob']
aug_prob = self._validate_keys(required_keys=required_keys, dict_to_val=aug_prob)
if aug_prob['noise_prob'] > np.random.rand():
audio = self._white_noise(audio)
if aug_prob['time_shift_prob'] > np.random.rand():
audio = self._time_shift(audio)
return audio
def _white_noise(self, signal: np.ndarray) -> np.ndarray:
"""
This function adds white noise on the audio signal provided.
:param signal: audio signal
:return: ndarray of audio signal with white noise.
"""
noise = self.random_float(0, signal.std(), signal.size)
noise_factor = self.random_float(0.05, 0.10)
return signal + (noise * noise_factor)
# TODO: understand!
# Randomly shift audio -> any sound at <t> time may get shifted to <t+shift> time
def _time_shift(self, audio):
shift = self.random_int(min_val=0, max_val=np.shape(audio)[0])
shift = -shift
audio = np.roll(audio, shift, axis=0)
return audio
# just a wrapper around all the spec augmentation functions
def spec_aug(self, spec, aug_prob):
required_keys = ['freq_mask_prob', 'time_mask_prob']
aug_prob = self._validate_keys(required_keys=required_keys, dict_to_val=aug_prob)
if aug_prob["freq_mask_prob"] > np.random.rand():
spec = self.freq_masking(spec)
if aug_prob["time_mask_prob"] > np.random.rand():
spec = self.time_masking(spec)
return spec
def freq_masking(self, spec):
mask = self.random_int(self.min_val_f, self.max_val_f)
num_r = self.random_int(0, spec.shape[0]) - 1
for j in range(num_r, num_r + min(mask, spec.shape[0] - num_r)):
spec[j] = 0
return spec
def time_masking(self, spec):
mask = self.random_int(self.min_val_t, self.max_val_t)
num_r = spec.shape[0]
num_c = self.random_int(0, spec.shape[1]) - 1
for i in range(0, num_r):
for j in range(num_c, num_c + min(mask, spec.shape[1] - num_c)):
spec[i, j] = 0
return spec
@staticmethod
def random_int(min_val=0, max_val=1, size=None):
return np.int32(np.random.uniform(min_val, max_val, size))
@staticmethod
def random_float(min_val=0.0, max_val=1.0, size=None):
return np.float32(np.random.uniform(min_val, max_val, size))
@staticmethod
def _validate_keys(required_keys: list, dict_to_val: dict = None, ) -> dict:
"""
:raises TypeError: if the required dictionary is not a dict.
:raises ValueError: if the required values are not between 0 and 1.
:raises KeyError: if the keys do not match the required keys.
:param dict_to_val:
:param required_keys:
:return:
"""
if type(dict_to_val) is not dict:
dict_to_val = {}
# Raise error if an extra key is provided
for key, value in dict_to_val.items():
if key not in required_keys:
raise KeyError(f"Wrong key: {key}, Please select the keys from {required_keys}")
if 1 < value or 0 > value:
raise ValueError("Please provide a valid dictionary with probability between 0-1.")
# Check for missing keys
missing_aug = [key for key in required_keys if key not in dict_to_val]
for aug in missing_aug:
dict_to_val[aug] = 0
return dict_to_val