-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathtfrecords_write.py
More file actions
178 lines (149 loc) · 8.19 KB
/
Copy pathtfrecords_write.py
File metadata and controls
178 lines (149 loc) · 8.19 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
import os
from typing import Union, Literal
import numpy as np
from colorama import Fore
from tensorflow import io as tf_io, train as tf_train
from pre_processing import LoadAudioFile, AudioAugmentation
class WriteTFRecord(LoadAudioFile, AudioAugmentation):
def __init__(self,
sr: int = 16000,
start_time: float = 0.0,
duration=3, mono: bool = True,
pad: bool = True):
LoadAudioFile.__init__(self, sr=sr,
start_time=start_time,
duration=duration,
mono=mono,
pad=pad)
AudioAugmentation.__init__(self,)
self.Extractor = None
self.Normalizer = None
self.feature_shape = None
def create_tfrecord(self,
tfrecord_file_name: Union[str, os.PathLike],
path_datasets: list[np.ndarray],
names: list[Literal["real", "fake"]],
n: list[int] = None,
audio_aug_prob: dict = None,
freq_aug_prob: dict = None,
verbose: bool = False,
is_bad_audio: bool = True):
# if length of the lists provided are not equal then raise error.
if len(path_datasets) != len(names):
raise ValueError(f'Number of "path_dataset" provided must equal'
f'to the number of their respective "names"!')
if len(path_datasets) != len(n):
raise ValueError(f'Number of "path_dataset" provided must equal'
f'to the number of their respective "n"!')
if len(names) != len(n):
raise ValueError(f'Number of "names" provided must equal to the "n"!')
# if any num in n is less than 1 then raise error
if any(num < 1 for num in n):
raise ValueError("list: n must contain 1 or greater value!")
# if any name in names is not 'real' or 'fake' then raise error.
if any(name != "real" and name != 'fake' for name in names):
raise ValueError("list: 'names' must be either 'real' or 'fake' according to the list: 'path_datasets'!")
tfrecord_writer_real = None
tfrecord_writer_fake = None
for path_dataset, name, number in zip(path_datasets, names, n):
if name == "real":
# Creating TFRecord first time
if tfrecord_writer_real is None:
real_path = os.path.join(os.getcwd(), 'tfrecords', f'{tfrecord_file_name}_{name}.tfrecords')
tfrecord_writer_real = tf_io.TFRecordWriter(real_path)
self._tfrecord_creator(path_dataset, tfrecord_writer_real, name, number,
verbose=verbose, is_bad_audio=is_bad_audio,
audio_aug_prob=audio_aug_prob, freq_aug_prob=freq_aug_prob)
elif name == "fake":
# Creating TFRecord first time
if tfrecord_writer_fake is None:
fake_path = os.path.join(os.getcwd(), 'tfrecords', f'{tfrecord_file_name}_{name}.tfrecords')
tfrecord_writer_fake = tf_io.TFRecordWriter(fake_path)
self._tfrecord_creator(path_dataset, tfrecord_writer_fake, name, number,
verbose=verbose, is_bad_audio=is_bad_audio,
audio_aug_prob=audio_aug_prob, freq_aug_prob=freq_aug_prob)
tfrecord_writer_fake.close()
tfrecord_writer_real.close()
def _tfrecord_creator(self, path_dataset: np.ndarray,
tfrecord_writer: tf_io.TFRecordWriter,
name: str,
n: int = 1,
audio_aug_prob: dict = None,
freq_aug_prob: dict = None,
verbose: bool = False,
is_bad_audio: bool = True, ):
# if n is less than 1 then the main writing loop will not work
if n < 1:
raise IndexError(f"n must be greater than 0, in {name}")
# x and z for keeping track of total files and augmented files, (for debugging reasons)
x, y, z = 0, 0, 0
for i, (path, label) in enumerate(path_dataset):
if verbose:
print(f'{Fore.BLUE}Loading... {Fore.GREEN}{path} {Fore.BLUE}- {Fore.GREEN}{label}', Fore.RESET)
x += 1
feat = None
# loop for extracting audio and features form audio
for j in range(n):
audio = self.load_audio(path, verbose=verbose)
if audio is None or type(audio) is None:
if is_bad_audio:
print(f'{Fore.BLUE}Couldn\'t load Path: '
f'{Fore.GREEN}{path} {Fore.BLUE}and label: {Fore.GREEN}{label}', Fore.RESET)
if not verbose:
print(f'\r{Fore.BLUE}{j}:{Fore.CYAN} Failed to load: {Fore.GREEN}{path}', Fore.RESET, end="")
break
# augment audio if range is more than 1 such that the loop, loops more than 1
if j > 0:
# augment every file loaded, n determines the number of time the file will be augmented starting
# from 2, so 2 == the file augmented once, 3 == the file augmented twice, and so on.
z += 1
try:
# inherited function
audio = self.time_aug(audio, aug_prob=audio_aug_prob)
except Exception as e:
raise Exception("While augmenting time of audio", type(e).__name__, e)
try:
feat = self.Extractor.extract(audio) # extracting audio features
except Exception as e:
raise Exception(f"Error name: {type(e).__name__} - {e} while extracting features and normalizing.")
self.feature_shape = feat.shape
if j > 0:
# augment every file loaded, n determines the number of time the file will be augmented starting
# from 2, so 2 == the file augmented once, 3 == the file augmented twice, and so on.
y += 1
try:
# inherited function
feat = self.spec_aug(feat, aug_prob=freq_aug_prob)
except Exception as e:
raise Exception("While augmenting spec of audio", type(e).__name__, e)
if y != z:
raise Exception("While augmenting audio and spectrogram features. (not equally augmented)")
# calling example function
example = self._create_examples(audio_features=feat, label=int(label))
tfrecord_writer.write(example)
print(f'\r{Fore.BLUE}{i}{Fore.CYAN}: shape{self.feature_shape},'
f' type:{type(feat[0, 0])}', Fore.RESET, end="")
print(f'\nWrote {x + z} files in {name}, {z} files were augmented')
@staticmethod
def _create_examples(audio_features: np.ndarray,
label: int,
verbose: bool = False):
if audio_features.dtype == np.float32:
flatten_array = audio_features.flatten()
audio_bytes = tf_io.serialize_tensor(flatten_array)
bytes_string_audio = audio_bytes.numpy()
else:
raise TypeError(f"'audio_features' content type must be np.float32")
audio_feature = tf_train.Feature(bytes_list=tf_train.BytesList(value=[bytes_string_audio]))
label_feature = tf_train.Feature(int64_list=tf_train.Int64List(value=[label]))
example = tf_train.Example(
features=tf_train.Features(feature={
'stft': audio_feature,
'label': label_feature,
})
)
if verbose:
print(f'{Fore.BLUE}Creating TFRecord example for label '
f'{Fore.GREEN}{audio_features} {Fore.BLUE}- {Fore.GREEN}{label}',
Fore.RESET)
return example.SerializeToString()