-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathfeature_extraction.py
More file actions
136 lines (101 loc) · 4.16 KB
/
Copy pathfeature_extraction.py
File metadata and controls
136 lines (101 loc) · 4.16 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
import librosa
import numpy as np
from librosa import feature
class ExtractAbsSTFT:
def __init__(self, *, n_fft, hop_length):
self.n_fft = n_fft
self.hop_length = hop_length
def extract(self, audio):
# leaving the last frame, just better for audio manipulation
stft = librosa.stft(audio, n_fft=self.n_fft, hop_length=self.hop_length)[:-1]
spectrogram = np.abs(stft) ** 2
return spectrogram
class ExtractSTFTDB:
def __init__(self, *, n_fft, hop_length):
self.n_fft = n_fft
self.hop_length = hop_length
def extract(self, audio):
# leaving the last frame, just better for audio manipulation
stft = librosa.stft(audio, n_fft=self.n_fft, hop_length=self.hop_length)[:-1]
spectrogram = np.abs(stft) ** 2
stft_db = librosa.power_to_db(spectrogram)
return stft_db
class ExtractLogSTFT:
def __init__(self, *, n_fft, hop_length):
self.n_fft = n_fft
self.hop_length = hop_length
def extract(self, audio):
# leaving the last frame, just better for audio manipulation
stft = librosa.stft(audio, n_fft=self.n_fft, hop_length=self.hop_length)[:-1]
spectrogram = np.abs(stft) ** 2
stft_db = librosa.power_to_db(spectrogram)
stft = np.log2(stft_db)
return stft
class ExtractCQT:
def __init__(self, *, n_bins: int = 128, bins_per_octave=18, hop_length=128):
self.n_bins = n_bins
self.bins_per_octave = bins_per_octave
self.hop_length = hop_length
def extract(self, audio):
cqt = librosa.cqt(audio, sr=16000, hop_length=self.hop_length, n_bins=self.n_bins,
bins_per_octave=self.bins_per_octave)
cqt = np.abs(cqt) ** 2
cqt_db = librosa.power_to_db(cqt)
return cqt_db
class ExtractSpecFlux:
def __init__(self, *, n_fft, hop_length):
self.n_fft = n_fft
self.hop_length = hop_length
def extract(self, audio):
"""
Extracts spectral flux from an audio file.
Args:
audio: the audio file.
Returns:
the spectral flux features.
"""
# Calculate spectrogram using librosa.stft
spectrogram = librosa.stft(y=audio, n_fft=self.n_fft, hop_length=self.hop_length)
# Convert spectrogram to magnitude (power) spectrogram
mag_spectrogram = np.abs(spectrogram)
# Convert power spectrogram to log scale
log_spectrogram = librosa.power_to_db(mag_spectrogram)
# Calculate spectral flux by difference along frequency axis
spectral_flux = np.diff(log_spectrogram, axis=1)
return spectral_flux
class ExtractSpecCentroid:
def __init__(self, *, n_fft, hop_length):
self.n_fft = n_fft
self.hop_length = hop_length
def extract(self, audio):
"""
Calculates the spectral centroid of a spectrogram.
Args:
audio: the audio file.
Returns:
A 1D numpy array containing the spectral centroid for each time frame.
"""
S = librosa.stft(audio, n_fft=self.n_fft, hop_length=self.hop_length)
spectrogram = np.abs(S) # Magnitude spectrogram
frequencies = np.arange(spectrogram.shape[0]) # Assuming linear frequency bins
spectral_centroid_tf = np.expand_dims(frequencies, axis=1) * spectrogram
spectral_centroid_tf = np.sum(spectral_centroid_tf, axis=0) / np.sum(spectrogram, axis=0)
return spectral_centroid_tf
class ExtractMFCCs:
def __init__(self, *, sr=16000, n_mfcc=128):
"""
:param sr: Sampling rate of the audio (default: 22050 Hz).
:param n_mfcc: Number of MFCC coefficients to extract (default: 20).
"""
self.sr = sr
self.n_mfcc = n_mfcc
def extract(self, audio):
"""
Extracts Mel-frequency cepstral coefficients (MFCCs) from an audio file.
Args:
audio: the audio file.
Returns:
A 2D numpy array representing the MFCC features.
"""
mfccs = librosa.feature.mfcc(y=audio, sr=self.sr, n_mfcc=self.n_mfcc)
return mfccs.T # Transpose for consistency with spectral flux