Repository navigation
Expand file tree
/
Copy pathpreprocessing.py
More file actions
241 lines (198 loc) · 6.88 KB
/
Copy pathpreprocessing.py
File metadata and controls
241 lines (198 loc) · 6.88 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
'''
Place coughs in ../cough and music in ../music
This is the place for creating audio chunks with corresponding data about if
and where there are coughs in the resulting audio chunks.
The resulting output is an array of objects. The objects need two variables:
the audio-waveform data and the position Data of the cough, given in frames.
For preparing the cough data in Terminal on cough folder run:
"for f in *.ogg
ffmpeg -i $f -c:a pcm_f32le $f.wav
end"
If this file is run as main you will hear randomly generated frames.
To use this as an import use:
"from preprocessing import getFrame"
The getFrame function will return sampleRate: int, audioData: np.ndarray.
'''
import string
import numpy as np
import scipy.io.wavfile as scpw
import os
import random
import resampy
from retry import retry
from pedalboard import Reverb
# import sounddevice as sd
rmsStepSize = 100 # Step size for RMS analysis in milliseconds
rmsThreshold = 0.01 # 0.5 # Threshold for cutting of silence
frameLength = 1 # Size of generated audio pieces in seconds
loudestCough = -6 # the loudest possible coughing volume in dB
quietestCough = -24 # the quietest possible coughing volume in dB
globalSampleRate = 32000 # sampleRate used for the output
reverb = Reverb()
reverb.dry_level = 0.5
def getRandomFile(directory: str):
'''
Returns a random path from chosen directory.
'''
chosenFile = random.choice(os.listdir(directory))
# print(chosenFile)
chosenPath = directory + chosenFile
return chosenPath
def sumToMono(audioData: np.ndarray):
'''
Sums a stereo audio file to mono.
'''
if len(audioData.shape) == 2:
data = np.sum(audioData, axis=1)
return data
else:
return audioData
def normaliseNdarray(audioData: np.ndarray):
'''
Returns biggest absolute value of ndarray for normalisation.
'''
absmax = max(audioData, key=abs)
audioData = np.divide(audioData, absmax) # normalise values
return audioData
def cutRandomFrame(audioData: np.ndarray, sampleRate: int):
'''
Randomly cuts a frame from a bigger audio file and returns it.
'''
sampleFrameLength = frameLengthInSamples(sampleRate, frameLength)
range = len(audioData) - sampleFrameLength
startPos = random.randrange(range)
endPos = startPos + sampleFrameLength
randomSecondAudio = audioData[startPos:endPos]
# print(f"start: {startPos} length: {len(randomSecondAudio)}")
return randomSecondAudio
def cutIntoFrames(audioData: np.ndarray, sampleRate: int):
'''
Cut a longer audio file into consecutive frames.
'''
sampleFrameLength = frameLengthInSamples(sampleRate, frameLength)
audioDataLength = len(audioData) - (len(audioData) % sampleFrameLength)
arraySplit = audioDataLength / sampleFrameLength
audioFrames = np.split(
audioData[0:audioDataLength], arraySplit)
return audioFrames
'''
def playNdarray(audioData: np.ndarray, sampleRate: int):
audioData = np.int16(audioData/np.max(np.abs(audioData)) * 32767)
sd.play(audioData, sampleRate, blocking=True)
return
'''
def RMS(audioData: np.ndarray):
'''
Return RMS Value over ndarray.
'''
return np.sqrt(abs(np.mean(audioData**2)))
def dbToA(db: float):
'''
Convert decibels to amplitude.
'''
amplitude = 10**(db/20)
return amplitude
def frameLengthInSamples(sampleRate: int, frameLength: int):
'''
Return frame Length in samples when given length in seconds and samplerate.
'''
frameLength = round(sampleRate * frameLength)
return frameLength
def removeSilence(audioData: np.ndarray, sampleRate: int):
'''
Remove silence at the beginning and end of an audio file.
'''
rmsSampleStep = round(rmsStepSize*(sampleRate/1000))
startPos = None
endPos = None
for i in range(0, len(audioData) - rmsSampleStep, rmsSampleStep):
rms = RMS(audioData[i:i+rmsSampleStep])
if startPos is None:
if rms > rmsThreshold:
startPos = i
else:
if endPos is None:
if rms <= rmsThreshold:
endPos = i
if startPos is None:
raise ValueError('Audio Sample is too quiet.')
if endPos is None:
endPos = len(audioData)
if endPos-startPos < frameLengthInSamples(sampleRate, frameLength):
raise ValueError('Audio Sample too short.')
else:
audioWithoutSilence = audioData[startPos:endPos]
return audioWithoutSilence
def addCoughToMusic(dataMusic: np.ndarray,
dataCough: np.ndarray,
sampleRate: int):
'''
Add a coughing frame to a music frame with randomized volume.
'''
minusThreeDb = 0.7079457843841379
lowerLimit = dbToA(quietestCough)
upperLimit = dbToA(loudestCough)
dataCough = dataCough * random.uniform(lowerLimit, upperLimit)
data = np.add(dataMusic*minusThreeDb, dataCough*minusThreeDb)
return data
@retry()
def processCough():
'''
Create a cough frame with reverb.
'''
reverb.wet_level = random.uniform(0.0, 0.5)
reverb.room_size = random.uniform(0.1, 0.6)
path = getRandomFile("./cough/")
sr, data = scpw.read(path)
data = sumToMono(data)
data = np.int16(data/np.max(np.abs(data)) * 32767)
data = reverb(data, sr)
data = removeSilence(data, sr)
data = cutRandomFrame(data, sr)
if sr != globalSampleRate:
data = resampy.resample(data, sr, globalSampleRate)
return globalSampleRate, data
@retry()
def processMusic():
'''
Create a music frame.
'''
sr, data = scpw.read(getRandomFile("./music/"))
data = sumToMono(data)
data = np.int16(data/np.max(np.abs(data)) * 32767)
data = cutRandomFrame(data, sr)
if sr != globalSampleRate:
data = resampy.resample(data, sr, globalSampleRate)
return globalSampleRate, data
def getFrame(withCough: bool, length: float):
'''
Create a frame containing music and optionally with a cough frame added.
'''
global frameLength
frameLength = length
srMusic, dataMusic = processMusic()
if withCough:
srCough, dataCough = processCough()
data = addCoughToMusic(dataMusic, dataCough, srMusic)
# vggish = waveform_to_examples(data, srMusic)
return srMusic, data
else:
# vggish = waveform_to_examples(dataMusic, srMusic)
return srMusic, dataMusic
def getTestFrames(path: string, length: float):
'''
Get consecutive frames from a longer audio file.
'''
global frameLength
frameLength = length
sr, data = scpw.read("./test_music/" + path)
data = sumToMono(data)
data = np.int16(data/np.max(np.abs(data)) * 32767)
if sr != globalSampleRate:
data = resampy.resample(data, sr, globalSampleRate)
data = cutIntoFrames(data, globalSampleRate)
return data
if __name__ == "__main__":
while True:
sr, data = getFrame(True, 1.0)
# playNdarray(data, sr)