diff --git a/examples/mic_vad_streaming/mic_vad_streaming.py b/examples/mic_vad_streaming/mic_vad_streaming.py old mode 100644 new mode 100755 index 259d4fef..6e7f4993 --- a/examples/mic_vad_streaming/mic_vad_streaming.py +++ b/examples/mic_vad_streaming/mic_vad_streaming.py @@ -1,12 +1,13 @@ import time, logging from datetime import datetime import threading, collections, queue, os, os.path -import wave -import pyaudio -import webrtcvad -from halo import Halo import deepspeech import numpy as np +import pyaudio +import wave +import webrtcvad +from halo import Halo +from scipy import signal logging.basicConfig(level=20) @@ -14,28 +15,61 @@ class Audio(object): """Streams raw audio from microphone. Data is received in a separate thread, and stored in a buffer, to be read from.""" FORMAT = pyaudio.paInt16 - RATE = 16000 + # Network/VAD rate-space + RATE_PROCESS = 16000 CHANNELS = 1 BLOCKS_PER_SECOND = 50 - BLOCK_SIZE = int(RATE / float(BLOCKS_PER_SECOND)) - def __init__(self, callback=None): + def __init__(self, callback=None, device=None, input_rate=RATE_PROCESS): def proxy_callback(in_data, frame_count, time_info, status): callback(in_data) return (None, pyaudio.paContinue) if callback is None: callback = lambda in_data: self.buffer_queue.put(in_data) self.buffer_queue = queue.Queue() - self.sample_rate = self.RATE - self.block_size = self.BLOCK_SIZE + self.device = device + self.input_rate = input_rate + self.sample_rate = self.RATE_PROCESS + self.block_size = int(self.RATE_PROCESS / float(self.BLOCKS_PER_SECOND)) + self.block_size_input = int(self.input_rate / float(self.BLOCKS_PER_SECOND)) self.pa = pyaudio.PyAudio() - self.stream = self.pa.open(format=self.FORMAT, - channels=self.CHANNELS, - rate=self.sample_rate, - input=True, - frames_per_buffer=self.block_size, - stream_callback=proxy_callback) + + kwargs = { + 'format': self.FORMAT, + 'channels': self.CHANNELS, + 'rate': self.input_rate, + 'input': True, + 'frames_per_buffer': self.block_size_input, + 'stream_callback': proxy_callback, + } + + # if not default device + if self.device: + kwargs['input_device_index'] = self.device + + self.stream = self.pa.open(**kwargs) self.stream.start_stream() + def resample(self, data, input_rate): + """ + Microphone may not support our native processing sampling rate, so + resample from input_rate to RATE_PROCESS here for webrtcvad and + deepspeech + + Args: + data (binary): Input audio stream + input_rate (int): Input audio rate to resample from + """ + data16 = np.fromstring(string=data, dtype=np.int16) + resample_size = int(len(data16) / self.input_rate * self.RATE_PROCESS) + resample = signal.resample(data16, resample_size) + resample16 = np.array(resample, dtype=np.int16) + return resample16.tostring() + + def read_resampled(self): + """Return a block of audio data resampled to 16000hz, blocking if necessary.""" + return self.resample(data=self.buffer_queue.get(), + input_rate=self.input_rate) + def read(self): """Return a block of audio data, blocking if necessary.""" return self.buffer_queue.get() @@ -58,17 +92,22 @@ class Audio(object): wf.writeframes(data) wf.close() + class VADAudio(Audio): """Filter & segment audio with voice activity detection.""" - def __init__(self, aggressiveness=3): - super().__init__() + def __init__(self, aggressiveness=3, device=None, input_rate=None): + super().__init__(device=device, input_rate=input_rate) self.vad = webrtcvad.Vad(aggressiveness) def frame_generator(self): """Generator that yields all audio frames from microphone.""" - while True: - yield self.read() + if self.input_rate == self.RATE_PROCESS: + while True: + yield self.read() + else: + while True: + yield self.read_resampled() def vad_collector(self, padding_ms=300, ratio=0.75, frames=None): """Generator that yields series of consecutive audio frames comprising each utterence, separated by yielding a single None. @@ -121,7 +160,9 @@ def main(ARGS): model.enableDecoderWithLM(ARGS.alphabet, ARGS.lm, ARGS.trie, ARGS.lm_alpha, ARGS.lm_beta) # Start audio with VAD - vad_audio = VADAudio(aggressiveness=ARGS.vad_aggressiveness) + vad_audio = VADAudio(aggressiveness=ARGS.vad_aggressiveness, + device=ARGS.device, + input_rate=ARGS.rate) print("Listening (ctrl-C to exit)...") frames = vad_audio.vad_collector() @@ -148,6 +189,7 @@ def main(ARGS): if __name__ == '__main__': BEAM_WIDTH = 500 + DEFAULT_SAMPLE_RATE = 16000 LM_ALPHA = 0.75 LM_BETA = 1.85 N_FEATURES = 26 @@ -171,6 +213,10 @@ if __name__ == '__main__': help="Path to the language model binary file. Default: lm.binary") parser.add_argument('-t', '--trie', default='trie', help="Path to the language model trie file created with native_client/generate_trie. Default: trie") + parser.add_argument('-d', '--device', type=int, default=None, + help="Device input index (Int) as listed by pyaudio.PyAudio.get_device_info_by_index(). If not provided, falls back to PyAudio.get_default_device()") + parser.add_argument('-r', '--rate', type=int, default=DEFAULT_SAMPLE_RATE, + help=f"Input device sample rate. Default: {DEFAULT_SAMPLE_RATE}. Your device may require 44100.") parser.add_argument('-nf', '--n_features', type=int, default=N_FEATURES, help=f"Number of MFCC features to use. Default: {N_FEATURES}") parser.add_argument('-nc', '--n_context', type=int, default=N_CONTEXT, diff --git a/taskcluster/test-training_upstream-linux-amd64-py34m-opt.yml b/taskcluster/test-training_upstream-linux-amd64-py34m-opt.yml deleted file mode 100644 index 6f65f902..00000000 --- a/taskcluster/test-training_upstream-linux-amd64-py34m-opt.yml +++ /dev/null @@ -1,12 +0,0 @@ -build: - template_file: test-linux-opt-base.tyml - dependencies: - - "linux-amd64-ctc-opt" - system_setup: - > - apt-get -qq -y install ${python.packages_trusty.apt} - args: - tests_cmdline: "${system.homedir.linux}/DeepSpeech/ds/tc-train-tests.sh 3.4.8:m" - metadata: - name: "DeepSpeech Linux AMD64 CPU upstream training Py3.4" - description: "Training a DeepSpeech LDC93S1 model for Linux/AMD64 using upstream TensorFlow Python 3.4, CPU only, optimized version" diff --git a/tc-tests-utils.sh b/tc-tests-utils.sh index f9d94ea2..45c0a327 100755 --- a/tc-tests-utils.sh +++ b/tc-tests-utils.sh @@ -381,7 +381,7 @@ install_nuget() nuget install NAudio cp NAudio*/lib/net35/NAudio.dll ${TASKCLUSTER_TMP_DIR}/ds/ cp ${PROJECT_NAME}.${DS_VERSION}/build/libdeepspeech.so ${TASKCLUSTER_TMP_DIR}/ds/ - cp ${PROJECT_NAME}.${DS_VERSION}/lib/net462/DeepSpeechClient.dll ${TASKCLUSTER_TMP_DIR}/ds/ + cp ${PROJECT_NAME}.${DS_VERSION}/lib/net46/DeepSpeechClient.dll ${TASKCLUSTER_TMP_DIR}/ds/ ls -hal ${TASKCLUSTER_TMP_DIR}/ds/ @@ -616,21 +616,21 @@ do_deepspeech_netframework_build() /p:Configuration=Release \ /p:Platform=x64 \ /p:TargetFrameworkVersion="v4.5" \ - /p:OutputPath=bin/x64/Release/v4.5 + /p:OutputPath=bin/nuget/x64/v4.5 MSYS2_ARG_CONV_EXCL='/' "${MSBUILD}" \ DeepSpeechClient/DeepSpeechClient.csproj \ /p:Configuration=Release \ /p:Platform=x64 \ /p:TargetFrameworkVersion="v4.6" \ - /p:OutputPath=bin/x64/Release/v4.6 + /p:OutputPath=bin/nuget/x64/v4.6 MSYS2_ARG_CONV_EXCL='/' "${MSBUILD}" \ DeepSpeechClient/DeepSpeechClient.csproj \ /p:Configuration=Release \ /p:Platform=x64 \ /p:TargetFrameworkVersion="v4.7" \ - /p:OutputPath=bin/x64/Release/v4.7 + /p:OutputPath=bin/nuget/x64/v4.7 MSYS2_ARG_CONV_EXCL='/' "${MSBUILD}" \ DeepSpeechConsole/DeepSpeechConsole.csproj \ @@ -658,13 +658,13 @@ do_nuget_build() # We copy the generated clients for .NET into the Nuget framework dirs mkdir -p nupkg/lib/net45/ - cp DeepSpeechClient/bin/x64/Release/v4.5/DeepSpeechClient.dll nupkg/lib/net45/ + cp DeepSpeechClient/bin/nuget/x64/v4.5/DeepSpeechClient.dll nupkg/lib/net45/ mkdir -p nupkg/lib/net46/ - cp DeepSpeechClient/bin/x64/Release/v4.6/DeepSpeechClient.dll nupkg/lib/net46/ + cp DeepSpeechClient/bin/nuget/x64/v4.6/DeepSpeechClient.dll nupkg/lib/net46/ mkdir -p nupkg/lib/net47/ - cp DeepSpeechClient/bin/x64/Release/v4.7/DeepSpeechClient.dll nupkg/lib/net47/ + cp DeepSpeechClient/bin/nuget/x64/v4.7/DeepSpeechClient.dll nupkg/lib/net47/ PROJECT_VERSION=$(shell cat ../../../VERSION | tr -d '\n' | tr -d '\r') sed \