diff options
Diffstat (limited to 'client/mic.py')
| -rw-r--r-- | client/mic.py | 296 |
1 files changed, 296 insertions, 0 deletions
diff --git a/client/mic.py b/client/mic.py new file mode 100644 index 0000000..71f49ab --- /dev/null +++ b/client/mic.py @@ -0,0 +1,296 @@ +""" + The Mic class handles all interactions with the microphone and speaker. +""" + +import os +import json +from wave import open as open_audio +import audioop +import pyaudio +import alteration + + +# quirky bug where first import doesn't work +try: + import pocketsphinx as ps +except: + import pocketsphinx as ps + + +class Mic: + + speechRec = None + speechRec_persona = None + + def __init__(self, lmd, dictd, lmd_persona, dictd_persona, lmd_music=None, dictd_music=None): + """ + Initiates the pocketsphinx instance. + + Arguments: + lmd -- filename of the full language model + dictd -- filename of the full dictionary (.dic) + lmd_persona -- filename of the 'Persona' language model (containing, e.g., 'Jasper') + dictd_persona -- filename of the 'Persona' dictionary (.dic) + """ + + hmdir = "/usr/local/share/pocketsphinx/model/hmm/en_US/hub4wsj_sc_8k" + + if lmd_music and dictd_music: + self.speechRec_music = ps.Decoder(hmm = hmdir, lm = lmd_music, dict = dictd_music) + self.speechRec_persona = ps.Decoder( + hmm=hmdir, lm=lmd_persona, dict=dictd_persona) + self.speechRec = ps.Decoder(hmm=hmdir, lm=lmd, dict=dictd) + + def transcribe(self, audio_file_path, PERSONA_ONLY=False, MUSIC=False): + """ + Performs TTS, transcribing an audio file and returning the result. + + Arguments: + audio_file_path -- the path to the audio file to-be transcribed + PERSONA_ONLY -- if True, uses the 'Persona' language model and dictionary + MUSIC -- if True, uses the 'Music' language model and dictionary + """ + + wavFile = file(audio_file_path, 'rb') + wavFile.seek(44) + + if MUSIC: + self.speechRec_music.decode_raw(wavFile) + result = self.speechRec_music.get_hyp() + elif PERSONA_ONLY: + self.speechRec_persona.decode_raw(wavFile) + result = self.speechRec_persona.get_hyp() + else: + self.speechRec.decode_raw(wavFile) + result = self.speechRec.get_hyp() + + print "===================" + print "JASPER: " + result[0] + print "===================" + + return result[0] + + def getScore(self, data): + rms = audioop.rms(data, 2) + score = rms / 3 + return score + + def fetchThreshold(self): + + # TODO: Consolidate all of these variables from the next three + # functions + THRESHOLD_MULTIPLIER = 1.8 + AUDIO_FILE = "passive.wav" + RATE = 16000 + CHUNK = 1024 + + # number of seconds to allow to establish threshold + THRESHOLD_TIME = 1 + + # number of seconds to listen before forcing restart + LISTEN_TIME = 10 + + # prepare recording stream + audio = pyaudio.PyAudio() + stream = audio.open(format=pyaudio.paInt16, + channels=1, + rate=RATE, + input=True, + frames_per_buffer=CHUNK) + + # stores the audio data + frames = [] + + # stores the lastN score values + lastN = [i for i in range(20)] + + # calculate the long run average, and thereby the proper threshold + for i in range(0, RATE / CHUNK * THRESHOLD_TIME): + + data = stream.read(CHUNK) + frames.append(data) + + # save this data point as a score + lastN.pop(0) + lastN.append(self.getScore(data)) + average = sum(lastN) / len(lastN) + + # this will be the benchmark to cause a disturbance over! + THRESHOLD = average * THRESHOLD_MULTIPLIER + + return THRESHOLD + + def passiveListen(self, PERSONA): + """ + Listens for PERSONA in everyday sound + Times out after LISTEN_TIME, so needs to be restarted + """ + + THRESHOLD_MULTIPLIER = 1.8 + AUDIO_FILE = "passive.wav" + RATE = 16000 + CHUNK = 1024 + + # number of seconds to allow to establish threshold + THRESHOLD_TIME = 1 + + # number of seconds to listen before forcing restart + LISTEN_TIME = 10 + + # prepare recording stream + audio = pyaudio.PyAudio() + stream = audio.open(format=pyaudio.paInt16, + channels=1, + rate=RATE, + input=True, + frames_per_buffer=CHUNK) + + # stores the audio data + frames = [] + + # stores the lastN score values + lastN = [i for i in range(30)] + + # calculate the long run average, and thereby the proper threshold + for i in range(0, RATE / CHUNK * THRESHOLD_TIME): + + data = stream.read(CHUNK) + frames.append(data) + + # save this data point as a score + lastN.pop(0) + lastN.append(self.getScore(data)) + average = sum(lastN) / len(lastN) + + # this will be the benchmark to cause a disturbance over! + THRESHOLD = average * THRESHOLD_MULTIPLIER + + # save some memory for sound data + frames = [] + + # flag raised when sound disturbance detected + didDetect = False + + # start passively listening for disturbance above threshold + for i in range(0, RATE / CHUNK * LISTEN_TIME): + + data = stream.read(CHUNK) + frames.append(data) + score = self.getScore(data) + + if score > THRESHOLD: + didDetect = True + break + + # no use continuing if no flag raised + if not didDetect: + print "No disturbance detected" + return + + # cutoff any recording before this disturbance was detected + frames = frames[-20:] + + # otherwise, let's keep recording for few seconds and save the file + DELAY_MULTIPLIER = 1 + for i in range(0, RATE / CHUNK * DELAY_MULTIPLIER): + + data = stream.read(CHUNK) + frames.append(data) + + # save the audio data + stream.stop_stream() + stream.close() + audio.terminate() + write_frames = open_audio(AUDIO_FILE, 'wb') + write_frames.setnchannels(1) + write_frames.setsampwidth(audio.get_sample_size(pyaudio.paInt16)) + write_frames.setframerate(RATE) + write_frames.writeframes(''.join(frames)) + write_frames.close() + + # check if PERSONA was said + transcribed = self.transcribe(AUDIO_FILE, PERSONA_ONLY=True) + + if PERSONA in transcribed: + return (THRESHOLD, PERSONA) + + return (False, transcribed) + + def activeListen(self, THRESHOLD=None, LISTEN=True, MUSIC=False): + """ + Records until a second of silence or times out after 12 seconds + """ + + AUDIO_FILE = "active.wav" + RATE = 16000 + CHUNK = 1024 + LISTEN_TIME = 12 + + # user can request pre-recorded sound + if not LISTEN: + if not os.path.exists(AUDIO_FILE): + return None + + return self.transcribe(AUDIO_FILE) + + # check if no threshold provided + if THRESHOLD == None: + THRESHOLD = self.fetchThreshold() + + os.system("aplay -D hw:1,0 beep_hi.wav") + + # prepare recording stream + audio = pyaudio.PyAudio() + stream = audio.open(format=pyaudio.paInt16, + channels=1, + rate=RATE, + input=True, + frames_per_buffer=CHUNK) + + frames = [] + # increasing the range # results in longer pause after command + # generation + lastN = [THRESHOLD * 1.2 for i in range(30)] + + for i in range(0, RATE / CHUNK * LISTEN_TIME): + + data = stream.read(CHUNK) + frames.append(data) + score = self.getScore(data) + + lastN.pop(0) + lastN.append(score) + + average = sum(lastN) / float(len(lastN)) + + # TODO: 0.8 should not be a MAGIC NUMBER! + if average < THRESHOLD * 0.8: + break + + os.system("aplay -D hw:1,0 beep_lo.wav") + + # save the audio data + stream.stop_stream() + stream.close() + audio.terminate() + write_frames = open_audio(AUDIO_FILE, 'wb') + write_frames.setnchannels(1) + write_frames.setsampwidth(audio.get_sample_size(pyaudio.paInt16)) + write_frames.setframerate(RATE) + write_frames.writeframes(''.join(frames)) + write_frames.close() + + # DO SOME AMPLIFICATION + # os.system("sox "+AUDIO_FILE+" temp.wav vol 20dB") + + if MUSIC: + return self.transcribe(AUDIO_FILE, MUSIC=True) + + return self.transcribe(AUDIO_FILE) + + def say(self, phrase, OPTIONS=" -vdefault+m3 -p 40 -s 160 --stdout > say.wav"): + # alter phrase before speaking + phrase = alteration.clean(phrase) + + os.system("espeak " + json.dumps(phrase) + OPTIONS) + os.system("aplay -D hw:1,0 say.wav") |
