diff options
| author | schneefux <schneefux+commit@schneefux.xyz> | 2014-10-13 11:30:26 +0200 |
|---|---|---|
| committer | schneefux <schneefux+commit@schneefux.xyz> | 2014-10-13 11:30:26 +0200 |
| commit | 722ac02441b7e9978f908f6670e1fae731cbe6bb (patch) | |
| tree | 6c90ed884f7038aba574460054aebc5c2c4d6d94 /client | |
| parent | e0e2708dd240dbda6e6690b38a20ca418e5e77a8 (diff) | |
| parent | ac0a6d731ad179d62bf2be6be28ee862a020bdab (diff) | |
| download | jasper-client-722ac02441b7e9978f908f6670e1fae731cbe6bb.tar.gz jasper-client-722ac02441b7e9978f908f6670e1fae731cbe6bb.zip | |
Merge pull request #181 from Holzhaus/vocabcompiler-abstraction
Rewrite of vocabcompiler
Diffstat (limited to 'client')
| -rw-r--r-- | client/g2p.py | 168 | ||||
| -rw-r--r-- | client/modules/MPDControl.py | 42 | ||||
| -rw-r--r-- | client/requirements.txt | 1 | ||||
| -rw-r--r-- | client/stt.py | 88 | ||||
| -rw-r--r-- | client/test.py | 212 | ||||
| -rw-r--r-- | client/vocabcompiler.py | 425 |
6 files changed, 758 insertions, 178 deletions
diff --git a/client/g2p.py b/client/g2p.py index 89f2282..7b03386 100644 --- a/client/g2p.py +++ b/client/g2p.py @@ -1,69 +1,153 @@ # -*- coding: utf-8-*- import os -import tempfile -import subprocess import re +import subprocess +import tempfile +import logging + import yaml +import diagnose import jasperpath -PHONE_MATCH = re.compile(r'<s> (.*) </s>') -FST_MODEL = None +class PhonetisaurusG2P(object): + PATTERN = re.compile(r'^(?P<word>.+)\t(?P<precision>\d+\.\d+)\t<s> ' + + r'(?P<pronounciation>.*) </s>', re.MULTILINE) + + @classmethod + def execute(cls, fst_model, input, is_file=False, nbest=None): + logger = logging.getLogger(__name__) -# Try to get fst_model from config -profile_path = os.path.join(os.path.dirname(__file__), 'profile.yml') -if os.path.exists(profile_path): - with open(profile_path, 'r') as f: - profile = yaml.safe_load(f) - if ('pocketsphinx' in profile and - 'fst_model' in profile['pocketsphinx']): - FST_MODEL = profile['pocketsphinx']['fst_model'] + cmd = ['phonetisaurus-g2p', + '--model=%s' % fst_model, + '--input=%s' % input, + '--words'] -if not FST_MODEL: - FST_MODEL = os.path.join(jasperpath.APP_PATH, os.pardir, 'phonetisaurus', - 'g014b2b.fst') + if is_file: + cmd.append('--isfile') + if nbest is not None: + cmd.extend(['--nbest=%d' % nbest]) -def parseLine(line): - return PHONE_MATCH.search(line).group(1) + cmd = [str(x) for x in cmd] + try: + # FIXME: We can't just use subprocess.call and redirect stdout + # and stderr, because it looks like Phonetisaurus can't open + # an already opened file descriptor a second time. This is why + # we have to use this somehow hacky subprocess.Popen approach. + proc = subprocess.Popen(cmd, stdout=subprocess.PIPE, + stderr=subprocess.PIPE) + stdoutdata, stderrdata = proc.communicate() + if proc.returncode != 0: + logger.warning("Command '%s' return with exit status %d", + ' '.join(cmd), proc.returncode) + except OSError: + logger.error("Error occured while executing command '%s'", + ' '.join(cmd), exc_info=True) + stdoutdata, stderrdata = None, None + if stderrdata is not None: + for line in stderrdata.splitlines(): + message = line.strip() + if message: + logger.debug(message) + result = {} + if stdoutdata is not None: + for word, precision, pronounc in cls.PATTERN.findall(stdoutdata): + if word not in result: + result[word] = [] + result[word].append(pronounc) + return result -def parseOutput(output): - return PHONE_MATCH.findall(output) + @classmethod + def get_config(cls): + # FIXME: Replace this as soon as pull request + # jasperproject/jasper-client#128 has been merged + conf = {'fst_model': os.path.join(jasperpath.APP_PATH, os.pardir, + 'phonetisaurus', 'g014b2b.fst')} + # Try to get fst_model from config + profile_path = jasperpath.config('profile.yml') + if os.path.exists(profile_path): + with open(profile_path, 'r') as f: + profile = yaml.safe_load(f) + if 'pocketsphinx' in profile: + if 'fst_model' in profile['pocketsphinx']: + conf['fst_model'] = \ + profile['pocketsphinx']['fst_model'] + if 'nbest' in profile['pocketsphinx']: + conf['nbest'] = int(profile['pocketsphinx']['nbest']) + return conf -def translateWord(word): - out = subprocess.check_output( - ['phonetisaurus-g2p', '--model=%s' % FST_MODEL, '--input=%s' % word]) - return parseLine(out) + def __new__(cls, fst_model=None, *args, **kwargs): + if not diagnose.check_executable('phonetisaurus-g2p'): + raise OSError("Can't find command 'phonetisaurus-g2p'! Please " + + "check if Phonetisaurus is installed and in your " + + "$PATH.") + if fst_model is None or not os.access(fst_model, os.R_OK): + raise OSError("FST model '%r' does not exist! Can't create " + + "instance." % fst_model) + inst = object.__new__(cls, fst_model, *args, **kwargs) + return inst + def __init__(self, fst_model=None, nbest=None): + self._logger = logging.getLogger(__name__) -def translateWords(words): - full_text = '\n'.join(words) + self.fst_model = os.path.abspath(fst_model) + self._logger.debug("Using FST model: '%s'", self.fst_model) - with tempfile.NamedTemporaryFile(suffix='.g2p', delete=False) as f: - temp_filename = f.name - f.write(full_text) + self.nbest = nbest + if self.nbest is not None: + self._logger.debug("Will use the %d best results.", self.nbest) - output = translateFile(temp_filename) - os.remove(temp_filename) + def _translate_word(self, word): + return self.execute(self.fst_model, word, nbest=self.nbest) - return output + def _translate_words(self, words): + with tempfile.NamedTemporaryFile(suffix='.g2p', delete=False) as f: + # The 'delete=False' kwarg is kind of a hack, but Phonetisaurus + # won't work if we remove it, because it seems that I can't open + # a file descriptor a second time. + for word in words: + f.write("%s\n" % word) + tmp_fname = f.name + output = self.execute(self.fst_model, tmp_fname, is_file=True, + nbest=self.nbest) + os.remove(tmp_fname) + return output + def translate(self, words): + if type(words) is str or len(words) == 1: + self._logger.debug('Converting single word to phonemes') + output = self._translate_word(words if type(words) is str + else words[0]) + else: + self._logger.debug('Converting %d words to phonemes', len(words)) + output = self._translate_words(words) + self._logger.debug('G2P conversion returned phonemes for %d words', + len(output)) + return output -def translateFile(input_filename, output_filename=None): - out = subprocess.check_output( - ['phonetisaurus-g2p', '--model=%s' % FST_MODEL, - '--input=%s' % input_filename, '--words', '--isfile']) - out = parseOutput(out) +if __name__ == "__main__": + import pprint + import argparse + parser = argparse.ArgumentParser(description='Phonetisaurus G2P module') + parser.add_argument('fst_model', action='store', + help='Path to the FST Model') + parser.add_argument('--debug', action='store_true', + help='Show debug messages') + args = parser.parse_args() - if output_filename: - out = '\n'.join(out) + logging.basicConfig() + logger = logging.getLogger() + if args.debug: + logger.setLevel(logging.DEBUG) - with open(output_filename, "wb") as f: - f.write(out) + words = ['THIS', 'IS', 'A', 'TEST'] - return None + g2pconv = PhonetisaurusG2P(args.fst_model, nbest=3) + output = g2pconv.translate(words) - return out + pp = pprint.PrettyPrinter(indent=2) + pp.pprint(output) diff --git a/client/modules/MPDControl.py b/client/modules/MPDControl.py index eefdcbd..f23fc7d 100644 --- a/client/modules/MPDControl.py +++ b/client/modules/MPDControl.py @@ -1,11 +1,11 @@ # -*- coding: utf-8-*- -import os import re import logging import difflib import mpd -import g2p import stt +import vocabcompiler +import jasperpath from mic import Mic # Standard module stuff @@ -73,34 +73,22 @@ class MusicMode(object): self.music = mpdwrapper # index spotify playlists into new dictionary and language models - original = ["STOP", "CLOSE", "PLAY", "PAUSE", "NEXT", "PREVIOUS", - "LOUDER", "SOFTER", "LOWER", "HIGHER", "VOLUME", - "PLAYLIST"] + self.music.get_soup_playlist() - pronounced = g2p.translateWords(original) - zipped = zip(original, pronounced) - lines = ["%s %s" % (x, y) for x, y in zipped] + phrases = ["STOP", "CLOSE", "PLAY", "PAUSE", "NEXT", "PREVIOUS", + "LOUDER", "SOFTER", "LOWER", "HIGHER", "VOLUME", + "PLAYLIST"] + phrases.extend(self.music.get_soup_playlist()) - with open("dictionary_spotify.dic", "w") as f: - f.write("\n".join(lines) + "\n") - - with open("sentences_spotify.txt", "w") as f: - f.write("\n".join(original) + "\n") - f.write("<s> \n </s> \n") - f.close() - - # make language model - os.system("text2idngram -vocab sentences_spotify.txt < " + - "sentences_spotify.txt -idngram spotify.idngram") - os.system("idngram2lm -idngram spotify.idngram -vocab " + - "sentences_spotify.txt -arpa languagemodel_spotify.lm") + vocabulary_music = vocabcompiler.PocketsphinxVocabulary( + name='music', path=jasperpath.config('vocabularies')) + vocabulary_music.compile(phrases) # create a new mic with the new music models - self.mic = Mic( - mic.speaker, - mic.passive_stt_engine, - stt.PocketSphinxSTT(lmd_music="languagemodel_spotify.lm", - dictd_music="dictionary_spotify.dic") - ) + config = stt.PocketSphinxSTT.get_config() + + self.mic = Mic(mic.speaker, + mic.passive_stt_engine, + stt.PocketSphinxSTT(vocabulary_music=vocabulary_music, + **config)) def delegateInput(self, input): diff --git a/client/requirements.txt b/client/requirements.txt index dd0e474..f3508fe 100644 --- a/client/requirements.txt +++ b/client/requirements.txt @@ -10,3 +10,4 @@ python-mpd==0.3.0 pytz==2013b semantic==1.0.3 requests==2.1.0 +cmuclmtk==0.1.5 diff --git a/client/stt.py b/client/stt.py index cb06117..9ac1a18 100644 --- a/client/stt.py +++ b/client/stt.py @@ -11,6 +11,7 @@ import requests import yaml import jasperpath import diagnose +import vocabcompiler class TranscriptionMode: @@ -45,23 +46,19 @@ class PocketSphinxSTT(AbstractSTTEngine): SLUG = 'sphinx' - def __init__(self, lmd=jasperpath.config("languagemodel.lm"), - dictd=jasperpath.config("dictionary.dic"), - lmd_persona=jasperpath.data("languagemodel_persona.lm"), - dictd_persona=jasperpath.data("dictionary_persona.dic"), - lmd_music=None, dictd_music=None, - hmm_dir="/usr/local/share/pocketsphinx/model/hmm/en_US/" + - "hub4wsj_sc_8k"): + def __init__(self, vocabulary=None, vocabulary_keyword=None, + vocabulary_music=None, hmm_dir="/usr/local/share/" + + "pocketsphinx/model/hmm/en_US/hub4wsj_sc_8k"): + """ Initiates the pocketsphinx instance. Arguments: - speaker -- handles platform-independent audio output - lmd -- filename of the full language model - dictd -- filename of the full dictionary (.dic) - lmd_persona -- filename of the 'Persona' language model (containing, - e.g., 'Jasper') - dictd_persona -- filename of the 'Persona' dictionary (.dic) + vocabulary -- a PocketsphinxVocabulary instance + vocabulary_keyword -- a PocketsphinxVocabulary instance + (containing, e.g., 'Jasper') + vocabulary_music -- (optional) a PocketsphinxVocabulary instance + hmm_dir -- the path of the Hidden Markov Model (HMM) """ self._logger = logging.getLogger(__name__) @@ -84,16 +81,19 @@ class PocketSphinxSTT(AbstractSTTEngine): self._logfiles[TranscriptionMode.NORMAL] = f.name self._decoders = {} - if lmd_music and dictd_music: + if vocabulary_music is not None: self._decoders[TranscriptionMode.MUSIC] = \ - ps.Decoder(hmm=hmm_dir, lm=lmd_music, dict=dictd_music, - logfn=self._logfiles[TranscriptionMode.MUSIC]) + ps.Decoder(hmm=hmm_dir, + logfn=self._logfiles[TranscriptionMode.MUSIC], + **vocabulary_music.decoder_kwargs) self._decoders[TranscriptionMode.KEYWORD] = \ - ps.Decoder(hmm=hmm_dir, lm=lmd_persona, dict=dictd_persona, - logfn=self._logfiles[TranscriptionMode.KEYWORD]) + ps.Decoder(hmm=hmm_dir, + logfn=self._logfiles[TranscriptionMode.KEYWORD], + **vocabulary_keyword.decoder_kwargs) self._decoders[TranscriptionMode.NORMAL] = \ - ps.Decoder(hmm=hmm_dir, lm=lmd, dict=dictd, - logfn=self._logfiles[TranscriptionMode.NORMAL]) + ps.Decoder(hmm=hmm_dir, + logfn=self._logfiles[TranscriptionMode.NORMAL], + **vocabulary.decoder_kwargs) def __del__(self): for filename in self._logfiles.values(): @@ -106,27 +106,45 @@ class PocketSphinxSTT(AbstractSTTEngine): # HMM dir # Try to get hmm_dir from config profile_path = os.path.join(os.path.dirname(__file__), 'profile.yml') + + name_default = 'default' + path_default = jasperpath.config('vocabularies') + + name_keyword = 'keyword' + path_keyword = jasperpath.config('vocabularies') + if os.path.exists(profile_path): with open(profile_path, 'r') as f: profile = yaml.safe_load(f) if 'pocketsphinx' in profile: if 'hmm_dir' in profile['pocketsphinx']: config['hmm_dir'] = profile['pocketsphinx']['hmm_dir'] - if 'lmd' in profile['pocketsphinx']: - config['lmd'] = profile['pocketsphinx']['lmd'] - if 'dictd' in profile['pocketsphinx']: - config['dictd'] = profile['pocketsphinx']['dictd'] - if 'lmd_persona' in profile['pocketsphinx']: - config['lmd_persona'] = \ - profile['pocketsphinx']['lmd_persona'] - if 'dictd_persona' in profile['pocketsphinx']: - config['dictd_persona'] = \ - profile['pocketsphinx']['dictd_persona'] - if 'lmd_music' in profile['pocketsphinx']: - config['lmd'] = profile['pocketsphinx']['lmd_music'] - if 'dictd_music' in profile['pocketsphinx']: - config['dictd_music'] = \ - profile['pocketsphinx']['dictd_music'] + + if 'vocabulary_default_name' in profile['pocketsphinx']: + name_default = \ + profile['pocketsphinx']['vocabulary_default_name'] + + if 'vocabulary_default_path' in profile['pocketsphinx']: + path_default = \ + profile['pocketsphinx']['vocabulary_default_path'] + + if 'vocabulary_keyword_name' in profile['pocketsphinx']: + name_keyword = \ + profile['pocketsphinx']['vocabulary_keyword_name'] + + if 'vocabulary_keyword_path' in profile['pocketsphinx']: + path_keyword = \ + profile['pocketsphinx']['vocabulary_keyword_path'] + + config['vocabulary'] = vocabcompiler.PocketsphinxVocabulary( + name_default, path=path_default) + config['vocabulary_keyword'] = vocabcompiler.PocketsphinxVocabulary( + name_keyword, path=path_keyword) + + config['vocabulary'].compile(vocabcompiler.get_all_phrases()) + config['vocabulary_keyword'].compile( + vocabcompiler.get_keyword_phrases()) + return config def transcribe(self, fp, mode=TranscriptionMode.NORMAL): diff --git a/client/test.py b/client/test.py index 6858700..0a414f4 100644 --- a/client/test.py +++ b/client/test.py @@ -4,6 +4,9 @@ import os import sys import unittest import logging +import tempfile +import shutil +import contextlib import argparse from mock import patch, Mock @@ -24,39 +27,131 @@ DEFAULT_PROFILE = { } -class UnorderedList(list): +class TestVocabCompiler(unittest.TestCase): - def __eq__(self, other): - return sorted(self) == sorted(other) + def testPhraseExtraction(self): + expected_phrases = ['MOCK'] + mock_module = Mock() + mock_module.WORDS = ['MOCK'] -class TestVocabCompiler(unittest.TestCase): + with patch.object(brain.Brain, 'get_modules', + classmethod(lambda cls: [mock_module])): + extracted_phrases = vocabcompiler.get_all_phrases() + self.assertEqual(expected_phrases, extracted_phrases) - def testWordExtraction(self): - sentences = "temp_sentences.txt" - dictionary = "temp_dictionary.dic" - languagemodel = "temp_languagemodel.lm" + def testKeywordPhraseExtraction(self): + expected_phrases = ['MOCK'] - words = [] + with tempfile.TemporaryFile() as f: + # We can't use mock_open here, because it doesn't seem to work + # with the 'for line in f' syntax + f.write("MOCK\n") + f.seek(0) + with patch('%s.open' % vocabcompiler.__name__, + return_value=f, create=True): + extracted_phrases = vocabcompiler.get_keyword_phrases() + self.assertEqual(expected_phrases, extracted_phrases) + + +class TestVocabulary(unittest.TestCase): + VOCABULARY = vocabcompiler.DummyVocabulary + + @contextlib.contextmanager + def do_in_tempdir(self): + tempdir = tempfile.mkdtemp() + yield tempdir + shutil.rmtree(tempdir) + + def testVocabulary(self): + phrases = ['GOOD BAD UGLY'] + with self.do_in_tempdir() as tempdir: + self.vocab = self.VOCABULARY(path=tempdir) + self.assertIsNone(self.vocab.compiled_revision) + self.assertFalse(self.vocab.is_compiled) + self.assertFalse(self.vocab.matches_phrases(phrases)) + + # We're now testing error handling. To avoid flooding the + # output with error messages that are catched anyway, + # we'll temporarly disable logging. Otherwise, error log + # messages and traceback would be printed so that someone + # might think that tests failed even though they succeeded. + logging.disable(logging.ERROR) + with self.assertRaises(OSError): + with patch('os.makedirs', side_effect=OSError('test')): + self.vocab.compile(phrases) + with self.assertRaises(OSError): + with patch('%s.open' % vocabcompiler.__name__, + create=True, + side_effect=OSError('test')): + self.vocab.compile(phrases) + + class StrangeCompilationError(Exception): + pass + with patch.object(self.vocab, '_compile_vocabulary', + side_effect=StrangeCompilationError('test')): + with self.assertRaises(StrangeCompilationError): + self.vocab.compile(phrases) + with self.assertRaises(StrangeCompilationError): + with patch('os.remove', + side_effect=OSError('test')): + self.vocab.compile(phrases) + # Re-enable logging again + logging.disable(logging.NOTSET) + + self.vocab.compile(phrases) + self.assertIsInstance(self.vocab.compiled_revision, str) + self.assertTrue(self.vocab.is_compiled) + self.assertTrue(self.vocab.matches_phrases(phrases)) + self.vocab.compile(phrases) + self.vocab.compile(phrases, force=True) + + +class TestPocketsphinxVocabulary(TestVocabulary): + + VOCABULARY = vocabcompiler.PocketsphinxVocabulary + + def testVocabulary(self): + super(TestPocketsphinxVocabulary, self).testVocabulary() + self.assertIsInstance(self.vocab.decoder_kwargs, dict) + self.assertIn('lm', self.vocab.decoder_kwargs) + self.assertIn('dict', self.vocab.decoder_kwargs) - mock_module = Mock() - mock_module.WORDS = [ - 'MOCK' - ] - words.extend(mock_module.WORDS) +class TestPatchedPocketsphinxVocabulary(TestPocketsphinxVocabulary): - with patch.object(g2p, 'translateWords') as translateWords: - with patch.object(vocabcompiler, 'text2lm') as text2lm: - with patch.object(brain.Brain, 'get_modules', - classmethod(lambda cls: [mock_module])): - vocabcompiler.compile(sentences, dictionary, languagemodel) + def testVocabulary(self): - translateWords.assert_called_once_with( - UnorderedList(words)) - self.assertTrue(text2lm.called) - os.remove(sentences) - os.remove(dictionary) + def write_test_vocab(text, output_file): + with open(output_file, "w") as f: + for word in text.split(' '): + f.write("%s\n" % word) + + def write_test_lm(text, output_file, **kwargs): + with open(output_file, "w") as f: + f.write("TEST") + + class DummyG2P(object): + def __init__(self, *args, **kwargs): + pass + + @classmethod + def get_config(self, *args, **kwargs): + return {} + + def translate(self, *args, **kwargs): + return {'GOOD': ['G UH D', + 'G UW D'], + 'BAD': ['B AE D'], + 'UGLY': ['AH G L IY']} + + with patch('vocabcompiler.cmuclmtk', + create=True) as mocked_cmuclmtk: + mocked_cmuclmtk.text2vocab = write_test_vocab + mocked_cmuclmtk.text2lm = write_test_lm + with patch('vocabcompiler.PhonetisaurusG2P', DummyG2P): + super(TestPatchedPocketsphinxVocabulary, + self).testVocabulary() class TestMic(unittest.TestCase): @@ -66,7 +161,7 @@ class TestMic(unittest.TestCase): self.time_clip = jasperpath.data('audio', 'time.wav') from stt import PocketSphinxSTT - self.stt = PocketSphinxSTT() + self.stt = PocketSphinxSTT(**PocketSphinxSTT.get_config()) def testTranscribeJasper(self): """ @@ -89,22 +184,55 @@ class TestMic(unittest.TestCase): class TestG2P(unittest.TestCase): def setUp(self): - self.translations = { - 'GOOD': 'G UH D', - 'BAD': 'B AE D', - 'UGLY': 'AH G L IY' - } + self.g2pconverter = g2p.PhonetisaurusG2P( + **g2p.PhonetisaurusG2P.get_config()) + self.words = ['GOOD', 'BAD', 'UGLY'] + + def testTranslateWord(self): + for word in self.words: + self.assertIn(word, self.g2pconverter.translate(word).keys()) + + def testTranslateWords(self): + results = self.g2pconverter.translate(self.words).keys() + for word in self.words: + self.assertIn(word, results) + + +class TestPatchedG2P(TestG2P): + class DummyProc(object): + def __init__(self, *args, **kwargs): + self.returncode = 0 + + def communicate(self): + return ("GOOD\t9.20477\t<s> G UH D </s>\n" + + "GOOD\t14.4036\t<s> G UW D </s>\n" + + "GOOD\t16.0258\t<s> G UH D IY </s>\n" + + "BAD\t0.7416\t<s> B AE D </s>\n" + + "BAD\t12.5495\t<s> B AA D </s>\n" + + "BAD\t13.6745\t<s> B AH D </s>\n" + + "UGLY\t12.572\t<s> AH G L IY </s>\n" + + "UGLY\t17.9278\t<s> Y UW G L IY </s>\n" + + "UGLY\t18.9617\t<s> AH G L AY </s>\n", "") + + def setUp(self): + with patch('g2p.diagnose.check_executable', + return_value=True): + with tempfile.NamedTemporaryFile() as f: + conf = g2p.PhonetisaurusG2P.get_config().items() + with patch.object(g2p.PhonetisaurusG2P, 'get_config', + classmethod(lambda cls: dict( + conf + [('fst_model', f.name)]))): + super(self.__class__, self).setUp() def testTranslateWord(self): - for word in self.translations: - translation = self.translations[word] - self.assertEqual(g2p.translateWord(word), translation) + with patch('subprocess.Popen', + return_value=TestPatchedG2P.DummyProc()): + super(self.__class__, self).testTranslateWord() def testTranslateWords(self): - words = self.translations.keys() - # preserve ordering - translations = [self.translations[w] for w in words] - self.assertEqual(g2p.translateWords(words), translations) + with patch('subprocess.Popen', + return_value=TestPatchedG2P.DummyProc()): + super(self.__class__, self).testTranslateWords() class TestDiagnose(unittest.TestCase): @@ -275,10 +403,14 @@ if __name__ == '__main__': # Change CWD to jasperpath.LIB_PATH os.chdir(jasperpath.LIB_PATH) - test_cases = [TestBrain, TestModules, TestVocabCompiler, TestTTS, - TestDiagnose] - if not args.light: + test_cases = [TestBrain, TestModules, TestDiagnose, TestTTS, + TestVocabCompiler, TestVocabulary] + if args.light: + test_cases.append(TestPatchedG2P) + test_cases.append(TestPatchedPocketsphinxVocabulary) + else: test_cases.append(TestG2P) + test_cases.append(TestPocketsphinxVocabulary) test_cases.append(TestMic) suite = unittest.TestSuite() diff --git a/client/vocabcompiler.py b/client/vocabcompiler.py index 52b1f0a..7eef5c0 100644 --- a/client/vocabcompiler.py +++ b/client/vocabcompiler.py @@ -1,56 +1,413 @@ # -*- coding: utf-8-*- """ Iterates over all the WORDS variables in the modules and creates a -dictionary for the client. +vocabulary for the respective stt_engine if needed. """ import os -import g2p -from brain import Brain +import tempfile +import logging +import hashlib +from abc import ABCMeta, abstractmethod, abstractproperty +import brain +import jasperpath -def text2lm(in_filename, out_filename): - """Wrapper around the language model compilation tools""" - def text2idngram(in_filename, out_filename): - cmd = "text2idngram -vocab %s < %s -idngram temp.idngram" % ( - out_filename, in_filename) - os.system(cmd) +from g2p import PhonetisaurusG2P +try: + import cmuclmtk +except ImportError: + logging.getLogger(__name__).error("Error importing CMUCLMTK module. " + + "PocketsphinxVocabulary will not work " + + "correctly.", exc_info=True) - def idngram2lm(in_filename, out_filename): - cmd = "idngram2lm -idngram temp.idngram -vocab %s -arpa %s" % ( - in_filename, out_filename) - os.system(cmd) - text2idngram(in_filename, in_filename) - idngram2lm(in_filename, out_filename) +class AbstractVocabulary(object): + """ + Abstract base class for Vocabulary classes. + + Please note that subclasses have to implement the compile_vocabulary() + method and set a string as the PATH_PREFIX class attribute. + """ + __metaclass__ = ABCMeta + + @classmethod + def phrases_to_revision(self, phrases): + """ + Calculates a revision from phrases by using the SHA1 hash function. + + Arguments: + phrases -- a list of phrases + + Returns: + A revision string for given phrases. + """ + sorted_phrases = sorted(phrases) + joined_phrases = '\n'.join(sorted_phrases) + sha1 = hashlib.sha1() + sha1.update(joined_phrases) + return sha1.hexdigest() + + def __init__(self, name='default', path='.'): + """ + Initializes a new Vocabulary instance. + + Optional Arguments: + name -- (optional) the name of the vocabulary (Default: 'default') + path -- (optional) the path in which the vocabulary exists or will + be created (Default: '.') + """ + self.name = name + self.path = os.path.abspath(os.path.join(path, self.PATH_PREFIX, name)) + self._logger = logging.getLogger(__name__) + + @property + def revision_file(self): + """ + Returns: + The path of the the revision file as string + """ + return os.path.join(self.path, 'revision') + + @abstractproperty + def is_compiled(self): + """ + Checks if the vocabulary is compiled by checking if the revision file + is readable. This method should be overridden by subclasses to check + for class-specific additional files, too. + + Returns: + True if the dictionary is compiled, else False + """ + return os.access(self.revision_file, os.R_OK) + + @property + def compiled_revision(self): + """ + Reads the compiled revision from the revision file. + + Returns: + the revision of this vocabulary (i.e. the string + inside the revision file), or None if is_compiled + if False + """ + if not self.is_compiled: + return None + with open(self.revision_file, 'r') as f: + revision = f.read().strip() + self._logger.debug("compiled_revision is '%s'", revision) + return revision + + def matches_phrases(self, phrases): + """ + Convenience method to check if this vocabulary exactly contains the + phrases passed to this method. + + Arguments: + phrases -- a list of phrases + + Returns: + True if phrases exactly matches the phrases inside this + vocabulary. + + """ + return (self.compiled_revision == self.phrases_to_revision(phrases)) + + def compile(self, phrases, force=False): + """ + Compiles this vocabulary. If the force argument is True, compilation + will be forced regardless of necessity (which means that the + preliminary check if the current revision already equals the + revision after compilation will be skipped). + This method is not meant to be overridden by subclasses - use the + _compile_vocabulary()-method instead. + + Arguments: + phrases -- a list of phrases that this vocabulary will contain + force -- (optional) forces compilation (Default: False) + + Returns: + The revision of the compiled vocabulary + """ + revision = self.phrases_to_revision(phrases) + if not force and self.compiled_revision == revision: + self._logger.debug('Compilation not neccessary, compiled ' + + 'version matches phrases.') + return revision + + if not os.path.exists(self.path): + self._logger.debug("Vocabulary dir '%s' does not exist, " + + "creating...", self.path) + try: + os.makedirs(self.path) + except OSError: + self._logger.error("Couldn't create vocabulary dir '%s'", + self.path, exc_info=True) + raise + try: + with open(self.revision_file, 'w') as f: + f.write(revision) + except (OSError, IOError): + self._logger.error("Couldn't write revision file in '%s'", + self.revision_file, exc_info=True) + raise + else: + self._logger.info('Starting compilation...') + try: + self._compile_vocabulary(phrases) + except Exception as e: + self._logger.error("Fatal compilation Error occured, " + + "cleaning up...", exc_info=True) + try: + os.remove(self.revision_file) + except OSError: + pass + raise e + else: + self._logger.info('Compilation done.') + return revision + + @abstractmethod + def _compile_vocabulary(self, phrases): + """ + Abstract method that should be overridden in subclasses with custom + compilation code. + + Arguments: + phrases -- a list of phrases that this vocabulary will contain + """ + + +class DummyVocabulary(AbstractVocabulary): + + PATH_PREFIX = 'dummy-vocabulary' + + @property + def is_compiled(self): + """ + Checks if the vocabulary is compiled by checking if the revision + file is readable. + + Returns: + True if this vocabulary has been compiled, else False + """ + return super(self.__class__, self).is_compiled + + def _compile_vocabulary(self, phrases): + """ + Does nothing (because this is a dummy class for testing purposes). + """ + pass + + +class PocketsphinxVocabulary(AbstractVocabulary): + + PATH_PREFIX = 'pocketsphinx-vocabulary' + + @property + def languagemodel_file(self): + """ + Returns: + The path of the the pocketsphinx languagemodel file as string + """ + return os.path.join(self.path, 'languagemodel') + @property + def dictionary_file(self): + """ + Returns: + The path of the pocketsphinx dictionary file as string + """ + return os.path.join(self.path, 'dictionary') -def compile(sentences, dictionary, languagemodel): + @property + def is_compiled(self): + """ + Checks if the vocabulary is compiled by checking if the revision, + languagemodel and dictionary files are readable. + + Returns: + True if this vocabulary has been compiled, else False + """ + return (super(self.__class__, self).is_compiled and + os.access(self.languagemodel_file, os.R_OK) and + os.access(self.dictionary_file, os.R_OK)) + + @property + def decoder_kwargs(self): + """ + Convenience property to use this Vocabulary with the __init__() method + of the pocketsphinx.Decoder class. + + Returns: + A dict containing kwargs for the pocketsphinx.Decoder.__init__() + method. + + Example: + decoder = pocketsphinx.Decoder(**vocab_instance.decoder_kwargs, + hmm='/path/to/hmm') + + """ + return {'lm': self.languagemodel_file, 'dict': self.dictionary_file} + + def _compile_vocabulary(self, phrases): + """ + Compiles the vocabulary to the Pocketsphinx format by creating a + languagemodel and a dictionary. + + Arguments: + phrases -- a list of phrases that this vocabulary will contain + """ + text = " ".join([("<s> %s </s>" % phrase) for phrase in phrases]) + self._logger.debug('Compiling languagemodel...') + vocabulary = self._compile_languagemodel(text, self.languagemodel_file) + self._logger.debug('Starting dictionary...') + self._compile_dictionary(vocabulary, self.dictionary_file) + + def _compile_languagemodel(self, text, output_file): + """ + Compiles the languagemodel from a text. + + Arguments: + text -- the text the languagemodel will be generated from + output_file -- the path of the file this languagemodel will + be written to + + Returns: + A list of all unique words this vocabulary contains. + """ + with tempfile.NamedTemporaryFile(suffix='.vocab', delete=False) as f: + vocab_file = f.name + + # Create vocab file from text + self._logger.debug("Creating vocab file: '%s'", vocab_file) + cmuclmtk.text2vocab(text, vocab_file) + + # Create language model from text + self._logger.debug("Creating languagemodel file: '%s'", output_file) + cmuclmtk.text2lm(text, output_file, vocab_file=vocab_file) + + # Get words from vocab file + self._logger.debug("Getting words from vocab file and removing it " + + "afterwards...") + words = [] + with open(vocab_file, 'r') as f: + for line in f: + line = line.strip() + if not line.startswith('#') and line not in ('<s>', '</s>'): + words.append(line) + os.remove(vocab_file) + + return words + + def _compile_dictionary(self, words, output_file): + """ + Compiles the dictionary from a list of words. + + Arguments: + words -- a list of all unique words this vocabulary contains + output_file -- the path of the file this dictionary will + be written to + """ + # create the dictionary + self._logger.debug("Getting phonemes for %d words...", len(words)) + g2pconverter = PhonetisaurusG2P(**PhonetisaurusG2P.get_config()) + phonemes = g2pconverter.translate(words) + + self._logger.debug("Creating dict file: '%s'", output_file) + with open(output_file, "w") as f: + for word, pronounciations in phonemes.items(): + for i, pronounciation in enumerate(pronounciations, start=1): + if i == 1: + line = "%s\t%s\n" % (word, pronounciation) + else: + line = "%s(%d)\t%s\n" % (word, i, pronounciation) + f.write(line) + + +def get_phrases_from_module(module): + """ + Gets phrases from a module. + + Arguments: + module -- a module reference + + Returns: + The list of phrases in this module. + """ + return module.WORDS if hasattr(module, 'WORDS') else [] + + +def get_keyword_phrases(): """ - Gets the words and creates the dictionary + Gets the keyword phrases from the keywords file in the jasper data dir. + + Returns: + A list of keyword phrases. + """ + phrases = [] + + with open(jasperpath.data('keyword_phrases'), mode="r") as f: + for line in f: + phrase = line.strip() + if phrase: + phrases.append(phrase) + + return phrases + + +def get_all_phrases(): """ + Gets phrases from all modules. - modules = Brain.get_modules() + Returns: + A list of phrases in all modules plus additional phrases passed to this + function. + """ + phrases = [] - words = [] + modules = brain.Brain.get_modules() for module in modules: - words.extend(module.WORDS) + phrases.extend(get_phrases_from_module(module)) + + return sorted(list(set(phrases))) - words = list(set(words)) +if __name__ == '__main__': + import shutil + import argparse - # create the dictionary - pronounced = g2p.translateWords(words) - zipped = zip(words, pronounced) - lines = ["%s %s" % (x, y) for x, y in zipped] + parser = argparse.ArgumentParser(description='Vocabcompiler Demo') + parser.add_argument('--base-dir', action='store', + help='the directory in which the vocabulary will be ' + + 'compiled.') + parser.add_argument('--debug', action='store_true', + help='show debug messages') + args = parser.parse_args() - with open(dictionary, "w") as f: - f.write("\n".join(lines) + "\n") + logging.basicConfig(level=logging.DEBUG if args.debug else logging.INFO) + base_dir = args.base_dir if args.base_dir else tempfile.mkdtemp() - # create the language model - with open(sentences, "w") as f: - f.write("\n".join(words) + "\n") - f.write("<s> \n </s> \n") - f.close() + phrases = get_all_phrases() + print "Module phrases: %r" % phrases - # make language model - text2lm(sentences, languagemodel) + for subclass in AbstractVocabulary.__subclasses__(): + if hasattr(subclass, 'PATH_PREFIX'): + vocab = subclass(path=base_dir) + print("Vocabulary in: %s" % vocab.path) + print("Revision file: %s" % vocab.revision_file) + print("Compiled revision: %s" % vocab.compiled_revision) + print("Is compiled: %r" % vocab.is_compiled) + print("Matches phrases: %r" % vocab.matches_phrases(phrases)) + if not vocab.is_compiled or not vocab.matches_phrases(phrases): + print("Compiling...") + vocab.compile(phrases) + print("") + print("Vocabulary in: %s" % vocab.path) + print("Revision file: %s" % vocab.revision_file) + print("Compiled revision: %s" % vocab.compiled_revision) + print("Is compiled: %r" % vocab.is_compiled) + print("Matches phrases: %r" % vocab.matches_phrases(phrases)) + print("") + if not args.base_dir: + print("Removing temporary directory '%s'..." % base_dir) + shutil.rmtree(base_dir) |
