From c30292e5f2b38430f5c5f17a52ae20cc16131233 Mon Sep 17 00:00:00 2001 From: schneefux Date: Wed, 17 Sep 2014 18:54:51 +0200 Subject: Rewritten vocabcompiler and separated it from pocketsphinx logic The pocketsphinx part of vocabcompiler now uses the cmuclmtk wrapper libary for compilation of the languagemodel/dictionary. A revision check has been implemented, so that vocabulary won't get recompiled if there's no need. Proper integration into jasper.py, client/stt.py and client/test.py is still missing due to pending pull requests that change these modules. --- client/vocabcompiler.py | 362 +++++++++++++++++++++++++++++++++++++++++++----- 1 file changed, 329 insertions(+), 33 deletions(-) (limited to 'client/vocabcompiler.py') diff --git a/client/vocabcompiler.py b/client/vocabcompiler.py index 52b1f0a..0e498a2 100644 --- a/client/vocabcompiler.py +++ b/client/vocabcompiler.py @@ -1,56 +1,352 @@ # -*- coding: utf-8-*- """ Iterates over all the WORDS variables in the modules and creates a -dictionary for the client. +vocabulary for the respective stt_engine if needed. """ import os +import tempfile +import logging +import hashlib +from abc import ABCMeta, abstractmethod, abstractproperty + +import cmuclmtk import g2p -from brain import Brain +import brain + + +class AbstractVocabulary(object): + """ + Abstract base class for Vocabulary classes. + + Please note that subclasses have to implement the compile_vocabulary() + method and set a string as the PATH_PREFIX class attribute. + """ + __metaclass__ = ABCMeta + + @classmethod + def phrases_to_revision(self, phrases): + """ + Calculates a revision from phrases by using the SHA1 hash function. + + Arguments: + phrases -- a list of phrases + + Returns: + A revision string for given phrases. + """ + sorted_phrases = sorted(phrases) + joined_phrases = '\n'.join(sorted_phrases) + sha1 = hashlib.sha1() + sha1.update(joined_phrases) + return sha1.hexdigest() + + def __init__(self, name='default', path='.'): + """ + Initializes a new Vocabulary instance. + + Optional Arguments: + name -- (optional) the name of the vocabulary (Default: 'default') + path -- (optional) the path in which the vocabulary exists or will + be created (Default: '.') + """ + self.name = name + self.path = os.path.abspath(os.path.join(path, self.PATH_PREFIX, name)) + self._logger = logging.getLogger(__name__) + + @property + def revision_file(self): + """ + Returns: + The path of the the revision file as string + """ + return os.path.join(self.path, 'revision') + + @abstractproperty + def is_compiled(self): + """ + Checks if the vocabulary is compiled by checking if the revision file + is readable. This method should be overridden by subclasses to check + for class-specific additional files, too. + + Returns: + True if the dictionary is compiled, else False + """ + return os.access(self.revision_file, os.R_OK) + + @property + def compiled_revision(self): + """ + Reads the compiled revision from the revision file. + + Returns: + the revision of this vocabulary (i.e. the string + inside the revision file), or None if is_compiled + if False + """ + if not self.is_compiled: + return None + with open(self.revision_file, 'r') as f: + revision = f.read().strip() + self._logger.debug("compiled_revision is '%s'", revision) + return revision + + def matches_phrases(self, phrases): + """ + Convenience method to check if this vocabulary exactly contains the + phrases passed to this method. + + Arguments: + phrases -- a list of phrases + + Returns: + True if phrases exactly matches the phrases inside this + vocabulary. + + """ + return (self.compiled_revision == self.phrases_to_revision(phrases)) + + def compile(self, phrases, force=False): + """ + Compiles this vocabulary. If the force argument is True, compilation + will be forced regardless of necessity (which means that the + preliminary check if the current revision already equals the + revision after compilation will be skipped). + This method is not meant to be overridden by subclasses - use the + _compile_vocabulary()-method instead. + + Arguments: + phrases -- a list of phrases that this vocabulary will contain + force -- (optional) forces compilation (Default: False) + + Returns: + The revision of the compiled vocabulary + """ + revision = self.phrases_to_revision(phrases) + if not force and self.compiled_revision == revision: + self._logger.debug('Compilation not neccessary, compiled ' + + 'version matches phrases.') + return revision + + if not os.path.exists(self.path): + try: + os.makedirs(self.path) + except OSError: + self._logger.error("Couldn't create vocabulary dir '%s'", + self.path, exc_info=True) + raise + try: + with open(self.revision_file, 'w') as f: + f.write(revision) + except (OSError, IOError): + self._logger.error("Couldn't write revision file in '%s'", + self.revision_file, exc_info=True) + raise + else: + try: + self._logger.debug('Starting compilation...') + self._compile_vocabulary(phrases) + except Exception as e: + self._logger.error("Fatal compilation Error occured, " + + "cleaning up...", exc_info=True) + try: + os.remove(self.revision_file) + except OSError: + pass + raise e + return revision + + @abstractmethod + def _compile_vocabulary(self, phrases): + """ + Abstract method that should be overridden in subclasses with custom + compilation code. + + Arguments: + phrases -- a list of phrases that this vocabulary will contain + """ + pass + +class PocketsphinxVocabulary(AbstractVocabulary): -def text2lm(in_filename, out_filename): - """Wrapper around the language model compilation tools""" - def text2idngram(in_filename, out_filename): - cmd = "text2idngram -vocab %s < %s -idngram temp.idngram" % ( - out_filename, in_filename) - os.system(cmd) + PATH_PREFIX = 'pocketsphinx-vocabulary' - def idngram2lm(in_filename, out_filename): - cmd = "idngram2lm -idngram temp.idngram -vocab %s -arpa %s" % ( - in_filename, out_filename) - os.system(cmd) + @property + def languagemodel_file(self): + """ + Returns: + The path of the the pocketsphinx languagemodel file as string + """ + return os.path.join(self.path, 'languagemodel') - text2idngram(in_filename, in_filename) - idngram2lm(in_filename, out_filename) + @property + def dictionary_file(self): + """ + Returns: + The path of the pocketsphinx dictionary file as string + """ + return os.path.join(self.path, 'dictionary') + @property + def is_compiled(self): + """ + Checks if the vocabulary is compiled by checking if the revision, + languagemodel and dictionary files are readable. -def compile(sentences, dictionary, languagemodel): + Returns: + True if this vocabulary has been compiled, else False + """ + return (super(self.__class__, self).is_compiled and + os.access(self.languagemodel_file, os.R_OK) and + os.access(self.dictionary_file, os.R_OK)) + + @property + def decoder_kwargs(self): + """ + Convenience property to use this Vocabulary with the __init__() method + of the pocketsphinx.Decoder class. + + Returns: + A dict containing kwargs for the pocketsphinx.Decoder.__init__() + method. + + Example: + decoder = pocketsphinx.Decoder(**vocab_instance.decoder_kwargs, + hmm='/path/to/hmm') + + """ + return {'lm': self.languagemodel_file, 'dict': self.dictionary_file} + + def _compile_vocabulary(self, phrases): + """ + Compiles the vocabulary to the Pocketsphinx format by creating a + languagemodel and a dictionary. + + Arguments: + phrases -- a list of phrases that this vocabulary will contain + """ + text = " ".join([(" %s " % phrase) for phrase in phrases]) + vocabulary = self._compile_languagemodel(text, self.languagemodel_file) + self._compile_dictionary(vocabulary, self.dictionary_file) + + def _compile_languagemodel(self, text, output_file): + """ + Compiles the languagemodel from a text. + + Arguments: + text -- the text the languagemodel will be generated from + output_file -- the path of the file this languagemodel will + be written to + + Returns: + A list of all unique words this vocabulary contains. + """ + with tempfile.NamedTemporaryFile(suffix='.vocab', delete=False) as f: + vocab_file = f.name + + # Create vocab file from text + cmuclmtk.text2vocab(text, vocab_file) + + # Create language model from text + cmuclmtk.text2lm(text, output_file, vocab_file=vocab_file) + + # Get words from vocab file + words = [] + with open(vocab_file, 'r') as f: + for line in f: + line = line.strip() + if not line.startswith('#') and line not in ('', ''): + words.append(line) + + os.remove(vocab_file) + + return words + + def _compile_dictionary(self, words, output_file): + """ + Compiles the dictionary from a list of words. + + Arguments: + words -- a list of all unique words this vocabulary contains + output_file -- the path of the file this dictionary will + be written to + """ + # create the dictionary + pronounced = g2p.translateWords(words) + zipped = zip(words, pronounced) + lines = ["%s %s" % (x, y) for x, y in zipped] + + with open(output_file, "w") as f: + for line in lines: + f.write("%s\n" % line) + + +def get_phrases_from_module(module): """ - Gets the words and creates the dictionary + Gets phrases from a module. + + Arguments: + module -- a module reference + + Returns: + The list of phrases in this module. + """ + return module.WORDS if hasattr(module, 'WORDS') else [] + + +def get_all_phrases(): """ + Gets phrases from all modules. - modules = Brain.get_modules() + Returns: + A list of phrases in all modules plus additional phrases passed to this + function. + """ + phrases = [] - words = [] + modules = brain.Brain.get_modules() for module in modules: - words.extend(module.WORDS) + phrases.extend(get_phrases_from_module(module)) + + return sorted(list(set(phrases))) - words = list(set(words)) +if __name__ == '__main__': + import shutil + import argparse - # create the dictionary - pronounced = g2p.translateWords(words) - zipped = zip(words, pronounced) - lines = ["%s %s" % (x, y) for x, y in zipped] + parser = argparse.ArgumentParser(description='Vocabcompiler Demo') + parser.add_argument('--base-dir', action='store', + help='the directory in which the vocabulary will be ' + + 'compiled.') + parser.add_argument('--debug', action='store_true', + help='show debug messages') + args = parser.parse_args() - with open(dictionary, "w") as f: - f.write("\n".join(lines) + "\n") + logging.basicConfig(level=logging.DEBUG if args.debug else logging.INFO) + base_dir = args.base_dir if args.base_dir else tempfile.mkdtemp() - # create the language model - with open(sentences, "w") as f: - f.write("\n".join(words) + "\n") - f.write(" \n \n") - f.close() + phrases = get_all_phrases() + print "Module phrases: %r" % phrases - # make language model - text2lm(sentences, languagemodel) + for subclass in AbstractVocabulary.__subclasses__(): + if hasattr(subclass, 'PATH_PREFIX'): + vocab = subclass(path=base_dir) + print("Vocabulary in: %s" % vocab.path) + print("Revision file: %s" % vocab.revision_file) + print("Compiled revision: %s" % vocab.compiled_revision) + print("Is compiled: %r" % vocab.is_compiled) + print("Matches phrases: %r" % vocab.matches_phrases(phrases)) + if not vocab.is_compiled or not vocab.matches_phrases(phrases): + print("Compiling...") + vocab.compile(phrases) + print("") + print("Vocabulary in: %s" % vocab.path) + print("Revision file: %s" % vocab.revision_file) + print("Compiled revision: %s" % vocab.compiled_revision) + print("Is compiled: %r" % vocab.is_compiled) + print("Matches phrases: %r" % vocab.matches_phrases(phrases)) + print("") + if not args.base_dir: + print("Removing temporary directory '%s'..." % base_dir) + shutil.rmtree(base_dir) -- cgit v1.3.1 From 9633fdeb0118f5e05c7e9eebd154a49f5ef85209 Mon Sep 17 00:00:00 2001 From: schneefux Date: Wed, 24 Sep 2014 19:10:43 +0200 Subject: Added DummyVocabulary class --- client/vocabcompiler.py | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) (limited to 'client/vocabcompiler.py') diff --git a/client/vocabcompiler.py b/client/vocabcompiler.py index 0e498a2..a7d2c4b 100644 --- a/client/vocabcompiler.py +++ b/client/vocabcompiler.py @@ -168,6 +168,28 @@ class AbstractVocabulary(object): pass +class DummyVocabulary(AbstractVocabulary): + + PATH_PREFIX = 'dummy-vocabulary' + + @property + def is_compiled(self): + """ + Checks if the vocabulary is compiled by checking if the revision + file is readable. + + Returns: + True if this vocabulary has been compiled, else False + """ + return super(self.__class__, self).is_compiled + + def _compile_vocabulary(self, phrases): + """ + Does nothing (because this is a dummy class for testing purposes). + """ + pass + + class PocketsphinxVocabulary(AbstractVocabulary): PATH_PREFIX = 'pocketsphinx-vocabulary' -- cgit v1.3.1 From 0b681b489996ea8b9a9a5d3fc670af841994085d Mon Sep 17 00:00:00 2001 From: schneefux Date: Thu, 25 Sep 2014 17:44:29 +0200 Subject: Add error handler for cmuclmtk import to vocabcompiler --- client/vocabcompiler.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) (limited to 'client/vocabcompiler.py') diff --git a/client/vocabcompiler.py b/client/vocabcompiler.py index a7d2c4b..9481ef6 100644 --- a/client/vocabcompiler.py +++ b/client/vocabcompiler.py @@ -10,9 +10,12 @@ import logging import hashlib from abc import ABCMeta, abstractmethod, abstractproperty -import cmuclmtk import g2p import brain +try: + import cmuclmtk +except: + logging.getLogger(__name__).error("Error importing CMUCLMTK module. PocketsphinxVocabulary will not work correctly.", exc_info=True) class AbstractVocabulary(object): -- cgit v1.3.1 From 6e48647677f47ea7990d70757ec23eb2fea9e874 Mon Sep 17 00:00:00 2001 From: schneefux Date: Thu, 25 Sep 2014 22:24:20 +0200 Subject: Improve logging in vocabcompiler.py --- client/vocabcompiler.py | 19 ++++++++++++++++--- 1 file changed, 16 insertions(+), 3 deletions(-) (limited to 'client/vocabcompiler.py') diff --git a/client/vocabcompiler.py b/client/vocabcompiler.py index 9481ef6..ae5dc7d 100644 --- a/client/vocabcompiler.py +++ b/client/vocabcompiler.py @@ -15,7 +15,9 @@ import brain try: import cmuclmtk except: - logging.getLogger(__name__).error("Error importing CMUCLMTK module. PocketsphinxVocabulary will not work correctly.", exc_info=True) + logging.getLogger(__name__).error("Error importing CMUCLMTK module. " + + "PocketsphinxVocabulary will not work " + + "correctly.", exc_info=True) class AbstractVocabulary(object): @@ -132,6 +134,8 @@ class AbstractVocabulary(object): return revision if not os.path.exists(self.path): + self._logger.debug("Vocabulary dir '%s' does not exist, " + + "creating...", self.path) try: os.makedirs(self.path) except OSError: @@ -146,8 +150,8 @@ class AbstractVocabulary(object): self.revision_file, exc_info=True) raise else: + self._logger.info('Starting compilation...') try: - self._logger.debug('Starting compilation...') self._compile_vocabulary(phrases) except Exception as e: self._logger.error("Fatal compilation Error occured, " + @@ -157,6 +161,8 @@ class AbstractVocabulary(object): except OSError: pass raise e + else: + self._logger.info('Compilation done.') return revision @abstractmethod @@ -252,7 +258,9 @@ class PocketsphinxVocabulary(AbstractVocabulary): phrases -- a list of phrases that this vocabulary will contain """ text = " ".join([(" %s " % phrase) for phrase in phrases]) + self._logger.debug('Compiling languagemodel...') vocabulary = self._compile_languagemodel(text, self.languagemodel_file) + self._logger.debug('Starting dictionary...') self._compile_dictionary(vocabulary, self.dictionary_file) def _compile_languagemodel(self, text, output_file): @@ -271,19 +279,22 @@ class PocketsphinxVocabulary(AbstractVocabulary): vocab_file = f.name # Create vocab file from text + self._logger.debug("Creating vocab file: '%s'", vocab_file) cmuclmtk.text2vocab(text, vocab_file) # Create language model from text + self._logger.debug("Creating languagemodel file: '%s'", output_file) cmuclmtk.text2lm(text, output_file, vocab_file=vocab_file) # Get words from vocab file + self._logger.debug("Getting words from vocab file and removing it " + + "afterwards...") words = [] with open(vocab_file, 'r') as f: for line in f: line = line.strip() if not line.startswith('#') and line not in ('', ''): words.append(line) - os.remove(vocab_file) return words @@ -298,10 +309,12 @@ class PocketsphinxVocabulary(AbstractVocabulary): be written to """ # create the dictionary + self._logger.debug("Getting phonemes for %d words...", len(words)) pronounced = g2p.translateWords(words) zipped = zip(words, pronounced) lines = ["%s %s" % (x, y) for x, y in zipped] + self._logger.debug("Creating dict file: '%s'", output_file) with open(output_file, "w") as f: for line in lines: f.write("%s\n" % line) -- cgit v1.3.1 From 382c21d2ec5ca18d6932d38db8c23432aeac33dd Mon Sep 17 00:00:00 2001 From: schneefux Date: Fri, 26 Sep 2014 16:03:59 +0200 Subject: Use newer cmuclmtk libary version and catch ImportError --- client/requirements.txt | 2 +- client/vocabcompiler.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) (limited to 'client/vocabcompiler.py') diff --git a/client/requirements.txt b/client/requirements.txt index 57b6bc6..f3508fe 100644 --- a/client/requirements.txt +++ b/client/requirements.txt @@ -10,4 +10,4 @@ python-mpd==0.3.0 pytz==2013b semantic==1.0.3 requests==2.1.0 -cmuclmtk==0.1.2 +cmuclmtk==0.1.5 diff --git a/client/vocabcompiler.py b/client/vocabcompiler.py index ae5dc7d..c59bd6e 100644 --- a/client/vocabcompiler.py +++ b/client/vocabcompiler.py @@ -14,7 +14,7 @@ import g2p import brain try: import cmuclmtk -except: +except ImportError: logging.getLogger(__name__).error("Error importing CMUCLMTK module. " + "PocketsphinxVocabulary will not work " + "correctly.", exc_info=True) -- cgit v1.3.1 From 0aacc7df0f106cbca991c2d7d0d86d5ec3abafad Mon Sep 17 00:00:00 2001 From: schneefux Date: Sat, 27 Sep 2014 14:41:02 +0200 Subject: Add method to get keyword phrases to vocabcompiler This way they can be used by other STT engines as well --- client/vocabcompiler.py | 20 ++++++++++++++++++++ static/keyword_phrases | 18 ++++++++++++++++++ 2 files changed, 38 insertions(+) create mode 100644 static/keyword_phrases (limited to 'client/vocabcompiler.py') diff --git a/client/vocabcompiler.py b/client/vocabcompiler.py index c59bd6e..8edee6f 100644 --- a/client/vocabcompiler.py +++ b/client/vocabcompiler.py @@ -12,6 +12,8 @@ from abc import ABCMeta, abstractmethod, abstractproperty import g2p import brain +import jasperpath + try: import cmuclmtk except ImportError: @@ -333,6 +335,24 @@ def get_phrases_from_module(module): return module.WORDS if hasattr(module, 'WORDS') else [] +def get_keyword_phrases(): + """ + Gets the keyword phrases from the keywords file in the jasper data dir. + + Returns: + A list of keyword phrases. + """ + phrases = [] + + with open(jasperpath.data('keyword_phrases'), mode="r") as f: + for line in f: + phrase = line.strip() + if phrase: + phrases.append(phrase) + + return phrases + + def get_all_phrases(): """ Gets phrases from all modules. diff --git a/static/keyword_phrases b/static/keyword_phrases new file mode 100644 index 0000000..d118b13 --- /dev/null +++ b/static/keyword_phrases @@ -0,0 +1,18 @@ +BE +BEING +BUT +DID +FIRST +IN +IS +IT +JASPER +NOW +OF +ON +RIGHT +SAY +WHAT +WHICH +WITH +WORK \ No newline at end of file -- cgit v1.3.1 From faa0df57224fd5c7374c831af67e01bedfd30e1b Mon Sep 17 00:00:00 2001 From: schneefux Date: Sun, 28 Sep 2014 19:18:18 +0200 Subject: Rewritten G2P code --- client/g2p.py | 225 +++++++++++++++++++++++++++++++++++------------- client/vocabcompiler.py | 16 ++-- 2 files changed, 173 insertions(+), 68 deletions(-) (limited to 'client/vocabcompiler.py') diff --git a/client/g2p.py b/client/g2p.py index 89f2282..8ce7997 100644 --- a/client/g2p.py +++ b/client/g2p.py @@ -1,69 +1,170 @@ # -*- coding: utf-8-*- import os -import tempfile -import subprocess +import sys import re -import yaml +import subprocess +import tempfile +import shutil +import logging +if sys.version_info < (3, 3): + import distutils.spawn import jasperpath +import yaml -PHONE_MATCH = re.compile(r' (.*) ') - -FST_MODEL = None - -# Try to get fst_model from config -profile_path = os.path.join(os.path.dirname(__file__), 'profile.yml') -if os.path.exists(profile_path): - with open(profile_path, 'r') as f: - profile = yaml.safe_load(f) - if ('pocketsphinx' in profile and - 'fst_model' in profile['pocketsphinx']): - FST_MODEL = profile['pocketsphinx']['fst_model'] - -if not FST_MODEL: - FST_MODEL = os.path.join(jasperpath.APP_PATH, os.pardir, 'phonetisaurus', - 'g014b2b.fst') - - -def parseLine(line): - return PHONE_MATCH.search(line).group(1) - - -def parseOutput(output): - return PHONE_MATCH.findall(output) - - -def translateWord(word): - out = subprocess.check_output( - ['phonetisaurus-g2p', '--model=%s' % FST_MODEL, '--input=%s' % word]) - return parseLine(out) - - -def translateWords(words): - full_text = '\n'.join(words) - - with tempfile.NamedTemporaryFile(suffix='.g2p', delete=False) as f: - temp_filename = f.name - f.write(full_text) - - output = translateFile(temp_filename) - os.remove(temp_filename) - - return output - - -def translateFile(input_filename, output_filename=None): - out = subprocess.check_output( - ['phonetisaurus-g2p', '--model=%s' % FST_MODEL, - '--input=%s' % input_filename, '--words', '--isfile']) - out = parseOutput(out) - - if output_filename: - out = '\n'.join(out) - - with open(output_filename, "wb") as f: - f.write(out) - - return None - return out +class PhonetisaurusG2P(object): + PATTERN = re.compile(r'^(?P.+)\t(?P\d+\.\d+)\t ' + + r'(?P.*) ', re.MULTILINE) + + @classmethod + def executable_found(cls): + if sys.version_info < (3, 3): + cmd_exists = distutils.spawn.find_executable + else: + cmd_exists = shutil.which + # Required binary for this class + cmd = 'phonetisaurus-g2p' + if not cmd_exists(cmd): + return False + return True + + @classmethod + def execute(cls, fst_model, input, is_file=False, nbest=None): + logger = logging.getLogger(__name__) + + cmd = ['phonetisaurus-g2p', + '--model=%s' % fst_model, + '--input=%s' % input, + '--words'] + + if is_file: + cmd.append('--isfile') + + if nbest is not None: + cmd.extend(['--nbest=%d' % nbest]) + + cmd = [str(x) for x in cmd] + with tempfile.SpooledTemporaryFile() as err_f: + try: + # FIXME: We can't just use subprocess.call and redirect stdout + # and stderr, because it looks like Phonetisaurus can't open + # an already opened file descriptor a second time. This is why + # we have to use this somehow hacky subprocess.Popen approach. + proc = subprocess.Popen(cmd, stdout=subprocess.PIPE, + stderr=subprocess.PIPE) + stdoutdata, stderrdata = proc.communicate() + returncode = proc.returncode + if returncode != 0: + logger.warning("Command '%s' return with exit status %d", + ' '.join(cmd), returncode) + except OSError: + logger.error("Error occured while executing command '%s'", + ' '.join(cmd), exc_info=True) + stdoutdata, stderrdata = None, None + if stderrdata is not None: + for line in stderrdata.splitlines(): + message = line.strip() + if message: + logger.debug(message) + + result = {} + if stdoutdata is not None: + for word, precision, pronounc in cls.PATTERN.findall(stdoutdata): + if word not in result: + result[word] = [] + result[word].append(pronounc) + return result + + @classmethod + def get_config(cls): + # FIXME: Replace this as soon as pull request + # jasperproject/jasper-client#128 has been merged + + conf = {'fst_model': os.path.join(jasperpath.APP_PATH, os.pardir, + 'phonetisaurus', 'g014b2b.fst')} + # Try to get fst_model from config + profile_path = os.path.join(os.path.dirname(__file__), 'profile.yml') + if os.path.exists(profile_path): + with open(profile_path, 'r') as f: + profile = yaml.safe_load(f) + if 'pocketsphinx' in profile: + if 'fst_model' in profile['pocketsphinx']: + conf['fst_model'] = \ + profile['pocketsphinx']['fst_model'] + if 'nbest' in profile['pocketsphinx']: + conf['nbest'] = int(profile['pocketsphinx']['nbest']) + print conf + return conf + + def __new__(cls, fst_model=None, *args, **kwargs): + if not cls.executable_found(): + raise OSError("Can't find command 'phonetisaurus-g2p'! Please " + + "check if Phonetisaurus is installed and in your " + + "$PATH.") + if fst_model is None or not os.access(fst_model, os.R_OK): + raise OSError("FST model '%r' does not exist! Can't create " + + "instance." % fst_model) + inst = object.__new__(cls, fst_model, *args, **kwargs) + return inst + + def __init__(self, fst_model=None, nbest=None): + self._logger = logging.getLogger(__name__) + + self.fst_model = os.path.abspath(fst_model) + self._logger.debug("Using FST model: '%s'", self.fst_model) + + self.nbest = nbest + if self.nbest is not None: + self._logger.debug("Will use the %d best results.", self.nbest) + + def _translate_word(self, word): + return self.execute(self.fst_model, word, nbest=self.nbest) + + def _translate_words(self, words): + with tempfile.NamedTemporaryFile(suffix='.g2p', delete=False) as f: + # The 'delete=False' kwarg is kind of a hack, but Phonetisaurus + # won't work if we remove it, because it seems that I can't open + # a file descriptor a second time. + for word in words: + f.write("%s\n" % word) + tmp_fname = f.name + output = self.execute(self.fst_model, tmp_fname, is_file=True, + nbest=self.nbest) + os.remove(tmp_fname) + return output + + def translate(self, words): + if type(words) is str or len(words) == 1: + self._logger.debug('Converting single word to phonemes') + output = self._translate_word(words if type(words) is str + else words[0]) + else: + self._logger.debug('Converting %d words to phonemes', len(words)) + output = self._translate_words(words) + self._logger.debug('G2P conversion returned phonemes for %d words', + len(output)) + return output + +if __name__ == "__main__": + import pprint + import argparse + parser = argparse.ArgumentParser(description='Phonetisaurus G2P module') + parser.add_argument('fst_model', action='store', + help='Path to the FST Model') + parser.add_argument('--debug', action='store_true', + help='Show debug messages') + args = parser.parse_args() + + logging.basicConfig() + logger = logging.getLogger() + if args.debug: + logger.setLevel(logging.DEBUG) + + words = ['THIS', 'IS', 'A', 'TEST'] + + g2pconv = PhonetisaurusG2P(args.fst_model, nbest=3) + output = g2pconv.translate(words) + + pp = pprint.PrettyPrinter(indent=2) + pp.pprint(output) diff --git a/client/vocabcompiler.py b/client/vocabcompiler.py index 8edee6f..1cfe15e 100644 --- a/client/vocabcompiler.py +++ b/client/vocabcompiler.py @@ -10,10 +10,10 @@ import logging import hashlib from abc import ABCMeta, abstractmethod, abstractproperty -import g2p import brain import jasperpath +from g2p import PhonetisaurusG2P try: import cmuclmtk except ImportError: @@ -312,14 +312,18 @@ class PocketsphinxVocabulary(AbstractVocabulary): """ # create the dictionary self._logger.debug("Getting phonemes for %d words...", len(words)) - pronounced = g2p.translateWords(words) - zipped = zip(words, pronounced) - lines = ["%s %s" % (x, y) for x, y in zipped] + g2pconverter = PhonetisaurusG2P(**PhonetisaurusG2P.get_config()) + phonemes = g2pconverter.translate(words) self._logger.debug("Creating dict file: '%s'", output_file) with open(output_file, "w") as f: - for line in lines: - f.write("%s\n" % line) + for word, pronounciations in phonemes.items(): + for i, pronounciation in enumerate(pronounciations, start=1): + if i == 1: + line = "%s\t%s\n" % (word, pronounciation) + else: + line = "%s(%d)\t%s\n" % (word, i, pronounciation) + f.write(line) def get_phrases_from_module(module): -- cgit v1.3.1 From f9db756c18723abcacd54ad0667f69babca4c695 Mon Sep 17 00:00:00 2001 From: schneefux Date: Mon, 6 Oct 2014 18:11:32 +0200 Subject: PEP8 style fixes in test.py and vocabcompiler.py --- client/test.py | 3 ++- client/vocabcompiler.py | 2 +- 2 files changed, 3 insertions(+), 2 deletions(-) (limited to 'client/vocabcompiler.py') diff --git a/client/test.py b/client/test.py index 7f31f37..0034974 100644 --- a/client/test.py +++ b/client/test.py @@ -86,7 +86,8 @@ class TestMic(unittest.TestCase): class TestG2P(unittest.TestCase): def setUp(self): - self.g2pconverter = g2p.PhonetisaurusG2P(**g2p.PhonetisaurusG2P.get_config()) + self.g2pconverter = g2p.PhonetisaurusG2P( + **g2p.PhonetisaurusG2P.get_config()) self.words = ['GOOD', 'BAD', 'UGLY'] def testTranslateWord(self): diff --git a/client/vocabcompiler.py b/client/vocabcompiler.py index 1cfe15e..d0f0124 100644 --- a/client/vocabcompiler.py +++ b/client/vocabcompiler.py @@ -323,7 +323,7 @@ class PocketsphinxVocabulary(AbstractVocabulary): line = "%s\t%s\n" % (word, pronounciation) else: line = "%s(%d)\t%s\n" % (word, i, pronounciation) - f.write(line) + f.write(line) def get_phrases_from_module(module): -- cgit v1.3.1 From f91ed45347ae10e4fed612ecb81037d421bb785c Mon Sep 17 00:00:00 2001 From: schneefux Date: Wed, 8 Oct 2014 19:55:31 +0200 Subject: Remove unneccessary pass statement from AbstractVocabulary class This should raise vocabcompiler test coverage to 100%. Whohooo! --- client/vocabcompiler.py | 1 - 1 file changed, 1 deletion(-) (limited to 'client/vocabcompiler.py') diff --git a/client/vocabcompiler.py b/client/vocabcompiler.py index d0f0124..7eef5c0 100644 --- a/client/vocabcompiler.py +++ b/client/vocabcompiler.py @@ -176,7 +176,6 @@ class AbstractVocabulary(object): Arguments: phrases -- a list of phrases that this vocabulary will contain """ - pass class DummyVocabulary(AbstractVocabulary): -- cgit v1.3.1