diff options
| author | schneefux <schneefux+commit@schneefux.xyz> | 2014-09-10 17:30:34 +0200 |
|---|---|---|
| committer | schneefux <schneefux+commit@schneefux.xyz> | 2014-09-11 10:41:14 -0400 |
| commit | 2e59ddec511c6a84c1953845d21898390cfd8b77 (patch) | |
| tree | 42818d81cdf178780eb9f8306e5eccc50f1d7242 /client | |
| parent | 2b6f6469189cbff93a16a2025fa3560e5086667c (diff) | |
| download | jasper-client-2e59ddec511c6a84c1953845d21898390cfd8b77.tar.gz jasper-client-2e59ddec511c6a84c1953845d21898390cfd8b77.zip | |
Move vocabcompiler into client folder
Diffstat (limited to 'client')
| -rw-r--r-- | client/vocabcompiler.py | 69 |
1 files changed, 69 insertions, 0 deletions
diff --git a/client/vocabcompiler.py b/client/vocabcompiler.py new file mode 100644 index 0000000..1b4d94c --- /dev/null +++ b/client/vocabcompiler.py @@ -0,0 +1,69 @@ +# -*- coding: utf-8-*- +""" + Iterates over all the WORDS variables in the modules and creates a dictionary for the client. +""" + +import os +import sys +import glob + +mod_path = os.path.abspath('modules/') + +sys.path.append(mod_path) + +import g2p + + +def text2lm(in_filename, out_filename): + """Wrapper around the language model compilation tools""" + def text2idngram(in_filename, out_filename): + cmd = "text2idngram -vocab %s < %s -idngram temp.idngram" % (out_filename, + in_filename) + os.system(cmd) + + def idngram2lm(in_filename, out_filename): + cmd = "idngram2lm -idngram temp.idngram -vocab %s -arpa %s" % ( + in_filename, out_filename) + os.system(cmd) + + text2idngram(in_filename, in_filename) + idngram2lm(in_filename, out_filename) + + +def compile(sentences, dictionary, languagemodel): + """ + Gets the words and creates the dictionary + """ + + m = [os.path.basename(f)[:-3] + for f in glob.glob(os.path.dirname("../client/modules/") + "/*.py")] + + words = [] + for module_name in m: + try: + exec("import %s" % module_name) + eval("words.extend(%s.WORDS)" % module_name) + except: + pass # module probably doesn't have the property + + words = list(set(words)) + + # for spotify module + words.extend(["MUSIC", "SPOTIFY"]) + + # create the dictionary + pronounced = g2p.translateWords(words) + zipped = zip(words, pronounced) + lines = ["%s %s" % (x, y) for x, y in zipped] + + with open(dictionary, "w") as f: + f.write("\n".join(lines) + "\n") + + # create the language model + with open(sentences, "w") as f: + f.write("\n".join(words) + "\n") + f.write("<s> \n </s> \n") + f.close() + + # make language model + text2lm(sentences, languagemodel) |
