From 7786b9e586f504f20ec28be9213aff18abcc81a2 Mon Sep 17 00:00:00 2001 From: schneefux Date: Mon, 2 Feb 2015 10:14:19 +0100 Subject: add option to use XSAMPA fst models --- client/g2p.py | 27 ++- client/phonetic_helper.py | 479 ++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 498 insertions(+), 8 deletions(-) create mode 100644 client/phonetic_helper.py diff --git a/client/g2p.py b/client/g2p.py index 9595dbf..8de18f9 100644 --- a/client/g2p.py +++ b/client/g2p.py @@ -10,13 +10,14 @@ import yaml import diagnose import jasperpath +import phonetic_helper class PhonetisaurusG2P(object): PATTERN = re.compile(r'^(?P.+)\t(?P\d+\.\d+)\t(?: )?' + r'(?P.*?)(?: )?$', re.MULTILINE) @classmethod - def execute(cls, fst_model, input, is_file=False, nbest=None): + def execute(cls, fst_model, fst_is_xsampa, input, is_file=False, nbest=None): logger = logging.getLogger(__name__) cmd = ['phonetisaurus-g2p', @@ -60,6 +61,8 @@ class PhonetisaurusG2P(object): for word, precision, pronounc in cls.PATTERN.findall(stdoutdata): if word not in result: result[word] = [] + if fst_is_xsampa: + pronounc = phonetic_helper.xsampa2xarpabet("", pronounc.replace(" ", "")) result[word].append(pronounc) return result @@ -69,7 +72,8 @@ class PhonetisaurusG2P(object): # jasperproject/jasper-client#128 has been merged conf = {'fst_model': os.path.join(jasperpath.APP_PATH, os.pardir, - 'phonetisaurus', 'g014b2b.fst')} + 'phonetisaurus', 'g014b2b.fst'), + 'fst_is_xsampa': False} # Try to get fst_model from config profile_path = jasperpath.config('profile.yml') if os.path.exists(profile_path): @@ -79,11 +83,14 @@ class PhonetisaurusG2P(object): if 'fst_model' in profile['pocketsphinx']: conf['fst_model'] = \ profile['pocketsphinx']['fst_model'] + if 'fst_is_xsampa' in profile['pocketsphinx']: + conf['fst_is_xsampa'] = \ + profile['pocketsphinx']['fst_is_xsampa'] if 'nbest' in profile['pocketsphinx']: conf['nbest'] = int(profile['pocketsphinx']['nbest']) return conf - def __new__(cls, fst_model=None, *args, **kwargs): + def __new__(cls, fst_model=None, fst_is_xsampa=False, *args, **kwargs): if not diagnose.check_executable('phonetisaurus-g2p'): raise OSError("Can't find command 'phonetisaurus-g2p'! Please " + "check if Phonetisaurus is installed and in your " + @@ -91,21 +98,23 @@ class PhonetisaurusG2P(object): if fst_model is None or not os.access(fst_model, os.R_OK): raise OSError(("FST model '%r' does not exist! Can't create " + "instance.") % fst_model) - inst = object.__new__(cls, fst_model, *args, **kwargs) + inst = object.__new__(cls, fst_model, fst_model, *args, **kwargs) return inst - def __init__(self, fst_model=None, nbest=None): + def __init__(self, fst_model=None, fst_is_xsampa=False, nbest=None): self._logger = logging.getLogger(__name__) self.fst_model = os.path.abspath(fst_model) self._logger.debug("Using FST model: '%s'", self.fst_model) + self.fst_is_xsampa = fst_is_xsampa + self.nbest = nbest if self.nbest is not None: self._logger.debug("Will use the %d best results.", self.nbest) def _translate_word(self, word): - return self.execute(self.fst_model, word, nbest=self.nbest) + return self.execute(self.fst_model, self.fst_is_xsampa, word, nbest=self.nbest) def _translate_words(self, words): with tempfile.NamedTemporaryFile(suffix='.g2p', delete=False) as f: @@ -115,7 +124,7 @@ class PhonetisaurusG2P(object): for word in words: f.write("%s\n" % word) tmp_fname = f.name - output = self.execute(self.fst_model, tmp_fname, is_file=True, + output = self.execute(self.fst_model, self.fst_is_xsampa, tmp_fname, is_file=True, nbest=self.nbest) os.remove(tmp_fname) return output @@ -138,6 +147,8 @@ if __name__ == "__main__": parser = argparse.ArgumentParser(description='Phonetisaurus G2P module') parser.add_argument('fst_model', action='store', help='Path to the FST Model') + parser.add_argument('fst_is_xsampa', action='store_true', + help='Convert XSAMPA to Arpabet (required depending on your model)') parser.add_argument('--debug', action='store_true', help='Show debug messages') args = parser.parse_args() @@ -149,7 +160,7 @@ if __name__ == "__main__": words = ['THIS', 'IS', 'A', 'TEST'] - g2pconv = PhonetisaurusG2P(args.fst_model, nbest=3) + g2pconv = PhonetisaurusG2P(args.fst_model, args.fst_is_xsampa, nbest=3) output = g2pconv.translate(words) pp = pprint.PrettyPrinter(indent=2) diff --git a/client/phonetic_helper.py b/client/phonetic_helper.py new file mode 100644 index 0000000..556b92b --- /dev/null +++ b/client/phonetic_helper.py @@ -0,0 +1,479 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- + +# +# Copyright 2013, 2014 Guenter Bartsch +# +# This program is free software: you can redistribute it and/or modify +# it under the terms of the GNU Lesser General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# This program is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU Lesser General Public License for more details. +# +# You should have received a copy of the GNU Lesser General Public License +# along with this program. If not, see . +# + +import sys +import xml.sax +import re +import random +import unittest + +# +# big phoneme table +# +# entries: +# ( IPA, XSAMPA, MARY ) +# + +MAX_PHONEME_LENGTH = 2 + +big_phoneme_table = [ + + # + # stop + # + + ( u'p' , 'p' , 'p' ), + ( u'b' , 'b' , 'b' ), + ( u't' , 't' , 't' ), + ( u'd' , 'd' , 'd' ), + ( u'k' , 'k' , 'k' ), + ( u'g' , 'g' , 'g' ), + ( u'ʔ' , '?' , '?' ), + + # + # 2 consonants + # + + ( u'pf' , 'pf' , 'pf' ), + ( u'ts' , 'ts' , 'ts' ), + ( u'tʃ' , 'tS' , 'tS' ), + ( u'dʒ' , 'dZ' , 'dZ' ), + + # + # fricative + # + + ( u'f' , 'f' , 'f' ), + ( u'v' , 'v' , 'v' ), + ( u'θ' , 'T' , 'T' ), + ( u'ð' , 'D' , 'D' ), + ( u's' , 's' , 's' ), + ( u'z' , 'z' , 'z' ), + ( u'ʃ' , 'S' , 'S' ), + ( u'ʒ' , 'Z' , 'Z' ), + ( u'ç' , 'C' , 'C' ), + ( u'j' , 'j' , 'j' ), + ( u'x' , 'x' , 'x' ), + ( u'ʁ' , 'R' , 'R' ), + ( u'h' , 'h' , 'h' ), + ( u'ɥ' , 'H' , 'H' ), + + # + # nasal + # + + ( u'm' , 'm' , 'm' ), + ( u'n' , 'n' , 'n' ), + ( u'ɳ' , 'N' , 'N' ), + + # + # liquid + # + + ( u'l' , 'l' , 'l' ), + ( u'r' , 'r' , 'r' ), + + # + # glide + # + + ( u'w' , 'w' , 'w' ), + # see above ( u'j' , 'j' , 'j' ), + + # + # vowels: monophongs + # + + # front + ( u'i' , 'i' , 'i' ), + ( u'ɪ' , 'I' , 'I' ), + ( u'y' , 'y' , 'y' ), + ( u'ʏ' , 'Y' , 'Y' ), + ( u'e' , 'e' , 'e' ), + ( u'ø' , '2' , '2' ), + ( u'œ' , '9' , '9' ), + ( u'ɛ' , 'E' , 'E' ), + ( u'æ' , '{' , '{' ), + ( u'a' , 'a' , 'a' ), + + # central + ( u'ʌ' , 'V' , 'V' ), + ( u'ə' , '@' , '@' ), + ( u'ɐ' , '6' , '6' ), + ( u'ɜ' , '3' , 'r=' ), + + # back + ( u'u' , 'u' , 'u' ), + ( u'ʊ' , 'U' , 'U' ), + ( u'o' , 'o' , 'o' ), + ( u'ɔ' , 'O' , 'O' ), + ( u'ɑ' , 'A' , 'A' ), + ( u'ɒ' , 'Q' , 'Q' ), + + # diphtongs + + ( u'aɪ' , 'aI' , 'aI' ), + ( u'ɔɪ' , 'OI' , 'OI' ), + ( u'aʊ' , 'aU' , 'aU' ), + ( u'ɔʏ' , 'OY' , 'OY' ), + + # + # misc + # + ( u'ː' , ':' , ':' ), + ( u'-' , '-' , '-' ), + ( u'\'' , '\'' , '\'' ), + ] + +IPA_normalization = { + u':' : u'ː', + u'?' : u'ʔ', + u'ɾ' : u'ʁ', + u'ɡ' : u'g', + u'ŋ' : u'ɳ', + u' ' : None, + u'(' : None, + u')' : None, + u'\u02c8' : u'\'', + u'\u032f' : None, + u'\u0329' : None, + u'\u02cc' : None, + u'\u200d' : None, + u'\u0279' : None, + } + +XSAMPA_normalization = { + ' ': None, + '~': None, + '0': 'O', + ',': None, + } + +def _normalize (s, norm_table): + + buf = "" + + for c in s: + + if c in norm_table: + + x = norm_table[c] + if x: + buf += x + else: + buf += c + + return buf + +def _translate (graph, s, f_idx, t_idx, spaces=False): + + buf = "" + i = 0 + l = len(s) + + while i < l: + + found = False + + for pl in range(MAX_PHONEME_LENGTH, 0, -1): + + if i + pl > l: + continue + + substr = s[i : i+pl ] + + #print u"i: %s, pl: %d, substr: '%s'" % (i, pl, substr) + + for pe in big_phoneme_table: + p_f = pe[f_idx] + p_t = pe[t_idx] + + if substr == p_f: + buf += p_t + i += pl + if i l: + continue + + substr = s[i : i+pl ] + + #print u"i: %s, pl: %d, substr: '%s'" % (i, pl, substr) + + for pe in xs2xa_table: + p_f = pe[0] + p_t = pe[1] + + if substr == p_f: + if len(buf)>0: + buf += ' ' + buf += p_t + i += pl + found = True + break + + if found: + break + + if not found: + + p = s[i] + + msg = u"xsampa2xarpabet: graph:'%s' - s:'%s' Phoneme not found: '%s' (%d) '%s'" % (graph, s, p, ord(p), s[i:]) + + raise Exception (msg.encode('UTF8')) + + return buf + +class TestPhoneticAlphabets (unittest.TestCase): + + def setUp(self): + self.seq = range(10) + + def test_ipa(self): + + res = ipa2xsampa ("EISENBAHN", u"ˈaɪ̯zən̩ˌbaːn") + #print "res: %s" % res + self.assertEqual (res, "'aIz@nba:n") + + res = ipa2xsampa ("DIPHTONGTEST", u"aɪɔɪaʊɜ'") + #print "res: %s" % res + self.assertEqual (res, "aIOIaU3'") + + res = ipa2mary ("EISENBAHN", u"ˈaɪ̯zən̩ˌbaːn") + #print "res: %s" % res + self.assertEqual (res, "'aIz@nba:n") + + res = ipa2mary ("DIPHTONGTEST", u"aɪɔɪaʊɜ'") + #print "res: %s" % res + self.assertEqual (res, "aIOIaUr='") + + def test_xarpa(self): + + res = xsampa2xarpabet ("JAHRHUNDERTE", "ja:6-'hUn-d6-t@") + #print "res: %s" % res + self.assertEqual (res, "Y AAH EX HH UU N D EX T AX") + + res = xsampa2xarpabet ("ABGESCHRIEBEN", "'ap-g@-SRi:-b@n") + #print "res: %s" % res + self.assertEqual (res, "AH P G AX SH RR IIH B AX N") + + res = xsampa2xarpabet ("ZUGEGRIFFEN", "'tsu:-g@-gRI-f@n") + #print "res: %s" % res + self.assertEqual (res, "TS UUH G AX G RR IH F AX N") + + res = xsampa2xarpabet ("AUSLEGUNG", "'aU-sle:-gUN") + #print "res: %s" % res + self.assertEqual (res, "AW S L EEH G UU NG") + + def test_xarpa_unique(self): + + # all xarpa transcriptions have to be unique + + uniq_xs = set() + uniq_xa = set() + + for entry in xs2xa_table: + xs = entry[0] + xa = entry[1] + #print (u"xs: %s, xa: %s" % (xs, xa)).encode('utf8') + self.assertFalse (xa in uniq_xa) + uniq_xa.add(xa) + self.assertFalse (xs in uniq_xs) + uniq_xs.add(xs) + + + +if __name__ == "__main__": + + unittest.main() + -- cgit v1.3.1