Upload text/ with huggingface_hub

Browse files

Files changed (7) hide show

text/LICENSE +19 -0
text/__init__.py +66 -0
text/__pycache__/__init__.cpython-38.pyc +0 -0
text/__pycache__/cleaners.cpython-38.pyc +0 -0
text/__pycache__/symbols.cpython-38.pyc +0 -0
text/cleaners.py +138 -0
text/symbols.py +25 -0

text/LICENSE ADDED Viewed

	@@ -0,0 +1,19 @@

+Copyright (c) 2017 Keith Ito
+Permission is hereby granted, free of charge, to any person obtaining a copy
+of this software and associated documentation files (the "Software"), to deal
+in the Software without restriction, including without limitation the rights
+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+copies of the Software, and to permit persons to whom the Software is
+furnished to do so, subject to the following conditions:
+The above copyright notice and this permission notice shall be included in
+all copies or substantial portions of the Software.
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
+THE SOFTWARE.

text/__init__.py ADDED Viewed

	@@ -0,0 +1,66 @@

+""" from https://github.com/keithito/tacotron """
+from text import cleaners
+from text.symbols import symbols,symbols_zh
+# Mappings from symbol to numeric ID and vice versa:
+# _symbol_to_id = {s: i for i, s in enumerate(symbols)}
+# _id_to_symbol = {i: s for i, s in enumerate(symbols)}
+chinese_mode = True
+if chinese_mode:
+  _symbol_to_id = {s: i for i, s in enumerate(symbols_zh)}
+  _id_to_symbol = {i: s for i, s in enumerate(symbols_zh)}
+else:
+  _symbol_to_id = {s: i for i, s in enumerate(symbols)}
+  _id_to_symbol = {i: s for i, s in enumerate(symbols)}
+def text_to_sequence(text, cleaner_names, ):
+  '''Converts a string of text to a sequence of IDs corresponding to the symbols in the text.
+    Args:
+      text: string to convert to a sequence
+      cleaner_names: names of the cleaner functions to run the text through
+    Returns:
+      List of integers corresponding to the symbols in the text
+  '''
+  sequence = []
+  clean_text = _clean_text(text, cleaner_names)
+  for symbol in clean_text:
+    if symbol not in _symbol_to_id.keys():
+      coutinue
+    symbol_id = _symbol_to_id[symbol]
+    sequence += [symbol_id]
+  return sequence
+def cleaned_text_to_sequence(cleaned_text, chinese_mode=True):
+  '''Converts a string of text to a sequence of IDs corresponding to the symbols in the text.
+    Args:
+      text: string to convert to a sequence
+    Returns:
+      List of integers corresponding to the symbols in the text
+  '''
+  # if chinese_mode:
+  #   sequence = [_symbol_to_id_zh[symbol] for symbol in cleaned_text]
+  # else:
+  sequence = [_symbol_to_id[symbol] for symbol in cleaned_text]
+  return sequence
+def sequence_to_text(sequence):
+  '''Converts a sequence of IDs back to a string'''
+  result = ''
+  for symbol_id in sequence:
+    s = _id_to_symbol[symbol_id]
+    result += s
+  return result
+def _clean_text(text, cleaner_names):
+  for name in cleaner_names:
+    cleaner = getattr(cleaners, name)
+    if not cleaner:
+      raise Exception('Unknown cleaner: %s' % name)
+    text = cleaner(text)
+  return text

text/__pycache__/__init__.cpython-38.pyc ADDED Viewed

Binary file (2.42 kB). View file

text/__pycache__/cleaners.cpython-38.pyc ADDED Viewed

Binary file (3.82 kB). View file

text/__pycache__/symbols.cpython-38.pyc ADDED Viewed

Binary file (831 Bytes). View file

text/cleaners.py ADDED Viewed

	@@ -0,0 +1,138 @@

+""" from https://github.com/keithito/tacotron """
+'''
+Cleaners are transformations that run over the input text at both training and eval time.
+Cleaners can be selected by passing a comma-delimited list of cleaner names as the "cleaners"
+hyperparameter. Some cleaners are English-specific. You'll typically want to use:
+  1. "english_cleaners" for English text
+  2. "transliteration_cleaners" for non-English text that can be transliterated to ASCII using
+     the Unidecode library (https://pypi.python.org/pypi/Unidecode)
+  3. "basic_cleaners" if you do not want to transliterate (in this case, you should also update
+     the symbols in symbols.py to match your data).
+'''
+import re
+from unidecode import unidecode
+from phonemizer import phonemize
+from pypinyin import Style, pinyin
+from pypinyin.style._utils import get_finals, get_initials
+# Regular expression matching whitespace:
+_whitespace_re = re.compile(r'\s+')
+# List of (regular expression, replacement) pairs for abbreviations:
+_abbreviations = [(re.compile('\\b%s\\.' % x[0], re.IGNORECASE), x[1]) for x in [
+  ('mrs', 'misess'),
+  ('mr', 'mister'),
+  ('dr', 'doctor'),
+  ('st', 'saint'),
+  ('co', 'company'),
+  ('jr', 'junior'),
+  ('maj', 'major'),
+  ('gen', 'general'),
+  ('drs', 'doctors'),
+  ('rev', 'reverend'),
+  ('lt', 'lieutenant'),
+  ('hon', 'honorable'),
+  ('sgt', 'sergeant'),
+  ('capt', 'captain'),
+  ('esq', 'esquire'),
+  ('ltd', 'limited'),
+  ('col', 'colonel'),
+  ('ft', 'fort'),
+]]
+def expand_abbreviations(text):
+  for regex, replacement in _abbreviations:
+    text = re.sub(regex, replacement, text)
+  return text
+def expand_numbers(text):
+  return normalize_numbers(text)
+def lowercase(text):
+  return text.lower()
+def collapse_whitespace(text):
+  return re.sub(_whitespace_re, ' ', text)
+def convert_to_ascii(text):
+  return unidecode(text)
+def basic_cleaners(text):
+  '''Basic pipeline that lowercases and collapses whitespace without transliteration.'''
+  text = lowercase(text)
+  text = collapse_whitespace(text)
+  return text
+def transliteration_cleaners(text):
+  '''Pipeline for non-English text that transliterates to ASCII.'''
+  text = convert_to_ascii(text)
+  text = lowercase(text)
+  text = collapse_whitespace(text)
+  return text
+def english_cleaners(text):
+  '''Pipeline for English text, including abbreviation expansion.'''
+  text = convert_to_ascii(text)
+  text = lowercase(text)
+  text = expand_abbreviations(text)
+  phonemes = phonemize(text, language='en-us', backend='espeak', strip=True)
+  phonemes = collapse_whitespace(phonemes)
+  return phonemes
+def english_cleaners2(text):
+  '''Pipeline for English text, including abbreviation expansion. + punctuation + stress'''
+  text = convert_to_ascii(text)
+  text = lowercase(text)
+  text = expand_abbreviations(text)
+  phonemes = phonemize(text, language='en-us', backend='espeak', strip=True, preserve_punctuation=True, with_stress=True)
+  phonemes = collapse_whitespace(phonemes)
+  return phonemes
+def chinese_cleaners1(text):
+    from pypinyin import Style, pinyin
+    phones = [phone[0] for phone in pinyin(text, style=Style.TONE3)]
+    return ' '.join(phones)
+def chinese_cleaners2(text):
+  phones = [
+      p
+      for phone in pinyin(text, style=Style.TONE3)
+      for p in [
+          get_initials(phone[0], strict=True),
+          get_finals(phone[0][:-1], strict=True) + phone[0][-1]
+          if phone[0][-1].isdigit()
+          else get_finals(phone[0], strict=True)
+          if phone[0][-1].isalnum()
+          else phone[0],
+      ]
+      # Remove the case of individual tones as a phoneme
+      if len(p) != 0 and not p.isdigit()
+  ]
+  return phones
+  # return phonemes
+if __name__ == '__main__':
+  res = chinese_cleaners2('这是语音测试！')
+  print(res)
+  res = chinese_cleaners1('"第一，南京不是发展的不行，是大家对他期望很高，')
+  print(res)
+  res = english_cleaners2('this is a club test for one train.GDP')
+  print(res)

text/symbols.py ADDED Viewed

	@@ -0,0 +1,25 @@

+""" from https://github.com/keithito/tacotron """
+'''
+Defines the set of symbols used in text input to the model.
+'''
+_pad        = '_'
+_punctuation = ';:,.!?¡¿—…"«»“” '
+_punctuation_zh = '；：，。！？-“”《》、（）ＢＰ…—~.\·『』・ '
+_letters = 'ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz'
+_numbers = '1234567890'
+_others = ''
+_letters_ipa = "ɑɐɒæɓʙβɔɕçɗɖðʤəɘɚɛɜɝɞɟʄɡɠɢʛɦɧħɥʜɨɪʝɭɬɫɮʟɱɯɰŋɳɲɴøɵɸθœɶʘɹɺɾɻʀʁɽʂʃʈʧʉʊʋⱱʌɣɤʍχʎʏʑʐʒʔʡʕʢǀǁǂǃˈˌːˑʼʴʰʱʲʷˠˤ˞↓↑→↗↘'̩'ᵻ"
+# Export all symbols:
+symbols = [_pad] + list(_punctuation) + list(_letters) + list(_letters_ipa)
+symbols_zh = [_pad] + list(_punctuation_zh) +  list(_letters) + list(_numbers)
+# Special symbol ids
+SPACE_ID = symbols.index(" ")