Spaces:

projecte-aina
/

matxa-alvocat-tts-ca

Running

App Files Files Community

wetdog commited on Mar 6, 2024

Commit

daa90f5

1 Parent(s): 92df4f5

add models, configs and utils

Browse files

Files changed (14) hide show

README.md +3 -3
config_22khz.yaml +24 -0
matcha_hifigan_multispeaker_cat.onnx +3 -0
matcha_multispeaker_cat_opset_15.onnx +3 -0
mel_spec_22khz.onnx +3 -0
requirements.txt +6 -0
text/LICENSE +19 -0
text/__init__.py +64 -0
text/__pycache__/__init__.cpython-310.pyc +0 -0
text/__pycache__/cleaners.cpython-310.pyc +0 -0
text/__pycache__/symbols.cpython-310.pyc +0 -0
text/cleaners.py +133 -0
text/symbols.py +16 -0
utils.py +41 -0

README.md CHANGED Viewed

@@ -1,8 +1,8 @@
 ---
-title: Tts Vocos Onnx
-emoji: 🌍
 colorFrom: purple
-colorTo: gray
 sdk: docker
 pinned: false
 license: apache-2.0

 ---
+title: tts vocos Onnx Comparison
+emoji: 🐨
 colorFrom: purple
+colorTo: yellow
 sdk: docker
 pinned: false
 license: apache-2.0

config_22khz.yaml ADDED Viewed

	@@ -0,0 +1,24 @@

+feature_extractor:
+  class_path: vocos.feature_extractors.MelSpectrogramFeatures
+  init_args:
+    sample_rate: 22050
+    n_fft: 1024
+    hop_length: 256
+    n_mels: 80
+    padding: center
+backbone:
+  class_path: vocos.models.VocosBackbone
+  init_args:
+    input_channels: 80
+    dim: 512
+    intermediate_dim: 1536
+    num_layers: 8
+head:
+  class_path: vocos.heads.ISTFTHead
+  init_args:
+    dim: 512
+    n_fft: 1024
+    hop_length: 256
+    padding: center

matcha_hifigan_multispeaker_cat.onnx ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c5927b5a9a5f7890d4a8c353266ff00a1d9c4376eb1294020ffe43afa622b72f
+size 142073725

matcha_multispeaker_cat_opset_15.onnx ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e5b53370f69b8f4ca3d510634b644f6d815f34ee7a2944d0fb3a5588f6286b88
+size 102285286

mel_spec_22khz.onnx ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:15485817350df1e1cf50f75058497ec4b5273acb8903591bb41c6b5fb62daf2b
+size 53870258

requirements.txt ADDED Viewed

	@@ -0,0 +1,6 @@

+onnxruntime
+phonemizer
+torch
+unidecode
+gradio
+soundfile

text/LICENSE ADDED Viewed

	@@ -0,0 +1,19 @@

+Copyright (c) 2017 Keith Ito
+Permission is hereby granted, free of charge, to any person obtaining a copy
+of this software and associated documentation files (the "Software"), to deal
+in the Software without restriction, including without limitation the rights
+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+copies of the Software, and to permit persons to whom the Software is
+furnished to do so, subject to the following conditions:
+The above copyright notice and this permission notice shall be included in
+all copies or substantial portions of the Software.
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
+THE SOFTWARE.

text/__init__.py ADDED Viewed

	@@ -0,0 +1,64 @@

+""" from https://github.com/keithito/tacotron """
+from text import cleaners
+from text.symbols import symbols
+# Mappings from symbol to numeric ID and vice versa:
+_symbol_to_id = {s: i for i, s in enumerate(symbols)}
+_id_to_symbol = {i: s for i, s in enumerate(symbols)}
+def text_to_sequence(text, cleaner_names):
+    """Converts a string of text to a sequence of IDs corresponding to the symbols in the text.
+    Args:
+      text: string to convert to a sequence
+      cleaner_names: names of the cleaner functions to run the text through
+    Returns:
+      List of integers corresponding to the symbols in the text
+    """
+    sequence = []
+    clean_text = _clean_text(text, cleaner_names)
+    for symbol in clean_text:
+        if symbol in _symbol_to_id.keys():
+            symbol_id = _symbol_to_id[symbol]
+            sequence += [symbol_id]
+        else:
+            continue
+    return sequence
+def cleaned_text_to_sequence(cleaned_text):
+    """Converts a string of text to a sequence of IDs corresponding to the symbols in the text.
+    Args:
+      text: string to convert to a sequence
+    Returns:
+      List of integers corresponding to the symbols in the text
+    """
+    sequence = []
+    for symbol in cleaned_text:
+        if symbol in _symbol_to_id.keys():
+            symbol_id = _symbol_to_id[symbol]
+            sequence += [symbol_id]
+        else:
+            continue
+    return sequence
+def sequence_to_text(sequence):
+    """Converts a sequence of IDs back to a string"""
+    result = ""
+    for symbol_id in sequence:
+        s = _id_to_symbol[symbol_id]
+        result += s
+    return result
+def _clean_text(text, cleaner_names):
+    for name in cleaner_names:
+        cleaner = getattr(cleaners, name)
+        if not cleaner:
+            raise Exception("Unknown cleaner: %s" % name)
+        text = cleaner(text)
+    return text

text/__pycache__/__init__.cpython-310.pyc ADDED Viewed

Binary file (2.04 kB). View file

text/__pycache__/cleaners.cpython-310.pyc ADDED Viewed

Binary file (3.19 kB). View file

text/__pycache__/symbols.cpython-310.pyc ADDED Viewed

Binary file (693 Bytes). View file

text/cleaners.py ADDED Viewed

	@@ -0,0 +1,133 @@

+""" from https://github.com/keithito/tacotron """
+"""
+Cleaners are transformations that run over the input text at both training and eval time.
+Cleaners can be selected by passing a comma-delimited list of cleaner names as the "cleaners"
+hyperparameter. Some cleaners are English-specific. You'll typically want to use:
+  1. "english_cleaners" for English text
+  2. "transliteration_cleaners" for non-English text that can be transliterated to ASCII using
+     the Unidecode library (https://pypi.python.org/pypi/Unidecode)
+  3. "basic_cleaners" if you do not want to transliterate (in this case, you should also update
+     the symbols in symbols.py to match your data).
+"""
+import re
+from unidecode import unidecode
+from phonemizer import phonemize
+from phonemizer.backend import EspeakBackend
+backend = EspeakBackend("ca", preserve_punctuation=True, with_stress=True)
+backend_en = EspeakBackend("en-us", preserve_punctuation=True, with_stress=True)
+# Regular expression matching whitespace:
+_whitespace_re = re.compile(r"\s+")
+# List of (regular expression, replacement) pairs for abbreviations:
+_abbreviations = [
+    (re.compile("\\b%s\\." % x[0], re.IGNORECASE), x[1])
+    for x in [
+        ("mrs", "misess"),
+        ("mr", "mister"),
+        ("dr", "doctor"),
+        ("st", "saint"),
+        ("co", "company"),
+        ("jr", "junior"),
+        ("maj", "major"),
+        ("gen", "general"),
+        ("drs", "doctors"),
+        ("rev", "reverend"),
+        ("lt", "lieutenant"),
+        ("hon", "honorable"),
+        ("sgt", "sergeant"),
+        ("capt", "captain"),
+        ("esq", "esquire"),
+        ("ltd", "limited"),
+        ("col", "colonel"),
+        ("ft", "fort"),
+    ]
+]
+def expand_abbreviations(text):
+    for regex, replacement in _abbreviations:
+        text = re.sub(regex, replacement, text)
+    return text
+def expand_numbers(text):
+    return normalize_numbers(text)
+def lowercase(text):
+    return text.lower()
+def collapse_whitespace(text):
+    return re.sub(_whitespace_re, " ", text)
+def convert_to_ascii(text):
+    return unidecode(text)
+def basic_cleaners(text):
+    """Basic pipeline that lowercases and collapses whitespace without transliteration."""
+    text = lowercase(text)
+    text = collapse_whitespace(text)
+    return text
+def transliteration_cleaners(text):
+    """Pipeline for non-English text that transliterates to ASCII."""
+    text = convert_to_ascii(text)
+    text = lowercase(text)
+    text = collapse_whitespace(text)
+    return text
+def english_cleaners(text):
+    """Pipeline for English text, including abbreviation expansion."""
+    text = convert_to_ascii(text)
+    text = lowercase(text)
+    text = expand_abbreviations(text)
+    phonemes = phonemize(text, language="en-us", backend="espeak", strip=True)
+    phonemes = collapse_whitespace(phonemes)
+    return phonemes
+def english_cleaners2(text):
+    """Pipeline for English text, including abbreviation expansion. + punctuation + stress"""
+    text = convert_to_ascii(text)
+    text = lowercase(text)
+    text = expand_abbreviations(text)
+    phonemes = phonemize(
+        text,
+        language="en-us",
+        backend="espeak",
+        strip=True,
+        preserve_punctuation=True,
+        with_stress=True,
+    )
+    phonemes = collapse_whitespace(phonemes)
+    return phonemes
+def english_cleaners3(text):
+    """Pipeline for English text, including abbreviation expansion. + punctuation + stress"""
+    text = convert_to_ascii(text)
+    text = lowercase(text)
+    text = expand_abbreviations(text)
+    phonemes = backend_en.phonemize([text], strip=True)[0]
+    phonemes = collapse_whitespace(phonemes)
+    return phonemes
+def catalan_cleaners(text):
+    """Pipeline for catalan text, including punctuation + stress"""
+    #text = convert_to_ascii(text)
+    text = lowercase(text)
+    #text = expand_abbreviations(text)
+    phonemes = backend.phonemize([text], strip=True)[0]
+    phonemes = collapse_whitespace(phonemes)
+    return phonemes

text/symbols.py ADDED Viewed

	@@ -0,0 +1,16 @@

+""" from https://github.com/keithito/tacotron """
+"""
+Defines the set of symbols used in text input to the model.
+"""
+_pad = "_"
+_punctuation = ';:,.!?¡¿—…"«»“” '
+_letters = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz"
+_letters_ipa = "ɑɐɒæɓʙβɔɕçɗɖðʤəɘɚɛɜɝɞɟʄɡɠɢʛɦɧħɥʜɨɪʝɭɬɫɮʟɱɯɰŋɳɲɴøɵɸθœɶʘɹɺɾɻʀʁɽʂʃʈʧʉʊʋⱱʌɣɤʍχʎʏʑʐʒʔʡʕʢǀǁǂǃˈˌːˑʼʴʰʱʲʷˠˤ˞↓↑→↗↘'̩'ᵻ"
+# Export all symbols:
+symbols = [_pad] + list(_punctuation) + list(_letters) + list(_letters_ipa)
+# Special symbol ids
+SPACE_ID = symbols.index(" ")

utils.py ADDED Viewed

	@@ -0,0 +1,41 @@

+import json
+class HParams:
+    def __init__(self, **kwargs):
+        for k, v in kwargs.items():
+            if type(v) == dict:
+                v = HParams(**v)
+            self[k] = v
+    def keys(self):
+        return self.__dict__.keys()
+    def items(self):
+        return self.__dict__.items()
+    def values(self):
+        return self.__dict__.values()
+    def __len__(self):
+        return len(self.__dict__)
+    def __getitem__(self, key):
+        return getattr(self, key)
+    def __setitem__(self, key, value):
+        return setattr(self, key, value)
+    def __contains__(self, key):
+        return key in self.__dict__
+    def __repr__(self):
+        return self.__dict__.__repr__()
+def get_hparams_from_file(config_path):
+    with open(config_path, "r") as f:
+        data = f.read()
+    config = json.loads(data)
+    hparams = HParams(**config)
+    return hparams