Update Tacotron2 French model

Browse files

Files changed (4) hide show

README.md +93 -0
config.yml +86 -0
model.h5 +3 -0
processor.json +1 -0

README.md ADDED Viewed

	@@ -0,0 +1,93 @@

+---
+tags:
+- tensorflowtts
+- audio
+- text-to-speech
+- text-to-mel
+language: fr
+license: apache-2.0
+datasets:
+- synpaflex
+widget:
+- text: "Oh, je voudrais tant que tu te souviennes Des jours heureux quand nous étions amis"
+---
+# Tacotron 2 with Guided Attention trained on Synpaflex (Fr)
+This repository provides a pretrained [Tacotron2](https://arxiv.org/abs/1712.05884) trained with [Guided Attention](https://arxiv.org/abs/1710.08969) on Synpaflex dataset (Fr). For a detail of the model, we encourage you to read more about
+[TensorFlowTTS](https://github.com/TensorSpeech/TensorFlowTTS).
+## Install TensorFlowTTS
+First of all, please install TensorFlowTTS with the following command:
+```
+pip install TensorFlowTTS
+```
+### Converting your Text to Mel Spectrogram
+```python
+import numpy as np
+import soundfile as sf
+import yaml
+import tensorflow as tf
+from tensorflow_tts.inference import AutoProcessor
+from tensorflow_tts.inference import TFAutoModel
+processor = AutoProcessor.from_pretrained("tensorspeech/tts-tacotron2-synpaflex-fr")
+tacotron2 = TFAutoModel.from_pretrained("tensorspeech/tts-tacotron2-synpaflex-fr")
+text = "Oh, je voudrais tant que tu te souviennes Des jours heureux quand nous étions amis"
+input_ids = processor.text_to_sequence(text)
+decoder_output, mel_outputs, stop_token_prediction, alignment_history = tacotron2.inference(
+    input_ids=tf.expand_dims(tf.convert_to_tensor(input_ids, dtype=tf.int32), 0),
+    input_lengths=tf.convert_to_tensor([len(input_ids)], tf.int32),
+    speaker_ids=tf.convert_to_tensor([0], dtype=tf.int32),
+)
+```
+#### Referencing Tacotron 2
+```
+@article{DBLP:journals/corr/abs-1712-05884,
+  author    = {Jonathan Shen and
+               Ruoming Pang and
+               Ron J. Weiss and
+               Mike Schuster and
+               Navdeep Jaitly and
+               Zongheng Yang and
+               Zhifeng Chen and
+               Yu Zhang and
+               Yuxuan Wang and
+               R. J. Skerry{-}Ryan and
+               Rif A. Saurous and
+               Yannis Agiomyrgiannakis and
+               Yonghui Wu},
+  title     = {Natural {TTS} Synthesis by Conditioning WaveNet on Mel Spectrogram
+               Predictions},
+  journal   = {CoRR},
+  volume    = {abs/1712.05884},
+  year      = {2017},
+  url       = {http://arxiv.org/abs/1712.05884},
+  archivePrefix = {arXiv},
+  eprint    = {1712.05884},
+  timestamp = {Thu, 28 Nov 2019 08:59:52 +0100},
+  biburl    = {https://dblp.org/rec/journals/corr/abs-1712-05884.bib},
+  bibsource = {dblp computer science bibliography, https://dblp.org}
+}
+```
+#### Referencing TensorFlowTTS
+```
+@misc{TFTTS,
+    author = {Minh Nguyen, Alejandro Miguel Velasquez, Erogol, Kuan Chen, Dawid Kobus, Takuya Ebata,
+    Trinh Le and Yunchao He},
+    title = {TensorflowTTS},
+    year = {2020},
+    publisher = {GitHub},
+    journal = {GitHub repository},
+    howpublished = {\\url{https://github.com/TensorSpeech/TensorFlowTTS}},
+  }
+```

config.yml ADDED Viewed

	@@ -0,0 +1,86 @@

+# This is the hyperparameter configuration file for Tacotron2 v1.
+# Please make sure this is adjusted for the synpaflex dataset. If you want to
+# apply to the other dataset, you might need to carefully change some parameters.
+# This configuration performs 200k iters but 65k iters is enough to get a good models.
+###########################################################
+#                FEATURE EXTRACTION SETTING               #
+###########################################################
+hop_size: 256            # Hop size.
+format: "npy"
+###########################################################
+#              NETWORK ARCHITECTURE SETTING               #
+###########################################################
+model_type: "tacotron2"
+tacotron2_params:
+    dataset: synpaflex
+    embedding_hidden_size: 512
+    initializer_range: 0.02
+    embedding_dropout_prob: 0.1
+    n_speakers: 1
+    n_conv_encoder: 5
+    encoder_conv_filters: 512
+    encoder_conv_kernel_sizes: 5
+    encoder_conv_activation: 'relu'
+    encoder_conv_dropout_rate: 0.5
+    encoder_lstm_units: 256
+    n_prenet_layers: 2
+    prenet_units: 256
+    prenet_activation: 'relu'
+    prenet_dropout_rate: 0.5
+    n_lstm_decoder: 1
+    reduction_factor: 1
+    decoder_lstm_units: 1024
+    attention_dim: 128
+    attention_filters: 32
+    attention_kernel: 31
+    n_mels: 80
+    n_conv_postnet: 5
+    postnet_conv_filters: 512
+    postnet_conv_kernel_sizes: 5
+    postnet_dropout_rate: 0.1
+    attention_type: "lsa"
+###########################################################
+#                  DATA LOADER SETTING                    #
+###########################################################
+batch_size: 32              # Batch size for each GPU with assuming that gradient_accumulation_steps == 1.
+remove_short_samples: true # Whether to remove samples the length of which are less than batch_max_steps.
+allow_cache: true           # Whether to allow cache in dataset. If true, it requires cpu memory.
+mel_length_threshold: 32    # remove all targets has mel_length <= 32
+is_shuffle: true            # shuffle dataset after each epoch.
+use_fixed_shapes: true      # use_fixed_shapes for training (2x speed-up)
+                            # refer (https://github.com/dathudeptrai/TensorflowTTS/issues/34#issuecomment-642309118)
+###########################################################
+#             OPTIMIZER & SCHEDULER SETTING               #
+###########################################################
+optimizer_params:
+    initial_learning_rate: 0.001
+    end_learning_rate: 0.00001
+    decay_steps: 150000          # < train_max_steps is recommend.
+    warmup_proportion: 0.02
+    weight_decay: 0.001
+gradient_accumulation_steps: 1
+var_train_expr: null  # trainable variable expr (eg. 'embeddings|decoder_cell' )
+                      # must separate by |. if var_train_expr is null then we
+                      # training all variables.
+###########################################################
+#                    INTERVAL SETTING                     #
+###########################################################
+train_max_steps: 200000                 # Number of training steps.
+save_interval_steps: 2000               # Interval steps to save checkpoint.
+eval_interval_steps: 500                # Interval steps to evaluate the network.
+log_interval_steps: 200                 # Interval steps to record the training log.
+start_schedule_teacher_forcing: 200001  # don't need to apply schedule teacher forcing.
+start_ratio_value: 0.5                  # start ratio of scheduled teacher forcing.
+schedule_decay_steps: 50000             # decay step scheduled teacher forcing.
+end_ratio_value: 0.0                    # end ratio of scheduled teacher forcing.
+###########################################################
+#                     OTHER SETTING                       #
+###########################################################
+num_save_intermediate_results: 1  # Number of results to be saved as intermediate results.

model.h5 ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:7761e61d0dd3bbe9387ff6191d1507d9fd308d6117c8d3ec2f8151c6f9ea4470
+size 127842184

processor.json ADDED Viewed

	@@ -0,0 +1 @@

+ {"symbol_to_id": {"pad": 0, "!": 1, "/": 2, "'": 3, "(": 4, ")": 5, ",": 6, "-": 7, ".": 8, ":": 9, ";": 10, "?": 11, " ": 12, "A": 13, "B": 14, "C": 15, "D": 16, "E": 17, "F": 18, "G": 19, "H": 20, "I": 21, "J": 22, "K": 23, "L": 24, "M": 25, "N": 26, "O": 27, "P": 28, "Q": 29, "R": 30, "S": 31, "T": 32, "U": 33, "V": 34, "W": 35, "X": 36, "Y": 37, "Z": 38, "a": 39, "b": 40, "c": 41, "d": 42, "e": 43, "f": 44, "g": 45, "h": 46, "i": 47, "j": 48, "k": 49, "l": 50, "m": 51, "n": 52, "o": 53, "p": 54, "q": 55, "r": 56, "s": 57, "t": 58, "u": 59, "v": 60, "w": 61, "x": 62, "y": 63, "z": 64, "\u00e9": 65, "\u00e8": 66, "\u00e0": 67, "\u00f9": 68, "\u00e2": 69, "\u00ea": 70, "\u00ee": 71, "\u00f4": 72, "\u00fb": 73, "\u00e7": 74, "\u00e4": 75, "\u00eb": 76, "\u00ef": 77, "\u00f6": 78, "\u00fc": 79, "\u00ff": 80, "\u0153": 81, "\u00e6": 82, "eos": 83}, "id_to_symbol": {"0": "pad", "1": "!", "2": "/", "3": "'", "4": "(", "5": ")", "6": ",", "7": "-", "8": ".", "9": ":", "10": ";", "11": "?", "12": " ", "13": "A", "14": "B", "15": "C", "16": "D", "17": "E", "18": "F", "19": "G", "20": "H", "21": "I", "22": "J", "23": "K", "24": "L", "25": "M", "26": "N", "27": "O", "28": "P", "29": "Q", "30": "R", "31": "S", "32": "T", "33": "U", "34": "V", "35": "W", "36": "X", "37": "Y", "38": "Z", "39": "a", "40": "b", "41": "c", "42": "d", "43": "e", "44": "f", "45": "g", "46": "h", "47": "i", "48": "j", "49": "k", "50": "l", "51": "m", "52": "n", "53": "o", "54": "p", "55": "q", "56": "r", "57": "s", "58": "t", "59": "u", "60": "v", "61": "w", "62": "x", "63": "y", "64": "z", "65": "\u00e9", "66": "\u00e8", "67": "\u00e0", "68": "\u00f9", "69": "\u00e2", "70": "\u00ea", "71": "\u00ee", "72": "\u00f4", "73": "\u00fb", "74": "\u00e7", "75": "\u00e4", "76": "\u00eb", "77": "\u00ef", "78": "\u00f6", "79": "\u00fc", "80": "\u00ff", "81": "\u0153", "82": "\u00e6", "83": "eos"}, "speakers_map": {"synpaflex": 0}, "processor_name": "SynpaflexProcessor"}