coqui2

Sleeping

App Files Files Community

coqui2 / recipes /thorsten_DE /speedy_speech /train_speedy_speech.py

Adoetz

Upload 833 files

17ed7d8 verified 2 months ago

raw

history blame contribute delete

3.58 kB

	import os

	from trainer import Trainer, TrainerArgs

	from TTS.config import BaseAudioConfig, BaseDatasetConfig
	from TTS.tts.configs.speedy_speech_config import SpeedySpeechConfig
	from TTS.tts.datasets import load_tts_samples
	from TTS.tts.models.forward_tts import ForwardTTS
	from TTS.tts.utils.text.tokenizer import TTSTokenizer
	from TTS.utils.audio import AudioProcessor
	from TTS.utils.downloaders import download_thorsten_de

	output_path = os.path.dirname(os.path.abspath(__file__))
	dataset_config = BaseDatasetConfig(
	formatter="thorsten", meta_file_train="metadata.csv", path=os.path.join(output_path, "../thorsten-de/")
	)

	# download dataset if not already present
	if not os.path.exists(dataset_config.path):
	print("Downloading dataset")
	download_thorsten_de(os.path.split(os.path.abspath(dataset_config.path))[0])

	audio_config = BaseAudioConfig(
	sample_rate=22050,
	do_trim_silence=True,
	trim_db=60.0,
	signal_norm=False,
	mel_fmin=0.0,
	mel_fmax=8000,
	spec_gain=1.0,
	log_func="np.log",
	ref_level_db=20,
	preemphasis=0.0,
	)

	config = SpeedySpeechConfig(
	run_name="speedy_speech_thorsten-de",
	audio=audio_config,
	batch_size=32,
	eval_batch_size=16,
	num_loader_workers=4,
	num_eval_loader_workers=4,
	compute_input_seq_cache=True,
	run_eval=True,
	test_delay_epochs=-1,
	epochs=1000,
	min_audio_len=11050, # need to up min_audio_len to avois speedy speech error
	text_cleaner="phoneme_cleaners",
	use_phonemes=True,
	phoneme_language="de",
	phoneme_cache_path=os.path.join(output_path, "phoneme_cache"),
	precompute_num_workers=4,
	print_step=50,
	print_eval=False,
	mixed_precision=False,
	test_sentences=[
	"Es hat mich viel Zeit gekostet ein Stimme zu entwickeln, jetzt wo ich sie habe werde ich nicht mehr schweigen.",
	"Sei eine Stimme, kein Echo.",
	"Es tut mir Leid David. Das kann ich leider nicht machen.",
	"Dieser Kuchen ist großartig. Er ist so lecker und feucht.",
	"Vor dem 22. November 1963.",
	],
	max_seq_len=500000,
	output_path=output_path,
	datasets=[dataset_config],
	)

	# INITIALIZE THE AUDIO PROCESSOR
	# Audio processor is used for feature extraction and audio I/O.
	# It mainly serves to the dataloader and the training loggers.
	ap = AudioProcessor.init_from_config(config)

	# INITIALIZE THE TOKENIZER
	# Tokenizer is used to convert text to sequences of token IDs.
	# If characters are not defined in the config, default characters are passed to the config
	tokenizer, config = TTSTokenizer.init_from_config(config)

	# LOAD DATA SAMPLES
	# Each sample is a list of ```[text, audio_file_path, speaker_name]```
	# You can define your custom sample loader returning the list of samples.
	# Or define your custom formatter and pass it to the `load_tts_samples`.
	# Check `TTS.tts.datasets.load_tts_samples` for more details.
	train_samples, eval_samples = load_tts_samples(
	dataset_config,
	eval_split=True,
	eval_split_max_size=config.eval_split_max_size,
	eval_split_size=config.eval_split_size,
	)

	# init model
	model = ForwardTTS(config, ap, tokenizer)

	# INITIALIZE THE TRAINER
	# Trainer provides a generic API to train all the 🐸TTS models with all its perks like mixed-precision training,
	# distributed training, etc.
	trainer = Trainer(
	TrainerArgs(), config, output_path, model=model, train_samples=train_samples, eval_samples=eval_samples
	)

	# AND... 3,2,1... 🚀
	trainer.fit()