before changes
This commit is contained in:
@@ -0,0 +1,121 @@
|
||||
{
|
||||
"model": "speaker_encoder",
|
||||
"run_name": "speaker_encoder",
|
||||
"run_description": "resnet speaker encoder trained with commonvoice all languages dev and train, Voxceleb 1 dev and Voxceleb 2 dev",
|
||||
"epochs": 100000,
|
||||
"batch_size": null,
|
||||
"eval_batch_size": null,
|
||||
"mixed_precision": false,
|
||||
"run_eval": true,
|
||||
"test_delay_epochs": 0,
|
||||
"print_eval": false,
|
||||
"print_step": 50,
|
||||
"tb_plot_step": 100,
|
||||
"tb_model_param_stats": false,
|
||||
"save_step": 1000,
|
||||
"checkpoint": true,
|
||||
"keep_all_best": false,
|
||||
"keep_after": 10000,
|
||||
"num_loader_workers": 8,
|
||||
"num_val_loader_workers": 0,
|
||||
"use_noise_augment": false,
|
||||
"output_path": "../checkpoints/speaker_encoder/language_balanced/normalized/angleproto-4-samples-by-speakers/",
|
||||
"distributed_backend": "nccl",
|
||||
"distributed_url": "tcp://localhost:54321",
|
||||
"audio": {
|
||||
"fft_size": 512,
|
||||
"win_length": 400,
|
||||
"hop_length": 160,
|
||||
"frame_shift_ms": null,
|
||||
"frame_length_ms": null,
|
||||
"stft_pad_mode": "reflect",
|
||||
"sample_rate": 16000,
|
||||
"resample": false,
|
||||
"preemphasis": 0.97,
|
||||
"ref_level_db": 20,
|
||||
"do_sound_norm": false,
|
||||
"do_trim_silence": false,
|
||||
"trim_db": 60,
|
||||
"power": 1.5,
|
||||
"griffin_lim_iters": 60,
|
||||
"num_mels": 64,
|
||||
"mel_fmin": 0.0,
|
||||
"mel_fmax": 8000.0,
|
||||
"spec_gain": 20,
|
||||
"signal_norm": false,
|
||||
"min_level_db": -100,
|
||||
"symmetric_norm": false,
|
||||
"max_norm": 4.0,
|
||||
"clip_norm": false,
|
||||
"stats_path": null,
|
||||
"do_rms_norm": true,
|
||||
"db_level": -27.0
|
||||
},
|
||||
"datasets": [
|
||||
{
|
||||
"name": "voxceleb2",
|
||||
"path": "/workspace/scratch/ecasanova/datasets/VoxCeleb/vox2_dev_aac/",
|
||||
"meta_file_train": null,
|
||||
"ununsed_speakers": null,
|
||||
"meta_file_val": null,
|
||||
"meta_file_attn_mask": "",
|
||||
"language": "voxceleb"
|
||||
}
|
||||
],
|
||||
"model_params": {
|
||||
"model_name": "resnet",
|
||||
"input_dim": 64,
|
||||
"use_torch_spec": true,
|
||||
"log_input": true,
|
||||
"proj_dim": 512
|
||||
},
|
||||
"audio_augmentation": {
|
||||
"p": 0.5,
|
||||
"rir": {
|
||||
"rir_path": "/workspace/store/ecasanova/ComParE/RIRS_NOISES/simulated_rirs/",
|
||||
"conv_mode": "full"
|
||||
},
|
||||
"additive": {
|
||||
"sounds_path": "/workspace/store/ecasanova/ComParE/musan/",
|
||||
"speech": {
|
||||
"min_snr_in_db": 13,
|
||||
"max_snr_in_db": 20,
|
||||
"min_num_noises": 1,
|
||||
"max_num_noises": 1
|
||||
},
|
||||
"noise": {
|
||||
"min_snr_in_db": 0,
|
||||
"max_snr_in_db": 15,
|
||||
"min_num_noises": 1,
|
||||
"max_num_noises": 1
|
||||
},
|
||||
"music": {
|
||||
"min_snr_in_db": 5,
|
||||
"max_snr_in_db": 15,
|
||||
"min_num_noises": 1,
|
||||
"max_num_noises": 1
|
||||
}
|
||||
},
|
||||
"gaussian": {
|
||||
"p": 0.0,
|
||||
"min_amplitude": 0.0,
|
||||
"max_amplitude": 1e-05
|
||||
}
|
||||
},
|
||||
"storage": {
|
||||
"sample_from_storage_p": 0.5,
|
||||
"storage_size": 40
|
||||
},
|
||||
"max_train_step": 1000000,
|
||||
"loss": "angleproto",
|
||||
"grad_clip": 3.0,
|
||||
"lr": 0.0001,
|
||||
"lr_decay": false,
|
||||
"warmup_steps": 4000,
|
||||
"wd": 1e-06,
|
||||
"steps_plot_stats": 100,
|
||||
"num_speakers_in_batch": 100,
|
||||
"num_utters_per_speaker": 4,
|
||||
"skip_speakers": true,
|
||||
"voice_len": 2.0
|
||||
}
|
||||
Binary file not shown.
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,226 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import sys
|
||||
import os
|
||||
import argparse
|
||||
|
||||
import torch
|
||||
|
||||
# load coqui-ai/trainer libraries
|
||||
from trainer import Trainer, TrainerArgs
|
||||
|
||||
# load coqui-ai/TTS libraries
|
||||
from TTS.tts.configs.shared_configs import BaseDatasetConfig
|
||||
from TTS.tts.configs.vits_config import VitsConfig
|
||||
from TTS.tts.datasets import load_tts_samples
|
||||
from TTS.tts.models.vits import Vits, VitsArgs, VitsAudioConfig
|
||||
|
||||
|
||||
#
|
||||
# parse command line arguments
|
||||
#
|
||||
def parse_cmdline_args():
|
||||
parser = argparse.ArgumentParser(
|
||||
description = "Code to clone a voice from a given set of voice samples and a multi-speaker baseline model")
|
||||
parser.add_argument("--baseline_model_path", type = str, required = True,
|
||||
help = "Path to multi-speaker baseline model (VITS model)")
|
||||
parser.add_argument("--speaker_dataset_path", type = str, required = True,
|
||||
help = "Path to voice cloning dataset")
|
||||
parser.add_argument("--speaker_embeddings_path", type = str, required = True,
|
||||
help = "Path to speaker's embeddings file")
|
||||
parser.add_argument("--output_path", type = str, default = "results/cloned-voices",
|
||||
help = "Path to store trained / generated assets")
|
||||
parser.add_argument("--batch_size", type = int, default = 96, # 96 is suitable for AWS g5 instances
|
||||
help = "Batch size for training run")
|
||||
parser.add_argument("--max_epochs", type = int, default = 200, # 200 for batch_size 96 (with the 22.050 sampling rate multi-speaker model
|
||||
help = "Maximum number of epochs for training run") # 2000 for batch_size 64 and 1500 for batch_size 96 (with the initial 16k sampling rate VCTK 0.80 model)
|
||||
parser.add_argument("--use_cpu", default = False, action = "store_true", # untested!!!
|
||||
help = "Signal that CPU should be used even if a CUDA-device is available")
|
||||
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
|
||||
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
|
||||
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
#
|
||||
# main training method (voice cloning)
|
||||
#
|
||||
def main(args):
|
||||
if args.output_format == "txt":
|
||||
print("Commencing training of a new multi-speaker potion-voice baseline model:")
|
||||
print("")
|
||||
print(" + Baseline multi-speaker model path: {}" . format(args.baseline_model_path))
|
||||
print(" + Voice training dataset path : {}" . format(args.speaker_dataset_path))
|
||||
print(" + Speaker embeddings path : {}" . format(args.speaker_embeddings_path))
|
||||
print(" + Output path : {}" . format(args.output_path))
|
||||
print(" + Batch size : {}" . format(args.batch_size))
|
||||
print(" + Training runs (max epochs) : {}" . format(args.max_epochs))
|
||||
print("")
|
||||
|
||||
# determine whether CUDA support is available and set device parameters accordingly
|
||||
use_cuda = torch.cuda.is_available()
|
||||
if args.output_format == "txt":
|
||||
print(" + CUDA availability : {}" . format(use_cuda))
|
||||
|
||||
if args.use_cpu:
|
||||
device = "cpu"
|
||||
device_torch = False
|
||||
elif use_cuda:
|
||||
device = "cuda"
|
||||
device_torch = torch.device("cuda")
|
||||
else:
|
||||
device = "cpu"
|
||||
device_torch = False
|
||||
if args.output_format == "txt":
|
||||
print(" + Compute device used : {}" . format(device))
|
||||
print("")
|
||||
|
||||
# define training data set
|
||||
dataset_config = BaseDatasetConfig(formatter = "vctk_old", language = "en-us", path = args.speaker_dataset_path)
|
||||
|
||||
# set VITS training parameters
|
||||
audio_config = VitsAudioConfig(
|
||||
sample_rate = 22050,
|
||||
win_length = 1024,
|
||||
hop_length = 256,
|
||||
num_mels = 80,
|
||||
mel_fmin = 0,
|
||||
mel_fmax = None,
|
||||
)
|
||||
|
||||
vitsArgs = VitsArgs(
|
||||
use_speaker_embedding = False,
|
||||
use_d_vector_file = True,
|
||||
d_vector_file = [args.speaker_embeddings_path],
|
||||
d_vector_dim = 512,
|
||||
num_layers_text_encoder = 10
|
||||
)
|
||||
|
||||
config = VitsConfig(
|
||||
model_args = vitsArgs,
|
||||
audio = audio_config,
|
||||
run_name = "vits_potion_clone",
|
||||
use_speaker_embedding = False,
|
||||
use_d_vector_file = True,
|
||||
d_vector_file = [args.speaker_embeddings_path],
|
||||
d_vector_dim = 512,
|
||||
batch_size = args.batch_size,
|
||||
eval_batch_size = 8,
|
||||
batch_group_size = 0, # changing this to 5 (VITS training default) slows training down, but doesn't have any positive training effects
|
||||
num_loader_workers = 4,
|
||||
num_eval_loader_workers = 4,
|
||||
run_eval = True,
|
||||
eval_split_size = 2, # fix size of eval dataset (default 1% approach requires at least 100 voice samples!)
|
||||
test_delay_epochs = -1,
|
||||
epochs = args.max_epochs,
|
||||
text_cleaner = "english_cleaners",
|
||||
use_phonemes = False,
|
||||
phoneme_language = "en-us",
|
||||
phoneme_cache_path = os.path.join(args.output_path, "phoneme_cache"),
|
||||
compute_input_seq_cache = True,
|
||||
print_step = 50,
|
||||
print_eval = True,
|
||||
mixed_precision = True,
|
||||
max_text_len = 325,
|
||||
output_path = args.output_path,
|
||||
|
||||
save_checkpoints = True,
|
||||
save_step = 200,
|
||||
|
||||
datasets = [dataset_config],
|
||||
cudnn_benchmark = False,
|
||||
#characters = {
|
||||
# "pad": "_",
|
||||
# "eos": "&",
|
||||
# "bos": "*",
|
||||
# "characters": "!¡'(),-.:;¿?abcdefghijklmnopqrstuvwxyz «°±µ»$%&‘’‚“`”„",
|
||||
# "punctuations": "!¡'(),-.:;¿? ",
|
||||
# "phonemes": None,
|
||||
# "unique": True
|
||||
#},
|
||||
test_sentences = [
|
||||
["It took me quite a long time to develop a voice, and now that I have it I'm not going to be silent."],
|
||||
["Be a voice, not an echo."],
|
||||
["I'm sorry Dave. I'm afraid I can't do that."],
|
||||
["This cake is great. It's so delicious and moist."],
|
||||
["Prior to November 22, 1963."],
|
||||
["Hey! Sandra."],
|
||||
["Hey! Andrew."],
|
||||
["Hey, Michelle."],
|
||||
["Hey! George."],
|
||||
["Hey there, Rachel."]
|
||||
]
|
||||
)
|
||||
|
||||
# load training samples
|
||||
train_samples, eval_samples = load_tts_samples(config.datasets, eval_split = True, eval_split_max_size = config.eval_split_max_size, eval_split_size = config.eval_split_size)
|
||||
|
||||
# init VITS model
|
||||
model = Vits.init_from_config(config)
|
||||
|
||||
# init voice cloning
|
||||
trainer = Trainer(
|
||||
TrainerArgs(restore_path = args.baseline_model_path, use_ddp = False),
|
||||
config,
|
||||
args.output_path,
|
||||
model = model,
|
||||
train_samples = train_samples,
|
||||
eval_samples = eval_samples
|
||||
)
|
||||
|
||||
# trigger voice cloning (aka single speaker training)
|
||||
try:
|
||||
trainer.fit()
|
||||
except (KeyboardInterrupt, SystemExit):
|
||||
print("Training stopped manually (via keyboard interrupt)! Bye.")
|
||||
exit(0)
|
||||
|
||||
# determine required adjustment for speech synthesizing (i.e., the scaling factor for the duration predictor)
|
||||
# take the duration of the test sentence and calculate the difference to corresponding reference samples
|
||||
# set config.model_args["length_scale"] accordingly and save the updated config asset
|
||||
|
||||
# exit gracefully
|
||||
if args.output_format == "txt":
|
||||
print("")
|
||||
print("Completed voice cloning. The resulting model(s) can be found at:")
|
||||
print(" --> {}" . format(args.output_path))
|
||||
print("")
|
||||
print("Done; bye.")
|
||||
print("")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# parse command line arguments
|
||||
args = parse_cmdline_args()
|
||||
|
||||
# clear command line arguments to avoid triggering argparse features part of Trainer / coqpit imports
|
||||
# Traceback (most recent call last):
|
||||
# File "train_multispeaker_baseline_model.py", line 208, in <module>
|
||||
# main(args)
|
||||
# File "train_multispeaker_baseline_model.py", line 177, in main
|
||||
# trainer = Trainer(
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 360, in __init__
|
||||
# config, new_fields = self.init_training(args, coqpit_overrides, config)
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 594, in init_training
|
||||
# config.parse_known_args(coqpit_overrides, relaxed_parser=True)
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 843, in parse_known_args
|
||||
# parser = self.init_argparse(arg_prefix=arg_prefix, relaxed_parser=relaxed_parser)
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 881, in init_argparse
|
||||
# _init_argparse(
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 529, in _init_argparse
|
||||
# parser = _init_argparse(
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 550, in _init_argparse
|
||||
# return default.init_argparse(
|
||||
# AttributeError: 'str' object has no attribute 'init_argparse'
|
||||
sys.argv = [sys.argv[0]]
|
||||
|
||||
# ensure the output path exists
|
||||
os.makedirs(args.output_path, exist_ok = True)
|
||||
|
||||
main(args)
|
||||
|
||||
|
||||
### USAGE:
|
||||
### $ python3 TTS/TTS/bin/resample.py --input_dir voice_dataset_path/person_82/wav48/1 --output_sr 16000
|
||||
### $ python3 clone_voice.py [with argument]
|
||||
@@ -0,0 +1,764 @@
|
||||
# potion-voice **voice-cloning** *Installation and Usage Guide*
|
||||
|
||||
In this guide, you will find more detailed instructions and examples for the following tasks:
|
||||
|
||||
+ Setting up a new AWS GPU-backed EC2 instance suitable for training new potion-voice models;
|
||||
+ Setting up software environment and (optionally) prepare data sets for training new potion-voice models;
|
||||
+ Training and evaluating new potion-voice models; and
|
||||
+ Usage examples for voice cloning and speech synthesizing.
|
||||
|
||||
## Set Up AWS GPU-backed Compute Node (non-production)
|
||||
|
||||
1. Set up baseline & connect to remote node:
|
||||
|
||||
+ GPU-enabled Compute Node (e.g., g5.2xlarge by default)
|
||||
+ We recommend a GPU-enabled Compute Node with 256GB root partition (volume type: gp3; 64GB for swapfile) and 512GB secondary SDD holding all dev / data files)
|
||||
+ Inbound ports: SSH and TensorBoard (e.g., port 6006)
|
||||
+ Ubuntu 22.04 LTS (Server) Installation
|
||||
+ SSH into the EC2 instance
|
||||
|
||||
1. Secure / update baseline
|
||||
|
||||
```sh
|
||||
$ sudo apt-get update
|
||||
$ sudo apt-get upgrade
|
||||
$ sudo apt-get install linux-aws linux-headers-aws linux-image-aws
|
||||
```
|
||||
|
||||
1. Disable unattended upgrades. Enter the below command and select 'No'. These Upgrades might cause version mismatch between nvidia-drivers and cuda.
|
||||
|
||||
```sh
|
||||
$ sudo dpkg-reconfigure -plow unattended-upgrades
|
||||
Replacing config file /etc/apt/apt.conf.d/20auto-upgrades with new version
|
||||
```
|
||||
|
||||
1. Set up secondary disk (used as dev / data volume)
|
||||
|
||||
```sh
|
||||
$ sudo lsblk
|
||||
|
||||
NAME MAJ:MIN RM SIZE RO TYPE MOUNTPOINT
|
||||
[...]
|
||||
nvme1n1 259:0 0 500G 0 disk
|
||||
[...]
|
||||
|
||||
$ sudo mkfs -t ext4 /dev/nvme1n1
|
||||
|
||||
mke2fs 1.45.5 (07-Jan-2020)
|
||||
Creating filesystem with 524288000 4k blocks and 131072000 inodes
|
||||
Filesystem UUID: 90327770-ba4d-4003-9136-964b4388ffb6
|
||||
Superblock backups stored on blocks:
|
||||
32768, 98304, 163840, 229376, 294912, 819200, 884736, 1605632, 2654208,
|
||||
4096000, 7962624, 11239424, 20480000, 23887872, 71663616, 78675968,
|
||||
102400000, 214990848, 512000000
|
||||
|
||||
Allocating group tables: done
|
||||
Writing inode tables: done
|
||||
Creating journal (262144 blocks): done
|
||||
Writing superblocks and filesystem accounting information: done
|
||||
|
||||
$ mkdir DEV_PATH
|
||||
```
|
||||
|
||||
+ Edit `/etc/fstab` and add
|
||||
|
||||
```txt
|
||||
/dev/nvme1n1 DEV_PATH ext4 defaults,nofail 0 2
|
||||
```
|
||||
|
||||
```sh
|
||||
$ sudo mount -a
|
||||
$ sudo chown -R ubuntu:ubuntu DEV_PATH
|
||||
$ mkdir DEV_PATH/data
|
||||
```
|
||||
|
||||
1. Create a swap file (training is memory intensive; so, add a swap file!)
|
||||
|
||||
+ Use the `dd` command to create a swap file on the root file system
|
||||
+ Note: The size of the swap file is the block size option multiplied by the count option in the dd command. Adjust these values to determine the desired swap file size.
|
||||
+ Note: The block size you specify should be less than the available memory on the instance or you receive a "memory exhausted" error.
|
||||
|
||||
+ Set up the swap file (of size 64 GB [512 MB x 128]).
|
||||
|
||||
```sh
|
||||
$ sudo dd if=/dev/zero of=/swapfile bs=512M count=128
|
||||
128+0 records in
|
||||
128+0 records out
|
||||
68719476736 bytes (69 GB, 64 GiB) copied, 336.416 s, 204 MB/s
|
||||
```
|
||||
|
||||
+ Update the read and write permissions for the swap file:
|
||||
|
||||
```sh
|
||||
$ sudo chmod 600 /swapfile
|
||||
```
|
||||
|
||||
+ Set up a Linux swap area:
|
||||
|
||||
```sh
|
||||
$ sudo mkswap /swapfile
|
||||
Setting up swapspace version 1, size = 64 GiB
|
||||
no label, UUID=1dfc20ce-ed64-4e69-8fa7-a800bbea4617
|
||||
```
|
||||
|
||||
+ Make the swap file available for immediate use by adding the swap file to swap space:
|
||||
|
||||
```sh
|
||||
$ sudo swapon /swapfile
|
||||
```
|
||||
|
||||
+ Verify that the procedure was successful:
|
||||
|
||||
```sh
|
||||
$ sudo swapon -s
|
||||
Filename Type Size Used Priority
|
||||
/swapfile file 67108860 0 -2
|
||||
```
|
||||
|
||||
+ Enable the swap file at boot time by editing the `/etc/fstab` file. Add the following new line at the end of the file:
|
||||
|
||||
```txt
|
||||
/swapfile swap swap defaults 0 0
|
||||
```
|
||||
|
||||
1. Install NVIDIA drivers / CUDA support (pytorch required version 11.6 or 12)
|
||||
|
||||
```sh
|
||||
wget https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.0-1_all.deb
|
||||
sudo dpkg -i cuda-keyring_1.0-1_all.deb
|
||||
sudo apt-get update
|
||||
sudo apt-get -y install cuda-12-0
|
||||
```
|
||||
|
||||
+ Reboot the instance and ensure all drivers load automatically
|
||||
|
||||
```sh
|
||||
$ sudo reboot
|
||||
```
|
||||
|
||||
+ Reconnect to the instance and verify NVIDIA drivers / CUDA support are as expected
|
||||
|
||||
```sh
|
||||
$ nvidia-smi
|
||||
|
||||
Tue Jan 17 08:20:53 2023
|
||||
+-----------------------------------------------------------------------------+
|
||||
| NVIDIA-SMI 525.60.13 Driver Version: 525.60.13 CUDA Version: 12.0 |
|
||||
|-------------------------------+----------------------+----------------------+
|
||||
| GPU Name Persistence-M| Bus-Id Disp.A | Volatile Uncorr. ECC |
|
||||
| Fan Temp Perf Pwr:Usage/Cap| Memory-Usage | GPU-Util Compute M. |
|
||||
| | | MIG M. |
|
||||
|===============================+======================+======================|
|
||||
| 0 NVIDIA A10G On | 00000000:00:1E.0 Off | 0 |
|
||||
| 0% 19C P8 16W / 300W | 0MiB / 23028MiB | 0% Default |
|
||||
| | | N/A |
|
||||
+-------------------------------+----------------------+----------------------+
|
||||
|
||||
+-----------------------------------------------------------------------------+
|
||||
| Processes: |
|
||||
| GPU GI CI PID Type Process name GPU Memory |
|
||||
| ID ID Usage |
|
||||
|=============================================================================|
|
||||
| No running processes found |
|
||||
+-----------------------------------------------------------------------------+
|
||||
```
|
||||
|
||||
## Set Up Software Environment
|
||||
|
||||
1. Set up Python 3 (v3.10) development environment
|
||||
|
||||
```sh
|
||||
$ sudo apt-get install python3-dev python3-pip python3-wheel python3-venv
|
||||
```
|
||||
|
||||
1. Set up Phoneme back-end
|
||||
|
||||
```sh
|
||||
$ sudo apt-get install espeak-ng espeak-ng-espeak
|
||||
```
|
||||
|
||||
1. Set up required tools / standard dependencies
|
||||
|
||||
```sh
|
||||
$ sudo apt-get install ffmpeg unzip git
|
||||
```
|
||||
|
||||
1. Set up AWS Command Line Interface
|
||||
|
||||
```sh
|
||||
$ sudo apt-get install awscli
|
||||
$ aws configure
|
||||
|
||||
AWS Access Key ID [None]: xxxxxxxxxx
|
||||
AWS Secret Access Key [None]: yyyyyyyyyy
|
||||
Default region name [None]: us-west-2
|
||||
Default output format [None]: json
|
||||
|
||||
$ aws configure set default.s3.max_concurrent_requests 50
|
||||
```
|
||||
|
||||
1. (dev install only) Copy and extract training data sets from AWS
|
||||
|
||||
```sh
|
||||
$ cd DEV_PATH/data
|
||||
|
||||
### VCTK v 0.92
|
||||
$ aws s3 cp s3://potion-datasets/VCTK/VCTK-Corpus-0.92/VCTK-Corpus-0.92.tgz .
|
||||
download: s3://potion-datasets/VCTK/VCTK-Corpus-0.92/VCTK-Corpus-0.92.tgz to ./VCTK-Corpus-0.92.tgz
|
||||
|
||||
$ tar -xzvf VCTK-Corpus-0.92.tgz
|
||||
VCTK-Corpus-0.92/
|
||||
VCTK-Corpus-0.92/README.txt
|
||||
VCTK-Corpus-0.92/update.txt
|
||||
VCTK-Corpus-0.92/license_text
|
||||
VCTK-Corpus-0.92/txt/
|
||||
[...]
|
||||
VCTK-Corpus-0.92/wav48_silence_trimmed/p238/p238_191_mic1.flac
|
||||
VCTK-Corpus-0.92/wav48_silence_trimmed/p238/p238_267_mic2.flac
|
||||
|
||||
$ rm VCTK-Corpus-0.92.tgz
|
||||
|
||||
### LibriTTS train-clean-360 subset
|
||||
$ aws s3 cp s3://potion-datasets/LibriTTS/train-clean-360.tar.gz .
|
||||
download: s3://potion-datasets/LibriTTS/train-clean-360.tar.gz to ./train-clean-360.tar.gz
|
||||
|
||||
$ tar -xzvf train-clean-360.tar.gz
|
||||
./LibriTTS/train-clean-360/
|
||||
./LibriTTS/train-clean-360/2272/
|
||||
./LibriTTS/train-clean-360/2272/152265/
|
||||
./LibriTTS/train-clean-360/2272/152265/2272_152265_000032_000001.original.txt
|
||||
./LibriTTS/train-clean-360/2272/152265/2272_152265_000012_000001.wav
|
||||
[...]
|
||||
LibriTTS/reader_book.tsv
|
||||
LibriTTS/speakers.tsv
|
||||
|
||||
$ rm train-clean-360.tar.gz
|
||||
|
||||
### Potion salutation recordings
|
||||
$ aws s3 cp s3://potion-datasets/potion-voice-datasets/potion-salut-corpus_20221026.tgz .
|
||||
download: s3://potion-datasets/potion-voice-datasets/potion-salut-corpus_20221026.tgz to ./potion-salut-corpus_20221019.tgz
|
||||
|
||||
$ tar -xzvf potion-salut-corpus_20221026.tgz
|
||||
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/
|
||||
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/wav48/
|
||||
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/wav48/POTION_6192d9c9a563df5c87ecb8bd/
|
||||
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/wav48/POTION_6192d9c9a563df5c87ecb8bd/POTION_6192d9c9a563df5c87ecb8bd_334.wav
|
||||
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/wav48/POTION_6192d9c9a563df5c87ecb8bd/POTION_6192d9c9a563df5c87ecb8bd_473.wav
|
||||
[...]
|
||||
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/txt/POTION_62d82d269cbde00027b66007/POTION_62d82d269cbde00027b66007_197.txt
|
||||
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/speaker-info.txt
|
||||
|
||||
$ rm potion-salut-corpus_20221026.tgz
|
||||
```
|
||||
|
||||
1. Create a virtual potion-voice-cloner working environment
|
||||
|
||||
```sh
|
||||
$ cd DEV_PATH
|
||||
$ python3 -m venv potion-voice_venv
|
||||
$ cd potion-voice_venv/
|
||||
$ source bin/activate
|
||||
(potion-voice_venv) $
|
||||
```
|
||||
|
||||
1. Clone the potion-voice GitHub repository
|
||||
|
||||
```sh
|
||||
(potion-voice_venv) $ cd DEV_PATH/potion-voice_venv/
|
||||
(potion-voice_venv) $ python3 -m pip install --upgrade pip
|
||||
(potion-voice_venv) $ git clone https://github.com/potion/potion-voice.git
|
||||
```
|
||||
|
||||
1. Install potion-voice requirements (dependencies) and test that PyTorch is working with the GPU properly
|
||||
|
||||
```sh
|
||||
(potion-voice_venv) $ cd DEV_PATH/potion-voice_venv/potion-voice/
|
||||
(potion-voice_venv) $ python3 -m pip install -r ./requirements.dev.txt
|
||||
(potion-voice_venv) $ python3
|
||||
Python 3.10.6 (main, Nov 14 2022, 16:10:14) [GCC 11.3.0] on linux
|
||||
Type "help", "copyright", "credits" or "license" for more information.
|
||||
>>> import torch
|
||||
>>> torch.cuda.is_available()
|
||||
True
|
||||
>>> torch.cuda.get_device_name(0)
|
||||
'NVIDIA A10G'
|
||||
>>> quit()
|
||||
```
|
||||
|
||||
1. Install TTS dependencies
|
||||
|
||||
```sh
|
||||
(potion-voice_venv) $ cd voice-cloning/
|
||||
(potion-voice_venv) $ git clone --depth 1 --branch v0.10.2 https://github.com/coqui-ai/TTS
|
||||
(potion-voice_venv) $ python3 -m pip install -e TTS/
|
||||
```
|
||||
|
||||
+ Note 1: Installing requirements will ask for GitHub token twice! The second request is for a dependent package, which is also a private repo.
|
||||
|
||||
+ Note 2: Separate requirements files have been added for development (local versus AWS) and production usage (for GPU and CPU-only deployment).
|
||||
|
||||
## Training New potion-voice Models (Multi-speaker Baseline & Voice Cloning)
|
||||
|
||||
### Preprocess Dataset(s) Required for Multi-speaker Baseline Model Training
|
||||
|
||||
1. For each dataset, ensure that the sampling rate matches and speaker embeddings are precomputed.
|
||||
|
||||
```sh
|
||||
(potion-voice_venv) $ python3 prepare_datasets.py --dataset_preset vctk --dataset_archive_path ~/datasets/VCTK_v0.92/VCTK-Corpus-0.92.tgz --sampling_rate 22050
|
||||
Commencing preparation of dataset for multi-speaker baseline model training:
|
||||
|
||||
+ Dataset preset: vctk
|
||||
+ Dataset : /home/[REDACTED_HOMEDIR_USERNAME_3]/datasets/VCTK_v0.92/VCTK-Corpus-0.92.tgz
|
||||
+ Output path : results/datasets
|
||||
+ Sampling rate : 22050
|
||||
|
||||
>>> Extracting archive ...
|
||||
>>> Resampling audio files to 16000Hz ...
|
||||
Resampling the audio files...
|
||||
Found 88328 files...
|
||||
100%|████████████████████████████████████████████████████████████████████████████████| 88328/88328 [18:25<00:00, 79.88it/s]
|
||||
Done !
|
||||
>>> Computing speaker embeddings ...
|
||||
> Found 44283 files in /home/[REDACTED_HOMEDIR_USERNAME_3]/work/potion-repos/potion-voice_venv/potion-voice/voice-cloning/results/datasets/VCTK-Corpus-0.92
|
||||
> Model fully restored.
|
||||
> Setting up Audio Processor...
|
||||
[...]
|
||||
100%|████████████████████████████████████████████████████████████████████████████████| 44283/44283 [06:18<00:00, 116.99it/s]
|
||||
Speaker embeddings saved at: results/datasets/VCTK-Corpus-0.92/speakers.pth
|
||||
>>> Extracting original archive again (overwritting previously resampled files)...
|
||||
>>> Resampling audio files to 22050Hz ...
|
||||
Resampling the audio files...
|
||||
Found 88328 files...
|
||||
100%|████████████████████████████████████████████████████████████████████████████████| 88328/88328 [20:48<00:00, 70.74it/s]
|
||||
Done !
|
||||
|
||||
Completed preparing voice dataset for multi-speaker baseline model training; generated asset locations are as follows:
|
||||
--> results/datasets/VCTK-Corpus-0.92
|
||||
--> results/datasets/VCTK-Corpus-0.92/speakers.pth
|
||||
|
||||
Done; bye.
|
||||
```
|
||||
|
||||
### Train New potion-voice Multi-speaker Baseline Model
|
||||
|
||||
1. To train a new baseline model:
|
||||
|
||||
```sh
|
||||
(potion-voice_venv) $ python3 train_multispeaker_baseline_model.py
|
||||
|
||||
usage: train_multispeaker_baseline_model.py [-h] --datasets {VCTK,LibriTTS_tc360,POTION_Salut} [{VCTK,LibriTTS_tc360,POTION_Salut} ...] [--output_path OUTPUT_PATH] [--batch_size BATCH_SIZE] [--max_epochs MAX_EPOCHS]
|
||||
|
||||
Code to train multi-speaker baseline model
|
||||
|
||||
options:
|
||||
-h, --help show this help message and exit
|
||||
--datasets {VCTK,LibriTTS_tc360,POTION_Salut} [{VCTK,LibriTTS_tc360,POTION_Salut} ...]
|
||||
List of training datasets to be included in training run.
|
||||
--output_path OUTPUT_PATH
|
||||
Path to store trained / generated assets
|
||||
--batch_size BATCH_SIZE
|
||||
Batch size for training run
|
||||
--max_epochs MAX_EPOCHS
|
||||
Maximum number of epochs for training run
|
||||
```
|
||||
|
||||
Using the default settings, training a new multi-speaker baseline model (on an AWS g5.2xlarge instance) takes 5-7 days (100 epochs with 32 batch size and all 3 datasets (i.e., VCTK, LibriTTS_tc360, andpotion_Salut)).
|
||||
|
||||
1. At the end of a training run, there will be the following files in the result folder:
|
||||
|
||||
```txt
|
||||
results/baseline-models/vits_vctk-March-23-2022_03+43AM-0000000/
|
||||
|-- best_model.pth .................................... best model using avg_loss_0 (NOT the best model; suggest to ignore for now)
|
||||
|-- best_model_19096.pth .............................. same as best_model.pth (suggest to ignore for now)
|
||||
|-- checkpoint_300000.pth ............................. fifth last checkpoint
|
||||
|-- checkpoint_310000.pth ............................. fourth last checkpoint
|
||||
|-- checkpoint_320000.pth ............................. third last checkpoint
|
||||
|-- checkpoint_330000.pth ............................. second last checkpoint
|
||||
|-- checkpoint_340000.pth ............................. last checkpoint
|
||||
|-- config.json ....................................... configuration file
|
||||
|-- events.out.tfevents.1648007016.ip-172-31-83-225 ... event log for entire training run including eval samples and charts (view via tensorboard)
|
||||
|-- speakers.pth ...................................... speaker embeddings
|
||||
|-- trainer_0_log.txt ................................. training log
|
||||
|-- train_multispeaker_baseline_model.py .............. copy of the training script
|
||||
```
|
||||
|
||||
Use the event log to determine which of the checkpoints corresponds to the best model.
|
||||
|
||||
### Clone a Voice based on the Mutli-speaker Baseline Model
|
||||
|
||||
1. To clone a new voice, you need at least 10 voice samples (ideally 30). Those voice recordings (and their corresponding transcription files) have to be arranged as follows (and compressed into a `.tgz`, `.tbz` or `.zip` archive):
|
||||
|
||||
```txt
|
||||
VOICE_DATASET_PATH/txt/1/1_001.txt
|
||||
VOICE_DATASET_PATH/txt/1/1_002.txt
|
||||
VOICE_DATASET_PATH/txt/1/1_003.txt
|
||||
...
|
||||
VOICE_DATASET_PATH/txt/1/1_029.txt
|
||||
VOICE_DATASET_PATH/txt/1/1_030.txt
|
||||
VOICE_DATASET_PATH/wav48/1/1_001.wav
|
||||
VOICE_DATASET_PATH/wav48/1/1_002.wav
|
||||
VOICE_DATASET_PATH/wav48/1/1_003.wav
|
||||
...
|
||||
VOICE_DATASET_PATH/wav48/1/1_028.wav
|
||||
VOICE_DATASET_PATH/wav48/1/1_029.wav
|
||||
VOICE_DATASET_PATH/wav48/1/1_030.wav
|
||||
```
|
||||
|
||||
1. Next, pre-process audio recordings to fit the format of audio samples (i.e., sampling rate) and pre-compute speaker embeddings:
|
||||
|
||||
```sh
|
||||
(potion-voice_venv) $ python prepare_datasets.py --dataset_preset potion_voice_cloning --dataset_archive_path ~/datasets/potion\ Recordings/potion-voice\ recordings/user123.tgz
|
||||
|
||||
Commencing preparation of dataset for multi-speaker baseline model training:
|
||||
|
||||
+ Dataset preset: potion_voice_cloning
|
||||
+ Dataset : /home/[REDACTED_HOMEDIR_USERNAME_3]/datasets/potion Recordings/potion-voice recordings/user123.tgz
|
||||
+ Output path : results/datasets
|
||||
+ Sampling rate : 22050
|
||||
|
||||
>>> Extracting archive ...
|
||||
>>> Resampling audio files to 16000Hz ...
|
||||
Resampling the audio files...
|
||||
Found 30 files...
|
||||
100%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 30/30 [00:00<00:00, 39.40it/s]
|
||||
Done !
|
||||
>>> Extracting original archive again (overwritting previously resampled files)...
|
||||
>>> Resampling audio files to 22050Hz ...
|
||||
Resampling the audio files...
|
||||
Found 30 files...
|
||||
100%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 30/30 [00:00<00:00, 37.61it/s]
|
||||
Done !
|
||||
|
||||
Completed preparing voice dataset for multi-speaker baseline model training; generated asset locations are as follows:
|
||||
--> results/datasets/sr22050/user123
|
||||
--> results/datasets/sr22050/user123/speakers.pth
|
||||
|
||||
Done; bye.
|
||||
```
|
||||
|
||||
1. Finally, trigger voice cloning:
|
||||
|
||||
```sh
|
||||
(potion-voice_venv) $ python3 clone_voice.py [-h] --baseline_model_path BASELINE_MODEL_PATH --speaker_dataset_path SPEAKER_DATASET_PATH --speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH [--output_path OUTPUT_PATH] [--batch_size BATCH_SIZE] [--max_epochs MAX_EPOCHS] [--use_cpu] [--output_format {txt,json}]
|
||||
|
||||
Code to clone a voice from a given set of voice samples and a multi-speaker baseline model
|
||||
|
||||
options:
|
||||
-h, --help show this help message and exit
|
||||
--baseline_model_path BASELINE_MODEL_PATH
|
||||
Path to multi-speaker baseline model (VITS model)
|
||||
--speaker_dataset_path SPEAKER_DATASET_PATH
|
||||
Path to voice cloning dataset
|
||||
--speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH
|
||||
Path to speaker's embeddings file
|
||||
--output_path OUTPUT_PATH
|
||||
Path to store trained / generated assets
|
||||
--batch_size BATCH_SIZE
|
||||
Batch size for training run
|
||||
--max_epochs MAX_EPOCHS
|
||||
Maximum number of epochs for training run
|
||||
--use_cpu Signal that CPU should be used even if a CUDA-device is available
|
||||
--output_format {txt,json}
|
||||
Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting
|
||||
```
|
||||
|
||||
Using the default settings and 30 audio samples, cloning a new voice (on an AWS g5.2xlarge instance) takes about one hour.
|
||||
|
||||
1. At the end of a voice cloning run, there will be the following files in the result folder:
|
||||
|
||||
```txt
|
||||
results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/
|
||||
|-- best_model_365097.pth .................. best model using avg_loss_0 (save to use)
|
||||
|-- best_model.pth ......................... same as best_model_365097.pth
|
||||
|-- checkpoint_365200.pth .................. last checkpoint
|
||||
|-- clone_voice.py ......................... copy of the clone_voice script used in this run
|
||||
|-- config.json ............................ configuration file
|
||||
|-- events.out.tfevents.1672195961.rigel ... event log for entire voice cloning run including eval samples and charts (view via tensorboard)
|
||||
|-- speakers.pth ........................... speaker's embeddings file
|
||||
|-- trainer_0_log.txt ...................... training log file
|
||||
```
|
||||
|
||||
Use the event log to confirm that the best model is indeed giving the best outputs.
|
||||
|
||||
### Monitoring Training Progress
|
||||
|
||||
Using tensorboard / tensorboardX, training progress (for both, multi-speaker baseline training and voice cloning) can be monitored and evaluation samples can be accessed.
|
||||
|
||||
1. Ensure AWS Security Group settings (inbound) are set appropriamust include:
|
||||
|
||||
```txt
|
||||
HTTPS TCP 443 0.0.0.0/0
|
||||
Custom_TCP TCP 6006 0.0.0.0/0
|
||||
```
|
||||
|
||||
+ Server-side, launch the tensorboard service:
|
||||
|
||||
```sh
|
||||
(potion-voice_venv) $ tensorboard --logdir=./results/baseline-models/vits_vctk-March-07-2022_09+47AM-0000000/ --host 0.0.0.0
|
||||
TensorFlow installation not found - running with reduced feature set.
|
||||
|
||||
NOTE: Using experimental fast data loading logic. To disable, pass
|
||||
"--load_fast=false" and report issues on GitHub. More details:
|
||||
https://github.com/tensorflow/tensorboard/issues/4784
|
||||
|
||||
TensorBoard 2.8.0 at http://0.0.0.0:6006/ (Press CTRL+C to quit)
|
||||
```
|
||||
|
||||
+ Locally, point your preferred Web browser to <http://PUBLIC_IPv4_DNS:6006/>
|
||||
|
||||
### Minimise a Cloned Voice
|
||||
|
||||
To minimise the size of a trained model, run the followng script which removes optimiser and discriminator components from the model -- those are only required for training but not for inference:
|
||||
|
||||
```sh
|
||||
(potion-voice_venv) $ python3 minimize_cloned_voice_model.py [-h] --voice_model_asset_path VOICE_MODEL_ASSET_PATH [--voice_model_name VOICE_MODEL_NAME] [--voice_model_config_name VOICE_MODEL_CONFIG_NAME] [--minimise_suffix MINIMISE_SUFFIX] [--overwrite_assets] [--output_format {txt,json}]
|
||||
|
||||
Code to minimise (i.e., remove optimiser & discriminator) a cloned voice model
|
||||
|
||||
options:
|
||||
-h, --help show this help message and exit
|
||||
--voice_model_asset_path VOICE_MODEL_ASSET_PATH
|
||||
Path to directory storing cloned voice model and the corresponding configuration and speaker files
|
||||
--voice_model_name VOICE_MODEL_NAME
|
||||
Name of the (best) cloned voice model
|
||||
--voice_model_config_name VOICE_MODEL_CONFIG_NAME
|
||||
Name of the config file for the cloned voice model
|
||||
--minimise_suffix MINIMISE_SUFFIX
|
||||
Suffix to be used for minimised model and its assets (i.e., new config file)
|
||||
--overwrite_assets Signal whether existing model assets should be overwritten or not (default: do not overwrite)
|
||||
--output_format {txt,json}
|
||||
Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting
|
||||
```
|
||||
|
||||
1. Command-line output sample for output format option "txt":
|
||||
|
||||
```sh
|
||||
$ python3 minimize_cloned_voice_model.py --voice_model_asset_path results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/
|
||||
Minimising given voice model:
|
||||
|
||||
+ Cloned voice model file path : results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/best_model.pth
|
||||
+ Cloned voice model config file : results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/config.json
|
||||
|
||||
> Using model: vits
|
||||
> Setting up Audio Processor...
|
||||
[...]
|
||||
Completed minimising cloned voice model. The resulting (modified) assets can be found at:
|
||||
--> Minimised voice model path : results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/best_model_light.pth
|
||||
--> Minimised voice model config path: results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/config_light.json
|
||||
|
||||
Done; bye.
|
||||
```
|
||||
|
||||
### Scoring a Cloned Voice
|
||||
|
||||
1. To score a cloned voice, run the following command:
|
||||
|
||||
```sh
|
||||
(potion-voice_venv) $ python3 score_cloned_voice.py [-h] --voice_dataset_path VOICE_DATASET_PATH --voice_model_path VOICE_MODEL_PATH --voice_model_config_path VOICE_MODEL_CONFIG_PATH --speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH [--temp_path TEMP_PATH] [--keep_temp] [--use_cpu] [--output_format {txt,json}]
|
||||
|
||||
Compute quality score for a given voice model (cloned voice) wrt. a given set of voice recordings (original voice))
|
||||
|
||||
options:
|
||||
-h, --help show this help message and exit
|
||||
--voice_dataset_path VOICE_DATASET_PATH
|
||||
Path to set of voice recordings (original voice)
|
||||
--voice_model_path VOICE_MODEL_PATH
|
||||
Path to cloned voice model
|
||||
--voice_model_config_path VOICE_MODEL_CONFIG_PATH
|
||||
Path to config file for the cloned voice model
|
||||
--speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH
|
||||
Path to speaker's embeddings file (i.e., pre-computed embeddings typically stored with the speaker's dataset)
|
||||
--temp_path TEMP_PATH
|
||||
Path to store temporary speech assets
|
||||
--keep_temp Signal that temporary assets used for scoring should not be deleted once done
|
||||
--use_cpu Signal that CPU should be used even if a CUDA-device is available
|
||||
--output_format {txt,json}
|
||||
Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting
|
||||
```
|
||||
|
||||
1. Command-line output sample for output format option "txt":
|
||||
|
||||
```sh
|
||||
$ python3 score_cloned_voice.py --voice_dataset_path results/datasets/sr22050/michael/wav48/1/ --voice_model_path results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/best_model.pth --voice_model_config_path results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/config.json --speaker_embeddings_path results/datasets/sr22050/michael/speakers.pth
|
||||
Computing similarity score for a given voice model (cloned voice) wrt. a given set of voice recordings (original voice):
|
||||
|
||||
+ Original voice recordings path: results/datasets/sr22050/michael/wav48/1/
|
||||
+ Cloned voice model file path : results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/best_model.pth
|
||||
+ Cloned voice model config file: results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/config.json
|
||||
+ Speaker embeddings file : results/datasets/sr22050/michael/speakers.pth
|
||||
|
||||
+ CUDA availability : True
|
||||
+ Compute device used : cuda
|
||||
|
||||
+ No. of speakers : 1
|
||||
+ Speaker's names : ['VCTK_old_1']
|
||||
+ No. of embeddings : 30
|
||||
|
||||
> Using model: vits
|
||||
> Setting up Audio Processor...
|
||||
|
||||
Loaded the voice encoder model on cuda in 0.01 seconds.
|
||||
|
||||
Completed computing similarity score for the two sets of recordings. The resulting similarity score is:
|
||||
--> 0.9127510190010071
|
||||
|
||||
Done; bye.
|
||||
```
|
||||
|
||||
1. Command-line output sample for output format option "json":
|
||||
|
||||
```sh
|
||||
(potion-voice_venv)$ python3 score_cloned_voice.py --voice_dataset_path results/datasets/sr22050/michael/wav48/1/ --voice_model_path results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/best_model.pth --voice_model_config_path results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/config.json --speaker_embeddings_path results/datasets/sr22050/michael/speakers.pth --output_format json
|
||||
> Using model: vits
|
||||
> Setting up Audio Processor...
|
||||
[...]
|
||||
Loaded the voice encoder model on cuda in 0.01 seconds.
|
||||
{"success": true, "in": {"voice_dataset_path": "results/datasets/sr22050/michael/wav48/1/", "voice_model_path": "results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/best_model.pth"}, "out": {"score": 0.91}}
|
||||
```
|
||||
|
||||
## Usage Examples for Speech Synthesizing
|
||||
|
||||
1. To generate speech for a given cloned voice, run the following command:
|
||||
|
||||
```sh
|
||||
(potion-voice_venv) $ python3 synthesize_speech.py [-h] --voice_model_path VOICE_MODEL_PATH --voice_model_config_path VOICE_MODEL_CONFIG_PATH --speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH --txt TXT [--output_path OUTPUT_PATH] [--target_sampling_rate TARGET_SAMPLING_RATE] [--speech_sample_wav_path SPEECH_SAMPLE_WAV_PATH] [--speech_sample_txt SPEECH_SAMPLE_TXT] [--trim_silence] [--use_cpu] [--output_format {txt,json}]
|
||||
|
||||
Code to synthesize speech for a given voice model
|
||||
|
||||
options:
|
||||
-h, --help show this help message and exit
|
||||
--voice_model_path VOICE_MODEL_PATH
|
||||
Path to cloned voice model
|
||||
--voice_model_config_path VOICE_MODEL_CONFIG_PATH
|
||||
Path to config file for the cloned voice model
|
||||
--speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH
|
||||
Path to speaker's embeddings file (i.e., pre-computed embeddings typically stored with the speaker's dataset)
|
||||
--txt TXT Text to synthesize
|
||||
--output_path OUTPUT_PATH
|
||||
Path to store generated speech assets
|
||||
--target_sampling_rate TARGET_SAMPLING_RATE
|
||||
Desired sampling rate (in Hz) for output file
|
||||
--speech_sample_wav_path SPEECH_SAMPLE_WAV_PATH
|
||||
Path to a sample utterance of the speaker (used for style transfer)
|
||||
--speech_sample_txt SPEECH_SAMPLE_TXT
|
||||
Text of the sample utterance of the speaker (used for style transfer)
|
||||
--trim_silence Signal whether to trim silence from synthesised speech
|
||||
--use_cpu Signal that CPU should be used even if a CUDA-device is available
|
||||
--output_format {txt,json}
|
||||
Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting
|
||||
```
|
||||
|
||||
1. Command-line output sample for output format option "txt":
|
||||
|
||||
```sh
|
||||
(potion-voice_venv) $ python3 synthesize_speech.py --voice_model_path results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/best_model.pth --voice_model_config_path results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/config.json --speaker_embeddings_path results/datasets/sr22050/[REDACTED_HOMEDIR_USERNAME_2]/speakers.pth --txt "Hi person_82, it works!"
|
||||
Commencing speech synthesizing:
|
||||
|
||||
+ Voice model file path : results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/best_model.pth
|
||||
+ Voice model config file: results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/config.json
|
||||
+ Speaker embeddings file: results/datasets/sr22050/[REDACTED_HOMEDIR_USERNAME_2]/speakers.pth
|
||||
+ Output path : results/speech
|
||||
+ Text to synthesize : Hi person_82, it works!
|
||||
|
||||
+ CUDA availability : True
|
||||
+ Compute device used : cuda
|
||||
+ No. of speakers : 1
|
||||
+ Speaker's names : ['VCTK_old_1']
|
||||
+ No. of embeddings : 30
|
||||
|
||||
> Using model: vits
|
||||
> Setting up Audio Processor...
|
||||
[...]
|
||||
>>> Saving original output to : results/speech/b4189e9e-6142-4dad-8577-6de77087ffd1.wav
|
||||
>>> Saving resampled output to: results/speech/b4189e9e-6142-4dad-8577-6de77087ffd1_sr48000.wav
|
||||
|
||||
Speech synthesizing has completed. Bye.
|
||||
```
|
||||
|
||||
1. Command-line output sample for output format option "txt":
|
||||
|
||||
```sh
|
||||
(potion-voice_venv) $ python3 synthesize_speech.py --voice_model_path results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/best_model_light.pth --voice_model_config_path results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/config_light.json --speaker_embeddings_path results/datasets/sr22050/[REDACTED_HOMEDIR_USERNAME_2]/speakers.pth --txt "Hi person_82, it works!" --output_format json
|
||||
> Using model: vits
|
||||
> Setting up Audio Processor...
|
||||
[...]
|
||||
{"success": true, "in": {"voice_model_path": "results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/best_model_light.pth", "voice_model_config_path": "results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/config_light.json", "speaker_embeddings_path": "results/datasets/sr22050/[REDACTED_HOMEDIR_USERNAME_2]/speakers.pth"}, "out": {"speech_original_path": "results/speech/c99e494c-e1f9-4c12-9095-255cf7db792b.wav", "speech_resampled_path": "results/speech/c99e494c-e1f9-4c12-9095-255cf7db792b_sr48000.wav"}}
|
||||
```
|
||||
|
||||
### Scoring a Synthesised Salutation
|
||||
|
||||
1. To score a synthesised salutation, run the following command:
|
||||
|
||||
```sh
|
||||
(potion-voice_venv)$ python3 score_salutation.py [-h] --recording_path RECORDING_PATH --first_name FIRST_NAME [--output_format {txt,json}]
|
||||
|
||||
Score a given salutation recording wrt. its desired content, the actual salutation recording, and a generated transcription (using Potion's internal Transciption API) of the recording.
|
||||
|
||||
optional arguments:
|
||||
-h, --help show this help message and exit
|
||||
--recording_path RECORDING_PATH
|
||||
Path to salutation recoding (.wav audio file)
|
||||
--first_name FIRST_NAME
|
||||
First name that the salutation recoding is meant to use
|
||||
--output_format {txt,json}
|
||||
Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting
|
||||
```
|
||||
|
||||
1. Command-line output sample for output format option "txt":
|
||||
|
||||
```sh
|
||||
(potion-voice_venv)$ python3 score_salutation.py --recording_path /home/[REDACTED_HOMEDIR_USERNAME_3]/person_82_-_Hey_person_83.wav --first_name person_83
|
||||
Commencing scoring of the given salutation recording:
|
||||
|
||||
+ Salutation recording path: /home/[REDACTED_HOMEDIR_USERNAME_3]/person_82_-_Hey_person_83.wav
|
||||
+ Salutation first name : person_83
|
||||
|
||||
>> Salutation score : 0.892155
|
||||
|
||||
Done; bye.
|
||||
```
|
||||
|
||||
1. Command-line output sample for output format option "json":
|
||||
|
||||
```sh
|
||||
(potion-voice_venv)$ python3 score_salutation.py --recording_path /home/[REDACTED_HOMEDIR_USERNAME_3]/person_82_-_Hey_person_83.wav --first_name person_83 --output_format json
|
||||
{"in": {"recording_path": "/home/[REDACTED_HOMEDIR_USERNAME_3]/person_82_-_Hey_person_83.wav", "first_name": "person_83"}, "out": {"score": 0.89}}
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
1. How to better monitor GPU load / utilisation?
|
||||
|
||||
+ Install an interactive NVIDIA-GPU process viewer such as `nvitop`:
|
||||
|
||||
```sh
|
||||
$ python3 -m pip install nvitop
|
||||
Collecting nvitop
|
||||
[...]
|
||||
Installing collected packages: nvidia-ml-py, termcolor, psutil, nvitop
|
||||
Successfully installed nvidia-ml-py-11.495.46 nvitop-0.8.0 psutil-5.9.2 termcolor-2.0.1
|
||||
````
|
||||
|
||||
+ Run via command-line: `nvitop`:
|
||||
|
||||
```sh
|
||||
Tue Sep 13 02:09:07 2022
|
||||
╒═════════════════════════════════════════════════════════════════════════════╕
|
||||
│ NVITOP 0.8.0 Driver Version: 515.65.01 CUDA Driver Version: 11.7 │
|
||||
├───────────────────────────────┬──────────────────────┬──────────────────────┤
|
||||
│ GPU Name Persistence-M│ Bus-Id Disp.A │ Volatile Uncorr. ECC │
|
||||
│ Fan Temp Perf Pwr:Usage/Cap│ Memory-Usage │ GPU-Util Compute M. │
|
||||
╞═══════════════════════════════╪══════════════════════╪══════════════════════╪══════════════════════════╕
|
||||
│ 0 A10G On │ 00000000:00:1E.0 Off │ 0 │ MEM: █████████▊ 65.1% │
|
||||
│ 0% 47C P0 192W / 300W │ 14982MiB / 22.49GiB │ 100% Default │ UTL: ███████████████ MAX │
|
||||
╘═══════════════════════════════╧══════════════════════╧══════════════════════╧══════════════════════════╛
|
||||
[ CPU: ██████████▏ 18.1% ] ( Load Average: 1.07 1.11 1.04 )
|
||||
[ MEM: ███████████▎ 20.2% ] [ SWP: ▏ 0.3% ]
|
||||
|
||||
╒════════════════════════════════════════════════════════════════════════════════════════════════════════╕
|
||||
│ Processes: ubuntu@ip-172-31-95-84 │
|
||||
│ GPU PID USER GPU-MEM %SM %CPU %MEM TIME COMMAND │
|
||||
╞════════════════════════════════════════════════════════════════════════════════════════════════════════╡
|
||||
│ 0 2100 C ubuntu 14463MiB 90 103.7 9.6 5.4 days python3 train_multispeaker_baseline_model.py │
|
||||
╘════════════════════════════════════════════════════════════════════════════════════════════════════════╛
|
||||
```
|
||||
@@ -0,0 +1,120 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import sys
|
||||
import os
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
|
||||
import json
|
||||
import torch
|
||||
|
||||
from TTS.config import load_config
|
||||
from TTS.tts.models import setup_model as setup_tts_model
|
||||
|
||||
|
||||
#
|
||||
# parse command line arguments
|
||||
#
|
||||
def parse_cmdline_args():
|
||||
parser = argparse.ArgumentParser(
|
||||
description = "Code to minimise (i.e., remove optimiser & discriminator) a cloned voice model")
|
||||
parser.add_argument("--voice_model_asset_path", type = str, required = True,
|
||||
help = "Path to directory storing cloned voice model and the corresponding configuration and speaker files")
|
||||
parser.add_argument("--voice_model_name", type = str, default = "best_model.pth",
|
||||
help = "Name of the (best) cloned voice model")
|
||||
parser.add_argument("--voice_model_config_name", type = str, default = "config.json",
|
||||
help = "Name of the config file for the cloned voice model")
|
||||
parser.add_argument("--minimise_suffix", type = str, default = "light",
|
||||
help = "Suffix to be used for minimised model and its assets (i.e., new config file)")
|
||||
parser.add_argument("--overwrite_assets", default = False, action = "store_true",
|
||||
help = "Signal whether existing model assets should be overwritten or not (default: do not overwrite)")
|
||||
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
|
||||
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
|
||||
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
#
|
||||
# utility function to expand the name of a given filename (infront of the extension)
|
||||
#
|
||||
def append_suffix_to_filename(fname, fname_suffix):
|
||||
fpath = Path(fname)
|
||||
|
||||
return "{0}_{2}{1}" . format(fpath.stem, fpath.suffix, fname_suffix)
|
||||
|
||||
|
||||
#
|
||||
# save a lightweight (i.e., without optimiser and discriminator) model of the given cloned voice and corresponding config assets
|
||||
#
|
||||
def main(args):
|
||||
# set variables
|
||||
output_path = args.voice_model_asset_path
|
||||
model_path = os.path.join(args.voice_model_asset_path, args.voice_model_name)
|
||||
model_config_path = os.path.join(args.voice_model_asset_path, args.voice_model_config_name)
|
||||
model_light_path = os.path.join(output_path, append_suffix_to_filename(args.voice_model_name, args.minimise_suffix))
|
||||
model_light_config_path = os.path.join(output_path, append_suffix_to_filename(args.voice_model_config_name, args.minimise_suffix))
|
||||
|
||||
if not args.overwrite_assets:
|
||||
# ensure target output files do not already exist
|
||||
if (Path(model_light_path).exists()) or (Path(model_light_config_path).exists()):
|
||||
sys.exit("Naming conflict: Model asset files ({} and/or {}) exist already!" . format (model_light_path, model_light_config_path))
|
||||
|
||||
if args.output_format == "txt":
|
||||
print("Minimising given voice model:")
|
||||
print("")
|
||||
print(" + Cloned voice model file path : {}" . format(model_path))
|
||||
print(" + Cloned voice model config file : {}" . format(model_config_path))
|
||||
print("")
|
||||
elif args.output_format == "json":
|
||||
json_data = {
|
||||
"success": False,
|
||||
"in": {
|
||||
"voice_model_path": format(model_path),
|
||||
"voice_model_config_path": format(model_config_path)
|
||||
},
|
||||
"out": {
|
||||
"voice_model_light_path": "",
|
||||
"voice_model_light_config_path": ""
|
||||
}
|
||||
}
|
||||
|
||||
# load model
|
||||
config = load_config(model_config_path)
|
||||
|
||||
# init model
|
||||
model = setup_tts_model(config = config)
|
||||
|
||||
# load checkpoint / model
|
||||
model.load_checkpoint(config, model_path, eval = True)
|
||||
model.disc = None
|
||||
model_state = model.state_dict()
|
||||
state = {
|
||||
"model": model_state
|
||||
}
|
||||
|
||||
torch.save(state, model_light_path)
|
||||
|
||||
config.model_args["init_discriminator"] = False
|
||||
config.save_json(model_light_config_path)
|
||||
|
||||
# exit gracefully
|
||||
if args.output_format == "txt":
|
||||
print("Completed minimising cloned voice model. The resulting (modified) assets can be found at:")
|
||||
print(" --> Minimised voice model path : {}" . format(model_light_path))
|
||||
print(" --> Minimised voice model config path: {}" . format(model_light_config_path))
|
||||
print("")
|
||||
print("Done; bye.")
|
||||
print("")
|
||||
elif args.output_format == "json":
|
||||
json_data["out"]["voice_model_light_path"] = format(model_light_path)
|
||||
json_data["out"]["voice_model_light_config_path"]: format(model_light_config_path)
|
||||
json_data["success"] = True
|
||||
print(json.dumps(json_data))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# parse command line arguments
|
||||
args = parse_cmdline_args()
|
||||
|
||||
main(args)
|
||||
@@ -0,0 +1,191 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import sys
|
||||
import os
|
||||
import argparse
|
||||
|
||||
# load coqui-ai/TTS libraries
|
||||
from TTS.bin.resample import resample_files
|
||||
from TTS.bin.compute_embeddings import compute_embeddings
|
||||
|
||||
import train_config as tc
|
||||
|
||||
|
||||
#
|
||||
# parse command line arguments
|
||||
#
|
||||
def parse_cmdline_args():
|
||||
parser = argparse.ArgumentParser(
|
||||
description = "Code to prepare voice dataset for multi-speaker baseline model training (i.e., adjust sampling rate and compute speaker embeddings).")
|
||||
parser.add_argument("--dataset_preset", type = str, choices = ("VCTK", "LibriTTS_tc360", "DAPS", "POTION_Salut", "potion_voice_cloning"), required = True,
|
||||
help = "Path the voice dataset archive (.zip, .tar.gz, .tgz, .tar.bz2, and .tbz are supported)")
|
||||
parser.add_argument("--dataset_archive_path", type = str, required = True,
|
||||
help = "Path the voice dataset archive (.zip, .tar.gz, .tgz, .tar.bz2, and .tbz are supported)")
|
||||
parser.add_argument("--output_path", type = str, default = "results/datasets",
|
||||
help = "Path to store augmented dataset")
|
||||
parser.add_argument("--sampling_rate", type = int, default = 22050, choices = (16000, 22050, 32000, 48000), # 32k & 48k are untested
|
||||
help = "Sampling rate for training run")
|
||||
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
#
|
||||
# utility functuion to extract archives (zip, tar, tgz, ...)
|
||||
# - returns first entry in archive (typically the main directory name contained in the archive)
|
||||
#
|
||||
def extract_archive(archive_path, dest_path):
|
||||
|
||||
from zipfile import ZipFile
|
||||
import tarfile
|
||||
|
||||
if archive_path.endswith('.zip'):
|
||||
opener, getnames, mode = ZipFile, ZipFile.namelist, 'r'
|
||||
|
||||
elif (archive_path.endswith('.tar.gz')) or (archive_path.endswith('.tgz')):
|
||||
opener, getnames, mode = tarfile.open, tarfile.TarFile.getnames, 'r:gz'
|
||||
|
||||
elif (archive_path.endswith('.tar.bz2')) or (archive_path.endswith('.tbz')):
|
||||
opener, getnames, mode = tarfile.open, tarfile.TarFile.getnames, 'r:bz2'
|
||||
|
||||
else:
|
||||
print("Extracting archive " + archive_path + " is not supported.")
|
||||
return
|
||||
|
||||
# extract archive
|
||||
with opener(archive_path, mode) as archive:
|
||||
archive_dir = archive.getnames()[0]
|
||||
archive.extractall(path = dest_path)
|
||||
|
||||
return archive_dir
|
||||
|
||||
|
||||
#
|
||||
# main training method (VITS multi-speaker model)
|
||||
#
|
||||
def main(args):
|
||||
print("Commencing preparation of dataset for multi-speaker baseline model training:")
|
||||
print("")
|
||||
print(" + Dataset preset: {}" . format(args.dataset_preset))
|
||||
print(" + Dataset : {}" . format(args.dataset_archive_path))
|
||||
print(" + Output path : {}" . format(args.output_path))
|
||||
print(" + Sampling rate : {}" . format(args.sampling_rate))
|
||||
print("")
|
||||
|
||||
# set parameters according to dataset preset
|
||||
if args.dataset_preset == "VCTK":
|
||||
DATASET_NAME = tc.VCTK_DATASET_NAME
|
||||
DATASET_FORMATTER = tc.VCTK_DATASET_FORMATTER
|
||||
DATASET_FILE_FORMAT = tc.VCTK_DATASET_FILE_FORMAT
|
||||
NO_EVAL = False
|
||||
elif args.dataset_preset == "LibriTTS_tc360":
|
||||
DATASET_NAME = tc.LIBRITTS_TC360_DATASET_NAME
|
||||
DATASET_FORMATTER = tc.LIBRITTS_TC360_DATASET_FORMATTER
|
||||
DATASET_FILE_FORMAT = tc.LIBRITTS_TC360_DATASET_FILE_FORMAT
|
||||
NO_EVAL = False
|
||||
elif args.dataset_preset == "POTION_Salut":
|
||||
DATASET_NAME = tc.POTION_SALUT_DATASET_NAME
|
||||
DATASET_FORMATTER = tc.POTION_SALUT_DATASET_FORMATTER
|
||||
DATASET_FILE_FORMAT = tc.POTION_SALUT_DATASET_FILE_FORMAT
|
||||
NO_EVAL = False
|
||||
elif args.dataset_preset == "potion_voice_cloning":
|
||||
DATASET_NAME = tc.POTION_SALUT_DATASET_NAME
|
||||
DATASET_FORMATTER = tc.POTION_SALUT_DATASET_FORMATTER
|
||||
DATASET_FILE_FORMAT = tc.POTION_SALUT_DATASET_FILE_FORMAT
|
||||
NO_EVAL = True
|
||||
|
||||
# define sampling rate for computing speaker embeddings
|
||||
SPK_EMB_SAMPLING_RATE = 16000
|
||||
|
||||
# define the number of threads used during audio resampling
|
||||
NUM_RESAMPLE_THREADS = 10
|
||||
|
||||
# extract dataset archive
|
||||
print(f">>> Extracting archive ...")
|
||||
dataset_root = extract_archive(args.dataset_archive_path, os.path.join(args.output_path, "sr" + str(args.sampling_rate)))
|
||||
|
||||
# set dataset path (there should only be ONE directory in the extracted archive location)
|
||||
dataset_path = os.path.join(args.output_path, "sr" + str(args.sampling_rate), dataset_root)
|
||||
|
||||
# ensure the dataset_path exists
|
||||
os.makedirs(dataset_path, exist_ok = True)
|
||||
|
||||
# resample dataset for speaker embeddings computation
|
||||
print(f">>> Resampling audio files to 16000Hz ...")
|
||||
resample_files(dataset_path, 16000, file_ext = DATASET_FILE_FORMAT, n_jobs = NUM_RESAMPLE_THREADS)
|
||||
|
||||
# compute speaker embeddings
|
||||
SPEAKER_ENCODER_CHECKPOINT_PATH = "assets/speaker_encoder_model/model_se.pth.tar"
|
||||
SPEAKER_ENCODER_CONFIG_PATH = "assets/speaker_encoder_model/config_se.json"
|
||||
|
||||
# init list speaker embeddings/d-vectors to be used during the training
|
||||
d_vector_files = []
|
||||
|
||||
# check if the speakers embeddings are already computated, if not compute them
|
||||
embeddings_file = os.path.join(dataset_path, "speakers.pth")
|
||||
|
||||
if not os.path.isfile(embeddings_file):
|
||||
print(f">>> Computing speaker embeddings ...")
|
||||
compute_embeddings(
|
||||
SPEAKER_ENCODER_CHECKPOINT_PATH,
|
||||
SPEAKER_ENCODER_CONFIG_PATH,
|
||||
embeddings_file,
|
||||
old_spakers_file = None,
|
||||
config_dataset_path = None,
|
||||
formatter_name = DATASET_FORMATTER,
|
||||
dataset_name = DATASET_NAME,
|
||||
dataset_path = dataset_path,
|
||||
meta_file_train = "",
|
||||
meta_file_val = "",
|
||||
disable_cuda = False,
|
||||
no_eval = NO_EVAL
|
||||
)
|
||||
|
||||
d_vector_files.append(embeddings_file)
|
||||
|
||||
# if targetted sampling rate is not the same as that used for computing speaker embeddings, replace and resample audio files
|
||||
if not args.sampling_rate == SPK_EMB_SAMPLING_RATE:
|
||||
print(f">>> Extracting original archive again (overwritting previously resampled files)...")
|
||||
extract_archive(args.dataset_archive_path, os.path.join(args.output_path, "sr" + str(args.sampling_rate)))
|
||||
print(f">>> Resampling audio files to {args.sampling_rate}Hz ...")
|
||||
resample_files(dataset_path, args.sampling_rate, file_ext = DATASET_FILE_FORMAT, n_jobs = NUM_RESAMPLE_THREADS)
|
||||
|
||||
# exit gracefully
|
||||
print("")
|
||||
print("Completed preparing voice dataset for multi-speaker baseline model training; generated asset locations are as follows:")
|
||||
print(" --> {}" . format(dataset_path))
|
||||
print(" --> {}" . format(embeddings_file))
|
||||
print("")
|
||||
print("Done; bye.")
|
||||
print("")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# parse command line arguments
|
||||
args = parse_cmdline_args()
|
||||
|
||||
# clear command line arguments to avoid triggering argparse features part of Trainer / coqpit imports
|
||||
# Traceback (most recent call last):
|
||||
# File "train_multispeaker_baseline_model.py", line 208, in <module>
|
||||
# main(args)
|
||||
# File "train_multispeaker_baseline_model.py", line 177, in main
|
||||
# trainer = Trainer(
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 360, in __init__
|
||||
# config, new_fields = self.init_training(args, coqpit_overrides, config)
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 594, in init_training
|
||||
# config.parse_known_args(coqpit_overrides, relaxed_parser=True)
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 843, in parse_known_args
|
||||
# parser = self.init_argparse(arg_prefix=arg_prefix, relaxed_parser=relaxed_parser)
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 881, in init_argparse
|
||||
# _init_argparse(
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 529, in _init_argparse
|
||||
# parser = _init_argparse(
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 550, in _init_argparse
|
||||
# return default.init_argparse(
|
||||
# AttributeError: 'str' object has no attribute 'init_argparse'
|
||||
sys.argv = [sys.argv[0]]
|
||||
|
||||
# ensure the output path exists
|
||||
os.makedirs(args.output_path, exist_ok = True)
|
||||
|
||||
main(args)
|
||||
@@ -0,0 +1,179 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import os
|
||||
import argparse
|
||||
import shutil
|
||||
|
||||
import uuid
|
||||
import json
|
||||
import torch
|
||||
|
||||
from TTS.TTS.tts.utils.speakers import SpeakerManager
|
||||
from utils.synthesize_utils import init_synth, synthesize, save_waveform
|
||||
from utils.scoring_utils import init_scoring_vocoder, score_speaker_similarity
|
||||
|
||||
|
||||
#
|
||||
# parse command line arguments
|
||||
#
|
||||
def parse_cmdline_args():
|
||||
parser = argparse.ArgumentParser(
|
||||
description = "Compute quality score for a given voice model (cloned voice) wrt. a given set of voice recordings (original voice))")
|
||||
parser.add_argument("--voice_dataset_path", type = str, required = True,
|
||||
help = "Path to set of voice recordings (original voice)")
|
||||
parser.add_argument("--voice_model_path", type = str, required = True,
|
||||
help = "Path to cloned voice model")
|
||||
parser.add_argument("--voice_model_config_path", type = str, required = True,
|
||||
help = "Path to config file for the cloned voice model")
|
||||
parser.add_argument('--speaker_embeddings_path', type = str, required = True,
|
||||
help = "Path to speaker's embeddings file (i.e., pre-computed embeddings typically stored with the speaker's dataset)")
|
||||
parser.add_argument("--temp_path", type = str, default = "temp",
|
||||
help = "Path to store temporary speech assets")
|
||||
parser.add_argument("--keep_temp", default = False, action = "store_true",
|
||||
help = "Signal that temporary assets used for scoring should not be deleted once done")
|
||||
parser.add_argument("--use_cpu", default = False, action = "store_true", # untested!!!
|
||||
help = "Signal that CPU should be used even if a CUDA-device is available")
|
||||
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
|
||||
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
|
||||
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
#
|
||||
# main training method (voice cloning)
|
||||
#
|
||||
def main(args):
|
||||
# define assets required for using a pretrained voice
|
||||
MODEL_PATH = args.voice_model_path
|
||||
CONFIG_PATH = args.voice_model_config_path
|
||||
SPK_EMBEDDINGS_PATH = args.speaker_embeddings_path
|
||||
|
||||
# set default score
|
||||
sim_score = -1.0
|
||||
|
||||
if args.output_format == "txt":
|
||||
print("Computing similarity score for a given voice model (cloned voice) wrt. a given set of voice recordings (original voice):")
|
||||
print("")
|
||||
print(" + Original voice recordings path: {}" . format(args.voice_dataset_path))
|
||||
print(" + Cloned voice model file path : {}" . format(MODEL_PATH))
|
||||
print(" + Cloned voice model config file: {}" . format(CONFIG_PATH))
|
||||
print(" + Speaker embeddings file : {}" . format(SPK_EMBEDDINGS_PATH))
|
||||
print("")
|
||||
elif args.output_format == "json":
|
||||
json_data = {
|
||||
"success": False,
|
||||
"in": {
|
||||
"voice_dataset_path": format(args.voice_dataset_path),
|
||||
"voice_model_path": format(MODEL_PATH)
|
||||
},
|
||||
"out": {
|
||||
"score": sim_score
|
||||
}
|
||||
}
|
||||
|
||||
# determine whether CUDA support is available and set device parameters accordingly
|
||||
use_cuda = torch.cuda.is_available()
|
||||
if args.output_format == "txt":
|
||||
print(" + CUDA availability : {}" . format(use_cuda))
|
||||
|
||||
if args.use_cpu:
|
||||
device = "cpu"
|
||||
USE_CUDA = False
|
||||
elif use_cuda:
|
||||
device = "cuda"
|
||||
USE_CUDA = True
|
||||
else:
|
||||
device = "cpu"
|
||||
USE_CUDA = False
|
||||
if args.output_format == "txt":
|
||||
print(" + Compute device used : {}" . format(device))
|
||||
print("")
|
||||
|
||||
# score the cloned voice (wrt. similarity to recorded voice)
|
||||
# 1. generate 20 samples (5 x samples from Potion's Web-site; 5 x salutations; 10 x test sentences from other research papers)
|
||||
# 2. compute similarity score (training samples versus generated samples)
|
||||
scoring_sentences = [
|
||||
"Book more meetings, build more trust, and close more sales using Potion.",
|
||||
"Free forever. As long as you hustle. No credit card required.",
|
||||
"Don't send plain old boring text emails. Send Potion.",
|
||||
"What distinguished you from everyone else?",
|
||||
"We absolutely ensure that you see increased engagement in your outreach efforts.",
|
||||
"Hi there, Samuel. Hope things are going well for you.",
|
||||
"Hey person_93. I wanted to reach out to see if you are interested to learn mode about our services.",
|
||||
"Hi person_90. I noticed you and I are both members of the Green Movement on LinkedIn, and that you just opened a new office in Austin.",
|
||||
"Hey person_96. Could your team handle an extra 20 leads a week?",
|
||||
"Hi person_95. For every 100 cold emails you send, you'll only get one reply. That's a lot of effort for little reward.",
|
||||
"Prosecutors have opened a massive investigation into allegations of fixing games and illegal betting.",
|
||||
"Feedback must be timely and accurate throughout the project.",
|
||||
"Humans also judge distance by using the relative sizes of objects.",
|
||||
"If this is true then those who tend to think creatively really are somehow different.",
|
||||
"But really in the grand scheme of things this information is insignificant.",
|
||||
"About half the people who are infected also lose weight.",
|
||||
"The second half of the book focuses on argument and essay writing.",
|
||||
"He loves to watch me drink this stuff.",
|
||||
"Funding is always an issue after the fact.",
|
||||
"Let us encourage each other."
|
||||
]
|
||||
|
||||
# init speaker manager
|
||||
speaker_manager = None
|
||||
speaker_manager = SpeakerManager(d_vectors_file_path = SPK_EMBEDDINGS_PATH)
|
||||
if args.output_format == "txt":
|
||||
print(" + No. of speakers : {}" . format(speaker_manager.num_speakers))
|
||||
print(" + Speaker's names : {}" . format(speaker_manager.embedding_names))
|
||||
print(" + No. of embeddings : {}" . format(speaker_manager.num_embeddings))
|
||||
print("")
|
||||
|
||||
# assert that only one speaker is present in the embedding's file
|
||||
assert speaker_manager.num_speakers == 1, f"Number of speakers in the given embedding's file MUST be one; found {speaker_manager.num_speakers} speakers!"
|
||||
|
||||
# initialise speech synthesization
|
||||
voice_config, voice_model = init_synth(CONFIG_PATH, MODEL_PATH, speaker_embeddings_file = SPK_EMBEDDINGS_PATH, use_cuda = USE_CUDA)
|
||||
|
||||
# synthesize speech for all scoring sentences
|
||||
output_path = os.path.join(args.temp_path, str(uuid.uuid4()))
|
||||
|
||||
# create temp path (exit if it already exists)
|
||||
os.makedirs(output_path, exist_ok = False)
|
||||
|
||||
for cnt, txt in enumerate(scoring_sentences):
|
||||
# speaker embeddings provided, use it together with the given model (cloned or baseline model) to synthesize speech
|
||||
waveform = synthesize(voice_config, voice_model, txt, speaker_manager.get_mean_embedding(speaker_manager.embedding_names[0], speaker_manager.num_embeddings), USE_CUDA)
|
||||
|
||||
# save the results
|
||||
output_fname = os.path.join(output_path, "{:02d}" . format(cnt) + ".wav")
|
||||
save_waveform(voice_model, waveform, output_fname)
|
||||
|
||||
# initialize voice envcoder used for scoring
|
||||
scoring_vocoder = init_scoring_vocoder()
|
||||
|
||||
# determine similarity score
|
||||
sim_score = score_speaker_similarity(scoring_vocoder, args.voice_dataset_path, output_path)
|
||||
|
||||
# clean up
|
||||
if not args.keep_temp:
|
||||
shutil.rmtree(output_path)
|
||||
|
||||
# exit gracefully
|
||||
if args.output_format == "txt":
|
||||
print("")
|
||||
print("Completed computing similarity score for the two sets of recordings. The resulting similarity score is:")
|
||||
print(" --> {}" . format(sim_score))
|
||||
print("")
|
||||
print("Done; bye.")
|
||||
print("")
|
||||
elif args.output_format == "json":
|
||||
json_data["out"]["score"] = round(float(sim_score), 2)
|
||||
json_data["success"] = True
|
||||
print(json.dumps(json_data))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# parse command line arguments
|
||||
args = parse_cmdline_args()
|
||||
|
||||
# ensure the temp path exists
|
||||
os.makedirs(args.temp_path, exist_ok = True)
|
||||
|
||||
main(args)
|
||||
@@ -0,0 +1,219 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import os
|
||||
import argparse
|
||||
import glob
|
||||
import shutil
|
||||
|
||||
import uuid
|
||||
import json
|
||||
import torch
|
||||
|
||||
from TTS.tts.utils.speakers import SpeakerManager
|
||||
from utils.synthesize_utils import init_synth, synthesize, save_waveform
|
||||
from utils.scoring_utils import init_scoring_vocoder, score_speaker_similarity
|
||||
|
||||
|
||||
#
|
||||
# What do we need?
|
||||
# -> list of models to test
|
||||
# -> test db (user recordings, speaker embeddings, reference to their voice in the multi-speaker model)
|
||||
# |- user
|
||||
# |- speaker.pth
|
||||
# |- userid.txt
|
||||
# |- txt
|
||||
# |- wav48
|
||||
#
|
||||
|
||||
|
||||
#
|
||||
# parse command line arguments
|
||||
#
|
||||
def parse_cmdline_args():
|
||||
parser = argparse.ArgumentParser(
|
||||
description = "Given a list of models, compute quality scores to determine the top-5 (human-perceived) models.")
|
||||
parser.add_argument("--models_path", type = str, required = True,
|
||||
help = "Path to a collection of models and their config file to be used for testing.")
|
||||
parser.add_argument("--speaker_embeddings_path_list", type = str, nargs = "+", required = True,
|
||||
help = "List of paths to the speaker embeddings files of the data sets used to train the models.")
|
||||
parser.add_argument("--test_dataset_path", type = str, required = True,
|
||||
help = "Path to a set of user recordings with speaker embedding and voice id (the users' ids in the models to be tested)")
|
||||
parser.add_argument("--temp_path", type = str, default = "temp",
|
||||
help = "Path to store temporary speech assets")
|
||||
parser.add_argument("--keep_temp", default = False, action = "store_true",
|
||||
help = "Signal that temporary assets used for scoring should not be deleted once done")
|
||||
parser.add_argument("--use_cpu", default = False, action = "store_true", # untested!!!
|
||||
help = "Signal that CPU should be used even if a CUDA-device is available")
|
||||
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
|
||||
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
|
||||
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
#
|
||||
# main training method (voice cloning)
|
||||
#
|
||||
def main(args):
|
||||
# define assets required for using a pretrained voice
|
||||
MODELS_PATH = args.models_path
|
||||
MODEL_CONFIG_PATH = os.path.join(MODELS_PATH, "config.json")
|
||||
MODEL_SPK_EMB_PATH_LIST = args.speaker_embeddings_path_list
|
||||
DATASET_PATH = args.test_dataset_path
|
||||
USER_ID_FNAME = "userid.txt"
|
||||
|
||||
# set default score
|
||||
sim_score_avg = -1.0
|
||||
|
||||
if args.output_format == "txt":
|
||||
print("Computing similarity score for a given voice model (cloned voice) wrt. a given set of voice recordings (original voice):")
|
||||
print("")
|
||||
print(" + Multi-speaker models path : {}" . format(MODELS_PATH))
|
||||
print(" + Multi-speaker model config file: {}" . format(MODEL_CONFIG_PATH))
|
||||
print(" + Multi-speaker embeddings file : {}" . format(MODEL_SPK_EMB_PATH_LIST))
|
||||
print(" + Test dataset path : {}" . format(DATASET_PATH))
|
||||
#print(" + Speaker embeddings filename : {}" . format(SPK_EMBEDDINGS_FNAME))
|
||||
print(" + User ID filename : {}" . format(USER_ID_FNAME))
|
||||
print("")
|
||||
elif args.output_format == "json":
|
||||
json_data = {
|
||||
"success": False,
|
||||
"in": {
|
||||
"models_path": format(MODELS_PATH),
|
||||
"dataset_path": format(DATASET_PATH)
|
||||
},
|
||||
"out": {
|
||||
"best_model": None,
|
||||
"top_5_models": None
|
||||
}
|
||||
}
|
||||
|
||||
# determine whether CUDA support is available and set device parameters accordingly
|
||||
use_cuda = torch.cuda.is_available()
|
||||
if args.output_format == "txt":
|
||||
print(" + CUDA availability : {}" . format(use_cuda))
|
||||
|
||||
if args.use_cpu:
|
||||
device = "cpu"
|
||||
USE_CUDA = False
|
||||
elif use_cuda:
|
||||
device = "cuda"
|
||||
USE_CUDA = True
|
||||
else:
|
||||
device = "cpu"
|
||||
USE_CUDA = False
|
||||
if args.output_format == "txt":
|
||||
print(" + Compute device used : {}" . format(device))
|
||||
print("")
|
||||
|
||||
# score the cloned voice (wrt. similarity to recorded voice)
|
||||
# 1. generate 20 samples (5 x samples from Potion's Web-site; 5 x salutations; 10 x test sentences from other research papers)
|
||||
# 2. compute similarity score (training samples versus generated samples)
|
||||
scoring_sentences = [
|
||||
"Book more meetings, build more trust, and close more sales using Potion.",
|
||||
"Free forever. As long as you hustle. No credit card required.",
|
||||
"Don't send plain old boring text emails. Send Potion.",
|
||||
"What distinguished you from everyone else?",
|
||||
"We absolutely ensure that you see increased engagement in your outreach efforts.",
|
||||
"Hi there, Samuel. Hope things are going well for you.",
|
||||
"Hey person_93. I wanted to reach out to see if you are interested to learn mode about our services.",
|
||||
"Hi person_90. I noticed you and I are both members of the Green Movement on LinkedIn, and that you just opened a new office in Austin.",
|
||||
"Hey person_96. Could your team handle an extra 20 leads a week?",
|
||||
"Hi person_95. For every 100 cold emails you send, you'll only get one reply. That's a lot of effort for little reward.",
|
||||
"Prosecutors have opened a massive investigation into allegations of fixing games and illegal betting.",
|
||||
"Feedback must be timely and accurate throughout the project.",
|
||||
"Humans also judge distance by using the relative sizes of objects.",
|
||||
"If this is true then those who tend to think creatively really are somehow different.",
|
||||
"But really in the grand scheme of things this information is insignificant.",
|
||||
"About half the people who are infected also lose weight.",
|
||||
"The second half of the book focuses on argument and essay writing.",
|
||||
"He loves to watch me drink this stuff.",
|
||||
"Funding is always an issue after the fact.",
|
||||
"Let us encourage each other."
|
||||
]
|
||||
|
||||
|
||||
# init scoring tracker
|
||||
sim_score = {}
|
||||
for model_fname in glob.glob(os.path.join(MODELS_PATH, "check*.pth")):
|
||||
# init scoring tracker
|
||||
sim_score[os.path.basename(model_fname)] = []
|
||||
|
||||
# score each moddel for every user
|
||||
for user_dir in os.listdir(DATASET_PATH):
|
||||
|
||||
# get user's speaker id / name
|
||||
with open(os.path.join(DATASET_PATH, user_dir, USER_ID_FNAME), 'r') as f:
|
||||
user_data = json.load(f)
|
||||
print("Speaker name: {}" . format(user_data["speaker_name"]))
|
||||
|
||||
# init speaker manager
|
||||
speaker_manager = None
|
||||
speaker_manager = SpeakerManager(d_vectors_file_path = MODEL_SPK_EMB_PATH_LIST)
|
||||
#speaker_manager = SpeakerManager(speaker_id_file_path = os.path.join(MODELS_PATH, "speakers.pth"))
|
||||
|
||||
print("Number of speakers:", speaker_manager.num_speakers)
|
||||
print("Speaker names :", speaker_manager.speaker_names)
|
||||
#print("Embedding names :", speaker_manager.embedding_names)
|
||||
|
||||
# assert that the user is indeed present in the embedding's file
|
||||
#assert speaker_manager.num_speakers == 1, f"Number of speakers in the given embedding's file MUST be one; found {speaker_manager.num_speakers} speakers!"
|
||||
|
||||
for model_fname in glob.glob(os.path.join(MODELS_PATH, "check*.pth")):
|
||||
|
||||
# initialise speech synthesization
|
||||
voice_config, voice_model = init_synth(MODEL_CONFIG_PATH, model_fname, speaker_embeddings_file = MODEL_SPK_EMB_PATH_LIST, use_cuda = USE_CUDA)
|
||||
|
||||
# synthesize speech for all scoring sentences
|
||||
output_path = os.path.join(args.temp_path, str(uuid.uuid4()))
|
||||
|
||||
# create temp path (exit if it already exists)
|
||||
os.makedirs(output_path, exist_ok = False)
|
||||
|
||||
for cnt, txt in enumerate(scoring_sentences):
|
||||
# speaker embeddings provided, use it together with the given model (cloned or baseline model) to synthesize speech
|
||||
#waveform = synthesize(voice_config, voice_model, txt, speaker_manager.get_mean_embedding(user_data["speaker_name"]), USE_CUDA)
|
||||
waveform = synthesize(voice_config, voice_model, txt, speaker_embeddings = speaker_manager.get_mean_embedding(user_data["speaker_name"], num_samples = None, randomize = False), use_cuda = USE_CUDA)
|
||||
#waveform = synthesize(voice_config, voice_model, txt, speaker_embeddings = speaker_manager.get_mean_embedding(user_data["speaker_name"]), speaker_id = speaker_manager.name_to_id[user_data["speaker_name"]], use_cuda = USE_CUDA)
|
||||
#waveform = synthesize(voice_config, voice_model, txt, speaker_id = speaker_manager.name_to_id[user_data["speaker_name"]], use_cuda = USE_CUDA)
|
||||
|
||||
# save the results
|
||||
output_fname = os.path.join(output_path, "{:02d}" . format(cnt) + ".wav")
|
||||
save_waveform(voice_config, voice_model, waveform, output_fname)
|
||||
|
||||
# initialize voice envcoder used for scoring
|
||||
scoring_vocoder = init_scoring_vocoder()
|
||||
|
||||
# determine similarity score
|
||||
sim_score[os.path.basename(model_fname)].append(score_speaker_similarity(scoring_vocoder, os.path.join(DATASET_PATH, user_dir, "wav48", "1"), output_path))
|
||||
|
||||
# clean up
|
||||
if not args.keep_temp:
|
||||
shutil.rmtree(output_path)
|
||||
|
||||
# determine the top-5 checkpoints (or fewer if there are less than 5 entries)
|
||||
top_5_checkpoints = [(k, sum(v) / len(v)) for k, v in sorted(sim_score.items(), key = lambda item: sum(item[1]) / len(item[1]), reverse = True)[:5]]
|
||||
|
||||
# exit gracefully
|
||||
if args.output_format == "txt":
|
||||
print("")
|
||||
print("Completed computing similarity score for the two sets of recordings. The best and the top-5 models based on their similarity scores are:")
|
||||
print(" --> Best model : {}" . format(top_5_checkpoints[0]))
|
||||
print(" --> Top-5 models: {}" . format(top_5_checkpoints))
|
||||
print("")
|
||||
print("Done; bye.")
|
||||
print("")
|
||||
elif args.output_format == "json":
|
||||
json_data["out"]["best_model"] = top_5_checkpoints[0]
|
||||
json_data["out"]["top_5_models"] = top_5_checkpoints
|
||||
json_data["success"] = True
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# parse command line arguments
|
||||
args = parse_cmdline_args()
|
||||
|
||||
# ensure the temp path exists
|
||||
os.makedirs(args.temp_path, exist_ok = True)
|
||||
|
||||
main(args)
|
||||
@@ -0,0 +1,161 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import argparse
|
||||
|
||||
import re
|
||||
from itertools import combinations
|
||||
import json
|
||||
|
||||
from utils.matching_utils import match_name_textualsim, match_name_mra
|
||||
from utils.transcription_utils import get_transcription
|
||||
|
||||
|
||||
#
|
||||
# parse command line arguments
|
||||
#
|
||||
def parse_cmdline_args():
|
||||
parser = argparse.ArgumentParser(
|
||||
description = "Score a given salutation recording wrt. its desired content, the actual salutation recording, and a generated transcription (using Potion's internal Transciption API) of the recording.")
|
||||
parser.add_argument("--recording_path", type = str, required = True,
|
||||
help = "Path to salutation recoding (.wav audio file)")
|
||||
parser.add_argument("--first_name", type = str, required = True,
|
||||
help = "First name that the salutation recoding is meant to use")
|
||||
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
|
||||
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
|
||||
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
#
|
||||
# load dictionary of common first names (Name DB source: World Gender Name Dictionary v2.0; https://dataverse.harvard.edu/dataset.xhtml?persistentId=doi:10.7910/DVN/MSEGSJ)
|
||||
#
|
||||
def load_names():
|
||||
NAME_DICTIONARY = "./assets/wgnd_2_0_unique_names_only_limited_special_chars.csv"
|
||||
|
||||
# removing the characters
|
||||
with open(NAME_DICTIONARY) as f:
|
||||
names_list = [line.rstrip() for line in f]
|
||||
|
||||
names_set = set(names_list)
|
||||
|
||||
return names_set
|
||||
|
||||
|
||||
#
|
||||
# auxilliary function to generate a list of all combinations of words from a given list of words (w/o chaningthe order of words)
|
||||
#
|
||||
def get_combinations(word_list):
|
||||
comb_list = word_list.copy()
|
||||
for start, end in combinations(range(len(word_list)), 2):
|
||||
comb_list.append(' '.join(word for word in word_list[start:end + 1]))
|
||||
|
||||
return comb_list
|
||||
|
||||
|
||||
#
|
||||
# main method
|
||||
#
|
||||
def main(args):
|
||||
|
||||
# set default score
|
||||
score = -1.0
|
||||
|
||||
if args.output_format == "txt":
|
||||
print("Commencing scoring of the given salutation recording:")
|
||||
print("")
|
||||
print(" + Salutation recording path: {}" . format(args.recording_path))
|
||||
print(" + Salutation first name : {}" . format(args.first_name))
|
||||
print("")
|
||||
elif args.output_format == "json":
|
||||
json_data = {
|
||||
"success": False,
|
||||
"in": {
|
||||
"recording_path": format(args.recording_path),
|
||||
"first_name": format(args.first_name)
|
||||
},
|
||||
"out": {
|
||||
"score": score
|
||||
}
|
||||
}
|
||||
|
||||
# obtain a transcription for the given salutation recording
|
||||
trans_success, trans_txt, trans_score = get_transcription(args.recording_path)
|
||||
|
||||
# proceed if a transcription was obtained successfully
|
||||
if trans_success:
|
||||
# check given first name against name database (Name DB source: https://dataverse.harvard.edu/dataset.xhtml?persistentId=doi:10.7910/DVN/MSEGSJ)
|
||||
names_set = load_names()
|
||||
|
||||
# ensure all words / letters are lower case only
|
||||
first_name = args.first_name.lower()
|
||||
trans_txt = trans_txt.lower()
|
||||
|
||||
name_valid = False
|
||||
if first_name in names_set:
|
||||
name_valid = True
|
||||
else:
|
||||
# cannot compute advanced score for a name that we do not have in our first name database (i.e., fallback to confidence score from transcription service)
|
||||
if args.output_format == "txt":
|
||||
print("Unknown first name: {}" . format(args.first_name))
|
||||
print("")
|
||||
score = trans_score
|
||||
|
||||
if name_valid:
|
||||
#
|
||||
trans_candidate_names = re.findall(r" ([a-zA-Z_-]+)", trans_txt)
|
||||
|
||||
if len(trans_candidate_names) >= 2:
|
||||
trans_candidate_names = get_combinations(trans_candidate_names)
|
||||
|
||||
cand_names_real = []
|
||||
for cand_name in trans_candidate_names:
|
||||
# check is cname is a valid name
|
||||
if cand_name.lower() in names_set:
|
||||
cand_names_real.append(cand_name)
|
||||
|
||||
if cand_names_real:
|
||||
# name similarity with first_name
|
||||
for cand_name_real in cand_names_real:
|
||||
# check is cname is a valid name
|
||||
jaro, lev = match_name_textualsim(first_name, cand_name_real)
|
||||
mra = match_name_mra(first_name, cand_name_real)
|
||||
|
||||
#print("Scores ({}): {} -- {} -- {} -- {}" . format(cand_name_real, trans_score, jaro, lev, mra))
|
||||
|
||||
# score if the normalised Jaro-Winkler distance >= 0.875
|
||||
# OR
|
||||
# the normalised Jaro-Winkler distance >= 0.75 and the normalised Levenshtein distance is >= 0.7
|
||||
# OR
|
||||
# the normalised MRA >= 0.75
|
||||
# else average
|
||||
if jaro > 0.875:
|
||||
score = (jaro + trans_score) / 2
|
||||
break
|
||||
elif (jaro >= 0.75) and (lev >= 0.7):
|
||||
score = (((jaro + lev) / 2) + trans_score) / 2
|
||||
break
|
||||
elif mra >= 0.75:
|
||||
score = (mra + trans_score) / 2
|
||||
break
|
||||
else:
|
||||
score_new = (((jaro + lev + mra) / 3) + trans_score) / 2
|
||||
if score_new > score:
|
||||
score = score_new
|
||||
|
||||
if args.output_format == "txt":
|
||||
print(" >> Salutation score : {}" . format(score))
|
||||
print("")
|
||||
print("Done; bye.")
|
||||
print("")
|
||||
elif args.output_format == "json":
|
||||
json_data["out"]["score"] = round(score, 2)
|
||||
json_data["success"] = True
|
||||
print(json.dumps(json_data))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# parse command line arguments
|
||||
args = parse_cmdline_args()
|
||||
|
||||
main(args)
|
||||
@@ -0,0 +1,144 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import os
|
||||
import argparse
|
||||
import subprocess
|
||||
|
||||
import json
|
||||
import uuid
|
||||
import torch
|
||||
|
||||
from TTS.tts.utils.speakers import SpeakerManager
|
||||
from utils.synthesize_utils import init_synth, synthesize, save_waveform
|
||||
|
||||
#
|
||||
# parse command line arguments
|
||||
#
|
||||
def parse_cmdline_args():
|
||||
parser = argparse.ArgumentParser(
|
||||
description = "Code to synthesize speech for a given voice model")
|
||||
parser.add_argument("--voice_model_path", type = str, required = True,
|
||||
help = "Path to cloned voice model")
|
||||
parser.add_argument("--voice_model_config_path", type = str, required = True,
|
||||
help = "Path to config file for the cloned voice model")
|
||||
parser.add_argument('--speaker_embeddings_path', type = str, required = True,
|
||||
help = "Path to speaker's embeddings file (i.e., pre-computed embeddings typically stored with the speaker's dataset)")
|
||||
parser.add_argument("--txt", type = str, required = True,
|
||||
help = "Text to synthesize")
|
||||
parser.add_argument("--output_path", type = str, default = "results/speech",
|
||||
help = "Path to store generated speech assets")
|
||||
parser.add_argument("--target_sampling_rate", type = int, default = 48000,
|
||||
help = "Desired sampling rate (in Hz) for output file")
|
||||
parser.add_argument('--speech_sample_wav_path', type = str, default = None,
|
||||
help = "Path to a sample utterance of the speaker (used for style transfer)")
|
||||
parser.add_argument('--speech_sample_txt', type = str, default = None,
|
||||
help = "Text of the sample utterance of the speaker (used for style transfer)")
|
||||
parser.add_argument("--trim_silence", default = True, action = "store_false",
|
||||
help = "Signal whether to trim silence from synthesised speech")
|
||||
parser.add_argument("--use_cpu", default = False, action = "store_true", # untested!!!
|
||||
help = "Signal that CPU should be used even if a CUDA-device is available")
|
||||
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
|
||||
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
|
||||
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
#
|
||||
# main speech synthesizing method
|
||||
#
|
||||
def main(args):
|
||||
# define assets required for using a pretrained voice
|
||||
MODEL_PATH = args.voice_model_path
|
||||
CONFIG_PATH = args.voice_model_config_path
|
||||
SPK_EMBEDDINGS_PATH = args.speaker_embeddings_path
|
||||
|
||||
if args.output_format == "txt":
|
||||
print("Commencing speech synthesizing:")
|
||||
print("")
|
||||
print(" + Voice model file path : {}" . format(MODEL_PATH))
|
||||
print(" + Voice model config file: {}" . format(CONFIG_PATH))
|
||||
print(" + Speaker embeddings file: {}" . format(SPK_EMBEDDINGS_PATH))
|
||||
print(" + Output path : {}" . format(args.output_path))
|
||||
print(" + Text to synthesize : {}" . format(args.txt))
|
||||
print("")
|
||||
elif args.output_format == "json":
|
||||
json_data = {
|
||||
"success": False,
|
||||
"in": {
|
||||
"voice_model_path": format(MODEL_PATH),
|
||||
"voice_model_config_path": format(CONFIG_PATH),
|
||||
"speaker_embeddings_path": format(SPK_EMBEDDINGS_PATH)
|
||||
},
|
||||
"out": {
|
||||
"speech_original_path": "",
|
||||
"speech_resampled_path": ""
|
||||
}
|
||||
}
|
||||
|
||||
# determine whether CUDA support is available and set device parameters accordingly
|
||||
use_cuda = torch.cuda.is_available()
|
||||
if args.output_format == "txt":
|
||||
print(" + CUDA availability : {}" . format(use_cuda))
|
||||
|
||||
if args.use_cpu:
|
||||
device = "cpu"
|
||||
USE_CUDA = False
|
||||
elif use_cuda:
|
||||
device = "cuda"
|
||||
USE_CUDA = True
|
||||
else:
|
||||
device = "cpu"
|
||||
USE_CUDA = False
|
||||
if args.output_format == "txt":
|
||||
print(" + Compute device used : {}" . format(device))
|
||||
|
||||
# init speaker manager
|
||||
speaker_manager = None
|
||||
speaker_manager = SpeakerManager(d_vectors_file_path = SPK_EMBEDDINGS_PATH)
|
||||
if args.output_format == "txt":
|
||||
print(" + No. of speakers : {}" . format(speaker_manager.num_speakers))
|
||||
print(" + Speaker's names : {}" . format(speaker_manager.embedding_names))
|
||||
print(" + No. of embeddings : {}" . format(speaker_manager.num_embeddings))
|
||||
print("")
|
||||
|
||||
# assert that only one speaker is present in the embedding's file
|
||||
assert speaker_manager.num_speakers == 1, f"Number of speakers in the given embedding's file MUST be one; found {speaker_manager.num_speakers} speakers!"
|
||||
|
||||
# initialise speech synthesization
|
||||
voice_config, voice_model = init_synth(CONFIG_PATH, MODEL_PATH, speaker_embeddings_file = SPK_EMBEDDINGS_PATH, use_cuda = USE_CUDA)
|
||||
|
||||
# synthesize speech
|
||||
waveform = synthesize(voice_config, voice_model, args.txt, speaker_embeddings = speaker_manager.get_mean_embedding(speaker_manager.embedding_names[0], speaker_manager.num_embeddings), speech_sample_wav = args.speech_sample_wav_path, speech_sample_txt = args.speech_sample_txt, use_cuda = USE_CUDA, trim_silence = args.trim_silence)
|
||||
|
||||
# save the synthesize speech
|
||||
output_fname_prefix = str(uuid.uuid4())
|
||||
output_fname = output_fname_prefix + ".wav"
|
||||
save_waveform(voice_config, voice_model, waveform, os.path.join(args.output_path, output_fname))
|
||||
|
||||
# convert the synthesize speech waveform to the target sampling rate
|
||||
output_resampled_fname = output_fname_prefix + "_sr" + str(args.target_sampling_rate) + ".wav"
|
||||
subprocess.run(["ffmpeg", "-i", os.path.join(args.output_path, output_fname), "-ar", str(args.target_sampling_rate), os.path.join(args.output_path, output_resampled_fname)], check=True)
|
||||
|
||||
# exit gracefully
|
||||
if args.output_format == "txt":
|
||||
print("")
|
||||
print(">>> Saving origianl output to : {}" . format(os.path.join(args.output_path, output_fname)))
|
||||
print(">>> Saving resampled output to: {}" . format(os.path.join(args.output_path, output_resampled_fname)))
|
||||
print("")
|
||||
print("Speech synthesizing has completed. Bye.")
|
||||
elif args.output_format == "json":
|
||||
json_data["out"]["speech_original_path"] = format(os.path.join(args.output_path, output_fname))
|
||||
json_data["out"]["speech_resampled_path"] = format(os.path.join(args.output_path, output_resampled_fname))
|
||||
json_data["success"] = True
|
||||
print(json.dumps(json_data))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# parse command line arguments
|
||||
args = parse_cmdline_args()
|
||||
|
||||
# ensure the output path exists
|
||||
os.makedirs(args.output_path, exist_ok = True)
|
||||
|
||||
main(args)
|
||||
@@ -0,0 +1,48 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
####################################################################################
|
||||
### ###
|
||||
### Configuration File for Multi-Speaker Baseline Model Training & Voice Cloning ###
|
||||
### ###
|
||||
####################################################################################
|
||||
|
||||
import os
|
||||
|
||||
# Data Sets
|
||||
|
||||
## VCTK (v0.92), sampling rate: 48000
|
||||
|
||||
VCTK_PRESET = "VCTK"
|
||||
VCTK_DATASET_NAME = "VCTK"
|
||||
VCTK_DATASET_FORMATTER = "vctk"
|
||||
VCTK_DATASET_FILE_FORMAT = "flac"
|
||||
VCTK_DATASET_PATH = "results/datasets/sr22050/VCTK-Corpus-0.92"
|
||||
VCTK_SPK_EMB_PATH = os.path.join(VCTK_DATASET_PATH, "speakers.pth")
|
||||
|
||||
## LibriTTS TC360, sampling rate: 24000
|
||||
|
||||
LIBRITTS_TC360_PRESET = "LibriTTS_tc360"
|
||||
LIBRITTS_TC360_DATASET_NAME = "LibtriTTS-tc360"
|
||||
LIBRITTS_TC360_DATASET_FORMATTER = "libri_tts"
|
||||
LIBRITTS_TC360_DATASET_FILE_FORMAT = "wav"
|
||||
LIBRITTS_TC360_DATASET_PATH = "results/datasets/sr22050/LibriTTS/train-clean-360"
|
||||
LIBRITTS_TC360_SPK_EMB_PATH = os.path.join(LIBRITTS_TC360_DATASET_PATH, "speakers.pth")
|
||||
|
||||
## DAPS
|
||||
|
||||
## Potion salutation recordings
|
||||
|
||||
POTION_SALUT_PRESET = "POTION_Salut"
|
||||
POTION_SALUT_DATASET_NAME = "potion-Salut"
|
||||
POTION_SALUT_DATASET_FORMATTER = "vctk_old"
|
||||
POTION_SALUT_DATASET_FILE_FORMAT = "wav"
|
||||
POTION_SALUT_DATASET_PATH = "results/datasets/sr22050/potion-salut-corpus-4ac24ce8-8405-4b70-8b48-018d4492f6e9"
|
||||
POTION_SALUT_SPK_EMB_PATH = os.path.join(POTION_SALUT_DATASET_PATH, "speakers.pth")
|
||||
|
||||
## Potion voice cloning recordings
|
||||
|
||||
POTION_SALUT_PRESET = "potion_voice_cloning"
|
||||
POTION_SALUT_DATASET_NAME = ""
|
||||
POTION_SALUT_DATASET_FORMATTER = "vctk_old"
|
||||
POTION_SALUT_DATASET_FILE_FORMAT = "wav"
|
||||
@@ -0,0 +1,238 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import sys
|
||||
import os
|
||||
import argparse
|
||||
|
||||
import torch
|
||||
|
||||
# load coqui-ai/trainer libraries
|
||||
from trainer import Trainer, TrainerArgs
|
||||
|
||||
# load coqui-ai/TTS libraries
|
||||
from TTS.tts.configs.shared_configs import BaseDatasetConfig
|
||||
from TTS.tts.configs.vits_config import VitsConfig
|
||||
from TTS.tts.datasets import load_tts_samples
|
||||
from TTS.tts.models.vits import Vits, VitsArgs, VitsAudioConfig
|
||||
|
||||
import train_config as tc
|
||||
|
||||
#
|
||||
# parse command line arguments
|
||||
#
|
||||
def parse_cmdline_args():
|
||||
parser = argparse.ArgumentParser(
|
||||
description = "Code to train multi-speaker baseline model")
|
||||
parser.add_argument("--datasets", type = str, nargs = "+", required = True,
|
||||
choices = (tc.VCTK_PRESET, tc.LIBRITTS_TC360_PRESET, tc.POTION_SALUT_PRESET),
|
||||
help = "List of training datasets to be included in training run.")
|
||||
parser.add_argument("--output_path", type = str, default = "results/baseline-models",
|
||||
help = "Path to store trained / generated assets")
|
||||
parser.add_argument("--batch_size", type = int, default = 32, # 96 is suitable for AWS g5 instances using VCTK v0.80 only
|
||||
help = "Batch size for training run") # 32 is suitable for AWS g5 instances using VCTK v0.92, LibriTTS 360 and Potion salutations
|
||||
parser.add_argument("--max_epochs", type = int, default = 100, # 250 for batch size 64 (with VCTK only)
|
||||
help = "Maximum number of epochs for training run") # 100 for batch size 32 (with VCTK v0.92, LibriTTS 360 and POTION_Salut)
|
||||
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
#
|
||||
# main training method (VITS multi-speaker model)
|
||||
#
|
||||
def main(args):
|
||||
print("Commencing training of a new multi-speaker potion-voice baseline model:")
|
||||
print("")
|
||||
print(" + Datasets : {}" . format(args.datasets))
|
||||
print(" + Output path : {}" . format(args.output_path))
|
||||
print(" + Batch size : {}" . format(args.batch_size))
|
||||
print(" + Training runs (max epochs): {}" . format(args.max_epochs))
|
||||
print("")
|
||||
|
||||
# determine whether CUDA support is available and set device parameters accordingly
|
||||
use_cuda = torch.cuda.is_available()
|
||||
print(" + CUDA availability : {}" . format(use_cuda))
|
||||
|
||||
if use_cuda:
|
||||
device = "cuda"
|
||||
device_torch = torch.device("cuda")
|
||||
else:
|
||||
device = "cpu"
|
||||
device_torch = torch.device("cpu")
|
||||
print(" + Compute device used : {}" . format(device))
|
||||
print("")
|
||||
|
||||
# define training data sets
|
||||
dataset_config_list = []
|
||||
speaker_embeddings_list = []
|
||||
|
||||
# VCTK (v0.92)
|
||||
if tc.VCTK_PRESET in args.datasets:
|
||||
vctk_dataset_config = BaseDatasetConfig(dataset_name = tc.VCTK_DATASET_NAME, formatter = tc.VCTK_DATASET_FORMATTER, language = "en-us", path = tc.VCTK_DATASET_PATH)
|
||||
dataset_config_list.append(vctk_dataset_config)
|
||||
speaker_embeddings_list.append(tc.VCTK_SPK_EMB_PATH)
|
||||
|
||||
# LibriTTS
|
||||
if tc.LIBRITTS_TC360_PRESET in args.datasets:
|
||||
libritts_dataset_config = BaseDatasetConfig(dataset_name = tc.LIBRITTS_TC360_DATASET_NAME, formatter = tc.LIBRITTS_TC360_DATASET_FORMATTER, language = "en-us", path = tc.LIBRITTS_TC360_DATASET_PATH)
|
||||
dataset_config_list.append(libritts_dataset_config)
|
||||
speaker_embeddings_list.append(tc.LIBRITTS_TC360_SPK_EMB_PATH)
|
||||
|
||||
# DAPS
|
||||
|
||||
# Potion recordings dataset
|
||||
if tc.POTION_SALUT_PRESET in args.datasets:
|
||||
potion_dataset_config = BaseDatasetConfig(dataset_name = tc.POTION_SALUT_DATASET_NAME, formatter = tc.POTION_SALUT_DATASET_FORMATTER, language = "en-us", path = tc.POTION_SALUT_DATASET_PATH)
|
||||
dataset_config_list.append(potion_dataset_config)
|
||||
speaker_embeddings_list.append(tc.POTION_SALUT_SPK_EMB_PATH)
|
||||
|
||||
# set VITS training parameters
|
||||
audio_config = VitsAudioConfig(
|
||||
sample_rate = 22050,
|
||||
win_length = 1024,
|
||||
hop_length = 256,
|
||||
num_mels = 80,
|
||||
mel_fmin = 0,
|
||||
mel_fmax = None,
|
||||
)
|
||||
|
||||
vitsArgs = VitsArgs(
|
||||
use_speaker_embedding = False,
|
||||
use_d_vector_file = True,
|
||||
d_vector_file = speaker_embeddings_list,
|
||||
d_vector_dim = 512,
|
||||
num_layers_text_encoder = 10
|
||||
)
|
||||
|
||||
config = VitsConfig(
|
||||
model_args = vitsArgs,
|
||||
audio = audio_config,
|
||||
run_name = "vits_potion",
|
||||
use_speaker_embedding = False,
|
||||
use_d_vector_file = True,
|
||||
d_vector_file = speaker_embeddings_list,
|
||||
d_vector_dim = 512,
|
||||
batch_size = args.batch_size,
|
||||
eval_batch_size = 16,
|
||||
batch_group_size = 0, # changing this to 5 (VITS training default) slows training down, but doesn't have any positive training effects
|
||||
num_loader_workers = 4,
|
||||
num_eval_loader_workers = 4,
|
||||
run_eval = True,
|
||||
test_delay_epochs = -1,
|
||||
epochs = args.max_epochs,
|
||||
text_cleaner = "english_cleaners",
|
||||
use_phonemes = False,
|
||||
phoneme_language = "en-us",
|
||||
phoneme_cache_path = os.path.join(args.output_path, "phoneme_cache"),
|
||||
compute_input_seq_cache = True,
|
||||
print_step = 50,
|
||||
print_eval = True,
|
||||
mixed_precision = True,
|
||||
max_text_len = 325,
|
||||
output_path = args.output_path,
|
||||
|
||||
save_checkpoints = True,
|
||||
save_step = 5000,
|
||||
save_n_checkpoints = 20,
|
||||
save_all_best = True,
|
||||
|
||||
datasets = dataset_config_list,
|
||||
cudnn_benchmark = False,
|
||||
#characters = {
|
||||
# "pad": "_",
|
||||
# "eos": "&",
|
||||
# "bos": "*",
|
||||
# "characters": "!¡'(),-.:;¿?abcdefghijklmnopqrstuvwxyz «°±µ»$%&‘’‚“`”„",
|
||||
# "punctuations": "!¡'(),-.:;¿? ",
|
||||
# "phonemes": None,
|
||||
# "unique": True
|
||||
#},
|
||||
test_sentences = [
|
||||
# VCTK
|
||||
["It took me quite a long time to develop a voice, and now that I have it I'm not going to be silent.", "VCTK_p299"], # 299 - F, American, California
|
||||
["Hey! Sandra.", "VCTK_p302"], # 302 - M, Canadian, Montreal
|
||||
["I'm sorry Dave. I'm afraid I can't do that.", "VCTK_p308"], # 308 - F, American, Alabama
|
||||
["This cake is great. It's so delicious and moist.", "VCTK_p334"], # 334 - M, American, Chicago
|
||||
["Prior to November 22, 1963.", "VCTK_p363"], # 363 - M, Canadian, Toronto
|
||||
["It took me quite a long time to develop a voice, and now that I have it I'm not going to be silent.", "VCTK_p376"], # 376 - M, Indian
|
||||
|
||||
# LibriTTS
|
||||
["It took me quite a long time to develop a voice, and now that I have it I'm not going to be silent.", "LTTS_38"], # 38 - M - train-clean-360 R. Francis Smith
|
||||
["Hey! Sandra.", "LTTS_22"], # 22 - F - train-clean-360 Michelle Crandall
|
||||
["I'm sorry Dave. I'm afraid I can't do that.", "LTTS_329"], # 329 - M - train-clean-360 Todd Cranston-Cuebas
|
||||
["This cake is great. It's so delicious and moist.", "LTTS_224"], # 224 - F - train-clean-360 Caitlin Kelly
|
||||
["Prior to November 22, 1963.", "LTTS_339"], # 339 - F - train-clean-360 Heather Ordover
|
||||
["It took me quite a long time to develop a voice, and now that I have it I'm not going to be silent.", "LTTS_1779"], # 1779 - F - train-clean-360 Cynthia Zocca
|
||||
|
||||
# DAPS
|
||||
|
||||
# Potion Salutation Recordings
|
||||
["Hey! Andrew.", "VCTK_old_POTION_6231a04a9f5f7707120b4215"], # Potion user
|
||||
["Hey, Michelle.", "VCTK_old_POTION_628c025a09943300254532ae"], # Potion user
|
||||
["Hey! George.", "VCTK_old_POTION_6189854312bfc264a528c3c3"], # Potion user
|
||||
["Hey there, Rachel.", "VCTK_old_POTION_63163f7b2be1c500219f3d04"], # Potion user
|
||||
#["I'm sorry Dave. I'm afraid I can't do that.", "VCTK_old_POTION_63222943fd2bff2e1c651b3b"] # Potion user - not in yet
|
||||
]
|
||||
)
|
||||
|
||||
# load training samples
|
||||
train_samples, eval_samples = load_tts_samples(config.datasets, eval_split = True, eval_split_max_size = config.eval_split_max_size, eval_split_size = config.eval_split_size)
|
||||
|
||||
# init VITS model
|
||||
model = Vits.init_from_config(config)
|
||||
|
||||
# init multi-speaker training
|
||||
trainer = Trainer(
|
||||
TrainerArgs(),
|
||||
config,
|
||||
args.output_path,
|
||||
model = model,
|
||||
train_samples = train_samples,
|
||||
eval_samples = eval_samples
|
||||
)
|
||||
|
||||
# trigger model training
|
||||
try:
|
||||
trainer.fit()
|
||||
except (KeyboardInterrupt, SystemExit):
|
||||
print("Training stopped manually (via keyboard interrupt)! Bye.")
|
||||
exit(0)
|
||||
|
||||
# exit gracefully
|
||||
print("")
|
||||
print("Completed training a new multi-speaker potion-voice baseline model, which can be found at:")
|
||||
print(" --> {}" . format(args.output_path))
|
||||
print("")
|
||||
print("Done; bye.")
|
||||
print("")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# parse command line arguments
|
||||
args = parse_cmdline_args()
|
||||
|
||||
# clear command line arguments to avoid triggering argparse features part of Trainer / coqpit imports
|
||||
# Traceback (most recent call last):
|
||||
# File "train_multispeaker_baseline_model.py", line 208, in <module>
|
||||
# main(args)
|
||||
# File "train_multispeaker_baseline_model.py", line 177, in main
|
||||
# trainer = Trainer(
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 360, in __init__
|
||||
# config, new_fields = self.init_training(args, coqpit_overrides, config)
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 594, in init_training
|
||||
# config.parse_known_args(coqpit_overrides, relaxed_parser=True)
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 843, in parse_known_args
|
||||
# parser = self.init_argparse(arg_prefix=arg_prefix, relaxed_parser=relaxed_parser)
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 881, in init_argparse
|
||||
# _init_argparse(
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 529, in _init_argparse
|
||||
# parser = _init_argparse(
|
||||
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 550, in _init_argparse
|
||||
# return default.init_argparse(
|
||||
# AttributeError: 'str' object has no attribute 'init_argparse'
|
||||
sys.argv = [sys.argv[0]]
|
||||
|
||||
# ensure the output path exists
|
||||
os.makedirs(args.output_path, exist_ok = True)
|
||||
|
||||
main(args)
|
||||
@@ -0,0 +1,22 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import textdistance
|
||||
|
||||
|
||||
#
|
||||
# Name matching via textual similarity search
|
||||
# + Returns two (normalised) distance measures: the Jaro-Winkler Distance and the regular Levenshtein Distance
|
||||
#
|
||||
def match_name_textualsim(name1, name2):
|
||||
jaro_winkler = textdistance.jaro_winkler.normalized_similarity(name1, name2)
|
||||
levenshtein = textdistance.levenshtein.normalized_similarity(name1, name2)
|
||||
|
||||
return jaro_winkler, levenshtein
|
||||
|
||||
|
||||
#
|
||||
# Name matching via phonetic matching algorithm (using the normalised Match Rating Approach)
|
||||
#
|
||||
def match_name_mra(name1, name2):
|
||||
return textdistance.mra.normalized_similarity(name1, name2)
|
||||
@@ -0,0 +1,37 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
from pathlib import Path
|
||||
from itertools import groupby
|
||||
|
||||
import numpy as np
|
||||
|
||||
from resemblyzer import preprocess_wav, VoiceEncoder
|
||||
|
||||
|
||||
def init_scoring_vocoder():
|
||||
|
||||
# initialise voice encoder (using CUDA by default; CPU as fallback)
|
||||
encoder = VoiceEncoder()
|
||||
|
||||
return encoder
|
||||
|
||||
|
||||
def score_speaker_similarity(scoring_vocoder, spk_a_fpaths, spk_b_fpaths):
|
||||
|
||||
# filepaths to waveforms
|
||||
wav_fpaths = list(Path(spk_a_fpaths).glob("*.wav")) + list(Path(spk_b_fpaths).glob("*.wav"))
|
||||
|
||||
# group the wavs per speaker and load them using the preprocessing function provided with Resemblyzer to load wavs in memory
|
||||
# - normalizes the volume, trims long silences and resamples the wav to the correct sampling rate
|
||||
speaker_wavs = {speaker: list(map(preprocess_wav, wav_fpaths)) for speaker, wav_fpaths in groupby(wav_fpaths, lambda wav_fpath: wav_fpath.parent.stem)}
|
||||
|
||||
# compute similarity between two speaker embeddings
|
||||
# - divides the utterances of each speaker in groups of identical size and embed each group as a speaker embedding
|
||||
spk_embeds_a = np.array([scoring_vocoder.embed_speaker(wavs[:len(wavs) // 2]) for wavs in speaker_wavs.values()])
|
||||
spk_embeds_b = np.array([scoring_vocoder.embed_speaker(wavs[len(wavs) // 2:]) for wavs in speaker_wavs.values()])
|
||||
spk_sim_matrix = np.inner(spk_embeds_a, spk_embeds_b)
|
||||
|
||||
sim_score = np.average([spk_sim_matrix[0, 1], spk_sim_matrix[1, 0]])
|
||||
|
||||
return(sim_score)
|
||||
@@ -0,0 +1,77 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import numpy as np
|
||||
|
||||
# load coqui-ai/TTS libraries
|
||||
from TTS.config import load_config
|
||||
from TTS.tts.models import setup_model as setup_tts_model
|
||||
from TTS.tts.utils.synthesis import synthesis, trim_silence
|
||||
|
||||
|
||||
def init_synth(config_path, voice_model_path, speakers_file_path = None, speaker_embeddings_file = None, use_cuda = True, use_phonemes = False):
|
||||
|
||||
# load config and customise config parameters (those that are different during training and inference / synthesizing)
|
||||
config = load_config(config_path)
|
||||
|
||||
if not speakers_file_path is None:
|
||||
config.use_speaker_embedding = True,
|
||||
config.use_d_vector_file = False,
|
||||
config.speakers_file = speakers_file_path
|
||||
config.model_args["use_speaker_embedding"] = True,
|
||||
config.model_args["use_d_vector_file"] = False,
|
||||
config.model_args["speakers_file"] = speakers_file_path
|
||||
else:
|
||||
config.d_vector_file = speaker_embeddings_file
|
||||
config.model_args["d_vector_file"] = speaker_embeddings_file
|
||||
|
||||
# set whether or not phonemes are used
|
||||
config.use_phonemes = use_phonemes
|
||||
|
||||
# load cloned voice model
|
||||
model = setup_tts_model(config = config)
|
||||
model.load_checkpoint(config, voice_model_path, eval = True)
|
||||
|
||||
if use_cuda:
|
||||
model.cuda()
|
||||
|
||||
return config, model
|
||||
|
||||
|
||||
def synthesize(config, voice_model, txt, speaker_embeddings = None, speaker_id = None, speech_sample_wav = None, speech_sample_txt = None, use_cuda = True, trim_silence = True):
|
||||
|
||||
# disable language selection
|
||||
#language_id = 0
|
||||
language_id = None
|
||||
|
||||
# set default voice encoder
|
||||
use_gl = True
|
||||
|
||||
# synthesize voice
|
||||
outputs = synthesis(
|
||||
model = voice_model,
|
||||
text = txt,
|
||||
CONFIG = config,
|
||||
use_cuda = use_cuda,
|
||||
speaker_id = speaker_id,
|
||||
style_wav = speech_sample_wav,
|
||||
style_text = speech_sample_txt,
|
||||
use_griffin_lim = use_gl,
|
||||
do_trim_silence = trim_silence,
|
||||
d_vector = speaker_embeddings,
|
||||
language_id = language_id
|
||||
)
|
||||
|
||||
waveform = outputs["wav"]
|
||||
waveform = waveform.squeeze()
|
||||
|
||||
# trim silence (disabled due to some "TypeError: 'bool' object is not callable" bug that needs to be investigated)
|
||||
#if (config.audio["do_trim_silence"]) or (trim_silence):
|
||||
# waveform = trim_silence(waveform, voice_model.ap)
|
||||
|
||||
return waveform
|
||||
|
||||
|
||||
def save_waveform(config, voice_model, waveform, out_path):
|
||||
wav = np.array(waveform)
|
||||
voice_model.ap.save_wav(wav, out_path, config["audio"].sample_rate)
|
||||
@@ -0,0 +1,94 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import sys
|
||||
import os
|
||||
|
||||
import requests
|
||||
from requests.structures import CaseInsensitiveDict
|
||||
import json
|
||||
|
||||
from time import sleep
|
||||
|
||||
|
||||
# set transcription service API endpoint and token (retrieved from operating system's ENV variables)
|
||||
# + sample endpoints:
|
||||
# - [dev] "https://development.sendpotion.com/api/transcript"
|
||||
# - [staging] "https://staging.sendpotion.com/api/transcript"
|
||||
API_ENDPOINT = os.environ.get("TRANSCRIPTION_API_ENDPOINT")
|
||||
API_TOKEN = os.environ.get("TRANSCRIPTION_API_TOKEN")
|
||||
|
||||
|
||||
#
|
||||
# Using potions internal transcription API endpoint, get a transcription for a given (wav) audio recording
|
||||
# + returns a triple:
|
||||
# - Boolean ......... indicating success (True) or failure (False)
|
||||
# - String / None ... transcription text (or None in failure case)
|
||||
# - Float / None .... transcription confidence score (or None in failure case)
|
||||
#
|
||||
def get_transcription(wav_fname):
|
||||
|
||||
# validate that transcription service API endpoint and token are set
|
||||
if (API_ENDPOINT is None) or (API_TOKEN is None):
|
||||
# terminate
|
||||
print("TRANSCRIPTION_API_ENDPOINT and TRANSCRIPTION_API_TOKEN environment variables MUST be set!")
|
||||
sys.exit(1)
|
||||
|
||||
# set request header to contain (bearer) API token
|
||||
headers = CaseInsensitiveDict()
|
||||
headers["Accept"] = "application/json"
|
||||
headers["Authorization"] = "Bearer " + str(API_TOKEN)
|
||||
|
||||
# set files field (data is empty)
|
||||
files = {'wav': open(wav_fname, 'rb')}
|
||||
|
||||
# issue POST request and save response as response object
|
||||
response = requests.post(url = API_ENDPOINT, headers = headers, files = files)
|
||||
|
||||
# test for auth error
|
||||
# test for timeout
|
||||
|
||||
# check if the status code is not an error code (i.e., 4xx or 5xx)
|
||||
success = False
|
||||
if response:
|
||||
# extracting response text
|
||||
response_text = response.text
|
||||
response_json = json.loads(response_text)
|
||||
#print(response_json)
|
||||
|
||||
if response.ok: # synch call
|
||||
success = True
|
||||
trans_text = response_json["transcriptObj"]["text"]
|
||||
trans_score = float(response_json["transcriptObj"]["confidence"])
|
||||
else: # fallback to asynch call
|
||||
# wait up to 60 seconds for the transcription to be ready; try every 5 seconds
|
||||
wait = 0
|
||||
|
||||
while wait < 60:
|
||||
sleep(5)
|
||||
wait += 5
|
||||
|
||||
# issue GET request using the previously returned reqiestId and save response as response object
|
||||
response_get = requests.get(url = API_ENDPOINT + ':' + response_json["requestId"])
|
||||
|
||||
# check if the status code is not an error code (i.e., 4xx or 5xx)
|
||||
if response_get.ok:
|
||||
response_get_text = response_get.text
|
||||
response_get_json = json.loads(response_get_text)
|
||||
|
||||
success = True
|
||||
trans_text = response_get_json["transcriptObj"]["text"]
|
||||
trans_score = float(response_get_json["transcriptObj"]["confidence"])
|
||||
break
|
||||
|
||||
# in case no successful response is received even after a 60 seconds waiting period -> proceed without transcription
|
||||
#if not response_get.ok:
|
||||
# print("Response: FAILED.")
|
||||
|
||||
#else:
|
||||
# print("ERROR: {} ({})" . format(response.status_code, response.text))
|
||||
|
||||
if success:
|
||||
return success, trans_text, trans_score
|
||||
else:
|
||||
return False, None, None
|
||||
Reference in New Issue
Block a user