before changes

This commit is contained in:
2026-10-05 16:14:53 -04:00
parent dd6e6f7cfd
commit f39555ce15
94 changed files with 3285321 additions and 1 deletions

View File

@@ -0,0 +1,121 @@
{
"model": "speaker_encoder",
"run_name": "speaker_encoder",
"run_description": "resnet speaker encoder trained with commonvoice all languages dev and train, Voxceleb 1 dev and Voxceleb 2 dev",
"epochs": 100000,
"batch_size": null,
"eval_batch_size": null,
"mixed_precision": false,
"run_eval": true,
"test_delay_epochs": 0,
"print_eval": false,
"print_step": 50,
"tb_plot_step": 100,
"tb_model_param_stats": false,
"save_step": 1000,
"checkpoint": true,
"keep_all_best": false,
"keep_after": 10000,
"num_loader_workers": 8,
"num_val_loader_workers": 0,
"use_noise_augment": false,
"output_path": "../checkpoints/speaker_encoder/language_balanced/normalized/angleproto-4-samples-by-speakers/",
"distributed_backend": "nccl",
"distributed_url": "tcp://localhost:54321",
"audio": {
"fft_size": 512,
"win_length": 400,
"hop_length": 160,
"frame_shift_ms": null,
"frame_length_ms": null,
"stft_pad_mode": "reflect",
"sample_rate": 16000,
"resample": false,
"preemphasis": 0.97,
"ref_level_db": 20,
"do_sound_norm": false,
"do_trim_silence": false,
"trim_db": 60,
"power": 1.5,
"griffin_lim_iters": 60,
"num_mels": 64,
"mel_fmin": 0.0,
"mel_fmax": 8000.0,
"spec_gain": 20,
"signal_norm": false,
"min_level_db": -100,
"symmetric_norm": false,
"max_norm": 4.0,
"clip_norm": false,
"stats_path": null,
"do_rms_norm": true,
"db_level": -27.0
},
"datasets": [
{
"name": "voxceleb2",
"path": "/workspace/scratch/ecasanova/datasets/VoxCeleb/vox2_dev_aac/",
"meta_file_train": null,
"ununsed_speakers": null,
"meta_file_val": null,
"meta_file_attn_mask": "",
"language": "voxceleb"
}
],
"model_params": {
"model_name": "resnet",
"input_dim": 64,
"use_torch_spec": true,
"log_input": true,
"proj_dim": 512
},
"audio_augmentation": {
"p": 0.5,
"rir": {
"rir_path": "/workspace/store/ecasanova/ComParE/RIRS_NOISES/simulated_rirs/",
"conv_mode": "full"
},
"additive": {
"sounds_path": "/workspace/store/ecasanova/ComParE/musan/",
"speech": {
"min_snr_in_db": 13,
"max_snr_in_db": 20,
"min_num_noises": 1,
"max_num_noises": 1
},
"noise": {
"min_snr_in_db": 0,
"max_snr_in_db": 15,
"min_num_noises": 1,
"max_num_noises": 1
},
"music": {
"min_snr_in_db": 5,
"max_snr_in_db": 15,
"min_num_noises": 1,
"max_num_noises": 1
}
},
"gaussian": {
"p": 0.0,
"min_amplitude": 0.0,
"max_amplitude": 1e-05
}
},
"storage": {
"sample_from_storage_p": 0.5,
"storage_size": 40
},
"max_train_step": 1000000,
"loss": "angleproto",
"grad_clip": 3.0,
"lr": 0.0001,
"lr_decay": false,
"warmup_steps": 4000,
"wd": 1e-06,
"steps_plot_stats": 100,
"num_speakers_in_batch": 100,
"num_utters_per_speaker": 4,
"skip_speakers": true,
"voice_len": 2.0
}

View File

@@ -0,0 +1,226 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import sys
import os
import argparse
import torch
# load coqui-ai/trainer libraries
from trainer import Trainer, TrainerArgs
# load coqui-ai/TTS libraries
from TTS.tts.configs.shared_configs import BaseDatasetConfig
from TTS.tts.configs.vits_config import VitsConfig
from TTS.tts.datasets import load_tts_samples
from TTS.tts.models.vits import Vits, VitsArgs, VitsAudioConfig
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Code to clone a voice from a given set of voice samples and a multi-speaker baseline model")
parser.add_argument("--baseline_model_path", type = str, required = True,
help = "Path to multi-speaker baseline model (VITS model)")
parser.add_argument("--speaker_dataset_path", type = str, required = True,
help = "Path to voice cloning dataset")
parser.add_argument("--speaker_embeddings_path", type = str, required = True,
help = "Path to speaker's embeddings file")
parser.add_argument("--output_path", type = str, default = "results/cloned-voices",
help = "Path to store trained / generated assets")
parser.add_argument("--batch_size", type = int, default = 96, # 96 is suitable for AWS g5 instances
help = "Batch size for training run")
parser.add_argument("--max_epochs", type = int, default = 200, # 200 for batch_size 96 (with the 22.050 sampling rate multi-speaker model
help = "Maximum number of epochs for training run") # 2000 for batch_size 64 and 1500 for batch_size 96 (with the initial 16k sampling rate VCTK 0.80 model)
parser.add_argument("--use_cpu", default = False, action = "store_true", # untested!!!
help = "Signal that CPU should be used even if a CUDA-device is available")
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
return parser.parse_args()
#
# main training method (voice cloning)
#
def main(args):
if args.output_format == "txt":
print("Commencing training of a new multi-speaker potion-voice baseline model:")
print("")
print(" + Baseline multi-speaker model path: {}" . format(args.baseline_model_path))
print(" + Voice training dataset path : {}" . format(args.speaker_dataset_path))
print(" + Speaker embeddings path : {}" . format(args.speaker_embeddings_path))
print(" + Output path : {}" . format(args.output_path))
print(" + Batch size : {}" . format(args.batch_size))
print(" + Training runs (max epochs) : {}" . format(args.max_epochs))
print("")
# determine whether CUDA support is available and set device parameters accordingly
use_cuda = torch.cuda.is_available()
if args.output_format == "txt":
print(" + CUDA availability : {}" . format(use_cuda))
if args.use_cpu:
device = "cpu"
device_torch = False
elif use_cuda:
device = "cuda"
device_torch = torch.device("cuda")
else:
device = "cpu"
device_torch = False
if args.output_format == "txt":
print(" + Compute device used : {}" . format(device))
print("")
# define training data set
dataset_config = BaseDatasetConfig(formatter = "vctk_old", language = "en-us", path = args.speaker_dataset_path)
# set VITS training parameters
audio_config = VitsAudioConfig(
sample_rate = 22050,
win_length = 1024,
hop_length = 256,
num_mels = 80,
mel_fmin = 0,
mel_fmax = None,
)
vitsArgs = VitsArgs(
use_speaker_embedding = False,
use_d_vector_file = True,
d_vector_file = [args.speaker_embeddings_path],
d_vector_dim = 512,
num_layers_text_encoder = 10
)
config = VitsConfig(
model_args = vitsArgs,
audio = audio_config,
run_name = "vits_potion_clone",
use_speaker_embedding = False,
use_d_vector_file = True,
d_vector_file = [args.speaker_embeddings_path],
d_vector_dim = 512,
batch_size = args.batch_size,
eval_batch_size = 8,
batch_group_size = 0, # changing this to 5 (VITS training default) slows training down, but doesn't have any positive training effects
num_loader_workers = 4,
num_eval_loader_workers = 4,
run_eval = True,
eval_split_size = 2, # fix size of eval dataset (default 1% approach requires at least 100 voice samples!)
test_delay_epochs = -1,
epochs = args.max_epochs,
text_cleaner = "english_cleaners",
use_phonemes = False,
phoneme_language = "en-us",
phoneme_cache_path = os.path.join(args.output_path, "phoneme_cache"),
compute_input_seq_cache = True,
print_step = 50,
print_eval = True,
mixed_precision = True,
max_text_len = 325,
output_path = args.output_path,
save_checkpoints = True,
save_step = 200,
datasets = [dataset_config],
cudnn_benchmark = False,
#characters = {
# "pad": "_",
# "eos": "&",
# "bos": "*",
# "characters": "!¡'(),-.:;¿?abcdefghijklmnopqrstuvwxyz «°±µ»$%&‘’‚“`”„",
# "punctuations": "!¡'(),-.:;¿? ",
# "phonemes": None,
# "unique": True
#},
test_sentences = [
["It took me quite a long time to develop a voice, and now that I have it I'm not going to be silent."],
["Be a voice, not an echo."],
["I'm sorry Dave. I'm afraid I can't do that."],
["This cake is great. It's so delicious and moist."],
["Prior to November 22, 1963."],
["Hey! Sandra."],
["Hey! Andrew."],
["Hey, Michelle."],
["Hey! George."],
["Hey there, Rachel."]
]
)
# load training samples
train_samples, eval_samples = load_tts_samples(config.datasets, eval_split = True, eval_split_max_size = config.eval_split_max_size, eval_split_size = config.eval_split_size)
# init VITS model
model = Vits.init_from_config(config)
# init voice cloning
trainer = Trainer(
TrainerArgs(restore_path = args.baseline_model_path, use_ddp = False),
config,
args.output_path,
model = model,
train_samples = train_samples,
eval_samples = eval_samples
)
# trigger voice cloning (aka single speaker training)
try:
trainer.fit()
except (KeyboardInterrupt, SystemExit):
print("Training stopped manually (via keyboard interrupt)! Bye.")
exit(0)
# determine required adjustment for speech synthesizing (i.e., the scaling factor for the duration predictor)
# take the duration of the test sentence and calculate the difference to corresponding reference samples
# set config.model_args["length_scale"] accordingly and save the updated config asset
# exit gracefully
if args.output_format == "txt":
print("")
print("Completed voice cloning. The resulting model(s) can be found at:")
print(" --> {}" . format(args.output_path))
print("")
print("Done; bye.")
print("")
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
# clear command line arguments to avoid triggering argparse features part of Trainer / coqpit imports
# Traceback (most recent call last):
# File "train_multispeaker_baseline_model.py", line 208, in <module>
# main(args)
# File "train_multispeaker_baseline_model.py", line 177, in main
# trainer = Trainer(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 360, in __init__
# config, new_fields = self.init_training(args, coqpit_overrides, config)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 594, in init_training
# config.parse_known_args(coqpit_overrides, relaxed_parser=True)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 843, in parse_known_args
# parser = self.init_argparse(arg_prefix=arg_prefix, relaxed_parser=relaxed_parser)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 881, in init_argparse
# _init_argparse(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 529, in _init_argparse
# parser = _init_argparse(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 550, in _init_argparse
# return default.init_argparse(
# AttributeError: 'str' object has no attribute 'init_argparse'
sys.argv = [sys.argv[0]]
# ensure the output path exists
os.makedirs(args.output_path, exist_ok = True)
main(args)
### USAGE:
### $ python3 TTS/TTS/bin/resample.py --input_dir voice_dataset_path/person_82/wav48/1 --output_sr 16000
### $ python3 clone_voice.py [with argument]

View File

@@ -0,0 +1,764 @@
# potion-voice **voice-cloning** *Installation and Usage Guide*
In this guide, you will find more detailed instructions and examples for the following tasks:
+ Setting up a new AWS GPU-backed EC2 instance suitable for training new potion-voice models;
+ Setting up software environment and (optionally) prepare data sets for training new potion-voice models;
+ Training and evaluating new potion-voice models; and
+ Usage examples for voice cloning and speech synthesizing.
## Set Up AWS GPU-backed Compute Node (non-production)
1. Set up baseline & connect to remote node:
+ GPU-enabled Compute Node (e.g., g5.2xlarge by default)
+ We recommend a GPU-enabled Compute Node with 256GB root partition (volume type: gp3; 64GB for swapfile) and 512GB secondary SDD holding all dev / data files)
+ Inbound ports: SSH and TensorBoard (e.g., port 6006)
+ Ubuntu 22.04 LTS (Server) Installation
+ SSH into the EC2 instance
1. Secure / update baseline
```sh
$ sudo apt-get update
$ sudo apt-get upgrade
$ sudo apt-get install linux-aws linux-headers-aws linux-image-aws
```
1. Disable unattended upgrades. Enter the below command and select 'No'. These Upgrades might cause version mismatch between nvidia-drivers and cuda.
```sh
$ sudo dpkg-reconfigure -plow unattended-upgrades
Replacing config file /etc/apt/apt.conf.d/20auto-upgrades with new version
```
1. Set up secondary disk (used as dev / data volume)
```sh
$ sudo lsblk
NAME MAJ:MIN RM SIZE RO TYPE MOUNTPOINT
[...]
nvme1n1 259:0 0 500G 0 disk
[...]
$ sudo mkfs -t ext4 /dev/nvme1n1
mke2fs 1.45.5 (07-Jan-2020)
Creating filesystem with 524288000 4k blocks and 131072000 inodes
Filesystem UUID: 90327770-ba4d-4003-9136-964b4388ffb6
Superblock backups stored on blocks:
32768, 98304, 163840, 229376, 294912, 819200, 884736, 1605632, 2654208,
4096000, 7962624, 11239424, 20480000, 23887872, 71663616, 78675968,
102400000, 214990848, 512000000
Allocating group tables: done
Writing inode tables: done
Creating journal (262144 blocks): done
Writing superblocks and filesystem accounting information: done
$ mkdir DEV_PATH
```
+ Edit `/etc/fstab` and add
```txt
/dev/nvme1n1 DEV_PATH ext4 defaults,nofail 0 2
```
```sh
$ sudo mount -a
$ sudo chown -R ubuntu:ubuntu DEV_PATH
$ mkdir DEV_PATH/data
```
1. Create a swap file (training is memory intensive; so, add a swap file!)
+ Use the `dd` command to create a swap file on the root file system
+ Note: The size of the swap file is the block size option multiplied by the count option in the dd command. Adjust these values to determine the desired swap file size.
+ Note: The block size you specify should be less than the available memory on the instance or you receive a "memory exhausted" error.
+ Set up the swap file (of size 64 GB [512 MB x 128]).
```sh
$ sudo dd if=/dev/zero of=/swapfile bs=512M count=128
128+0 records in
128+0 records out
68719476736 bytes (69 GB, 64 GiB) copied, 336.416 s, 204 MB/s
```
+ Update the read and write permissions for the swap file:
```sh
$ sudo chmod 600 /swapfile
```
+ Set up a Linux swap area:
```sh
$ sudo mkswap /swapfile
Setting up swapspace version 1, size = 64 GiB
no label, UUID=1dfc20ce-ed64-4e69-8fa7-a800bbea4617
```
+ Make the swap file available for immediate use by adding the swap file to swap space:
```sh
$ sudo swapon /swapfile
```
+ Verify that the procedure was successful:
```sh
$ sudo swapon -s
Filename Type Size Used Priority
/swapfile file 67108860 0 -2
```
+ Enable the swap file at boot time by editing the `/etc/fstab` file. Add the following new line at the end of the file:
```txt
/swapfile swap swap defaults 0 0
```
1. Install NVIDIA drivers / CUDA support (pytorch required version 11.6 or 12)
```sh
wget https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.0-1_all.deb
sudo dpkg -i cuda-keyring_1.0-1_all.deb
sudo apt-get update
sudo apt-get -y install cuda-12-0
```
+ Reboot the instance and ensure all drivers load automatically
```sh
$ sudo reboot
```
+ Reconnect to the instance and verify NVIDIA drivers / CUDA support are as expected
```sh
$ nvidia-smi
Tue Jan 17 08:20:53 2023
+-----------------------------------------------------------------------------+
| NVIDIA-SMI 525.60.13 Driver Version: 525.60.13 CUDA Version: 12.0 |
|-------------------------------+----------------------+----------------------+
| GPU Name Persistence-M| Bus-Id Disp.A | Volatile Uncorr. ECC |
| Fan Temp Perf Pwr:Usage/Cap| Memory-Usage | GPU-Util Compute M. |
| | | MIG M. |
|===============================+======================+======================|
| 0 NVIDIA A10G On | 00000000:00:1E.0 Off | 0 |
| 0% 19C P8 16W / 300W | 0MiB / 23028MiB | 0% Default |
| | | N/A |
+-------------------------------+----------------------+----------------------+
+-----------------------------------------------------------------------------+
| Processes: |
| GPU GI CI PID Type Process name GPU Memory |
| ID ID Usage |
|=============================================================================|
| No running processes found |
+-----------------------------------------------------------------------------+
```
## Set Up Software Environment
1. Set up Python 3 (v3.10) development environment
```sh
$ sudo apt-get install python3-dev python3-pip python3-wheel python3-venv
```
1. Set up Phoneme back-end
```sh
$ sudo apt-get install espeak-ng espeak-ng-espeak
```
1. Set up required tools / standard dependencies
```sh
$ sudo apt-get install ffmpeg unzip git
```
1. Set up AWS Command Line Interface
```sh
$ sudo apt-get install awscli
$ aws configure
AWS Access Key ID [None]: xxxxxxxxxx
AWS Secret Access Key [None]: yyyyyyyyyy
Default region name [None]: us-west-2
Default output format [None]: json
$ aws configure set default.s3.max_concurrent_requests 50
```
1. (dev install only) Copy and extract training data sets from AWS
```sh
$ cd DEV_PATH/data
### VCTK v 0.92
$ aws s3 cp s3://potion-datasets/VCTK/VCTK-Corpus-0.92/VCTK-Corpus-0.92.tgz .
download: s3://potion-datasets/VCTK/VCTK-Corpus-0.92/VCTK-Corpus-0.92.tgz to ./VCTK-Corpus-0.92.tgz
$ tar -xzvf VCTK-Corpus-0.92.tgz
VCTK-Corpus-0.92/
VCTK-Corpus-0.92/README.txt
VCTK-Corpus-0.92/update.txt
VCTK-Corpus-0.92/license_text
VCTK-Corpus-0.92/txt/
[...]
VCTK-Corpus-0.92/wav48_silence_trimmed/p238/p238_191_mic1.flac
VCTK-Corpus-0.92/wav48_silence_trimmed/p238/p238_267_mic2.flac
$ rm VCTK-Corpus-0.92.tgz
### LibriTTS train-clean-360 subset
$ aws s3 cp s3://potion-datasets/LibriTTS/train-clean-360.tar.gz .
download: s3://potion-datasets/LibriTTS/train-clean-360.tar.gz to ./train-clean-360.tar.gz
$ tar -xzvf train-clean-360.tar.gz
./LibriTTS/train-clean-360/
./LibriTTS/train-clean-360/2272/
./LibriTTS/train-clean-360/2272/152265/
./LibriTTS/train-clean-360/2272/152265/2272_152265_000032_000001.original.txt
./LibriTTS/train-clean-360/2272/152265/2272_152265_000012_000001.wav
[...]
LibriTTS/reader_book.tsv
LibriTTS/speakers.tsv
$ rm train-clean-360.tar.gz
### Potion salutation recordings
$ aws s3 cp s3://potion-datasets/potion-voice-datasets/potion-salut-corpus_20221026.tgz .
download: s3://potion-datasets/potion-voice-datasets/potion-salut-corpus_20221026.tgz to ./potion-salut-corpus_20221019.tgz
$ tar -xzvf potion-salut-corpus_20221026.tgz
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/wav48/
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/wav48/POTION_6192d9c9a563df5c87ecb8bd/
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/wav48/POTION_6192d9c9a563df5c87ecb8bd/POTION_6192d9c9a563df5c87ecb8bd_334.wav
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/wav48/POTION_6192d9c9a563df5c87ecb8bd/POTION_6192d9c9a563df5c87ecb8bd_473.wav
[...]
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/txt/POTION_62d82d269cbde00027b66007/POTION_62d82d269cbde00027b66007_197.txt
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/speaker-info.txt
$ rm potion-salut-corpus_20221026.tgz
```
1. Create a virtual potion-voice-cloner working environment
```sh
$ cd DEV_PATH
$ python3 -m venv potion-voice_venv
$ cd potion-voice_venv/
$ source bin/activate
(potion-voice_venv) $
```
1. Clone the potion-voice GitHub repository
```sh
(potion-voice_venv) $ cd DEV_PATH/potion-voice_venv/
(potion-voice_venv) $ python3 -m pip install --upgrade pip
(potion-voice_venv) $ git clone https://github.com/potion/potion-voice.git
```
1. Install potion-voice requirements (dependencies) and test that PyTorch is working with the GPU properly
```sh
(potion-voice_venv) $ cd DEV_PATH/potion-voice_venv/potion-voice/
(potion-voice_venv) $ python3 -m pip install -r ./requirements.dev.txt
(potion-voice_venv) $ python3
Python 3.10.6 (main, Nov 14 2022, 16:10:14) [GCC 11.3.0] on linux
Type "help", "copyright", "credits" or "license" for more information.
>>> import torch
>>> torch.cuda.is_available()
True
>>> torch.cuda.get_device_name(0)
'NVIDIA A10G'
>>> quit()
```
1. Install TTS dependencies
```sh
(potion-voice_venv) $ cd voice-cloning/
(potion-voice_venv) $ git clone --depth 1 --branch v0.10.2 https://github.com/coqui-ai/TTS
(potion-voice_venv) $ python3 -m pip install -e TTS/
```
+ Note 1: Installing requirements will ask for GitHub token twice! The second request is for a dependent package, which is also a private repo.
+ Note 2: Separate requirements files have been added for development (local versus AWS) and production usage (for GPU and CPU-only deployment).
## Training New potion-voice Models (Multi-speaker Baseline & Voice Cloning)
### Preprocess Dataset(s) Required for Multi-speaker Baseline Model Training
1. For each dataset, ensure that the sampling rate matches and speaker embeddings are precomputed.
```sh
(potion-voice_venv) $ python3 prepare_datasets.py --dataset_preset vctk --dataset_archive_path ~/datasets/VCTK_v0.92/VCTK-Corpus-0.92.tgz --sampling_rate 22050
Commencing preparation of dataset for multi-speaker baseline model training:
+ Dataset preset: vctk
+ Dataset : /home/[REDACTED_HOMEDIR_USERNAME_3]/datasets/VCTK_v0.92/VCTK-Corpus-0.92.tgz
+ Output path : results/datasets
+ Sampling rate : 22050
>>> Extracting archive ...
>>> Resampling audio files to 16000Hz ...
Resampling the audio files...
Found 88328 files...
100%|████████████████████████████████████████████████████████████████████████████████| 88328/88328 [18:25<00:00, 79.88it/s]
Done !
>>> Computing speaker embeddings ...
> Found 44283 files in /home/[REDACTED_HOMEDIR_USERNAME_3]/work/potion-repos/potion-voice_venv/potion-voice/voice-cloning/results/datasets/VCTK-Corpus-0.92
> Model fully restored.
> Setting up Audio Processor...
[...]
100%|████████████████████████████████████████████████████████████████████████████████| 44283/44283 [06:18<00:00, 116.99it/s]
Speaker embeddings saved at: results/datasets/VCTK-Corpus-0.92/speakers.pth
>>> Extracting original archive again (overwritting previously resampled files)...
>>> Resampling audio files to 22050Hz ...
Resampling the audio files...
Found 88328 files...
100%|████████████████████████████████████████████████████████████████████████████████| 88328/88328 [20:48<00:00, 70.74it/s]
Done !
Completed preparing voice dataset for multi-speaker baseline model training; generated asset locations are as follows:
--> results/datasets/VCTK-Corpus-0.92
--> results/datasets/VCTK-Corpus-0.92/speakers.pth
Done; bye.
```
### Train New potion-voice Multi-speaker Baseline Model
1. To train a new baseline model:
```sh
(potion-voice_venv) $ python3 train_multispeaker_baseline_model.py
usage: train_multispeaker_baseline_model.py [-h] --datasets {VCTK,LibriTTS_tc360,POTION_Salut} [{VCTK,LibriTTS_tc360,POTION_Salut} ...] [--output_path OUTPUT_PATH] [--batch_size BATCH_SIZE] [--max_epochs MAX_EPOCHS]
Code to train multi-speaker baseline model
options:
-h, --help show this help message and exit
--datasets {VCTK,LibriTTS_tc360,POTION_Salut} [{VCTK,LibriTTS_tc360,POTION_Salut} ...]
List of training datasets to be included in training run.
--output_path OUTPUT_PATH
Path to store trained / generated assets
--batch_size BATCH_SIZE
Batch size for training run
--max_epochs MAX_EPOCHS
Maximum number of epochs for training run
```
Using the default settings, training a new multi-speaker baseline model (on an AWS g5.2xlarge instance) takes 5-7 days (100 epochs with 32 batch size and all 3 datasets (i.e., VCTK, LibriTTS_tc360, andpotion_Salut)).
1. At the end of a training run, there will be the following files in the result folder:
```txt
results/baseline-models/vits_vctk-March-23-2022_03+43AM-0000000/
|-- best_model.pth .................................... best model using avg_loss_0 (NOT the best model; suggest to ignore for now)
|-- best_model_19096.pth .............................. same as best_model.pth (suggest to ignore for now)
|-- checkpoint_300000.pth ............................. fifth last checkpoint
|-- checkpoint_310000.pth ............................. fourth last checkpoint
|-- checkpoint_320000.pth ............................. third last checkpoint
|-- checkpoint_330000.pth ............................. second last checkpoint
|-- checkpoint_340000.pth ............................. last checkpoint
|-- config.json ....................................... configuration file
|-- events.out.tfevents.1648007016.ip-172-31-83-225 ... event log for entire training run including eval samples and charts (view via tensorboard)
|-- speakers.pth ...................................... speaker embeddings
|-- trainer_0_log.txt ................................. training log
|-- train_multispeaker_baseline_model.py .............. copy of the training script
```
Use the event log to determine which of the checkpoints corresponds to the best model.
### Clone a Voice based on the Mutli-speaker Baseline Model
1. To clone a new voice, you need at least 10 voice samples (ideally 30). Those voice recordings (and their corresponding transcription files) have to be arranged as follows (and compressed into a `.tgz`, `.tbz` or `.zip` archive):
```txt
VOICE_DATASET_PATH/txt/1/1_001.txt
VOICE_DATASET_PATH/txt/1/1_002.txt
VOICE_DATASET_PATH/txt/1/1_003.txt
...
VOICE_DATASET_PATH/txt/1/1_029.txt
VOICE_DATASET_PATH/txt/1/1_030.txt
VOICE_DATASET_PATH/wav48/1/1_001.wav
VOICE_DATASET_PATH/wav48/1/1_002.wav
VOICE_DATASET_PATH/wav48/1/1_003.wav
...
VOICE_DATASET_PATH/wav48/1/1_028.wav
VOICE_DATASET_PATH/wav48/1/1_029.wav
VOICE_DATASET_PATH/wav48/1/1_030.wav
```
1. Next, pre-process audio recordings to fit the format of audio samples (i.e., sampling rate) and pre-compute speaker embeddings:
```sh
(potion-voice_venv) $ python prepare_datasets.py --dataset_preset potion_voice_cloning --dataset_archive_path ~/datasets/potion\ Recordings/potion-voice\ recordings/user123.tgz
Commencing preparation of dataset for multi-speaker baseline model training:
+ Dataset preset: potion_voice_cloning
+ Dataset : /home/[REDACTED_HOMEDIR_USERNAME_3]/datasets/potion Recordings/potion-voice recordings/user123.tgz
+ Output path : results/datasets
+ Sampling rate : 22050
>>> Extracting archive ...
>>> Resampling audio files to 16000Hz ...
Resampling the audio files...
Found 30 files...
100%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 30/30 [00:00<00:00, 39.40it/s]
Done !
>>> Extracting original archive again (overwritting previously resampled files)...
>>> Resampling audio files to 22050Hz ...
Resampling the audio files...
Found 30 files...
100%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 30/30 [00:00<00:00, 37.61it/s]
Done !
Completed preparing voice dataset for multi-speaker baseline model training; generated asset locations are as follows:
--> results/datasets/sr22050/user123
--> results/datasets/sr22050/user123/speakers.pth
Done; bye.
```
1. Finally, trigger voice cloning:
```sh
(potion-voice_venv) $ python3 clone_voice.py [-h] --baseline_model_path BASELINE_MODEL_PATH --speaker_dataset_path SPEAKER_DATASET_PATH --speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH [--output_path OUTPUT_PATH] [--batch_size BATCH_SIZE] [--max_epochs MAX_EPOCHS] [--use_cpu] [--output_format {txt,json}]
Code to clone a voice from a given set of voice samples and a multi-speaker baseline model
options:
-h, --help show this help message and exit
--baseline_model_path BASELINE_MODEL_PATH
Path to multi-speaker baseline model (VITS model)
--speaker_dataset_path SPEAKER_DATASET_PATH
Path to voice cloning dataset
--speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH
Path to speaker's embeddings file
--output_path OUTPUT_PATH
Path to store trained / generated assets
--batch_size BATCH_SIZE
Batch size for training run
--max_epochs MAX_EPOCHS
Maximum number of epochs for training run
--use_cpu Signal that CPU should be used even if a CUDA-device is available
--output_format {txt,json}
Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting
```
Using the default settings and 30 audio samples, cloning a new voice (on an AWS g5.2xlarge instance) takes about one hour.
1. At the end of a voice cloning run, there will be the following files in the result folder:
```txt
results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/
|-- best_model_365097.pth .................. best model using avg_loss_0 (save to use)
|-- best_model.pth ......................... same as best_model_365097.pth
|-- checkpoint_365200.pth .................. last checkpoint
|-- clone_voice.py ......................... copy of the clone_voice script used in this run
|-- config.json ............................ configuration file
|-- events.out.tfevents.1672195961.rigel ... event log for entire voice cloning run including eval samples and charts (view via tensorboard)
|-- speakers.pth ........................... speaker's embeddings file
|-- trainer_0_log.txt ...................... training log file
```
Use the event log to confirm that the best model is indeed giving the best outputs.
### Monitoring Training Progress
Using tensorboard / tensorboardX, training progress (for both, multi-speaker baseline training and voice cloning) can be monitored and evaluation samples can be accessed.
1. Ensure AWS Security Group settings (inbound) are set appropriamust include:
```txt
HTTPS TCP 443 0.0.0.0/0
Custom_TCP TCP 6006 0.0.0.0/0
```
+ Server-side, launch the tensorboard service:
```sh
(potion-voice_venv) $ tensorboard --logdir=./results/baseline-models/vits_vctk-March-07-2022_09+47AM-0000000/ --host 0.0.0.0
TensorFlow installation not found - running with reduced feature set.
NOTE: Using experimental fast data loading logic. To disable, pass
"--load_fast=false" and report issues on GitHub. More details:
https://github.com/tensorflow/tensorboard/issues/4784
TensorBoard 2.8.0 at http://0.0.0.0:6006/ (Press CTRL+C to quit)
```
+ Locally, point your preferred Web browser to <http://PUBLIC_IPv4_DNS:6006/>
### Minimise a Cloned Voice
To minimise the size of a trained model, run the followng script which removes optimiser and discriminator components from the model -- those are only required for training but not for inference:
```sh
(potion-voice_venv) $ python3 minimize_cloned_voice_model.py [-h] --voice_model_asset_path VOICE_MODEL_ASSET_PATH [--voice_model_name VOICE_MODEL_NAME] [--voice_model_config_name VOICE_MODEL_CONFIG_NAME] [--minimise_suffix MINIMISE_SUFFIX] [--overwrite_assets] [--output_format {txt,json}]
Code to minimise (i.e., remove optimiser & discriminator) a cloned voice model
options:
-h, --help show this help message and exit
--voice_model_asset_path VOICE_MODEL_ASSET_PATH
Path to directory storing cloned voice model and the corresponding configuration and speaker files
--voice_model_name VOICE_MODEL_NAME
Name of the (best) cloned voice model
--voice_model_config_name VOICE_MODEL_CONFIG_NAME
Name of the config file for the cloned voice model
--minimise_suffix MINIMISE_SUFFIX
Suffix to be used for minimised model and its assets (i.e., new config file)
--overwrite_assets Signal whether existing model assets should be overwritten or not (default: do not overwrite)
--output_format {txt,json}
Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting
```
1. Command-line output sample for output format option "txt":
```sh
$ python3 minimize_cloned_voice_model.py --voice_model_asset_path results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/
Minimising given voice model:
+ Cloned voice model file path : results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/best_model.pth
+ Cloned voice model config file : results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/config.json
> Using model: vits
> Setting up Audio Processor...
[...]
Completed minimising cloned voice model. The resulting (modified) assets can be found at:
--> Minimised voice model path : results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/best_model_light.pth
--> Minimised voice model config path: results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/config_light.json
Done; bye.
```
### Scoring a Cloned Voice
1. To score a cloned voice, run the following command:
```sh
(potion-voice_venv) $ python3 score_cloned_voice.py [-h] --voice_dataset_path VOICE_DATASET_PATH --voice_model_path VOICE_MODEL_PATH --voice_model_config_path VOICE_MODEL_CONFIG_PATH --speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH [--temp_path TEMP_PATH] [--keep_temp] [--use_cpu] [--output_format {txt,json}]
Compute quality score for a given voice model (cloned voice) wrt. a given set of voice recordings (original voice))
options:
-h, --help show this help message and exit
--voice_dataset_path VOICE_DATASET_PATH
Path to set of voice recordings (original voice)
--voice_model_path VOICE_MODEL_PATH
Path to cloned voice model
--voice_model_config_path VOICE_MODEL_CONFIG_PATH
Path to config file for the cloned voice model
--speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH
Path to speaker's embeddings file (i.e., pre-computed embeddings typically stored with the speaker's dataset)
--temp_path TEMP_PATH
Path to store temporary speech assets
--keep_temp Signal that temporary assets used for scoring should not be deleted once done
--use_cpu Signal that CPU should be used even if a CUDA-device is available
--output_format {txt,json}
Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting
```
1. Command-line output sample for output format option "txt":
```sh
$ python3 score_cloned_voice.py --voice_dataset_path results/datasets/sr22050/michael/wav48/1/ --voice_model_path results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/best_model.pth --voice_model_config_path results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/config.json --speaker_embeddings_path results/datasets/sr22050/michael/speakers.pth
Computing similarity score for a given voice model (cloned voice) wrt. a given set of voice recordings (original voice):
+ Original voice recordings path: results/datasets/sr22050/michael/wav48/1/
+ Cloned voice model file path : results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/best_model.pth
+ Cloned voice model config file: results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/config.json
+ Speaker embeddings file : results/datasets/sr22050/michael/speakers.pth
+ CUDA availability : True
+ Compute device used : cuda
+ No. of speakers : 1
+ Speaker's names : ['VCTK_old_1']
+ No. of embeddings : 30
> Using model: vits
> Setting up Audio Processor...
Loaded the voice encoder model on cuda in 0.01 seconds.
Completed computing similarity score for the two sets of recordings. The resulting similarity score is:
--> 0.9127510190010071
Done; bye.
```
1. Command-line output sample for output format option "json":
```sh
(potion-voice_venv)$ python3 score_cloned_voice.py --voice_dataset_path results/datasets/sr22050/michael/wav48/1/ --voice_model_path results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/best_model.pth --voice_model_config_path results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/config.json --speaker_embeddings_path results/datasets/sr22050/michael/speakers.pth --output_format json
> Using model: vits
> Setting up Audio Processor...
[...]
Loaded the voice encoder model on cuda in 0.01 seconds.
{"success": true, "in": {"voice_dataset_path": "results/datasets/sr22050/michael/wav48/1/", "voice_model_path": "results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/best_model.pth"}, "out": {"score": 0.91}}
```
## Usage Examples for Speech Synthesizing
1. To generate speech for a given cloned voice, run the following command:
```sh
(potion-voice_venv) $ python3 synthesize_speech.py [-h] --voice_model_path VOICE_MODEL_PATH --voice_model_config_path VOICE_MODEL_CONFIG_PATH --speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH --txt TXT [--output_path OUTPUT_PATH] [--target_sampling_rate TARGET_SAMPLING_RATE] [--speech_sample_wav_path SPEECH_SAMPLE_WAV_PATH] [--speech_sample_txt SPEECH_SAMPLE_TXT] [--trim_silence] [--use_cpu] [--output_format {txt,json}]
Code to synthesize speech for a given voice model
options:
-h, --help show this help message and exit
--voice_model_path VOICE_MODEL_PATH
Path to cloned voice model
--voice_model_config_path VOICE_MODEL_CONFIG_PATH
Path to config file for the cloned voice model
--speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH
Path to speaker's embeddings file (i.e., pre-computed embeddings typically stored with the speaker's dataset)
--txt TXT Text to synthesize
--output_path OUTPUT_PATH
Path to store generated speech assets
--target_sampling_rate TARGET_SAMPLING_RATE
Desired sampling rate (in Hz) for output file
--speech_sample_wav_path SPEECH_SAMPLE_WAV_PATH
Path to a sample utterance of the speaker (used for style transfer)
--speech_sample_txt SPEECH_SAMPLE_TXT
Text of the sample utterance of the speaker (used for style transfer)
--trim_silence Signal whether to trim silence from synthesised speech
--use_cpu Signal that CPU should be used even if a CUDA-device is available
--output_format {txt,json}
Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting
```
1. Command-line output sample for output format option "txt":
```sh
(potion-voice_venv) $ python3 synthesize_speech.py --voice_model_path results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/best_model.pth --voice_model_config_path results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/config.json --speaker_embeddings_path results/datasets/sr22050/[REDACTED_HOMEDIR_USERNAME_2]/speakers.pth --txt "Hi person_82, it works!"
Commencing speech synthesizing:
+ Voice model file path : results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/best_model.pth
+ Voice model config file: results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/config.json
+ Speaker embeddings file: results/datasets/sr22050/[REDACTED_HOMEDIR_USERNAME_2]/speakers.pth
+ Output path : results/speech
+ Text to synthesize : Hi person_82, it works!
+ CUDA availability : True
+ Compute device used : cuda
+ No. of speakers : 1
+ Speaker's names : ['VCTK_old_1']
+ No. of embeddings : 30
> Using model: vits
> Setting up Audio Processor...
[...]
>>> Saving original output to : results/speech/b4189e9e-6142-4dad-8577-6de77087ffd1.wav
>>> Saving resampled output to: results/speech/b4189e9e-6142-4dad-8577-6de77087ffd1_sr48000.wav
Speech synthesizing has completed. Bye.
```
1. Command-line output sample for output format option "txt":
```sh
(potion-voice_venv) $ python3 synthesize_speech.py --voice_model_path results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/best_model_light.pth --voice_model_config_path results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/config_light.json --speaker_embeddings_path results/datasets/sr22050/[REDACTED_HOMEDIR_USERNAME_2]/speakers.pth --txt "Hi person_82, it works!" --output_format json
> Using model: vits
> Setting up Audio Processor...
[...]
{"success": true, "in": {"voice_model_path": "results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/best_model_light.pth", "voice_model_config_path": "results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/config_light.json", "speaker_embeddings_path": "results/datasets/sr22050/[REDACTED_HOMEDIR_USERNAME_2]/speakers.pth"}, "out": {"speech_original_path": "results/speech/c99e494c-e1f9-4c12-9095-255cf7db792b.wav", "speech_resampled_path": "results/speech/c99e494c-e1f9-4c12-9095-255cf7db792b_sr48000.wav"}}
```
### Scoring a Synthesised Salutation
1. To score a synthesised salutation, run the following command:
```sh
(potion-voice_venv)$ python3 score_salutation.py [-h] --recording_path RECORDING_PATH --first_name FIRST_NAME [--output_format {txt,json}]
Score a given salutation recording wrt. its desired content, the actual salutation recording, and a generated transcription (using Potion's internal Transciption API) of the recording.
optional arguments:
-h, --help show this help message and exit
--recording_path RECORDING_PATH
Path to salutation recoding (.wav audio file)
--first_name FIRST_NAME
First name that the salutation recoding is meant to use
--output_format {txt,json}
Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting
```
1. Command-line output sample for output format option "txt":
```sh
(potion-voice_venv)$ python3 score_salutation.py --recording_path /home/[REDACTED_HOMEDIR_USERNAME_3]/person_82_-_Hey_person_83.wav --first_name person_83
Commencing scoring of the given salutation recording:
+ Salutation recording path: /home/[REDACTED_HOMEDIR_USERNAME_3]/person_82_-_Hey_person_83.wav
+ Salutation first name : person_83
>> Salutation score : 0.892155
Done; bye.
```
1. Command-line output sample for output format option "json":
```sh
(potion-voice_venv)$ python3 score_salutation.py --recording_path /home/[REDACTED_HOMEDIR_USERNAME_3]/person_82_-_Hey_person_83.wav --first_name person_83 --output_format json
{"in": {"recording_path": "/home/[REDACTED_HOMEDIR_USERNAME_3]/person_82_-_Hey_person_83.wav", "first_name": "person_83"}, "out": {"score": 0.89}}
```
## Troubleshooting
1. How to better monitor GPU load / utilisation?
+ Install an interactive NVIDIA-GPU process viewer such as `nvitop`:
```sh
$ python3 -m pip install nvitop
Collecting nvitop
[...]
Installing collected packages: nvidia-ml-py, termcolor, psutil, nvitop
Successfully installed nvidia-ml-py-11.495.46 nvitop-0.8.0 psutil-5.9.2 termcolor-2.0.1
````
+ Run via command-line: `nvitop`:
```sh
Tue Sep 13 02:09:07 2022
╒═════════════════════════════════════════════════════════════════════════════╕
│ NVITOP 0.8.0 Driver Version: 515.65.01 CUDA Driver Version: 11.7 │
├───────────────────────────────┬──────────────────────┬──────────────────────┤
│ GPU Name Persistence-M│ Bus-Id Disp.A │ Volatile Uncorr. ECC │
│ Fan Temp Perf Pwr:Usage/Cap│ Memory-Usage │ GPU-Util Compute M. │
╞═══════════════════════════════╪══════════════════════╪══════════════════════╪══════════════════════════╕
│ 0 A10G On │ 00000000:00:1E.0 Off │ 0 │ MEM: █████████▊ 65.1% │
│ 0% 47C P0 192W / 300W │ 14982MiB / 22.49GiB │ 100% Default │ UTL: ███████████████ MAX │
╘═══════════════════════════════╧══════════════════════╧══════════════════════╧══════════════════════════╛
[ CPU: ██████████▏ 18.1% ] ( Load Average: 1.07 1.11 1.04 )
[ MEM: ███████████▎ 20.2% ] [ SWP: ▏ 0.3% ]
╒════════════════════════════════════════════════════════════════════════════════════════════════════════╕
│ Processes: ubuntu@ip-172-31-95-84 │
│ GPU PID USER GPU-MEM %SM %CPU %MEM TIME COMMAND │
╞════════════════════════════════════════════════════════════════════════════════════════════════════════╡
│ 0 2100 C ubuntu 14463MiB 90 103.7 9.6 5.4 days python3 train_multispeaker_baseline_model.py │
╘════════════════════════════════════════════════════════════════════════════════════════════════════════╛
```

View File

@@ -0,0 +1,120 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import sys
import os
import argparse
from pathlib import Path
import json
import torch
from TTS.config import load_config
from TTS.tts.models import setup_model as setup_tts_model
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Code to minimise (i.e., remove optimiser & discriminator) a cloned voice model")
parser.add_argument("--voice_model_asset_path", type = str, required = True,
help = "Path to directory storing cloned voice model and the corresponding configuration and speaker files")
parser.add_argument("--voice_model_name", type = str, default = "best_model.pth",
help = "Name of the (best) cloned voice model")
parser.add_argument("--voice_model_config_name", type = str, default = "config.json",
help = "Name of the config file for the cloned voice model")
parser.add_argument("--minimise_suffix", type = str, default = "light",
help = "Suffix to be used for minimised model and its assets (i.e., new config file)")
parser.add_argument("--overwrite_assets", default = False, action = "store_true",
help = "Signal whether existing model assets should be overwritten or not (default: do not overwrite)")
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
return parser.parse_args()
#
# utility function to expand the name of a given filename (infront of the extension)
#
def append_suffix_to_filename(fname, fname_suffix):
fpath = Path(fname)
return "{0}_{2}{1}" . format(fpath.stem, fpath.suffix, fname_suffix)
#
# save a lightweight (i.e., without optimiser and discriminator) model of the given cloned voice and corresponding config assets
#
def main(args):
# set variables
output_path = args.voice_model_asset_path
model_path = os.path.join(args.voice_model_asset_path, args.voice_model_name)
model_config_path = os.path.join(args.voice_model_asset_path, args.voice_model_config_name)
model_light_path = os.path.join(output_path, append_suffix_to_filename(args.voice_model_name, args.minimise_suffix))
model_light_config_path = os.path.join(output_path, append_suffix_to_filename(args.voice_model_config_name, args.minimise_suffix))
if not args.overwrite_assets:
# ensure target output files do not already exist
if (Path(model_light_path).exists()) or (Path(model_light_config_path).exists()):
sys.exit("Naming conflict: Model asset files ({} and/or {}) exist already!" . format (model_light_path, model_light_config_path))
if args.output_format == "txt":
print("Minimising given voice model:")
print("")
print(" + Cloned voice model file path : {}" . format(model_path))
print(" + Cloned voice model config file : {}" . format(model_config_path))
print("")
elif args.output_format == "json":
json_data = {
"success": False,
"in": {
"voice_model_path": format(model_path),
"voice_model_config_path": format(model_config_path)
},
"out": {
"voice_model_light_path": "",
"voice_model_light_config_path": ""
}
}
# load model
config = load_config(model_config_path)
# init model
model = setup_tts_model(config = config)
# load checkpoint / model
model.load_checkpoint(config, model_path, eval = True)
model.disc = None
model_state = model.state_dict()
state = {
"model": model_state
}
torch.save(state, model_light_path)
config.model_args["init_discriminator"] = False
config.save_json(model_light_config_path)
# exit gracefully
if args.output_format == "txt":
print("Completed minimising cloned voice model. The resulting (modified) assets can be found at:")
print(" --> Minimised voice model path : {}" . format(model_light_path))
print(" --> Minimised voice model config path: {}" . format(model_light_config_path))
print("")
print("Done; bye.")
print("")
elif args.output_format == "json":
json_data["out"]["voice_model_light_path"] = format(model_light_path)
json_data["out"]["voice_model_light_config_path"]: format(model_light_config_path)
json_data["success"] = True
print(json.dumps(json_data))
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
main(args)

View File

@@ -0,0 +1,191 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import sys
import os
import argparse
# load coqui-ai/TTS libraries
from TTS.bin.resample import resample_files
from TTS.bin.compute_embeddings import compute_embeddings
import train_config as tc
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Code to prepare voice dataset for multi-speaker baseline model training (i.e., adjust sampling rate and compute speaker embeddings).")
parser.add_argument("--dataset_preset", type = str, choices = ("VCTK", "LibriTTS_tc360", "DAPS", "POTION_Salut", "potion_voice_cloning"), required = True,
help = "Path the voice dataset archive (.zip, .tar.gz, .tgz, .tar.bz2, and .tbz are supported)")
parser.add_argument("--dataset_archive_path", type = str, required = True,
help = "Path the voice dataset archive (.zip, .tar.gz, .tgz, .tar.bz2, and .tbz are supported)")
parser.add_argument("--output_path", type = str, default = "results/datasets",
help = "Path to store augmented dataset")
parser.add_argument("--sampling_rate", type = int, default = 22050, choices = (16000, 22050, 32000, 48000), # 32k & 48k are untested
help = "Sampling rate for training run")
return parser.parse_args()
#
# utility functuion to extract archives (zip, tar, tgz, ...)
# - returns first entry in archive (typically the main directory name contained in the archive)
#
def extract_archive(archive_path, dest_path):
from zipfile import ZipFile
import tarfile
if archive_path.endswith('.zip'):
opener, getnames, mode = ZipFile, ZipFile.namelist, 'r'
elif (archive_path.endswith('.tar.gz')) or (archive_path.endswith('.tgz')):
opener, getnames, mode = tarfile.open, tarfile.TarFile.getnames, 'r:gz'
elif (archive_path.endswith('.tar.bz2')) or (archive_path.endswith('.tbz')):
opener, getnames, mode = tarfile.open, tarfile.TarFile.getnames, 'r:bz2'
else:
print("Extracting archive " + archive_path + " is not supported.")
return
# extract archive
with opener(archive_path, mode) as archive:
archive_dir = archive.getnames()[0]
archive.extractall(path = dest_path)
return archive_dir
#
# main training method (VITS multi-speaker model)
#
def main(args):
print("Commencing preparation of dataset for multi-speaker baseline model training:")
print("")
print(" + Dataset preset: {}" . format(args.dataset_preset))
print(" + Dataset : {}" . format(args.dataset_archive_path))
print(" + Output path : {}" . format(args.output_path))
print(" + Sampling rate : {}" . format(args.sampling_rate))
print("")
# set parameters according to dataset preset
if args.dataset_preset == "VCTK":
DATASET_NAME = tc.VCTK_DATASET_NAME
DATASET_FORMATTER = tc.VCTK_DATASET_FORMATTER
DATASET_FILE_FORMAT = tc.VCTK_DATASET_FILE_FORMAT
NO_EVAL = False
elif args.dataset_preset == "LibriTTS_tc360":
DATASET_NAME = tc.LIBRITTS_TC360_DATASET_NAME
DATASET_FORMATTER = tc.LIBRITTS_TC360_DATASET_FORMATTER
DATASET_FILE_FORMAT = tc.LIBRITTS_TC360_DATASET_FILE_FORMAT
NO_EVAL = False
elif args.dataset_preset == "POTION_Salut":
DATASET_NAME = tc.POTION_SALUT_DATASET_NAME
DATASET_FORMATTER = tc.POTION_SALUT_DATASET_FORMATTER
DATASET_FILE_FORMAT = tc.POTION_SALUT_DATASET_FILE_FORMAT
NO_EVAL = False
elif args.dataset_preset == "potion_voice_cloning":
DATASET_NAME = tc.POTION_SALUT_DATASET_NAME
DATASET_FORMATTER = tc.POTION_SALUT_DATASET_FORMATTER
DATASET_FILE_FORMAT = tc.POTION_SALUT_DATASET_FILE_FORMAT
NO_EVAL = True
# define sampling rate for computing speaker embeddings
SPK_EMB_SAMPLING_RATE = 16000
# define the number of threads used during audio resampling
NUM_RESAMPLE_THREADS = 10
# extract dataset archive
print(f">>> Extracting archive ...")
dataset_root = extract_archive(args.dataset_archive_path, os.path.join(args.output_path, "sr" + str(args.sampling_rate)))
# set dataset path (there should only be ONE directory in the extracted archive location)
dataset_path = os.path.join(args.output_path, "sr" + str(args.sampling_rate), dataset_root)
# ensure the dataset_path exists
os.makedirs(dataset_path, exist_ok = True)
# resample dataset for speaker embeddings computation
print(f">>> Resampling audio files to 16000Hz ...")
resample_files(dataset_path, 16000, file_ext = DATASET_FILE_FORMAT, n_jobs = NUM_RESAMPLE_THREADS)
# compute speaker embeddings
SPEAKER_ENCODER_CHECKPOINT_PATH = "assets/speaker_encoder_model/model_se.pth.tar"
SPEAKER_ENCODER_CONFIG_PATH = "assets/speaker_encoder_model/config_se.json"
# init list speaker embeddings/d-vectors to be used during the training
d_vector_files = []
# check if the speakers embeddings are already computated, if not compute them
embeddings_file = os.path.join(dataset_path, "speakers.pth")
if not os.path.isfile(embeddings_file):
print(f">>> Computing speaker embeddings ...")
compute_embeddings(
SPEAKER_ENCODER_CHECKPOINT_PATH,
SPEAKER_ENCODER_CONFIG_PATH,
embeddings_file,
old_spakers_file = None,
config_dataset_path = None,
formatter_name = DATASET_FORMATTER,
dataset_name = DATASET_NAME,
dataset_path = dataset_path,
meta_file_train = "",
meta_file_val = "",
disable_cuda = False,
no_eval = NO_EVAL
)
d_vector_files.append(embeddings_file)
# if targetted sampling rate is not the same as that used for computing speaker embeddings, replace and resample audio files
if not args.sampling_rate == SPK_EMB_SAMPLING_RATE:
print(f">>> Extracting original archive again (overwritting previously resampled files)...")
extract_archive(args.dataset_archive_path, os.path.join(args.output_path, "sr" + str(args.sampling_rate)))
print(f">>> Resampling audio files to {args.sampling_rate}Hz ...")
resample_files(dataset_path, args.sampling_rate, file_ext = DATASET_FILE_FORMAT, n_jobs = NUM_RESAMPLE_THREADS)
# exit gracefully
print("")
print("Completed preparing voice dataset for multi-speaker baseline model training; generated asset locations are as follows:")
print(" --> {}" . format(dataset_path))
print(" --> {}" . format(embeddings_file))
print("")
print("Done; bye.")
print("")
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
# clear command line arguments to avoid triggering argparse features part of Trainer / coqpit imports
# Traceback (most recent call last):
# File "train_multispeaker_baseline_model.py", line 208, in <module>
# main(args)
# File "train_multispeaker_baseline_model.py", line 177, in main
# trainer = Trainer(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 360, in __init__
# config, new_fields = self.init_training(args, coqpit_overrides, config)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 594, in init_training
# config.parse_known_args(coqpit_overrides, relaxed_parser=True)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 843, in parse_known_args
# parser = self.init_argparse(arg_prefix=arg_prefix, relaxed_parser=relaxed_parser)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 881, in init_argparse
# _init_argparse(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 529, in _init_argparse
# parser = _init_argparse(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 550, in _init_argparse
# return default.init_argparse(
# AttributeError: 'str' object has no attribute 'init_argparse'
sys.argv = [sys.argv[0]]
# ensure the output path exists
os.makedirs(args.output_path, exist_ok = True)
main(args)

View File

@@ -0,0 +1,179 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import os
import argparse
import shutil
import uuid
import json
import torch
from TTS.TTS.tts.utils.speakers import SpeakerManager
from utils.synthesize_utils import init_synth, synthesize, save_waveform
from utils.scoring_utils import init_scoring_vocoder, score_speaker_similarity
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Compute quality score for a given voice model (cloned voice) wrt. a given set of voice recordings (original voice))")
parser.add_argument("--voice_dataset_path", type = str, required = True,
help = "Path to set of voice recordings (original voice)")
parser.add_argument("--voice_model_path", type = str, required = True,
help = "Path to cloned voice model")
parser.add_argument("--voice_model_config_path", type = str, required = True,
help = "Path to config file for the cloned voice model")
parser.add_argument('--speaker_embeddings_path', type = str, required = True,
help = "Path to speaker's embeddings file (i.e., pre-computed embeddings typically stored with the speaker's dataset)")
parser.add_argument("--temp_path", type = str, default = "temp",
help = "Path to store temporary speech assets")
parser.add_argument("--keep_temp", default = False, action = "store_true",
help = "Signal that temporary assets used for scoring should not be deleted once done")
parser.add_argument("--use_cpu", default = False, action = "store_true", # untested!!!
help = "Signal that CPU should be used even if a CUDA-device is available")
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
return parser.parse_args()
#
# main training method (voice cloning)
#
def main(args):
# define assets required for using a pretrained voice
MODEL_PATH = args.voice_model_path
CONFIG_PATH = args.voice_model_config_path
SPK_EMBEDDINGS_PATH = args.speaker_embeddings_path
# set default score
sim_score = -1.0
if args.output_format == "txt":
print("Computing similarity score for a given voice model (cloned voice) wrt. a given set of voice recordings (original voice):")
print("")
print(" + Original voice recordings path: {}" . format(args.voice_dataset_path))
print(" + Cloned voice model file path : {}" . format(MODEL_PATH))
print(" + Cloned voice model config file: {}" . format(CONFIG_PATH))
print(" + Speaker embeddings file : {}" . format(SPK_EMBEDDINGS_PATH))
print("")
elif args.output_format == "json":
json_data = {
"success": False,
"in": {
"voice_dataset_path": format(args.voice_dataset_path),
"voice_model_path": format(MODEL_PATH)
},
"out": {
"score": sim_score
}
}
# determine whether CUDA support is available and set device parameters accordingly
use_cuda = torch.cuda.is_available()
if args.output_format == "txt":
print(" + CUDA availability : {}" . format(use_cuda))
if args.use_cpu:
device = "cpu"
USE_CUDA = False
elif use_cuda:
device = "cuda"
USE_CUDA = True
else:
device = "cpu"
USE_CUDA = False
if args.output_format == "txt":
print(" + Compute device used : {}" . format(device))
print("")
# score the cloned voice (wrt. similarity to recorded voice)
# 1. generate 20 samples (5 x samples from Potion's Web-site; 5 x salutations; 10 x test sentences from other research papers)
# 2. compute similarity score (training samples versus generated samples)
scoring_sentences = [
"Book more meetings, build more trust, and close more sales using Potion.",
"Free forever. As long as you hustle. No credit card required.",
"Don't send plain old boring text emails. Send Potion.",
"What distinguished you from everyone else?",
"We absolutely ensure that you see increased engagement in your outreach efforts.",
"Hi there, Samuel. Hope things are going well for you.",
"Hey person_93. I wanted to reach out to see if you are interested to learn mode about our services.",
"Hi person_90. I noticed you and I are both members of the Green Movement on LinkedIn, and that you just opened a new office in Austin.",
"Hey person_96. Could your team handle an extra 20 leads a week?",
"Hi person_95. For every 100 cold emails you send, you'll only get one reply. That's a lot of effort for little reward.",
"Prosecutors have opened a massive investigation into allegations of fixing games and illegal betting.",
"Feedback must be timely and accurate throughout the project.",
"Humans also judge distance by using the relative sizes of objects.",
"If this is true then those who tend to think creatively really are somehow different.",
"But really in the grand scheme of things this information is insignificant.",
"About half the people who are infected also lose weight.",
"The second half of the book focuses on argument and essay writing.",
"He loves to watch me drink this stuff.",
"Funding is always an issue after the fact.",
"Let us encourage each other."
]
# init speaker manager
speaker_manager = None
speaker_manager = SpeakerManager(d_vectors_file_path = SPK_EMBEDDINGS_PATH)
if args.output_format == "txt":
print(" + No. of speakers : {}" . format(speaker_manager.num_speakers))
print(" + Speaker's names : {}" . format(speaker_manager.embedding_names))
print(" + No. of embeddings : {}" . format(speaker_manager.num_embeddings))
print("")
# assert that only one speaker is present in the embedding's file
assert speaker_manager.num_speakers == 1, f"Number of speakers in the given embedding's file MUST be one; found {speaker_manager.num_speakers} speakers!"
# initialise speech synthesization
voice_config, voice_model = init_synth(CONFIG_PATH, MODEL_PATH, speaker_embeddings_file = SPK_EMBEDDINGS_PATH, use_cuda = USE_CUDA)
# synthesize speech for all scoring sentences
output_path = os.path.join(args.temp_path, str(uuid.uuid4()))
# create temp path (exit if it already exists)
os.makedirs(output_path, exist_ok = False)
for cnt, txt in enumerate(scoring_sentences):
# speaker embeddings provided, use it together with the given model (cloned or baseline model) to synthesize speech
waveform = synthesize(voice_config, voice_model, txt, speaker_manager.get_mean_embedding(speaker_manager.embedding_names[0], speaker_manager.num_embeddings), USE_CUDA)
# save the results
output_fname = os.path.join(output_path, "{:02d}" . format(cnt) + ".wav")
save_waveform(voice_model, waveform, output_fname)
# initialize voice envcoder used for scoring
scoring_vocoder = init_scoring_vocoder()
# determine similarity score
sim_score = score_speaker_similarity(scoring_vocoder, args.voice_dataset_path, output_path)
# clean up
if not args.keep_temp:
shutil.rmtree(output_path)
# exit gracefully
if args.output_format == "txt":
print("")
print("Completed computing similarity score for the two sets of recordings. The resulting similarity score is:")
print(" --> {}" . format(sim_score))
print("")
print("Done; bye.")
print("")
elif args.output_format == "json":
json_data["out"]["score"] = round(float(sim_score), 2)
json_data["success"] = True
print(json.dumps(json_data))
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
# ensure the temp path exists
os.makedirs(args.temp_path, exist_ok = True)
main(args)

View File

@@ -0,0 +1,219 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import os
import argparse
import glob
import shutil
import uuid
import json
import torch
from TTS.tts.utils.speakers import SpeakerManager
from utils.synthesize_utils import init_synth, synthesize, save_waveform
from utils.scoring_utils import init_scoring_vocoder, score_speaker_similarity
#
# What do we need?
# -> list of models to test
# -> test db (user recordings, speaker embeddings, reference to their voice in the multi-speaker model)
# |- user
# |- speaker.pth
# |- userid.txt
# |- txt
# |- wav48
#
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Given a list of models, compute quality scores to determine the top-5 (human-perceived) models.")
parser.add_argument("--models_path", type = str, required = True,
help = "Path to a collection of models and their config file to be used for testing.")
parser.add_argument("--speaker_embeddings_path_list", type = str, nargs = "+", required = True,
help = "List of paths to the speaker embeddings files of the data sets used to train the models.")
parser.add_argument("--test_dataset_path", type = str, required = True,
help = "Path to a set of user recordings with speaker embedding and voice id (the users' ids in the models to be tested)")
parser.add_argument("--temp_path", type = str, default = "temp",
help = "Path to store temporary speech assets")
parser.add_argument("--keep_temp", default = False, action = "store_true",
help = "Signal that temporary assets used for scoring should not be deleted once done")
parser.add_argument("--use_cpu", default = False, action = "store_true", # untested!!!
help = "Signal that CPU should be used even if a CUDA-device is available")
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
return parser.parse_args()
#
# main training method (voice cloning)
#
def main(args):
# define assets required for using a pretrained voice
MODELS_PATH = args.models_path
MODEL_CONFIG_PATH = os.path.join(MODELS_PATH, "config.json")
MODEL_SPK_EMB_PATH_LIST = args.speaker_embeddings_path_list
DATASET_PATH = args.test_dataset_path
USER_ID_FNAME = "userid.txt"
# set default score
sim_score_avg = -1.0
if args.output_format == "txt":
print("Computing similarity score for a given voice model (cloned voice) wrt. a given set of voice recordings (original voice):")
print("")
print(" + Multi-speaker models path : {}" . format(MODELS_PATH))
print(" + Multi-speaker model config file: {}" . format(MODEL_CONFIG_PATH))
print(" + Multi-speaker embeddings file : {}" . format(MODEL_SPK_EMB_PATH_LIST))
print(" + Test dataset path : {}" . format(DATASET_PATH))
#print(" + Speaker embeddings filename : {}" . format(SPK_EMBEDDINGS_FNAME))
print(" + User ID filename : {}" . format(USER_ID_FNAME))
print("")
elif args.output_format == "json":
json_data = {
"success": False,
"in": {
"models_path": format(MODELS_PATH),
"dataset_path": format(DATASET_PATH)
},
"out": {
"best_model": None,
"top_5_models": None
}
}
# determine whether CUDA support is available and set device parameters accordingly
use_cuda = torch.cuda.is_available()
if args.output_format == "txt":
print(" + CUDA availability : {}" . format(use_cuda))
if args.use_cpu:
device = "cpu"
USE_CUDA = False
elif use_cuda:
device = "cuda"
USE_CUDA = True
else:
device = "cpu"
USE_CUDA = False
if args.output_format == "txt":
print(" + Compute device used : {}" . format(device))
print("")
# score the cloned voice (wrt. similarity to recorded voice)
# 1. generate 20 samples (5 x samples from Potion's Web-site; 5 x salutations; 10 x test sentences from other research papers)
# 2. compute similarity score (training samples versus generated samples)
scoring_sentences = [
"Book more meetings, build more trust, and close more sales using Potion.",
"Free forever. As long as you hustle. No credit card required.",
"Don't send plain old boring text emails. Send Potion.",
"What distinguished you from everyone else?",
"We absolutely ensure that you see increased engagement in your outreach efforts.",
"Hi there, Samuel. Hope things are going well for you.",
"Hey person_93. I wanted to reach out to see if you are interested to learn mode about our services.",
"Hi person_90. I noticed you and I are both members of the Green Movement on LinkedIn, and that you just opened a new office in Austin.",
"Hey person_96. Could your team handle an extra 20 leads a week?",
"Hi person_95. For every 100 cold emails you send, you'll only get one reply. That's a lot of effort for little reward.",
"Prosecutors have opened a massive investigation into allegations of fixing games and illegal betting.",
"Feedback must be timely and accurate throughout the project.",
"Humans also judge distance by using the relative sizes of objects.",
"If this is true then those who tend to think creatively really are somehow different.",
"But really in the grand scheme of things this information is insignificant.",
"About half the people who are infected also lose weight.",
"The second half of the book focuses on argument and essay writing.",
"He loves to watch me drink this stuff.",
"Funding is always an issue after the fact.",
"Let us encourage each other."
]
# init scoring tracker
sim_score = {}
for model_fname in glob.glob(os.path.join(MODELS_PATH, "check*.pth")):
# init scoring tracker
sim_score[os.path.basename(model_fname)] = []
# score each moddel for every user
for user_dir in os.listdir(DATASET_PATH):
# get user's speaker id / name
with open(os.path.join(DATASET_PATH, user_dir, USER_ID_FNAME), 'r') as f:
user_data = json.load(f)
print("Speaker name: {}" . format(user_data["speaker_name"]))
# init speaker manager
speaker_manager = None
speaker_manager = SpeakerManager(d_vectors_file_path = MODEL_SPK_EMB_PATH_LIST)
#speaker_manager = SpeakerManager(speaker_id_file_path = os.path.join(MODELS_PATH, "speakers.pth"))
print("Number of speakers:", speaker_manager.num_speakers)
print("Speaker names :", speaker_manager.speaker_names)
#print("Embedding names :", speaker_manager.embedding_names)
# assert that the user is indeed present in the embedding's file
#assert speaker_manager.num_speakers == 1, f"Number of speakers in the given embedding's file MUST be one; found {speaker_manager.num_speakers} speakers!"
for model_fname in glob.glob(os.path.join(MODELS_PATH, "check*.pth")):
# initialise speech synthesization
voice_config, voice_model = init_synth(MODEL_CONFIG_PATH, model_fname, speaker_embeddings_file = MODEL_SPK_EMB_PATH_LIST, use_cuda = USE_CUDA)
# synthesize speech for all scoring sentences
output_path = os.path.join(args.temp_path, str(uuid.uuid4()))
# create temp path (exit if it already exists)
os.makedirs(output_path, exist_ok = False)
for cnt, txt in enumerate(scoring_sentences):
# speaker embeddings provided, use it together with the given model (cloned or baseline model) to synthesize speech
#waveform = synthesize(voice_config, voice_model, txt, speaker_manager.get_mean_embedding(user_data["speaker_name"]), USE_CUDA)
waveform = synthesize(voice_config, voice_model, txt, speaker_embeddings = speaker_manager.get_mean_embedding(user_data["speaker_name"], num_samples = None, randomize = False), use_cuda = USE_CUDA)
#waveform = synthesize(voice_config, voice_model, txt, speaker_embeddings = speaker_manager.get_mean_embedding(user_data["speaker_name"]), speaker_id = speaker_manager.name_to_id[user_data["speaker_name"]], use_cuda = USE_CUDA)
#waveform = synthesize(voice_config, voice_model, txt, speaker_id = speaker_manager.name_to_id[user_data["speaker_name"]], use_cuda = USE_CUDA)
# save the results
output_fname = os.path.join(output_path, "{:02d}" . format(cnt) + ".wav")
save_waveform(voice_config, voice_model, waveform, output_fname)
# initialize voice envcoder used for scoring
scoring_vocoder = init_scoring_vocoder()
# determine similarity score
sim_score[os.path.basename(model_fname)].append(score_speaker_similarity(scoring_vocoder, os.path.join(DATASET_PATH, user_dir, "wav48", "1"), output_path))
# clean up
if not args.keep_temp:
shutil.rmtree(output_path)
# determine the top-5 checkpoints (or fewer if there are less than 5 entries)
top_5_checkpoints = [(k, sum(v) / len(v)) for k, v in sorted(sim_score.items(), key = lambda item: sum(item[1]) / len(item[1]), reverse = True)[:5]]
# exit gracefully
if args.output_format == "txt":
print("")
print("Completed computing similarity score for the two sets of recordings. The best and the top-5 models based on their similarity scores are:")
print(" --> Best model : {}" . format(top_5_checkpoints[0]))
print(" --> Top-5 models: {}" . format(top_5_checkpoints))
print("")
print("Done; bye.")
print("")
elif args.output_format == "json":
json_data["out"]["best_model"] = top_5_checkpoints[0]
json_data["out"]["top_5_models"] = top_5_checkpoints
json_data["success"] = True
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
# ensure the temp path exists
os.makedirs(args.temp_path, exist_ok = True)
main(args)

View File

@@ -0,0 +1,161 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import argparse
import re
from itertools import combinations
import json
from utils.matching_utils import match_name_textualsim, match_name_mra
from utils.transcription_utils import get_transcription
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Score a given salutation recording wrt. its desired content, the actual salutation recording, and a generated transcription (using Potion's internal Transciption API) of the recording.")
parser.add_argument("--recording_path", type = str, required = True,
help = "Path to salutation recoding (.wav audio file)")
parser.add_argument("--first_name", type = str, required = True,
help = "First name that the salutation recoding is meant to use")
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
return parser.parse_args()
#
# load dictionary of common first names (Name DB source: World Gender Name Dictionary v2.0; https://dataverse.harvard.edu/dataset.xhtml?persistentId=doi:10.7910/DVN/MSEGSJ)
#
def load_names():
NAME_DICTIONARY = "./assets/wgnd_2_0_unique_names_only_limited_special_chars.csv"
# removing the characters
with open(NAME_DICTIONARY) as f:
names_list = [line.rstrip() for line in f]
names_set = set(names_list)
return names_set
#
# auxilliary function to generate a list of all combinations of words from a given list of words (w/o chaningthe order of words)
#
def get_combinations(word_list):
comb_list = word_list.copy()
for start, end in combinations(range(len(word_list)), 2):
comb_list.append(' '.join(word for word in word_list[start:end + 1]))
return comb_list
#
# main method
#
def main(args):
# set default score
score = -1.0
if args.output_format == "txt":
print("Commencing scoring of the given salutation recording:")
print("")
print(" + Salutation recording path: {}" . format(args.recording_path))
print(" + Salutation first name : {}" . format(args.first_name))
print("")
elif args.output_format == "json":
json_data = {
"success": False,
"in": {
"recording_path": format(args.recording_path),
"first_name": format(args.first_name)
},
"out": {
"score": score
}
}
# obtain a transcription for the given salutation recording
trans_success, trans_txt, trans_score = get_transcription(args.recording_path)
# proceed if a transcription was obtained successfully
if trans_success:
# check given first name against name database (Name DB source: https://dataverse.harvard.edu/dataset.xhtml?persistentId=doi:10.7910/DVN/MSEGSJ)
names_set = load_names()
# ensure all words / letters are lower case only
first_name = args.first_name.lower()
trans_txt = trans_txt.lower()
name_valid = False
if first_name in names_set:
name_valid = True
else:
# cannot compute advanced score for a name that we do not have in our first name database (i.e., fallback to confidence score from transcription service)
if args.output_format == "txt":
print("Unknown first name: {}" . format(args.first_name))
print("")
score = trans_score
if name_valid:
#
trans_candidate_names = re.findall(r" ([a-zA-Z_-]+)", trans_txt)
if len(trans_candidate_names) >= 2:
trans_candidate_names = get_combinations(trans_candidate_names)
cand_names_real = []
for cand_name in trans_candidate_names:
# check is cname is a valid name
if cand_name.lower() in names_set:
cand_names_real.append(cand_name)
if cand_names_real:
# name similarity with first_name
for cand_name_real in cand_names_real:
# check is cname is a valid name
jaro, lev = match_name_textualsim(first_name, cand_name_real)
mra = match_name_mra(first_name, cand_name_real)
#print("Scores ({}): {} -- {} -- {} -- {}" . format(cand_name_real, trans_score, jaro, lev, mra))
# score if the normalised Jaro-Winkler distance >= 0.875
# OR
# the normalised Jaro-Winkler distance >= 0.75 and the normalised Levenshtein distance is >= 0.7
# OR
# the normalised MRA >= 0.75
# else average
if jaro > 0.875:
score = (jaro + trans_score) / 2
break
elif (jaro >= 0.75) and (lev >= 0.7):
score = (((jaro + lev) / 2) + trans_score) / 2
break
elif mra >= 0.75:
score = (mra + trans_score) / 2
break
else:
score_new = (((jaro + lev + mra) / 3) + trans_score) / 2
if score_new > score:
score = score_new
if args.output_format == "txt":
print(" >> Salutation score : {}" . format(score))
print("")
print("Done; bye.")
print("")
elif args.output_format == "json":
json_data["out"]["score"] = round(score, 2)
json_data["success"] = True
print(json.dumps(json_data))
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
main(args)

View File

@@ -0,0 +1,144 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import os
import argparse
import subprocess
import json
import uuid
import torch
from TTS.tts.utils.speakers import SpeakerManager
from utils.synthesize_utils import init_synth, synthesize, save_waveform
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Code to synthesize speech for a given voice model")
parser.add_argument("--voice_model_path", type = str, required = True,
help = "Path to cloned voice model")
parser.add_argument("--voice_model_config_path", type = str, required = True,
help = "Path to config file for the cloned voice model")
parser.add_argument('--speaker_embeddings_path', type = str, required = True,
help = "Path to speaker's embeddings file (i.e., pre-computed embeddings typically stored with the speaker's dataset)")
parser.add_argument("--txt", type = str, required = True,
help = "Text to synthesize")
parser.add_argument("--output_path", type = str, default = "results/speech",
help = "Path to store generated speech assets")
parser.add_argument("--target_sampling_rate", type = int, default = 48000,
help = "Desired sampling rate (in Hz) for output file")
parser.add_argument('--speech_sample_wav_path', type = str, default = None,
help = "Path to a sample utterance of the speaker (used for style transfer)")
parser.add_argument('--speech_sample_txt', type = str, default = None,
help = "Text of the sample utterance of the speaker (used for style transfer)")
parser.add_argument("--trim_silence", default = True, action = "store_false",
help = "Signal whether to trim silence from synthesised speech")
parser.add_argument("--use_cpu", default = False, action = "store_true", # untested!!!
help = "Signal that CPU should be used even if a CUDA-device is available")
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
return parser.parse_args()
#
# main speech synthesizing method
#
def main(args):
# define assets required for using a pretrained voice
MODEL_PATH = args.voice_model_path
CONFIG_PATH = args.voice_model_config_path
SPK_EMBEDDINGS_PATH = args.speaker_embeddings_path
if args.output_format == "txt":
print("Commencing speech synthesizing:")
print("")
print(" + Voice model file path : {}" . format(MODEL_PATH))
print(" + Voice model config file: {}" . format(CONFIG_PATH))
print(" + Speaker embeddings file: {}" . format(SPK_EMBEDDINGS_PATH))
print(" + Output path : {}" . format(args.output_path))
print(" + Text to synthesize : {}" . format(args.txt))
print("")
elif args.output_format == "json":
json_data = {
"success": False,
"in": {
"voice_model_path": format(MODEL_PATH),
"voice_model_config_path": format(CONFIG_PATH),
"speaker_embeddings_path": format(SPK_EMBEDDINGS_PATH)
},
"out": {
"speech_original_path": "",
"speech_resampled_path": ""
}
}
# determine whether CUDA support is available and set device parameters accordingly
use_cuda = torch.cuda.is_available()
if args.output_format == "txt":
print(" + CUDA availability : {}" . format(use_cuda))
if args.use_cpu:
device = "cpu"
USE_CUDA = False
elif use_cuda:
device = "cuda"
USE_CUDA = True
else:
device = "cpu"
USE_CUDA = False
if args.output_format == "txt":
print(" + Compute device used : {}" . format(device))
# init speaker manager
speaker_manager = None
speaker_manager = SpeakerManager(d_vectors_file_path = SPK_EMBEDDINGS_PATH)
if args.output_format == "txt":
print(" + No. of speakers : {}" . format(speaker_manager.num_speakers))
print(" + Speaker's names : {}" . format(speaker_manager.embedding_names))
print(" + No. of embeddings : {}" . format(speaker_manager.num_embeddings))
print("")
# assert that only one speaker is present in the embedding's file
assert speaker_manager.num_speakers == 1, f"Number of speakers in the given embedding's file MUST be one; found {speaker_manager.num_speakers} speakers!"
# initialise speech synthesization
voice_config, voice_model = init_synth(CONFIG_PATH, MODEL_PATH, speaker_embeddings_file = SPK_EMBEDDINGS_PATH, use_cuda = USE_CUDA)
# synthesize speech
waveform = synthesize(voice_config, voice_model, args.txt, speaker_embeddings = speaker_manager.get_mean_embedding(speaker_manager.embedding_names[0], speaker_manager.num_embeddings), speech_sample_wav = args.speech_sample_wav_path, speech_sample_txt = args.speech_sample_txt, use_cuda = USE_CUDA, trim_silence = args.trim_silence)
# save the synthesize speech
output_fname_prefix = str(uuid.uuid4())
output_fname = output_fname_prefix + ".wav"
save_waveform(voice_config, voice_model, waveform, os.path.join(args.output_path, output_fname))
# convert the synthesize speech waveform to the target sampling rate
output_resampled_fname = output_fname_prefix + "_sr" + str(args.target_sampling_rate) + ".wav"
subprocess.run(["ffmpeg", "-i", os.path.join(args.output_path, output_fname), "-ar", str(args.target_sampling_rate), os.path.join(args.output_path, output_resampled_fname)], check=True)
# exit gracefully
if args.output_format == "txt":
print("")
print(">>> Saving origianl output to : {}" . format(os.path.join(args.output_path, output_fname)))
print(">>> Saving resampled output to: {}" . format(os.path.join(args.output_path, output_resampled_fname)))
print("")
print("Speech synthesizing has completed. Bye.")
elif args.output_format == "json":
json_data["out"]["speech_original_path"] = format(os.path.join(args.output_path, output_fname))
json_data["out"]["speech_resampled_path"] = format(os.path.join(args.output_path, output_resampled_fname))
json_data["success"] = True
print(json.dumps(json_data))
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
# ensure the output path exists
os.makedirs(args.output_path, exist_ok = True)
main(args)

View File

@@ -0,0 +1,48 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
####################################################################################
### ###
### Configuration File for Multi-Speaker Baseline Model Training & Voice Cloning ###
### ###
####################################################################################
import os
# Data Sets
## VCTK (v0.92), sampling rate: 48000
VCTK_PRESET = "VCTK"
VCTK_DATASET_NAME = "VCTK"
VCTK_DATASET_FORMATTER = "vctk"
VCTK_DATASET_FILE_FORMAT = "flac"
VCTK_DATASET_PATH = "results/datasets/sr22050/VCTK-Corpus-0.92"
VCTK_SPK_EMB_PATH = os.path.join(VCTK_DATASET_PATH, "speakers.pth")
## LibriTTS TC360, sampling rate: 24000
LIBRITTS_TC360_PRESET = "LibriTTS_tc360"
LIBRITTS_TC360_DATASET_NAME = "LibtriTTS-tc360"
LIBRITTS_TC360_DATASET_FORMATTER = "libri_tts"
LIBRITTS_TC360_DATASET_FILE_FORMAT = "wav"
LIBRITTS_TC360_DATASET_PATH = "results/datasets/sr22050/LibriTTS/train-clean-360"
LIBRITTS_TC360_SPK_EMB_PATH = os.path.join(LIBRITTS_TC360_DATASET_PATH, "speakers.pth")
## DAPS
## Potion salutation recordings
POTION_SALUT_PRESET = "POTION_Salut"
POTION_SALUT_DATASET_NAME = "potion-Salut"
POTION_SALUT_DATASET_FORMATTER = "vctk_old"
POTION_SALUT_DATASET_FILE_FORMAT = "wav"
POTION_SALUT_DATASET_PATH = "results/datasets/sr22050/potion-salut-corpus-4ac24ce8-8405-4b70-8b48-018d4492f6e9"
POTION_SALUT_SPK_EMB_PATH = os.path.join(POTION_SALUT_DATASET_PATH, "speakers.pth")
## Potion voice cloning recordings
POTION_SALUT_PRESET = "potion_voice_cloning"
POTION_SALUT_DATASET_NAME = ""
POTION_SALUT_DATASET_FORMATTER = "vctk_old"
POTION_SALUT_DATASET_FILE_FORMAT = "wav"

View File

@@ -0,0 +1,238 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import sys
import os
import argparse
import torch
# load coqui-ai/trainer libraries
from trainer import Trainer, TrainerArgs
# load coqui-ai/TTS libraries
from TTS.tts.configs.shared_configs import BaseDatasetConfig
from TTS.tts.configs.vits_config import VitsConfig
from TTS.tts.datasets import load_tts_samples
from TTS.tts.models.vits import Vits, VitsArgs, VitsAudioConfig
import train_config as tc
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Code to train multi-speaker baseline model")
parser.add_argument("--datasets", type = str, nargs = "+", required = True,
choices = (tc.VCTK_PRESET, tc.LIBRITTS_TC360_PRESET, tc.POTION_SALUT_PRESET),
help = "List of training datasets to be included in training run.")
parser.add_argument("--output_path", type = str, default = "results/baseline-models",
help = "Path to store trained / generated assets")
parser.add_argument("--batch_size", type = int, default = 32, # 96 is suitable for AWS g5 instances using VCTK v0.80 only
help = "Batch size for training run") # 32 is suitable for AWS g5 instances using VCTK v0.92, LibriTTS 360 and Potion salutations
parser.add_argument("--max_epochs", type = int, default = 100, # 250 for batch size 64 (with VCTK only)
help = "Maximum number of epochs for training run") # 100 for batch size 32 (with VCTK v0.92, LibriTTS 360 and POTION_Salut)
return parser.parse_args()
#
# main training method (VITS multi-speaker model)
#
def main(args):
print("Commencing training of a new multi-speaker potion-voice baseline model:")
print("")
print(" + Datasets : {}" . format(args.datasets))
print(" + Output path : {}" . format(args.output_path))
print(" + Batch size : {}" . format(args.batch_size))
print(" + Training runs (max epochs): {}" . format(args.max_epochs))
print("")
# determine whether CUDA support is available and set device parameters accordingly
use_cuda = torch.cuda.is_available()
print(" + CUDA availability : {}" . format(use_cuda))
if use_cuda:
device = "cuda"
device_torch = torch.device("cuda")
else:
device = "cpu"
device_torch = torch.device("cpu")
print(" + Compute device used : {}" . format(device))
print("")
# define training data sets
dataset_config_list = []
speaker_embeddings_list = []
# VCTK (v0.92)
if tc.VCTK_PRESET in args.datasets:
vctk_dataset_config = BaseDatasetConfig(dataset_name = tc.VCTK_DATASET_NAME, formatter = tc.VCTK_DATASET_FORMATTER, language = "en-us", path = tc.VCTK_DATASET_PATH)
dataset_config_list.append(vctk_dataset_config)
speaker_embeddings_list.append(tc.VCTK_SPK_EMB_PATH)
# LibriTTS
if tc.LIBRITTS_TC360_PRESET in args.datasets:
libritts_dataset_config = BaseDatasetConfig(dataset_name = tc.LIBRITTS_TC360_DATASET_NAME, formatter = tc.LIBRITTS_TC360_DATASET_FORMATTER, language = "en-us", path = tc.LIBRITTS_TC360_DATASET_PATH)
dataset_config_list.append(libritts_dataset_config)
speaker_embeddings_list.append(tc.LIBRITTS_TC360_SPK_EMB_PATH)
# DAPS
# Potion recordings dataset
if tc.POTION_SALUT_PRESET in args.datasets:
potion_dataset_config = BaseDatasetConfig(dataset_name = tc.POTION_SALUT_DATASET_NAME, formatter = tc.POTION_SALUT_DATASET_FORMATTER, language = "en-us", path = tc.POTION_SALUT_DATASET_PATH)
dataset_config_list.append(potion_dataset_config)
speaker_embeddings_list.append(tc.POTION_SALUT_SPK_EMB_PATH)
# set VITS training parameters
audio_config = VitsAudioConfig(
sample_rate = 22050,
win_length = 1024,
hop_length = 256,
num_mels = 80,
mel_fmin = 0,
mel_fmax = None,
)
vitsArgs = VitsArgs(
use_speaker_embedding = False,
use_d_vector_file = True,
d_vector_file = speaker_embeddings_list,
d_vector_dim = 512,
num_layers_text_encoder = 10
)
config = VitsConfig(
model_args = vitsArgs,
audio = audio_config,
run_name = "vits_potion",
use_speaker_embedding = False,
use_d_vector_file = True,
d_vector_file = speaker_embeddings_list,
d_vector_dim = 512,
batch_size = args.batch_size,
eval_batch_size = 16,
batch_group_size = 0, # changing this to 5 (VITS training default) slows training down, but doesn't have any positive training effects
num_loader_workers = 4,
num_eval_loader_workers = 4,
run_eval = True,
test_delay_epochs = -1,
epochs = args.max_epochs,
text_cleaner = "english_cleaners",
use_phonemes = False,
phoneme_language = "en-us",
phoneme_cache_path = os.path.join(args.output_path, "phoneme_cache"),
compute_input_seq_cache = True,
print_step = 50,
print_eval = True,
mixed_precision = True,
max_text_len = 325,
output_path = args.output_path,
save_checkpoints = True,
save_step = 5000,
save_n_checkpoints = 20,
save_all_best = True,
datasets = dataset_config_list,
cudnn_benchmark = False,
#characters = {
# "pad": "_",
# "eos": "&",
# "bos": "*",
# "characters": "!¡'(),-.:;¿?abcdefghijklmnopqrstuvwxyz «°±µ»$%&‘’‚“`”„",
# "punctuations": "!¡'(),-.:;¿? ",
# "phonemes": None,
# "unique": True
#},
test_sentences = [
# VCTK
["It took me quite a long time to develop a voice, and now that I have it I'm not going to be silent.", "VCTK_p299"], # 299 - F, American, California
["Hey! Sandra.", "VCTK_p302"], # 302 - M, Canadian, Montreal
["I'm sorry Dave. I'm afraid I can't do that.", "VCTK_p308"], # 308 - F, American, Alabama
["This cake is great. It's so delicious and moist.", "VCTK_p334"], # 334 - M, American, Chicago
["Prior to November 22, 1963.", "VCTK_p363"], # 363 - M, Canadian, Toronto
["It took me quite a long time to develop a voice, and now that I have it I'm not going to be silent.", "VCTK_p376"], # 376 - M, Indian
# LibriTTS
["It took me quite a long time to develop a voice, and now that I have it I'm not going to be silent.", "LTTS_38"], # 38 - M - train-clean-360 R. Francis Smith
["Hey! Sandra.", "LTTS_22"], # 22 - F - train-clean-360 Michelle Crandall
["I'm sorry Dave. I'm afraid I can't do that.", "LTTS_329"], # 329 - M - train-clean-360 Todd Cranston-Cuebas
["This cake is great. It's so delicious and moist.", "LTTS_224"], # 224 - F - train-clean-360 Caitlin Kelly
["Prior to November 22, 1963.", "LTTS_339"], # 339 - F - train-clean-360 Heather Ordover
["It took me quite a long time to develop a voice, and now that I have it I'm not going to be silent.", "LTTS_1779"], # 1779 - F - train-clean-360 Cynthia Zocca
# DAPS
# Potion Salutation Recordings
["Hey! Andrew.", "VCTK_old_POTION_6231a04a9f5f7707120b4215"], # Potion user
["Hey, Michelle.", "VCTK_old_POTION_628c025a09943300254532ae"], # Potion user
["Hey! George.", "VCTK_old_POTION_6189854312bfc264a528c3c3"], # Potion user
["Hey there, Rachel.", "VCTK_old_POTION_63163f7b2be1c500219f3d04"], # Potion user
#["I'm sorry Dave. I'm afraid I can't do that.", "VCTK_old_POTION_63222943fd2bff2e1c651b3b"] # Potion user - not in yet
]
)
# load training samples
train_samples, eval_samples = load_tts_samples(config.datasets, eval_split = True, eval_split_max_size = config.eval_split_max_size, eval_split_size = config.eval_split_size)
# init VITS model
model = Vits.init_from_config(config)
# init multi-speaker training
trainer = Trainer(
TrainerArgs(),
config,
args.output_path,
model = model,
train_samples = train_samples,
eval_samples = eval_samples
)
# trigger model training
try:
trainer.fit()
except (KeyboardInterrupt, SystemExit):
print("Training stopped manually (via keyboard interrupt)! Bye.")
exit(0)
# exit gracefully
print("")
print("Completed training a new multi-speaker potion-voice baseline model, which can be found at:")
print(" --> {}" . format(args.output_path))
print("")
print("Done; bye.")
print("")
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
# clear command line arguments to avoid triggering argparse features part of Trainer / coqpit imports
# Traceback (most recent call last):
# File "train_multispeaker_baseline_model.py", line 208, in <module>
# main(args)
# File "train_multispeaker_baseline_model.py", line 177, in main
# trainer = Trainer(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 360, in __init__
# config, new_fields = self.init_training(args, coqpit_overrides, config)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 594, in init_training
# config.parse_known_args(coqpit_overrides, relaxed_parser=True)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 843, in parse_known_args
# parser = self.init_argparse(arg_prefix=arg_prefix, relaxed_parser=relaxed_parser)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 881, in init_argparse
# _init_argparse(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 529, in _init_argparse
# parser = _init_argparse(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 550, in _init_argparse
# return default.init_argparse(
# AttributeError: 'str' object has no attribute 'init_argparse'
sys.argv = [sys.argv[0]]
# ensure the output path exists
os.makedirs(args.output_path, exist_ok = True)
main(args)

View File

@@ -0,0 +1,22 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import textdistance
#
# Name matching via textual similarity search
# + Returns two (normalised) distance measures: the Jaro-Winkler Distance and the regular Levenshtein Distance
#
def match_name_textualsim(name1, name2):
jaro_winkler = textdistance.jaro_winkler.normalized_similarity(name1, name2)
levenshtein = textdistance.levenshtein.normalized_similarity(name1, name2)
return jaro_winkler, levenshtein
#
# Name matching via phonetic matching algorithm (using the normalised Match Rating Approach)
#
def match_name_mra(name1, name2):
return textdistance.mra.normalized_similarity(name1, name2)

View File

@@ -0,0 +1,37 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
from pathlib import Path
from itertools import groupby
import numpy as np
from resemblyzer import preprocess_wav, VoiceEncoder
def init_scoring_vocoder():
# initialise voice encoder (using CUDA by default; CPU as fallback)
encoder = VoiceEncoder()
return encoder
def score_speaker_similarity(scoring_vocoder, spk_a_fpaths, spk_b_fpaths):
# filepaths to waveforms
wav_fpaths = list(Path(spk_a_fpaths).glob("*.wav")) + list(Path(spk_b_fpaths).glob("*.wav"))
# group the wavs per speaker and load them using the preprocessing function provided with Resemblyzer to load wavs in memory
# - normalizes the volume, trims long silences and resamples the wav to the correct sampling rate
speaker_wavs = {speaker: list(map(preprocess_wav, wav_fpaths)) for speaker, wav_fpaths in groupby(wav_fpaths, lambda wav_fpath: wav_fpath.parent.stem)}
# compute similarity between two speaker embeddings
# - divides the utterances of each speaker in groups of identical size and embed each group as a speaker embedding
spk_embeds_a = np.array([scoring_vocoder.embed_speaker(wavs[:len(wavs) // 2]) for wavs in speaker_wavs.values()])
spk_embeds_b = np.array([scoring_vocoder.embed_speaker(wavs[len(wavs) // 2:]) for wavs in speaker_wavs.values()])
spk_sim_matrix = np.inner(spk_embeds_a, spk_embeds_b)
sim_score = np.average([spk_sim_matrix[0, 1], spk_sim_matrix[1, 0]])
return(sim_score)

View File

@@ -0,0 +1,77 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import numpy as np
# load coqui-ai/TTS libraries
from TTS.config import load_config
from TTS.tts.models import setup_model as setup_tts_model
from TTS.tts.utils.synthesis import synthesis, trim_silence
def init_synth(config_path, voice_model_path, speakers_file_path = None, speaker_embeddings_file = None, use_cuda = True, use_phonemes = False):
# load config and customise config parameters (those that are different during training and inference / synthesizing)
config = load_config(config_path)
if not speakers_file_path is None:
config.use_speaker_embedding = True,
config.use_d_vector_file = False,
config.speakers_file = speakers_file_path
config.model_args["use_speaker_embedding"] = True,
config.model_args["use_d_vector_file"] = False,
config.model_args["speakers_file"] = speakers_file_path
else:
config.d_vector_file = speaker_embeddings_file
config.model_args["d_vector_file"] = speaker_embeddings_file
# set whether or not phonemes are used
config.use_phonemes = use_phonemes
# load cloned voice model
model = setup_tts_model(config = config)
model.load_checkpoint(config, voice_model_path, eval = True)
if use_cuda:
model.cuda()
return config, model
def synthesize(config, voice_model, txt, speaker_embeddings = None, speaker_id = None, speech_sample_wav = None, speech_sample_txt = None, use_cuda = True, trim_silence = True):
# disable language selection
#language_id = 0
language_id = None
# set default voice encoder
use_gl = True
# synthesize voice
outputs = synthesis(
model = voice_model,
text = txt,
CONFIG = config,
use_cuda = use_cuda,
speaker_id = speaker_id,
style_wav = speech_sample_wav,
style_text = speech_sample_txt,
use_griffin_lim = use_gl,
do_trim_silence = trim_silence,
d_vector = speaker_embeddings,
language_id = language_id
)
waveform = outputs["wav"]
waveform = waveform.squeeze()
# trim silence (disabled due to some "TypeError: 'bool' object is not callable" bug that needs to be investigated)
#if (config.audio["do_trim_silence"]) or (trim_silence):
# waveform = trim_silence(waveform, voice_model.ap)
return waveform
def save_waveform(config, voice_model, waveform, out_path):
wav = np.array(waveform)
voice_model.ap.save_wav(wav, out_path, config["audio"].sample_rate)

View File

@@ -0,0 +1,94 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import sys
import os
import requests
from requests.structures import CaseInsensitiveDict
import json
from time import sleep
# set transcription service API endpoint and token (retrieved from operating system's ENV variables)
# + sample endpoints:
# - [dev] "https://development.sendpotion.com/api/transcript"
# - [staging] "https://staging.sendpotion.com/api/transcript"
API_ENDPOINT = os.environ.get("TRANSCRIPTION_API_ENDPOINT")
API_TOKEN = os.environ.get("TRANSCRIPTION_API_TOKEN")
#
# Using potions internal transcription API endpoint, get a transcription for a given (wav) audio recording
# + returns a triple:
# - Boolean ......... indicating success (True) or failure (False)
# - String / None ... transcription text (or None in failure case)
# - Float / None .... transcription confidence score (or None in failure case)
#
def get_transcription(wav_fname):
# validate that transcription service API endpoint and token are set
if (API_ENDPOINT is None) or (API_TOKEN is None):
# terminate
print("TRANSCRIPTION_API_ENDPOINT and TRANSCRIPTION_API_TOKEN environment variables MUST be set!")
sys.exit(1)
# set request header to contain (bearer) API token
headers = CaseInsensitiveDict()
headers["Accept"] = "application/json"
headers["Authorization"] = "Bearer " + str(API_TOKEN)
# set files field (data is empty)
files = {'wav': open(wav_fname, 'rb')}
# issue POST request and save response as response object
response = requests.post(url = API_ENDPOINT, headers = headers, files = files)
# test for auth error
# test for timeout
# check if the status code is not an error code (i.e., 4xx or 5xx)
success = False
if response:
# extracting response text
response_text = response.text
response_json = json.loads(response_text)
#print(response_json)
if response.ok: # synch call
success = True
trans_text = response_json["transcriptObj"]["text"]
trans_score = float(response_json["transcriptObj"]["confidence"])
else: # fallback to asynch call
# wait up to 60 seconds for the transcription to be ready; try every 5 seconds
wait = 0
while wait < 60:
sleep(5)
wait += 5
# issue GET request using the previously returned reqiestId and save response as response object
response_get = requests.get(url = API_ENDPOINT + ':' + response_json["requestId"])
# check if the status code is not an error code (i.e., 4xx or 5xx)
if response_get.ok:
response_get_text = response_get.text
response_get_json = json.loads(response_get_text)
success = True
trans_text = response_get_json["transcriptObj"]["text"]
trans_score = float(response_get_json["transcriptObj"]["confidence"])
break
# in case no successful response is received even after a 60 seconds waiting period -> proceed without transcription
#if not response_get.ok:
# print("Response: FAILED.")
#else:
# print("ERROR: {} ({})" . format(response.status_code, response.text))
if success:
return success, trans_text, trans_score
else:
return False, None, None