before changes
This commit is contained in:
@@ -0,0 +1,22 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import textdistance
|
||||
|
||||
|
||||
#
|
||||
# Name matching via textual similarity search
|
||||
# + Returns two (normalised) distance measures: the Jaro-Winkler Distance and the regular Levenshtein Distance
|
||||
#
|
||||
def match_name_textualsim(name1, name2):
|
||||
jaro_winkler = textdistance.jaro_winkler.normalized_similarity(name1, name2)
|
||||
levenshtein = textdistance.levenshtein.normalized_similarity(name1, name2)
|
||||
|
||||
return jaro_winkler, levenshtein
|
||||
|
||||
|
||||
#
|
||||
# Name matching via phonetic matching algorithm (using the normalised Match Rating Approach)
|
||||
#
|
||||
def match_name_mra(name1, name2):
|
||||
return textdistance.mra.normalized_similarity(name1, name2)
|
||||
@@ -0,0 +1,37 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
from pathlib import Path
|
||||
from itertools import groupby
|
||||
|
||||
import numpy as np
|
||||
|
||||
from resemblyzer import preprocess_wav, VoiceEncoder
|
||||
|
||||
|
||||
def init_scoring_vocoder():
|
||||
|
||||
# initialise voice encoder (using CUDA by default; CPU as fallback)
|
||||
encoder = VoiceEncoder()
|
||||
|
||||
return encoder
|
||||
|
||||
|
||||
def score_speaker_similarity(scoring_vocoder, spk_a_fpaths, spk_b_fpaths):
|
||||
|
||||
# filepaths to waveforms
|
||||
wav_fpaths = list(Path(spk_a_fpaths).glob("*.wav")) + list(Path(spk_b_fpaths).glob("*.wav"))
|
||||
|
||||
# group the wavs per speaker and load them using the preprocessing function provided with Resemblyzer to load wavs in memory
|
||||
# - normalizes the volume, trims long silences and resamples the wav to the correct sampling rate
|
||||
speaker_wavs = {speaker: list(map(preprocess_wav, wav_fpaths)) for speaker, wav_fpaths in groupby(wav_fpaths, lambda wav_fpath: wav_fpath.parent.stem)}
|
||||
|
||||
# compute similarity between two speaker embeddings
|
||||
# - divides the utterances of each speaker in groups of identical size and embed each group as a speaker embedding
|
||||
spk_embeds_a = np.array([scoring_vocoder.embed_speaker(wavs[:len(wavs) // 2]) for wavs in speaker_wavs.values()])
|
||||
spk_embeds_b = np.array([scoring_vocoder.embed_speaker(wavs[len(wavs) // 2:]) for wavs in speaker_wavs.values()])
|
||||
spk_sim_matrix = np.inner(spk_embeds_a, spk_embeds_b)
|
||||
|
||||
sim_score = np.average([spk_sim_matrix[0, 1], spk_sim_matrix[1, 0]])
|
||||
|
||||
return(sim_score)
|
||||
@@ -0,0 +1,77 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import numpy as np
|
||||
|
||||
# load coqui-ai/TTS libraries
|
||||
from TTS.config import load_config
|
||||
from TTS.tts.models import setup_model as setup_tts_model
|
||||
from TTS.tts.utils.synthesis import synthesis, trim_silence
|
||||
|
||||
|
||||
def init_synth(config_path, voice_model_path, speakers_file_path = None, speaker_embeddings_file = None, use_cuda = True, use_phonemes = False):
|
||||
|
||||
# load config and customise config parameters (those that are different during training and inference / synthesizing)
|
||||
config = load_config(config_path)
|
||||
|
||||
if not speakers_file_path is None:
|
||||
config.use_speaker_embedding = True,
|
||||
config.use_d_vector_file = False,
|
||||
config.speakers_file = speakers_file_path
|
||||
config.model_args["use_speaker_embedding"] = True,
|
||||
config.model_args["use_d_vector_file"] = False,
|
||||
config.model_args["speakers_file"] = speakers_file_path
|
||||
else:
|
||||
config.d_vector_file = speaker_embeddings_file
|
||||
config.model_args["d_vector_file"] = speaker_embeddings_file
|
||||
|
||||
# set whether or not phonemes are used
|
||||
config.use_phonemes = use_phonemes
|
||||
|
||||
# load cloned voice model
|
||||
model = setup_tts_model(config = config)
|
||||
model.load_checkpoint(config, voice_model_path, eval = True)
|
||||
|
||||
if use_cuda:
|
||||
model.cuda()
|
||||
|
||||
return config, model
|
||||
|
||||
|
||||
def synthesize(config, voice_model, txt, speaker_embeddings = None, speaker_id = None, speech_sample_wav = None, speech_sample_txt = None, use_cuda = True, trim_silence = True):
|
||||
|
||||
# disable language selection
|
||||
#language_id = 0
|
||||
language_id = None
|
||||
|
||||
# set default voice encoder
|
||||
use_gl = True
|
||||
|
||||
# synthesize voice
|
||||
outputs = synthesis(
|
||||
model = voice_model,
|
||||
text = txt,
|
||||
CONFIG = config,
|
||||
use_cuda = use_cuda,
|
||||
speaker_id = speaker_id,
|
||||
style_wav = speech_sample_wav,
|
||||
style_text = speech_sample_txt,
|
||||
use_griffin_lim = use_gl,
|
||||
do_trim_silence = trim_silence,
|
||||
d_vector = speaker_embeddings,
|
||||
language_id = language_id
|
||||
)
|
||||
|
||||
waveform = outputs["wav"]
|
||||
waveform = waveform.squeeze()
|
||||
|
||||
# trim silence (disabled due to some "TypeError: 'bool' object is not callable" bug that needs to be investigated)
|
||||
#if (config.audio["do_trim_silence"]) or (trim_silence):
|
||||
# waveform = trim_silence(waveform, voice_model.ap)
|
||||
|
||||
return waveform
|
||||
|
||||
|
||||
def save_waveform(config, voice_model, waveform, out_path):
|
||||
wav = np.array(waveform)
|
||||
voice_model.ap.save_wav(wav, out_path, config["audio"].sample_rate)
|
||||
@@ -0,0 +1,94 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import sys
|
||||
import os
|
||||
|
||||
import requests
|
||||
from requests.structures import CaseInsensitiveDict
|
||||
import json
|
||||
|
||||
from time import sleep
|
||||
|
||||
|
||||
# set transcription service API endpoint and token (retrieved from operating system's ENV variables)
|
||||
# + sample endpoints:
|
||||
# - [dev] "https://development.sendpotion.com/api/transcript"
|
||||
# - [staging] "https://staging.sendpotion.com/api/transcript"
|
||||
API_ENDPOINT = os.environ.get("TRANSCRIPTION_API_ENDPOINT")
|
||||
API_TOKEN = os.environ.get("TRANSCRIPTION_API_TOKEN")
|
||||
|
||||
|
||||
#
|
||||
# Using potions internal transcription API endpoint, get a transcription for a given (wav) audio recording
|
||||
# + returns a triple:
|
||||
# - Boolean ......... indicating success (True) or failure (False)
|
||||
# - String / None ... transcription text (or None in failure case)
|
||||
# - Float / None .... transcription confidence score (or None in failure case)
|
||||
#
|
||||
def get_transcription(wav_fname):
|
||||
|
||||
# validate that transcription service API endpoint and token are set
|
||||
if (API_ENDPOINT is None) or (API_TOKEN is None):
|
||||
# terminate
|
||||
print("TRANSCRIPTION_API_ENDPOINT and TRANSCRIPTION_API_TOKEN environment variables MUST be set!")
|
||||
sys.exit(1)
|
||||
|
||||
# set request header to contain (bearer) API token
|
||||
headers = CaseInsensitiveDict()
|
||||
headers["Accept"] = "application/json"
|
||||
headers["Authorization"] = "Bearer " + str(API_TOKEN)
|
||||
|
||||
# set files field (data is empty)
|
||||
files = {'wav': open(wav_fname, 'rb')}
|
||||
|
||||
# issue POST request and save response as response object
|
||||
response = requests.post(url = API_ENDPOINT, headers = headers, files = files)
|
||||
|
||||
# test for auth error
|
||||
# test for timeout
|
||||
|
||||
# check if the status code is not an error code (i.e., 4xx or 5xx)
|
||||
success = False
|
||||
if response:
|
||||
# extracting response text
|
||||
response_text = response.text
|
||||
response_json = json.loads(response_text)
|
||||
#print(response_json)
|
||||
|
||||
if response.ok: # synch call
|
||||
success = True
|
||||
trans_text = response_json["transcriptObj"]["text"]
|
||||
trans_score = float(response_json["transcriptObj"]["confidence"])
|
||||
else: # fallback to asynch call
|
||||
# wait up to 60 seconds for the transcription to be ready; try every 5 seconds
|
||||
wait = 0
|
||||
|
||||
while wait < 60:
|
||||
sleep(5)
|
||||
wait += 5
|
||||
|
||||
# issue GET request using the previously returned reqiestId and save response as response object
|
||||
response_get = requests.get(url = API_ENDPOINT + ':' + response_json["requestId"])
|
||||
|
||||
# check if the status code is not an error code (i.e., 4xx or 5xx)
|
||||
if response_get.ok:
|
||||
response_get_text = response_get.text
|
||||
response_get_json = json.loads(response_get_text)
|
||||
|
||||
success = True
|
||||
trans_text = response_get_json["transcriptObj"]["text"]
|
||||
trans_score = float(response_get_json["transcriptObj"]["confidence"])
|
||||
break
|
||||
|
||||
# in case no successful response is received even after a 60 seconds waiting period -> proceed without transcription
|
||||
#if not response_get.ok:
|
||||
# print("Response: FAILED.")
|
||||
|
||||
#else:
|
||||
# print("ERROR: {} ({})" . format(response.status_code, response.text))
|
||||
|
||||
if success:
|
||||
return success, trans_text, trans_score
|
||||
else:
|
||||
return False, None, None
|
||||
Reference in New Issue
Block a user