before changes

This commit is contained in:
2026-10-05 16:14:53 -04:00
parent dd6e6f7cfd
commit f39555ce15
94 changed files with 3285321 additions and 1 deletions

View File

@@ -0,0 +1,22 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import textdistance
#
# Name matching via textual similarity search
# + Returns two (normalised) distance measures: the Jaro-Winkler Distance and the regular Levenshtein Distance
#
def match_name_textualsim(name1, name2):
jaro_winkler = textdistance.jaro_winkler.normalized_similarity(name1, name2)
levenshtein = textdistance.levenshtein.normalized_similarity(name1, name2)
return jaro_winkler, levenshtein
#
# Name matching via phonetic matching algorithm (using the normalised Match Rating Approach)
#
def match_name_mra(name1, name2):
return textdistance.mra.normalized_similarity(name1, name2)

View File

@@ -0,0 +1,37 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
from pathlib import Path
from itertools import groupby
import numpy as np
from resemblyzer import preprocess_wav, VoiceEncoder
def init_scoring_vocoder():
# initialise voice encoder (using CUDA by default; CPU as fallback)
encoder = VoiceEncoder()
return encoder
def score_speaker_similarity(scoring_vocoder, spk_a_fpaths, spk_b_fpaths):
# filepaths to waveforms
wav_fpaths = list(Path(spk_a_fpaths).glob("*.wav")) + list(Path(spk_b_fpaths).glob("*.wav"))
# group the wavs per speaker and load them using the preprocessing function provided with Resemblyzer to load wavs in memory
# - normalizes the volume, trims long silences and resamples the wav to the correct sampling rate
speaker_wavs = {speaker: list(map(preprocess_wav, wav_fpaths)) for speaker, wav_fpaths in groupby(wav_fpaths, lambda wav_fpath: wav_fpath.parent.stem)}
# compute similarity between two speaker embeddings
# - divides the utterances of each speaker in groups of identical size and embed each group as a speaker embedding
spk_embeds_a = np.array([scoring_vocoder.embed_speaker(wavs[:len(wavs) // 2]) for wavs in speaker_wavs.values()])
spk_embeds_b = np.array([scoring_vocoder.embed_speaker(wavs[len(wavs) // 2:]) for wavs in speaker_wavs.values()])
spk_sim_matrix = np.inner(spk_embeds_a, spk_embeds_b)
sim_score = np.average([spk_sim_matrix[0, 1], spk_sim_matrix[1, 0]])
return(sim_score)

View File

@@ -0,0 +1,77 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import numpy as np
# load coqui-ai/TTS libraries
from TTS.config import load_config
from TTS.tts.models import setup_model as setup_tts_model
from TTS.tts.utils.synthesis import synthesis, trim_silence
def init_synth(config_path, voice_model_path, speakers_file_path = None, speaker_embeddings_file = None, use_cuda = True, use_phonemes = False):
# load config and customise config parameters (those that are different during training and inference / synthesizing)
config = load_config(config_path)
if not speakers_file_path is None:
config.use_speaker_embedding = True,
config.use_d_vector_file = False,
config.speakers_file = speakers_file_path
config.model_args["use_speaker_embedding"] = True,
config.model_args["use_d_vector_file"] = False,
config.model_args["speakers_file"] = speakers_file_path
else:
config.d_vector_file = speaker_embeddings_file
config.model_args["d_vector_file"] = speaker_embeddings_file
# set whether or not phonemes are used
config.use_phonemes = use_phonemes
# load cloned voice model
model = setup_tts_model(config = config)
model.load_checkpoint(config, voice_model_path, eval = True)
if use_cuda:
model.cuda()
return config, model
def synthesize(config, voice_model, txt, speaker_embeddings = None, speaker_id = None, speech_sample_wav = None, speech_sample_txt = None, use_cuda = True, trim_silence = True):
# disable language selection
#language_id = 0
language_id = None
# set default voice encoder
use_gl = True
# synthesize voice
outputs = synthesis(
model = voice_model,
text = txt,
CONFIG = config,
use_cuda = use_cuda,
speaker_id = speaker_id,
style_wav = speech_sample_wav,
style_text = speech_sample_txt,
use_griffin_lim = use_gl,
do_trim_silence = trim_silence,
d_vector = speaker_embeddings,
language_id = language_id
)
waveform = outputs["wav"]
waveform = waveform.squeeze()
# trim silence (disabled due to some "TypeError: 'bool' object is not callable" bug that needs to be investigated)
#if (config.audio["do_trim_silence"]) or (trim_silence):
# waveform = trim_silence(waveform, voice_model.ap)
return waveform
def save_waveform(config, voice_model, waveform, out_path):
wav = np.array(waveform)
voice_model.ap.save_wav(wav, out_path, config["audio"].sample_rate)

View File

@@ -0,0 +1,94 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import sys
import os
import requests
from requests.structures import CaseInsensitiveDict
import json
from time import sleep
# set transcription service API endpoint and token (retrieved from operating system's ENV variables)
# + sample endpoints:
# - [dev] "https://development.sendpotion.com/api/transcript"
# - [staging] "https://staging.sendpotion.com/api/transcript"
API_ENDPOINT = os.environ.get("TRANSCRIPTION_API_ENDPOINT")
API_TOKEN = os.environ.get("TRANSCRIPTION_API_TOKEN")
#
# Using potions internal transcription API endpoint, get a transcription for a given (wav) audio recording
# + returns a triple:
# - Boolean ......... indicating success (True) or failure (False)
# - String / None ... transcription text (or None in failure case)
# - Float / None .... transcription confidence score (or None in failure case)
#
def get_transcription(wav_fname):
# validate that transcription service API endpoint and token are set
if (API_ENDPOINT is None) or (API_TOKEN is None):
# terminate
print("TRANSCRIPTION_API_ENDPOINT and TRANSCRIPTION_API_TOKEN environment variables MUST be set!")
sys.exit(1)
# set request header to contain (bearer) API token
headers = CaseInsensitiveDict()
headers["Accept"] = "application/json"
headers["Authorization"] = "Bearer " + str(API_TOKEN)
# set files field (data is empty)
files = {'wav': open(wav_fname, 'rb')}
# issue POST request and save response as response object
response = requests.post(url = API_ENDPOINT, headers = headers, files = files)
# test for auth error
# test for timeout
# check if the status code is not an error code (i.e., 4xx or 5xx)
success = False
if response:
# extracting response text
response_text = response.text
response_json = json.loads(response_text)
#print(response_json)
if response.ok: # synch call
success = True
trans_text = response_json["transcriptObj"]["text"]
trans_score = float(response_json["transcriptObj"]["confidence"])
else: # fallback to asynch call
# wait up to 60 seconds for the transcription to be ready; try every 5 seconds
wait = 0
while wait < 60:
sleep(5)
wait += 5
# issue GET request using the previously returned reqiestId and save response as response object
response_get = requests.get(url = API_ENDPOINT + ':' + response_json["requestId"])
# check if the status code is not an error code (i.e., 4xx or 5xx)
if response_get.ok:
response_get_text = response_get.text
response_get_json = json.loads(response_get_text)
success = True
trans_text = response_get_json["transcriptObj"]["text"]
trans_score = float(response_get_json["transcriptObj"]["confidence"])
break
# in case no successful response is received even after a 60 seconds waiting period -> proceed without transcription
#if not response_get.ok:
# print("Response: FAILED.")
#else:
# print("ERROR: {} ({})" . format(response.status_code, response.text))
if success:
return success, trans_text, trans_score
else:
return False, None, None