Files
project-work/worker-toolkit-potion-polyglot/repos/potion-voice/voice-cloning/utils/transcription_utils.py
2026-10-05 16:14:53 -04:00

95 lines
3.4 KiB
Python

#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import sys
import os
import requests
from requests.structures import CaseInsensitiveDict
import json
from time import sleep
# set transcription service API endpoint and token (retrieved from operating system's ENV variables)
# + sample endpoints:
# - [dev] "https://development.sendpotion.com/api/transcript"
# - [staging] "https://staging.sendpotion.com/api/transcript"
API_ENDPOINT = os.environ.get("TRANSCRIPTION_API_ENDPOINT")
API_TOKEN = os.environ.get("TRANSCRIPTION_API_TOKEN")
#
# Using potions internal transcription API endpoint, get a transcription for a given (wav) audio recording
# + returns a triple:
# - Boolean ......... indicating success (True) or failure (False)
# - String / None ... transcription text (or None in failure case)
# - Float / None .... transcription confidence score (or None in failure case)
#
def get_transcription(wav_fname):
# validate that transcription service API endpoint and token are set
if (API_ENDPOINT is None) or (API_TOKEN is None):
# terminate
print("TRANSCRIPTION_API_ENDPOINT and TRANSCRIPTION_API_TOKEN environment variables MUST be set!")
sys.exit(1)
# set request header to contain (bearer) API token
headers = CaseInsensitiveDict()
headers["Accept"] = "application/json"
headers["Authorization"] = "Bearer " + str(API_TOKEN)
# set files field (data is empty)
files = {'wav': open(wav_fname, 'rb')}
# issue POST request and save response as response object
response = requests.post(url = API_ENDPOINT, headers = headers, files = files)
# test for auth error
# test for timeout
# check if the status code is not an error code (i.e., 4xx or 5xx)
success = False
if response:
# extracting response text
response_text = response.text
response_json = json.loads(response_text)
#print(response_json)
if response.ok: # synch call
success = True
trans_text = response_json["transcriptObj"]["text"]
trans_score = float(response_json["transcriptObj"]["confidence"])
else: # fallback to asynch call
# wait up to 60 seconds for the transcription to be ready; try every 5 seconds
wait = 0
while wait < 60:
sleep(5)
wait += 5
# issue GET request using the previously returned reqiestId and save response as response object
response_get = requests.get(url = API_ENDPOINT + ':' + response_json["requestId"])
# check if the status code is not an error code (i.e., 4xx or 5xx)
if response_get.ok:
response_get_text = response_get.text
response_get_json = json.loads(response_get_text)
success = True
trans_text = response_get_json["transcriptObj"]["text"]
trans_score = float(response_get_json["transcriptObj"]["confidence"])
break
# in case no successful response is received even after a 60 seconds waiting period -> proceed without transcription
#if not response_get.ok:
# print("Response: FAILED.")
#else:
# print("ERROR: {} ({})" . format(response.status_code, response.text))
if success:
return success, trans_text, trans_score
else:
return False, None, None