fixed .gitignore, regrades

This commit is contained in:
2026-09-25 20:08:25 -04:00
parent fe1f0de788
commit 2aa9ab7ff4
492 changed files with 3327982 additions and 2 deletions

View File

@@ -0,0 +1,165 @@
# Byte-compiled / optimized / DLL files
__pycache__/
*.py[cod]
*$py.class
# C extensions
*.so
# Distribution / packaging
.Python
build/
develop-eggs/
dist/
downloads/
eggs/
.eggs/
lib/
lib64/
parts/
sdist/
var/
wheels/
pip-wheel-metadata/
share/python-wheels/
*.egg-info/
.installed.cfg
*.egg
MANIFEST
# PyInstaller
# Usually these files are written by a python script from a template
# before PyInstaller builds the exe, so as to inject date/other infos into it.
*.manifest
*.spec
# Installer logs
pip-log.txt
pip-delete-this-directory.txt
# Unit test / coverage reports
htmlcov/
.tox/
.nox/
.coverage
.coverage.*
.cache
nosetests.xml
coverage.xml
*.cover
*.py,cover
.hypothesis/
.pytest_cache/
# Translations
*.mo
*.pot
# Django stuff:
*.log
local_settings.py
db.sqlite3
db.sqlite3-journal
# Flask stuff:
instance/
.webassets-cache
# Scrapy stuff:
.scrapy
# Sphinx documentation
docs/_build/
# PyBuilder
target/
# Jupyter Notebook
.ipynb_checkpoints
# IPython
profile_default/
ipython_config.py
# pyenv
.python-version
# pipenv
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
# However, in case of collaboration, if having platform-specific dependencies or dependencies
# having no cross-platform support, pipenv may install dependencies that don't work, or not
# install all needed dependencies.
#Pipfile.lock
# PEP 582; used by e.g. github.com/David-OConnor/pyflow
__pypackages__/
# Celery stuff
celerybeat-schedule
celerybeat.pid
# SageMath parsed files
*.sage.py
# Environments
.env
.venv
env/
venv/
ENV/
env.bak/
venv.bak/
# Spyder project settings
.spyderproject
.spyproject
# Rope project settings
.ropeproject
# mkdocs documentation
/site
# mypy
.mypy_cache/
.dmypy.json
dmypy.json
# Pyre type checker
.pyre/
# potion-voice specific exclusions
voice-cloning/TTS/
voice-cloning/Trainer/
voice-cloning/temp/
voice-cloning/results/
voice-cloning/output/
voice-cloning/pretrained-models/
*.pkl
*.jpg
*.mp4
*.pth
*.pyc
__pycache__
*.h5
*.avi
*.wav
filelists/*.txt
evaluation/test_filelists/lr*.txt
*.pyc
*.mkv
*.gif
*.webm
*.mp3
node_modules
build-staging-ai/
build-production-ai/
.env.production.aws-code-deploy
.env.staging.aws-code-deploy
env-aws-code-deploy/
**/poc.js
**/yarn.lock
.prettierrc
**/package-lock.json

View File

@@ -0,0 +1,150 @@
{
"number": 1,
"title": "Initial commit",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/1",
"createdAt": "2022-04-21T13:53:32Z",
"mergedAt": "2022-04-21T13:53:54Z",
"closedAt": "2022-04-21T13:53:54Z",
"additions": 1003,
"deletions": 0,
"changedFiles": 6,
"isDraft": false,
"baseRefName": "main",
"headRefName": "initialCommit",
"author": {
"login": "author_unknown"
},
"mergedBy": {
"login": "author_unknown"
},
"mergeCommit": {
"oid": "367ecad92dbaf831382a9262e017409a4aa7ec38"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 4,
"nodes": [
{
"commit": {
"oid": "da4ff050ff67740bf97fa0292903604dcf4d8452",
"message": "Initial commit of Potion voice repo based on the VITS implementation by coqui-ai/TTS.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-04T16:13:45Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-04T16:13:45Z"
}
}
},
{
"commit": {
"oid": "f6c7d822d009121685f25fbff90f2727884775d4",
"message": "Updated usage documentation",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-07T15:30:00Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-07T15:30:00Z"
}
}
},
{
"commit": {
"oid": "70b01e98c91bf6e808d0bdf849226512220aa72c",
"message": "Updated exclusions.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-21T13:48:34Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-21T13:48:34Z"
}
}
},
{
"commit": {
"oid": "0f4cd72a1960b87471e2865b55f804f62189bccb",
"message": "Corrected exclusions.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-21T13:50:15Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-21T13:50:15Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": ".gitignore",
"additions": 8,
"deletions": 0,
"changeType": "MODIFIED"
},
{
"path": "requirements.txt",
"additions": 8,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/clone_voice.py",
"additions": 212,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/docs/potion-voice-cloning_Installation_Guide.md",
"additions": 437,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/synthesize_speech.py",
"additions": 139,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/train_multispeaker_baseline_model.py",
"additions": 199,
"deletions": 0,
"changeType": "ADDED"
}
]
}
}

View File

@@ -0,0 +1,101 @@
{
"number": 10,
"title": "updated code for using original text",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/10",
"createdAt": "2023-01-10T09:39:58Z",
"mergedAt": "2023-01-10T12:13:20Z",
"closedAt": "2023-01-10T12:13:20Z",
"additions": 2,
"deletions": 2,
"changedFiles": 1,
"isDraft": false,
"baseRefName": "main",
"headRefName": "update-voice-cloning-to-use-original-text",
"author": {
"login": "author_7"
},
"mergedBy": {
"login": "author_6"
},
"mergeCommit": {
"oid": "64766ed9370d9b2da5914b3416dc6ad9271cd156"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 2,
"nodes": [
{
"commit": {
"oid": "2e8d6cabb09f54db2ea37534ff1c0e25f0d87fba",
"message": "Uploaded code for using original text",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-10T09:38:17Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-10T09:38:17Z"
}
}
},
{
"commit": {
"oid": "caa711f0a6d297fa51d1bbafa0f9180356a0baa8",
"message": "Uploaded code for using original text",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-10T09:40:32Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-10T09:40:32Z"
}
}
}
]
},
"reviews": {
"nodes": [
{
"author": {
"login": "author_6"
},
"state": "APPROVED",
"body": "",
"submittedAt": "2023-01-10T12:13:14Z",
"url": "https://github.com/potion/potion-voice/pull/10#pullrequestreview-1242087815",
"comments": {
"nodes": []
}
}
]
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning-job-handler/index.js",
"additions": 2,
"deletions": 2,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,583 @@
{
"number": 11,
"title": "Voice ai v2 changes",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/11",
"createdAt": "2023-02-01T09:10:44Z",
"mergedAt": "2023-02-02T05:32:22Z",
"closedAt": "2023-02-02T05:32:22Z",
"additions": 200,
"deletions": 6819,
"changedFiles": 12,
"isDraft": false,
"baseRefName": "staging",
"headRefName": "voice-ai-v2-changes",
"author": {
"login": "author_6"
},
"mergedBy": {
"login": "author_7"
},
"mergeCommit": {
"oid": "14c3c3630a14ab220a13d4b0ce2ac7a2b9c2ab2d"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 28,
"nodes": [
{
"commit": {
"oid": "09895be273030cf75781b88a43581db15e2b537f",
"message": "Some cleanup",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-12T15:39:34Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-12T15:39:34Z"
}
}
},
{
"commit": {
"oid": "477c39922d6f1f8bf474d6aa215dbf7f745629af",
"message": "Some cleanup",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-12T15:41:46Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-12T15:41:46Z"
}
}
},
{
"commit": {
"oid": "81a3e340d28c15313cf363697ea35e401bd48c30",
"message": "Some cleanup",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-12T15:41:59Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-12T15:41:59Z"
}
}
},
{
"commit": {
"oid": "920ab8ac6ba9b428d30b655a12533e7491c17ad2",
"message": "Added todos",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-13T07:54:31Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-13T07:54:31Z"
}
}
},
{
"commit": {
"oid": "18e04e2e4e9ae7eef5d77d86cb83bf1efea7fbd4",
"message": "Added code for v2 changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-16T17:46:46Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-16T17:46:46Z"
}
}
},
{
"commit": {
"oid": "31996261302205e07e9135750c71ba6c227e81ee",
"message": "update the python command",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-16T18:36:48Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-16T18:36:48Z"
}
}
},
{
"commit": {
"oid": "d6a4ca9809bd3e61bce09d81534c85582a6648c4",
"message": "Added code for v2 changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-17T07:42:04Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-17T07:42:04Z"
}
}
},
{
"commit": {
"oid": "5dbe54bf0674a0323fb237d2549b3bccab8b1bae",
"message": "Added code for v2 changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-17T07:50:16Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-17T07:50:16Z"
}
}
},
{
"commit": {
"oid": "103d47f263421ab090097825883fe33f9cbd1830",
"message": "Added code for v2 changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-17T07:51:23Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-17T07:51:23Z"
}
}
},
{
"commit": {
"oid": "de9256d575777382caee28192d6e7321d4f6cf37",
"message": "Merge branch 'main' into voice-ai-v2-changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T15:30:59Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T15:30:59Z"
}
}
},
{
"commit": {
"oid": "4054eaeabf53f7a1477134496e26d1b323a89211",
"message": "Removed unwanted package",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:24:29Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:24:29Z"
}
}
},
{
"commit": {
"oid": "676ae4419c2c00340a6e0ddbc2ebedaf547b984b",
"message": "updated zip command",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:31:31Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:31:31Z"
}
}
},
{
"commit": {
"oid": "a1d7a29e837f4c508245ed6158c07accfe1ab550",
"message": "Updated path",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:47:26Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:47:26Z"
}
}
},
{
"commit": {
"oid": "9eac7f855681a903b6372d74a0252ffa3f4f6ceb",
"message": "Updated zippath",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:59:07Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:59:07Z"
}
}
},
{
"commit": {
"oid": "6f33b7b4c8d3b14e323a5e87852216e35da8806d",
"message": "Updated zippath",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:03:24Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:03:24Z"
}
}
},
{
"commit": {
"oid": "a4a22adba17913568643f56a7803b743055dde97",
"message": "Updated zippath",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:06:13Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:06:13Z"
}
}
},
{
"commit": {
"oid": "b1222eea17760fbf5c9cf6007637a6f01e7deab7",
"message": "Updated path for result",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:09:05Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:09:05Z"
}
}
},
{
"commit": {
"oid": "60193204e3783f20e156000ef48a2b9386002d88",
"message": "updated command",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:25:27Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:25:27Z"
}
}
},
{
"commit": {
"oid": "7960ed5dfa9afce6281350870977aa48f4f903d2",
"message": "updated command",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:31:20Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:31:20Z"
}
}
},
{
"commit": {
"oid": "633ecbafcda44c57d24ab4280aec53e9453c93ba",
"message": "Added changes for the synthesize command",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T21:57:29Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T21:57:29Z"
}
}
},
{
"commit": {
"oid": "6a0fdc17bb6c19d20338c1a6cae2bbea8f87dcb4",
"message": "updated voice-ai-changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-22T18:43:39Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-22T18:43:39Z"
}
}
},
{
"commit": {
"oid": "fd70943a1ca541e00852e7384f91c7b09c42d9d3",
"message": "Updated speaker path",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T08:52:25Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T08:52:25Z"
}
}
},
{
"commit": {
"oid": "0617f15153fdffd52f009bb0429d6878743649dc",
"message": "Updated speaker path",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T10:01:08Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T10:01:08Z"
}
}
},
{
"commit": {
"oid": "0e83ea6cd962df73dccf012582fcd31277e9398e",
"message": "Updated code for synthesis",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T11:16:01Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T11:16:01Z"
}
}
},
{
"commit": {
"oid": "f996adf6c1277fd78253badeb77d487c5ad1d328",
"message": "Updated code for synthesis",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T11:24:45Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T11:24:45Z"
}
}
},
{
"commit": {
"oid": "2aea05da7bfa3c057f3a31ed16639e461a395f53",
"message": "Updated job code",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T11:36:52Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T11:36:52Z"
}
}
},
{
"commit": {
"oid": "6a234309474cbc1e420fd092fefed97ec2c75aae",
"message": "Updated job code",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T14:24:36Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T14:24:36Z"
}
}
},
{
"commit": {
"oid": "a616fd33178b70105d5cb40694e74e2c1df124da",
"message": "Added condition for delete check",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T05:20:17Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T05:20:17Z"
}
}
}
]
},
"reviews": {
"nodes": [
{
"author": {
"login": "author_7"
},
"state": "APPROVED",
"body": "",
"submittedAt": "2023-02-02T05:32:09Z",
"url": "https://github.com/potion/potion-voice/pull/11#pullrequestreview-1280363999",
"comments": {
"nodes": []
}
}
]
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": ".gitignore",
"additions": 3,
"deletions": 0,
"changeType": "MODIFIED"
},
{
"path": ".prettierrc",
"additions": 0,
"deletions": 7,
"changeType": "REMOVED"
},
{
"path": "voice-cloning-job-handler/index.js",
"additions": 56,
"deletions": 11,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/package-lock.json",
"additions": 0,
"deletions": 1782,
"changeType": "REMOVED"
},
{
"path": "voice-cloning-job-handler/package.json",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/user_audio_profile/user_audio_profile_service.js",
"additions": 8,
"deletions": 10,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/voice_cloning/voice_cloning_service.js",
"additions": 0,
"deletions": 2,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/yarn.lock",
"additions": 0,
"deletions": 2475,
"changeType": "REMOVED"
},
{
"path": "voice-synthsizer-job-handler/index.js",
"additions": 53,
"deletions": 56,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/recording_salutation/index.js",
"additions": 3,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/recording_salutation/recording_salutation_model.js",
"additions": 76,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "yarn.lock",
"additions": 0,
"deletions": 2475,
"changeType": "REMOVED"
}
]
}
}

View File

@@ -0,0 +1,129 @@
{
"number": 12,
"title": "Ai 490 adv synth",
"body": "Added:\r\n+ optional speech sample waveform and text parameter support for style transfer\r\n+ upsampling of synthetic speech output to target sampling rate (48kHz by default)\r\n\r\nUpdated:\r\n+ usage documentation",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/12",
"createdAt": "2023-02-01T17:09:41Z",
"mergedAt": "2023-02-02T05:52:09Z",
"closedAt": "2023-02-02T05:52:09Z",
"additions": 42,
"deletions": 21,
"changedFiles": 3,
"isDraft": false,
"baseRefName": "staging",
"headRefName": "ai-490-adv-synth",
"author": {
"login": "author_unknown"
},
"mergedBy": {
"login": "author_7"
},
"mergeCommit": {
"oid": "9de3769f0cfe8a81dbccf331702452f2c0112987"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": [
{
"login": "author_6"
}
]
},
"commits": {
"totalCount": 2,
"nodes": [
{
"commit": {
"oid": "6645341cf0150d9c2f3766c885fe8891660e2ac5",
"message": "Synthesising audio with optional speech samples for style transfer; upsampling output to target sampling rate (48kHz as default).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-01T16:51:24Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-01T16:51:24Z"
}
}
},
{
"commit": {
"oid": "b2d1cd59e7bd8a0a1d8a1606664230882e988d15",
"message": "Cleaned up and documented extended synthesising approach.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-01T17:05:34Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-01T17:05:34Z"
}
}
}
]
},
"reviews": {
"nodes": [
{
"author": {
"login": "author_7"
},
"state": "APPROVED",
"body": "",
"submittedAt": "2023-02-02T05:33:12Z",
"url": "https://github.com/potion/potion-voice/pull/12#pullrequestreview-1280364688",
"comments": {
"nodes": []
}
},
{
"author": {
"login": "author_7"
},
"state": "APPROVED",
"body": "",
"submittedAt": "2023-02-02T05:38:09Z",
"url": "https://github.com/potion/potion-voice/pull/12#pullrequestreview-1280367849",
"comments": {
"nodes": []
}
}
]
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning/docs/potion-voice-cloning_Installation_Guide.md",
"additions": 10,
"deletions": 3,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/synthesize_speech.py",
"additions": 23,
"deletions": 8,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/utils/synthesize_utils.py",
"additions": 9,
"deletions": 10,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,76 @@
{
"number": 13,
"title": "Added sr48000 for wave",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/13",
"createdAt": "2023-02-02T05:50:19Z",
"mergedAt": "2023-02-02T05:53:02Z",
"closedAt": "2023-02-02T05:53:02Z",
"additions": 1,
"deletions": 1,
"changedFiles": 1,
"isDraft": false,
"baseRefName": "staging",
"headRefName": "fix-output-for-wav",
"author": {
"login": "author_7"
},
"mergedBy": {
"login": "author_6"
},
"mergeCommit": {
"oid": "e54a3b5cec759c2269cfb3e02c26f9674699a26b"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": [
{
"login": "author_6"
}
]
},
"commits": {
"totalCount": 1,
"nodes": [
{
"commit": {
"oid": "6facd07321ab07dd4bdf2b4decbdd652c24cb442",
"message": "Added sr48000 for wave",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T05:49:36Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T05:49:36Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-synthsizer-job-handler/index.js",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,107 @@
{
"number": 14,
"title": "New feature score model",
"body": "No impact on staging / prod ... just for dev / training.",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/14",
"createdAt": "2023-02-03T04:11:03Z",
"mergedAt": "2023-02-03T10:51:48Z",
"closedAt": "2023-02-03T10:51:48Z",
"additions": 220,
"deletions": 1,
"changedFiles": 2,
"isDraft": false,
"baseRefName": "staging",
"headRefName": "new-feature-score-model",
"author": {
"login": "author_unknown"
},
"mergedBy": {
"login": "author_7"
},
"mergeCommit": {
"oid": "4d32b78bf90cd62384f5a788c1d5e19f61e27007"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 2,
"nodes": [
{
"commit": {
"oid": "25c21387320b085eec8c3a223fcb4ca44d952247",
"message": "Add ffmpeg to system-wide install requirements (synthesize_speech requires this now, but it's missing from the documentation).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T15:17:59Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T15:25:31Z"
}
}
},
{
"commit": {
"oid": "18f64f968a0b75f2b26a11e337dd596413055f5f",
"message": "Added new capability to test and rank a set of multi-speaker models (using Resemblyzer-based voice similarity).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-03T04:08:02Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-03T04:08:02Z"
}
}
}
]
},
"reviews": {
"nodes": [
{
"author": {
"login": "author_7"
},
"state": "APPROVED",
"body": "",
"submittedAt": "2023-02-03T10:51:37Z",
"url": "https://github.com/potion/potion-voice/pull/14#pullrequestreview-1282766813",
"comments": {
"nodes": []
}
}
]
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning/docs/potion-voice-cloning_Installation_Guide.md",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/score_models.py",
"additions": 219,
"deletions": 0,
"changeType": "ADDED"
}
]
}
}

View File

@@ -0,0 +1,417 @@
{
"number": 15,
"title": "New feature score model",
"body": "Improved and documented the new find_best_* model scripts.",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/15",
"createdAt": "2023-02-07T15:46:36Z",
"mergedAt": "2023-04-21T04:50:02Z",
"closedAt": "2023-04-21T04:50:02Z",
"additions": 421,
"deletions": 68,
"changedFiles": 12,
"isDraft": false,
"baseRefName": "staging",
"headRefName": "new-feature-score-model",
"author": {
"login": "author_unknown"
},
"mergedBy": {
"login": "author_7"
},
"mergeCommit": {
"oid": "da6d4e590a2db6dde5c4ab55147d277f8856e05e"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": [
{
"login": "author_6"
},
{
"login": "author_7"
}
]
},
"commits": {
"totalCount": 18,
"nodes": [
{
"commit": {
"oid": "ba8c0bb2527ea447d4d3ac69a44ccb442419f1ce",
"message": "Refined voice cloning to support new capabilities to determine best model; updated recently adde capability to determine best multi-speaker model.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-06T17:33:28Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-06T17:33:28Z"
}
}
},
{
"commit": {
"oid": "bbfa2b24aecd57f0ff35f94815b004de33a81218",
"message": "Added json output option.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-07T09:54:31Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-07T09:54:31Z"
}
}
},
{
"commit": {
"oid": "3d19bbd67f252f521da92142897f4f4bd46bd1b3",
"message": "Added syntax & usage examples for new find_best_* scripts; improved code readability",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-07T15:42:44Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-07T15:42:44Z"
}
}
},
{
"commit": {
"oid": "f452e872f96136c1985f1ab30efc03e25e333ccc",
"message": "Minor bug fix: mispelling of variable corrected.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-13T07:49:08Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-13T07:49:08Z"
}
}
},
{
"commit": {
"oid": "b575d70b7331c56fc1ae1bef5055c20cbf80c450",
"message": "out.cloned_model_path now returns the full path, not just the path to the output folder.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-13T14:48:43Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-13T14:48:43Z"
}
}
},
{
"commit": {
"oid": "a2b8a49be22efb291a3ece06d4098ceb9853210e",
"message": "Added updates for new feature",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T06:45:44Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T06:45:44Z"
}
}
},
{
"commit": {
"oid": "a4e6cb973a5d0be91eac08d6400bbaa543ccb3e1",
"message": "updated env",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T06:50:29Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T06:50:29Z"
}
}
},
{
"commit": {
"oid": "e7eeb7e4e70698fc6ef9353bc9c4898efbb99ecc",
"message": "Added console",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T07:15:11Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T07:15:11Z"
}
}
},
{
"commit": {
"oid": "d965399eb9437e2d8623a4d89ffbfd5a85a3d2b1",
"message": "Bug fix: dataset naming conventions back in sync.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T09:58:10Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T09:58:10Z"
}
}
},
{
"commit": {
"oid": "d342b87cd82ba38bfc5f4ed680d0841754065c17",
"message": "Merge branch 'new-feature-score-model' into new-feature-updates",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T11:55:54Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T11:55:54Z"
}
}
},
{
"commit": {
"oid": "8d2af53303db512e74c497cc17b43f9e7e6d61e8",
"message": "Added changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T13:41:34Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T13:41:34Z"
}
}
},
{
"commit": {
"oid": "2dab89e74c7721718f844c984ca628ac5b95e12c",
"message": "Added path changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T15:11:43Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T15:11:43Z"
}
}
},
{
"commit": {
"oid": "ba03da7512be4d8e094fd8efe867af47eb73216d",
"message": "Added path changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T16:42:02Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T16:42:02Z"
}
}
},
{
"commit": {
"oid": "c04ece4f8c09085ce2db37e4dd1831e89a9adc53",
"message": "Added path changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T17:30:10Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T17:30:10Z"
}
}
},
{
"commit": {
"oid": "224358eb5313bd84653159c0b9285b593d7c543e",
"message": "Added changes to model name",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T18:15:55Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T18:16:01Z"
}
}
},
{
"commit": {
"oid": "bd02cf536700604a600554b9b22d4ce759660763",
"message": "Added changes to model name",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T18:20:20Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T18:20:25Z"
}
}
},
{
"commit": {
"oid": "70ce62652dcd27038eebeaa6aa237f31099850b2",
"message": "Added code for removing speakers",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-03-03T06:19:41Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-03-03T06:19:41Z"
}
}
},
{
"commit": {
"oid": "6f04f67524f32fe63e334373b5aa445703740b9c",
"message": "Merge pull request #17 from potion/new-feature-updates\n\nNew feature updates",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-03-08T19:59:07Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-03-08T19:59:07Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning-job-handler/index.js",
"additions": 53,
"deletions": 28,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/pm2-development.yml",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/pm2-production.yml",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/clone_voice.py",
"additions": 28,
"deletions": 4,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/docs/potion-voice-cloning_Installation_Guide.md",
"additions": 112,
"deletions": 6,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/find_best_cloned_model.py",
"additions": 205,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/find_best_multispeaker_model.py",
"additions": 4,
"deletions": 16,
"changeType": "RENAMED"
},
{
"path": "voice-cloning/prepare_datasets.py",
"additions": 3,
"deletions": 3,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/score_cloned_voice.py",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/train_config.py",
"additions": 6,
"deletions": 6,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/train_multispeaker_baseline_model.py",
"additions": 1,
"deletions": 0,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/index.js",
"additions": 6,
"deletions": 2,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,751 @@
{
"number": 16,
"title": "Staging > Main",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/16",
"createdAt": "2023-02-13T04:47:47Z",
"mergedAt": "2023-02-13T09:49:07Z",
"closedAt": "2023-02-13T09:49:07Z",
"additions": 463,
"deletions": 6842,
"changedFiles": 16,
"isDraft": false,
"baseRefName": "main",
"headRefName": "staging",
"author": {
"login": "author_7"
},
"mergedBy": {
"login": "author_6"
},
"mergeCommit": {
"oid": "b704d803d14b4a61516927e0e348a5cf3cef844e"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 37,
"nodes": [
{
"commit": {
"oid": "09895be273030cf75781b88a43581db15e2b537f",
"message": "Some cleanup",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-12T15:39:34Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-12T15:39:34Z"
}
}
},
{
"commit": {
"oid": "477c39922d6f1f8bf474d6aa215dbf7f745629af",
"message": "Some cleanup",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-12T15:41:46Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-12T15:41:46Z"
}
}
},
{
"commit": {
"oid": "81a3e340d28c15313cf363697ea35e401bd48c30",
"message": "Some cleanup",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-12T15:41:59Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-12T15:41:59Z"
}
}
},
{
"commit": {
"oid": "920ab8ac6ba9b428d30b655a12533e7491c17ad2",
"message": "Added todos",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-13T07:54:31Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-13T07:54:31Z"
}
}
},
{
"commit": {
"oid": "18e04e2e4e9ae7eef5d77d86cb83bf1efea7fbd4",
"message": "Added code for v2 changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-16T17:46:46Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-16T17:46:46Z"
}
}
},
{
"commit": {
"oid": "31996261302205e07e9135750c71ba6c227e81ee",
"message": "update the python command",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-16T18:36:48Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-16T18:36:48Z"
}
}
},
{
"commit": {
"oid": "d6a4ca9809bd3e61bce09d81534c85582a6648c4",
"message": "Added code for v2 changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-17T07:42:04Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-17T07:42:04Z"
}
}
},
{
"commit": {
"oid": "5dbe54bf0674a0323fb237d2549b3bccab8b1bae",
"message": "Added code for v2 changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-17T07:50:16Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-17T07:50:16Z"
}
}
},
{
"commit": {
"oid": "103d47f263421ab090097825883fe33f9cbd1830",
"message": "Added code for v2 changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-17T07:51:23Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-17T07:51:23Z"
}
}
},
{
"commit": {
"oid": "de9256d575777382caee28192d6e7321d4f6cf37",
"message": "Merge branch 'main' into voice-ai-v2-changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T15:30:59Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T15:30:59Z"
}
}
},
{
"commit": {
"oid": "4054eaeabf53f7a1477134496e26d1b323a89211",
"message": "Removed unwanted package",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:24:29Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:24:29Z"
}
}
},
{
"commit": {
"oid": "676ae4419c2c00340a6e0ddbc2ebedaf547b984b",
"message": "updated zip command",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:31:31Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:31:31Z"
}
}
},
{
"commit": {
"oid": "a1d7a29e837f4c508245ed6158c07accfe1ab550",
"message": "Updated path",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:47:26Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:47:26Z"
}
}
},
{
"commit": {
"oid": "9eac7f855681a903b6372d74a0252ffa3f4f6ceb",
"message": "Updated zippath",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:59:07Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T19:59:07Z"
}
}
},
{
"commit": {
"oid": "6f33b7b4c8d3b14e323a5e87852216e35da8806d",
"message": "Updated zippath",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:03:24Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:03:24Z"
}
}
},
{
"commit": {
"oid": "a4a22adba17913568643f56a7803b743055dde97",
"message": "Updated zippath",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:06:13Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:06:13Z"
}
}
},
{
"commit": {
"oid": "b1222eea17760fbf5c9cf6007637a6f01e7deab7",
"message": "Updated path for result",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:09:05Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:09:05Z"
}
}
},
{
"commit": {
"oid": "60193204e3783f20e156000ef48a2b9386002d88",
"message": "updated command",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:25:27Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:25:27Z"
}
}
},
{
"commit": {
"oid": "7960ed5dfa9afce6281350870977aa48f4f903d2",
"message": "updated command",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:31:20Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T20:31:20Z"
}
}
},
{
"commit": {
"oid": "633ecbafcda44c57d24ab4280aec53e9453c93ba",
"message": "Added changes for the synthesize command",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T21:57:29Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-18T21:57:29Z"
}
}
},
{
"commit": {
"oid": "6a0fdc17bb6c19d20338c1a6cae2bbea8f87dcb4",
"message": "updated voice-ai-changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-22T18:43:39Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-22T18:43:39Z"
}
}
},
{
"commit": {
"oid": "fd70943a1ca541e00852e7384f91c7b09c42d9d3",
"message": "Updated speaker path",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T08:52:25Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T08:52:25Z"
}
}
},
{
"commit": {
"oid": "0617f15153fdffd52f009bb0429d6878743649dc",
"message": "Updated speaker path",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T10:01:08Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T10:01:08Z"
}
}
},
{
"commit": {
"oid": "0e83ea6cd962df73dccf012582fcd31277e9398e",
"message": "Updated code for synthesis",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T11:16:01Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T11:16:01Z"
}
}
},
{
"commit": {
"oid": "f996adf6c1277fd78253badeb77d487c5ad1d328",
"message": "Updated code for synthesis",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T11:24:45Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T11:24:45Z"
}
}
},
{
"commit": {
"oid": "2aea05da7bfa3c057f3a31ed16639e461a395f53",
"message": "Updated job code",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T11:36:52Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T11:36:52Z"
}
}
},
{
"commit": {
"oid": "6a234309474cbc1e420fd092fefed97ec2c75aae",
"message": "Updated job code",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T14:24:36Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-01-25T14:24:36Z"
}
}
},
{
"commit": {
"oid": "6645341cf0150d9c2f3766c885fe8891660e2ac5",
"message": "Synthesising audio with optional speech samples for style transfer; upsampling output to target sampling rate (48kHz as default).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-01T16:51:24Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-01T16:51:24Z"
}
}
},
{
"commit": {
"oid": "b2d1cd59e7bd8a0a1d8a1606664230882e988d15",
"message": "Cleaned up and documented extended synthesising approach.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-01T17:05:34Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-01T17:05:34Z"
}
}
},
{
"commit": {
"oid": "a616fd33178b70105d5cb40694e74e2c1df124da",
"message": "Added condition for delete check",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T05:20:17Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T05:20:17Z"
}
}
},
{
"commit": {
"oid": "14c3c3630a14ab220a13d4b0ce2ac7a2b9c2ab2d",
"message": "Merge pull request #11 from potion/voice-ai-v2-changes\n\nVoice ai v2 changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T05:32:22Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T05:32:22Z"
}
}
},
{
"commit": {
"oid": "6facd07321ab07dd4bdf2b4decbdd652c24cb442",
"message": "Added sr48000 for wave",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T05:49:36Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T05:49:36Z"
}
}
},
{
"commit": {
"oid": "9de3769f0cfe8a81dbccf331702452f2c0112987",
"message": "Merge pull request #12 from potion/ai-490-adv-synth\n\nAi 490 adv synth",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T05:52:09Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T05:52:09Z"
}
}
},
{
"commit": {
"oid": "e54a3b5cec759c2269cfb3e02c26f9674699a26b",
"message": "Merge pull request #13 from potion/fix-output-for-wav\n\nAdded sr48000 for wave",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T05:53:01Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T05:53:01Z"
}
}
},
{
"commit": {
"oid": "25c21387320b085eec8c3a223fcb4ca44d952247",
"message": "Add ffmpeg to system-wide install requirements (synthesize_speech requires this now, but it's missing from the documentation).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T15:17:59Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-02T15:25:31Z"
}
}
},
{
"commit": {
"oid": "18f64f968a0b75f2b26a11e337dd596413055f5f",
"message": "Added new capability to test and rank a set of multi-speaker models (using Resemblyzer-based voice similarity).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-03T04:08:02Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-03T04:08:02Z"
}
}
},
{
"commit": {
"oid": "4d32b78bf90cd62384f5a788c1d5e19f61e27007",
"message": "Merge pull request #14 from potion/new-feature-score-model\n\nNew feature score model",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-03T10:51:48Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-03T10:51:48Z"
}
}
}
]
},
"reviews": {
"nodes": [
{
"author": {
"login": "author_6"
},
"state": "APPROVED",
"body": "",
"submittedAt": "2023-02-13T09:48:42Z",
"url": "https://github.com/potion/potion-voice/pull/16#pullrequestreview-1295270316",
"comments": {
"nodes": []
}
}
]
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": ".gitignore",
"additions": 3,
"deletions": 0,
"changeType": "MODIFIED"
},
{
"path": ".prettierrc",
"additions": 0,
"deletions": 7,
"changeType": "REMOVED"
},
{
"path": "voice-cloning-job-handler/index.js",
"additions": 56,
"deletions": 11,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/package-lock.json",
"additions": 0,
"deletions": 1782,
"changeType": "REMOVED"
},
{
"path": "voice-cloning-job-handler/package.json",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/user_audio_profile/user_audio_profile_service.js",
"additions": 8,
"deletions": 10,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/voice_cloning/voice_cloning_service.js",
"additions": 0,
"deletions": 2,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/yarn.lock",
"additions": 0,
"deletions": 2475,
"changeType": "REMOVED"
},
{
"path": "voice-cloning/docs/potion-voice-cloning_Installation_Guide.md",
"additions": 11,
"deletions": 4,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/score_models.py",
"additions": 219,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/synthesize_speech.py",
"additions": 23,
"deletions": 8,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/utils/synthesize_utils.py",
"additions": 9,
"deletions": 10,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/index.js",
"additions": 54,
"deletions": 57,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/recording_salutation/index.js",
"additions": 3,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/recording_salutation/recording_salutation_model.js",
"additions": 76,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "yarn.lock",
"additions": 0,
"deletions": 2475,
"changeType": "REMOVED"
}
]
}
}

View File

@@ -0,0 +1,263 @@
{
"number": 17,
"title": "New feature updates",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/17",
"createdAt": "2023-03-03T05:14:09Z",
"mergedAt": "2023-03-08T19:59:08Z",
"closedAt": "2023-03-08T19:59:08Z",
"additions": 61,
"deletions": 32,
"changedFiles": 4,
"isDraft": false,
"baseRefName": "new-feature-score-model",
"headRefName": "new-feature-updates",
"author": {
"login": "author_7"
},
"mergedBy": {
"login": "author_6"
},
"mergeCommit": {
"oid": "6f04f67524f32fe63e334373b5aa445703740b9c"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 11,
"nodes": [
{
"commit": {
"oid": "a2b8a49be22efb291a3ece06d4098ceb9853210e",
"message": "Added updates for new feature",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T06:45:44Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T06:45:44Z"
}
}
},
{
"commit": {
"oid": "a4e6cb973a5d0be91eac08d6400bbaa543ccb3e1",
"message": "updated env",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T06:50:29Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T06:50:29Z"
}
}
},
{
"commit": {
"oid": "e7eeb7e4e70698fc6ef9353bc9c4898efbb99ecc",
"message": "Added console",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T07:15:11Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T07:15:11Z"
}
}
},
{
"commit": {
"oid": "d342b87cd82ba38bfc5f4ed680d0841754065c17",
"message": "Merge branch 'new-feature-score-model' into new-feature-updates",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T11:55:54Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T11:55:54Z"
}
}
},
{
"commit": {
"oid": "8d2af53303db512e74c497cc17b43f9e7e6d61e8",
"message": "Added changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T13:41:34Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T13:41:34Z"
}
}
},
{
"commit": {
"oid": "2dab89e74c7721718f844c984ca628ac5b95e12c",
"message": "Added path changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T15:11:43Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T15:11:43Z"
}
}
},
{
"commit": {
"oid": "ba03da7512be4d8e094fd8efe867af47eb73216d",
"message": "Added path changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T16:42:02Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T16:42:02Z"
}
}
},
{
"commit": {
"oid": "c04ece4f8c09085ce2db37e4dd1831e89a9adc53",
"message": "Added path changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T17:30:10Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T17:30:10Z"
}
}
},
{
"commit": {
"oid": "224358eb5313bd84653159c0b9285b593d7c543e",
"message": "Added changes to model name",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T18:15:55Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T18:16:01Z"
}
}
},
{
"commit": {
"oid": "bd02cf536700604a600554b9b22d4ce759660763",
"message": "Added changes to model name",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T18:20:20Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-14T18:20:25Z"
}
}
},
{
"commit": {
"oid": "70ce62652dcd27038eebeaa6aa237f31099850b2",
"message": "Added code for removing speakers",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-03-03T06:19:41Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-03-03T06:19:41Z"
}
}
}
]
},
"reviews": {
"nodes": [
{
"author": {
"login": "author_6"
},
"state": "APPROVED",
"body": "",
"submittedAt": "2023-03-08T19:59:00Z",
"url": "https://github.com/potion/potion-voice/pull/17#pullrequestreview-1331346663",
"comments": {
"nodes": []
}
}
]
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning-job-handler/index.js",
"additions": 53,
"deletions": 28,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/pm2-development.yml",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/pm2-production.yml",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/index.js",
"additions": 6,
"deletions": 2,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,661 @@
{
"number": 18,
"title": "Refined voice cloning settings",
"body": "Update settings for voice cloning.",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/18",
"createdAt": "2023-04-21T05:10:41Z",
"mergedAt": "2023-06-21T10:58:16Z",
"closedAt": "2023-06-21T10:58:17Z",
"additions": 1152,
"deletions": 146,
"changedFiles": 15,
"isDraft": false,
"baseRefName": "staging",
"headRefName": "develop",
"author": {
"login": "author_unknown"
},
"mergedBy": {
"login": "author_7"
},
"mergeCommit": {
"oid": "54efcabc34b82b156ab06a0beab0f895d4c7edd0"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 32,
"nodes": [
{
"commit": {
"oid": "a47c5e000965b8c38d742a512d88c20d457cfd6a",
"message": "Refined checkpointing and enabled weighted sampler for multi-speaker baseline training.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-03T17:55:31Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-03T17:55:31Z"
}
}
},
{
"commit": {
"oid": "c58120f853b1c7549818d7f7194d6fb901ebbc39",
"message": "Refined checkpointing and enabled weighted sampler for multi-speaker baseline training.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-03T17:55:31Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-03T17:56:49Z"
}
}
},
{
"commit": {
"oid": "c53d9e44c068880953046b1ceab8e6da82b7f60c",
"message": "Merge branch 'develop' of https://github.com/potion/potion-voice into develop",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-03T18:02:10Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-02-03T18:02:10Z"
}
}
},
{
"commit": {
"oid": "237c552fb6dafcf22c2213e69692cd22d6df43e1",
"message": "Merge branch 'staging' into develop",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-04-21T05:13:59Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-04-21T05:13:59Z"
}
}
},
{
"commit": {
"oid": "452d734aefe8407bc0b018fa1974acbe1429dd05",
"message": "Added support for Potion Diverse and Mozilla Common Voice data-sets as well as for continuation of training runs.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-04-27T02:37:20Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-04-27T02:37:20Z"
}
}
},
{
"commit": {
"oid": "c811784a9e37b78698f913bc63c76868e14b902f",
"message": "Merge conflict resolved (naming convention diversion addressed.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-04-27T02:45:15Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-04-27T02:45:15Z"
}
}
},
{
"commit": {
"oid": "940e2d1af624e35dec9447b91495a2908c95c7b5",
"message": "Test sentences added based on dataset arguments; training parameters revised; phoneme usage supported as argument.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-05T15:35:00Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-05T15:35:00Z"
}
}
},
{
"commit": {
"oid": "d6dea65f95cc3a8d8f3e37f62cc022f908a1a936",
"message": "Add loss to config explicitely; use different resblock_type_decoder by default",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-09T07:09:26Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-09T07:09:26Z"
}
}
},
{
"commit": {
"oid": "4a9ddf2b561bf6240d3d03da8671e77b7e7ed78a",
"message": "phoneme support; updated training parameters; removed use_cpu option (unsupported).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-15T07:48:30Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-15T07:48:30Z"
}
}
},
{
"commit": {
"oid": "2563f8dee714cd8b83e72a511c37c7daa4db5c19",
"message": "Remove overwrite of phoneme usage; use setting from config.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-15T16:07:22Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-15T16:07:22Z"
}
}
},
{
"commit": {
"oid": "31dd16cc50d27f434b0afa719f44ad4565a9f17b",
"message": "Added new script to generate a merged speaker embeddings file (for multiple datasets).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-17T16:14:46Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-17T16:14:46Z"
}
}
},
{
"commit": {
"oid": "96680990014cc7c8943fdec303e04299ec5a5914",
"message": "Minor bug fix (argument misspelled).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-17T16:24:30Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-17T16:24:30Z"
}
}
},
{
"commit": {
"oid": "a70ed6dc7f33de96b19390bfa66a2e755e899232",
"message": "Removed unnecessary import; improved comments.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-18T02:27:06Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-18T02:27:06Z"
}
}
},
{
"commit": {
"oid": "f0a7f3cfa96afbb14defc01c2484559d13a9eb2a",
"message": "Added new voice conversion option (idea 1: convert from multi-speaker model; idea 2: clone voice, synthesize speech, convert into recorded template; ...).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-18T15:39:33Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-18T15:39:33Z"
}
}
},
{
"commit": {
"oid": "abb23d49febcb47434f8d5e4db467f86bdd242cb",
"message": "Replaces hard-coded spk embedding reference; commented code more thoroughly.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-18T16:45:58Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-18T16:45:58Z"
}
}
},
{
"commit": {
"oid": "2ed529263af58b0c8374306bd530ab58892df2e7",
"message": "Added new voice cloning approach (using fine-tuning via continue_path).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-19T08:53:49Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-19T08:53:49Z"
}
}
},
{
"commit": {
"oid": "b3d7d633595b26c75c34c6e876685ae0e3008977",
"message": "Spelling and formatting improvements.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-19T08:54:37Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-19T08:54:37Z"
}
}
},
{
"commit": {
"oid": "7836c1cbf74f408d76bd2777ff38dc116b596951",
"message": "Revised training parameters after testing (no freeze and unique phoneme cache per run).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-19T10:10:03Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-19T10:10:03Z"
}
}
},
{
"commit": {
"oid": "3f80bdcb2a4efdb4c7086c8e4985eca28d0262c8",
"message": "Fixed typo",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-19T10:13:20Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-19T10:13:20Z"
}
}
},
{
"commit": {
"oid": "d59d33fb0190c1b42148804e112e8c3578544fbb",
"message": "Explicitly name phonemizer (i.e., espeak).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-22T02:25:36Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-22T02:25:36Z"
}
}
},
{
"commit": {
"oid": "913e45d5b8d16859d74d1ae327835473ee7d4d8c",
"message": "Remove scaling factors for now; improved conversion settings.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-22T14:57:32Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-22T14:57:32Z"
}
}
},
{
"commit": {
"oid": "320b26f304a147b511d015487de6dcc62d4d17d9",
"message": "Added numpy import for assert statements.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-22T15:03:21Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-22T15:03:21Z"
}
}
},
{
"commit": {
"oid": "4ad5caa4edc3fd79c4fddc52f91fd7b8d67902ef",
"message": "Fixed mistake in target_auddio_embedding computation.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-23T07:48:57Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-23T07:48:57Z"
}
}
},
{
"commit": {
"oid": "2bf1db90a774abebce8f68cb92428807a0fd2c33",
"message": "Added clean-up steps to remove temporary files.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-23T15:01:35Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-23T15:01:35Z"
}
}
},
{
"commit": {
"oid": "5ae957485632e7e49295104ff9c7095ab3ec11d2",
"message": "Fixed typo.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-24T02:44:19Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-24T02:44:19Z"
}
}
},
{
"commit": {
"oid": "87aeefc7689f0cb76a0cc8455178015be11fb04e",
"message": "Added (but left them commented out) additional configuration options used in the YourTTS paper.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-24T02:45:21Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-24T02:45:21Z"
}
}
},
{
"commit": {
"oid": "0b29c3b698a063048762c42c90b6f653371e9135",
"message": "Added scale related synth parameters (to be tested / refined).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-24T02:56:14Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-24T02:56:14Z"
}
}
},
{
"commit": {
"oid": "e23cc5d5abb6d6f87d55606088e05edd6e15021f",
"message": "Added sampiling rate as argument (in preparation for higher quality training runs).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-24T15:02:21Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-24T15:02:21Z"
}
}
},
{
"commit": {
"oid": "1e036850c9c3514acfce3ab549ad1a7953b37ca2",
"message": "Added ffmpeg-normalize and require resemblyzer>=1.3.0.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-25T16:57:17Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-25T16:57:17Z"
}
}
},
{
"commit": {
"oid": "add8aec51cb065024e3658ba3abd24cddb3af0e4",
"message": "New cloning step and deployment guidance added; TTS requirements updated.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-25T16:58:39Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-05-25T16:58:39Z"
}
}
},
{
"commit": {
"oid": "397d59ea1071383b5cf94b584028f5040d83f951",
"message": "Fixed typo in requirements files.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-21T10:03:06Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-21T10:03:06Z"
}
}
},
{
"commit": {
"oid": "08d6314e0270b92a1dcab69b30ba57313badd7ea",
"message": "TTS repo assumes config.json name is fixed ... it's hardcoded; so, keep it as config.json.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-21T10:22:39Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-21T10:22:39Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": [
{
"author": {
"login": "author_unknown"
},
"body": "@author_7 @author_6 This is ready to be merged into staging!",
"createdAt": "2023-05-25T17:05:27Z",
"url": "https://github.com/potion/potion-voice/pull/18#issuecomment-1563235903"
}
]
},
"files": {
"nodes": [
{
"path": "requirements.dev.local.txt",
"additions": 2,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "requirements.dev.txt",
"additions": 2,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "requirements.prod.cpu.txt",
"additions": 2,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "requirements.prod.gpu.txt",
"additions": 2,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "requirements.txt",
"additions": 2,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/clone_voice.py",
"additions": 17,
"deletions": 22,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/clone_voice_via_continue.py",
"additions": 278,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/convert_voice.py",
"additions": 308,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/docs/potion-voice-cloning_Installation_Guide.md",
"additions": 153,
"deletions": 4,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/prepare_datasets.py",
"additions": 84,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/save_multispeaker_baseline_embeddings_file.py",
"additions": 107,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/synthesize_speech.py",
"additions": 5,
"deletions": 0,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/train_config.py",
"additions": 21,
"deletions": 2,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/train_multispeaker_baseline_model.py",
"additions": 168,
"deletions": 108,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/utils/synthesize_utils.py",
"additions": 1,
"deletions": 4,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,234 @@
{
"number": 19,
"title": "Update voice clone 23 05",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/19",
"createdAt": "2023-06-21T10:59:07Z",
"mergedAt": "2023-06-27T04:50:28Z",
"closedAt": "2023-06-27T04:50:29Z",
"additions": 182,
"deletions": 91,
"changedFiles": 4,
"isDraft": false,
"baseRefName": "staging",
"headRefName": "update-voice-clone-23-05",
"author": {
"login": "author_7"
},
"mergedBy": {
"login": "author_7"
},
"mergeCommit": {
"oid": "6d89d4f3c84b1f743d93d3b1f6cf70472e1866e5"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 10,
"nodes": [
{
"commit": {
"oid": "94b4cbeb8e5863366512cbcc413a4a99817bef3d",
"message": "Changed version of python package",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-03T14:11:56Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-03T14:11:56Z"
}
}
},
{
"commit": {
"oid": "b254ec4c2842e34c4fb8a807655330916114ea45",
"message": "Added code for new changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-03T14:30:27Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-03T14:30:27Z"
}
}
},
{
"commit": {
"oid": "d95cae5242691b67688cad0077d0eef1ccf89fb2",
"message": "Added changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-03T15:00:59Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-03T15:00:59Z"
}
}
},
{
"commit": {
"oid": "0d7dc5354f866e849645692cd8af93ddd47a624c",
"message": "Added changes for synthesizer",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-06T18:02:16Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-06T18:02:16Z"
}
}
},
{
"commit": {
"oid": "219835c25c561b6d6f964c3ac533e2481b483c94",
"message": "Added change",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-06T18:10:09Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-06T18:10:09Z"
}
}
},
{
"commit": {
"oid": "58d0c83a2b2525967598d0915d374937e7dfd416",
"message": "Added path change",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-06T18:19:08Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-06T18:19:08Z"
}
}
},
{
"commit": {
"oid": "5cb69b8e41743b1b5f7f66e65ea15ff64a0ea945",
"message": "Added changes for not to save the salutation to global",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T03:51:40Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T03:51:40Z"
}
}
},
{
"commit": {
"oid": "58f4b9346d1918f2f83ae249035c8e93d326046c",
"message": "Merge branch 'staging' into update-voice-clone-23-05",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T03:52:23Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T03:52:23Z"
}
}
},
{
"commit": {
"oid": "8aeccc682363a55001fbdb9ec67c861cfb6a8b63",
"message": "Added code for deleting directory",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T04:15:47Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T04:15:47Z"
}
}
},
{
"commit": {
"oid": "e5f6efbb72902997353580ee56b7893c64452165",
"message": "Added delete code for template",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T04:32:35Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T04:32:35Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning-job-handler/index.js",
"additions": 3,
"deletions": 2,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/index.js",
"additions": 170,
"deletions": 89,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/pm2-development.yml",
"additions": 5,
"deletions": 0,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/pm2-production.yml",
"additions": 4,
"deletions": 0,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,100 @@
{
"number": 2,
"title": "Initial commit",
"body": "Additional improvements for initial commit (coqiau/TTS v0.6.2 compatibility)",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/2",
"createdAt": "2022-04-21T14:49:50Z",
"mergedAt": "2022-04-21T14:50:00Z",
"closedAt": "2022-04-21T14:50:00Z",
"additions": 19,
"deletions": 14,
"changedFiles": 3,
"isDraft": false,
"baseRefName": "main",
"headRefName": "initialCommit",
"author": {
"login": "author_unknown"
},
"mergedBy": {
"login": "author_unknown"
},
"mergeCommit": {
"oid": "a1d5a6b458af59b350a684e6fade1d1109b6d236"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 2,
"nodes": [
{
"commit": {
"oid": "b6abea59339bea8d4923d53fafc3c0e0d7c27ec9",
"message": "Removed install requirements for coqi-ai/Trainer (now a TTS dependency); added pretrained model install requirements.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-21T14:44:27Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-21T14:44:27Z"
}
}
},
{
"commit": {
"oid": "5cc49e28dbfbe823e3043232bc1984bdaf58b05f",
"message": "coquai/TTS v0.6.2 compatibiliuty changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-21T14:46:02Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-21T14:46:02Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning/clone_voice.py",
"additions": 3,
"deletions": 3,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/docs/potion-voice-cloning_Installation_Guide.md",
"additions": 13,
"deletions": 8,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/train_multispeaker_baseline_model.py",
"additions": 3,
"deletions": 3,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,94 @@
{
"number": 21,
"title": "Dev find best fixup",
"body": "find_best_* improvements (suppress TTS-based command line output; track progress instead)",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/21",
"createdAt": "2023-07-21T05:14:32Z",
"mergedAt": "2023-07-21T05:14:54Z",
"closedAt": "2023-07-21T05:14:54Z",
"additions": 88,
"deletions": 53,
"changedFiles": 2,
"isDraft": false,
"baseRefName": "develop",
"headRefName": "dev-find-best-fixup",
"author": {
"login": "author_unknown"
},
"mergedBy": {
"login": "author_unknown"
},
"mergeCommit": {
"oid": "a6639dbbae3c7920b312b5077aea4a9b42c18e1e"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 2,
"nodes": [
{
"commit": {
"oid": "f456697d843e83a34b1b5c192ced548bd10ae9d5",
"message": "Add progress tracker and suppress default TTS command-line outputs.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-17T08:32:37Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-17T08:32:37Z"
}
}
},
{
"commit": {
"oid": "dacf5b23d8581e816104868923dfc5d872960e5e",
"message": "Add progress tracker and suppress default TTS command-line outputs (part 2).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-18T15:24:55Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-18T15:24:55Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning/find_best_cloned_model.py",
"additions": 42,
"deletions": 22,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/find_best_multispeaker_model.py",
"additions": 46,
"deletions": 31,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,130 @@
{
"number": 22,
"title": "Dev cpu only",
"body": "CPU-only processing support (tested)",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/22",
"createdAt": "2023-07-25T09:44:12Z",
"mergedAt": "2023-07-25T09:44:26Z",
"closedAt": "2023-07-25T09:44:26Z",
"additions": 604,
"deletions": 28,
"changedFiles": 8,
"isDraft": false,
"baseRefName": "develop",
"headRefName": "dev-cpu-only",
"author": {
"login": "author_unknown"
},
"mergedBy": {
"login": "author_unknown"
},
"mergeCommit": {
"oid": "19894f8d753f8b5ecf7fb8ee756b63c4498f0fe9"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 2,
"nodes": [
{
"commit": {
"oid": "31234863f06285dcb88732cff5eead330ab852ad",
"message": "Minor improvements to support CPU-only processing.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-25T07:52:47Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-25T07:52:47Z"
}
}
},
{
"commit": {
"oid": "c96dccfb58c49186ee21c4fb24eff96302b35ecf",
"message": "Added more usage examples and necessary Coqui.ai TTS changes.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-25T09:43:09Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-25T09:43:09Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "requirements.prod.cpu.txt",
"additions": 3,
"deletions": 4,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/clone_voice_via_continue.py",
"additions": 3,
"deletions": 3,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/docs/potion-voice-cloning_Installation_Guide_-_CPU_only.md",
"additions": 578,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/find_best_cloned_model.py",
"additions": 2,
"deletions": 2,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/find_best_multispeaker_model.py",
"additions": 4,
"deletions": 4,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/prepare_datasets.py",
"additions": 5,
"deletions": 6,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/score_cloned_voice.py",
"additions": 3,
"deletions": 3,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/synthesize_speech.py",
"additions": 6,
"deletions": 6,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,948 @@
{
"number": 23,
"title": "Develop",
"body": "Support of 48k Hz sampling rate as default; GPU and CPU-based usage for all scripts but baseline training run.\r\n\r\nSee voice-cloning/docs/Voice\\ Cloning\\ @\\ 48k\\ Hz\\ Sampling\\ Rate\\ -\\ Step-by-Step.txt for example usage.\r\n\r\nModel assets can be found at s3://potion-ai-models/potion-voice-2023-08/",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/23",
"createdAt": "2023-08-07T06:58:31Z",
"mergedAt": "2023-09-25T07:15:56Z",
"closedAt": "2023-09-25T07:15:56Z",
"additions": 4121,
"deletions": 180,
"changedFiles": 27,
"isDraft": false,
"baseRefName": "staging",
"headRefName": "develop",
"author": {
"login": "author_unknown"
},
"mergedBy": {
"login": "author_7"
},
"mergeCommit": {
"oid": "8d62d38e524bfc62be0b6ac0ebca155ec2c4dda7"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 46,
"nodes": [
{
"commit": {
"oid": "6caecc390d8da5c2a0e1028ef5a477a25d0af9b5",
"message": "Switch to 48k as default sampling rate; disable mixed_precision due to possible min()/max() runtime error.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-30T11:29:26Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-30T11:29:26Z"
}
}
},
{
"commit": {
"oid": "f456697d843e83a34b1b5c192ced548bd10ae9d5",
"message": "Add progress tracker and suppress default TTS command-line outputs.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-17T08:32:37Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-17T08:32:37Z"
}
}
},
{
"commit": {
"oid": "dacf5b23d8581e816104868923dfc5d872960e5e",
"message": "Add progress tracker and suppress default TTS command-line outputs (part 2).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-18T15:24:55Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-18T15:24:55Z"
}
}
},
{
"commit": {
"oid": "a6639dbbae3c7920b312b5077aea4a9b42c18e1e",
"message": "Merge pull request #21 from potion/dev-find-best-fixup\n\nDev find best fixup",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-21T05:14:54Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-21T05:14:54Z"
}
}
},
{
"commit": {
"oid": "31234863f06285dcb88732cff5eead330ab852ad",
"message": "Minor improvements to support CPU-only processing.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-25T07:52:47Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-25T07:52:47Z"
}
}
},
{
"commit": {
"oid": "c96dccfb58c49186ee21c4fb24eff96302b35ecf",
"message": "Added more usage examples and necessary Coqui.ai TTS changes.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-25T09:43:09Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-25T09:43:09Z"
}
}
},
{
"commit": {
"oid": "19894f8d753f8b5ecf7fb8ee756b63c4498f0fe9",
"message": "Merge pull request #22 from potion/dev-cpu-only\n\nDev cpu only",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-25T09:44:26Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-25T09:44:26Z"
}
}
},
{
"commit": {
"oid": "2fde687781a036630ad65a1b6438f2bd2fe860c8",
"message": "Updated requirements: git clone --depth 1 --branch v0.16.0 https://github.com/coqui-ai/TTS onwards addresses the CPU-only processing issues.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-26T07:45:02Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-26T07:45:02Z"
}
}
},
{
"commit": {
"oid": "5f74690d758f047d1aa24f03b72dd242cfdf9c14",
"message": "Efficiency improvements: Rearranged loop order and removed re-init of speaker mgr and vocoder.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-26T16:42:56Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-26T16:42:56Z"
}
}
},
{
"commit": {
"oid": "4b83dcd2b33940fb67aa0a2ea726e5f4825709af",
"message": "Moved tqdm to outermost loop.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-26T16:49:29Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-26T16:49:29Z"
}
}
},
{
"commit": {
"oid": "ff2efad2181eef60c95855a26a346db3a241ac33",
"message": "Efficiency improvement: Removed re-init of vocoder.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-26T16:58:54Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-26T16:58:54Z"
}
}
},
{
"commit": {
"oid": "bfab4e52d152e1c1adb5127d8c52e57e5926d6f2",
"message": "Enable use_speaker_encoder_as_loss byu default.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-27T08:45:46Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-27T08:45:46Z"
}
}
},
{
"commit": {
"oid": "c4f0738615226b541c2728d8b518c1bb8f422e59",
"message": "Disable use_speaker_encoder_as_loss until we have a 48k speaker encoder.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-27T10:14:48Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-07-27T10:14:48Z"
}
}
},
{
"commit": {
"oid": "ebc3aabb5b47b92667f5cf38c6ad3eabe927c29b",
"message": "Efficiency improvements: re-use config and speaker embeddings when multiple models are evaluated.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-01T05:19:00Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-01T05:19:00Z"
}
}
},
{
"commit": {
"oid": "efe67a3be7ab363098a6ac11a868c84db8401942",
"message": "Bug fix: removed unnecessary argument.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-01T05:51:24Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-01T05:51:24Z"
}
}
},
{
"commit": {
"oid": "9c25385822430a0be1b517eba819fe45598b9a22",
"message": "No more need to suppress cmd output; fix remove silence bug; code readability improvements; better error handling.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T04:12:11Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T04:12:11Z"
}
}
},
{
"commit": {
"oid": "b861f6e7ca39117ce9d9d115dea10efd0b9382ad",
"message": "Fixed Vits config & model imports / typing references.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T05:20:28Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T05:20:28Z"
}
}
},
{
"commit": {
"oid": "24db1203681e633095501ad987fb7de61acbe1bf",
"message": "Fixed error handling for speaker embeddings.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T05:28:06Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T05:28:06Z"
}
}
},
{
"commit": {
"oid": "a9aa70cd38e28c26b9ea9cf3a142607733b5355e",
"message": "Typing improvements; trim_silence improvements.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T06:58:39Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T06:58:39Z"
}
}
},
{
"commit": {
"oid": "5a9113804a0762d4c926a5df9b56cd9cde09681d",
"message": "Testing alternative sim_score routine.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T09:10:37Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T09:10:37Z"
}
}
},
{
"commit": {
"oid": "9d937c7020d4fb2f71873beac200e12db77e2863",
"message": "Improved speaker similarity scoring approach.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T09:59:09Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T09:59:09Z"
}
}
},
{
"commit": {
"oid": "e81db101ed94eca0a634b16bbe0dc27f40ceb81e",
"message": "Remove scoring test import.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T10:05:18Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T10:05:18Z"
}
}
},
{
"commit": {
"oid": "5b551ce70ad459b64a9cdd2ce491a8d7740c44b0",
"message": "Fix ValueError issue.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T11:09:13Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T11:09:13Z"
}
}
},
{
"commit": {
"oid": "77fb20ca06b2f8b76285d6eefe662def5c92301b",
"message": "Add unload model (clean-up) routine; code readibility improvements.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T16:54:23Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-02T16:54:23Z"
}
}
},
{
"commit": {
"oid": "afdf4cfd7cc0cf90710a2e602327db45d6a87088",
"message": "Add unload model (clean-up) routine.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-03T01:42:31Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-03T01:42:31Z"
}
}
},
{
"commit": {
"oid": "60c8b4fe11e23740c4887cd8b5740b49eb63c806",
"message": "Switching to 48k HZ sampling rate by default.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T01:07:45Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T01:07:45Z"
}
}
},
{
"commit": {
"oid": "e47841f80c1e73ac33abbc161504782c3b06959b",
"message": "sampling_rate check fixes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T01:28:27Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T01:28:27Z"
}
}
},
{
"commit": {
"oid": "f98ad3fa51ed45847377d0d648c63868c2c990c6",
"message": "Support minimise call for multiple models (i.e., skip saving / overwritting minimised config).",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T04:55:43Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T04:55:43Z"
}
}
},
{
"commit": {
"oid": "a4d002539ad67e0667e6642293123793d7b72a34",
"message": "Refreshed documentation to reflect 48k sampling rate setup and usage.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T06:52:42Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T06:52:42Z"
}
}
},
{
"commit": {
"oid": "a92f32ac694e18e19b2676847387a215d05e01ea",
"message": "Merge branch 'staging' into develop-07-08",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:03:54Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:03:54Z"
}
}
},
{
"commit": {
"oid": "1bd4b8d3039d8a1e2ac02476bec235da829c5b5c",
"message": "Merge branch 'staging' into develop-07-08",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:05:02Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:05:02Z"
}
}
},
{
"commit": {
"oid": "75e78312e8d92560c335f640248bef88e37ddcf7",
"message": "Merge pull request #24 from potion/develop-07-08\n\nDevelop 07 08 code updates",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:08:05Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:08:05Z"
}
}
},
{
"commit": {
"oid": "a3afe89e0e4289c33044e3b9401a5d0bda2be401",
"message": "Added code for cloning script",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:20:41Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:20:41Z"
}
}
},
{
"commit": {
"oid": "1c59012333173eba76cec6325608024beb1a669a",
"message": "Merge pull request #25 from potion/develop-07-08-updates\n\nAdded code for cloning script",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:21:18Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:21:18Z"
}
}
},
{
"commit": {
"oid": "ea4608475587cea4615a4980bc767708a186d57e",
"message": "Added changes for sr 48000",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T10:17:46Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T10:17:58Z"
}
}
},
{
"commit": {
"oid": "6defbb0bb874390f1b1b51b1e20355890c530493",
"message": "Merge pull request #26 from potion/develop-07-08-updates\n\nAdded changes for sr 48000",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T10:19:55Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T10:19:55Z"
}
}
},
{
"commit": {
"oid": "bb152205ade80927946b88008109f106107af8f9",
"message": "Extended similarity scoring approach to also includde naturalness assessment.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-13T16:09:30Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-13T16:09:30Z"
}
}
},
{
"commit": {
"oid": "34740bd2d5daf5411c3fd51dd81125c3e2ee2a12",
"message": "Merge branch 'develop' of https://github.com/potion/potion-voice into develop",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-13T16:11:05Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-13T16:11:05Z"
}
}
},
{
"commit": {
"oid": "9527d33af85af9dae620cdf2fb6e462c50c9979e",
"message": "Bug fix: Wrong path used for synth samples.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-14T01:04:16Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-14T01:04:16Z"
}
}
},
{
"commit": {
"oid": "49a20a14ab518f005305c22e14bdfe4efab2d7cf",
"message": "Added naturalness score to find_best_cloned_model; improved comments.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-14T07:52:52Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-14T07:52:52Z"
}
}
},
{
"commit": {
"oid": "9703a12253c7e870079654776d8458a06725c71c",
"message": "Bug fix: init quality score correction",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-14T07:55:21Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-14T07:55:21Z"
}
}
},
{
"commit": {
"oid": "1021bd8bd93694350d66caf9ee7e6c03d2e4b1d9",
"message": "voice quality score redefined: 3/4 sim_scoe and 1/4 nat_score.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-14T14:34:46Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-14T14:34:46Z"
}
}
},
{
"commit": {
"oid": "789d1bc6777f274153b00a071e62cee7b689ae40",
"message": "Readying updated scoring approach for staging.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-18T05:18:03Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-18T05:18:03Z"
}
}
},
{
"commit": {
"oid": "313d2bda3895b0de175b9692d0148c0f257d3f1c",
"message": "Bug fix: Adjust to prev name change of init_synth argument.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-18T05:23:51Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-18T05:23:51Z"
}
}
},
{
"commit": {
"oid": "5233e2e6e3655d3f5f77dd1abf23a482d7c3975e",
"message": "Init fixup; improved comments.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-18T05:52:22Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-18T05:52:22Z"
}
}
},
{
"commit": {
"oid": "acbc3c99460ab52b1d5e6cd8bbf92dfe1062fdfe",
"message": "Improved wording for console-based output.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-18T07:40:39Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-09-18T07:40:39Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": ".gitignore",
"additions": 2,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "requirements.dev.local.txt",
"additions": 6,
"deletions": 0,
"changeType": "MODIFIED"
},
{
"path": "requirements.dev.txt",
"additions": 9,
"deletions": 3,
"changeType": "MODIFIED"
},
{
"path": "requirements.prod.cpu.txt",
"additions": 9,
"deletions": 4,
"changeType": "MODIFIED"
},
{
"path": "requirements.prod.gpu.txt",
"additions": 6,
"deletions": 0,
"changeType": "MODIFIED"
},
{
"path": "requirements.txt",
"additions": 6,
"deletions": 0,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/index.js",
"additions": 3,
"deletions": 3,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/clone_voice.py",
"additions": 14,
"deletions": 12,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/clone_voice_via_continue.py",
"additions": 6,
"deletions": 5,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/docs/Voice Cloning @ 48k Hz Sampling Rate - Step-by-Step.txt",
"additions": 183,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/docs/potion-voice-cloning_Installation_Guide.md",
"additions": 20,
"deletions": 6,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/docs/potion-voice-cloning_Installation_Guide_-_CPU_only.md",
"additions": 539,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/find_best_cloned_model.py",
"additions": 58,
"deletions": 24,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/find_best_multispeaker_model.py",
"additions": 77,
"deletions": 37,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/minimize_cloned_voice_model.py",
"additions": 7,
"deletions": 3,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/prepare_datasets.py",
"additions": 5,
"deletions": 6,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/score_cloned_voice.py",
"additions": 18,
"deletions": 10,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/synthesize_speech.py",
"additions": 29,
"deletions": 18,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/train_multispeaker_baseline_model.py",
"additions": 32,
"deletions": 24,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/utils/NISQA/LICENSE",
"additions": 21,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/utils/NISQA/LICENSE_model_weights",
"additions": 437,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/utils/NISQA/NISQA_lib.py",
"additions": 2170,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/utils/NISQA/NISQA_model.py",
"additions": 238,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/utils/NISQA/nisqa_tts.tar",
"additions": 0,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/utils/misc_utils.py",
"additions": 38,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/utils/scoring_utils.py",
"additions": 98,
"deletions": 9,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/utils/synthesize_utils.py",
"additions": 90,
"deletions": 15,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,298 @@
{
"number": 24,
"title": "Develop 07 08 code updates",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/24",
"createdAt": "2023-08-07T08:05:54Z",
"mergedAt": "2023-08-07T08:08:05Z",
"closedAt": "2023-08-07T08:08:05Z",
"additions": 182,
"deletions": 91,
"changedFiles": 4,
"isDraft": false,
"baseRefName": "develop",
"headRefName": "develop-07-08",
"author": {
"login": "author_7"
},
"mergedBy": {
"login": "author_6"
},
"mergeCommit": {
"oid": "75e78312e8d92560c335f640248bef88e37ddcf7"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 14,
"nodes": [
{
"commit": {
"oid": "94b4cbeb8e5863366512cbcc413a4a99817bef3d",
"message": "Changed version of python package",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-03T14:11:56Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-03T14:11:56Z"
}
}
},
{
"commit": {
"oid": "b254ec4c2842e34c4fb8a807655330916114ea45",
"message": "Added code for new changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-03T14:30:27Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-03T14:30:27Z"
}
}
},
{
"commit": {
"oid": "d95cae5242691b67688cad0077d0eef1ccf89fb2",
"message": "Added changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-03T15:00:59Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-03T15:00:59Z"
}
}
},
{
"commit": {
"oid": "0d7dc5354f866e849645692cd8af93ddd47a624c",
"message": "Added changes for synthesizer",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-06T18:02:16Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-06T18:02:16Z"
}
}
},
{
"commit": {
"oid": "219835c25c561b6d6f964c3ac533e2481b483c94",
"message": "Added change",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-06T18:10:09Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-06T18:10:09Z"
}
}
},
{
"commit": {
"oid": "58d0c83a2b2525967598d0915d374937e7dfd416",
"message": "Added path change",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-06T18:19:08Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-06T18:19:08Z"
}
}
},
{
"commit": {
"oid": "54efcabc34b82b156ab06a0beab0f895d4c7edd0",
"message": "Merge pull request #18 from potion/develop\n\nRefined voice cloning settings",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-21T10:58:16Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-21T10:58:16Z"
}
}
},
{
"commit": {
"oid": "5cb69b8e41743b1b5f7f66e65ea15ff64a0ea945",
"message": "Added changes for not to save the salutation to global",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T03:51:40Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T03:51:40Z"
}
}
},
{
"commit": {
"oid": "58f4b9346d1918f2f83ae249035c8e93d326046c",
"message": "Merge branch 'staging' into update-voice-clone-23-05",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T03:52:23Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T03:52:23Z"
}
}
},
{
"commit": {
"oid": "8aeccc682363a55001fbdb9ec67c861cfb6a8b63",
"message": "Added code for deleting directory",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T04:15:47Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T04:15:47Z"
}
}
},
{
"commit": {
"oid": "e5f6efbb72902997353580ee56b7893c64452165",
"message": "Added delete code for template",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T04:32:35Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-22T04:32:35Z"
}
}
},
{
"commit": {
"oid": "6d89d4f3c84b1f743d93d3b1f6cf70472e1866e5",
"message": "Merge pull request #19 from potion/update-voice-clone-23-05\n\nUpdate voice clone 23 05",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-27T04:50:28Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-06-27T04:50:28Z"
}
}
},
{
"commit": {
"oid": "a92f32ac694e18e19b2676847387a215d05e01ea",
"message": "Merge branch 'staging' into develop-07-08",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:03:54Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:03:54Z"
}
}
},
{
"commit": {
"oid": "1bd4b8d3039d8a1e2ac02476bec235da829c5b5c",
"message": "Merge branch 'staging' into develop-07-08",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:05:02Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:05:02Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning-job-handler/index.js",
"additions": 3,
"deletions": 2,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/index.js",
"additions": 170,
"deletions": 89,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/pm2-development.yml",
"additions": 5,
"deletions": 0,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/pm2-production.yml",
"additions": 4,
"deletions": 0,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,72 @@
{
"number": 25,
"title": "Added code for cloning script",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/25",
"createdAt": "2023-08-07T08:21:11Z",
"mergedAt": "2023-08-07T08:21:18Z",
"closedAt": "2023-08-07T08:21:18Z",
"additions": 2,
"deletions": 2,
"changedFiles": 1,
"isDraft": false,
"baseRefName": "develop",
"headRefName": "develop-07-08-updates",
"author": {
"login": "author_7"
},
"mergedBy": {
"login": "author_7"
},
"mergeCommit": {
"oid": "1c59012333173eba76cec6325608024beb1a669a"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 1,
"nodes": [
{
"commit": {
"oid": "a3afe89e0e4289c33044e3b9401a5d0bda2be401",
"message": "Added code for cloning script",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:20:41Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T08:20:41Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning-job-handler/index.js",
"additions": 2,
"deletions": 2,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,72 @@
{
"number": 26,
"title": "Added changes for sr 48000",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/26",
"createdAt": "2023-08-07T10:19:45Z",
"mergedAt": "2023-08-07T10:19:55Z",
"closedAt": "2023-08-07T10:19:55Z",
"additions": 1,
"deletions": 1,
"changedFiles": 1,
"isDraft": false,
"baseRefName": "develop",
"headRefName": "develop-07-08-updates",
"author": {
"login": "author_7"
},
"mergedBy": {
"login": "author_7"
},
"mergeCommit": {
"oid": "6defbb0bb874390f1b1b51b1e20355890c530493"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 1,
"nodes": [
{
"commit": {
"oid": "ea4608475587cea4615a4980bc767708a186d57e",
"message": "Added changes for sr 48000",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T10:17:46Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-08-07T10:17:58Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning-job-handler/index.js",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,76 @@
{
"number": 27,
"title": "updated model name",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/27",
"createdAt": "2023-10-11T20:46:31Z",
"mergedAt": "2023-10-11T20:47:05Z",
"closedAt": "2023-10-11T20:47:05Z",
"additions": 2,
"deletions": 2,
"changedFiles": 1,
"isDraft": false,
"baseRefName": "staging",
"headRefName": "develop",
"author": {
"login": "author_7"
},
"mergedBy": {
"login": "author_7"
},
"mergeCommit": {
"oid": "7fdfc74c03cab06856eff2fac1ec470f40eb64ad"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": [
{
"login": "author_6"
}
]
},
"commits": {
"totalCount": 1,
"nodes": [
{
"commit": {
"oid": "040f8565569ff7d324db918f17e5aabfb15ba4ca",
"message": "updated model name",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-10-11T20:36:39Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-10-11T20:36:39Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning-job-handler/index.js",
"additions": 2,
"deletions": 2,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,508 @@
{
"number": 28,
"title": "GCP support, upgraded TTS, new model with cleaner data",
"body": "",
"state": "OPEN",
"url": "https://github.com/potion/potion-voice/pull/28",
"createdAt": "2023-11-29T07:42:08Z",
"mergedAt": null,
"closedAt": null,
"additions": 1816,
"deletions": 283,
"changedFiles": 15,
"isDraft": false,
"baseRefName": "staging",
"headRefName": "develop",
"author": {
"login": "author_unknown"
},
"mergedBy": {
"login": ""
},
"mergeCommit": {
"oid": "8615798d7f85f3f29ce99973f97bd9071f76f05b"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 23,
"nodes": [
{
"commit": {
"oid": "ad31e02f2a39af5e40fbefe30483e5d8119346b4",
"message": "Added GCP support details and revised config settings for 48k Hz sampling rate usage with TTS v0.20.6",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-11-23T08:45:24Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-11-23T08:45:24Z"
}
}
},
{
"commit": {
"oid": "c41b76ff84f69f00ebfdb1bfbea6c40b0c2529bd",
"message": "Richer config settings (added sampling rate-based configs) and updated training settings.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-19T08:36:49Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-19T08:36:49Z"
}
}
},
{
"commit": {
"oid": "fdf7496764cff1f7f754175b121d1eec5fce285b",
"message": "Improved error handling and robustness; added support for FLAC files - VCTK v0.92 preprocessing.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T08:00:56Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T08:00:56Z"
}
}
},
{
"commit": {
"oid": "55886541d20ef9be247610cf616f0167842ca8de",
"message": "Add support for 24k sampling rate.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T08:33:31Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T08:33:31Z"
}
}
},
{
"commit": {
"oid": "bf4d4c9b215894810755bcd9deb1d6a980c408cc",
"message": "Improved for directory name in extract_archive",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T09:08:51Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T09:08:51Z"
}
}
},
{
"commit": {
"oid": "90bf857da5add6b341a5514d13bc2a144b8a542c",
"message": "Bug fix in extract_archive",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T09:35:56Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T09:35:56Z"
}
}
},
{
"commit": {
"oid": "cd968b26a2214cb241a4b660f80c360493a38493",
"message": "Support VCTK v0.92 _mic[12] naming convention.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T12:43:27Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T12:43:27Z"
}
}
},
{
"commit": {
"oid": "c5abb3269df3c7e4d828ba019907ac0356349fd3",
"message": "Rename wav_dir to audio_dir and correct function call arguments.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T13:58:09Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T13:58:09Z"
}
}
},
{
"commit": {
"oid": "334aef03182d954606bf2dd6907a30f380ecc0dd",
"message": "Verify that after extraction the dataset root and its audio and transcription file directories are identified correctly.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T15:31:18Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T15:31:18Z"
}
}
},
{
"commit": {
"oid": "41fa5ff5a72366fb7466445c79dad5c8fd5ec598",
"message": "Improved quality assurance: Ensure that each speaker had transcriptions with matching audio files and vice versa.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T17:32:04Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2023-12-27T17:32:04Z"
}
}
},
{
"commit": {
"oid": "2e33a79e0465e5fb13f22b6fee2e1f54178860d9",
"message": "fixups and added support for training merged, VCTK-formatted datasets.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-10T03:26:07Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-10T03:26:07Z"
}
}
},
{
"commit": {
"oid": "b0a3c294729107b780d941dfc4d0930b055b187c",
"message": "Added support for training merged, VCTK-formatted datasets.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-10T03:26:37Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-10T03:26:37Z"
}
}
},
{
"commit": {
"oid": "2d7adac5070bd308eba84c7dc193b6c5b5effaa2",
"message": "New cloning approach with termination condition based on speaker similarity and voice naturalness scores.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-10T03:27:39Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-10T03:27:39Z"
}
}
},
{
"commit": {
"oid": "6161686a58dd16505b59107c406ff759eaf02e71",
"message": "Added minimum scoring thresholds for speaker similarity and naturalness; updated scoring parameters.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-10T06:28:07Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-10T06:28:07Z"
}
}
},
{
"commit": {
"oid": "665fadf062511880abfae5b1f1b2bbf1d15d0a8f",
"message": "Skip checkpoint scoring iff keyboard interrupt.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-10T08:12:42Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-10T08:12:42Z"
}
}
},
{
"commit": {
"oid": "3a03db620469cf6603b49cabe1201a22bf81775e",
"message": "Expand pattern to also pick up best_simnat_checkpoint_*.pth checkpoint files.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-11T09:50:30Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-11T09:50:30Z"
}
}
},
{
"commit": {
"oid": "b22f1051bb3374bc969e3962740a3610266f2928",
"message": "48k Voice cloning documentation now based on clone_voice_via_continue_n_natqa.py; adjusted default cloning parameters.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-11T17:34:07Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-11T17:34:07Z"
}
}
},
{
"commit": {
"oid": "2c34903fe6b62551fb5d6844af78a0da1eb3cdb6",
"message": "find_best_cloned_model.py now also checks for minimum quality nat & sim scores; returns None for best_model if they are not met.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-12T10:30:52Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-12T10:30:52Z"
}
}
},
{
"commit": {
"oid": "75e92b2b887e3f0ccf8ea4a5c536883d4b6e2570",
"message": "Add support for training speaker encoder model at 16k and 48k sampling rates.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-28T16:06:30Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-28T16:06:30Z"
}
}
},
{
"commit": {
"oid": "7241466763630caf6febb9ea620211b2dcdb1bd7",
"message": "Bug fix: add use_cuda parameters whenever required",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-29T09:30:17Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-29T09:30:17Z"
}
}
},
{
"commit": {
"oid": "530ff4c8e7bc89d6205cf43d7ccd13bd13c5124e",
"message": "Init logger if None is given.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-29T12:51:00Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-29T12:51:00Z"
}
}
},
{
"commit": {
"oid": "38e981c51b9232c6e441d96c2901167f43d7ce0b",
"message": "Evaluation parameter passing fix.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-29T17:43:33Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-29T17:43:33Z"
}
}
},
{
"commit": {
"oid": "85eddfd84d8f6c79ba99010fa82148641c97970f",
"message": "Bug fix: Eval routine output init missing.",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-30T01:53:17Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2024-01-30T01:53:17Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning/clone_voice_via_continue.py",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/clone_voice_via_continue_n_natqa.py",
"additions": 291,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/docs/Voice Cloning @ 48k Hz Sampling Rate - Step-by-Step.txt",
"additions": 62,
"deletions": 57,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/docs/potion-voice-cloning_Installation_Guide.md",
"additions": 20,
"deletions": 50,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/find_best_cloned_model.py",
"additions": 23,
"deletions": 11,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/find_best_multispeaker_model.py",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/prepare_datasets.py",
"additions": 262,
"deletions": 88,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/save_multispeaker_baseline_embeddings_file.py",
"additions": 11,
"deletions": 4,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/synthesize_speech.py",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/train_config.py",
"additions": 101,
"deletions": 46,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/train_multispeaker_baseline_model.py",
"additions": 44,
"deletions": 23,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/train_speaker_encoder.py",
"additions": 181,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/utils/scoring_utils.py",
"additions": 78,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/utils/speaker_encoder_utils.py",
"additions": 389,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/utils/trainer_utils.py",
"additions": 351,
"deletions": 0,
"changeType": "ADDED"
}
]
}
}

View File

@@ -0,0 +1,78 @@
{
"number": 3,
"title": "Further coquai/TTS v0.6.2 compatibiliuty changes",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/3",
"createdAt": "2022-04-21T14:56:26Z",
"mergedAt": "2022-04-21T14:56:32Z",
"closedAt": "2022-04-21T14:56:32Z",
"additions": 2,
"deletions": 2,
"changedFiles": 2,
"isDraft": false,
"baseRefName": "main",
"headRefName": "initialCommit",
"author": {
"login": "author_unknown"
},
"mergedBy": {
"login": "author_unknown"
},
"mergeCommit": {
"oid": "113758f7434d36f7d090959adb5430abd2d1159e"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 1,
"nodes": [
{
"commit": {
"oid": "c81a01c209362a2a4e092e2e1cb848a9e14eddfa",
"message": "Further coquai/TTS v0.6.2 compatibiliuty changes",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-21T14:55:40Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-21T14:55:40Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning/clone_voice.py",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning/train_multispeaker_baseline_model.py",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,930 @@
{
"number": 4,
"title": "Feature 4023 voice clone handler",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/4",
"createdAt": "2022-06-06T15:36:08Z",
"mergedAt": "2022-07-19T07:21:41Z",
"closedAt": "2022-07-19T07:21:41Z",
"additions": 6450,
"deletions": 2,
"changedFiles": 40,
"isDraft": false,
"baseRefName": "main",
"headRefName": "feature-4023-voice-clone-handler",
"author": {
"login": "author_6"
},
"mergedBy": {
"login": "author_6"
},
"mergeCommit": {
"oid": "198aadfae502ee2c24eb4d32a93439e47067f9ff"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 40,
"nodes": [
{
"commit": {
"oid": "485e9915301bf4fe864029b0b2a391f0d2339d29",
"message": "Updated git ignore file",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-27T21:49:22Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-27T21:49:22Z"
}
}
},
{
"commit": {
"oid": "1f70cf87c2c13455add0c56af95ab5a2211e7942",
"message": "Added base code for voice cloning",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-27T21:51:05Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-04-27T21:51:05Z"
}
}
},
{
"commit": {
"oid": "9133eb39182721fd3847d97bd656a10b6346b4d0",
"message": "Added code for db update and status update",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-10T18:39:00Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-10T18:39:00Z"
}
}
},
{
"commit": {
"oid": "8478a51640fc5744711c183ba5aabb6ed3cf1d55",
"message": "Added salutation service and imported s3 model",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-10T18:44:39Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-10T18:44:39Z"
}
}
},
{
"commit": {
"oid": "0878a129ce1f7fb77085ddbe018ccb6424bc11d2",
"message": "Updated the sqs code",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-10T19:00:24Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-10T19:00:24Z"
}
}
},
{
"commit": {
"oid": "d6a2d6c761b94a758f9e2d0f03a759da5be52fd5",
"message": "Added code for voice synthesizer",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-10T19:25:03Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-10T19:25:03Z"
}
}
},
{
"commit": {
"oid": "79bf6fde2abfd7063bc3a1ff1c2dce5b42e8eada",
"message": "Added code for db connect",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-12T10:34:15Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-12T10:34:15Z"
}
}
},
{
"commit": {
"oid": "d5c037a7257f39ba1767b5b0ac6ec18b8c34ea92",
"message": "Updated command",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-12T10:39:32Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-12T10:39:32Z"
}
}
},
{
"commit": {
"oid": "569a2484dabd73b7a1b812540f8723bb8423fb01",
"message": "Updated command",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-12T10:46:35Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-12T10:46:35Z"
}
}
},
{
"commit": {
"oid": "d6a1af1063e12d9b073163a4296dd9f04546d535",
"message": "Updated command",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-12T10:59:30Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-12T10:59:30Z"
}
}
},
{
"commit": {
"oid": "84e63f580b72956a97c27fa6d9d41d6a147991a5",
"message": "fixed bug",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-12T17:50:16Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-12T17:50:16Z"
}
}
},
{
"commit": {
"oid": "f726bf5cb38d0da590f51468b0ce3be275e17c88",
"message": "Added changes for log utils",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-12T17:59:29Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-12T17:59:29Z"
}
}
},
{
"commit": {
"oid": "3271a7075ae1b1734faad3fe4cdc45fffffd3a3e",
"message": "Updated the logger object usage",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-12T18:06:41Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-12T18:06:41Z"
}
}
},
{
"commit": {
"oid": "35cf55c62186b425fc6380706d36967ed714a57d",
"message": "fixed db issue",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-25T06:50:15Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-25T06:50:15Z"
}
}
},
{
"commit": {
"oid": "ee5e43fde5d3217a84182e770bda73db245df17e",
"message": "fixed issues",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-25T07:35:50Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-25T07:35:50Z"
}
}
},
{
"commit": {
"oid": "b73ef06d9d983567a6ec4e7c5d930631db22cb26",
"message": "Fixed python training model commond",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-26T11:30:33Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-26T11:30:33Z"
}
}
},
{
"commit": {
"oid": "45c85a0a69247e9fd0eea8ceafa032daecc590da",
"message": "Added code for s3 upload and db update",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-26T14:33:40Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-26T14:33:40Z"
}
}
},
{
"commit": {
"oid": "172fea390bc6ea626366a836989135c121249338",
"message": "Added changes for s3 upload",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-26T15:11:52Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-26T15:11:52Z"
}
}
},
{
"commit": {
"oid": "b3447c5db4ae841d7f867f7f85583f309f722713",
"message": "Added changes to s3",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-26T15:12:29Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-26T15:12:29Z"
}
}
},
{
"commit": {
"oid": "f9ad756321d6e58e2a3258787718b7acfee5d16e",
"message": "Added change for path",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-26T15:19:26Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-26T15:19:26Z"
}
}
},
{
"commit": {
"oid": "985d130ef5c87bc7ea7c04b0338e7edbd71eae18",
"message": "Merge branch 'main' into feature-4023-voice-clone-handler",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T15:31:09Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T15:31:09Z"
}
}
},
{
"commit": {
"oid": "d8e350c29929c2ec7b9a459d5174d342a2b92da5",
"message": "Added job and audio profile model and service",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T15:53:25Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T15:53:25Z"
}
}
},
{
"commit": {
"oid": "96a23f61aa14fce790f9f6a6b8b6182507751ee2",
"message": "Added salutation service code",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T17:48:36Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T17:48:36Z"
}
}
},
{
"commit": {
"oid": "c6643bbcd640ef3ee75ef3dc537dd8ecfcfb464e",
"message": "Addded changes to code",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T18:27:16Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T18:27:16Z"
}
}
},
{
"commit": {
"oid": "606d7c440b074fae8bd266d57a7c21774e9b9134",
"message": "updated the code for db connect",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T18:29:33Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T18:29:33Z"
}
}
},
{
"commit": {
"oid": "ba9f860ccc95288513f0e3b82ee9e1cf712320c8",
"message": "Updated bucket name",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T18:31:49Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T18:31:49Z"
}
}
},
{
"commit": {
"oid": "38f4bca91eb38b2b7f5e610b389a7dff982b8aa0",
"message": "Updated buffer size for stdout on exec",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T18:36:18Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T18:36:18Z"
}
}
},
{
"commit": {
"oid": "b6037e00b8b74fb44b94f2dd26f4a3788b8183dc",
"message": "Added db connection code",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T18:44:23Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T18:44:23Z"
}
}
},
{
"commit": {
"oid": "6af84e64003d73e00786dd26cfeb8fa876d9f1fc",
"message": "Added code update",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T18:46:16Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-05-31T18:46:16Z"
}
}
},
{
"commit": {
"oid": "887337ec5173ddb4cda763cc45e6622671e8a490",
"message": "Merge branch 'main' into feature-4023-voice-clone-handler",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-06T15:39:00Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-06T15:39:00Z"
}
}
},
{
"commit": {
"oid": "b66a19b55110a5e3e9f72ae1ef75644328274067",
"message": "changed the pretrained module path argument",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-06T15:43:53Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-06T15:43:53Z"
}
}
},
{
"commit": {
"oid": "9805441c5ad0b6560ba2fb4aa1357f0769006a7b",
"message": "fixed python base module path argument issue",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-06T15:54:17Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-06T15:54:17Z"
}
}
},
{
"commit": {
"oid": "86f768692f5e5b445ba54089afadfbe8f914a10e",
"message": "fixed python base module path argument issue",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T05:30:15Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T05:30:15Z"
}
}
},
{
"commit": {
"oid": "4a5876b26b1c29de26f223638cc2859882d9fe3c",
"message": "fixed python command new line issue",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T05:37:32Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T05:37:32Z"
}
}
},
{
"commit": {
"oid": "fb0aa3ab3404abf9cd464cac1bda8842f4d6a231",
"message": "fixed issues",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T06:42:40Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T06:42:40Z"
}
}
},
{
"commit": {
"oid": "c6c7d126c8ac1a519fd3033824ff493175369f6e",
"message": "Merge branch 'feature-4023-voice-clone-handler' of https://github.com/potion/potion-voice into feature-4023-voice-clone-handler",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T06:43:11Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T06:43:11Z"
}
}
},
{
"commit": {
"oid": "c18611d1a1851acade764fe9ba248aee9163a02c",
"message": "pm2 added name of process to avoid conflicts",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T06:51:21Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T06:51:21Z"
}
}
},
{
"commit": {
"oid": "c6d8567b362782f065103c0adaa93dc37323c5cc",
"message": "Update clone_voice.py",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T07:06:42Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T07:06:42Z"
}
}
},
{
"commit": {
"oid": "7bb3ad322186b5ea1e815c827971e2894921352f",
"message": "Fixed error for userId",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T09:45:07Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T09:45:15Z"
}
}
},
{
"commit": {
"oid": "1b1f50404b5b2ec6d019db0de574b35a5231fef2",
"message": "Added fix for unlink",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T09:56:55Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-06-07T09:56:58Z"
}
}
}
]
},
"reviews": {
"nodes": []
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": ".gitignore",
"additions": 25,
"deletions": 0,
"changeType": "MODIFIED"
},
{
"path": "app/services/s3/index.js",
"additions": 61,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "app/services/s3/s3_service.js",
"additions": 0,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "app/services/sqs/index.js",
"additions": 6,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "app/services/sqs/sqs_service.js",
"additions": 83,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "app/services/utils/bugsnag.js",
"additions": 18,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "app/services/utils/deleteFile.js",
"additions": 9,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "app/services/utils/index.js",
"additions": 19,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "app/services/utils/logService.js",
"additions": 14,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "app/services/voice_cloning/index.js",
"additions": 4,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "app/services/voice_cloning/voice_cloning_model.js",
"additions": 44,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "app/services/voice_cloning/voice_cloning_service.js",
"additions": 140,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "package-lock.json",
"additions": 1782,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "package.json",
"additions": 21,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning-job-handler/index.js",
"additions": 263,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning-job-handler/package-lock.json",
"additions": 1782,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning-job-handler/package.json",
"additions": 24,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning-job-handler/pm2-development.yml",
"additions": 15,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning-job-handler/user_audio_profile/index.js",
"additions": 4,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning-job-handler/user_audio_profile/user_audio_profile_model.js",
"additions": 40,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning-job-handler/user_audio_profile/user_audio_profile_service.js",
"additions": 142,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning-job-handler/voice_cloning/index.js",
"additions": 4,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning-job-handler/voice_cloning/voice_cloning_model.js",
"additions": 44,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning-job-handler/voice_cloning/voice_cloning_service.js",
"additions": 143,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-cloning/clone_voice.py",
"additions": 4,
"deletions": 2,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/index.js",
"additions": 269,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/job/index.js",
"additions": 4,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/job/job_model.js",
"additions": 54,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/job/job_service.js",
"additions": 136,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/package-lock.json",
"additions": 528,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/package.json",
"additions": 23,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/pm2-development.yml",
"additions": 14,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/recording/index.js",
"additions": 4,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/recording/recording_model.js",
"additions": 406,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/salutation/index.js",
"additions": 4,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/salutation/salutation_model.js",
"additions": 44,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/salutation/salutation_service.js",
"additions": 87,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/user_audio_profile/index.js",
"additions": 4,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/user_audio_profile/user_audio_profile_model.js",
"additions": 40,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/user_audio_profile/user_audio_profile_service.js",
"additions": 142,
"deletions": 0,
"changeType": "ADDED"
}
]
}
}

View File

@@ -0,0 +1,85 @@
{
"number": 5,
"title": "Added the filename fix",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/5",
"createdAt": "2022-07-25T08:52:45Z",
"mergedAt": "2022-08-02T05:49:56Z",
"closedAt": "2022-08-02T05:49:56Z",
"additions": 8,
"deletions": 2,
"changedFiles": 1,
"isDraft": false,
"baseRefName": "main",
"headRefName": "hotfix-update-filename",
"author": {
"login": "author_7"
},
"mergedBy": {
"login": "author_6"
},
"mergeCommit": {
"oid": "cd472f8ee9a500461672acece889b02bb3a8e84f"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 1,
"nodes": [
{
"commit": {
"oid": "470e8e7b3012a7e39755f8e3e9de3eb0a257406a",
"message": "Added the filename fix",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-07-25T08:52:19Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-07-25T08:52:19Z"
}
}
}
]
},
"reviews": {
"nodes": [
{
"author": {
"login": "author_6"
},
"state": "APPROVED",
"body": "",
"submittedAt": "2022-08-02T05:49:50Z",
"url": "https://github.com/potion/potion-voice/pull/5#pullrequestreview-1058152207",
"comments": {
"nodes": []
}
}
]
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning-job-handler/index.js",
"additions": 8,
"deletions": 2,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,97 @@
{
"number": 6,
"title": "Updated the cloud front access and code",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/6",
"createdAt": "2022-11-29T13:22:59Z",
"mergedAt": "2022-11-29T13:35:23Z",
"closedAt": "2022-11-29T13:35:23Z",
"additions": 26,
"deletions": 3,
"changedFiles": 3,
"isDraft": false,
"baseRefName": "main",
"headRefName": "fix-bucket-access-for-sentences",
"author": {
"login": "author_7"
},
"mergedBy": {
"login": "author_6"
},
"mergeCommit": {
"oid": "5e5907f8b9099f4b51212ea2b4615cc2b48f3dc8"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 1,
"nodes": [
{
"commit": {
"oid": "972fa9e89c08cbd799230eab43b80c9a7f80f0ce",
"message": "Updated the cloudfront access and code",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-11-29T13:22:35Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-11-29T13:22:35Z"
}
}
}
]
},
"reviews": {
"nodes": [
{
"author": {
"login": "author_6"
},
"state": "APPROVED",
"body": "",
"submittedAt": "2022-11-29T13:35:16Z",
"url": "https://github.com/potion/potion-voice/pull/6#pullrequestreview-1197567560",
"comments": {
"nodes": []
}
}
]
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning-job-handler/index.js",
"additions": 18,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/pm2-development.yml",
"additions": 4,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/pm2-production.yml",
"additions": 4,
"deletions": 1,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,91 @@
{
"number": 7,
"title": "Added change in yml file",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/7",
"createdAt": "2022-11-29T14:51:46Z",
"mergedAt": "2022-11-29T14:52:47Z",
"closedAt": "2022-11-29T14:52:47Z",
"additions": 2,
"deletions": 2,
"changedFiles": 2,
"isDraft": false,
"baseRefName": "main",
"headRefName": "fix-bucket-access-for-sentences",
"author": {
"login": "author_7"
},
"mergedBy": {
"login": "author_6"
},
"mergeCommit": {
"oid": "a1d067c2fa8a813a6d47f7bccfb8d08af3163e23"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 1,
"nodes": [
{
"commit": {
"oid": "26e0b4fa32500e73025b8c1c4941deddcceab6da",
"message": "Added change in yml file",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-11-29T14:51:14Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-11-29T14:51:14Z"
}
}
}
]
},
"reviews": {
"nodes": [
{
"author": {
"login": "author_6"
},
"state": "APPROVED",
"body": "",
"submittedAt": "2022-11-29T14:52:40Z",
"url": "https://github.com/potion/potion-voice/pull/7#pullrequestreview-1197714881",
"comments": {
"nodes": []
}
}
]
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning-job-handler/pm2-development.yml",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/pm2-production.yml",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,193 @@
{
"number": 8,
"title": "updated mongoose version 6.x",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/8",
"createdAt": "2022-12-16T15:03:54Z",
"mergedAt": "2022-12-16T15:31:52Z",
"closedAt": "2022-12-16T15:31:52Z",
"additions": 6351,
"deletions": 23,
"changedFiles": 13,
"isDraft": false,
"baseRefName": "main",
"headRefName": "PR-update-mongoose-version-to-6.x",
"author": {
"login": "author_9"
},
"mergedBy": {
"login": "author_6"
},
"mergeCommit": {
"oid": "89ba7c08942ecb39cfc90d447d142e8d9fd8a3dc"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": [
{
"login": "author_7"
}
]
},
"commits": {
"totalCount": 3,
"nodes": [
{
"commit": {
"oid": "c31772f5c452adfad9548001d5bd44409054f6c6",
"message": "updated mongoose version 6.x",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-12-16T15:03:22Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-12-16T15:03:22Z"
}
}
},
{
"commit": {
"oid": "a444056854eaea7ad553e643e49953ac9792e54f",
"message": "update development mongo uri",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-12-16T15:15:25Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-12-16T15:15:25Z"
}
}
},
{
"commit": {
"oid": "ad1ade5bd3c480591490118c644059dabf5d46cf",
"message": "update dev/staging mongo uri",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-12-16T15:26:26Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-12-16T15:26:26Z"
}
}
}
]
},
"reviews": {
"nodes": [
{
"author": {
"login": "author_6"
},
"state": "APPROVED",
"body": "",
"submittedAt": "2022-12-16T15:31:43Z",
"url": "https://github.com/potion/potion-voice/pull/8#pullrequestreview-1221040528",
"comments": {
"nodes": []
}
}
]
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": ".prettierrc",
"additions": 7,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "package.json",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/index.js",
"additions": 2,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/package.json",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/pm2-development.yml",
"additions": 2,
"deletions": 2,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/pm2-production.yml",
"additions": 2,
"deletions": 2,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/yarn.lock",
"additions": 2475,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "voice-synthsizer-job-handler/index.js",
"additions": 2,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/package.json",
"additions": 2,
"deletions": 2,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/pm2-development.yml",
"additions": 5,
"deletions": 6,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/pm2-production.yml",
"additions": 6,
"deletions": 7,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/yarn.lock",
"additions": 1371,
"deletions": 0,
"changeType": "ADDED"
},
{
"path": "yarn.lock",
"additions": 2475,
"deletions": 0,
"changeType": "ADDED"
}
]
}
}

View File

@@ -0,0 +1,119 @@
{
"number": 9,
"title": "update DB uri",
"body": "",
"state": "MERGED",
"url": "https://github.com/potion/potion-voice/pull/9",
"createdAt": "2022-12-19T11:00:51Z",
"mergedAt": "2023-01-10T12:12:57Z",
"closedAt": "2023-01-10T12:12:57Z",
"additions": 6,
"deletions": 6,
"changedFiles": 4,
"isDraft": false,
"baseRefName": "main",
"headRefName": "update-db-uri",
"author": {
"login": "author_9"
},
"mergedBy": {
"login": "author_6"
},
"mergeCommit": {
"oid": "7708b1a13f89398c7718d1b39eae1271a1c9df40"
},
"milestone": null,
"labels": {
"nodes": []
},
"assignees": {
"nodes": []
},
"requestedReviewers": {
"nodes": []
},
"commits": {
"totalCount": 2,
"nodes": [
{
"commit": {
"oid": "391d1d181208264195773de717e9be058e9fe471",
"message": "update DB uri",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-12-19T11:00:25Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-12-19T11:00:25Z"
}
}
},
{
"commit": {
"oid": "6c444d38397251f3c7d1ea7eb919bdb57df8eb7e",
"message": "update DB uri",
"author": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-12-19T11:04:01Z"
},
"committer": {
"name": "author_unknown",
"email": "author_unknown",
"date": "2022-12-19T11:04:01Z"
}
}
}
]
},
"reviews": {
"nodes": [
{
"author": {
"login": "author_6"
},
"state": "APPROVED",
"body": "",
"submittedAt": "2023-01-10T12:12:51Z",
"url": "https://github.com/potion/potion-voice/pull/9#pullrequestreview-1242087312",
"comments": {
"nodes": []
}
}
]
},
"comments": {
"nodes": []
},
"files": {
"nodes": [
{
"path": "voice-cloning-job-handler/pm2-development.yml",
"additions": 2,
"deletions": 2,
"changeType": "MODIFIED"
},
{
"path": "voice-cloning-job-handler/pm2-production.yml",
"additions": 2,
"deletions": 2,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/pm2-development.yml",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
},
{
"path": "voice-synthsizer-job-handler/pm2-production.yml",
"additions": 1,
"deletions": 1,
"changeType": "MODIFIED"
}
]
}
}

View File

@@ -0,0 +1,2 @@
# potion-voice
Potion's Text-to-Speech Service (multi-speaker baseline model training, voice cloning and speech synthesising)

View File

@@ -0,0 +1,61 @@
const AWS = require('aws-sdk')
const fs = require('fs')
const { stringifyError } = require('../utils/logService')
var s3 = new AWS.S3()
const fetchS3Object = async ({ fileName, bucket, filePath }) => {
filePath = filePath || '/tmp/' + fileName
console.log('Fetching', stringifyObj({ fileName, filePath }))
try {
var params = { Bucket: bucket, Key: fileName }
const downloadResult = await s3.getObject(params).promise()
fs.writeFileSync(filePath, downloadResult.Body, function (err) {
if (err) console.log(err.code, '-', err.message)
})
return filePath
} catch (error) {
console.log('Download from S3 Error:', stringifyError(error))
throw error
}
}
const upload = ({
filePath,
fileName,
bucket,
contentType,
fileType,
// accessControl,
}) => {
return new Promise((resolve, reject) => {
// accessControl = accessControl || 'public-read'
fs.readFile(filePath, function (err, data) {
if (err) reject(err)
const params = {
Bucket: bucket, // pass your bucket name
Key: fileName,
Body: data,
// ContentType: contentType,
// ContentDisposition: `inline; fileName=${fileName}.${fileType}`,
// ACL: accessControl,
}
s3.upload(params, function (err, data) {
if (err) {
reject(err)
console.log(`${fileName} Upload to s3`, stringifyError(err))
} else {
console.log(
`Successfully uploaded data ${fileName}`,
stringifyError(data)
)
resolve(data.Location)
}
})
})
})
}
module.exports = {
upload,
fetchS3Object,
}

View File

@@ -0,0 +1,6 @@
const AWS = require('aws-sdk')
AWS.config.update({ region: 'us-west-2' })
const sqsService = require('./sqs_service')
const sqs = new AWS.SQS({ apiVersion: '2012-11-05' })
module.exports = sqsService

View File

@@ -0,0 +1,83 @@
const AWS = require('aws-sdk')
const sqs = new AWS.SQS({ apiVersion: '2012-11-05' })
const StringifyUtils = require('../utils/logService')
const fetchMessageFromSQS = (sqsQueueUrl, waitTimeInSeconds = 0) => {
return new Promise((resolve, reject) => {
const params = {
WaitTimeSeconds: waitTimeInSeconds,
QueueUrl: sqsQueueUrl /* required */,
}
sqs.receiveMessage(params, function (err, data) {
if (err) {
reject(err)
console.log(
`ERROR in fetchJobFromSQS : `,
StringifyUtils.stringifyError(err)
)
} else {
resolve(data)
}
})
})
}
const deleteMessageFromSQS = (sqsQueueUrl, receiptHandle) => {
return new Promise((resolve, reject) => {
const params = {
ReceiptHandle: receiptHandle,
QueueUrl: sqsQueueUrl /* required */,
}
sqs.deleteMessage(params, function (err, data) {
if (err) {
reject(err)
console.log(
`ERROR in sending delete request to AWS.SQS : `,
StringifyUtils.stringifyError(err)
)
} else {
console.log(
'Successfully sent delete request to AWS.SQS',
StringifyUtils.stringifyError(data)
)
resolve(data)
}
})
})
}
const sendMessageToSQS = (sqsQueueUrl, message) => {
return new Promise((resolve, reject) => {
const params = {
MessageBody: message,
QueueUrl: sqsQueueUrl /* required */,
// MessageGroupId:
// process.env.POTION_APP_ENV ||
// '' + `_` + uuidV4() + '_' + new Date().toISOString(),
// MessageDeduplicationId: uuidV4() + `_` + new Date().toISOString()
}
sqs.sendMessage(params, function (err, data) {
if (err) {
reject(err)
console.log(
`ERROR in seding request to AWS.SQS : `,
StringifyUtils.stringifyError(err)
)
} else {
console.log(
'Successfully sent request to AWS.SQS',
StringifyUtils.stringifyError(data)
)
resolve(data.Location)
}
})
})
}
module.exports = {
fetchMessageFromSQS,
deleteMessageFromSQS,
sendMessageToSQS,
}

View File

@@ -0,0 +1,18 @@
const Bugsnag = require('@bugsnag/js')
const DEV_APP_ENVS = ['local-dev', 'development']
const handleError = (err, user) => {
const isDevEnv = DEV_APP_ENVS.includes(process.env.POTION_APP_ENV)
if (isDevEnv) {
return
}
if (user)
Bugsnag.notify(err, function (event) {
event.setUser(user._id, user.email, user.name)
})
else Bugsnag.notify(err)
}
module.exports = handleError

View File

@@ -0,0 +1,9 @@
const deleteFile = (filePath) => {
return new Promise((resolve) => {
require('fs').unlinkSync(filePath)
console.log(`[deleted] ${filePath}`)
resolve()
})
}
exports.deleteFile = deleteFile

View File

@@ -0,0 +1,19 @@
const fs = require('fs')
const importedModules = {}
const files = fs.readdirSync(__dirname)
for (const file of files) {
const fileNameWithoutExtension = file.replace(/\.[^.]*$/, '')
const fileExtension = file.split('.').pop()
if (
fileNameWithoutExtension !== 'index' &&
(fileExtension === 'js' || fileExtension === 'ts')
)
importedModules[fileNameWithoutExtension] = require(`./${file}`)[
fileNameWithoutExtension
]
}
module.exports = importedModules

View File

@@ -0,0 +1,14 @@
const stringifyError = (error = {}) => {
return JSON.stringify(error, Object.getOwnPropertyNames(error))
}
const potionErrorObj = (error = {}, details = {}) => {
const stringifiedError = stringifyError(error)
const stringifiedObject = JSON.stringify({
error: stringifiedError,
details,
})
return stringifiedObject
}
module.exports = { potionErrorObj, stringifyError }

View File

@@ -0,0 +1,4 @@
const VoiceCloning = require('./voice_cloning_model')
const VoiceCloningService = require('./voice_cloning_service')
module.exports = VoiceCloningService(VoiceCloning)

View File

@@ -0,0 +1,44 @@
const mongoose = require('mongoose')
const Schema = mongoose.Schema
const VoiceCloningSchema = Schema(
{
userId: {
type: Schema.Types.ObjectId,
ref: 'User',
required: true,
},
userAudioProfileId: {
type: Schema.Types.ObjectId,
ref: 'UserAudioProfile',
required: true,
},
status: {
type: String,
required: false,
default: 'created',
},
input: {
type: Schema.Types.Mixed,
default: null,
},
training_model: {
type: Schema.Types.Mixed,
default: null,
},
metadata: {
type: Schema.Types.Mixed,
default: null,
},
deleted: {
type: Boolean,
required: true,
default: false,
},
},
{
timestamps: true,
}
)
module.exports = mongoose.model('VoiceCloning', VoiceCloningSchema)

View File

@@ -0,0 +1,140 @@
const StringifyUtils = require('../utils/logService')
const create = (VoiceCloningModel) => async (data) => {
try {
const newModel = new VoiceCloningModel({ ...data })
const savedModel = await newModel.save()
return savedModel
} catch (error) {
const details = { data }
console.log(
'ERROR - VOICE CLONING SERVICE > create',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const insertMany = (VoiceCloningModel) => async (data) => {
try {
const inserted = await VoiceCloningModel.insertMany(data)
return inserted
} catch (error) {
const details = { data }
console.log(
'ERROR - VOICE CLONING SERVICE > insertMany',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const read = (VoiceCloningModel) => async (filter) => {
try {
const foundModel = await VoiceCloningModel.findOne({
...filter,
deleted: false,
})
return foundModel
} catch (error) {
const details = { filter }
console.log(
'ERROR - VOICE CLONING SERVICE > read',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const find = (VoiceCloningModel) => async (filter) => {
try {
const foundModels = await VoiceCloningModel.find({
...filter,
deleted: false,
})
return foundModels
} catch (error) {
const details = { filter }
console.log(
'ERROR - VOICE CLONING SERVICE > find',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const update = (VoiceCloningModel) => async (data) => {
try {
const updatedModel = await VoiceCloningModel.findOneAndUpdate(
{ _id: data._id },
data,
{
new: true,
}
)
return updatedModel
} catch (error) {
const details = { data }
console.log(
'ERROR - VOICE CLONING SERVICE > update',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const remove = (VoiceCloningModel) => async (filter) => {
try {
const updatedModel = await VoiceCloningModel.findOneAndUpdate(
{ ...filter },
{
$set: {
deleted: true,
},
},
{ new: true }
)
return updatedModel
} catch (error) {
const details = { filter }
console.log(
'ERROR - VOICE CLONING SERVICE > remove',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const removeMany = (VoiceCloningModel) => async (filter) => {
try {
const updatedModel = await VoiceCloningModel.updateMany(
{ ...filter },
{
$set: {
deleted: true,
},
},
{ new: true }
)
return updatedModel
} catch (error) {
const details = { filter }
console.log(
'ERROR - VOICE CLONING SERVICE > removeMany',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
module.exports = (VoiceCloningModel) => {
return {
create: create(VoiceCloningModel),
insertMany: insertMany(VoiceCloningModel),
read: read(VoiceCloningModel),
remove: remove(VoiceCloningModel),
removeMany: removeMany(VoiceCloningModel),
update: update(VoiceCloningModel),
find: find(VoiceCloningModel),
}
}

View File

@@ -0,0 +1,21 @@
{
"name": "potion-voice",
"version": "1.0.0",
"description": "This will handle the voice cloning jobs",
"main": "index.js",
"scripts": {},
"dependencies": {
"@bugsnag/js": "^7.3.5",
"aws-sdk": "^2.752.0",
"fs-extra": "^9.0.1",
"mongoose": "^6.8.0",
"pm2": "^5.2.0",
"rimraf": "^3.0.2",
"uuid": "^8.3.2"
},
"devDependencies": {
"aws-code-deploy": "^1.0.11"
},
"author": "potion Team",
"license": "ISC"
}

View File

@@ -0,0 +1,12 @@
protobuf>=3.8.0
uuid
numpy
torch
torchvision
torchaudio
tensorboard
tensorboardx
requests
git+https://scrubbed_1@example.com/potion/potion-voice-utils.git
resemblyzer
textdistance

View File

@@ -0,0 +1,12 @@
protobuf>=3.8.0
uuid
numpy
torch==1.12.1+cu116 # requires --extra-index-url https://download.pytorch.org/whl/cu116
torchvision==0.13.1+cu116 # requires --extra-index-url https://download.pytorch.org/whl/cu116
torchaudio==0.12.1 # requires --extra-index-url https://download.pytorch.org/whl/cu116
tensorboard
tensorboardx
requests
git+https://scrubbed_1@example.com/potion/potion-voice-utils.git
resemblyzer
textdistance

View File

@@ -0,0 +1,10 @@
protobuf>=3.8.0
uuid
numpy
torch==1.9.1
torchvision==0.10.1
torchaudio~=0.9.0
requests
git+https://scrubbed_1@example.com/potion/potion-voice-utils.git
resemblyzer
textdistance

View File

@@ -0,0 +1,10 @@
protobuf>=3.8.0
uuid
numpy
torch
torchvision
torchaudio
requests
git+https://scrubbed_1@example.com/potion/potion-voice-utils.git
resemblyzer
textdistance

View File

@@ -0,0 +1,11 @@
protobuf>=3.8.0
uuid
numpy
torch==1.9.1+cu111 # requires --find-links https://download.pytorch.org/whl/torch_stable.html
torchvision==0.10.1+cu111 # requires --find-links https://download.pytorch.org/whl/torch_stable.html
torchaudio~=0.9.0 # requires --find-links https://download.pytorch.org/whl/torch_stable.html
tensorboard
tensorboardx
requests
git+https://scrubbed_1@example.com/potion/potion-voice-utils.git
resemblyzer

View File

@@ -0,0 +1,332 @@
const fs = require('fs')
const https = require('https')
const exec = require('child_process').exec
const AWS = require('aws-sdk')
const Bugsnag = require('@bugsnag/js')
const mongoose = require('mongoose')
const version = require('./package.json').version
const sqs = require('../app/services/sqs')
const s3 = require('../app/services/s3')
const voiceCloningService = require('./voice_cloning')
const userAudioProfileService = require('./user_audio_profile')
AWS.config.update({ region: 'us-west-2' })
const sqsQueueUrl = process.env.SQS_URL
const mongoUriDev = process.env.MONGODB_URI_DEV
const mongoUriStaging = process.env.MONGODB_URI_STAGING
const mongoUriProd = process.env.MONGODB_URI_PROD
let throttleMessageFetching = true
const APP_ENV = process.env.POTION_APP_ENV
const cloudFrontUrlProd = process.env.CLOUDFRONT_URL_PROD
const cloudFrontUrlDev = process.env.CLOUDFRONT_URL_DEV
const cloudFrontUrlStaging = process.env.CLOUDFRONT_URL_STAGING
const updateUrl = (str, cloudFrontUrl) => {
const host = new URL(str).host
return str.replace(`https://${host}`, cloudFrontUrl)
}
function connectDB(dbUri, retryCount = 0) {
return new Promise((resolve, reject) => {
console.log('Connection Attempt : ', retryCount)
mongoose.set('strictQuery', true)
mongoose
.connect(dbUri)
.then((msg) => {
console.log('Connected to Mongo DB !')
resolve()
})
.catch((err) => {
console.log('Failed to connect dns mongo: ', err)
if (retryCount < 6) {
retryCount++
connectDB(dbUri, retryCount)
}
})
})
}
function execShellCommand(cmd, logPath) {
// const exec = require("child_process").exec;
return new Promise((resolve, reject) => {
exec(cmd, { maxBuffer: 1024 * 1000000 }, async (error, stdout, stderr) => {
if (error) {
console.log('Error while proccessing python command', error)
reject(error)
}
// console.log('Stdout --- ', stdout)
// console.log('Stderror --- ', stderr)
await fs.promises.writeFile(`${logPath}/error.log`, stderr)
await fs.promises.writeFile(`${logPath}/info.log`, stdout)
resolve()
})
})
}
async function getFile(waveUrl, path) {
return new Promise((resolve) => {
https.get(waveUrl, (res) => {
const writeStream = fs.createWriteStream(path)
res.pipe(writeStream)
writeStream.on('finish', () => {
writeStream.close()
resolve()
})
})
})
}
function pad(s) {
while (s.length < 3) s = '0' + s // IN future we will need padding to 4
return s
}
const processQueue = () => {
/* eslint-disable no-async-promise-executor */
return new Promise(async (resolve, reject) => {
try {
const response = await sqs.fetchMessageFromSQS(sqsQueueUrl)
if (
typeof response.Messages !== 'undefined' &&
response.Messages.length > 0
) {
throttleMessageFetching = false
const job = JSON.parse(response.Messages[0].Body)
const receiptHandle = response.Messages[0].ReceiptHandle
console.log('job===', job)
const { metadata, input, _id, userAudioProfileId } = job._doc
console.log('userAudioProfileId', userAudioProfileId)
console.log('_id', _id)
const { env } = job
console.log('env', env)
console.log('metadata------', metadata)
console.log('input', input)
const DB_URI =
env === 'production'
? mongoUriProd
: env === 'staging'
? mongoUriStaging
: mongoUriDev
console.log('DB_URI ', DB_URI)
await connectDB(DB_URI)
const cloudFrontUrl =
env === 'production'
? cloudFrontUrlProd
: env === 'staging'
? cloudFrontUrlStaging
: cloudFrontUrlDev
try {
await sqs.deleteMessageFromSQS(sqsQueueUrl, receiptHandle)
const { directoryName } = metadata
console.log('directoryName', directoryName)
const logPath = `/mnt/efs/potion-voice/${env}/${directoryName}`
if (!fs.existsSync(logPath)) {
fs.mkdirSync(logPath, { recursive: true })
}
// update the db model to processing
await voiceCloningService.update({ _id, status: 'processing' })
await userAudioProfileService.update({
_id: userAudioProfileId,
status: 'processing',
})
// create directory for userid-useraudioprofileid if not exist
const rootPath = `/tmp/${directoryName}`
const wavePath = `${rootPath}/wav48/1`
if (!fs.existsSync(wavePath)) {
fs.mkdirSync(wavePath, { recursive: true })
}
const txtPath = `${rootPath}/txt/1`
if (!fs.existsSync(txtPath)) {
fs.mkdirSync(txtPath, { recursive: true })
}
// download the training data files and put it in respective directories
for (let index = 0; index < input.length; index++) {
const item = input[index]
const { waveUrl, originalText } = item
// download wave file
const waveFilePath = `${wavePath}/1_${pad('' + (index + 1))}.wav`
await getFile(updateUrl(waveUrl, cloudFrontUrl), waveFilePath)
const txtFilePath = `${txtPath}/1_${pad('' + (index + 1))}.txt`
await fs.promises.writeFile(txtFilePath, originalText)
}
const zipFileName = directoryName + '.tgz'
// /tmp/directoryName.tgz
await execShellCommand(
`cd /tmp && tar czvf ${zipFileName} ${directoryName}`,
logPath
)
console.log('ZIP created ', zipFileName)
// re-sample audio
const SAMPLING_LABEL = `Time Taken for re-sampling ${directoryName}`
console.time(SAMPLING_LABEL)
const outputPath = `/mnt/efs/potion-voice/${env}/${directoryName}`
const samplingCommand = `python3 ../voice-cloning/prepare_datasets.py --dataset_preset potion_voice_cloning --dataset_archive_path /tmp/${zipFileName} --output_path ${outputPath}`
console.log('samplingCommand ', samplingCommand)
const samplingResponse = await execShellCommand(
samplingCommand,
logPath
)
console.timeEnd(SAMPLING_LABEL)
// /mnt/efs/potion-voice/${env}/speakrs.pth
// /mnt/efs/potion-voice/${env}/txt
// /mnt/efs/potion-voice/${env}/${directoryName}/wav
const outPath = `/mnt/efs/potion-voice/${env}/${directoryName}/sr22050/${directoryName}`
const resultsPath = outPath + '/results'
//update pth file for cloning
// clone the voice
const VOICE_CLONING_LABEL = `Time Taken for voice cloning ${directoryName}`
console.time(VOICE_CLONING_LABEL)
const trainingModelCommand = `python3 ../voice-cloning/clone_voice.py --baseline_model_path ../voice-cloning/pretrained-models/checkpoint_365000.pth --speaker_dataset_path ${outPath} --speaker_embeddings_path ${
outPath + '/speakers.pth'
} --output_path ${resultsPath}`
console.log('Training Model Command', trainingModelCommand)
const trainingResponse = await execShellCommand(
trainingModelCommand,
logPath
)
console.timeEnd(VOICE_CLONING_LABEL)
let generatedDirectoryName = ''
fs.readdirSync(`${resultsPath}/`).forEach((file) => {
if (file.includes('vits_potion_clone'))
// use output from above to get right path and directory name
generatedDirectoryName = file
})
// minimize cloning model
const VOICE_MINIMIZE_LABEL = `Time Taken for voice minimizing cloning ${directoryName}`
console.time(VOICE_MINIMIZE_LABEL)
const minimizeCloningModelCommand = `python3 ../voice-cloning/minimize_cloned_voice_model.py --voice_model_asset_path ${
resultsPath + '/' + generatedDirectoryName + '/'
} --voice_model_name checkpoint_365200.pth`
console.log(
'Minimize Cloning Model Command',
minimizeCloningModelCommand
)
const minimizeCloning = await execShellCommand(
minimizeCloningModelCommand,
logPath
)
console.timeEnd(VOICE_MINIMIZE_LABEL)
// Add the code to update location of generated model and status into DB
await voiceCloningService.update({ _id, status: 'completed' })
const training_model_path = {
voice_model_path: `${resultsPath}/${generatedDirectoryName}/checkpoint_365200.pth`,
voice_model_config_path: `${resultsPath}/${generatedDirectoryName}/config.json`,
voice_model_speakers_file_path: `${outPath}/speakers.pth`, // TODO update the name to voice model speakers embeddings
voice_model_light_path: `${resultsPath}/${generatedDirectoryName}/checkpoint_365200_light.pth`,
voice_model_config_light_path: `${resultsPath}/${generatedDirectoryName}/config_light.json`,
}
await userAudioProfileService.update({
_id: userAudioProfileId,
status: 'completed',
training_model_path,
})
// add code to put that model into S3
let keys = Object.keys(training_model_path)
const training_model_s3_path = {}
for (let index = 0; index < keys.length; index++) {
const path = training_model_path[keys[index]]
const s3Path = await s3.upload({
filePath: path,
fileName: `${directoryName}/${path.split('/').pop()}`,
bucket: `potion-voice-users-training-model/${env}`,
})
training_model_s3_path[keys[index]] = s3Path
}
// add S3 path to user audio profile model
await userAudioProfileService.update({
_id: userAudioProfileId,
training_model_s3_path,
})
} catch (error) {
console.log('error********************', error)
Bugsnag.notify(
new Error(
`Unable to train for voice cloning videos ` + JSON.stringify(job)
)
)
Bugsnag.notify(error)
// update the db to set status as error
await voiceCloningService.update({ _id, status: 'error' })
await userAudioProfileService.update({
_id: userAudioProfileId,
status: 'error',
})
resolve() // to continue working on new jobs
}
} else {
throttleMessageFetching = true
}
resolve()
} catch (error) {
console.error('Error while training voice clone', { error })
Bugsnag.notify(error)
resolve() // to continue working on new jobs
} finally {
mongoose.connection.close()
}
})
}
function sleep(ms) {
return new Promise((resolve) => {
setTimeout(resolve, ms)
})
}
const init = async () => {
console.log('potion Voice Clone Process Started')
Bugsnag.start({
appVersion: APP_ENV + version,
apiKey: process.env.BUGSNAG_BACKEND_KEY,
releaseStage: process.env.NODE_ENV,
})
try {
while (true) {
await processQueue()
if (throttleMessageFetching) await sleep(2000)
}
} catch (error) {
Bugsnag.notify(error)
}
}
init()

View File

@@ -0,0 +1,24 @@
{
"name": "voice-cloning-job-handler",
"version": "1.0.0",
"description": "This will handle the voice cloning jobs",
"main": "index.js",
"scripts": {
"deploy-production": "npx dotenv-cli -e ./app-scripts/env-aws-code-deploy/.env.production.aws-code-deploy node ./app-scripts/deploy-scripts/deploy-production.js",
"deploy-staging": "npx dotenv-cli -e ./app-scripts/env-aws-code-deploy/.env.staging.aws-code-deploy node ./app-scripts/deploy-scripts/deploy-staging.js"
},
"dependencies": {
"@bugsnag/js": "^7.3.5",
"aws-sdk": "^2.752.0",
"fs-extra": "^9.0.1",
"mongoose": "^6.8.0",
"pm2": "^5.2.0",
"rimraf": "^3.0.2",
"uuid": "^8.3.2"
},
"devDependencies": {
"aws-code-deploy": "^1.0.11"
},
"author": "potion Team",
"license": "ISC"
}

View File

@@ -0,0 +1,18 @@
apps:
- name: training-model
script: index.js
watch: false
autorestart: true
instances: 1
time: true
env:
NODE_ENV: 'staging'
POTION_APP_ENV: 'staging'
SQS_URL: 'https://sqs.us-west-2.amazonaws.com/[REDACTED_AWS_ACCOUNT_1961]/potion-voice-clone-ai-staging.fifo'
BUGSNAG_BACKEND_KEY: '[REDACTED_generic-api-key]'
MONGODB_URI_DEV: 'mongodb+srv://[REDACTED_MONGO_USER_deve]:scrubbed_1@example.com7.mongodb.net/potion_development?retryWrites=true&w=majority'
MONGODB_URI_STAGING: 'mongodb+srv://[REDACTED_MONGO_USER_stag]:scrubbed_2@example.com7.mongodb.net/potion_staging?retryWrites=true&w=majority'
MONGODB_URI_PROD: ''
CLOUDFRONT_URL_PROD: 'https://videoassets.sendpotion.com'
CLOUDFRONT_URL_STAGING: ''
CLOUDFRONT_URL_DEV: 'https://d2rmbzmoml90gd.cloudfront.net'

View File

@@ -0,0 +1,18 @@
apps:
- name: training-model
script: index.js
watch: false
autorestart: true
instances: 1
time: true
env:
NODE_ENV: 'production'
POTION_APP_ENV: 'production'
SQS_URL: 'https://sqs.us-west-2.amazonaws.com/[REDACTED_AWS_ACCOUNT_1961]/potion-voice-clone-ai-production.fifo'
BUGSNAG_BACKEND_KEY: '[REDACTED_generic-api-key]'
MONGODB_URI_DEV: 'mongodb+srv://[REDACTED_MONGO_USER_deve]:scrubbed_1@example.com7.mongodb.net/potion_development?retryWrites=true&w=majority'
MONGODB_URI_STAGING: 'mongodb+srv://[REDACTED_MONGO_USER_stag]:scrubbed_2@example.com7.mongodb.net/potion_staging?retryWrites=true&w=majority'
MONGODB_URI_PROD: 'mongodb+srv://[REDACTED_MONGO_USER_prod]:scrubbed_3@example.com.net/potion_production?retryWrites=true&w=majority'
CLOUDFRONT_URL_PROD: 'https://videoassets.sendpotion.com'
CLOUDFRONT_URL_STAGING: ''
CLOUDFRONT_URL_DEV: 'https://d2rmbzmoml90gd.cloudfront.net'

View File

@@ -0,0 +1,4 @@
const UserAudioProfile = require('./user_audio_profile_model')
const UserAudioProfileService = require('./user_audio_profile_service')
module.exports = UserAudioProfileService(UserAudioProfile)

View File

@@ -0,0 +1,40 @@
const mongoose = require('mongoose')
const Schema = mongoose.Schema
const UserAudioProfileSchema = Schema(
{
userId: {
type: Schema.Types.ObjectId,
ref: 'User',
required: true,
},
name: {
type: String,
required: true,
default: '',
},
status: {
type: String,
required: false,
default: 'created',
},
training_model_path: {
type: Schema.Types.Mixed,
default: null,
},
training_model_s3_path: {
type: Schema.Types.Mixed,
default: null,
},
deleted: {
type: Boolean,
required: true,
default: false,
},
},
{
timestamps: true,
}
)
module.exports = mongoose.model('UserAudioProfile', UserAudioProfileSchema)

View File

@@ -0,0 +1,140 @@
const StringifyUtils = require('../../app/services/utils/logService')
const create = (UserAudioProfileModel) => async (data) => {
try {
const newModel = new UserAudioProfileModel({ ...data })
const savedModel = await newModel.save()
return savedModel
} catch (error) {
const details = { data }
console.log(
'ERROR - USER AUDIO PROFILE SERVICE > create',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const insertMany = (UserAudioProfileModel) => async (data) => {
try {
const inserted = await UserAudioProfileModel.insertMany(data)
return inserted
} catch (error) {
const details = { data }
console.log(
'ERROR - USER AUDIO PROFILE SERVICE > insertMany',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const read = (UserAudioProfileModel) => async (filter) => {
try {
const foundModel = await UserAudioProfileModel.findOne({
...filter,
deleted: false,
})
return foundModel
} catch (error) {
const details = { filter }
console.log(
'ERROR - USER AUDIO PROFILE SERVICE > read',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const find = (UserAudioProfileModel) => async (filter) => {
try {
const foundModels = await UserAudioProfileModel.find({
...filter,
deleted: false,
})
return foundModels
} catch (error) {
const details = { filter }
console.log(
'ERROR - USER AUDIO PROFILE SERVICE > find',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const update = (UserAudioProfileModel) => async (data) => {
try {
const updatedModel = await UserAudioProfileModel.findOneAndUpdate(
{ _id: data._id },
data,
{
new: true,
}
)
return updatedModel
} catch (error) {
const details = { data }
console.log(
'ERROR - USER AUDIO PROFILE SERVICE > update',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const remove = (UserAudioProfileModel) => async (filter) => {
try {
const updatedModel = await UserAudioProfileModel.findOneAndUpdate(
{ ...filter },
{
$set: {
deleted: true,
},
},
{ new: true }
)
return updatedModel
} catch (error) {
const details = { filter }
console.log(
'ERROR - USER AUDIO PROFILE SERVICE > remove',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const removeMany = (UserAudioProfileModel) => async (filter) => {
try {
const updatedModel = await UserAudioProfileModel.updateMany(
{ ...filter },
{
$set: {
deleted: true,
},
},
{ new: true }
)
return updatedModel
} catch (error) {
const details = { filter }
console.log(
'ERROR - USER AUDIO PROFILE SERVICE > removeMany',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
module.exports = (UserAudioProfileModel) => {
return {
create: create(UserAudioProfileModel),
insertMany: insertMany(UserAudioProfileModel),
read: read(UserAudioProfileModel),
remove: remove(UserAudioProfileModel),
removeMany: removeMany(UserAudioProfileModel),
update: update(UserAudioProfileModel),
find: find(UserAudioProfileModel),
}
}

View File

@@ -0,0 +1,4 @@
const VoiceCloning = require('./voice_cloning_model')
const VoiceCloningService = require('./voice_cloning_service')
module.exports = VoiceCloningService(VoiceCloning)

View File

@@ -0,0 +1,44 @@
const mongoose = require('mongoose')
const Schema = mongoose.Schema
const VoiceCloningSchema = Schema(
{
userId: {
type: Schema.Types.ObjectId,
ref: 'User',
required: true,
},
userAudioProfileId: {
type: Schema.Types.ObjectId,
ref: 'UserAudioProfile',
required: true,
},
status: {
type: String,
required: false,
default: 'created',
},
input: {
type: Schema.Types.Mixed,
default: null,
},
training_model: {
type: Schema.Types.Mixed,
default: null,
},
metadata: {
type: Schema.Types.Mixed,
default: null,
},
deleted: {
type: Boolean,
required: true,
default: false,
},
},
{
timestamps: true,
}
)
module.exports = mongoose.model('VoiceCloning', VoiceCloningSchema)

View File

@@ -0,0 +1,141 @@
const StringifyUtils = require('../../app/services/utils/logService')
const create = (VoiceCloningModel) => async (data) => {
try {
const newModel = new VoiceCloningModel({ ...data })
const savedModel = await newModel.save()
return savedModel
} catch (error) {
const details = { data }
console.log(
'ERROR - VOICE CLONING SERVICE > create',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const insertMany = (VoiceCloningModel) => async (data) => {
try {
const inserted = await VoiceCloningModel.insertMany(data)
return inserted
} catch (error) {
const details = { data }
console.log(
'ERROR - VOICE CLONING SERVICE > insertMany',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const read = (VoiceCloningModel) => async (filter) => {
try {
const foundModel = await VoiceCloningModel.findOne({
...filter,
deleted: false,
})
return foundModel
} catch (error) {
const details = { filter }
console.log(
'ERROR - VOICE CLONING SERVICE > read',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const find = (VoiceCloningModel) => async (filter) => {
try {
const foundModels = await VoiceCloningModel.find({
...filter,
deleted: false,
})
return foundModels
} catch (error) {
const details = { filter }
console.log(
'ERROR - VOICE CLONING SERVICE > find',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const update = (VoiceCloningModel) => async (data) => {
try {
const updatedModel = await VoiceCloningModel.findOneAndUpdate(
{ _id: data._id },
data,
{
new: true,
}
)
return updatedModel
} catch (error) {
const details = { data }
console.log(
'ERROR - VOICE CLONING SERVICE > update',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const remove = (VoiceCloningModel) => async (filter) => {
try {
const updatedModel = await VoiceCloningModel.findOneAndUpdate(
{ ...filter },
{
$set: {
deleted: true,
},
},
{ new: true }
)
return updatedModel
} catch (error) {
const details = { filter }
console.log(
'ERROR - VOICE CLONING SERVICE > remove',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const removeMany = (VoiceCloningModel) => async (filter) => {
try {
const updatedModel = await VoiceCloningModel.updateMany(
{ ...filter },
{
$set: {
deleted: true,
},
},
{ new: true }
)
return updatedModel
} catch (error) {
const details = { filter }
console.log(
'ERROR - VOICE CLONING SERVICE > removeMany',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
module.exports = (VoiceCloningModel) => {
return {
create: create(VoiceCloningModel),
insertMany: insertMany(VoiceCloningModel),
read: read(VoiceCloningModel),
remove: remove(VoiceCloningModel),
removeMany: removeMany(VoiceCloningModel),
update: update(VoiceCloningModel),
find: find(VoiceCloningModel),
}
}

View File

@@ -0,0 +1,121 @@
{
"model": "speaker_encoder",
"run_name": "speaker_encoder",
"run_description": "resnet speaker encoder trained with commonvoice all languages dev and train, Voxceleb 1 dev and Voxceleb 2 dev",
"epochs": 100000,
"batch_size": null,
"eval_batch_size": null,
"mixed_precision": false,
"run_eval": true,
"test_delay_epochs": 0,
"print_eval": false,
"print_step": 50,
"tb_plot_step": 100,
"tb_model_param_stats": false,
"save_step": 1000,
"checkpoint": true,
"keep_all_best": false,
"keep_after": 10000,
"num_loader_workers": 8,
"num_val_loader_workers": 0,
"use_noise_augment": false,
"output_path": "../checkpoints/speaker_encoder/language_balanced/normalized/angleproto-4-samples-by-speakers/",
"distributed_backend": "nccl",
"distributed_url": "tcp://localhost:54321",
"audio": {
"fft_size": 512,
"win_length": 400,
"hop_length": 160,
"frame_shift_ms": null,
"frame_length_ms": null,
"stft_pad_mode": "reflect",
"sample_rate": 16000,
"resample": false,
"preemphasis": 0.97,
"ref_level_db": 20,
"do_sound_norm": false,
"do_trim_silence": false,
"trim_db": 60,
"power": 1.5,
"griffin_lim_iters": 60,
"num_mels": 64,
"mel_fmin": 0.0,
"mel_fmax": 8000.0,
"spec_gain": 20,
"signal_norm": false,
"min_level_db": -100,
"symmetric_norm": false,
"max_norm": 4.0,
"clip_norm": false,
"stats_path": null,
"do_rms_norm": true,
"db_level": -27.0
},
"datasets": [
{
"name": "voxceleb2",
"path": "/workspace/scratch/ecasanova/datasets/VoxCeleb/vox2_dev_aac/",
"meta_file_train": null,
"ununsed_speakers": null,
"meta_file_val": null,
"meta_file_attn_mask": "",
"language": "voxceleb"
}
],
"model_params": {
"model_name": "resnet",
"input_dim": 64,
"use_torch_spec": true,
"log_input": true,
"proj_dim": 512
},
"audio_augmentation": {
"p": 0.5,
"rir": {
"rir_path": "/workspace/store/ecasanova/ComParE/RIRS_NOISES/simulated_rirs/",
"conv_mode": "full"
},
"additive": {
"sounds_path": "/workspace/store/ecasanova/ComParE/musan/",
"speech": {
"min_snr_in_db": 13,
"max_snr_in_db": 20,
"min_num_noises": 1,
"max_num_noises": 1
},
"noise": {
"min_snr_in_db": 0,
"max_snr_in_db": 15,
"min_num_noises": 1,
"max_num_noises": 1
},
"music": {
"min_snr_in_db": 5,
"max_snr_in_db": 15,
"min_num_noises": 1,
"max_num_noises": 1
}
},
"gaussian": {
"p": 0.0,
"min_amplitude": 0.0,
"max_amplitude": 1e-05
}
},
"storage": {
"sample_from_storage_p": 0.5,
"storage_size": 40
},
"max_train_step": 1000000,
"loss": "angleproto",
"grad_clip": 3.0,
"lr": 0.0001,
"lr_decay": false,
"warmup_steps": 4000,
"wd": 1e-06,
"steps_plot_stats": 100,
"num_speakers_in_batch": 100,
"num_utters_per_speaker": 4,
"skip_speakers": true,
"voice_len": 2.0
}

View File

@@ -0,0 +1,226 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import sys
import os
import argparse
import torch
# load coqui-ai/trainer libraries
from trainer import Trainer, TrainerArgs
# load coqui-ai/TTS libraries
from TTS.tts.configs.shared_configs import BaseDatasetConfig
from TTS.tts.configs.vits_config import VitsConfig
from TTS.tts.datasets import load_tts_samples
from TTS.tts.models.vits import Vits, VitsArgs, VitsAudioConfig
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Code to clone a voice from a given set of voice samples and a multi-speaker baseline model")
parser.add_argument("--baseline_model_path", type = str, required = True,
help = "Path to multi-speaker baseline model (VITS model)")
parser.add_argument("--speaker_dataset_path", type = str, required = True,
help = "Path to voice cloning dataset")
parser.add_argument("--speaker_embeddings_path", type = str, required = True,
help = "Path to speaker's embeddings file")
parser.add_argument("--output_path", type = str, default = "results/cloned-voices",
help = "Path to store trained / generated assets")
parser.add_argument("--batch_size", type = int, default = 96, # 96 is suitable for AWS g5 instances
help = "Batch size for training run")
parser.add_argument("--max_epochs", type = int, default = 200, # 200 for batch_size 96 (with the 22.050 sampling rate multi-speaker model
help = "Maximum number of epochs for training run") # 2000 for batch_size 64 and 1500 for batch_size 96 (with the initial 16k sampling rate VCTK 0.80 model)
parser.add_argument("--use_cpu", default = False, action = "store_true", # untested!!!
help = "Signal that CPU should be used even if a CUDA-device is available")
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
return parser.parse_args()
#
# main training method (voice cloning)
#
def main(args):
if args.output_format == "txt":
print("Commencing training of a new multi-speaker potion-voice baseline model:")
print("")
print(" + Baseline multi-speaker model path: {}" . format(args.baseline_model_path))
print(" + Voice training dataset path : {}" . format(args.speaker_dataset_path))
print(" + Speaker embeddings path : {}" . format(args.speaker_embeddings_path))
print(" + Output path : {}" . format(args.output_path))
print(" + Batch size : {}" . format(args.batch_size))
print(" + Training runs (max epochs) : {}" . format(args.max_epochs))
print("")
# determine whether CUDA support is available and set device parameters accordingly
use_cuda = torch.cuda.is_available()
if args.output_format == "txt":
print(" + CUDA availability : {}" . format(use_cuda))
if args.use_cpu:
device = "cpu"
device_torch = False
elif use_cuda:
device = "cuda"
device_torch = torch.device("cuda")
else:
device = "cpu"
device_torch = False
if args.output_format == "txt":
print(" + Compute device used : {}" . format(device))
print("")
# define training data set
dataset_config = BaseDatasetConfig(formatter = "vctk_old", language = "en-us", path = args.speaker_dataset_path)
# set VITS training parameters
audio_config = VitsAudioConfig(
sample_rate = 22050,
win_length = 1024,
hop_length = 256,
num_mels = 80,
mel_fmin = 0,
mel_fmax = None,
)
vitsArgs = VitsArgs(
use_speaker_embedding = False,
use_d_vector_file = True,
d_vector_file = [args.speaker_embeddings_path],
d_vector_dim = 512,
num_layers_text_encoder = 10
)
config = VitsConfig(
model_args = vitsArgs,
audio = audio_config,
run_name = "vits_potion_clone",
use_speaker_embedding = False,
use_d_vector_file = True,
d_vector_file = [args.speaker_embeddings_path],
d_vector_dim = 512,
batch_size = args.batch_size,
eval_batch_size = 8,
batch_group_size = 0, # changing this to 5 (VITS training default) slows training down, but doesn't have any positive training effects
num_loader_workers = 4,
num_eval_loader_workers = 4,
run_eval = True,
eval_split_size = 2, # fix size of eval dataset (default 1% approach requires at least 100 voice samples!)
test_delay_epochs = -1,
epochs = args.max_epochs,
text_cleaner = "english_cleaners",
use_phonemes = False,
phoneme_language = "en-us",
phoneme_cache_path = os.path.join(args.output_path, "phoneme_cache"),
compute_input_seq_cache = True,
print_step = 50,
print_eval = True,
mixed_precision = True,
max_text_len = 325,
output_path = args.output_path,
save_checkpoints = True,
save_step = 200,
datasets = [dataset_config],
cudnn_benchmark = False,
#characters = {
# "pad": "_",
# "eos": "&",
# "bos": "*",
# "characters": "!¡'(),-.:;¿?abcdefghijklmnopqrstuvwxyz «°±µ»$%&‘’‚“`”„",
# "punctuations": "!¡'(),-.:;¿? ",
# "phonemes": None,
# "unique": True
#},
test_sentences = [
["It took me quite a long time to develop a voice, and now that I have it I'm not going to be silent."],
["Be a voice, not an echo."],
["I'm sorry Dave. I'm afraid I can't do that."],
["This cake is great. It's so delicious and moist."],
["Prior to November 22, 1963."],
["Hey! Sandra."],
["Hey! Andrew."],
["Hey, Michelle."],
["Hey! George."],
["Hey there, Rachel."]
]
)
# load training samples
train_samples, eval_samples = load_tts_samples(config.datasets, eval_split = True, eval_split_max_size = config.eval_split_max_size, eval_split_size = config.eval_split_size)
# init VITS model
model = Vits.init_from_config(config)
# init voice cloning
trainer = Trainer(
TrainerArgs(restore_path = args.baseline_model_path, use_ddp = False),
config,
args.output_path,
model = model,
train_samples = train_samples,
eval_samples = eval_samples
)
# trigger voice cloning (aka single speaker training)
try:
trainer.fit()
except (KeyboardInterrupt, SystemExit):
print("Training stopped manually (via keyboard interrupt)! Bye.")
exit(0)
# determine required adjustment for speech synthesizing (i.e., the scaling factor for the duration predictor)
# take the duration of the test sentence and calculate the difference to corresponding reference samples
# set config.model_args["length_scale"] accordingly and save the updated config asset
# exit gracefully
if args.output_format == "txt":
print("")
print("Completed voice cloning. The resulting model(s) can be found at:")
print(" --> {}" . format(args.output_path))
print("")
print("Done; bye.")
print("")
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
# clear command line arguments to avoid triggering argparse features part of Trainer / coqpit imports
# Traceback (most recent call last):
# File "train_multispeaker_baseline_model.py", line 208, in <module>
# main(args)
# File "train_multispeaker_baseline_model.py", line 177, in main
# trainer = Trainer(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 360, in __init__
# config, new_fields = self.init_training(args, coqpit_overrides, config)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 594, in init_training
# config.parse_known_args(coqpit_overrides, relaxed_parser=True)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 843, in parse_known_args
# parser = self.init_argparse(arg_prefix=arg_prefix, relaxed_parser=relaxed_parser)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 881, in init_argparse
# _init_argparse(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 529, in _init_argparse
# parser = _init_argparse(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 550, in _init_argparse
# return default.init_argparse(
# AttributeError: 'str' object has no attribute 'init_argparse'
sys.argv = [sys.argv[0]]
# ensure the output path exists
os.makedirs(args.output_path, exist_ok = True)
main(args)
### USAGE:
### $ python3 TTS/TTS/bin/resample.py --input_dir voice_dataset_path/person_82/wav48/1 --output_sr 16000
### $ python3 clone_voice.py [with argument]

View File

@@ -0,0 +1,764 @@
# potion-voice **voice-cloning** *Installation and Usage Guide*
In this guide, you will find more detailed instructions and examples for the following tasks:
+ Setting up a new AWS GPU-backed EC2 instance suitable for training new potion-voice models;
+ Setting up software environment and (optionally) prepare data sets for training new potion-voice models;
+ Training and evaluating new potion-voice models; and
+ Usage examples for voice cloning and speech synthesizing.
## Set Up AWS GPU-backed Compute Node (non-production)
1. Set up baseline & connect to remote node:
+ GPU-enabled Compute Node (e.g., g5.2xlarge by default)
+ We recommend a GPU-enabled Compute Node with 256GB root partition (volume type: gp3; 64GB for swapfile) and 512GB secondary SDD holding all dev / data files)
+ Inbound ports: SSH and TensorBoard (e.g., port 6006)
+ Ubuntu 22.04 LTS (Server) Installation
+ SSH into the EC2 instance
1. Secure / update baseline
```sh
$ sudo apt-get update
$ sudo apt-get upgrade
$ sudo apt-get install linux-aws linux-headers-aws linux-image-aws
```
1. Disable unattended upgrades. Enter the below command and select 'No'. These Upgrades might cause version mismatch between nvidia-drivers and cuda.
```sh
$ sudo dpkg-reconfigure -plow unattended-upgrades
Replacing config file /etc/apt/apt.conf.d/20auto-upgrades with new version
```
1. Set up secondary disk (used as dev / data volume)
```sh
$ sudo lsblk
NAME MAJ:MIN RM SIZE RO TYPE MOUNTPOINT
[...]
nvme1n1 259:0 0 500G 0 disk
[...]
$ sudo mkfs -t ext4 /dev/nvme1n1
mke2fs 1.45.5 (07-Jan-2020)
Creating filesystem with 524288000 4k blocks and 131072000 inodes
Filesystem UUID: 90327770-ba4d-4003-9136-964b4388ffb6
Superblock backups stored on blocks:
32768, 98304, 163840, 229376, 294912, 819200, 884736, 1605632, 2654208,
4096000, 7962624, 11239424, 20480000, 23887872, 71663616, 78675968,
102400000, 214990848, 512000000
Allocating group tables: done
Writing inode tables: done
Creating journal (262144 blocks): done
Writing superblocks and filesystem accounting information: done
$ mkdir DEV_PATH
```
+ Edit `/etc/fstab` and add
```txt
/dev/nvme1n1 DEV_PATH ext4 defaults,nofail 0 2
```
```sh
$ sudo mount -a
$ sudo chown -R ubuntu:ubuntu DEV_PATH
$ mkdir DEV_PATH/data
```
1. Create a swap file (training is memory intensive; so, add a swap file!)
+ Use the `dd` command to create a swap file on the root file system
+ Note: The size of the swap file is the block size option multiplied by the count option in the dd command. Adjust these values to determine the desired swap file size.
+ Note: The block size you specify should be less than the available memory on the instance or you receive a "memory exhausted" error.
+ Set up the swap file (of size 64 GB [512 MB x 128]).
```sh
$ sudo dd if=/dev/zero of=/swapfile bs=512M count=128
128+0 records in
128+0 records out
68719476736 bytes (69 GB, 64 GiB) copied, 336.416 s, 204 MB/s
```
+ Update the read and write permissions for the swap file:
```sh
$ sudo chmod 600 /swapfile
```
+ Set up a Linux swap area:
```sh
$ sudo mkswap /swapfile
Setting up swapspace version 1, size = 64 GiB
no label, UUID=1dfc20ce-ed64-4e69-8fa7-a800bbea4617
```
+ Make the swap file available for immediate use by adding the swap file to swap space:
```sh
$ sudo swapon /swapfile
```
+ Verify that the procedure was successful:
```sh
$ sudo swapon -s
Filename Type Size Used Priority
/swapfile file 67108860 0 -2
```
+ Enable the swap file at boot time by editing the `/etc/fstab` file. Add the following new line at the end of the file:
```txt
/swapfile swap swap defaults 0 0
```
1. Install NVIDIA drivers / CUDA support (pytorch required version 11.6 or 12)
```sh
wget https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.0-1_all.deb
sudo dpkg -i cuda-keyring_1.0-1_all.deb
sudo apt-get update
sudo apt-get -y install cuda-12-0
```
+ Reboot the instance and ensure all drivers load automatically
```sh
$ sudo reboot
```
+ Reconnect to the instance and verify NVIDIA drivers / CUDA support are as expected
```sh
$ nvidia-smi
Tue Jan 17 08:20:53 2023
+-----------------------------------------------------------------------------+
| NVIDIA-SMI 525.60.13 Driver Version: 525.60.13 CUDA Version: 12.0 |
|-------------------------------+----------------------+----------------------+
| GPU Name Persistence-M| Bus-Id Disp.A | Volatile Uncorr. ECC |
| Fan Temp Perf Pwr:Usage/Cap| Memory-Usage | GPU-Util Compute M. |
| | | MIG M. |
|===============================+======================+======================|
| 0 NVIDIA A10G On | 00000000:00:1E.0 Off | 0 |
| 0% 19C P8 16W / 300W | 0MiB / 23028MiB | 0% Default |
| | | N/A |
+-------------------------------+----------------------+----------------------+
+-----------------------------------------------------------------------------+
| Processes: |
| GPU GI CI PID Type Process name GPU Memory |
| ID ID Usage |
|=============================================================================|
| No running processes found |
+-----------------------------------------------------------------------------+
```
## Set Up Software Environment
1. Set up Python 3 (v3.10) development environment
```sh
$ sudo apt-get install python3-dev python3-pip python3-wheel python3-venv
```
1. Set up Phoneme back-end
```sh
$ sudo apt-get install espeak-ng espeak-ng-espeak
```
1. Set up required tools / standard dependencies
```sh
$ sudo apt-get install ffmpeg unzip git
```
1. Set up AWS Command Line Interface
```sh
$ sudo apt-get install awscli
$ aws configure
AWS Access Key ID [None]: xxxxxxxxxx
AWS Secret Access Key [None]: yyyyyyyyyy
Default region name [None]: us-west-2
Default output format [None]: json
$ aws configure set default.s3.max_concurrent_requests 50
```
1. (dev install only) Copy and extract training data sets from AWS
```sh
$ cd DEV_PATH/data
### VCTK v 0.92
$ aws s3 cp s3://potion-datasets/VCTK/VCTK-Corpus-0.92/VCTK-Corpus-0.92.tgz .
download: s3://potion-datasets/VCTK/VCTK-Corpus-0.92/VCTK-Corpus-0.92.tgz to ./VCTK-Corpus-0.92.tgz
$ tar -xzvf VCTK-Corpus-0.92.tgz
VCTK-Corpus-0.92/
VCTK-Corpus-0.92/README.txt
VCTK-Corpus-0.92/update.txt
VCTK-Corpus-0.92/license_text
VCTK-Corpus-0.92/txt/
[...]
VCTK-Corpus-0.92/wav48_silence_trimmed/p238/p238_191_mic1.flac
VCTK-Corpus-0.92/wav48_silence_trimmed/p238/p238_267_mic2.flac
$ rm VCTK-Corpus-0.92.tgz
### LibriTTS train-clean-360 subset
$ aws s3 cp s3://potion-datasets/LibriTTS/train-clean-360.tar.gz .
download: s3://potion-datasets/LibriTTS/train-clean-360.tar.gz to ./train-clean-360.tar.gz
$ tar -xzvf train-clean-360.tar.gz
./LibriTTS/train-clean-360/
./LibriTTS/train-clean-360/2272/
./LibriTTS/train-clean-360/2272/152265/
./LibriTTS/train-clean-360/2272/152265/2272_152265_000032_000001.original.txt
./LibriTTS/train-clean-360/2272/152265/2272_152265_000012_000001.wav
[...]
LibriTTS/reader_book.tsv
LibriTTS/speakers.tsv
$ rm train-clean-360.tar.gz
### Potion salutation recordings
$ aws s3 cp s3://potion-datasets/potion-voice-datasets/potion-salut-corpus_20221026.tgz .
download: s3://potion-datasets/potion-voice-datasets/potion-salut-corpus_20221026.tgz to ./potion-salut-corpus_20221019.tgz
$ tar -xzvf potion-salut-corpus_20221026.tgz
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/wav48/
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/wav48/POTION_6192d9c9a563df5c87ecb8bd/
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/wav48/POTION_6192d9c9a563df5c87ecb8bd/POTION_6192d9c9a563df5c87ecb8bd_334.wav
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/wav48/POTION_6192d9c9a563df5c87ecb8bd/POTION_6192d9c9a563df5c87ecb8bd_473.wav
[...]
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/txt/POTION_62d82d269cbde00027b66007/POTION_62d82d269cbde00027b66007_197.txt
potion-salut-corpus-94de499c-b770-4e4c-97fc-6add91befe1b/speaker-info.txt
$ rm potion-salut-corpus_20221026.tgz
```
1. Create a virtual potion-voice-cloner working environment
```sh
$ cd DEV_PATH
$ python3 -m venv potion-voice_venv
$ cd potion-voice_venv/
$ source bin/activate
(potion-voice_venv) $
```
1. Clone the potion-voice GitHub repository
```sh
(potion-voice_venv) $ cd DEV_PATH/potion-voice_venv/
(potion-voice_venv) $ python3 -m pip install --upgrade pip
(potion-voice_venv) $ git clone https://github.com/potion/potion-voice.git
```
1. Install potion-voice requirements (dependencies) and test that PyTorch is working with the GPU properly
```sh
(potion-voice_venv) $ cd DEV_PATH/potion-voice_venv/potion-voice/
(potion-voice_venv) $ python3 -m pip install -r ./requirements.dev.txt
(potion-voice_venv) $ python3
Python 3.10.6 (main, Nov 14 2022, 16:10:14) [GCC 11.3.0] on linux
Type "help", "copyright", "credits" or "license" for more information.
>>> import torch
>>> torch.cuda.is_available()
True
>>> torch.cuda.get_device_name(0)
'NVIDIA A10G'
>>> quit()
```
1. Install TTS dependencies
```sh
(potion-voice_venv) $ cd voice-cloning/
(potion-voice_venv) $ git clone --depth 1 --branch v0.10.2 https://github.com/coqui-ai/TTS
(potion-voice_venv) $ python3 -m pip install -e TTS/
```
+ Note 1: Installing requirements will ask for GitHub token twice! The second request is for a dependent package, which is also a private repo.
+ Note 2: Separate requirements files have been added for development (local versus AWS) and production usage (for GPU and CPU-only deployment).
## Training New potion-voice Models (Multi-speaker Baseline & Voice Cloning)
### Preprocess Dataset(s) Required for Multi-speaker Baseline Model Training
1. For each dataset, ensure that the sampling rate matches and speaker embeddings are precomputed.
```sh
(potion-voice_venv) $ python3 prepare_datasets.py --dataset_preset vctk --dataset_archive_path ~/datasets/VCTK_v0.92/VCTK-Corpus-0.92.tgz --sampling_rate 22050
Commencing preparation of dataset for multi-speaker baseline model training:
+ Dataset preset: vctk
+ Dataset : /home/[REDACTED_HOMEDIR_USERNAME_3]/datasets/VCTK_v0.92/VCTK-Corpus-0.92.tgz
+ Output path : results/datasets
+ Sampling rate : 22050
>>> Extracting archive ...
>>> Resampling audio files to 16000Hz ...
Resampling the audio files...
Found 88328 files...
100%|████████████████████████████████████████████████████████████████████████████████| 88328/88328 [18:25<00:00, 79.88it/s]
Done !
>>> Computing speaker embeddings ...
> Found 44283 files in /home/[REDACTED_HOMEDIR_USERNAME_3]/work/potion-repos/potion-voice_venv/potion-voice/voice-cloning/results/datasets/VCTK-Corpus-0.92
> Model fully restored.
> Setting up Audio Processor...
[...]
100%|████████████████████████████████████████████████████████████████████████████████| 44283/44283 [06:18<00:00, 116.99it/s]
Speaker embeddings saved at: results/datasets/VCTK-Corpus-0.92/speakers.pth
>>> Extracting original archive again (overwritting previously resampled files)...
>>> Resampling audio files to 22050Hz ...
Resampling the audio files...
Found 88328 files...
100%|████████████████████████████████████████████████████████████████████████████████| 88328/88328 [20:48<00:00, 70.74it/s]
Done !
Completed preparing voice dataset for multi-speaker baseline model training; generated asset locations are as follows:
--> results/datasets/VCTK-Corpus-0.92
--> results/datasets/VCTK-Corpus-0.92/speakers.pth
Done; bye.
```
### Train New potion-voice Multi-speaker Baseline Model
1. To train a new baseline model:
```sh
(potion-voice_venv) $ python3 train_multispeaker_baseline_model.py
usage: train_multispeaker_baseline_model.py [-h] --datasets {VCTK,LibriTTS_tc360,POTION_Salut} [{VCTK,LibriTTS_tc360,POTION_Salut} ...] [--output_path OUTPUT_PATH] [--batch_size BATCH_SIZE] [--max_epochs MAX_EPOCHS]
Code to train multi-speaker baseline model
options:
-h, --help show this help message and exit
--datasets {VCTK,LibriTTS_tc360,POTION_Salut} [{VCTK,LibriTTS_tc360,POTION_Salut} ...]
List of training datasets to be included in training run.
--output_path OUTPUT_PATH
Path to store trained / generated assets
--batch_size BATCH_SIZE
Batch size for training run
--max_epochs MAX_EPOCHS
Maximum number of epochs for training run
```
Using the default settings, training a new multi-speaker baseline model (on an AWS g5.2xlarge instance) takes 5-7 days (100 epochs with 32 batch size and all 3 datasets (i.e., VCTK, LibriTTS_tc360, andpotion_Salut)).
1. At the end of a training run, there will be the following files in the result folder:
```txt
results/baseline-models/vits_vctk-March-23-2022_03+43AM-0000000/
|-- best_model.pth .................................... best model using avg_loss_0 (NOT the best model; suggest to ignore for now)
|-- best_model_19096.pth .............................. same as best_model.pth (suggest to ignore for now)
|-- checkpoint_300000.pth ............................. fifth last checkpoint
|-- checkpoint_310000.pth ............................. fourth last checkpoint
|-- checkpoint_320000.pth ............................. third last checkpoint
|-- checkpoint_330000.pth ............................. second last checkpoint
|-- checkpoint_340000.pth ............................. last checkpoint
|-- config.json ....................................... configuration file
|-- events.out.tfevents.1648007016.ip-172-31-83-225 ... event log for entire training run including eval samples and charts (view via tensorboard)
|-- speakers.pth ...................................... speaker embeddings
|-- trainer_0_log.txt ................................. training log
|-- train_multispeaker_baseline_model.py .............. copy of the training script
```
Use the event log to determine which of the checkpoints corresponds to the best model.
### Clone a Voice based on the Mutli-speaker Baseline Model
1. To clone a new voice, you need at least 10 voice samples (ideally 30). Those voice recordings (and their corresponding transcription files) have to be arranged as follows (and compressed into a `.tgz`, `.tbz` or `.zip` archive):
```txt
VOICE_DATASET_PATH/txt/1/1_001.txt
VOICE_DATASET_PATH/txt/1/1_002.txt
VOICE_DATASET_PATH/txt/1/1_003.txt
...
VOICE_DATASET_PATH/txt/1/1_029.txt
VOICE_DATASET_PATH/txt/1/1_030.txt
VOICE_DATASET_PATH/wav48/1/1_001.wav
VOICE_DATASET_PATH/wav48/1/1_002.wav
VOICE_DATASET_PATH/wav48/1/1_003.wav
...
VOICE_DATASET_PATH/wav48/1/1_028.wav
VOICE_DATASET_PATH/wav48/1/1_029.wav
VOICE_DATASET_PATH/wav48/1/1_030.wav
```
1. Next, pre-process audio recordings to fit the format of audio samples (i.e., sampling rate) and pre-compute speaker embeddings:
```sh
(potion-voice_venv) $ python prepare_datasets.py --dataset_preset potion_voice_cloning --dataset_archive_path ~/datasets/potion\ Recordings/potion-voice\ recordings/user123.tgz
Commencing preparation of dataset for multi-speaker baseline model training:
+ Dataset preset: potion_voice_cloning
+ Dataset : /home/[REDACTED_HOMEDIR_USERNAME_3]/datasets/potion Recordings/potion-voice recordings/user123.tgz
+ Output path : results/datasets
+ Sampling rate : 22050
>>> Extracting archive ...
>>> Resampling audio files to 16000Hz ...
Resampling the audio files...
Found 30 files...
100%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 30/30 [00:00<00:00, 39.40it/s]
Done !
>>> Extracting original archive again (overwritting previously resampled files)...
>>> Resampling audio files to 22050Hz ...
Resampling the audio files...
Found 30 files...
100%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 30/30 [00:00<00:00, 37.61it/s]
Done !
Completed preparing voice dataset for multi-speaker baseline model training; generated asset locations are as follows:
--> results/datasets/sr22050/user123
--> results/datasets/sr22050/user123/speakers.pth
Done; bye.
```
1. Finally, trigger voice cloning:
```sh
(potion-voice_venv) $ python3 clone_voice.py [-h] --baseline_model_path BASELINE_MODEL_PATH --speaker_dataset_path SPEAKER_DATASET_PATH --speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH [--output_path OUTPUT_PATH] [--batch_size BATCH_SIZE] [--max_epochs MAX_EPOCHS] [--use_cpu] [--output_format {txt,json}]
Code to clone a voice from a given set of voice samples and a multi-speaker baseline model
options:
-h, --help show this help message and exit
--baseline_model_path BASELINE_MODEL_PATH
Path to multi-speaker baseline model (VITS model)
--speaker_dataset_path SPEAKER_DATASET_PATH
Path to voice cloning dataset
--speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH
Path to speaker's embeddings file
--output_path OUTPUT_PATH
Path to store trained / generated assets
--batch_size BATCH_SIZE
Batch size for training run
--max_epochs MAX_EPOCHS
Maximum number of epochs for training run
--use_cpu Signal that CPU should be used even if a CUDA-device is available
--output_format {txt,json}
Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting
```
Using the default settings and 30 audio samples, cloning a new voice (on an AWS g5.2xlarge instance) takes about one hour.
1. At the end of a voice cloning run, there will be the following files in the result folder:
```txt
results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/
|-- best_model_365097.pth .................. best model using avg_loss_0 (save to use)
|-- best_model.pth ......................... same as best_model_365097.pth
|-- checkpoint_365200.pth .................. last checkpoint
|-- clone_voice.py ......................... copy of the clone_voice script used in this run
|-- config.json ............................ configuration file
|-- events.out.tfevents.1672195961.rigel ... event log for entire voice cloning run including eval samples and charts (view via tensorboard)
|-- speakers.pth ........................... speaker's embeddings file
|-- trainer_0_log.txt ...................... training log file
```
Use the event log to confirm that the best model is indeed giving the best outputs.
### Monitoring Training Progress
Using tensorboard / tensorboardX, training progress (for both, multi-speaker baseline training and voice cloning) can be monitored and evaluation samples can be accessed.
1. Ensure AWS Security Group settings (inbound) are set appropriamust include:
```txt
HTTPS TCP 443 0.0.0.0/0
Custom_TCP TCP 6006 0.0.0.0/0
```
+ Server-side, launch the tensorboard service:
```sh
(potion-voice_venv) $ tensorboard --logdir=./results/baseline-models/vits_vctk-March-07-2022_09+47AM-0000000/ --host 0.0.0.0
TensorFlow installation not found - running with reduced feature set.
NOTE: Using experimental fast data loading logic. To disable, pass
"--load_fast=false" and report issues on GitHub. More details:
https://github.com/tensorflow/tensorboard/issues/4784
TensorBoard 2.8.0 at http://0.0.0.0:6006/ (Press CTRL+C to quit)
```
+ Locally, point your preferred Web browser to <http://PUBLIC_IPv4_DNS:6006/>
### Minimise a Cloned Voice
To minimise the size of a trained model, run the followng script which removes optimiser and discriminator components from the model -- those are only required for training but not for inference:
```sh
(potion-voice_venv) $ python3 minimize_cloned_voice_model.py [-h] --voice_model_asset_path VOICE_MODEL_ASSET_PATH [--voice_model_name VOICE_MODEL_NAME] [--voice_model_config_name VOICE_MODEL_CONFIG_NAME] [--minimise_suffix MINIMISE_SUFFIX] [--overwrite_assets] [--output_format {txt,json}]
Code to minimise (i.e., remove optimiser & discriminator) a cloned voice model
options:
-h, --help show this help message and exit
--voice_model_asset_path VOICE_MODEL_ASSET_PATH
Path to directory storing cloned voice model and the corresponding configuration and speaker files
--voice_model_name VOICE_MODEL_NAME
Name of the (best) cloned voice model
--voice_model_config_name VOICE_MODEL_CONFIG_NAME
Name of the config file for the cloned voice model
--minimise_suffix MINIMISE_SUFFIX
Suffix to be used for minimised model and its assets (i.e., new config file)
--overwrite_assets Signal whether existing model assets should be overwritten or not (default: do not overwrite)
--output_format {txt,json}
Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting
```
1. Command-line output sample for output format option "txt":
```sh
$ python3 minimize_cloned_voice_model.py --voice_model_asset_path results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/
Minimising given voice model:
+ Cloned voice model file path : results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/best_model.pth
+ Cloned voice model config file : results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/config.json
> Using model: vits
> Setting up Audio Processor...
[...]
Completed minimising cloned voice model. The resulting (modified) assets can be found at:
--> Minimised voice model path : results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/best_model_light.pth
--> Minimised voice model config path: results/cloned-voices/vits_potion_clone-December-28-2022_10+52AM-1327031/config_light.json
Done; bye.
```
### Scoring a Cloned Voice
1. To score a cloned voice, run the following command:
```sh
(potion-voice_venv) $ python3 score_cloned_voice.py [-h] --voice_dataset_path VOICE_DATASET_PATH --voice_model_path VOICE_MODEL_PATH --voice_model_config_path VOICE_MODEL_CONFIG_PATH --speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH [--temp_path TEMP_PATH] [--keep_temp] [--use_cpu] [--output_format {txt,json}]
Compute quality score for a given voice model (cloned voice) wrt. a given set of voice recordings (original voice))
options:
-h, --help show this help message and exit
--voice_dataset_path VOICE_DATASET_PATH
Path to set of voice recordings (original voice)
--voice_model_path VOICE_MODEL_PATH
Path to cloned voice model
--voice_model_config_path VOICE_MODEL_CONFIG_PATH
Path to config file for the cloned voice model
--speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH
Path to speaker's embeddings file (i.e., pre-computed embeddings typically stored with the speaker's dataset)
--temp_path TEMP_PATH
Path to store temporary speech assets
--keep_temp Signal that temporary assets used for scoring should not be deleted once done
--use_cpu Signal that CPU should be used even if a CUDA-device is available
--output_format {txt,json}
Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting
```
1. Command-line output sample for output format option "txt":
```sh
$ python3 score_cloned_voice.py --voice_dataset_path results/datasets/sr22050/michael/wav48/1/ --voice_model_path results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/best_model.pth --voice_model_config_path results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/config.json --speaker_embeddings_path results/datasets/sr22050/michael/speakers.pth
Computing similarity score for a given voice model (cloned voice) wrt. a given set of voice recordings (original voice):
+ Original voice recordings path: results/datasets/sr22050/michael/wav48/1/
+ Cloned voice model file path : results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/best_model.pth
+ Cloned voice model config file: results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/config.json
+ Speaker embeddings file : results/datasets/sr22050/michael/speakers.pth
+ CUDA availability : True
+ Compute device used : cuda
+ No. of speakers : 1
+ Speaker's names : ['VCTK_old_1']
+ No. of embeddings : 30
> Using model: vits
> Setting up Audio Processor...
Loaded the voice encoder model on cuda in 0.01 seconds.
Completed computing similarity score for the two sets of recordings. The resulting similarity score is:
--> 0.9127510190010071
Done; bye.
```
1. Command-line output sample for output format option "json":
```sh
(potion-voice_venv)$ python3 score_cloned_voice.py --voice_dataset_path results/datasets/sr22050/michael/wav48/1/ --voice_model_path results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/best_model.pth --voice_model_config_path results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/config.json --speaker_embeddings_path results/datasets/sr22050/michael/speakers.pth --output_format json
> Using model: vits
> Setting up Audio Processor...
[...]
Loaded the voice encoder model on cuda in 0.01 seconds.
{"success": true, "in": {"voice_dataset_path": "results/datasets/sr22050/michael/wav48/1/", "voice_model_path": "results/cloned-voices/vits_potion_clone-December-28-2022_01+09AM-1327031/best_model.pth"}, "out": {"score": 0.91}}
```
## Usage Examples for Speech Synthesizing
1. To generate speech for a given cloned voice, run the following command:
```sh
(potion-voice_venv) $ python3 synthesize_speech.py [-h] --voice_model_path VOICE_MODEL_PATH --voice_model_config_path VOICE_MODEL_CONFIG_PATH --speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH --txt TXT [--output_path OUTPUT_PATH] [--target_sampling_rate TARGET_SAMPLING_RATE] [--speech_sample_wav_path SPEECH_SAMPLE_WAV_PATH] [--speech_sample_txt SPEECH_SAMPLE_TXT] [--trim_silence] [--use_cpu] [--output_format {txt,json}]
Code to synthesize speech for a given voice model
options:
-h, --help show this help message and exit
--voice_model_path VOICE_MODEL_PATH
Path to cloned voice model
--voice_model_config_path VOICE_MODEL_CONFIG_PATH
Path to config file for the cloned voice model
--speaker_embeddings_path SPEAKER_EMBEDDINGS_PATH
Path to speaker's embeddings file (i.e., pre-computed embeddings typically stored with the speaker's dataset)
--txt TXT Text to synthesize
--output_path OUTPUT_PATH
Path to store generated speech assets
--target_sampling_rate TARGET_SAMPLING_RATE
Desired sampling rate (in Hz) for output file
--speech_sample_wav_path SPEECH_SAMPLE_WAV_PATH
Path to a sample utterance of the speaker (used for style transfer)
--speech_sample_txt SPEECH_SAMPLE_TXT
Text of the sample utterance of the speaker (used for style transfer)
--trim_silence Signal whether to trim silence from synthesised speech
--use_cpu Signal that CPU should be used even if a CUDA-device is available
--output_format {txt,json}
Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting
```
1. Command-line output sample for output format option "txt":
```sh
(potion-voice_venv) $ python3 synthesize_speech.py --voice_model_path results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/best_model.pth --voice_model_config_path results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/config.json --speaker_embeddings_path results/datasets/sr22050/[REDACTED_HOMEDIR_USERNAME_2]/speakers.pth --txt "Hi person_82, it works!"
Commencing speech synthesizing:
+ Voice model file path : results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/best_model.pth
+ Voice model config file: results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/config.json
+ Speaker embeddings file: results/datasets/sr22050/[REDACTED_HOMEDIR_USERNAME_2]/speakers.pth
+ Output path : results/speech
+ Text to synthesize : Hi person_82, it works!
+ CUDA availability : True
+ Compute device used : cuda
+ No. of speakers : 1
+ Speaker's names : ['VCTK_old_1']
+ No. of embeddings : 30
> Using model: vits
> Setting up Audio Processor...
[...]
>>> Saving original output to : results/speech/b4189e9e-6142-4dad-8577-6de77087ffd1.wav
>>> Saving resampled output to: results/speech/b4189e9e-6142-4dad-8577-6de77087ffd1_sr48000.wav
Speech synthesizing has completed. Bye.
```
1. Command-line output sample for output format option "txt":
```sh
(potion-voice_venv) $ python3 synthesize_speech.py --voice_model_path results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/best_model_light.pth --voice_model_config_path results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/config_light.json --speaker_embeddings_path results/datasets/sr22050/[REDACTED_HOMEDIR_USERNAME_2]/speakers.pth --txt "Hi person_82, it works!" --output_format json
> Using model: vits
> Setting up Audio Processor...
[...]
{"success": true, "in": {"voice_model_path": "results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/best_model_light.pth", "voice_model_config_path": "results/cloned-voices/vits_potion_clone-December-28-2022_08+30AM-1327031/config_light.json", "speaker_embeddings_path": "results/datasets/sr22050/[REDACTED_HOMEDIR_USERNAME_2]/speakers.pth"}, "out": {"speech_original_path": "results/speech/c99e494c-e1f9-4c12-9095-255cf7db792b.wav", "speech_resampled_path": "results/speech/c99e494c-e1f9-4c12-9095-255cf7db792b_sr48000.wav"}}
```
### Scoring a Synthesised Salutation
1. To score a synthesised salutation, run the following command:
```sh
(potion-voice_venv)$ python3 score_salutation.py [-h] --recording_path RECORDING_PATH --first_name FIRST_NAME [--output_format {txt,json}]
Score a given salutation recording wrt. its desired content, the actual salutation recording, and a generated transcription (using Potion's internal Transciption API) of the recording.
optional arguments:
-h, --help show this help message and exit
--recording_path RECORDING_PATH
Path to salutation recoding (.wav audio file)
--first_name FIRST_NAME
First name that the salutation recoding is meant to use
--output_format {txt,json}
Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting
```
1. Command-line output sample for output format option "txt":
```sh
(potion-voice_venv)$ python3 score_salutation.py --recording_path /home/[REDACTED_HOMEDIR_USERNAME_3]/person_82_-_Hey_person_83.wav --first_name person_83
Commencing scoring of the given salutation recording:
+ Salutation recording path: /home/[REDACTED_HOMEDIR_USERNAME_3]/person_82_-_Hey_person_83.wav
+ Salutation first name : person_83
>> Salutation score : 0.892155
Done; bye.
```
1. Command-line output sample for output format option "json":
```sh
(potion-voice_venv)$ python3 score_salutation.py --recording_path /home/[REDACTED_HOMEDIR_USERNAME_3]/person_82_-_Hey_person_83.wav --first_name person_83 --output_format json
{"in": {"recording_path": "/home/[REDACTED_HOMEDIR_USERNAME_3]/person_82_-_Hey_person_83.wav", "first_name": "person_83"}, "out": {"score": 0.89}}
```
## Troubleshooting
1. How to better monitor GPU load / utilisation?
+ Install an interactive NVIDIA-GPU process viewer such as `nvitop`:
```sh
$ python3 -m pip install nvitop
Collecting nvitop
[...]
Installing collected packages: nvidia-ml-py, termcolor, psutil, nvitop
Successfully installed nvidia-ml-py-11.495.46 nvitop-0.8.0 psutil-5.9.2 termcolor-2.0.1
````
+ Run via command-line: `nvitop`:
```sh
Tue Sep 13 02:09:07 2022
╒═════════════════════════════════════════════════════════════════════════════╕
│ NVITOP 0.8.0 Driver Version: 515.65.01 CUDA Driver Version: 11.7 │
├───────────────────────────────┬──────────────────────┬──────────────────────┤
│ GPU Name Persistence-M│ Bus-Id Disp.A │ Volatile Uncorr. ECC │
│ Fan Temp Perf Pwr:Usage/Cap│ Memory-Usage │ GPU-Util Compute M. │
╞═══════════════════════════════╪══════════════════════╪══════════════════════╪══════════════════════════╕
│ 0 A10G On │ 00000000:00:1E.0 Off │ 0 │ MEM: █████████▊ 65.1% │
│ 0% 47C P0 192W / 300W │ 14982MiB / 22.49GiB │ 100% Default │ UTL: ███████████████ MAX │
╘═══════════════════════════════╧══════════════════════╧══════════════════════╧══════════════════════════╛
[ CPU: ██████████▏ 18.1% ] ( Load Average: 1.07 1.11 1.04 )
[ MEM: ███████████▎ 20.2% ] [ SWP: ▏ 0.3% ]
╒════════════════════════════════════════════════════════════════════════════════════════════════════════╕
│ Processes: ubuntu@ip-172-31-95-84 │
│ GPU PID USER GPU-MEM %SM %CPU %MEM TIME COMMAND │
╞════════════════════════════════════════════════════════════════════════════════════════════════════════╡
│ 0 2100 C ubuntu 14463MiB 90 103.7 9.6 5.4 days python3 train_multispeaker_baseline_model.py │
╘════════════════════════════════════════════════════════════════════════════════════════════════════════╛
```

View File

@@ -0,0 +1,120 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import sys
import os
import argparse
from pathlib import Path
import json
import torch
from TTS.config import load_config
from TTS.tts.models import setup_model as setup_tts_model
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Code to minimise (i.e., remove optimiser & discriminator) a cloned voice model")
parser.add_argument("--voice_model_asset_path", type = str, required = True,
help = "Path to directory storing cloned voice model and the corresponding configuration and speaker files")
parser.add_argument("--voice_model_name", type = str, default = "best_model.pth",
help = "Name of the (best) cloned voice model")
parser.add_argument("--voice_model_config_name", type = str, default = "config.json",
help = "Name of the config file for the cloned voice model")
parser.add_argument("--minimise_suffix", type = str, default = "light",
help = "Suffix to be used for minimised model and its assets (i.e., new config file)")
parser.add_argument("--overwrite_assets", default = False, action = "store_true",
help = "Signal whether existing model assets should be overwritten or not (default: do not overwrite)")
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
return parser.parse_args()
#
# utility function to expand the name of a given filename (infront of the extension)
#
def append_suffix_to_filename(fname, fname_suffix):
fpath = Path(fname)
return "{0}_{2}{1}" . format(fpath.stem, fpath.suffix, fname_suffix)
#
# save a lightweight (i.e., without optimiser and discriminator) model of the given cloned voice and corresponding config assets
#
def main(args):
# set variables
output_path = args.voice_model_asset_path
model_path = os.path.join(args.voice_model_asset_path, args.voice_model_name)
model_config_path = os.path.join(args.voice_model_asset_path, args.voice_model_config_name)
model_light_path = os.path.join(output_path, append_suffix_to_filename(args.voice_model_name, args.minimise_suffix))
model_light_config_path = os.path.join(output_path, append_suffix_to_filename(args.voice_model_config_name, args.minimise_suffix))
if not args.overwrite_assets:
# ensure target output files do not already exist
if (Path(model_light_path).exists()) or (Path(model_light_config_path).exists()):
sys.exit("Naming conflict: Model asset files ({} and/or {}) exist already!" . format (model_light_path, model_light_config_path))
if args.output_format == "txt":
print("Minimising given voice model:")
print("")
print(" + Cloned voice model file path : {}" . format(model_path))
print(" + Cloned voice model config file : {}" . format(model_config_path))
print("")
elif args.output_format == "json":
json_data = {
"success": False,
"in": {
"voice_model_path": format(model_path),
"voice_model_config_path": format(model_config_path)
},
"out": {
"voice_model_light_path": "",
"voice_model_light_config_path": ""
}
}
# load model
config = load_config(model_config_path)
# init model
model = setup_tts_model(config = config)
# load checkpoint / model
model.load_checkpoint(config, model_path, eval = True)
model.disc = None
model_state = model.state_dict()
state = {
"model": model_state
}
torch.save(state, model_light_path)
config.model_args["init_discriminator"] = False
config.save_json(model_light_config_path)
# exit gracefully
if args.output_format == "txt":
print("Completed minimising cloned voice model. The resulting (modified) assets can be found at:")
print(" --> Minimised voice model path : {}" . format(model_light_path))
print(" --> Minimised voice model config path: {}" . format(model_light_config_path))
print("")
print("Done; bye.")
print("")
elif args.output_format == "json":
json_data["out"]["voice_model_light_path"] = format(model_light_path)
json_data["out"]["voice_model_light_config_path"]: format(model_light_config_path)
json_data["success"] = True
print(json.dumps(json_data))
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
main(args)

View File

@@ -0,0 +1,191 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import sys
import os
import argparse
# load coqui-ai/TTS libraries
from TTS.bin.resample import resample_files
from TTS.bin.compute_embeddings import compute_embeddings
import train_config as tc
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Code to prepare voice dataset for multi-speaker baseline model training (i.e., adjust sampling rate and compute speaker embeddings).")
parser.add_argument("--dataset_preset", type = str, choices = ("VCTK", "LibriTTS_tc360", "DAPS", "POTION_Salut", "potion_voice_cloning"), required = True,
help = "Path the voice dataset archive (.zip, .tar.gz, .tgz, .tar.bz2, and .tbz are supported)")
parser.add_argument("--dataset_archive_path", type = str, required = True,
help = "Path the voice dataset archive (.zip, .tar.gz, .tgz, .tar.bz2, and .tbz are supported)")
parser.add_argument("--output_path", type = str, default = "results/datasets",
help = "Path to store augmented dataset")
parser.add_argument("--sampling_rate", type = int, default = 22050, choices = (16000, 22050, 32000, 48000), # 32k & 48k are untested
help = "Sampling rate for training run")
return parser.parse_args()
#
# utility functuion to extract archives (zip, tar, tgz, ...)
# - returns first entry in archive (typically the main directory name contained in the archive)
#
def extract_archive(archive_path, dest_path):
from zipfile import ZipFile
import tarfile
if archive_path.endswith('.zip'):
opener, getnames, mode = ZipFile, ZipFile.namelist, 'r'
elif (archive_path.endswith('.tar.gz')) or (archive_path.endswith('.tgz')):
opener, getnames, mode = tarfile.open, tarfile.TarFile.getnames, 'r:gz'
elif (archive_path.endswith('.tar.bz2')) or (archive_path.endswith('.tbz')):
opener, getnames, mode = tarfile.open, tarfile.TarFile.getnames, 'r:bz2'
else:
print("Extracting archive " + archive_path + " is not supported.")
return
# extract archive
with opener(archive_path, mode) as archive:
archive_dir = archive.getnames()[0]
archive.extractall(path = dest_path)
return archive_dir
#
# main training method (VITS multi-speaker model)
#
def main(args):
print("Commencing preparation of dataset for multi-speaker baseline model training:")
print("")
print(" + Dataset preset: {}" . format(args.dataset_preset))
print(" + Dataset : {}" . format(args.dataset_archive_path))
print(" + Output path : {}" . format(args.output_path))
print(" + Sampling rate : {}" . format(args.sampling_rate))
print("")
# set parameters according to dataset preset
if args.dataset_preset == "VCTK":
DATASET_NAME = tc.VCTK_DATASET_NAME
DATASET_FORMATTER = tc.VCTK_DATASET_FORMATTER
DATASET_FILE_FORMAT = tc.VCTK_DATASET_FILE_FORMAT
NO_EVAL = False
elif args.dataset_preset == "LibriTTS_tc360":
DATASET_NAME = tc.LIBRITTS_TC360_DATASET_NAME
DATASET_FORMATTER = tc.LIBRITTS_TC360_DATASET_FORMATTER
DATASET_FILE_FORMAT = tc.LIBRITTS_TC360_DATASET_FILE_FORMAT
NO_EVAL = False
elif args.dataset_preset == "POTION_Salut":
DATASET_NAME = tc.POTION_SALUT_DATASET_NAME
DATASET_FORMATTER = tc.POTION_SALUT_DATASET_FORMATTER
DATASET_FILE_FORMAT = tc.POTION_SALUT_DATASET_FILE_FORMAT
NO_EVAL = False
elif args.dataset_preset == "potion_voice_cloning":
DATASET_NAME = tc.POTION_SALUT_DATASET_NAME
DATASET_FORMATTER = tc.POTION_SALUT_DATASET_FORMATTER
DATASET_FILE_FORMAT = tc.POTION_SALUT_DATASET_FILE_FORMAT
NO_EVAL = True
# define sampling rate for computing speaker embeddings
SPK_EMB_SAMPLING_RATE = 16000
# define the number of threads used during audio resampling
NUM_RESAMPLE_THREADS = 10
# extract dataset archive
print(f">>> Extracting archive ...")
dataset_root = extract_archive(args.dataset_archive_path, os.path.join(args.output_path, "sr" + str(args.sampling_rate)))
# set dataset path (there should only be ONE directory in the extracted archive location)
dataset_path = os.path.join(args.output_path, "sr" + str(args.sampling_rate), dataset_root)
# ensure the dataset_path exists
os.makedirs(dataset_path, exist_ok = True)
# resample dataset for speaker embeddings computation
print(f">>> Resampling audio files to 16000Hz ...")
resample_files(dataset_path, 16000, file_ext = DATASET_FILE_FORMAT, n_jobs = NUM_RESAMPLE_THREADS)
# compute speaker embeddings
SPEAKER_ENCODER_CHECKPOINT_PATH = "assets/speaker_encoder_model/model_se.pth.tar"
SPEAKER_ENCODER_CONFIG_PATH = "assets/speaker_encoder_model/config_se.json"
# init list speaker embeddings/d-vectors to be used during the training
d_vector_files = []
# check if the speakers embeddings are already computated, if not compute them
embeddings_file = os.path.join(dataset_path, "speakers.pth")
if not os.path.isfile(embeddings_file):
print(f">>> Computing speaker embeddings ...")
compute_embeddings(
SPEAKER_ENCODER_CHECKPOINT_PATH,
SPEAKER_ENCODER_CONFIG_PATH,
embeddings_file,
old_spakers_file = None,
config_dataset_path = None,
formatter_name = DATASET_FORMATTER,
dataset_name = DATASET_NAME,
dataset_path = dataset_path,
meta_file_train = "",
meta_file_val = "",
disable_cuda = False,
no_eval = NO_EVAL
)
d_vector_files.append(embeddings_file)
# if targetted sampling rate is not the same as that used for computing speaker embeddings, replace and resample audio files
if not args.sampling_rate == SPK_EMB_SAMPLING_RATE:
print(f">>> Extracting original archive again (overwritting previously resampled files)...")
extract_archive(args.dataset_archive_path, os.path.join(args.output_path, "sr" + str(args.sampling_rate)))
print(f">>> Resampling audio files to {args.sampling_rate}Hz ...")
resample_files(dataset_path, args.sampling_rate, file_ext = DATASET_FILE_FORMAT, n_jobs = NUM_RESAMPLE_THREADS)
# exit gracefully
print("")
print("Completed preparing voice dataset for multi-speaker baseline model training; generated asset locations are as follows:")
print(" --> {}" . format(dataset_path))
print(" --> {}" . format(embeddings_file))
print("")
print("Done; bye.")
print("")
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
# clear command line arguments to avoid triggering argparse features part of Trainer / coqpit imports
# Traceback (most recent call last):
# File "train_multispeaker_baseline_model.py", line 208, in <module>
# main(args)
# File "train_multispeaker_baseline_model.py", line 177, in main
# trainer = Trainer(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 360, in __init__
# config, new_fields = self.init_training(args, coqpit_overrides, config)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 594, in init_training
# config.parse_known_args(coqpit_overrides, relaxed_parser=True)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 843, in parse_known_args
# parser = self.init_argparse(arg_prefix=arg_prefix, relaxed_parser=relaxed_parser)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 881, in init_argparse
# _init_argparse(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 529, in _init_argparse
# parser = _init_argparse(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 550, in _init_argparse
# return default.init_argparse(
# AttributeError: 'str' object has no attribute 'init_argparse'
sys.argv = [sys.argv[0]]
# ensure the output path exists
os.makedirs(args.output_path, exist_ok = True)
main(args)

View File

@@ -0,0 +1,179 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import os
import argparse
import shutil
import uuid
import json
import torch
from TTS.TTS.tts.utils.speakers import SpeakerManager
from utils.synthesize_utils import init_synth, synthesize, save_waveform
from utils.scoring_utils import init_scoring_vocoder, score_speaker_similarity
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Compute quality score for a given voice model (cloned voice) wrt. a given set of voice recordings (original voice))")
parser.add_argument("--voice_dataset_path", type = str, required = True,
help = "Path to set of voice recordings (original voice)")
parser.add_argument("--voice_model_path", type = str, required = True,
help = "Path to cloned voice model")
parser.add_argument("--voice_model_config_path", type = str, required = True,
help = "Path to config file for the cloned voice model")
parser.add_argument('--speaker_embeddings_path', type = str, required = True,
help = "Path to speaker's embeddings file (i.e., pre-computed embeddings typically stored with the speaker's dataset)")
parser.add_argument("--temp_path", type = str, default = "temp",
help = "Path to store temporary speech assets")
parser.add_argument("--keep_temp", default = False, action = "store_true",
help = "Signal that temporary assets used for scoring should not be deleted once done")
parser.add_argument("--use_cpu", default = False, action = "store_true", # untested!!!
help = "Signal that CPU should be used even if a CUDA-device is available")
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
return parser.parse_args()
#
# main training method (voice cloning)
#
def main(args):
# define assets required for using a pretrained voice
MODEL_PATH = args.voice_model_path
CONFIG_PATH = args.voice_model_config_path
SPK_EMBEDDINGS_PATH = args.speaker_embeddings_path
# set default score
sim_score = -1.0
if args.output_format == "txt":
print("Computing similarity score for a given voice model (cloned voice) wrt. a given set of voice recordings (original voice):")
print("")
print(" + Original voice recordings path: {}" . format(args.voice_dataset_path))
print(" + Cloned voice model file path : {}" . format(MODEL_PATH))
print(" + Cloned voice model config file: {}" . format(CONFIG_PATH))
print(" + Speaker embeddings file : {}" . format(SPK_EMBEDDINGS_PATH))
print("")
elif args.output_format == "json":
json_data = {
"success": False,
"in": {
"voice_dataset_path": format(args.voice_dataset_path),
"voice_model_path": format(MODEL_PATH)
},
"out": {
"score": sim_score
}
}
# determine whether CUDA support is available and set device parameters accordingly
use_cuda = torch.cuda.is_available()
if args.output_format == "txt":
print(" + CUDA availability : {}" . format(use_cuda))
if args.use_cpu:
device = "cpu"
USE_CUDA = False
elif use_cuda:
device = "cuda"
USE_CUDA = True
else:
device = "cpu"
USE_CUDA = False
if args.output_format == "txt":
print(" + Compute device used : {}" . format(device))
print("")
# score the cloned voice (wrt. similarity to recorded voice)
# 1. generate 20 samples (5 x samples from Potion's Web-site; 5 x salutations; 10 x test sentences from other research papers)
# 2. compute similarity score (training samples versus generated samples)
scoring_sentences = [
"Book more meetings, build more trust, and close more sales using Potion.",
"Free forever. As long as you hustle. No credit card required.",
"Don't send plain old boring text emails. Send Potion.",
"What distinguished you from everyone else?",
"We absolutely ensure that you see increased engagement in your outreach efforts.",
"Hi there, Samuel. Hope things are going well for you.",
"Hey person_93. I wanted to reach out to see if you are interested to learn mode about our services.",
"Hi person_90. I noticed you and I are both members of the Green Movement on LinkedIn, and that you just opened a new office in Austin.",
"Hey person_96. Could your team handle an extra 20 leads a week?",
"Hi person_95. For every 100 cold emails you send, you'll only get one reply. That's a lot of effort for little reward.",
"Prosecutors have opened a massive investigation into allegations of fixing games and illegal betting.",
"Feedback must be timely and accurate throughout the project.",
"Humans also judge distance by using the relative sizes of objects.",
"If this is true then those who tend to think creatively really are somehow different.",
"But really in the grand scheme of things this information is insignificant.",
"About half the people who are infected also lose weight.",
"The second half of the book focuses on argument and essay writing.",
"He loves to watch me drink this stuff.",
"Funding is always an issue after the fact.",
"Let us encourage each other."
]
# init speaker manager
speaker_manager = None
speaker_manager = SpeakerManager(d_vectors_file_path = SPK_EMBEDDINGS_PATH)
if args.output_format == "txt":
print(" + No. of speakers : {}" . format(speaker_manager.num_speakers))
print(" + Speaker's names : {}" . format(speaker_manager.embedding_names))
print(" + No. of embeddings : {}" . format(speaker_manager.num_embeddings))
print("")
# assert that only one speaker is present in the embedding's file
assert speaker_manager.num_speakers == 1, f"Number of speakers in the given embedding's file MUST be one; found {speaker_manager.num_speakers} speakers!"
# initialise speech synthesization
voice_config, voice_model = init_synth(CONFIG_PATH, MODEL_PATH, speaker_embeddings_file = SPK_EMBEDDINGS_PATH, use_cuda = USE_CUDA)
# synthesize speech for all scoring sentences
output_path = os.path.join(args.temp_path, str(uuid.uuid4()))
# create temp path (exit if it already exists)
os.makedirs(output_path, exist_ok = False)
for cnt, txt in enumerate(scoring_sentences):
# speaker embeddings provided, use it together with the given model (cloned or baseline model) to synthesize speech
waveform = synthesize(voice_config, voice_model, txt, speaker_manager.get_mean_embedding(speaker_manager.embedding_names[0], speaker_manager.num_embeddings), USE_CUDA)
# save the results
output_fname = os.path.join(output_path, "{:02d}" . format(cnt) + ".wav")
save_waveform(voice_model, waveform, output_fname)
# initialize voice envcoder used for scoring
scoring_vocoder = init_scoring_vocoder()
# determine similarity score
sim_score = score_speaker_similarity(scoring_vocoder, args.voice_dataset_path, output_path)
# clean up
if not args.keep_temp:
shutil.rmtree(output_path)
# exit gracefully
if args.output_format == "txt":
print("")
print("Completed computing similarity score for the two sets of recordings. The resulting similarity score is:")
print(" --> {}" . format(sim_score))
print("")
print("Done; bye.")
print("")
elif args.output_format == "json":
json_data["out"]["score"] = round(float(sim_score), 2)
json_data["success"] = True
print(json.dumps(json_data))
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
# ensure the temp path exists
os.makedirs(args.temp_path, exist_ok = True)
main(args)

View File

@@ -0,0 +1,219 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import os
import argparse
import glob
import shutil
import uuid
import json
import torch
from TTS.tts.utils.speakers import SpeakerManager
from utils.synthesize_utils import init_synth, synthesize, save_waveform
from utils.scoring_utils import init_scoring_vocoder, score_speaker_similarity
#
# What do we need?
# -> list of models to test
# -> test db (user recordings, speaker embeddings, reference to their voice in the multi-speaker model)
# |- user
# |- speaker.pth
# |- userid.txt
# |- txt
# |- wav48
#
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Given a list of models, compute quality scores to determine the top-5 (human-perceived) models.")
parser.add_argument("--models_path", type = str, required = True,
help = "Path to a collection of models and their config file to be used for testing.")
parser.add_argument("--speaker_embeddings_path_list", type = str, nargs = "+", required = True,
help = "List of paths to the speaker embeddings files of the data sets used to train the models.")
parser.add_argument("--test_dataset_path", type = str, required = True,
help = "Path to a set of user recordings with speaker embedding and voice id (the users' ids in the models to be tested)")
parser.add_argument("--temp_path", type = str, default = "temp",
help = "Path to store temporary speech assets")
parser.add_argument("--keep_temp", default = False, action = "store_true",
help = "Signal that temporary assets used for scoring should not be deleted once done")
parser.add_argument("--use_cpu", default = False, action = "store_true", # untested!!!
help = "Signal that CPU should be used even if a CUDA-device is available")
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
return parser.parse_args()
#
# main training method (voice cloning)
#
def main(args):
# define assets required for using a pretrained voice
MODELS_PATH = args.models_path
MODEL_CONFIG_PATH = os.path.join(MODELS_PATH, "config.json")
MODEL_SPK_EMB_PATH_LIST = args.speaker_embeddings_path_list
DATASET_PATH = args.test_dataset_path
USER_ID_FNAME = "userid.txt"
# set default score
sim_score_avg = -1.0
if args.output_format == "txt":
print("Computing similarity score for a given voice model (cloned voice) wrt. a given set of voice recordings (original voice):")
print("")
print(" + Multi-speaker models path : {}" . format(MODELS_PATH))
print(" + Multi-speaker model config file: {}" . format(MODEL_CONFIG_PATH))
print(" + Multi-speaker embeddings file : {}" . format(MODEL_SPK_EMB_PATH_LIST))
print(" + Test dataset path : {}" . format(DATASET_PATH))
#print(" + Speaker embeddings filename : {}" . format(SPK_EMBEDDINGS_FNAME))
print(" + User ID filename : {}" . format(USER_ID_FNAME))
print("")
elif args.output_format == "json":
json_data = {
"success": False,
"in": {
"models_path": format(MODELS_PATH),
"dataset_path": format(DATASET_PATH)
},
"out": {
"best_model": None,
"top_5_models": None
}
}
# determine whether CUDA support is available and set device parameters accordingly
use_cuda = torch.cuda.is_available()
if args.output_format == "txt":
print(" + CUDA availability : {}" . format(use_cuda))
if args.use_cpu:
device = "cpu"
USE_CUDA = False
elif use_cuda:
device = "cuda"
USE_CUDA = True
else:
device = "cpu"
USE_CUDA = False
if args.output_format == "txt":
print(" + Compute device used : {}" . format(device))
print("")
# score the cloned voice (wrt. similarity to recorded voice)
# 1. generate 20 samples (5 x samples from Potion's Web-site; 5 x salutations; 10 x test sentences from other research papers)
# 2. compute similarity score (training samples versus generated samples)
scoring_sentences = [
"Book more meetings, build more trust, and close more sales using Potion.",
"Free forever. As long as you hustle. No credit card required.",
"Don't send plain old boring text emails. Send Potion.",
"What distinguished you from everyone else?",
"We absolutely ensure that you see increased engagement in your outreach efforts.",
"Hi there, Samuel. Hope things are going well for you.",
"Hey person_93. I wanted to reach out to see if you are interested to learn mode about our services.",
"Hi person_90. I noticed you and I are both members of the Green Movement on LinkedIn, and that you just opened a new office in Austin.",
"Hey person_96. Could your team handle an extra 20 leads a week?",
"Hi person_95. For every 100 cold emails you send, you'll only get one reply. That's a lot of effort for little reward.",
"Prosecutors have opened a massive investigation into allegations of fixing games and illegal betting.",
"Feedback must be timely and accurate throughout the project.",
"Humans also judge distance by using the relative sizes of objects.",
"If this is true then those who tend to think creatively really are somehow different.",
"But really in the grand scheme of things this information is insignificant.",
"About half the people who are infected also lose weight.",
"The second half of the book focuses on argument and essay writing.",
"He loves to watch me drink this stuff.",
"Funding is always an issue after the fact.",
"Let us encourage each other."
]
# init scoring tracker
sim_score = {}
for model_fname in glob.glob(os.path.join(MODELS_PATH, "check*.pth")):
# init scoring tracker
sim_score[os.path.basename(model_fname)] = []
# score each moddel for every user
for user_dir in os.listdir(DATASET_PATH):
# get user's speaker id / name
with open(os.path.join(DATASET_PATH, user_dir, USER_ID_FNAME), 'r') as f:
user_data = json.load(f)
print("Speaker name: {}" . format(user_data["speaker_name"]))
# init speaker manager
speaker_manager = None
speaker_manager = SpeakerManager(d_vectors_file_path = MODEL_SPK_EMB_PATH_LIST)
#speaker_manager = SpeakerManager(speaker_id_file_path = os.path.join(MODELS_PATH, "speakers.pth"))
print("Number of speakers:", speaker_manager.num_speakers)
print("Speaker names :", speaker_manager.speaker_names)
#print("Embedding names :", speaker_manager.embedding_names)
# assert that the user is indeed present in the embedding's file
#assert speaker_manager.num_speakers == 1, f"Number of speakers in the given embedding's file MUST be one; found {speaker_manager.num_speakers} speakers!"
for model_fname in glob.glob(os.path.join(MODELS_PATH, "check*.pth")):
# initialise speech synthesization
voice_config, voice_model = init_synth(MODEL_CONFIG_PATH, model_fname, speaker_embeddings_file = MODEL_SPK_EMB_PATH_LIST, use_cuda = USE_CUDA)
# synthesize speech for all scoring sentences
output_path = os.path.join(args.temp_path, str(uuid.uuid4()))
# create temp path (exit if it already exists)
os.makedirs(output_path, exist_ok = False)
for cnt, txt in enumerate(scoring_sentences):
# speaker embeddings provided, use it together with the given model (cloned or baseline model) to synthesize speech
#waveform = synthesize(voice_config, voice_model, txt, speaker_manager.get_mean_embedding(user_data["speaker_name"]), USE_CUDA)
waveform = synthesize(voice_config, voice_model, txt, speaker_embeddings = speaker_manager.get_mean_embedding(user_data["speaker_name"], num_samples = None, randomize = False), use_cuda = USE_CUDA)
#waveform = synthesize(voice_config, voice_model, txt, speaker_embeddings = speaker_manager.get_mean_embedding(user_data["speaker_name"]), speaker_id = speaker_manager.name_to_id[user_data["speaker_name"]], use_cuda = USE_CUDA)
#waveform = synthesize(voice_config, voice_model, txt, speaker_id = speaker_manager.name_to_id[user_data["speaker_name"]], use_cuda = USE_CUDA)
# save the results
output_fname = os.path.join(output_path, "{:02d}" . format(cnt) + ".wav")
save_waveform(voice_config, voice_model, waveform, output_fname)
# initialize voice envcoder used for scoring
scoring_vocoder = init_scoring_vocoder()
# determine similarity score
sim_score[os.path.basename(model_fname)].append(score_speaker_similarity(scoring_vocoder, os.path.join(DATASET_PATH, user_dir, "wav48", "1"), output_path))
# clean up
if not args.keep_temp:
shutil.rmtree(output_path)
# determine the top-5 checkpoints (or fewer if there are less than 5 entries)
top_5_checkpoints = [(k, sum(v) / len(v)) for k, v in sorted(sim_score.items(), key = lambda item: sum(item[1]) / len(item[1]), reverse = True)[:5]]
# exit gracefully
if args.output_format == "txt":
print("")
print("Completed computing similarity score for the two sets of recordings. The best and the top-5 models based on their similarity scores are:")
print(" --> Best model : {}" . format(top_5_checkpoints[0]))
print(" --> Top-5 models: {}" . format(top_5_checkpoints))
print("")
print("Done; bye.")
print("")
elif args.output_format == "json":
json_data["out"]["best_model"] = top_5_checkpoints[0]
json_data["out"]["top_5_models"] = top_5_checkpoints
json_data["success"] = True
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
# ensure the temp path exists
os.makedirs(args.temp_path, exist_ok = True)
main(args)

View File

@@ -0,0 +1,161 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import argparse
import re
from itertools import combinations
import json
from utils.matching_utils import match_name_textualsim, match_name_mra
from utils.transcription_utils import get_transcription
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Score a given salutation recording wrt. its desired content, the actual salutation recording, and a generated transcription (using Potion's internal Transciption API) of the recording.")
parser.add_argument("--recording_path", type = str, required = True,
help = "Path to salutation recoding (.wav audio file)")
parser.add_argument("--first_name", type = str, required = True,
help = "First name that the salutation recoding is meant to use")
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
return parser.parse_args()
#
# load dictionary of common first names (Name DB source: World Gender Name Dictionary v2.0; https://dataverse.harvard.edu/dataset.xhtml?persistentId=doi:10.7910/DVN/MSEGSJ)
#
def load_names():
NAME_DICTIONARY = "./assets/wgnd_2_0_unique_names_only_limited_special_chars.csv"
# removing the characters
with open(NAME_DICTIONARY) as f:
names_list = [line.rstrip() for line in f]
names_set = set(names_list)
return names_set
#
# auxilliary function to generate a list of all combinations of words from a given list of words (w/o chaningthe order of words)
#
def get_combinations(word_list):
comb_list = word_list.copy()
for start, end in combinations(range(len(word_list)), 2):
comb_list.append(' '.join(word for word in word_list[start:end + 1]))
return comb_list
#
# main method
#
def main(args):
# set default score
score = -1.0
if args.output_format == "txt":
print("Commencing scoring of the given salutation recording:")
print("")
print(" + Salutation recording path: {}" . format(args.recording_path))
print(" + Salutation first name : {}" . format(args.first_name))
print("")
elif args.output_format == "json":
json_data = {
"success": False,
"in": {
"recording_path": format(args.recording_path),
"first_name": format(args.first_name)
},
"out": {
"score": score
}
}
# obtain a transcription for the given salutation recording
trans_success, trans_txt, trans_score = get_transcription(args.recording_path)
# proceed if a transcription was obtained successfully
if trans_success:
# check given first name against name database (Name DB source: https://dataverse.harvard.edu/dataset.xhtml?persistentId=doi:10.7910/DVN/MSEGSJ)
names_set = load_names()
# ensure all words / letters are lower case only
first_name = args.first_name.lower()
trans_txt = trans_txt.lower()
name_valid = False
if first_name in names_set:
name_valid = True
else:
# cannot compute advanced score for a name that we do not have in our first name database (i.e., fallback to confidence score from transcription service)
if args.output_format == "txt":
print("Unknown first name: {}" . format(args.first_name))
print("")
score = trans_score
if name_valid:
#
trans_candidate_names = re.findall(r" ([a-zA-Z_-]+)", trans_txt)
if len(trans_candidate_names) >= 2:
trans_candidate_names = get_combinations(trans_candidate_names)
cand_names_real = []
for cand_name in trans_candidate_names:
# check is cname is a valid name
if cand_name.lower() in names_set:
cand_names_real.append(cand_name)
if cand_names_real:
# name similarity with first_name
for cand_name_real in cand_names_real:
# check is cname is a valid name
jaro, lev = match_name_textualsim(first_name, cand_name_real)
mra = match_name_mra(first_name, cand_name_real)
#print("Scores ({}): {} -- {} -- {} -- {}" . format(cand_name_real, trans_score, jaro, lev, mra))
# score if the normalised Jaro-Winkler distance >= 0.875
# OR
# the normalised Jaro-Winkler distance >= 0.75 and the normalised Levenshtein distance is >= 0.7
# OR
# the normalised MRA >= 0.75
# else average
if jaro > 0.875:
score = (jaro + trans_score) / 2
break
elif (jaro >= 0.75) and (lev >= 0.7):
score = (((jaro + lev) / 2) + trans_score) / 2
break
elif mra >= 0.75:
score = (mra + trans_score) / 2
break
else:
score_new = (((jaro + lev + mra) / 3) + trans_score) / 2
if score_new > score:
score = score_new
if args.output_format == "txt":
print(" >> Salutation score : {}" . format(score))
print("")
print("Done; bye.")
print("")
elif args.output_format == "json":
json_data["out"]["score"] = round(score, 2)
json_data["success"] = True
print(json.dumps(json_data))
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
main(args)

View File

@@ -0,0 +1,144 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import os
import argparse
import subprocess
import json
import uuid
import torch
from TTS.tts.utils.speakers import SpeakerManager
from utils.synthesize_utils import init_synth, synthesize, save_waveform
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Code to synthesize speech for a given voice model")
parser.add_argument("--voice_model_path", type = str, required = True,
help = "Path to cloned voice model")
parser.add_argument("--voice_model_config_path", type = str, required = True,
help = "Path to config file for the cloned voice model")
parser.add_argument('--speaker_embeddings_path', type = str, required = True,
help = "Path to speaker's embeddings file (i.e., pre-computed embeddings typically stored with the speaker's dataset)")
parser.add_argument("--txt", type = str, required = True,
help = "Text to synthesize")
parser.add_argument("--output_path", type = str, default = "results/speech",
help = "Path to store generated speech assets")
parser.add_argument("--target_sampling_rate", type = int, default = 48000,
help = "Desired sampling rate (in Hz) for output file")
parser.add_argument('--speech_sample_wav_path', type = str, default = None,
help = "Path to a sample utterance of the speaker (used for style transfer)")
parser.add_argument('--speech_sample_txt', type = str, default = None,
help = "Text of the sample utterance of the speaker (used for style transfer)")
parser.add_argument("--trim_silence", default = True, action = "store_false",
help = "Signal whether to trim silence from synthesised speech")
parser.add_argument("--use_cpu", default = False, action = "store_true", # untested!!!
help = "Signal that CPU should be used even if a CUDA-device is available")
parser.add_argument("--output_format", type = str, choices = ["txt", "json"], default = "txt",
help = "Output format; available choices include 'txt' for human readible text and 'json' for JSON formatting")
return parser.parse_args()
#
# main speech synthesizing method
#
def main(args):
# define assets required for using a pretrained voice
MODEL_PATH = args.voice_model_path
CONFIG_PATH = args.voice_model_config_path
SPK_EMBEDDINGS_PATH = args.speaker_embeddings_path
if args.output_format == "txt":
print("Commencing speech synthesizing:")
print("")
print(" + Voice model file path : {}" . format(MODEL_PATH))
print(" + Voice model config file: {}" . format(CONFIG_PATH))
print(" + Speaker embeddings file: {}" . format(SPK_EMBEDDINGS_PATH))
print(" + Output path : {}" . format(args.output_path))
print(" + Text to synthesize : {}" . format(args.txt))
print("")
elif args.output_format == "json":
json_data = {
"success": False,
"in": {
"voice_model_path": format(MODEL_PATH),
"voice_model_config_path": format(CONFIG_PATH),
"speaker_embeddings_path": format(SPK_EMBEDDINGS_PATH)
},
"out": {
"speech_original_path": "",
"speech_resampled_path": ""
}
}
# determine whether CUDA support is available and set device parameters accordingly
use_cuda = torch.cuda.is_available()
if args.output_format == "txt":
print(" + CUDA availability : {}" . format(use_cuda))
if args.use_cpu:
device = "cpu"
USE_CUDA = False
elif use_cuda:
device = "cuda"
USE_CUDA = True
else:
device = "cpu"
USE_CUDA = False
if args.output_format == "txt":
print(" + Compute device used : {}" . format(device))
# init speaker manager
speaker_manager = None
speaker_manager = SpeakerManager(d_vectors_file_path = SPK_EMBEDDINGS_PATH)
if args.output_format == "txt":
print(" + No. of speakers : {}" . format(speaker_manager.num_speakers))
print(" + Speaker's names : {}" . format(speaker_manager.embedding_names))
print(" + No. of embeddings : {}" . format(speaker_manager.num_embeddings))
print("")
# assert that only one speaker is present in the embedding's file
assert speaker_manager.num_speakers == 1, f"Number of speakers in the given embedding's file MUST be one; found {speaker_manager.num_speakers} speakers!"
# initialise speech synthesization
voice_config, voice_model = init_synth(CONFIG_PATH, MODEL_PATH, speaker_embeddings_file = SPK_EMBEDDINGS_PATH, use_cuda = USE_CUDA)
# synthesize speech
waveform = synthesize(voice_config, voice_model, args.txt, speaker_embeddings = speaker_manager.get_mean_embedding(speaker_manager.embedding_names[0], speaker_manager.num_embeddings), speech_sample_wav = args.speech_sample_wav_path, speech_sample_txt = args.speech_sample_txt, use_cuda = USE_CUDA, trim_silence = args.trim_silence)
# save the synthesize speech
output_fname_prefix = str(uuid.uuid4())
output_fname = output_fname_prefix + ".wav"
save_waveform(voice_config, voice_model, waveform, os.path.join(args.output_path, output_fname))
# convert the synthesize speech waveform to the target sampling rate
output_resampled_fname = output_fname_prefix + "_sr" + str(args.target_sampling_rate) + ".wav"
subprocess.run(["ffmpeg", "-i", os.path.join(args.output_path, output_fname), "-ar", str(args.target_sampling_rate), os.path.join(args.output_path, output_resampled_fname)], check=True)
# exit gracefully
if args.output_format == "txt":
print("")
print(">>> Saving origianl output to : {}" . format(os.path.join(args.output_path, output_fname)))
print(">>> Saving resampled output to: {}" . format(os.path.join(args.output_path, output_resampled_fname)))
print("")
print("Speech synthesizing has completed. Bye.")
elif args.output_format == "json":
json_data["out"]["speech_original_path"] = format(os.path.join(args.output_path, output_fname))
json_data["out"]["speech_resampled_path"] = format(os.path.join(args.output_path, output_resampled_fname))
json_data["success"] = True
print(json.dumps(json_data))
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
# ensure the output path exists
os.makedirs(args.output_path, exist_ok = True)
main(args)

View File

@@ -0,0 +1,48 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
####################################################################################
### ###
### Configuration File for Multi-Speaker Baseline Model Training & Voice Cloning ###
### ###
####################################################################################
import os
# Data Sets
## VCTK (v0.92), sampling rate: 48000
VCTK_PRESET = "VCTK"
VCTK_DATASET_NAME = "VCTK"
VCTK_DATASET_FORMATTER = "vctk"
VCTK_DATASET_FILE_FORMAT = "flac"
VCTK_DATASET_PATH = "results/datasets/sr22050/VCTK-Corpus-0.92"
VCTK_SPK_EMB_PATH = os.path.join(VCTK_DATASET_PATH, "speakers.pth")
## LibriTTS TC360, sampling rate: 24000
LIBRITTS_TC360_PRESET = "LibriTTS_tc360"
LIBRITTS_TC360_DATASET_NAME = "LibtriTTS-tc360"
LIBRITTS_TC360_DATASET_FORMATTER = "libri_tts"
LIBRITTS_TC360_DATASET_FILE_FORMAT = "wav"
LIBRITTS_TC360_DATASET_PATH = "results/datasets/sr22050/LibriTTS/train-clean-360"
LIBRITTS_TC360_SPK_EMB_PATH = os.path.join(LIBRITTS_TC360_DATASET_PATH, "speakers.pth")
## DAPS
## Potion salutation recordings
POTION_SALUT_PRESET = "POTION_Salut"
POTION_SALUT_DATASET_NAME = "potion-Salut"
POTION_SALUT_DATASET_FORMATTER = "vctk_old"
POTION_SALUT_DATASET_FILE_FORMAT = "wav"
POTION_SALUT_DATASET_PATH = "results/datasets/sr22050/potion-salut-corpus-4ac24ce8-8405-4b70-8b48-018d4492f6e9"
POTION_SALUT_SPK_EMB_PATH = os.path.join(POTION_SALUT_DATASET_PATH, "speakers.pth")
## Potion voice cloning recordings
POTION_SALUT_PRESET = "potion_voice_cloning"
POTION_SALUT_DATASET_NAME = ""
POTION_SALUT_DATASET_FORMATTER = "vctk_old"
POTION_SALUT_DATASET_FILE_FORMAT = "wav"

View File

@@ -0,0 +1,238 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import sys
import os
import argparse
import torch
# load coqui-ai/trainer libraries
from trainer import Trainer, TrainerArgs
# load coqui-ai/TTS libraries
from TTS.tts.configs.shared_configs import BaseDatasetConfig
from TTS.tts.configs.vits_config import VitsConfig
from TTS.tts.datasets import load_tts_samples
from TTS.tts.models.vits import Vits, VitsArgs, VitsAudioConfig
import train_config as tc
#
# parse command line arguments
#
def parse_cmdline_args():
parser = argparse.ArgumentParser(
description = "Code to train multi-speaker baseline model")
parser.add_argument("--datasets", type = str, nargs = "+", required = True,
choices = (tc.VCTK_PRESET, tc.LIBRITTS_TC360_PRESET, tc.POTION_SALUT_PRESET),
help = "List of training datasets to be included in training run.")
parser.add_argument("--output_path", type = str, default = "results/baseline-models",
help = "Path to store trained / generated assets")
parser.add_argument("--batch_size", type = int, default = 32, # 96 is suitable for AWS g5 instances using VCTK v0.80 only
help = "Batch size for training run") # 32 is suitable for AWS g5 instances using VCTK v0.92, LibriTTS 360 and Potion salutations
parser.add_argument("--max_epochs", type = int, default = 100, # 250 for batch size 64 (with VCTK only)
help = "Maximum number of epochs for training run") # 100 for batch size 32 (with VCTK v0.92, LibriTTS 360 and POTION_Salut)
return parser.parse_args()
#
# main training method (VITS multi-speaker model)
#
def main(args):
print("Commencing training of a new multi-speaker potion-voice baseline model:")
print("")
print(" + Datasets : {}" . format(args.datasets))
print(" + Output path : {}" . format(args.output_path))
print(" + Batch size : {}" . format(args.batch_size))
print(" + Training runs (max epochs): {}" . format(args.max_epochs))
print("")
# determine whether CUDA support is available and set device parameters accordingly
use_cuda = torch.cuda.is_available()
print(" + CUDA availability : {}" . format(use_cuda))
if use_cuda:
device = "cuda"
device_torch = torch.device("cuda")
else:
device = "cpu"
device_torch = torch.device("cpu")
print(" + Compute device used : {}" . format(device))
print("")
# define training data sets
dataset_config_list = []
speaker_embeddings_list = []
# VCTK (v0.92)
if tc.VCTK_PRESET in args.datasets:
vctk_dataset_config = BaseDatasetConfig(dataset_name = tc.VCTK_DATASET_NAME, formatter = tc.VCTK_DATASET_FORMATTER, language = "en-us", path = tc.VCTK_DATASET_PATH)
dataset_config_list.append(vctk_dataset_config)
speaker_embeddings_list.append(tc.VCTK_SPK_EMB_PATH)
# LibriTTS
if tc.LIBRITTS_TC360_PRESET in args.datasets:
libritts_dataset_config = BaseDatasetConfig(dataset_name = tc.LIBRITTS_TC360_DATASET_NAME, formatter = tc.LIBRITTS_TC360_DATASET_FORMATTER, language = "en-us", path = tc.LIBRITTS_TC360_DATASET_PATH)
dataset_config_list.append(libritts_dataset_config)
speaker_embeddings_list.append(tc.LIBRITTS_TC360_SPK_EMB_PATH)
# DAPS
# Potion recordings dataset
if tc.POTION_SALUT_PRESET in args.datasets:
potion_dataset_config = BaseDatasetConfig(dataset_name = tc.POTION_SALUT_DATASET_NAME, formatter = tc.POTION_SALUT_DATASET_FORMATTER, language = "en-us", path = tc.POTION_SALUT_DATASET_PATH)
dataset_config_list.append(potion_dataset_config)
speaker_embeddings_list.append(tc.POTION_SALUT_SPK_EMB_PATH)
# set VITS training parameters
audio_config = VitsAudioConfig(
sample_rate = 22050,
win_length = 1024,
hop_length = 256,
num_mels = 80,
mel_fmin = 0,
mel_fmax = None,
)
vitsArgs = VitsArgs(
use_speaker_embedding = False,
use_d_vector_file = True,
d_vector_file = speaker_embeddings_list,
d_vector_dim = 512,
num_layers_text_encoder = 10
)
config = VitsConfig(
model_args = vitsArgs,
audio = audio_config,
run_name = "vits_potion",
use_speaker_embedding = False,
use_d_vector_file = True,
d_vector_file = speaker_embeddings_list,
d_vector_dim = 512,
batch_size = args.batch_size,
eval_batch_size = 16,
batch_group_size = 0, # changing this to 5 (VITS training default) slows training down, but doesn't have any positive training effects
num_loader_workers = 4,
num_eval_loader_workers = 4,
run_eval = True,
test_delay_epochs = -1,
epochs = args.max_epochs,
text_cleaner = "english_cleaners",
use_phonemes = False,
phoneme_language = "en-us",
phoneme_cache_path = os.path.join(args.output_path, "phoneme_cache"),
compute_input_seq_cache = True,
print_step = 50,
print_eval = True,
mixed_precision = True,
max_text_len = 325,
output_path = args.output_path,
save_checkpoints = True,
save_step = 5000,
save_n_checkpoints = 20,
save_all_best = True,
datasets = dataset_config_list,
cudnn_benchmark = False,
#characters = {
# "pad": "_",
# "eos": "&",
# "bos": "*",
# "characters": "!¡'(),-.:;¿?abcdefghijklmnopqrstuvwxyz «°±µ»$%&‘’‚“`”„",
# "punctuations": "!¡'(),-.:;¿? ",
# "phonemes": None,
# "unique": True
#},
test_sentences = [
# VCTK
["It took me quite a long time to develop a voice, and now that I have it I'm not going to be silent.", "VCTK_p299"], # 299 - F, American, California
["Hey! Sandra.", "VCTK_p302"], # 302 - M, Canadian, Montreal
["I'm sorry Dave. I'm afraid I can't do that.", "VCTK_p308"], # 308 - F, American, Alabama
["This cake is great. It's so delicious and moist.", "VCTK_p334"], # 334 - M, American, Chicago
["Prior to November 22, 1963.", "VCTK_p363"], # 363 - M, Canadian, Toronto
["It took me quite a long time to develop a voice, and now that I have it I'm not going to be silent.", "VCTK_p376"], # 376 - M, Indian
# LibriTTS
["It took me quite a long time to develop a voice, and now that I have it I'm not going to be silent.", "LTTS_38"], # 38 - M - train-clean-360 R. Francis Smith
["Hey! Sandra.", "LTTS_22"], # 22 - F - train-clean-360 Michelle Crandall
["I'm sorry Dave. I'm afraid I can't do that.", "LTTS_329"], # 329 - M - train-clean-360 Todd Cranston-Cuebas
["This cake is great. It's so delicious and moist.", "LTTS_224"], # 224 - F - train-clean-360 Caitlin Kelly
["Prior to November 22, 1963.", "LTTS_339"], # 339 - F - train-clean-360 Heather Ordover
["It took me quite a long time to develop a voice, and now that I have it I'm not going to be silent.", "LTTS_1779"], # 1779 - F - train-clean-360 Cynthia Zocca
# DAPS
# Potion Salutation Recordings
["Hey! Andrew.", "VCTK_old_POTION_6231a04a9f5f7707120b4215"], # Potion user
["Hey, Michelle.", "VCTK_old_POTION_628c025a09943300254532ae"], # Potion user
["Hey! George.", "VCTK_old_POTION_6189854312bfc264a528c3c3"], # Potion user
["Hey there, Rachel.", "VCTK_old_POTION_63163f7b2be1c500219f3d04"], # Potion user
#["I'm sorry Dave. I'm afraid I can't do that.", "VCTK_old_POTION_63222943fd2bff2e1c651b3b"] # Potion user - not in yet
]
)
# load training samples
train_samples, eval_samples = load_tts_samples(config.datasets, eval_split = True, eval_split_max_size = config.eval_split_max_size, eval_split_size = config.eval_split_size)
# init VITS model
model = Vits.init_from_config(config)
# init multi-speaker training
trainer = Trainer(
TrainerArgs(),
config,
args.output_path,
model = model,
train_samples = train_samples,
eval_samples = eval_samples
)
# trigger model training
try:
trainer.fit()
except (KeyboardInterrupt, SystemExit):
print("Training stopped manually (via keyboard interrupt)! Bye.")
exit(0)
# exit gracefully
print("")
print("Completed training a new multi-speaker potion-voice baseline model, which can be found at:")
print(" --> {}" . format(args.output_path))
print("")
print("Done; bye.")
print("")
if __name__ == "__main__":
# parse command line arguments
args = parse_cmdline_args()
# clear command line arguments to avoid triggering argparse features part of Trainer / coqpit imports
# Traceback (most recent call last):
# File "train_multispeaker_baseline_model.py", line 208, in <module>
# main(args)
# File "train_multispeaker_baseline_model.py", line 177, in main
# trainer = Trainer(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 360, in __init__
# config, new_fields = self.init_training(args, coqpit_overrides, config)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/trainer/trainer.py", line 594, in init_training
# config.parse_known_args(coqpit_overrides, relaxed_parser=True)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 843, in parse_known_args
# parser = self.init_argparse(arg_prefix=arg_prefix, relaxed_parser=relaxed_parser)
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 881, in init_argparse
# _init_argparse(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 529, in _init_argparse
# parser = _init_argparse(
# File "/home/ubuntu/dev/potion-voice_venv/lib/python3.8/site-packages/coqpit/coqpit.py", line 550, in _init_argparse
# return default.init_argparse(
# AttributeError: 'str' object has no attribute 'init_argparse'
sys.argv = [sys.argv[0]]
# ensure the output path exists
os.makedirs(args.output_path, exist_ok = True)
main(args)

View File

@@ -0,0 +1,22 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import textdistance
#
# Name matching via textual similarity search
# + Returns two (normalised) distance measures: the Jaro-Winkler Distance and the regular Levenshtein Distance
#
def match_name_textualsim(name1, name2):
jaro_winkler = textdistance.jaro_winkler.normalized_similarity(name1, name2)
levenshtein = textdistance.levenshtein.normalized_similarity(name1, name2)
return jaro_winkler, levenshtein
#
# Name matching via phonetic matching algorithm (using the normalised Match Rating Approach)
#
def match_name_mra(name1, name2):
return textdistance.mra.normalized_similarity(name1, name2)

View File

@@ -0,0 +1,37 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
from pathlib import Path
from itertools import groupby
import numpy as np
from resemblyzer import preprocess_wav, VoiceEncoder
def init_scoring_vocoder():
# initialise voice encoder (using CUDA by default; CPU as fallback)
encoder = VoiceEncoder()
return encoder
def score_speaker_similarity(scoring_vocoder, spk_a_fpaths, spk_b_fpaths):
# filepaths to waveforms
wav_fpaths = list(Path(spk_a_fpaths).glob("*.wav")) + list(Path(spk_b_fpaths).glob("*.wav"))
# group the wavs per speaker and load them using the preprocessing function provided with Resemblyzer to load wavs in memory
# - normalizes the volume, trims long silences and resamples the wav to the correct sampling rate
speaker_wavs = {speaker: list(map(preprocess_wav, wav_fpaths)) for speaker, wav_fpaths in groupby(wav_fpaths, lambda wav_fpath: wav_fpath.parent.stem)}
# compute similarity between two speaker embeddings
# - divides the utterances of each speaker in groups of identical size and embed each group as a speaker embedding
spk_embeds_a = np.array([scoring_vocoder.embed_speaker(wavs[:len(wavs) // 2]) for wavs in speaker_wavs.values()])
spk_embeds_b = np.array([scoring_vocoder.embed_speaker(wavs[len(wavs) // 2:]) for wavs in speaker_wavs.values()])
spk_sim_matrix = np.inner(spk_embeds_a, spk_embeds_b)
sim_score = np.average([spk_sim_matrix[0, 1], spk_sim_matrix[1, 0]])
return(sim_score)

View File

@@ -0,0 +1,77 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import numpy as np
# load coqui-ai/TTS libraries
from TTS.config import load_config
from TTS.tts.models import setup_model as setup_tts_model
from TTS.tts.utils.synthesis import synthesis, trim_silence
def init_synth(config_path, voice_model_path, speakers_file_path = None, speaker_embeddings_file = None, use_cuda = True, use_phonemes = False):
# load config and customise config parameters (those that are different during training and inference / synthesizing)
config = load_config(config_path)
if not speakers_file_path is None:
config.use_speaker_embedding = True,
config.use_d_vector_file = False,
config.speakers_file = speakers_file_path
config.model_args["use_speaker_embedding"] = True,
config.model_args["use_d_vector_file"] = False,
config.model_args["speakers_file"] = speakers_file_path
else:
config.d_vector_file = speaker_embeddings_file
config.model_args["d_vector_file"] = speaker_embeddings_file
# set whether or not phonemes are used
config.use_phonemes = use_phonemes
# load cloned voice model
model = setup_tts_model(config = config)
model.load_checkpoint(config, voice_model_path, eval = True)
if use_cuda:
model.cuda()
return config, model
def synthesize(config, voice_model, txt, speaker_embeddings = None, speaker_id = None, speech_sample_wav = None, speech_sample_txt = None, use_cuda = True, trim_silence = True):
# disable language selection
#language_id = 0
language_id = None
# set default voice encoder
use_gl = True
# synthesize voice
outputs = synthesis(
model = voice_model,
text = txt,
CONFIG = config,
use_cuda = use_cuda,
speaker_id = speaker_id,
style_wav = speech_sample_wav,
style_text = speech_sample_txt,
use_griffin_lim = use_gl,
do_trim_silence = trim_silence,
d_vector = speaker_embeddings,
language_id = language_id
)
waveform = outputs["wav"]
waveform = waveform.squeeze()
# trim silence (disabled due to some "TypeError: 'bool' object is not callable" bug that needs to be investigated)
#if (config.audio["do_trim_silence"]) or (trim_silence):
# waveform = trim_silence(waveform, voice_model.ap)
return waveform
def save_waveform(config, voice_model, waveform, out_path):
wav = np.array(waveform)
voice_model.ap.save_wav(wav, out_path, config["audio"].sample_rate)

View File

@@ -0,0 +1,94 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import sys
import os
import requests
from requests.structures import CaseInsensitiveDict
import json
from time import sleep
# set transcription service API endpoint and token (retrieved from operating system's ENV variables)
# + sample endpoints:
# - [dev] "https://development.sendpotion.com/api/transcript"
# - [staging] "https://staging.sendpotion.com/api/transcript"
API_ENDPOINT = os.environ.get("TRANSCRIPTION_API_ENDPOINT")
API_TOKEN = os.environ.get("TRANSCRIPTION_API_TOKEN")
#
# Using potions internal transcription API endpoint, get a transcription for a given (wav) audio recording
# + returns a triple:
# - Boolean ......... indicating success (True) or failure (False)
# - String / None ... transcription text (or None in failure case)
# - Float / None .... transcription confidence score (or None in failure case)
#
def get_transcription(wav_fname):
# validate that transcription service API endpoint and token are set
if (API_ENDPOINT is None) or (API_TOKEN is None):
# terminate
print("TRANSCRIPTION_API_ENDPOINT and TRANSCRIPTION_API_TOKEN environment variables MUST be set!")
sys.exit(1)
# set request header to contain (bearer) API token
headers = CaseInsensitiveDict()
headers["Accept"] = "application/json"
headers["Authorization"] = "Bearer " + str(API_TOKEN)
# set files field (data is empty)
files = {'wav': open(wav_fname, 'rb')}
# issue POST request and save response as response object
response = requests.post(url = API_ENDPOINT, headers = headers, files = files)
# test for auth error
# test for timeout
# check if the status code is not an error code (i.e., 4xx or 5xx)
success = False
if response:
# extracting response text
response_text = response.text
response_json = json.loads(response_text)
#print(response_json)
if response.ok: # synch call
success = True
trans_text = response_json["transcriptObj"]["text"]
trans_score = float(response_json["transcriptObj"]["confidence"])
else: # fallback to asynch call
# wait up to 60 seconds for the transcription to be ready; try every 5 seconds
wait = 0
while wait < 60:
sleep(5)
wait += 5
# issue GET request using the previously returned reqiestId and save response as response object
response_get = requests.get(url = API_ENDPOINT + ':' + response_json["requestId"])
# check if the status code is not an error code (i.e., 4xx or 5xx)
if response_get.ok:
response_get_text = response_get.text
response_get_json = json.loads(response_get_text)
success = True
trans_text = response_get_json["transcriptObj"]["text"]
trans_score = float(response_get_json["transcriptObj"]["confidence"])
break
# in case no successful response is received even after a 60 seconds waiting period -> proceed without transcription
#if not response_get.ok:
# print("Response: FAILED.")
#else:
# print("ERROR: {} ({})" . format(response.status_code, response.text))
if success:
return success, trans_text, trans_score
else:
return False, None, None

View File

@@ -0,0 +1,267 @@
const fs = require('fs')
const exec = require('child_process').exec
const AWS = require('aws-sdk')
const Bugsnag = require('@bugsnag/js')
const uuid = require('uuid').v4
const version = require('./package.json').version
const sqs = require('../app/services/sqs')
const s3 = require('../app/services/s3')
const userAudioProfileService = require('./user_audio_profile')
const recordingModel = require('./recording')
const recordingSalutationModel = require('./recording_salutation')
const jobService = require('./job')
const salutationService = require('./salutation')
let throttleMessageFetching = true
AWS.config.update({ region: 'us-west-2' })
const sqsQueueUrl = process.env.SQS_URL
const mongoUriDev = process.env.MONGODB_URI_DEV
const mongoUriStaging = process.env.MONGODB_URI_STAGING
const mongoUriProd = process.env.MONGODB_URI_PROD
const APP_ENV = process.env.POTION_APP_ENV
const mongoose = require('mongoose')
function execShellCommand(cmd) {
// const exec = require("child_process").exec;
return new Promise((resolve, reject) => {
exec(cmd, { maxBuffer: 1024 * 1000000 }, (error, stdout, stderr) => {
if (error) {
console.log('Error while processing python command', error)
reject(error)
}
console.log('Stdout --- ', stdout)
console.log('Std error --- ', stderr)
resolve(stdout || stderr)
})
})
}
function connectDB(dbUri, retryCount = 0) {
return new Promise((resolve, reject) => {
console.log('Connection Attempt : ', retryCount)
mongoose.set('strictQuery', true)
mongoose
.connect(dbUri)
.then((msg) => {
console.log('Connected to Mongo DB !')
resolve()
})
.catch((err) => {
console.log('Failed to connect dns mongo: ', err)
if (retryCount < 6) {
retryCount++
connectDB(dbUri, retryCount)
}
})
})
}
const processQueue = () => {
/* eslint-disable no-async-promise-executor */
return new Promise(async (resolve, reject) => {
try {
const response = await sqs.fetchMessageFromSQS(sqsQueueUrl)
if (
typeof response.Messages !== 'undefined' &&
response.Messages.length > 0
) {
throttleMessageFetching = false
const job = JSON.parse(response.Messages[0].Body)
const receiptHandle = response.Messages[0].ReceiptHandle
try {
await sqs.deleteMessageFromSQS(sqsQueueUrl, receiptHandle)
const {
userAudioProfileId,
text,
firstName,
salutationId,
recordingId,
baseUrlForPotionAi,
env,
} = job
const DB_URI =
env === 'production'
? mongoUriProd
: env === 'staging'
? mongoUriStaging
: mongoUriDev
console.log('DB_URI ', DB_URI)
await connectDB(DB_URI)
// read the path for the training model for the this users audio profile
const userAudioProfile = await userAudioProfileService.find({
_id: userAudioProfileId,
status: 'completed',
})
if (userAudioProfile) {
const { training_model_path, userId } = userAudioProfile[0]
const {
voice_model_light_path,
voice_model_config_light_path,
voice_model_speakers_file_path, // name for speakers embeddings file path
} = training_model_path
const outputPath = `/tmp/${uuid()}/`
if (!fs.existsSync(outputPath)) {
fs.mkdirSync(outputPath, { recursive: true })
}
const AI_COMMAND = `python3 ../voice-cloning/synthesize_speech.py --voice_model_path ${voice_model_light_path} --voice_model_config_path ${voice_model_config_light_path} --speaker_embeddings_path ${voice_model_speakers_file_path} --txt "${text}" --output_path ${outputPath}`
console.log('AI_COMMAND ', AI_COMMAND)
const SYNTHESIZE_AI_LABEL = `Time consumed by AI` + Math.random()
console.time(SYNTHESIZE_AI_LABEL)
const aiResponse = await execShellCommand(AI_COMMAND)
console.timeEnd(SYNTHESIZE_AI_LABEL)
let generatedFileName = ''
fs.readdirSync(`${outputPath}`).forEach((file) => {
if (file.includes('sr48000.wav')) generatedFileName = file
})
// upload the file to s3
const uploadParams = {
filePath: `${outputPath}${generatedFileName}`,
bucket: `recordings-${env}`,
fileName: `${uuid()}_salutation_${firstName.replace(
'-',
'_'
)}.wav`,
contentType: 'audio/x-wav',
fileType: 'wav',
}
console.time('Time to Upload video on S3')
const greetingUploadResponse = await s3.upload(uploadParams)
console.timeEnd('Time to Upload video on S3')
// Create new entry with the s3 path to salutation collection for the user and its profile id
// upsert the salutation
await salutationService.updateOrCreate(
{
firstName: firstName,
salutationVideo: greetingUploadResponse,
userAudioProfileId,
},
userId
)
// update the dynamic recordings for the current dynamic video with salutation url
const salutationToUpdate = await recordingSalutationModel.findOne({
_id: salutationId,
deleted: false,
})
const recordingToUpdate = await recordingModel.findOne({
_id: recordingId,
deleted: false,
})
if (
salutationToUpdate &&
salutationToUpdate.deleted === false &&
recordingToUpdate
) {
const jobsToInsert = []
await recordingSalutationModel.findOneAndUpdate(
{
_id: salutationId,
},
{
$set: {
salutationVideo: greetingUploadResponse,
},
}
)
const jobData = {
originalGreeting: recordingToUpdate.masterSalutationVideoUrl,
originalVideo:
recordingToUpdate.originalVideoUrl ||
recordingToUpdate.urls[0].url,
cropTimestamp: recordingToUpdate.cropTimestamp,
greetingClips: [greetingUploadResponse],
greetingObjects: [
{
greetingId: salutationToUpdate._id,
firstName: firstName,
videoUrl: greetingUploadResponse,
},
],
requestOrigin: baseUrlForPotionAi,
environment: env,
recordingId: recordingToUpdate._id,
salutation: salutationToUpdate._id,
dynamicVideoType: recordingToUpdate.dynamicVideoType,
}
jobsToInsert.push({
firstName,
recordingId: recordingToUpdate._id,
userId: recordingToUpdate.userId,
salutationId: salutationToUpdate._id,
metadata: jobData,
})
// create the job for the ai to create processing
if (jobsToInsert.length) {
await jobService.insertMany(jobsToInsert)
}
}
fs.unlinkSync(`${outputPath}${generatedFileName}`)
console.log(`[deleted] ${outputPath}${generatedFileName}`)
} else {
Bugsnag.notify(
new Error(
`audio profile training model not found ` + JSON.stringify(job)
)
)
resolve() // to continue working on new jobs
}
} catch (error) {
console.error('Error while synthesizing audio', { error })
Bugsnag.notify(
new Error(`Unable to synthesize audio ` + JSON.stringify(job))
)
Bugsnag.notify(error)
resolve() // to continue working on new jobs
}
} else {
throttleMessageFetching = true
}
resolve()
} catch (error) {
console.error('Error while synthesizing audio', { error })
Bugsnag.notify(error)
resolve() // to continue working on new jobs
} finally {
mongoose.connection.close()
}
})
}
function sleep(ms) {
return new Promise((resolve) => {
setTimeout(resolve, ms)
})
}
const init = async () => {
Bugsnag.start({
appVersion: APP_ENV + version,
apiKey: process.env.BUGSNAG_BACKEND_KEY,
releaseStage: process.env.NODE_ENV,
})
try {
while (true) {
await processQueue()
if (throttleMessageFetching) await sleep(2000)
}
} catch (error) {
Bugsnag.notify(error)
}
}
init()

View File

@@ -0,0 +1,4 @@
const Job = require('./job_model')
const JobService = require('./job_service')
module.exports = JobService(Job)

View File

@@ -0,0 +1,54 @@
const mongoose = require('mongoose')
const Schema = mongoose.Schema
const JobSchema = Schema(
{
recordingId: {
type: Schema.Types.ObjectId,
required: false
},
userId: {
type: Schema.Types.ObjectId,
required: false
},
salutationId: {
type: Schema.Types.ObjectId,
required: false
},
type: {
type: String,
required: false,
default: 'ai-job'
},
firstName: {
type: String,
default: ''
},
weight: {
type: Number,
default: 0
},
email: {
type: String,
default: ''
},
status: {
type: String,
required: false,
default: 'created'
},
metadata: {
type: Schema.Types.Mixed,
default: null
},
deleted: {
type: Boolean,
required: true,
default: false
}
},
{
timestamps: true
}
)
module.exports = mongoose.model('Job', JobSchema)

View File

@@ -0,0 +1,136 @@
const StringifyUtils = require('../../app/services/utils/logService')
const create = (Job) => async (jobData) => {
try {
const newJob = new Job({ ...jobData })
const savedJob = await newJob.save()
return savedJob
} catch (error) {
const details = { jobData }
console.log(
'ERROR - JOB SERVICE > create',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const insertMany = (Job) => async (jobData) => {
try {
const inserted = await Job.insertMany(jobData)
return inserted
} catch (error) {
const details = { jobData }
console.log(
'ERROR - JOB SERVICE > insertMany',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const read = (Job) => async (filter) => {
try {
const foundJob = await Job.findOne({
...filter,
deleted: false,
})
return foundJob
} catch (error) {
const details = { filter }
console.log(
'ERROR - JOB SERVICE > read',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const find = (Job) => async (filter) => {
try {
const foundJobs = await Job.find({
...filter,
deleted: false,
})
return foundJobs
} catch (error) {
const details = { filter }
console.log(
'ERROR - JOB SERVICE > find',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const update = (Job) => async (job) => {
try {
const updatedJob = await Job.findOneAndUpdate({ _id: job._id }, job, {
new: true,
})
return updatedJob
} catch (error) {
const details = { job }
console.log(
'ERROR - JOB SERVICE > update',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const remove = (Job) => async (filter) => {
try {
const updatedJob = await Job.findOneAndUpdate(
{ ...filter },
{
$set: {
deleted: true,
},
},
{ new: true }
)
return updatedJob
} catch (error) {
const details = { filter }
console.log(
'ERROR - JOB SERVICE > remove',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const removeMany = (Job) => async (filter) => {
try {
const updatedJob = await Job.updateMany(
{ ...filter },
{
$set: {
deleted: true,
},
},
{ new: true }
)
return updatedJob
} catch (error) {
const details = { filter }
console.log(
'ERROR - JOB SERVICE > removeMany',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
module.exports = (Job) => {
return {
create: create(Job),
insertMany: insertMany(Job),
read: read(Job),
remove: remove(Job),
removeMany: removeMany(Job),
update: update(Job),
find: find(Job),
}
}

View File

@@ -0,0 +1,23 @@
{
"name": "voice-synthesizer-job-handler",
"version": "1.0.0",
"description": "This will handle the voice synthesizer jobs",
"main": "index.js",
"scripts": {
"deploy-production": "npx dotenv-cli -e ./app-scripts/env-aws-code-deploy/.env.production.aws-code-deploy node ./app-scripts/deploy-scripts/deploy-production.js",
"deploy-staging": "npx dotenv-cli -e ./app-scripts/env-aws-code-deploy/.env.staging.aws-code-deploy node ./app-scripts/deploy-scripts/deploy-staging.js"
},
"dependencies": {
"@bugsnag/js": "^7.3.5",
"aws-sdk": "^2.752.0",
"fs-extra": "^9.0.1",
"mongoose": "^6.8.0",
"rimraf": "^3.0.2",
"uuid": "^8.3.2"
},
"devDependencies": {
"aws-code-deploy": "^1.0.11"
},
"author": "potion Team",
"license": "ISC"
}

View File

@@ -0,0 +1,13 @@
apps:
- name: synthsizer-job
script: index.js
watch: false
autorestart: true
instances: 1
time: true
env:
NODE_ENV: 'production'
SQS_URL: 'https://sqs.us-west-2.amazonaws.com/[REDACTED_AWS_ACCOUNT_1961]/potion-voice-synthesizer-ai-staging.fifo'
APP_ENV: 'development'
BUGSNAG_BACKEND_KEY: '[REDACTED_generic-api-key]'
MONGODB_URI_DEV: 'mongodb+srv://[REDACTED_MONGO_USER_deve]:scrubbed_1@example.com7.mongodb.net/potion_development?retryWrites=true&w=majority'

View File

@@ -0,0 +1,14 @@
apps:
- name: synthsizer-job
script: index.js
watch: false
autorestart: true
instances: 1
time: true
env:
NODE_ENV: 'production'
SQS_URL: 'https://sqs.us-west-2.amazonaws.com/[REDACTED_AWS_ACCOUNT_1961]/potion-voice-synthesizer-ai-production.fifo'
APP_ENV: 'production'
BUGSNAG_BACKEND_KEY: '[REDACTED_generic-api-key]'
MONGODB_URI_DEV: 'mongodb+srv://[REDACTED_MONGO_USER_deve]:scrubbed_1@example.com7.mongodb.net/potion_development?retryWrites=true&w=majority'
MONGODB_URI_PROD: 'mongodb+srv://[REDACTED_MONGO_USER_prod]:scrubbed_2@example.com.net/potion_production?retryWrites=true&w=majority'

View File

@@ -0,0 +1,4 @@
const Recording = require('./recording_model')
module.exports = Recording

View File

@@ -0,0 +1,406 @@
const mongoose = require('mongoose')
const Schema = mongoose.Schema
const RecordingSchema = Schema(
{
userId: {
type: Schema.Types.ObjectId,
ref: 'User',
required: true,
},
urls: [
new mongoose.Schema(
{
quality: {
type: String,
required: false,
default: '',
},
url: {
type: String,
required: false,
default: '',
},
},
{ _id: false }
),
],
faceVideoUrl: {
type: String,
required: false,
default: '',
},
title: {
type: String,
required: false,
default: '',
},
type: {
type: String,
required: false,
default: 'video/webm',
},
duration: {
type: String,
required: false,
default: '',
},
screenRecording: {
type: Boolean,
required: false,
default: false,
},
uploadedRecording: {
type: Boolean,
required: false,
default: false,
},
ctaClickCount: {
type: Number,
default: 0,
},
previewThumbnails: [
new mongoose.Schema(
{
size: {
type: String,
required: false,
default: '',
},
url: {
type: String,
required: false,
default: '',
},
},
{ _id: false }
),
],
previewGifs: [
new mongoose.Schema(
{
size: {
type: String,
required: false,
default: '',
},
url: {
type: String,
required: false,
default: '',
},
},
{ _id: false }
),
],
previewGifsVersion: {
type: Number,
required: false,
default: 0,
},
videoInitialGifUrl: {
type: String,
require: false,
},
unfirlGifUrl: {
type: String,
require: false,
},
deleted: {
type: Boolean,
required: true,
default: '0',
},
subtitles: [
new mongoose.Schema({
kind: {
type: String,
required: true,
default: 'subtitles',
},
label: {
type: String,
required: true,
default: 'English',
},
srclang: {
type: String,
required: true,
default: 'en',
},
url: {
type: String,
required: false,
default: '',
},
transcriptId: {
type: String,
required: false,
default: '',
},
transcriptionPending: {
type: Boolean,
required: true,
default: true,
},
isDefault: {
type: Boolean,
required: true,
default: true,
},
}),
],
views: [
new mongoose.Schema({
deviceId: {
type: String,
required: false,
default: '',
},
startedAt: {
type: String,
required: false,
default: '',
},
viewedDuration: {
type: String,
required: false,
default: '',
},
}),
],
draft: {
type: Boolean,
require: true,
default: true,
},
notified: {
type: Boolean,
require: false,
default: false,
},
dynamic: {
type: Boolean,
require: false,
},
dynamicVideoProcessing: {
type: Boolean,
require: false,
default: false,
},
dynamicVideoProcessingError: {
type: Boolean,
require: false,
default: false,
},
dynamicVideoTemplate: {
type: Boolean,
require: false,
default: false,
},
dynamicVideoTemplateBackgroundUrl: {
type: String,
},
dynamicVideoTemplateGenerationStatus: {
type: String,
require: false,
default: '',
},
dynamicVideoType: {
type: String,
default: 'video',
},
templateRecordingId: {
type: Schema.Types.ObjectId,
default: null,
},
autoGenerated: {
type: Boolean,
require: false,
},
masterRecordingId: {
type: Schema.Types.ObjectId,
require: false,
ref: 'Recordings',
default: null,
},
dynamicRecordings: [
new mongoose.Schema(
{
recordingId: {
type: Schema.Types.ObjectId,
},
firstName: {
type: String,
required: false,
default: '',
},
salutationVideo: {
type: String,
default: null,
},
salutationVideoProcessed: {
type: String,
default: null,
},
screenRecordingProcessed: {
type: String,
},
transcriptId: {
type: String,
},
processed: {
type: Boolean,
default: false,
},
inProgress: {
type: Boolean,
default: false,
},
deleted: {
type: Boolean,
default: false,
},
backgroundScreenUrl: {
type: String,
},
backgroundScreenFileUrl: {
type: String,
},
slug: {
type: String,
},
status: {
type: String,
default: null,
},
},
{
timestamps: true,
}
),
],
masterSalutationVideoUrl: {
type: String,
default: null,
},
cropTimestamp: {
type: Number,
default: 0,
},
audioURL: {
type: String,
default: null,
},
videoLogoUrl: {
type: String,
default: null,
},
videoLogoUrls: [
new mongoose.Schema({
url: {
type: String,
default: null,
},
logoSetting: {
type: Schema.Types.Mixed,
default: null,
},
}),
],
videoLogoSetting: {
type: Schema.Types.Mixed,
default: null,
},
muteVideo: {
type: String,
require: false,
default: 'no',
},
autoPlayVideo: {
type: String,
default: null,
},
flipVideo: {
type: String,
require: false,
default: null,
},
showVideoSubtitle: {
type: String,
require: false,
default: null,
},
backgroundChangeStatus: {
type: String,
default: null,
},
backgroundImageUrl: {
type: String,
require: false,
default: null,
},
backgroundImageName: {
type: String,
require: false,
default: null,
},
videoMatteUrl: {
type: String,
require: false,
default: null,
},
originalVideoUrl: {
type: String,
require: false,
default: null,
},
isProcessingAssets: {
type: Boolean,
default: false,
},
audioDeviceId: {
type: String,
require: false,
default: null,
},
audioDeviceName: {
type: String,
require: false,
default: null,
},
defaultVolume: {
type: Number,
default: 1.0,
},
calendlyLink: {
type: String,
required: false,
},
addCalendarToVideo: {
type: String,
require: false,
default: null,
},
ctaButtonToggle: {
type: String,
require: false,
default: null,
},
ctaButtonText: {
type: String,
require: false,
},
ctaButtonURL: {
type: String,
require: false,
},
lastDynamicProcessCompletedAt: {
type: Date,
default: null,
},
},
{
timestamps: true,
}
)
module.exports = mongoose.model('Recordings', RecordingSchema)

View File

@@ -0,0 +1,3 @@
const RecordingSalutationModel = require('./recording_salutation_model')
module.exports = RecordingSalutationModel

View File

@@ -0,0 +1,76 @@
const mongoose = require('mongoose')
const Schema = mongoose.Schema
const RecordingSalutationSchema = Schema(
{
userId: {
type: Schema.Types.ObjectId,
ref: 'User',
required: true
},
masterRecordingId: {
type: Schema.Types.ObjectId,
ref: 'Recordings',
required: true
},
recordingId: {
type: Schema.Types.ObjectId,
ref: 'Recordings',
required: false
},
firstName: {
type: String,
required: false,
default: ''
},
salutationVideo: {
type: String,
default: null
},
salutationVideoProcessed: {
type: String,
default: null
},
screenRecordingProcessed: {
type: String
},
transcriptId: {
type: String
},
processed: {
type: Boolean,
default: false
},
inProgress: {
type: Boolean,
default: false
},
deleted: {
type: Boolean,
default: false
},
backgroundScreenUrl: {
type: String
},
backgroundScreenFileUrl: {
type: String
},
slug: {
type: String
},
status: {
type: String,
default: null
}
},
{
timestamps: true
}
)
const RecordingSalutationModel = mongoose.model(
'recording_salutations',
RecordingSalutationSchema
)
module.exports = RecordingSalutationModel

View File

@@ -0,0 +1,4 @@
const SalutationModel = require('./salutation_model')
const SalutationService = require('./salutation_service')
module.exports = SalutationService(SalutationModel)

View File

@@ -0,0 +1,44 @@
const mongoose = require('mongoose')
const Schema = mongoose.Schema
const SalutationSchema = Schema(
{
userId: {
type: Schema.Types.ObjectId,
ref: 'User',
required: true
},
firstName: {
type: String,
required: false,
default: ''
},
salutationVideo: {
type: String,
default: null
},
transcriptId: {
type: String,
default: null,
required: false
},
userAudioProfileId: {
type: Schema.Types.ObjectId,
default: null,
required: false
},
transcriptObj: {
type: Object,
default: null,
required: false
},
deleted: {
type: Boolean,
default: false
}
},
{
timestamps: true
}
)
module.exports = mongoose.model('Salutations', SalutationSchema)

View File

@@ -0,0 +1,87 @@
const create = (Salutation) => async (salutationData, userId) => {
const newSalutation = new Salutation({ ...salutationData, userId })
const savedSalutation = await newSalutation.save()
return savedSalutation
}
const read = (Salutation) => async (filter) => {
const foundSalutation = await Salutation.findOne({
...filter,
deleted: false,
})
return foundSalutation
}
const find = (Salutation) => async (filter) => {
const foundSalutations = await Salutation.find({
...filter,
deleted: false,
})
return foundSalutations
}
const update = (Salutation) => async (salutation, userId) => {
const updatedSalutation = await Salutation.findOneAndUpdate(
{ _id: salutation._id, userId },
salutation,
{ new: true }
)
return updatedSalutation
}
const modify = (Salutation) => async (salutation) => {
const findQuery = salutation._id
? { _id: salutation._id }
: { transcriptId: salutation.transcriptId }
const updatedSalutation = await Salutation.findOneAndUpdate(
findQuery,
salutation,
{ new: true }
)
return updatedSalutation
}
const updateOrCreate =
(Salutation) =>
async ({ firstName, salutationVideo, userAudioProfileId }, userId) => {
const salutation = await read(Salutation)({
firstName,
userAudioProfileId,
userId,
})
if (salutation) {
salutation.salutationVideo = salutationVideo
return await update(Salutation)(salutation, userId)
} else {
return await create(Salutation)(
{ firstName, salutationVideo, userAudioProfileId },
userId
)
}
}
const remove = (Salutation) => async (filter, userId) => {
const updatedSalutation = await Salutation.findOneAndUpdate(
{ ...filter, userId },
{
$set: {
deleted: true,
},
},
{ new: true }
)
return updatedSalutation
}
module.exports = (Salutation) => {
return {
create: create(Salutation),
read: read(Salutation),
remove: remove(Salutation),
update: update(Salutation),
updateOrCreate: updateOrCreate(Salutation),
find: find(Salutation),
modify: modify(Salutation),
}
}

View File

@@ -0,0 +1,4 @@
const UserAudioProfile = require('./user_audio_profile_model')
const UserAudioProfileService = require('./user_audio_profile_service')
module.exports = UserAudioProfileService(UserAudioProfile)

View File

@@ -0,0 +1,40 @@
const mongoose = require('mongoose')
const Schema = mongoose.Schema
const UserAudioProfileSchema = Schema(
{
userId: {
type: Schema.Types.ObjectId,
ref: 'User',
required: true,
},
name: {
type: String,
required: true,
default: '',
},
status: {
type: String,
required: false,
default: 'created',
},
training_model_path: {
type: Schema.Types.Mixed,
default: null,
},
training_model_s3_path: {
type: Schema.Types.Mixed,
default: null,
},
deleted: {
type: Boolean,
required: true,
default: false,
},
},
{
timestamps: true,
}
)
module.exports = mongoose.model('UserAudioProfile', UserAudioProfileSchema)

View File

@@ -0,0 +1,142 @@
const StringifyUtils = require('../../app/services/utils/logService')
const create = (UserAudioProfileModel) => async (data) => {
try {
const newModel = new UserAudioProfileModel({ ...data })
const savedModel = await newModel.save()
return savedModel
} catch (error) {
const details = { data }
console.log(
'ERROR - USER AUDIO PROFILE SERVICE > create',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const insertMany = (UserAudioProfileModel) => async (data) => {
try {
const inserted = await UserAudioProfileModel.insertMany(data)
return inserted
} catch (error) {
const details = { data }
console.log(
'ERROR - USER AUDIO PROFILE SERVICE > insertMany',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const read = (UserAudioProfileModel) => async (filter) => {
try {
const foundModel = await UserAudioProfileModel.findOne({
...filter,
deleted: false
})
return foundModel
} catch (error) {
const details = { filter }
console.log(
'ERROR - USER AUDIO PROFILE SERVICE > read',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const find = (UserAudioProfileModel) => async (filter) => {
try {
const foundModels = await UserAudioProfileModel.find({
...filter,
deleted: false
})
return foundModels
} catch (error) {
const details = { filter }
console.log(
'ERROR - USER AUDIO PROFILE SERVICE > find',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const update = (UserAudioProfileModel) => async (data) => {
console.log('ua data', data)
try {
const updatedModel = await UserAudioProfileModel.findOneAndUpdate(
{ _id: data._id },
data,
{
new: true
}
)
console.log('ua updatedModel', updatedModel)
return updatedModel
} catch (error) {
const details = { data }
console.log(
'ERROR - USER AUDIO PROFILE SERVICE > update',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const remove = (UserAudioProfileModel) => async (filter) => {
try {
const updatedModel = await UserAudioProfileModel.findOneAndUpdate(
{ ...filter },
{
$set: {
deleted: true
}
},
{ new: true }
)
return updatedModel
} catch (error) {
const details = { filter }
console.log(
'ERROR - USER AUDIO PROFILE SERVICE > remove',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
const removeMany = (UserAudioProfileModel) => async (filter) => {
try {
const updatedModel = await UserAudioProfileModel.updateMany(
{ ...filter },
{
$set: {
deleted: true
}
},
{ new: true }
)
return updatedModel
} catch (error) {
const details = { filter }
console.log(
'ERROR - USER AUDIO PROFILE SERVICE > removeMany',
StringifyUtils.potionErrorObj(error, details)
)
throw error
}
}
module.exports = (UserAudioProfileModel) => {
return {
create: create(UserAudioProfileModel),
insertMany: insertMany(UserAudioProfileModel),
read: read(UserAudioProfileModel),
remove: remove(UserAudioProfileModel),
removeMany: removeMany(UserAudioProfileModel),
update: update(UserAudioProfileModel),
find: find(UserAudioProfileModel)
}
}