{ "number": 28, "title": "GCP support, upgraded TTS, new model with cleaner data", "body": "", "state": "OPEN", "url": "https://github.com/potion/potion-voice/pull/28", "createdAt": "2023-11-29T07:42:08Z", "mergedAt": null, "closedAt": null, "additions": 1816, "deletions": 283, "changedFiles": 15, "isDraft": false, "baseRefName": "staging", "headRefName": "develop", "author": { "login": "author_unknown" }, "mergedBy": { "login": "" }, "mergeCommit": { "oid": "8615798d7f85f3f29ce99973f97bd9071f76f05b" }, "milestone": null, "labels": { "nodes": [] }, "assignees": { "nodes": [] }, "requestedReviewers": { "nodes": [] }, "commits": { "totalCount": 23, "nodes": [ { "commit": { "oid": "ad31e02f2a39af5e40fbefe30483e5d8119346b4", "message": "Added GCP support details and revised config settings for 48k Hz sampling rate usage with TTS v0.20.6", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2023-11-23T08:45:24Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2023-11-23T08:45:24Z" } } }, { "commit": { "oid": "c41b76ff84f69f00ebfdb1bfbea6c40b0c2529bd", "message": "Richer config settings (added sampling rate-based configs) and updated training settings.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-19T08:36:49Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-19T08:36:49Z" } } }, { "commit": { "oid": "fdf7496764cff1f7f754175b121d1eec5fce285b", "message": "Improved error handling and robustness; added support for FLAC files - VCTK v0.92 preprocessing.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T08:00:56Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T08:00:56Z" } } }, { "commit": { "oid": "55886541d20ef9be247610cf616f0167842ca8de", "message": "Add support for 24k sampling rate.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T08:33:31Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T08:33:31Z" } } }, { "commit": { "oid": "bf4d4c9b215894810755bcd9deb1d6a980c408cc", "message": "Improved for directory name in extract_archive", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T09:08:51Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T09:08:51Z" } } }, { "commit": { "oid": "90bf857da5add6b341a5514d13bc2a144b8a542c", "message": "Bug fix in extract_archive", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T09:35:56Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T09:35:56Z" } } }, { "commit": { "oid": "cd968b26a2214cb241a4b660f80c360493a38493", "message": "Support VCTK v0.92 _mic[12] naming convention.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T12:43:27Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T12:43:27Z" } } }, { "commit": { "oid": "c5abb3269df3c7e4d828ba019907ac0356349fd3", "message": "Rename wav_dir to audio_dir and correct function call arguments.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T13:58:09Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T13:58:09Z" } } }, { "commit": { "oid": "334aef03182d954606bf2dd6907a30f380ecc0dd", "message": "Verify that after extraction the dataset root and its audio and transcription file directories are identified correctly.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T15:31:18Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T15:31:18Z" } } }, { "commit": { "oid": "41fa5ff5a72366fb7466445c79dad5c8fd5ec598", "message": "Improved quality assurance: Ensure that each speaker had transcriptions with matching audio files and vice versa.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T17:32:04Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2023-12-27T17:32:04Z" } } }, { "commit": { "oid": "2e33a79e0465e5fb13f22b6fee2e1f54178860d9", "message": "fixups and added support for training merged, VCTK-formatted datasets.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-10T03:26:07Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-10T03:26:07Z" } } }, { "commit": { "oid": "b0a3c294729107b780d941dfc4d0930b055b187c", "message": "Added support for training merged, VCTK-formatted datasets.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-10T03:26:37Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-10T03:26:37Z" } } }, { "commit": { "oid": "2d7adac5070bd308eba84c7dc193b6c5b5effaa2", "message": "New cloning approach with termination condition based on speaker similarity and voice naturalness scores.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-10T03:27:39Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-10T03:27:39Z" } } }, { "commit": { "oid": "6161686a58dd16505b59107c406ff759eaf02e71", "message": "Added minimum scoring thresholds for speaker similarity and naturalness; updated scoring parameters.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-10T06:28:07Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-10T06:28:07Z" } } }, { "commit": { "oid": "665fadf062511880abfae5b1f1b2bbf1d15d0a8f", "message": "Skip checkpoint scoring iff keyboard interrupt.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-10T08:12:42Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-10T08:12:42Z" } } }, { "commit": { "oid": "3a03db620469cf6603b49cabe1201a22bf81775e", "message": "Expand pattern to also pick up best_simnat_checkpoint_*.pth checkpoint files.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-11T09:50:30Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-11T09:50:30Z" } } }, { "commit": { "oid": "b22f1051bb3374bc969e3962740a3610266f2928", "message": "48k Voice cloning documentation now based on clone_voice_via_continue_n_natqa.py; adjusted default cloning parameters.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-11T17:34:07Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-11T17:34:07Z" } } }, { "commit": { "oid": "2c34903fe6b62551fb5d6844af78a0da1eb3cdb6", "message": "find_best_cloned_model.py now also checks for minimum quality nat & sim scores; returns None for best_model if they are not met.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-12T10:30:52Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-12T10:30:52Z" } } }, { "commit": { "oid": "75e92b2b887e3f0ccf8ea4a5c536883d4b6e2570", "message": "Add support for training speaker encoder model at 16k and 48k sampling rates.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-28T16:06:30Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-28T16:06:30Z" } } }, { "commit": { "oid": "7241466763630caf6febb9ea620211b2dcdb1bd7", "message": "Bug fix: add use_cuda parameters whenever required", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-29T09:30:17Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-29T09:30:17Z" } } }, { "commit": { "oid": "530ff4c8e7bc89d6205cf43d7ccd13bd13c5124e", "message": "Init logger if None is given.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-29T12:51:00Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-29T12:51:00Z" } } }, { "commit": { "oid": "38e981c51b9232c6e441d96c2901167f43d7ce0b", "message": "Evaluation parameter passing fix.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-29T17:43:33Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-29T17:43:33Z" } } }, { "commit": { "oid": "85eddfd84d8f6c79ba99010fa82148641c97970f", "message": "Bug fix: Eval routine output init missing.", "author": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-30T01:53:17Z" }, "committer": { "name": "author_unknown", "email": "author_unknown", "date": "2024-01-30T01:53:17Z" } } } ] }, "reviews": { "nodes": [] }, "comments": { "nodes": [] }, "files": { "nodes": [ { "path": "voice-cloning/clone_voice_via_continue.py", "additions": 1, "deletions": 1, "changeType": "MODIFIED" }, { "path": "voice-cloning/clone_voice_via_continue_n_natqa.py", "additions": 291, "deletions": 0, "changeType": "ADDED" }, { "path": "voice-cloning/docs/Voice Cloning @ 48k Hz Sampling Rate - Step-by-Step.txt", "additions": 62, "deletions": 57, "changeType": "MODIFIED" }, { "path": "voice-cloning/docs/potion-voice-cloning_Installation_Guide.md", "additions": 20, "deletions": 50, "changeType": "MODIFIED" }, { "path": "voice-cloning/find_best_cloned_model.py", "additions": 23, "deletions": 11, "changeType": "MODIFIED" }, { "path": "voice-cloning/find_best_multispeaker_model.py", "additions": 1, "deletions": 1, "changeType": "MODIFIED" }, { "path": "voice-cloning/prepare_datasets.py", "additions": 262, "deletions": 88, "changeType": "MODIFIED" }, { "path": "voice-cloning/save_multispeaker_baseline_embeddings_file.py", "additions": 11, "deletions": 4, "changeType": "MODIFIED" }, { "path": "voice-cloning/synthesize_speech.py", "additions": 1, "deletions": 1, "changeType": "MODIFIED" }, { "path": "voice-cloning/train_config.py", "additions": 101, "deletions": 46, "changeType": "MODIFIED" }, { "path": "voice-cloning/train_multispeaker_baseline_model.py", "additions": 44, "deletions": 23, "changeType": "MODIFIED" }, { "path": "voice-cloning/train_speaker_encoder.py", "additions": 181, "deletions": 0, "changeType": "ADDED" }, { "path": "voice-cloning/utils/scoring_utils.py", "additions": 78, "deletions": 1, "changeType": "MODIFIED" }, { "path": "voice-cloning/utils/speaker_encoder_utils.py", "additions": 389, "deletions": 0, "changeType": "ADDED" }, { "path": "voice-cloning/utils/trainer_utils.py", "additions": 351, "deletions": 0, "changeType": "ADDED" } ] } }